hermit-detcore 0.4.0

Detcore: the deterministic scheduler and syscall determinization core of the Hermit execution engine.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
/*
 * Copyright (c) Meta Platforms, Inc. and affiliates.
 * All rights reserved.
 *
 * This source code is licensed under the BSD-style license found in the
 * LICENSE file in the root directory of this source tree.
 */

//! Hash the bytes a syscall moved through a guest buffer, at the syscall
//! boundary.
//!
//! WHAT GAP THIS FILLS. `--detlog-stack` and `--detlog-heap` hash a whole named
//! mapping, so what they can see is decided by where the guest happened to
//! ALLOCATE a buffer rather than by what the syscall did. Measured 2026-08-20,
//! three runs per cell, running the same netlink `RTM_GETLINK` exchange and
//! changing only the receive buffer's home:
//!
//! | buffer home     | `--detlog-stack` | `--detlog-heap` | both   |
//! |-----------------|------------------|-----------------|--------|
//! | `[stack]`       | CAUGHT           | MISSED          | CAUGHT |
//! | `[heap]` (brk)  | MISSED           | CAUGHT          | CAUGHT |
//! | BSS / static    | MISSED           | MISSED          | MISSED |
//! | anonymous mmap  | MISSED           | MISSED          | MISSED |
//!
//! Two of the four are invisible even with both flags on, and anonymous mmap is
//! not a corner case: it is where glibc puts any `malloc` above the 128 KiB
//! `M_MMAP_THRESHOLD`. Taking the address and length from the SYSCALL ARGUMENTS
//! instead makes all four rows irrelevant by construction.
//!
//! WHAT IT CATCHES THAT `--verify` CANNOT. `--verify` compares the INFO record,
//! and a syscall whose output buffer is typed as a bare pointer in Reverie
//! prints the address, not the contents (`reverie-syscalls/src/syscalls.rs`
//! carries standing TODOs saying exactly this for `Read` and `Write`). So a
//! `recvmsg` that returns a stable `Ok(1468)` while four bytes of its payload
//! differ produces a character-identical record and `--verify` reports
//! `bitwise_parity: true`. Measured on one QEMU/Linux boot, 278,824 of 632,228
//! syscalls (44.1%) move bytes through a buffer whose content the log never
//! shows.
//!
//! COST SHAPE, and it differs from the whole-mapping flags in kind rather than
//! degree. `--detlog-heap` is `O(syscalls x region_size)` -- it re-reads the
//! entire heap after every syscall, hashing 10.9 TB per boot to watch a 16.49
//! MiB heap. This is `O(bytes the syscalls actually returned)`: 139.1 MB per
//! boot, measured by summing real return values.

use reverie::Error;
use reverie::Guest;
use reverie::Tool;
use reverie::syscalls::Addr;
use reverie::syscalls::AddrMut;
use reverie::syscalls::Errno;
use reverie::syscalls::MemoryAccess;
use reverie::syscalls::Syscall;
use reverie::syscalls::SyscallInfo;

use crate::digest::Digest;
use crate::types::DetTid;

/// One contiguous run of guest bytes a completed syscall moved.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct BufferExtent {
    /// Guest virtual address of the first byte.
    pub addr: u64,
    /// Number of bytes actually moved, which is bounded by the syscall's return
    /// value and not by the buffer's declared capacity.
    pub len: u64,
}

/// Bind successful RNG output to the very array the synchronous handler used.
/// The return count, not the declared capacity, bounds each observed extent.
pub(crate) fn rng_readv_extents(
    iovecs: &[crate::iovecs::ImportedIovec],
    written: usize,
) -> Result<Vec<BufferExtent>, Error> {
    let mut output = rng_observation_vec(iovecs.len(), "imported geometry")?;
    let mut remaining = written;
    for iov in iovecs {
        let take = remaining.min(iov.len);
        remaining -= take;
        if take > 0 {
            output.push(BufferExtent {
                addr: iov.base as u64,
                len: take as u64,
            });
        }
    }
    Ok(output)
}

/// RNG observation happens after output and any shared-cursor commit. A
/// recoverable reservation failure must stop the tool, not become guest errno.
/// This does not promise to catch physical OOM or other logging allocations.
fn rng_observation_vec<T>(capacity: usize, purpose: &str) -> Result<Vec<T>, Error> {
    let mut output = Vec::new();
    output.try_reserve_exact(capacity).map_err(|error| {
        Error::Tool(anyhow::anyhow!(
            "RNG vector observation after output commit: cannot reserve {purpose}: {error}"
        ))
    })?;
    Ok(output)
}

fn observed_extents(
    memory: &impl MemoryAccess,
    call: &Syscall,
    ret: i64,
    rng_output: Option<&[BufferExtent]>,
) -> Result<Vec<BufferExtent>, Error> {
    if let Some(output) = rng_output {
        if ret <= 0 {
            return Ok(Vec::new());
        }
        let mut observed = rng_observation_vec(output.len(), "observed geometry")?;
        let mut remaining = ret as u64;
        for extent in output {
            let take = remaining.min(extent.len);
            remaining -= take;
            if take > 0 {
                observed.push(BufferExtent {
                    addr: extent.addr,
                    len: take,
                });
            }
        }
        return Ok(observed);
    }
    extents(memory, call, ret)
}

/// Direction of travel, recorded so a reader can tell a value the kernel
/// produced from one the guest produced without knowing every syscall.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Direction {
    /// The kernel filled the buffer: an unvirtualized field here is a leak INTO
    /// the guest, which is the netlink case.
    In,
    /// The guest filled the buffer: a difference here means the guest computed
    /// different bytes, which is how divergence reaches a file or a device.
    Out,
}

impl Direction {
    fn as_str(self) -> &'static str {
        match self {
            Direction::In => "in",
            Direction::Out => "out",
        }
    }
}

/// Clamp a declared buffer to what the syscall says it actually moved.
///
/// `capacity` is the argument the guest passed; `moved` is the return value.
/// Hashing `capacity` would fold in bytes the syscall never wrote, which are
/// whatever was in the buffer beforehand -- for a stack buffer that is previous
/// frames, so the hash would report divergence for an unrelated reason.
fn clamp(addr: Option<u64>, capacity: usize, moved: i64) -> Vec<BufferExtent> {
    let Some(addr) = addr else { return Vec::new() };
    let moved = u64::try_from(moved).unwrap_or(0);
    let len = moved.min(capacity as u64);
    if len == 0 {
        return Vec::new();
    }
    vec![BufferExtent { addr, len }]
}

/// A buffer written wholly in place, whose size does not depend on the return
/// value (`poll`'s `revents` are rewritten across the whole array regardless of
/// how many descriptors were ready).
fn whole(addr: Option<u64>, len: u64) -> Vec<BufferExtent> {
    match addr {
        Some(addr) if len > 0 => vec![BufferExtent { addr, len }],
        _ => Vec::new(),
    }
}

/// Walk an `iovec` array and return the segments the syscall actually filled,
/// bounded by `moved`.
///
/// Mirrors the existing traversal in `crate::syscalls::io`: clamp the count to
/// `UIO_MAXIOV`, skip null/empty segments, and stop once `moved` bytes are
/// accounted for. Under `MSG_TRUNC` the returned count can exceed the buffers'
/// capacity, which is why the running remainder rather than `moved` alone
/// bounds each segment.
fn iovec_extents<M: MemoryAccess>(
    memory: &M,
    iov_addr: usize,
    iov_count: usize,
    moved: i64,
) -> Result<Vec<BufferExtent>, Error> {
    let remaining = u64::try_from(moved).unwrap_or(0);
    if iov_addr == 0 || iov_count == 0 || remaining == 0 {
        return Ok(Vec::new());
    }
    let iov_count = iov_count.min(libc::UIO_MAXIOV as usize);
    let iov_address: AddrMut<'_, libc::iovec> = AddrMut::from_raw(iov_addr).ok_or(Errno::EFAULT)?;
    // SAFETY: `iovec` is a plain C record; an all-zero value is a valid staging
    // value that `read_values` immediately overwrites.
    let mut iovecs: Vec<libc::iovec> = (0..iov_count)
        .map(|_| unsafe { std::mem::zeroed() })
        .collect();
    memory.read_values(iov_address.into(), &mut iovecs)?;
    Ok(iovec_extents_from_slice(&iovecs, moved))
}

/// Apply the return-value bound after the guest `iovec` array has been read.
///
/// This is separate from the memory access so the short-transfer rule can be
/// tested directly for every syscall family that shares it.
fn iovec_extents_from_slice(iovecs: &[libc::iovec], moved: i64) -> Vec<BufferExtent> {
    let mut remaining = u64::try_from(moved).unwrap_or(0);

    let mut out = Vec::new();
    for iov in iovecs {
        if remaining == 0 {
            break;
        }
        if iov.iov_base.is_null() || iov.iov_len == 0 {
            continue;
        }
        let take = (iov.iov_len as u64).min(remaining);
        out.push(BufferExtent {
            addr: iov.iov_base as u64,
            len: take,
        });
        remaining -= take;
    }
    out
}

/// Return the guest `iovec` array described by a vectored I/O syscall.
///
/// Keep the family in one match so adding a syscall variant cannot update the
/// log direction without also making its complete, return-value-bounded iovec
/// prefix available to [`extents`].
fn iovec_extent_arguments(call: &Syscall) -> Option<(usize, usize)> {
    match call {
        Syscall::Readv(call) => Some((call.iov().map_or(0, |p| p.as_raw()), call.len())),
        Syscall::Preadv(call) => Some((call.iov().map_or(0, |p| p.as_raw()), call.iov_len())),
        Syscall::Preadv2(call) => Some((
            call.iov().map_or(0, |p| p.as_raw()),
            usize::try_from(call.iov_len()).unwrap_or(usize::MAX),
        )),
        Syscall::Writev(call) => Some((call.iov().map_or(0, |p| p.as_raw()), call.len())),
        Syscall::Pwritev(call) => Some((call.iov().map_or(0, |p| p.as_raw()), call.iov_len())),
        Syscall::Pwritev2(call) => Some((
            call.iov().map_or(0, |p| p.as_raw()),
            usize::try_from(call.iov_len()).unwrap_or(usize::MAX),
        )),
        _ => None,
    }
}

/// Read a `msghdr` out of the guest and walk the iovecs it points at.
fn msghdr_extents<M: MemoryAccess>(
    memory: &M,
    msg_addr: usize,
    moved: i64,
) -> Result<Vec<BufferExtent>, Error> {
    if msg_addr == 0 {
        return Ok(Vec::new());
    }
    let address: AddrMut<'_, libc::msghdr> = AddrMut::from_raw(msg_addr).ok_or(Errno::EFAULT)?;
    let message: libc::msghdr = memory.read_value(address)?;
    iovec_extents(memory, message.msg_iov as usize, message.msg_iovlen, moved)
}

/// Number of completed messages whose per-message lengths are meaningful.
fn completed_mmsghdr_count(vlen: u32, completed: i64) -> usize {
    usize::try_from(completed)
        .unwrap_or(0)
        .min(vlen as usize)
        .min(libc::UIO_MAXIOV as usize)
}

/// Walk the `mmsghdr` array a batch send or receive completed.
///
/// `moved` here is a COUNT OF MESSAGES, not a byte count -- which is why
/// these calls cannot share the `clamp`/`msghdr_extents` path that every other
/// send or receive uses. Each completed message carries its own byte count in
/// `msg_len`, so each is walked separately and bounded by that; treating the
/// batch as one buffer would let one message's length run into the next
/// message's memory.
fn mmsghdr_extents<M: MemoryAccess>(
    memory: &M,
    mmsg_addr: usize,
    vlen: u32,
    delivered: i64,
) -> Result<Vec<BufferExtent>, Error> {
    let count = completed_mmsghdr_count(vlen, delivered);
    if mmsg_addr == 0 || count == 0 {
        return Ok(Vec::new());
    }
    let address: AddrMut<'_, libc::mmsghdr> = AddrMut::from_raw(mmsg_addr).ok_or(Errno::EFAULT)?;
    // SAFETY: `mmsghdr` is a plain C record; an all-zero value is a valid
    // staging value that `read_values` immediately overwrites.
    let mut headers: Vec<libc::mmsghdr> =
        (0..count).map(|_| unsafe { std::mem::zeroed() }).collect();
    memory.read_values(address.into(), &mut headers)?;

    let mut out = Vec::new();
    for header in &headers {
        out.extend(iovec_extents(
            memory,
            header.msg_hdr.msg_iov as usize,
            header.msg_hdr.msg_iovlen,
            i64::from(header.msg_len),
        )?);
    }
    Ok(out)
}

/// Whether this syscall's return value bounds what it wrote.
///
/// True for the ordinary case, where the return IS the byte count, so a return
/// of 0 or less means nothing was written. FALSE for `poll`/`ppoll`, whose
/// return is a COUNT OF READY DESCRIPTORS: the kernel rewrites `revents` on
/// every entry even when it returns 0. Measured -- `poll(timeout=0)` on two
/// never-ready pipe fds returns 0 and still moves `revents` from a poisoned
/// 0x7FFF to 0 on both entries.
///
/// This exists as a predicate consulted BY the single `ret <= 0` guard, rather
/// than as poll arms placed above it, so the guard cannot be reordered back
/// into swallowing the zero-ready case. The original gap was exactly that: the
/// guard sat above the poll arms, and the unit test called `whole` directly, so
/// it could not see the short-circuit above the code it exercised.
/// Every syscall `extents` below returns a non-empty result for.
///
/// KEEP THIS IN STEP WITH THE `extents` MATCH -- it is the same set written
/// twice, once as `Syscall::` arms (which need constructed calls and a `Guest`
/// to exercise) and once as bare `Sysno`s (which a subscription can be asked
/// about). The second spelling exists so the relationship between this check
/// and what Detcore actually intercepts can be ASSERTED rather than assumed;
/// see `passthru_opt_leaves_io_buffer_hashing_blind_only_for_getcwd` in
/// `lib.rs`. Adding an arm to `extents` without adding it here does not break
/// that test, so add both.
#[cfg(test)]
pub(crate) const HASHED_SYSCALLS: &[reverie::syscalls::Sysno] = &[
    // Bytes the kernel produced.
    reverie::syscalls::Sysno::read,
    reverie::syscalls::Sysno::pread64,
    reverie::syscalls::Sysno::recvfrom,
    reverie::syscalls::Sysno::getrandom,
    reverie::syscalls::Sysno::getcwd,
    reverie::syscalls::Sysno::getdents64,
    reverie::syscalls::Sysno::readlink,
    reverie::syscalls::Sysno::readlinkat,
    reverie::syscalls::Sysno::recvmsg,
    reverie::syscalls::Sysno::recvmmsg,
    reverie::syscalls::Sysno::readv,
    reverie::syscalls::Sysno::preadv,
    reverie::syscalls::Sysno::preadv2,
    // Bytes the guest produced.
    reverie::syscalls::Sysno::write,
    reverie::syscalls::Sysno::pwrite64,
    reverie::syscalls::Sysno::sendto,
    reverie::syscalls::Sysno::sendmsg,
    reverie::syscalls::Sysno::sendmmsg,
    reverie::syscalls::Sysno::writev,
    reverie::syscalls::Sysno::pwritev,
    reverie::syscalls::Sysno::pwritev2,
    // Rewritten in place across the whole array.
    reverie::syscalls::Sysno::poll,
    reverie::syscalls::Sysno::ppoll,
];

fn ret_gates_output(call: &Syscall) -> bool {
    !matches!(call, Syscall::Poll(_) | Syscall::Ppoll(_))
}

/// The extents a completed syscall moved, or `None` when this syscall has no
/// output buffer worth hashing.
///
/// Only syscalls whose buffer CONTENT the INFO record does not already show are
/// listed. `clock_gettime` and `newfstatat`, for instance, are absent because
/// Reverie's typed display already dereferences and prints their output.
fn extents<M: MemoryAccess>(
    memory: &M,
    call: &Syscall,
    ret: i64,
) -> Result<Vec<BufferExtent>, Error> {
    // Nothing was written on a failed or empty call -- for every syscall whose
    // return value is a byte count. `ret_gates_output` is what keeps the poll
    // family out of this, and it is a predicate rather than an arm placed above
    // so the exclusion cannot be undone by moving code.
    if ret <= 0 && ret_gates_output(call) {
        return Ok(Vec::new());
    }
    if let Some((iov_addr, iov_count)) = iovec_extent_arguments(call) {
        return iovec_extents(memory, iov_addr, iov_count, ret);
    }
    let raw = |a: Option<AddrMut<'_, u8>>| a.map(|p| p.as_raw() as u64);
    Ok(match call {
        // Bytes the kernel produced.
        Syscall::Read(c) => clamp(raw(c.buf()), c.len(), ret),
        Syscall::Pread64(c) => clamp(raw(c.buf()), c.len(), ret),
        Syscall::Recvfrom(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.len(), ret),
        Syscall::Getrandom(c) => clamp(raw(c.buf()), c.buflen(), ret),
        Syscall::Getcwd(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.size(), ret),
        Syscall::Getdents64(c) => clamp(
            c.dirent().map(|p| p.as_raw() as u64),
            c.count() as usize,
            ret,
        ),
        Syscall::Readlink(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.bufsize(), ret),
        Syscall::Readlinkat(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.buf_len(), ret),
        Syscall::Recvmsg(c) => msghdr_extents(memory, c.msg().map_or(0, |p| p.as_raw()), ret)?,
        // `ret` is a MESSAGE count here, not a byte count; see
        // `mmsghdr_extents`. recvmmsg is one of the four receive syscalls
        // that could reach a NETLINK_SOCK_DIAG dump without passing the
        // sock_diag sanitizer, so leaving it unhashed left this check blind
        // to exactly the bypass it would otherwise have reported.
        Syscall::Recvmmsg(c) => {
            mmsghdr_extents(memory, c.mmsg().map_or(0, |p| p.as_raw()), c.vlen(), ret)?
        }
        // Bytes the guest produced. These never reach stdout/stderr for a QEMU
        // boot -- measured, all 234,872 writes went to fds 7/12/14/11/13/4/8/19/23
        // and none to fd 1 or 2 -- so `--verify`'s stdout/stderr comparison does
        // not cover them either.
        Syscall::Write(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.len(), ret),
        Syscall::Pwrite64(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.len(), ret),
        Syscall::Sendto(c) => clamp(c.buf().map(|p| p.as_raw() as u64), c.size(), ret),
        Syscall::Sendmsg(c) => msghdr_extents(memory, c.msg().map_or(0, |p| p.as_raw()), ret)?,
        // Like recvmmsg, `ret` counts completed messages and each completed
        // header's `msg_len` bounds the bytes consumed from that message.
        Syscall::Sendmmsg(c) => {
            mmsghdr_extents(memory, c.msgvec().map_or(0, |p| p.as_raw()), c.vlen(), ret)?
        }
        // Rewritten in place across the WHOLE array: `poll` sets `revents` on
        // every entry, not just on the `ret` that were ready, so the extent is
        // the array and not a prefix of it -- and it is reached even when
        // `ret == 0`, via `ret_gates_output`.
        Syscall::Poll(c) => whole(
            c.fds().map(|p| p.as_raw() as u64),
            c.nfds() * std::mem::size_of::<libc::pollfd>() as u64,
        ),
        Syscall::Ppoll(c) => whole(
            c.fds().map(|p| p.as_raw() as u64),
            c.nfds() * std::mem::size_of::<libc::pollfd>() as u64,
        ),

        _ => Vec::new(),
    })
}

/// Which way the bytes travelled, for the record's label.
fn direction(call: &Syscall) -> Direction {
    match call {
        Syscall::Write(_)
        | Syscall::Pwrite64(_)
        | Syscall::Sendto(_)
        | Syscall::Sendmsg(_)
        | Syscall::Sendmmsg(_)
        | Syscall::Writev(_)
        | Syscall::Pwritev(_)
        | Syscall::Pwritev2(_) => Direction::Out,
        _ => Direction::In,
    }
}

/// Emit one deterministic record per buffer a completed syscall moved.
///
/// Digests that LOCATE a divergence instead of only detecting one.
///
/// The whole-extent digest says two buffers differ and nothing else -- not
/// which bytes, not which field. Comparing two logs of per-chunk digests names
/// the FIRST DIFFERING CHUNK, which is the offset and a bounded window around
/// it. That is what a content divergence needs to be classifiable, and it costs
/// no retained bytes: the guest's data never leaves the guest.
///
/// ⚠️ ONE GUEST READ, NOT ONE PER CHUNK. Reading guest memory is the expensive
/// half of this check -- `compute_hash_range` does a `read_values` per call --
/// so the extent is read ONCE and every digest is taken from that local copy.
/// The guest-memory cost is therefore identical to the single-digest version
/// this replaces; only local hashing and log width grow.
///
/// The chunk count is CAPPED so a large buffer cannot produce an unbounded log
/// line: at most [`CHUNK_CAP`] digests always span the whole extent, so the
/// window is `max(CHUNK_MIN, ceil(len / CHUNK_CAP))`. A buffer that fits in one
/// chunk emits no chunk list, because a single chunk repeats what the
/// whole-extent digest already said.
const CHUNK_CAP: usize = 8;
const CHUNK_MIN: usize = 256;

fn extent_digests<G, T>(
    guest: &mut G,
    addr: u64,
    len: u64,
    rng_output: bool,
) -> Result<(Digest, usize, Vec<String>), Error>
where
    G: Guest<T>,
    T: Tool,
{
    let size = len as usize;
    let mut buf = if rng_output {
        let mut buf = rng_observation_vec(size, "digest payload")?;
        buf.resize(size, 0u8);
        buf
    } else {
        vec![0u8; size]
    };
    if size > 0 {
        let start = Addr::<u8>::from_raw(addr as usize).ok_or(Errno::EFAULT)?;
        guest.memory().read_values(start, buf.as_mut_slice())?;
    }
    let whole = Digest::new(buf.as_slice());
    let (chunk, chunks) = chunk_digests(buf.as_slice());
    Ok((whole, chunk, chunks))
}

/// The locating half, split out from the guest read so it can be bracketed.
///
/// Returns the chunk width and one short digest per chunk, or an empty vector
/// when the extent fits in a single chunk and the list would only repeat the
/// whole-extent digest.
fn chunk_digests(buf: &[u8]) -> (usize, Vec<String>) {
    let size = buf.len();
    let chunk = std::cmp::max(CHUNK_MIN, size.div_ceil(CHUNK_CAP));
    if size <= chunk {
        return (chunk, Vec::new());
    }
    let chunks = buf
        .chunks(chunk)
        // Short prefix per chunk: this locates a window, it does not
        // authenticate one, and eight full SHA-256 digests per buffer would
        // dominate the line without making the location any sharper. Truncated
        // explicitly rather than with a `{:.8}` precision, which `Digest`'s
        // Display does not honour -- it delegates to LowerHex and ignores the
        // formatter's precision, so the specifier silently printed the full
        // digest.
        .map(|c| {
            Digest::new(c)
                .to_string()
                .chars()
                .take(8)
                .collect::<String>()
        })
        .collect();
    (chunk, chunks)
}

/// ⚠️ THE GUARD IS FIRST AND THAT IS THE POINT. Everything below it touches
/// guest memory: for `recvmsg` the extents cannot even be computed without
/// reading a `msghdr` and an `iovec` array out of the guest. That is
/// preparatory work done BEFORE the `detlog!`, which is exactly the shape that
/// made `--detlog-stack` and `--detlog-heap` cost 4.36x and 4.76x on a boot
/// with logging off, producing 123 bytes of log. `detlog_observed!()` is
/// checked before any of it so the disabled path is genuinely inert, which is
/// the property a default-on check has to have.
pub(crate) fn detlog_io_buffers<G, T>(
    guest: &mut G,
    call: &Syscall,
    ret: i64,
    dettid: DetTid,
    rng_output: Option<&[BufferExtent]>,
) -> Result<(), Error>
where
    G: Guest<T>,
    T: Tool,
{
    if !crate::detlog_observed!() {
        return Ok(());
    }
    let dir = direction(call).as_str();
    let name = call.name();
    // NAME THE DESCRIPTOR THE BYTES CAME THROUGH.
    //
    // A content divergence is classified by WHAT the differing bytes are, and
    // the descriptor is the cheapest available answer to that. Without it,
    // identifying one real case cost reading backward five syscalls through a
    // 5.8 MB log to find `socket(16, 524291, 0) = Ok(11)` and recognising
    // AF_NETLINK by hand; with it, "this is a netlink dump" is on the line that
    // diverged.
    //
    // Safe to add to a byte-for-byte compared surface because the value is
    // ALREADY compared and already stable: the `[syscall]` lines carry these
    // same fds, and across two verify pairs measured 2026-08-25 every
    // fd-bearing syscall line matched with zero differing lines. `-` covers the
    // calls with no single fd argument (getrandom, getcwd, readlink), which is
    // a fact about the call rather than a missing lookup.
    let fd = match crate::syscalls::helpers::get_fd(*call) {
        Some(fd) => fd.to_string(),
        None => "-".to_string(),
    };
    let moved_extents = {
        let memory = guest.memory();
        observed_extents(&memory, call, ret, rng_output)?
    };
    for extent in moved_extents {
        let (whole, chunk, chunks) =
            extent_digests(guest, extent.addr, extent.len, rng_output.is_some()).map_err(
                |error| {
                    if rng_output.is_some() {
                        Error::Tool(anyhow::Error::new(error).context(format!(
                            "RNG vector observation after output commit: digest read at {:#x}+{}",
                            extent.addr, extent.len
                        )))
                    } else {
                        error
                    }
                },
            )?;
        let located = if chunks.is_empty() {
            String::new()
        } else {
            format!(" chunks={}:{}", chunk, chunks.join(","))
        };
        crate::detlog!(
            "[iobuf][dtid {}] {} {} fd={} {:#x}+{}->{}{}",
            dettid,
            name,
            dir,
            fd,
            extent.addr,
            extent.len,
            whole,
            located
        );
    }
    Ok(())
}

#[cfg(test)]
mod event_tests {
    include!("io_buffers/user_access_event_tests.rs");
    use std::io::IoSlice;
    use std::io::IoSliceMut;
    use std::sync::Arc;
    use std::sync::Mutex;

    use nix::fcntl::OFlag;
    use reverie::GlobalRPC;
    use reverie::GlobalTool;
    use reverie::Pid;
    use tokio::sync::oneshot;
    use tracing::Event;
    use tracing::Id;
    use tracing::Level;
    use tracing::Metadata;
    use tracing::Subscriber;
    use tracing::field::Field;
    use tracing::field::Visit;
    use tracing::span::Attributes;
    use tracing::span::Record;

    use super::*;
    use crate::Config;
    use crate::Detcore;
    use crate::GlobalState;
    use crate::ThreadState;
    use crate::fd::DetFd;
    use crate::fd::FdType;
    use crate::resources::Permission;
    use crate::resources::ResourceID;
    use crate::tool_global::GlobalRequest;
    use crate::tool_global::GlobalResponse;
    use crate::tool_global::ResumeStatus;
    use crate::types::DetPid;
    use crate::types::LogicalTime;
    use crate::types::OpenFileId;

    const FD: i32 = 3;
    const IOV: usize = 0x1000;
    const FIRST_DEST: usize = 0x2000;
    const RETRY_DEST: usize = 0x3000;
    const PIPE_BYTES: &[u8; 4] = b"pipe";
    const CANARY: u8 = 0xa5;

    // Numeric guest addresses never become host slices. The fixed arena also
    // makes corrupted descriptors fail promptly rather than allocate by length.
    #[derive(Clone)]
    struct EventMemory(Arc<Mutex<Vec<u8>>>, Arc<Mutex<MemoryReads>>);

    #[derive(Default)]
    struct MemoryReads {
        imported_entries: usize,
        import_error: Option<(usize, Errno)>,
        import_failed: bool,
        after_import_error: Vec<&'static str>,
        observer_reads: Vec<(usize, usize)>,
        digest_error: Option<Errno>,
        user_copy_audit: bool,
        copy_done: bool,
        copy_actions: std::collections::VecDeque<(usize, Result<usize, Errno>)>,
        copy_lengths: Vec<usize>,
        after_copy: Vec<&'static str>,
    }

    impl EventMemory {
        fn new() -> Self {
            Self(
                Arc::new(Mutex::new(vec![CANARY; 0x4000])),
                Arc::new(Mutex::new(MemoryReads::default())),
            )
        }

        fn put_iovec(&self, index: usize, base: usize, len: usize) {
            let start = IOV + index * std::mem::size_of::<libc::iovec>();
            let mut bytes = self.0.lock().unwrap();
            bytes[start..start + 8].copy_from_slice(&base.to_ne_bytes());
            bytes[start + 8..start + 16].copy_from_slice(&len.to_ne_bytes());
        }

        fn iovec(&self, index: usize) -> (usize, usize) {
            let address = IOV + index * std::mem::size_of::<libc::iovec>();
            let iov: libc::iovec = self
                .read_value(Addr::<libc::iovec>::from_raw(address).unwrap())
                .unwrap();
            (iov.iov_base as usize, iov.iov_len)
        }

        fn bytes(&self, address: usize, len: usize) -> Vec<u8> {
            self.0.lock().unwrap()[address..address + len].to_vec()
        }

        fn copy_read(&self, start: usize, buf: &mut [u8]) -> Result<(), Errno> {
            let end = start.checked_add(buf.len()).ok_or(Errno::EFAULT)?;
            let bytes = self.0.lock().unwrap();
            buf.copy_from_slice(bytes.get(start..end).ok_or(Errno::EFAULT)?);
            Ok(())
        }
    }

    impl MemoryAccess for EventMemory {
        fn write_with_user_access(
            &mut self,
            addr: AddrMut<u8>,
            buf: &[u8],
        ) -> Result<usize, Errno> {
            let (written, outcome) = {
                let mut audit = self.1.lock().unwrap();
                audit.copy_lengths.push(buf.len());
                if audit.import_failed {
                    audit.after_import_error.push("user-write");
                }
                audit
                    .copy_actions
                    .pop_front()
                    .unwrap_or((buf.len(), Ok(buf.len())))
            };
            assert!(written <= buf.len());
            let start = addr.as_raw();
            let end = start.checked_add(written).ok_or(Errno::EFAULT)?;
            self.0
                .lock()
                .unwrap()
                .get_mut(start..end)
                .ok_or(Errno::EFAULT)?
                .copy_from_slice(&buf[..written]);
            self.1.lock().unwrap().copy_done = true;
            outcome
        }

        fn read_vectored(
            &self,
            _remote: &[IoSlice],
            _local: &mut [IoSliceMut],
        ) -> Result<usize, Errno> {
            panic!("event fixture must use its scalar memory override")
        }

        fn write_vectored(
            &mut self,
            _local: &[IoSlice],
            _remote: &mut [IoSliceMut],
        ) -> Result<usize, Errno> {
            panic!("event fixture must use its scalar memory override")
        }

        fn read<'a, A>(&self, addr: A, buf: &mut [u8]) -> Result<usize, Errno>
        where
            A: Into<Addr<'a, u8>>,
        {
            let start = addr.into().as_raw();
            let mut reads = self.1.lock().unwrap();
            reads.observer_reads.push((start, buf.len()));
            if reads.import_failed {
                reads.after_import_error.push("memory-read");
            }
            if reads.user_copy_audit && reads.copy_done {
                reads.after_copy.push("memory-read");
            }
            if start == FIRST_DEST
                && let Some(error) = reads.digest_error
            {
                return Err(error);
            }
            drop(reads);
            self.copy_read(start, buf)?;
            Ok(buf.len())
        }

        fn read_exact_with_user_access<'a, A>(&self, addr: A, buf: &mut [u8]) -> Result<(), Errno>
        where
            A: Into<Addr<'a, u8>>,
        {
            {
                let mut reads = self.1.lock().unwrap();
                reads.imported_entries += 1;
                if reads.user_copy_audit && reads.copy_done {
                    reads.after_copy.push("user-read");
                }
                if reads.import_failed {
                    reads.after_import_error.push("user-read");
                }
                if let Some((attempt, error)) = reads.import_error
                    && reads.imported_entries == attempt
                {
                    reads.import_failed = true;
                    return Err(error);
                }
            }
            self.copy_read(addr.into().as_raw(), buf)
        }

        fn write(&mut self, addr: AddrMut<u8>, buf: &[u8]) -> Result<usize, Errno> {
            assert!(
                !self.1.lock().unwrap().user_copy_audit,
                "random output used debugger write"
            );
            let start = addr.as_raw();
            let end = start.checked_add(buf.len()).ok_or(Errno::EFAULT)?;
            let mut bytes = self.0.lock().unwrap();
            bytes
                .get_mut(start..end)
                .ok_or(Errno::EFAULT)?
                .copy_from_slice(buf);
            Ok(buf.len())
        }
    }

    type RetryGate = (oneshot::Sender<()>, oneshot::Receiver<()>);

    struct EventGuest {
        config: Config,
        thread: ThreadState<()>,
        memory: EventMemory,
        injected_iovecs: Vec<(usize, usize)>,
        injected_zero_reads: usize,
        polls: Mutex<Vec<u32>>,
        /// `Resources::backend_runtime_bootstrap` of each resource request,
        /// in the order the requests arrive.
        request_marks: Mutex<Vec<bool>>,
        releases: Mutex<usize>,
        retry_gate: Mutex<Option<RetryGate>>,
        /// What this fake backend answers to
        /// `Guest::is_backend_runtime_bootstrap`.
        backend_runtime_bootstrap: bool,
    }

    impl EventGuest {
        fn audit_call(&self, operation: &'static str) -> bool {
            let mut audit = self.memory.1.lock().unwrap();
            if audit.user_copy_audit && audit.copy_done {
                audit.after_copy.push(operation);
            }
            if audit.import_failed {
                audit.after_import_error.push(operation);
            }
            audit.user_copy_audit
        }
    }

    struct UnusedStack;
    struct UnusedStackGuard;

    impl Drop for UnusedStackGuard {
        fn drop(&mut self) {}
    }

    impl reverie::Stack for UnusedStack {
        type StackGuard = UnusedStackGuard;

        fn size(&self) -> usize {
            panic!("readv event must not use a guest stack")
        }

        fn capacity(&self) -> usize {
            panic!("readv event must not use a guest stack")
        }

        fn push<'stack, T>(&mut self, _value: T) -> Addr<'stack, T> {
            panic!("readv event must not use a guest stack")
        }

        fn reserve<'stack, T>(&mut self) -> AddrMut<'stack, T> {
            panic!("readv event must not use a guest stack")
        }

        fn commit(self) -> Result<Self::StackGuard, Errno> {
            panic!("readv event must not use a guest stack")
        }
    }

    #[reverie::tool]
    impl GlobalRPC<GlobalState> for EventGuest {
        async fn send_rpc(
            &self,
            message: <GlobalState as GlobalTool>::Request,
        ) -> <GlobalState as GlobalTool>::Response {
            let response = match message.2 {
                GlobalRequest::RequestResources(request, _) => {
                    assert_eq!(request.resources.len(), 1);
                    assert_eq!(
                        request.resources.get(&ResourceID::InternalIOPolling),
                        Some(&Permission::W)
                    );
                    self.polls.lock().unwrap().push(request.poll_attempt);
                    self.request_marks
                        .lock()
                        .unwrap()
                        .push(request.backend_runtime_bootstrap);
                    match request.poll_attempt {
                        0 => {}
                        1 => {
                            assert_eq!(self.injected_iovecs, [(FIRST_DEST, 8)]);
                            let (parked, resume) = self.retry_gate.lock().unwrap().take().unwrap();
                            parked.send(()).unwrap();
                            resume.await.unwrap();
                        }
                        attempt => panic!("unexpected extra readv poll {attempt}"),
                    }
                    GlobalResponse::RequestResources(ResumeStatus::Normal)
                }
                GlobalRequest::ReleaseAllResources => {
                    self.audit_call("release");
                    *self.releases.lock().unwrap() += 1;
                    GlobalResponse::ReleaseAllResources(())
                }
                request => panic!("unexpected readv event RPC: {request:?}"),
            };
            (None, response)
        }

        fn config(&self) -> &Config {
            &self.config
        }
    }

    #[reverie::tool]
    impl Guest<Detcore> for EventGuest {
        type Memory = EventMemory;
        type Stack = UnusedStack;

        fn tid(&self) -> Pid {
            Pid::from_raw(self.thread.dettid.as_raw())
        }

        fn pid(&self) -> Pid {
            Pid::from_raw(self.thread.detpid.unwrap().as_raw())
        }

        fn ppid(&self) -> Option<Pid> {
            None
        }

        fn is_backend_runtime_bootstrap(&self) -> bool {
            self.backend_runtime_bootstrap
        }

        fn memory(&self) -> Self::Memory {
            self.memory.clone()
        }

        fn thread_state_mut(&mut self) -> &mut ThreadState<()> {
            &mut self.thread
        }

        fn thread_state(&self) -> &ThreadState<()> {
            &self.thread
        }

        async fn regs(&mut self) -> libc::user_regs_struct {
            if self.audit_call("regs") {
                return libc::user_regs_struct {
                    rip: 0x4000,
                    eflags: 2,
                    ..unsafe { std::mem::zeroed() }
                };
            }
            panic!("buffer observation must not request register evidence")
        }

        async fn set_regs(&mut self, _regs: libc::user_regs_struct) -> Result<(), Error> {
            assert!(self.audit_call("set-regs"));
            Ok(())
        }

        fn detlog_memory_regions(&self) -> Option<Vec<reverie::DetlogMemoryRegion>> {
            if !self.audit_call("memory-regions") {
                return None;
            }
            Some(vec![
                reverie::DetlogMemoryRegion {
                    kind: reverie::DetlogRegionKind::Stack,
                    start: FIRST_DEST as u64,
                    end: (FIRST_DEST + 8) as u64,
                },
                reverie::DetlogMemoryRegion {
                    kind: reverie::DetlogRegionKind::Heap,
                    start: RETRY_DEST as u64,
                    end: (RETRY_DEST + 8) as u64,
                },
            ])
        }

        async fn stack(&mut self) -> Self::Stack {
            panic!("readv event must not use a guest stack")
        }

        async fn daemonize(&mut self) {
            panic!("readv event must not daemonize")
        }

        async fn inject<S: SyscallInfo>(&mut self, syscall: S) -> Result<i64, Errno> {
            let (number, args) = syscall.into_parts();
            if let Syscall::Read(call) = Syscall::from_raw(number, args) {
                assert_eq!(call.len(), 0);
                self.injected_zero_reads += 1;
                if call.fd() == -1 {
                    return Err(Errno::EBADF);
                }
                assert_eq!(call.fd(), FD);
                assert_ne!(
                    self.thread.with_detfd(FD, |fd| fd.ty()).unwrap(),
                    FdType::Rng
                );
                return Err(Errno::EINVAL);
            }
            let Syscall::Readv(call) = Syscall::from_raw(number, args) else {
                panic!("unexpected injected syscall {number}");
            };
            assert_eq!(call.fd(), FD);
            assert_eq!(call.iov().unwrap().as_raw(), IOV);
            assert_eq!(call.len(), 1);
            assert_eq!(
                self.thread.with_detfd(FD, |fd| fd.ty()).unwrap(),
                FdType::Pipe,
                "an RNG readv must be emulated without injection"
            );
            // Each actual nonblocking kernel attempt imports its own array.
            // The first attempt moves nothing; only the retry writes bytes.
            let (base, len) = self.memory.iovec(0);
            self.injected_iovecs.push((base, len));
            match self.injected_iovecs.len() {
                1 => Err(Errno::EAGAIN),
                2 => {
                    assert!(len >= PIPE_BYTES.len());
                    self.memory
                        .write_exact(AddrMut::from_raw(base).unwrap(), PIPE_BYTES)?;
                    Ok(PIPE_BYTES.len() as i64)
                }
                attempt => panic!("unexpected extra readv injection {attempt}"),
            }
        }

        async fn tail_inject<S: SyscallInfo>(&mut self, _syscall: S) -> reverie::Never {
            panic!("readv must return through the event observer")
        }

        fn set_timer(&mut self, _schedule: reverie::TimerSchedule) -> Result<(), Error> {
            if self.audit_call("timer") {
                return Ok(());
            }
            panic!("event fixture has no PMU timeslice")
        }

        fn set_timer_precise(&mut self, _schedule: reverie::TimerSchedule) -> Result<(), Error> {
            if self.audit_call("timer") {
                return Ok(());
            }
            panic!("event fixture has no PMU timeslice")
        }

        fn read_clock(&mut self) -> Result<u64, Error> {
            if self.audit_call("clock") {
                return Ok(0);
            }
            panic!("event fixture must not read a host clock")
        }
    }

    fn event_guest(ty: FdType, retry_gate: Option<RetryGate>) -> (Detcore, EventGuest) {
        let config = Config {
            seed: 0,
            rng_seed: Some(0),
            sequentialize_threads: true,
            recordreplay_modes: false,
            record_preemptions: false,
            max_timeslice: None,
            detlog_io_buffers: true,
            detlog_heap: false,
            detlog_stack: false,
            detlog_regs: false,
            backend_is_kvm: true,
            syscall_clobbers_virtualized_by_backend: true,
            ..Config::default()
        };
        let pid = DetPid::from_raw(1);
        let mut thread = ThreadState::new(pid, &config, ());
        thread.detpid = Some(pid);
        let fd = DetFd::new(FD, OFlag::O_RDONLY, ty, OpenFileId::new(pid, 99));
        if ty == FdType::Pipe {
            fd.set_physically_nonblocking();
        }
        thread
            .file_metadata
            .lock()
            .unwrap()
            .file_handles
            .insert(FD, fd);
        let tool = <Detcore as Tool>::new(Pid::from_raw(1), &config);
        let guest = EventGuest {
            config,
            thread,
            memory: EventMemory::new(),
            injected_iovecs: Vec::new(),
            injected_zero_reads: 0,
            polls: Mutex::new(Vec::new()),
            request_marks: Mutex::new(Vec::new()),
            releases: Mutex::new(0),
            retry_gate: Mutex::new(retry_gate),
            backend_runtime_bootstrap: false,
        };
        (tool, guest)
    }

    /// Collects the INFO messages that contain the second field.
    #[derive(Clone)]
    struct BufferLog(Arc<Mutex<Vec<String>>>, &'static str);

    impl Default for BufferLog {
        fn default() -> Self {
            Self(Arc::default(), "[iobuf]")
        }
    }

    struct MessageVisitor(Option<String>);

    impl Visit for MessageVisitor {
        fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) {
            if field.name() == "message" {
                self.0 = Some(format!("{value:?}"));
            }
        }
    }

    impl Subscriber for BufferLog {
        fn enabled(&self, metadata: &Metadata<'_>) -> bool {
            *metadata.level() == Level::INFO
        }

        fn new_span(&self, _span: &Attributes<'_>) -> Id {
            Id::from_u64(1)
        }

        fn record(&self, _span: &Id, _values: &Record<'_>) {}

        fn record_follows_from(&self, _span: &Id, _follows: &Id) {}

        fn event(&self, event: &Event<'_>) {
            let mut visitor = MessageVisitor(None);
            event.record(&mut visitor);
            if let Some(message) = visitor.0
                && message.contains(self.1)
            {
                self.0.lock().unwrap().push(message);
            }
        }

        fn enter(&self, _span: &Id) {}

        fn exit(&self, _span: &Id) {}
    }

    fn readv(count: usize) -> Syscall {
        reverie::syscalls::Readv::new()
            .with_fd(FD)
            .with_iov(Addr::from_raw(IOV))
            .with_len(count)
            .into()
    }

    fn assert_extent(message: &str, address: usize, bytes: &[u8]) {
        assert_named_extent(message, "readv", address, bytes);
    }

    fn assert_named_extent(message: &str, name: &str, address: usize, bytes: &[u8]) {
        let expected = format!(
            "[iobuf][dtid 1] {name} in fd={FD} {address:#x}+{}->{}",
            bytes.len(),
            Digest::new(bytes)
        );
        assert!(
            message.contains(&expected),
            "expected {expected}, got {message}"
        );
    }

    fn rng_vector_calls(count: usize) -> [(&'static str, Syscall, bool); 4] {
        let preadv = reverie::syscalls::Preadv::new()
            .with_fd(FD)
            .with_iov(Addr::from_raw(IOV))
            .with_iov_len(count)
            .with_pos_l(19)
            .with_pos_h(u64::MAX);
        let preadv2 = reverie::syscalls::Preadv2::new()
            .with_fd(FD)
            .with_iov(Addr::from_raw(IOV))
            .with_iov_len(count as u64)
            .with_pos_l(19)
            .with_pos_h(u64::MAX)
            .with_flags(0);
        [
            ("readv", readv(count), true),
            ("preadv", preadv.into(), false),
            ("preadv2", preadv2.into(), false),
            ("preadv2", preadv2.with_pos_l(u64::MAX).into(), true),
        ]
    }

    /// A syscall issued while the backend reports its own runtime bootstrap
    /// (https://github.com/rrnewton/hermit/issues/3338) is counted and fully
    /// handled, but charges neither the thread's logical clock nor process CPU
    /// time. The first guest syscall after the window resumes from exactly the
    /// clock value the window started at: it charges one syscall's cost, the
    /// same as a run that never had the window.
    #[tokio::test(flavor = "current_thread")]
    async fn backend_runtime_bootstrap_syscall_is_handled_but_not_charged_to_guest_time() {
        struct Observed {
            result: i64,
            bytes: Vec<u8>,
            syscall_count: u64,
            random_offset: u64,
            clock_before: LogicalTime,
            clock_after: LogicalTime,
            system_before: LogicalTime,
            system_after: LogicalTime,
            process_system_before: LogicalTime,
            process_system_after: LogicalTime,
        }

        async fn one_readv(tool: &Detcore, guest: &mut EventGuest) -> Observed {
            let memory = guest.memory.clone();
            memory.put_iovec(0, FIRST_DEST, 8);
            let clock_before = guest.thread.thread_logical_time.as_nanos();
            let system_before = guest.thread.thread_logical_time.system_cpu_time();
            let process_system_before = guest.thread.process_cpu_time().system;
            let result = tool.handle_syscall_event(guest, readv(1)).await.unwrap();
            Observed {
                result,
                bytes: memory.bytes(FIRST_DEST, 8),
                syscall_count: guest.thread.stats.syscall_count,
                random_offset: guest
                    .thread
                    .with_detfd(FD, |fd| fd.random_device_offset())
                    .unwrap(),
                clock_before,
                clock_after: guest.thread.thread_logical_time.as_nanos(),
                system_before,
                system_after: guest.thread.thread_logical_time.system_cpu_time(),
                process_system_before,
                process_system_after: guest.thread.process_cpu_time().system,
            }
        }

        // Reference: an ordinary guest syscall is charged.
        let (tool, mut plain) = event_guest(FdType::Rng, None);
        let guest_call = one_readv(&tool, &mut plain).await;
        assert!(
            guest_call.clock_after > guest_call.clock_before,
            "a guest syscall must advance logical time"
        );
        assert!(guest_call.system_after > guest_call.system_before);
        assert!(guest_call.process_system_after > guest_call.process_system_before);

        // The same syscall inside the backend's bootstrap window.
        let (tool, mut booting) = event_guest(FdType::Rng, None);
        booting.backend_runtime_bootstrap = true;
        let bootstrap_call = one_readv(&tool, &mut booting).await;
        // Still counted and handled exactly like the guest syscall: same
        // result, same emulated random bytes written, same RNG cursor advance,
        // same resource release.
        assert_eq!(bootstrap_call.result, guest_call.result);
        assert_eq!(bootstrap_call.bytes, guest_call.bytes);
        assert_eq!(bootstrap_call.random_offset, guest_call.random_offset);
        assert_eq!(bootstrap_call.syscall_count, 1);
        assert_eq!(*booting.releases.lock().unwrap(), 1);
        // But not charged to the guest.
        assert_eq!(
            bootstrap_call.clock_after, bootstrap_call.clock_before,
            "a backend-bootstrap syscall must not advance guest logical time"
        );
        assert_eq!(bootstrap_call.system_after, bootstrap_call.system_before);
        assert_eq!(
            bootstrap_call.process_system_after,
            bootstrap_call.process_system_before
        );
        assert_eq!(bootstrap_call.clock_before, guest_call.clock_before);

        // The window closes; the next guest syscall resumes from the same clock
        // value (no reset, no jump) and is charged one syscall's cost, so the
        // clock matches the run that never had the window.
        booting.backend_runtime_bootstrap = false;
        let after_window = one_readv(&tool, &mut booting).await;
        assert_eq!(after_window.syscall_count, 2);
        assert_eq!(after_window.clock_before, bootstrap_call.clock_after);
        assert_eq!(after_window.clock_after, guest_call.clock_after);
        assert_eq!(after_window.system_after, guest_call.system_after);
        assert_eq!(
            after_window.process_system_after,
            guest_call.process_system_after
        );
        assert!(after_window.clock_after > after_window.clock_before);
    }

    /// The bootstrapping thread's clock keeps moving inside a backend-runtime
    /// bootstrap window (finding F1 of
    /// https://github.com/rrnewton/hermit/pull/3430#issuecomment-5928691696).
    /// A syscall that observes virtual time is charged exactly as outside the
    /// window, so two reads in a row see different times. Other syscalls are
    /// left uncharged only up to `MAX_UNCHARGED_BOOTSTRAP_SYSCALLS` per window;
    /// the syscall that reaches the cap logs one info line, the next one is
    /// charged its normal cost, and a new window starts counting from zero. The
    /// harness's clock is syscall-driven (`max_timeslice: None`),
    /// the configuration in which nothing else would advance this thread.
    #[tokio::test(flavor = "current_thread")]
    async fn backend_runtime_bootstrap_window_charges_time_reads_and_caps_uncharged_syscalls() {
        use reverie::syscalls::Sysno;

        use crate::syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS;
        use crate::syscall_time::observes_virtual_time;

        const USAGE: usize = RETRY_DEST;

        // getrusage(RUSAGE_THREAD) is the time read this harness can execute:
        // it reports the thread's own CPU time without a global-clock RPC, and
        // every charged syscall adds to that CPU time.
        fn thread_rusage() -> Syscall {
            reverie::syscalls::Getrusage::new()
                .with_who(libc::RUSAGE_THREAD)
                .with_usage(AddrMut::from_raw(USAGE))
                .into()
        }

        /// Runs one syscall and returns how far it advanced the thread's
        /// logical time, in nanoseconds.
        async fn advance(tool: &Detcore, guest: &mut EventGuest, call: Syscall) -> u64 {
            guest.memory.put_iovec(0, FIRST_DEST, 8);
            let before = guest.thread.thread_logical_time.as_nanos().as_nanos();
            tool.handle_syscall_event(guest, call).await.unwrap();
            guest.thread.thread_logical_time.as_nanos().as_nanos() - before
        }

        /// The system CPU time the last getrusage wrote, in nanoseconds.
        fn reported_system_ns(guest: &EventGuest) -> u64 {
            let usage: libc::rusage = guest
                .memory
                .read_value(Addr::<libc::rusage>::from_raw(USAGE).unwrap())
                .unwrap();
            usage.ru_stime.tv_sec as u64 * 1_000_000_000 + usage.ru_stime.tv_usec as u64 * 1_000
        }

        // Every syscall whose main result is a clock or timer value.
        const TIME_READS: [Sysno; 11] = [
            Sysno::gettimeofday,
            Sysno::time,
            Sysno::clock_gettime,
            Sysno::sysinfo,
            Sysno::times,
            Sysno::getrusage,
            Sysno::timerfd_gettime,
            Sysno::timer_gettime,
            Sysno::getitimer,
            Sysno::adjtimex,
            Sysno::clock_adjtime,
        ];
        for sysno in TIME_READS {
            assert!(observes_virtual_time(sysno), "{sysno} reads virtual time");
        }
        assert!(!observes_virtual_time(Sysno::readv));
        assert!(!observes_virtual_time(Sysno::clock_getres));

        // Reference costs outside any window.
        let (tool, mut plain) = event_guest(FdType::Rng, None);
        let time_read_cost = advance(&tool, &mut plain, thread_rusage()).await;
        let first_read_outside = reported_system_ns(&plain);
        let readv_cost = advance(&tool, &mut plain, readv(1)).await;
        assert!(time_read_cost > 0 && readv_cost > 0);

        // (1) Time reads inside the window are charged like time reads outside.
        let (tool, mut booting) = event_guest(FdType::Rng, None);
        booting.backend_runtime_bootstrap = true;
        let first_read_cost = advance(&tool, &mut booting, thread_rusage()).await;
        let first_read = reported_system_ns(&booting);
        let second_read_cost = advance(&tool, &mut booting, thread_rusage()).await;
        let second_read = reported_system_ns(&booting);
        assert_eq!(
            first_read_cost, time_read_cost,
            "a time read inside the window must advance logical time as it does outside"
        );
        assert_eq!(second_read_cost, time_read_cost);
        assert_eq!(first_read, first_read_outside);
        assert_ne!(
            second_read, first_read,
            "two time reads inside the window must observe different times"
        );
        assert_eq!(second_read - first_read, time_read_cost);
        // The other time reads are charged inside the window as well, and
        // none of them counts toward the cap.
        for sysno in TIME_READS {
            assert!(
                booting.thread.charge_syscall_time(true, sysno),
                "{sysno} inside the window must be charged"
            );
        }
        assert_eq!(booting.thread.uncharged_bootstrap_syscalls, 0);

        // (2) The first MAX_UNCHARGED_BOOTSTRAP_SYSCALLS other syscalls of the
        // window are uncharged; the next one, and every one after it, is charged.
        // The window logs one info line, at the syscall that reaches the cap.
        let cap_logs = BufferLog(Arc::default(), "reached its cap");
        let _subscriber = tracing::subscriber::set_default(cap_logs.clone());
        for index in 0..MAX_UNCHARGED_BOOTSTRAP_SYSCALLS {
            if index == MAX_UNCHARGED_BOOTSTRAP_SYSCALLS - 1 {
                assert!(
                    cap_logs.0.lock().unwrap().is_empty(),
                    "the cap line was logged before the window reached the cap"
                );
            }
            assert_eq!(
                advance(&tool, &mut booting, readv(1)).await,
                0,
                "uncharged syscall {index} of the window advanced logical time"
            );
        }
        assert_eq!(
            booting.thread.uncharged_bootstrap_syscalls,
            MAX_UNCHARGED_BOOTSTRAP_SYSCALLS
        );
        let cap_line = vec![format!(
            "[dtid {}] backend runtime bootstrap window reached its cap of {} uncharged syscalls; every further syscall in this window is charged",
            booting.thread.dettid, MAX_UNCHARGED_BOOTSTRAP_SYSCALLS
        )];
        assert_eq!(
            *cap_logs.0.lock().unwrap(),
            cap_line,
            "reaching the cap must log exactly one info line"
        );
        assert_eq!(
            advance(&tool, &mut booting, readv(1)).await,
            readv_cost,
            "the first syscall past the cap must be charged one syscall cost"
        );
        assert_eq!(advance(&tool, &mut booting, readv(1)).await, readv_cost);
        assert_eq!(
            booting.thread.uncharged_bootstrap_syscalls,
            MAX_UNCHARGED_BOOTSTRAP_SYSCALLS
        );
        assert_eq!(
            *cap_logs.0.lock().unwrap(),
            cap_line,
            "syscalls charged past the cap must not log the cap line again"
        );

        // (3) Closing the window resets the count, so a reopened window (the
        // next exec'd image) starts uncharged again.
        booting.backend_runtime_bootstrap = false;
        assert_eq!(advance(&tool, &mut booting, readv(1)).await, readv_cost);
        assert_eq!(booting.thread.uncharged_bootstrap_syscalls, 0);
        booting.backend_runtime_bootstrap = true;
        assert_eq!(advance(&tool, &mut booting, readv(1)).await, 0);
        assert_eq!(booting.thread.uncharged_bootstrap_syscalls, 1);
        assert_eq!(*cap_logs.0.lock().unwrap(), cap_line);

        // (4) The reopened window has the whole cap again, and reaching it logs
        // the window's own info line: the line is once per window, not once per
        // thread or per process.
        for index in 1..MAX_UNCHARGED_BOOTSTRAP_SYSCALLS {
            if index == MAX_UNCHARGED_BOOTSTRAP_SYSCALLS - 1 {
                assert_eq!(
                    *cap_logs.0.lock().unwrap(),
                    cap_line,
                    "the reopened window logged the cap line before reaching the cap"
                );
            }
            assert_eq!(
                advance(&tool, &mut booting, readv(1)).await,
                0,
                "uncharged syscall {index} of the reopened window advanced logical time"
            );
        }
        assert_eq!(
            booting.thread.uncharged_bootstrap_syscalls,
            MAX_UNCHARGED_BOOTSTRAP_SYSCALLS
        );
        let two_cap_lines = vec![cap_line[0].clone(), cap_line[0].clone()];
        assert_eq!(
            *cap_logs.0.lock().unwrap(),
            two_cap_lines,
            "a reopened window that reaches the cap must log its own info line"
        );
        assert_eq!(
            advance(&tool, &mut booting, readv(1)).await,
            readv_cost,
            "the first syscall past the reopened window's cap must be charged"
        );
        assert_eq!(*cap_logs.0.lock().unwrap(), two_cap_lines);
    }

    /// The resource requests a syscall makes are marked as backend-runtime
    /// bootstrap work exactly when that syscall's cost is withheld, so the
    /// scheduler withholds their turn time as well
    /// (https://github.com/rrnewton/hermit/issues/3517). The same blocking pipe
    /// readv, which makes two IO-polling requests, is unmarked as a guest
    /// syscall, marked inside the window, and unmarked again inside the window
    /// once the window's cap makes it a charged syscall. The mark never outlives
    /// the syscall that set it.
    #[tokio::test(flavor = "current_thread")]
    async fn bootstrap_syscall_marks_its_resource_requests_only_while_it_is_uncharged() {
        use crate::syscall_time::MAX_UNCHARGED_BOOTSTRAP_SYSCALLS;

        /// Runs one blocking pipe readv (EAGAIN, then four bytes) and returns
        /// the marks of the requests it made, the clock advance, and the guest.
        async fn pipe_readv(
            in_window: bool,
            uncharged_so_far: u32,
        ) -> (Vec<bool>, u64, EventGuest) {
            let (parked_tx, parked_rx) = oneshot::channel();
            let (resume_tx, resume_rx) = oneshot::channel();
            let (tool, mut guest) = event_guest(FdType::Pipe, Some((parked_tx, resume_rx)));
            guest.backend_runtime_bootstrap = in_window;
            guest.thread.uncharged_bootstrap_syscalls = uncharged_so_far;
            guest.memory.put_iovec(0, FIRST_DEST, 8);
            let before = guest.thread.thread_logical_time.as_nanos().as_nanos();
            let event = tool.handle_syscall_event(&mut guest, readv(1));
            let resume = async {
                parked_rx.await.unwrap();
                resume_tx.send(()).unwrap();
            };
            let (result, ()) = tokio::time::timeout(std::time::Duration::from_secs(2), async {
                tokio::join!(event, resume)
            })
            .await
            .expect("readv did not cross its retry resource wait");
            assert_eq!(result.unwrap(), 4);
            assert_eq!(*guest.polls.lock().unwrap(), [0, 1]);
            assert!(
                !guest.thread.in_uncharged_bootstrap_syscall,
                "the mark must be cleared when the syscall finishes"
            );
            let advance = guest.thread.thread_logical_time.as_nanos().as_nanos() - before;
            let marks = guest.request_marks.lock().unwrap().clone();
            (marks, advance, guest)
        }

        let (guest_marks, guest_cost, _) = pipe_readv(false, 0).await;
        assert!(guest_cost > 0, "a guest syscall must advance logical time");
        assert_eq!(
            guest_marks,
            [false, false],
            "a guest syscall's requests must not be marked"
        );

        let (window_marks, window_cost, booting) = pipe_readv(true, 0).await;
        assert_eq!(window_cost, 0, "the window's syscall must be uncharged");
        assert_eq!(booting.thread.uncharged_bootstrap_syscalls, 1);
        assert_eq!(
            window_marks,
            [true, true],
            "every request of an uncharged window syscall must be marked"
        );

        let (capped_marks, capped_cost, _) =
            pipe_readv(true, MAX_UNCHARGED_BOOTSTRAP_SYSCALLS).await;
        assert_eq!(
            capped_cost, guest_cost,
            "a window syscall past the cap must be charged"
        );
        assert_eq!(
            capped_marks,
            [false, false],
            "a charged window syscall's requests must not be marked"
        );
    }

    #[tokio::test(flavor = "current_thread")]
    async fn pipe_readv_event_observes_destination_imported_after_retry_wait() {
        let logs = BufferLog::default();
        let _subscriber = tracing::subscriber::set_default(logs.clone());
        let (parked_tx, parked_rx) = oneshot::channel();
        let (resume_tx, resume_rx) = oneshot::channel();
        let (tool, mut guest) = event_guest(FdType::Pipe, Some((parked_tx, resume_rx)));
        let memory = guest.memory.clone();
        memory.put_iovec(0, FIRST_DEST, 8);

        // This is the real Detcore syscall-event dispatch and observation path,
        // including its nonblocking EAGAIN retry. No helper supplies geometry.
        let event = tool.handle_syscall_event(&mut guest, readv(1));
        let mutate_during_retry = async {
            parked_rx.await.unwrap();
            assert_eq!(memory.iovec(0), (FIRST_DEST, 8));
            assert_eq!(memory.bytes(FIRST_DEST, 8), [CANARY; 8]);
            memory.put_iovec(0, RETRY_DEST, 8);
            resume_tx.send(()).unwrap();
        };
        let (result, ()) = tokio::time::timeout(std::time::Duration::from_secs(2), async {
            tokio::join!(event, mutate_during_retry)
        })
        .await
        .expect("readv did not cross its retry resource wait");
        assert_eq!(result.unwrap(), 4);
        assert_eq!(guest.injected_iovecs, [(FIRST_DEST, 8), (RETRY_DEST, 8)]);
        assert_eq!(*guest.polls.lock().unwrap(), [0, 1]);
        assert_eq!(*guest.releases.lock().unwrap(), 1);
        assert_eq!(guest.thread.stats.syscall_count, 1);
        assert_eq!(memory.bytes(FIRST_DEST, 8), [CANARY; 8]);
        assert_eq!(
            memory.bytes(RETRY_DEST, 4).as_slice(),
            PIPE_BYTES.as_slice()
        );
        assert_eq!(memory.bytes(RETRY_DEST + 4, 4), [CANARY; 4]);
        let messages = logs.0.lock().unwrap();
        assert_eq!(messages.len(), 1, "{messages:?}");
        assert_extent(&messages[0], RETRY_DEST, PIPE_BYTES);
    }

    #[tokio::test(flavor = "current_thread")]
    async fn rng_readv_event_observes_imported_array_after_output_overwrites_it() {
        for (name, call, advances) in rng_vector_calls(2) {
            let logs = BufferLog::default();
            let _subscriber = tracing::subscriber::set_default(logs.clone());
            let (tool, mut guest) = event_guest(FdType::Rng, None);
            let memory = guest.memory.clone();
            let second_iovec = IOV + std::mem::size_of::<libc::iovec>();
            memory.put_iovec(0, second_iovec, 16);
            memory.put_iovec(1, FIRST_DEST, 4);
            let original_second_iovec = memory.bytes(second_iovec, 16);

            let result = tool.handle_syscall_event(&mut guest, call).await;
            assert_eq!(result.unwrap(), 20);
            let expected = if advances {
                [
                    41, 114, 187, 4, 77, 150, 223, 40, 113, 186, 3, 76, 149, 222, 39, 112, 185, 2,
                    75, 148,
                ]
            } else {
                // Literal seed0 stream at explicit offset19, distinct from cursor0.
                [
                    148, 221, 38, 111, 184, 1, 74, 147, 220, 37, 110, 183, 0, 73, 146, 219, 36,
                    109, 182, 255,
                ]
            };
            assert_eq!(memory.bytes(second_iovec, 16), expected[..16]);
            assert_ne!(memory.bytes(second_iovec, 16), original_second_iovec);
            assert_eq!(memory.bytes(FIRST_DEST, 4), expected[16..]);
            assert_eq!(memory.bytes(FIRST_DEST + 4, 4), [CANARY; 4]);
            assert!(guest.injected_iovecs.is_empty());
            assert!(guest.polls.lock().unwrap().is_empty());
            assert_eq!(*guest.releases.lock().unwrap(), 1);
            assert_eq!(guest.thread.stats.syscall_count, 1);
            assert_eq!(
                guest
                    .thread
                    .with_detfd(FD, |fd| fd.random_device_offset())
                    .unwrap(),
                if advances { 20 } else { 0 }
            );
            let messages = logs.0.lock().unwrap();
            assert_eq!(messages.len(), 2, "{messages:?}");
            assert_named_extent(&messages[0], name, second_iovec, &expected[..16]);
            assert_named_extent(&messages[1], name, FIRST_DEST, &expected[16..]);
        }
    }

    #[tokio::test(flavor = "current_thread")]
    async fn rng_readv_event_digest_failure_is_terminal_after_bytes_and_cursor_commit() {
        for (name, call, advances) in rng_vector_calls(2) {
            for fault in [Errno::EFAULT, Errno::EIO] {
                let logs = BufferLog::default();
                let _subscriber = tracing::subscriber::set_default(logs.clone());
                let (tool, mut guest) = event_guest(FdType::Rng, None);
                let memory = guest.memory.clone();
                memory.put_iovec(0, RETRY_DEST, 3);
                memory.put_iovec(1, FIRST_DEST, 5);
                memory.1.lock().unwrap().digest_error = Some(fault);

                let result = tool.handle_syscall_event(&mut guest, call).await;
                let Err(Error::Tool(error)) = result else {
                    panic!("completed RNG readv digest failure became guest result: {result:?}");
                };
                assert!(
                    error
                        .to_string()
                        .contains("RNG vector observation after output commit")
                );
                assert!(error.chain().any(|cause| {
                matches!(cause.downcast_ref::<Error>(), Some(Error::Errno(error)) if *error == fault)
            }));
                let expected = if advances {
                    [41, 114, 187, 4, 77, 150, 223, 40]
                } else {
                    [148, 221, 38, 111, 184, 1, 74, 147]
                };
                assert_eq!(memory.bytes(RETRY_DEST, 3), expected[..3]);
                assert_eq!(memory.bytes(RETRY_DEST + 3, 1), [CANARY]);
                assert_eq!(memory.bytes(FIRST_DEST, 5), expected[3..]);
                assert_eq!(memory.bytes(FIRST_DEST + 5, 1), [CANARY]);
                assert_eq!(
                    guest
                        .thread
                        .with_detfd(FD, |fd| fd.random_device_offset())
                        .unwrap(),
                    if advances { 8 } else { 0 }
                );
                assert_eq!(*guest.releases.lock().unwrap(), 1);
                assert!(guest.injected_iovecs.is_empty());
                let reads = memory.1.lock().unwrap();
                assert_eq!(reads.imported_entries, 2);
                assert_eq!(reads.observer_reads, [(RETRY_DEST, 3), (FIRST_DEST, 5)]);
                let messages = logs.0.lock().unwrap();
                assert_eq!(messages.len(), 1);
                assert_named_extent(&messages[0], name, RETRY_DEST, &expected[..3]);
            }
        }
    }

    #[tokio::test(flavor = "current_thread")]
    async fn rng_readv_event_observer_configuration_and_subscription_are_inert() {
        for (name, call, advances) in rng_vector_calls((1_usize << 32) | 2) {
            for (configured, subscribed) in [(true, true), (false, true), (true, false)] {
                let logs = BufferLog::default();
                let dispatch = if subscribed {
                    tracing::Dispatch::new(logs.clone())
                } else {
                    tracing::Dispatch::new(tracing::subscriber::NoSubscriber::default())
                };
                let _subscriber = tracing::dispatcher::set_default(&dispatch);
                let (mut tool, mut guest) = event_guest(FdType::Rng, None);
                tool.cfg.detlog_io_buffers = configured;
                guest.config.detlog_io_buffers = configured;
                let memory = guest.memory.clone();
                memory.put_iovec(0, FIRST_DEST, 3);
                memory.put_iovec(1, RETRY_DEST, 5);

                // Import and evidence must narrow the same raw count. In particular,
                // the observer must not fetch 1024 descriptors after a two-entry read.
                let result = tool.handle_syscall_event(&mut guest, call).await;
                assert_eq!(result.unwrap(), 8);
                let expected = if advances {
                    [41, 114, 187, 4, 77, 150, 223, 40]
                } else {
                    [148, 221, 38, 111, 184, 1, 74, 147]
                };
                assert_eq!(memory.bytes(FIRST_DEST, 3), expected[..3]);
                assert_eq!(memory.bytes(FIRST_DEST + 3, 1), [CANARY]);
                assert_eq!(memory.bytes(RETRY_DEST, 5), expected[3..]);
                assert_eq!(memory.bytes(RETRY_DEST + 5, 1), [CANARY]);
                assert_eq!(
                    guest
                        .thread
                        .with_detfd(FD, |fd| fd.random_device_offset())
                        .unwrap(),
                    if advances { 8 } else { 0 }
                );
                assert_eq!(*guest.releases.lock().unwrap(), 1);
                assert!(guest.injected_iovecs.is_empty());
                let reads = memory.1.lock().unwrap();
                assert_eq!(reads.imported_entries, 2);
                let messages = logs.0.lock().unwrap();
                if configured && subscribed {
                    assert_eq!(reads.observer_reads, [(FIRST_DEST, 3), (RETRY_DEST, 5)]);
                    assert_eq!(messages.len(), 2);
                    assert_named_extent(&messages[0], name, FIRST_DEST, &expected[..3]);
                    assert_named_extent(&messages[1], name, RETRY_DEST, &expected[3..]);
                } else {
                    assert!(
                        reads.observer_reads.is_empty(),
                        "disabled observer read guest bytes"
                    );
                    assert!(messages.is_empty(), "disabled observer emitted hashes");
                }
            }
        }
    }

    #[tokio::test(flavor = "current_thread")]
    async fn rng_readv_event_zero_and_malformed_requests_have_no_observer_effects() {
        for (count, length, expected, imports) in [
            (0, 3, Ok(0), 0),
            (1025, 3, Err(Errno::EINVAL), 0),
            (1, usize::MAX, Err(Errno::EINVAL), 1),
        ] {
            for (_, call, _) in rng_vector_calls(count) {
                let logs = BufferLog::default();
                let _subscriber = tracing::subscriber::set_default(logs.clone());
                let (tool, mut guest) = event_guest(FdType::Rng, None);
                let memory = guest.memory.clone();
                memory.put_iovec(0, FIRST_DEST, length);
                let result = tool.handle_syscall_event(&mut guest, call).await;
                assert_eq!(
                    result.map_err(|error| error.into_errno().unwrap()),
                    expected
                );
                assert_eq!(memory.bytes(FIRST_DEST, 8), [CANARY; 8]);
                assert_eq!(
                    guest
                        .thread
                        .with_detfd(FD, |fd| fd.random_device_offset())
                        .unwrap(),
                    0
                );
                assert_eq!(*guest.releases.lock().unwrap(), 1);
                let reads = memory.1.lock().unwrap();
                assert_eq!(reads.imported_entries, imports);
                assert!(reads.observer_reads.is_empty());
                assert!(logs.0.lock().unwrap().is_empty());
            }
        }
    }

    #[tokio::test(flavor = "current_thread")]
    async fn rng_scalar_zero_read_checks_access_without_touching_replay_placeholders() {
        for mode in [
            OFlag::O_RDONLY,
            OFlag::O_RDWR,
            OFlag::O_WRONLY,
            OFlag::O_ACCMODE,
            OFlag::O_PATH,
        ] {
            for address in [0, usize::MAX] {
                let logs = BufferLog::default();
                let _subscriber = tracing::subscriber::set_default(logs.clone());
                let (tool, mut guest) = event_guest(FdType::Rng, None);
                let fd = DetFd::new(
                    FD,
                    mode,
                    FdType::Rng,
                    OpenFileId::new(DetTid::from_raw(1), 99),
                );
                fd.advance_random_device_offset(7);
                guest
                    .thread
                    .file_metadata
                    .lock()
                    .unwrap()
                    .file_handles
                    .insert(FD, fd);
                let call = reverie::syscalls::Read::new()
                    .with_fd(FD)
                    .with_buf(AddrMut::from_raw(address))
                    .with_len(0);
                let result = tool.handle_syscall_event(&mut guest, call.into()).await;
                let expected = if mode == OFlag::O_RDONLY || mode == OFlag::O_RDWR {
                    if address == 0 {
                        Ok(0)
                    } else {
                        Err(Errno::EFAULT)
                    }
                } else {
                    Err(Errno::EBADF)
                };
                assert_eq!(
                    result.map_err(|error| error.into_errno().unwrap()),
                    expected
                );
                assert_eq!(
                    guest
                        .thread
                        .with_detfd(FD, |fd| fd.random_device_offset())
                        .unwrap(),
                    7
                );
                assert!(guest.injected_iovecs.is_empty());
                assert_eq!(guest.injected_zero_reads, 0);
                assert!(guest.polls.lock().unwrap().is_empty());
                let reads = guest.memory.1.lock().unwrap();
                assert_eq!(reads.imported_entries, 0);
                assert!(reads.observer_reads.is_empty());
                assert!(logs.0.lock().unwrap().is_empty());
            }
        }
    }

    #[tokio::test(flavor = "current_thread")]
    async fn non_rng_zero_reads_keep_the_backend_result() {
        for (fd, expected) in [(FD, Errno::EINVAL), (-1, Errno::EBADF)] {
            let (tool, mut guest) = event_guest(FdType::Regular, None);
            let call = reverie::syscalls::Read::new().with_fd(fd).with_len(0);
            let result = tool.handle_syscall_event(&mut guest, call.into()).await;
            assert!(matches!(result, Err(Error::Errno(error)) if error == expected));
            // The incumbent fd precheck rejects -1 before the handler's
            // injected zero-read path. A valid ordinary fd still reaches it.
            assert_eq!(guest.injected_zero_reads, usize::from(fd == FD));
            assert!(guest.memory.1.lock().unwrap().observer_reads.is_empty());
        }
    }
}

#[cfg(test)]
mod tests {
    use reverie::syscalls;
    use reverie::syscalls::LocalMemory;

    #[test]
    fn rng_observation_capacity_failure_is_a_tool_error() {
        let error = rng_observation_vec::<u8>(usize::MAX, "digest payload").unwrap_err();
        let Error::Tool(error) = error else {
            panic!("post-commit allocation failure must not become guest errno");
        };
        assert!(error.to_string().contains(
            "RNG vector observation after output commit: cannot reserve digest payload:"
        ));
        assert!(error.to_string().contains("digest payload"));
    }

    /// A divergence must be LOCATED, not merely detected -- that is the whole
    /// reason the chunk list exists, so it is asserted rather than assumed.
    #[test]
    fn chunk_digests_name_the_first_differing_window() {
        let len = 1468; // the real netlink dump size
        let left = vec![0xABu8; len];
        let mut right = left.clone();
        let offset = 900;
        right[offset] ^= 0xFF; // one flipped byte, nothing else
        let (chunk, l) = chunk_digests(&left);
        let (_, r) = chunk_digests(&right);
        assert_eq!(chunk, 256);
        assert_eq!(l.len(), r.len());
        let first_diff = l.iter().zip(&r).position(|(a, b)| a != b);
        assert_eq!(
            first_diff,
            Some(offset / chunk),
            "must name the window holding the flip"
        );
        // and ONLY that window: a single-byte change must not smear.
        assert_eq!(l.iter().zip(&r).filter(|(a, b)| a != b).count(), 1);
    }

    #[test]
    fn identical_buffers_produce_identical_chunk_lists() {
        let buf = vec![7u8; 1468];
        assert_eq!(chunk_digests(&buf).1, chunk_digests(&buf.clone()).1);
    }

    /// A buffer that fits in one chunk gets no list: it would only repeat the
    /// whole-extent digest, and this check is always on.
    #[test]
    fn a_single_chunk_extent_emits_no_list() {
        assert!(chunk_digests(&vec![0u8; CHUNK_MIN]).1.is_empty());
        assert!(chunk_digests(&[]).1.is_empty());
    }

    /// The log line must stay bounded however large the buffer is.
    #[test]
    fn the_chunk_count_is_capped_and_still_spans_the_extent() {
        for len in [2048usize, 65536, 1 << 20] {
            let (chunk, chunks) = chunk_digests(&vec![0u8; len]);
            assert!(
                chunks.len() <= CHUNK_CAP,
                "len {len} produced {} chunks",
                chunks.len()
            );
            assert!(
                chunk * chunks.len() >= len,
                "chunks must span the whole extent"
            );
        }
    }

    use super::*;

    #[test]
    fn clamp_bounds_the_extent_by_what_was_moved_not_by_capacity() {
        // A 4096-byte buffer that received 10 bytes must hash 10, not 4096:
        // the other 4086 are whatever was there before, which for a stack
        // buffer is previous frames and would report unrelated divergence.
        assert_eq!(
            clamp(Some(0x1000), 4096, 10),
            vec![BufferExtent {
                addr: 0x1000,
                len: 10
            }]
        );
    }

    #[test]
    fn clamp_never_exceeds_capacity() {
        // MSG_TRUNC lets a receive report more bytes than the buffer held.
        assert_eq!(
            clamp(Some(0x1000), 64, 4096),
            vec![BufferExtent {
                addr: 0x1000,
                len: 64
            }]
        );
    }

    #[test]
    fn nothing_is_hashed_for_a_null_buffer_or_an_empty_move() {
        assert!(clamp(None, 4096, 10).is_empty());
        assert!(clamp(Some(0x1000), 4096, 0).is_empty());
        assert!(clamp(Some(0x1000), 4096, -1).is_empty());
    }

    #[test]
    fn poll_hashes_the_whole_array_not_a_prefix() {
        // `poll` writes `revents` on every entry regardless of the return
        // value, so three descriptors are 3 * sizeof(pollfd) = 24 bytes even
        // when the call returns 0 ready.
        assert_eq!(
            whole(Some(0x2000), 3 * std::mem::size_of::<libc::pollfd>() as u64),
            vec![BufferExtent {
                addr: 0x2000,
                len: 24
            }]
        );
        assert!(whole(None, 24).is_empty());
        assert!(whole(Some(0x2000), 0).is_empty());
    }

    /// The gap this file had: `poll(...) = 0` produced no record at all,
    /// because the `ret <= 0` guard sat above the poll arms. The pre-existing
    /// test called `whole()` directly, so it passed throughout -- it could not
    /// see a short-circuit above the code it called.
    ///
    /// This drives `ret_gates_output`, which is the predicate the single guard
    /// consults, so the property is tested where it is decided rather than one
    /// level away from it.
    #[test]
    fn the_poll_family_is_exempt_from_the_return_value_guard() {
        let nfds = 3;
        for call in [
            Syscall::Poll(
                syscalls::Poll::new()
                    .with_fds(AddrMut::from_raw(0x2000))
                    .with_nfds(nfds),
            ),
            Syscall::Ppoll(
                syscalls::Ppoll::new()
                    .with_fds(AddrMut::from_raw(0x2000))
                    .with_nfds(nfds),
            ),
        ] {
            assert!(
                !ret_gates_output(&call),
                "{} returns a count of READY DESCRIPTORS, not a byte count, so a \
                 zero return must not suppress its extent",
                call.name()
            );
        }

        // Everything else must stay gated, or a failed or empty call would hash
        // bytes the kernel never wrote.
        for call in [
            Syscall::Read(syscalls::Read::new()),
            Syscall::Recvmsg(syscalls::Recvmsg::new()),
            Syscall::Recvmmsg(syscalls::Recvmmsg::new()),
            Syscall::Sendmmsg(syscalls::Sendmmsg::new()),
        ] {
            assert!(
                ret_gates_output(&call),
                "{} returns a byte or message count, so a zero return means \
                 nothing was written",
                call.name()
            );
        }
    }

    #[test]
    fn direction_separates_kernel_produced_from_guest_produced() {
        assert_eq!(Direction::In.as_str(), "in");
        assert_eq!(Direction::Out.as_str(), "out");

        assert_eq!(
            direction(&Syscall::Preadv2(syscalls::Preadv2::new())),
            Direction::In
        );
        assert_eq!(
            direction(&Syscall::Pwritev2(syscalls::Pwritev2::new())),
            Direction::Out
        );
        assert_eq!(
            direction(&Syscall::Sendmmsg(syscalls::Sendmmsg::new())),
            Direction::Out
        );
    }

    #[test]
    fn vectored_v2_calls_are_hashed_over_their_complete_iovec_prefix() {
        for sysno in [syscalls::Sysno::preadv2, syscalls::Sysno::pwritev2] {
            assert!(
                HASHED_SYSCALLS.contains(&sysno),
                "{sysno:?} must remain in strict io-buffer observation"
            );
        }

        let first = [0_u8; 4];
        let second = [0_u8; 8];
        let third = [0_u8; 16];
        let iovecs = [
            libc::iovec {
                iov_base: first.as_ptr() as *mut libc::c_void,
                iov_len: first.len(),
            },
            libc::iovec {
                iov_base: second.as_ptr() as *mut libc::c_void,
                iov_len: second.len(),
            },
            libc::iovec {
                iov_base: third.as_ptr() as *mut libc::c_void,
                iov_len: third.len(),
            },
        ];
        let iov = Addr::from_ptr(iovecs.as_ptr());
        let calls = [
            Syscall::Preadv2(syscalls::Preadv2::new().with_iov(iov).with_iov_len(3)),
            Syscall::Pwritev2(syscalls::Pwritev2::new().with_iov(iov).with_iov_len(3)),
        ];
        let expected = vec![
            BufferExtent {
                addr: first.as_ptr() as u64,
                len: 4,
            },
            BufferExtent {
                addr: second.as_ptr() as u64,
                len: 2,
            },
        ];
        let memory = LocalMemory::new();

        for call in calls {
            assert_eq!(
                iovec_extent_arguments(&call),
                Some((iovecs.as_ptr() as usize, 3))
            );
            assert_eq!(extents(&memory, &call, 6).unwrap(), expected);
            assert!(extents(&memory, &call, 0).unwrap().is_empty());
            assert!(extents(&memory, &call, -1).unwrap().is_empty());
        }
    }

    #[test]
    fn vectored_io_extents_stop_at_the_completed_byte_count() {
        let iovecs = [
            libc::iovec {
                iov_base: 0x1000usize as *mut libc::c_void,
                iov_len: 4,
            },
            libc::iovec {
                iov_base: 0x2000usize as *mut libc::c_void,
                iov_len: 8,
            },
            libc::iovec {
                iov_base: 0x3000usize as *mut libc::c_void,
                iov_len: 16,
            },
        ];

        assert_eq!(
            iovec_extents_from_slice(&iovecs, 6),
            vec![
                BufferExtent {
                    addr: 0x1000,
                    len: 4,
                },
                BufferExtent {
                    addr: 0x2000,
                    len: 2,
                },
            ]
        );
        assert!(iovec_extents_from_slice(&iovecs, 0).is_empty());
        assert!(iovec_extents_from_slice(&iovecs, -1).is_empty());
    }

    #[test]
    fn sendmmsg_hashes_only_completed_messages_and_each_reported_byte_prefix() {
        assert!(HASHED_SYSCALLS.contains(&syscalls::Sysno::sendmmsg));

        let first = [0_u8; 4];
        let second = [0_u8; 8];
        let third = [0_u8; 16];
        let first_iovecs = [
            libc::iovec {
                iov_base: first.as_ptr() as *mut libc::c_void,
                iov_len: first.len(),
            },
            libc::iovec {
                iov_base: second.as_ptr() as *mut libc::c_void,
                iov_len: second.len(),
            },
            libc::iovec {
                iov_base: third.as_ptr() as *mut libc::c_void,
                iov_len: third.len(),
            },
        ];
        let fourth = [0_u8; 5];
        let fifth = [0_u8; 7];
        let second_iovecs = [
            libc::iovec {
                iov_base: fourth.as_ptr() as *mut libc::c_void,
                iov_len: fourth.len(),
            },
            libc::iovec {
                iov_base: fifth.as_ptr() as *mut libc::c_void,
                iov_len: fifth.len(),
            },
        ];

        // SAFETY: `mmsghdr` is a plain C record and every field consumed by
        // `extents` is initialized below before LocalMemory reads it.
        let mut headers: [libc::mmsghdr; 2] = unsafe { std::mem::zeroed() };
        headers[0].msg_hdr.msg_iov = first_iovecs.as_ptr() as *mut libc::iovec;
        headers[0].msg_hdr.msg_iovlen = first_iovecs.len();
        headers[0].msg_len = 6;
        headers[1].msg_hdr.msg_iov = second_iovecs.as_ptr() as *mut libc::iovec;
        headers[1].msg_hdr.msg_iovlen = second_iovecs.len();
        headers[1].msg_len = 9;

        let call = Syscall::Sendmmsg(
            syscalls::Sendmmsg::new()
                .with_msgvec(Addr::from_ptr(headers.as_ptr().cast::<libc::msghdr>()))
                .with_vlen(2),
        );
        let memory = LocalMemory::new();
        let first_message = vec![
            BufferExtent {
                addr: first.as_ptr() as u64,
                len: 4,
            },
            BufferExtent {
                addr: second.as_ptr() as u64,
                len: 2,
            },
        ];
        assert_eq!(extents(&memory, &call, 1).unwrap(), first_message);

        let both_messages = vec![
            BufferExtent {
                addr: first.as_ptr() as u64,
                len: 4,
            },
            BufferExtent {
                addr: second.as_ptr() as u64,
                len: 2,
            },
            BufferExtent {
                addr: fourth.as_ptr() as u64,
                len: 5,
            },
            BufferExtent {
                addr: fifth.as_ptr() as u64,
                len: 4,
            },
        ];
        assert_eq!(extents(&memory, &call, 2).unwrap(), both_messages);
        assert!(extents(&memory, &call, 0).unwrap().is_empty());
        assert!(extents(&memory, &call, -1).unwrap().is_empty());
        assert_eq!(completed_mmsghdr_count(1, 2), 1);
    }
}