rust-hdf5 0.6.1

Pure Rust HDF5 library with full read/write and SWMR support
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
2268
2269
2270
2271
2272
2273
2274
2275
2276
2277
2278
2279
2280
2281
2282
2283
2284
2285
2286
2287
2288
2289
2290
2291
2292
2293
2294
2295
2296
2297
2298
2299
2300
2301
2302
2303
2304
2305
2306
2307
2308
2309
2310
2311
2312
2313
2314
2315
2316
2317
2318
2319
2320
2321
2322
2323
2324
2325
2326
2327
2328
2329
2330
2331
2332
2333
2334
2335
2336
2337
2338
2339
2340
2341
2342
2343
2344
2345
2346
2347
2348
2349
2350
2351
2352
2353
2354
2355
2356
2357
2358
2359
2360
2361
2362
2363
2364
2365
2366
2367
2368
2369
2370
2371
2372
2373
2374
2375
2376
2377
2378
2379
2380
2381
2382
2383
2384
2385
2386
2387
2388
2389
2390
2391
2392
2393
2394
2395
2396
2397
2398
2399
2400
2401
2402
2403
2404
2405
2406
2407
2408
2409
2410
2411
2412
2413
2414
2415
2416
2417
2418
2419
2420
2421
2422
2423
2424
2425
2426
2427
2428
2429
2430
2431
2432
2433
2434
2435
2436
2437
2438
2439
2440
2441
2442
2443
2444
2445
2446
2447
2448
2449
2450
2451
2452
2453
2454
2455
2456
2457
2458
2459
2460
2461
2462
2463
2464
2465
2466
2467
2468
2469
2470
2471
2472
2473
2474
2475
2476
2477
2478
2479
2480
2481
2482
2483
2484
2485
2486
2487
2488
2489
2490
2491
2492
2493
2494
2495
2496
2497
2498
2499
2500
2501
2502
2503
2504
2505
2506
2507
2508
2509
2510
2511
2512
2513
2514
2515
2516
2517
2518
2519
2520
2521
2522
2523
2524
2525
2526
2527
2528
2529
2530
2531
2532
2533
2534
2535
2536
2537
2538
2539
2540
2541
2542
2543
2544
2545
2546
2547
2548
2549
2550
2551
2552
2553
2554
2555
2556
2557
2558
2559
2560
2561
2562
2563
2564
2565
2566
2567
2568
2569
2570
2571
2572
2573
2574
2575
2576
2577
2578
2579
2580
2581
2582
2583
2584
2585
2586
2587
2588
2589
2590
2591
2592
2593
2594
2595
2596
2597
2598
2599
2600
2601
2602
2603
2604
2605
2606
2607
2608
2609
2610
2611
2612
2613
2614
2615
2616
2617
2618
2619
2620
2621
2622
2623
2624
2625
2626
2627
2628
2629
2630
2631
2632
2633
2634
2635
2636
2637
2638
2639
2640
2641
2642
2643
2644
2645
2646
2647
2648
2649
2650
2651
2652
2653
2654
2655
2656
2657
2658
2659
2660
2661
2662
2663
2664
2665
2666
2667
2668
2669
2670
2671
2672
2673
2674
2675
2676
2677
2678
2679
2680
2681
2682
2683
2684
2685
2686
2687
2688
2689
2690
2691
2692
2693
2694
2695
2696
2697
2698
2699
2700
2701
2702
2703
2704
2705
2706
2707
2708
2709
2710
2711
2712
2713
2714
2715
2716
2717
2718
2719
2720
2721
2722
2723
2724
2725
2726
2727
2728
2729
2730
2731
2732
2733
2734
2735
2736
2737
2738
2739
2740
2741
2742
2743
2744
2745
2746
2747
2748
2749
2750
2751
2752
2753
2754
2755
2756
2757
2758
2759
2760
2761
2762
2763
2764
2765
2766
2767
2768
2769
2770
2771
2772
2773
2774
2775
2776
2777
2778
2779
2780
2781
2782
2783
2784
2785
2786
2787
2788
2789
2790
2791
2792
2793
2794
2795
2796
2797
2798
2799
2800
2801
2802
2803
2804
2805
2806
2807
2808
2809
2810
2811
2812
2813
2814
2815
2816
2817
2818
2819
2820
2821
2822
2823
2824
2825
2826
2827
2828
2829
2830
2831
2832
2833
2834
2835
2836
2837
2838
2839
2840
2841
2842
2843
2844
2845
2846
2847
2848
2849
2850
2851
2852
2853
2854
2855
2856
2857
2858
2859
2860
2861
2862
2863
2864
2865
2866
2867
2868
2869
2870
2871
2872
2873
2874
2875
2876
2877
2878
2879
2880
2881
2882
2883
2884
2885
2886
2887
2888
2889
2890
2891
2892
2893
2894
2895
2896
2897
2898
2899
2900
2901
2902
2903
2904
2905
2906
2907
2908
2909
2910
2911
2912
2913
2914
2915
2916
2917
2918
2919
2920
2921
2922
2923
2924
2925
2926
2927
2928
2929
2930
2931
2932
2933
2934
2935
2936
2937
2938
2939
2940
2941
2942
2943
2944
2945
2946
2947
2948
2949
2950
2951
2952
2953
2954
2955
2956
2957
2958
2959
2960
2961
2962
2963
2964
2965
2966
2967
2968
2969
2970
2971
2972
2973
2974
2975
2976
2977
2978
2979
2980
2981
2982
2983
2984
2985
2986
2987
2988
2989
2990
2991
2992
2993
2994
2995
2996
2997
2998
2999
3000
3001
3002
3003
3004
3005
3006
3007
3008
3009
3010
3011
3012
3013
3014
3015
3016
3017
3018
3019
3020
3021
3022
3023
3024
3025
3026
3027
3028
3029
3030
3031
3032
3033
3034
3035
3036
3037
3038
3039
3040
3041
3042
3043
3044
3045
3046
3047
3048
3049
3050
3051
3052
3053
3054
3055
3056
3057
3058
3059
3060
3061
3062
3063
3064
3065
3066
3067
3068
3069
3070
3071
3072
3073
3074
3075
3076
3077
3078
3079
3080
3081
3082
3083
3084
3085
3086
3087
3088
3089
3090
3091
3092
3093
3094
3095
3096
3097
3098
3099
3100
3101
3102
3103
3104
3105
3106
3107
3108
3109
3110
3111
3112
3113
3114
3115
3116
3117
3118
3119
3120
3121
3122
3123
3124
3125
3126
3127
3128
3129
3130
3131
3132
3133
3134
3135
3136
3137
3138
3139
3140
3141
3142
3143
3144
3145
3146
3147
3148
3149
3150
3151
3152
3153
3154
3155
3156
3157
3158
3159
3160
3161
3162
3163
3164
3165
3166
3167
3168
3169
3170
3171
3172
3173
3174
3175
3176
3177
3178
3179
3180
3181
3182
3183
3184
3185
3186
3187
3188
3189
3190
3191
3192
3193
3194
3195
3196
3197
3198
3199
3200
3201
3202
3203
3204
3205
3206
3207
3208
3209
3210
3211
3212
3213
3214
3215
3216
3217
3218
3219
3220
3221
3222
3223
3224
3225
3226
3227
3228
3229
3230
3231
3232
3233
3234
3235
3236
3237
3238
3239
3240
3241
3242
3243
3244
3245
3246
3247
3248
3249
3250
3251
3252
3253
3254
3255
3256
3257
3258
3259
3260
3261
3262
3263
3264
3265
3266
3267
3268
3269
3270
3271
3272
3273
3274
3275
3276
3277
3278
3279
3280
3281
3282
3283
3284
3285
3286
3287
3288
3289
3290
3291
3292
3293
3294
3295
3296
3297
3298
3299
3300
3301
3302
3303
3304
3305
3306
3307
3308
3309
3310
3311
3312
3313
3314
3315
3316
3317
3318
3319
3320
3321
3322
3323
3324
3325
3326
3327
3328
3329
3330
3331
3332
3333
3334
3335
3336
3337
3338
3339
3340
3341
3342
3343
3344
3345
3346
3347
3348
3349
3350
3351
3352
3353
3354
3355
3356
3357
3358
3359
3360
3361
3362
3363
3364
3365
3366
3367
3368
3369
3370
3371
3372
3373
3374
3375
3376
3377
3378
3379
3380
3381
3382
3383
3384
3385
3386
3387
3388
3389
3390
3391
3392
3393
3394
3395
3396
3397
3398
3399
3400
3401
3402
3403
3404
3405
3406
3407
3408
3409
3410
3411
3412
3413
3414
3415
3416
3417
3418
3419
3420
3421
3422
3423
3424
3425
3426
3427
3428
3429
3430
3431
3432
3433
3434
3435
3436
3437
3438
3439
3440
3441
3442
3443
3444
3445
3446
3447
3448
3449
3450
3451
3452
3453
3454
3455
3456
3457
3458
3459
3460
3461
3462
3463
3464
3465
3466
3467
3468
3469
3470
3471
3472
3473
3474
3475
3476
3477
3478
3479
3480
3481
3482
3483
3484
3485
3486
3487
3488
3489
3490
3491
3492
3493
3494
3495
3496
3497
3498
3499
3500
3501
3502
3503
3504
3505
3506
3507
3508
3509
3510
3511
3512
3513
3514
3515
3516
3517
3518
3519
3520
3521
3522
3523
3524
3525
3526
3527
3528
3529
3530
3531
3532
3533
3534
3535
3536
3537
3538
3539
3540
3541
3542
3543
3544
3545
3546
3547
3548
3549
3550
3551
3552
3553
3554
3555
3556
3557
3558
3559
3560
3561
3562
3563
3564
3565
3566
3567
3568
3569
3570
3571
3572
3573
3574
3575
3576
3577
3578
3579
3580
3581
3582
3583
3584
3585
3586
3587
3588
3589
3590
3591
3592
3593
3594
3595
3596
3597
3598
3599
3600
3601
3602
3603
3604
3605
3606
3607
3608
3609
3610
3611
3612
3613
3614
3615
3616
3617
3618
3619
3620
3621
3622
3623
3624
3625
3626
3627
3628
3629
3630
3631
3632
3633
3634
3635
3636
3637
3638
3639
3640
3641
3642
3643
3644
3645
3646
3647
3648
3649
3650
3651
3652
3653
3654
3655
3656
3657
3658
3659
3660
3661
3662
3663
3664
3665
3666
3667
3668
3669
3670
3671
3672
3673
3674
3675
3676
3677
3678
3679
3680
3681
3682
3683
3684
3685
3686
3687
3688
3689
3690
3691
3692
3693
3694
3695
3696
3697
3698
3699
3700
3701
3702
3703
3704
3705
3706
3707
3708
3709
3710
3711
3712
3713
3714
3715
3716
3717
3718
3719
3720
3721
3722
3723
3724
3725
3726
3727
3728
3729
3730
3731
3732
3733
3734
3735
3736
3737
3738
3739
3740
3741
3742
3743
3744
3745
3746
3747
3748
3749
3750
3751
3752
3753
3754
3755
3756
3757
3758
3759
3760
3761
3762
3763
3764
3765
3766
3767
3768
3769
3770
3771
3772
3773
3774
3775
3776
3777
3778
3779
3780
3781
3782
3783
3784
3785
3786
3787
3788
3789
3790
3791
3792
3793
3794
3795
3796
3797
3798
3799
3800
3801
3802
3803
3804
3805
3806
3807
3808
3809
3810
3811
3812
3813
3814
3815
3816
3817
3818
3819
3820
3821
3822
3823
3824
3825
3826
3827
3828
3829
3830
3831
3832
3833
3834
3835
3836
3837
3838
3839
3840
3841
3842
3843
3844
3845
3846
3847
3848
3849
3850
3851
3852
3853
3854
3855
3856
3857
3858
3859
3860
3861
3862
3863
3864
3865
3866
3867
3868
3869
3870
3871
3872
3873
3874
3875
3876
3877
3878
3879
3880
3881
3882
3883
3884
3885
3886
3887
3888
3889
3890
3891
3892
3893
3894
3895
3896
3897
3898
3899
3900
3901
3902
3903
3904
3905
3906
3907
3908
3909
3910
3911
3912
3913
3914
3915
3916
3917
3918
3919
3920
3921
3922
3923
3924
3925
3926
3927
3928
3929
3930
3931
3932
3933
3934
3935
3936
3937
3938
3939
3940
3941
3942
3943
3944
3945
3946
3947
3948
3949
3950
3951
3952
3953
3954
3955
3956
3957
3958
3959
3960
3961
3962
3963
3964
3965
3966
3967
3968
3969
3970
3971
3972
3973
3974
3975
3976
3977
3978
3979
3980
3981
3982
3983
3984
3985
3986
3987
3988
3989
3990
3991
3992
3993
3994
3995
3996
3997
3998
3999
4000
4001
4002
4003
4004
4005
4006
4007
4008
4009
4010
4011
4012
4013
4014
4015
4016
4017
4018
4019
4020
4021
4022
4023
4024
4025
4026
4027
4028
4029
4030
4031
4032
4033
4034
4035
4036
4037
4038
4039
4040
4041
4042
4043
4044
4045
4046
4047
4048
4049
4050
4051
4052
4053
4054
4055
4056
4057
4058
4059
4060
4061
4062
4063
4064
4065
4066
4067
4068
4069
4070
4071
4072
4073
4074
4075
4076
4077
4078
4079
4080
4081
4082
4083
4084
4085
4086
4087
4088
4089
4090
4091
4092
4093
4094
4095
4096
4097
4098
4099
4100
4101
4102
4103
4104
4105
4106
4107
4108
4109
4110
4111
4112
4113
4114
4115
4116
4117
4118
4119
4120
4121
4122
4123
4124
4125
4126
4127
4128
4129
4130
4131
4132
4133
4134
4135
4136
4137
4138
4139
4140
4141
4142
4143
4144
4145
4146
4147
4148
4149
4150
4151
4152
4153
4154
4155
4156
4157
4158
4159
4160
4161
4162
4163
4164
4165
4166
4167
4168
4169
4170
4171
4172
4173
4174
4175
4176
4177
4178
4179
4180
4181
4182
4183
4184
4185
4186
4187
4188
4189
4190
4191
4192
4193
4194
4195
4196
4197
4198
4199
4200
4201
4202
4203
4204
4205
4206
4207
4208
4209
4210
4211
4212
4213
4214
4215
4216
4217
4218
4219
4220
4221
4222
4223
4224
4225
4226
4227
4228
4229
4230
4231
4232
4233
4234
4235
4236
4237
4238
4239
4240
4241
4242
4243
4244
4245
4246
4247
4248
4249
4250
4251
4252
4253
4254
4255
4256
4257
4258
4259
4260
4261
4262
4263
4264
4265
4266
4267
4268
4269
4270
4271
4272
4273
4274
4275
4276
4277
4278
4279
4280
4281
4282
4283
4284
4285
4286
4287
4288
4289
4290
4291
4292
4293
4294
4295
4296
4297
4298
4299
4300
4301
4302
4303
4304
4305
4306
4307
4308
4309
4310
4311
4312
4313
4314
4315
4316
4317
4318
4319
4320
4321
4322
4323
4324
4325
4326
4327
4328
4329
4330
4331
4332
4333
4334
4335
4336
4337
4338
4339
4340
4341
4342
4343
4344
4345
4346
4347
4348
4349
4350
4351
4352
4353
4354
4355
4356
4357
4358
4359
4360
4361
4362
4363
4364
4365
4366
4367
4368
4369
4370
4371
4372
4373
4374
4375
4376
4377
4378
4379
4380
4381
4382
4383
4384
4385
4386
4387
4388
4389
4390
4391
4392
4393
4394
4395
4396
4397
4398
4399
4400
4401
4402
4403
4404
4405
4406
4407
4408
4409
4410
4411
4412
4413
4414
4415
4416
4417
4418
4419
4420
4421
4422
4423
4424
4425
4426
4427
4428
4429
4430
4431
4432
4433
4434
4435
4436
4437
4438
4439
4440
4441
4442
4443
4444
4445
4446
4447
4448
4449
4450
4451
4452
4453
4454
4455
4456
4457
4458
4459
4460
4461
4462
4463
4464
4465
4466
4467
4468
4469
4470
4471
4472
4473
4474
4475
4476
4477
4478
4479
4480
4481
4482
4483
4484
4485
4486
4487
4488
4489
4490
4491
4492
4493
4494
4495
4496
4497
4498
4499
4500
4501
4502
4503
4504
4505
4506
4507
4508
4509
4510
4511
4512
4513
4514
4515
4516
4517
4518
4519
4520
4521
4522
4523
4524
4525
4526
4527
4528
4529
4530
4531
4532
4533
4534
4535
4536
4537
4538
4539
4540
4541
4542
4543
4544
4545
4546
4547
4548
4549
4550
4551
4552
4553
4554
4555
4556
4557
4558
4559
4560
4561
4562
4563
4564
4565
4566
4567
4568
4569
4570
4571
4572
4573
4574
4575
4576
4577
4578
4579
4580
4581
4582
4583
4584
4585
4586
4587
4588
4589
4590
4591
4592
4593
4594
4595
4596
4597
4598
4599
4600
4601
4602
4603
4604
4605
4606
4607
4608
4609
4610
4611
4612
4613
4614
4615
4616
4617
4618
4619
4620
4621
4622
4623
4624
4625
4626
4627
4628
4629
4630
4631
4632
4633
4634
4635
4636
4637
4638
4639
4640
4641
4642
4643
4644
4645
4646
4647
4648
4649
4650
4651
4652
4653
4654
4655
4656
4657
4658
4659
4660
4661
4662
4663
4664
4665
4666
4667
4668
4669
4670
4671
4672
4673
4674
4675
4676
4677
4678
4679
4680
4681
4682
4683
4684
4685
4686
4687
4688
4689
4690
4691
4692
4693
4694
4695
4696
4697
4698
4699
4700
4701
4702
4703
4704
4705
4706
4707
4708
4709
4710
4711
4712
4713
4714
4715
4716
4717
4718
4719
4720
4721
4722
4723
4724
4725
4726
4727
4728
4729
4730
4731
4732
4733
4734
4735
4736
4737
4738
4739
4740
4741
4742
4743
4744
4745
4746
4747
4748
4749
4750
4751
4752
4753
4754
4755
4756
4757
4758
4759
4760
4761
4762
4763
4764
4765
4766
4767
4768
4769
4770
4771
4772
4773
4774
4775
4776
4777
4778
4779
4780
4781
4782
4783
4784
4785
4786
4787
4788
4789
4790
4791
4792
4793
4794
4795
4796
4797
4798
4799
4800
4801
4802
4803
4804
4805
4806
4807
4808
4809
4810
4811
4812
4813
4814
4815
4816
4817
4818
4819
4820
4821
4822
4823
4824
4825
4826
4827
4828
4829
4830
4831
4832
4833
4834
4835
4836
4837
4838
4839
4840
4841
4842
4843
4844
4845
4846
4847
4848
4849
4850
4851
4852
4853
4854
4855
4856
4857
4858
4859
4860
4861
4862
4863
4864
4865
4866
4867
4868
4869
4870
4871
4872
4873
4874
4875
4876
4877
4878
4879
4880
4881
4882
4883
4884
4885
4886
4887
4888
4889
4890
4891
4892
4893
4894
4895
4896
4897
4898
4899
4900
4901
4902
4903
4904
4905
4906
4907
4908
4909
4910
4911
4912
4913
4914
4915
4916
4917
4918
4919
4920
4921
4922
4923
4924
4925
4926
4927
4928
4929
4930
4931
4932
4933
4934
4935
4936
4937
4938
4939
4940
4941
4942
4943
4944
4945
4946
4947
4948
4949
4950
4951
4952
4953
4954
4955
4956
4957
4958
4959
4960
4961
4962
4963
4964
4965
4966
4967
4968
4969
4970
4971
4972
4973
4974
4975
4976
4977
4978
4979
4980
4981
4982
4983
4984
4985
4986
4987
4988
4989
4990
4991
4992
4993
4994
4995
4996
4997
4998
4999
5000
5001
5002
5003
5004
5005
5006
5007
5008
5009
5010
5011
5012
5013
5014
5015
5016
5017
5018
5019
5020
5021
5022
5023
5024
5025
5026
5027
5028
5029
5030
5031
5032
5033
5034
5035
5036
5037
5038
5039
5040
5041
5042
5043
5044
5045
5046
5047
5048
5049
5050
5051
5052
5053
5054
5055
5056
5057
5058
5059
5060
5061
5062
5063
5064
5065
5066
5067
5068
5069
5070
5071
5072
5073
5074
5075
5076
5077
5078
5079
5080
5081
5082
5083
5084
5085
5086
5087
5088
5089
5090
5091
5092
5093
5094
5095
5096
5097
5098
5099
5100
5101
5102
5103
5104
5105
5106
5107
5108
5109
5110
5111
5112
5113
5114
5115
5116
5117
5118
5119
5120
5121
5122
5123
5124
5125
5126
5127
5128
5129
5130
5131
5132
5133
5134
5135
5136
5137
5138
5139
5140
5141
5142
5143
5144
5145
5146
5147
5148
5149
5150
5151
5152
5153
5154
5155
5156
5157
5158
5159
5160
5161
5162
5163
5164
5165
5166
5167
5168
5169
5170
5171
5172
5173
5174
5175
5176
5177
5178
5179
5180
5181
5182
5183
5184
5185
5186
5187
5188
5189
5190
5191
5192
5193
5194
5195
5196
5197
5198
5199
5200
5201
5202
5203
5204
5205
5206
5207
5208
5209
5210
5211
5212
5213
5214
5215
5216
5217
5218
5219
5220
5221
5222
5223
5224
5225
5226
5227
5228
5229
5230
5231
5232
5233
5234
5235
5236
5237
5238
5239
5240
5241
5242
5243
5244
5245
5246
5247
5248
5249
5250
5251
5252
5253
5254
5255
5256
5257
5258
5259
5260
5261
5262
5263
5264
5265
5266
5267
5268
5269
5270
5271
5272
5273
5274
5275
5276
5277
5278
5279
5280
5281
5282
5283
5284
5285
5286
5287
5288
5289
5290
5291
5292
5293
5294
5295
5296
5297
5298
5299
5300
5301
5302
5303
5304
5305
5306
5307
5308
5309
5310
5311
5312
5313
5314
5315
5316
5317
5318
5319
5320
5321
5322
5323
5324
5325
5326
5327
5328
5329
5330
5331
5332
5333
5334
5335
5336
5337
5338
5339
5340
5341
5342
5343
5344
5345
5346
5347
5348
5349
5350
5351
5352
5353
5354
5355
5356
5357
5358
5359
5360
5361
5362
5363
5364
5365
5366
5367
5368
5369
5370
5371
5372
5373
5374
5375
5376
5377
5378
5379
5380
5381
5382
5383
5384
5385
5386
5387
5388
5389
5390
5391
5392
5393
5394
5395
5396
5397
5398
5399
5400
5401
5402
5403
5404
5405
5406
5407
5408
5409
5410
5411
5412
5413
5414
5415
5416
5417
5418
5419
5420
5421
5422
5423
5424
5425
5426
5427
5428
5429
5430
5431
5432
5433
5434
5435
5436
5437
5438
5439
5440
5441
5442
5443
5444
5445
5446
5447
5448
5449
5450
5451
5452
5453
5454
5455
5456
5457
5458
5459
5460
5461
5462
5463
5464
5465
5466
5467
5468
5469
5470
5471
5472
5473
5474
5475
5476
5477
5478
5479
5480
5481
5482
5483
5484
5485
5486
5487
5488
5489
5490
5491
5492
5493
5494
5495
5496
5497
5498
5499
5500
5501
5502
5503
5504
5505
5506
5507
5508
5509
5510
5511
5512
5513
5514
5515
5516
5517
5518
5519
5520
5521
5522
5523
5524
5525
5526
5527
5528
5529
5530
5531
5532
5533
5534
5535
5536
5537
5538
5539
5540
5541
5542
5543
5544
5545
5546
5547
5548
5549
5550
5551
5552
5553
5554
5555
5556
5557
5558
5559
5560
5561
5562
5563
5564
5565
5566
5567
5568
5569
5570
5571
5572
5573
5574
5575
5576
5577
5578
5579
5580
5581
5582
5583
5584
5585
5586
5587
5588
5589
5590
5591
5592
5593
5594
5595
5596
5597
5598
5599
5600
5601
5602
5603
5604
5605
5606
5607
5608
5609
5610
5611
5612
5613
5614
5615
5616
5617
5618
5619
5620
5621
5622
5623
5624
5625
5626
5627
5628
5629
5630
5631
5632
5633
5634
5635
5636
5637
5638
5639
5640
5641
5642
5643
5644
5645
5646
5647
5648
5649
5650
5651
5652
5653
5654
5655
5656
5657
5658
5659
5660
5661
5662
5663
5664
5665
5666
5667
5668
5669
5670
5671
5672
5673
5674
5675
5676
5677
5678
5679
5680
5681
5682
5683
5684
5685
5686
5687
5688
5689
5690
5691
5692
5693
5694
5695
5696
5697
5698
5699
5700
5701
5702
5703
5704
5705
5706
5707
5708
5709
5710
5711
5712
5713
5714
5715
5716
5717
5718
5719
5720
5721
5722
5723
5724
5725
5726
5727
5728
5729
5730
5731
5732
5733
5734
5735
5736
5737
5738
5739
5740
5741
5742
5743
5744
5745
5746
5747
5748
5749
5750
5751
5752
5753
5754
5755
5756
5757
5758
5759
5760
5761
5762
5763
5764
5765
5766
5767
5768
5769
5770
5771
5772
5773
5774
5775
5776
5777
5778
5779
5780
5781
5782
5783
5784
5785
5786
5787
5788
5789
5790
5791
5792
5793
5794
5795
5796
5797
5798
5799
5800
5801
5802
5803
5804
5805
5806
5807
5808
5809
5810
5811
5812
5813
5814
5815
5816
5817
5818
5819
5820
5821
5822
5823
5824
5825
5826
5827
5828
5829
5830
5831
5832
5833
5834
5835
5836
5837
5838
5839
5840
5841
5842
5843
5844
5845
5846
5847
5848
5849
5850
5851
5852
5853
5854
5855
5856
5857
5858
5859
5860
5861
5862
5863
5864
5865
5866
5867
5868
5869
5870
5871
5872
5873
5874
5875
5876
5877
5878
5879
5880
5881
5882
5883
5884
5885
5886
5887
5888
5889
5890
5891
5892
5893
5894
5895
5896
5897
5898
5899
5900
5901
5902
5903
5904
5905
5906
5907
5908
5909
5910
5911
5912
5913
5914
5915
5916
5917
5918
5919
5920
5921
5922
5923
5924
5925
5926
5927
5928
5929
5930
5931
5932
5933
5934
5935
5936
5937
5938
5939
5940
5941
5942
5943
5944
5945
5946
5947
5948
5949
5950
5951
5952
5953
5954
5955
5956
5957
5958
5959
5960
5961
5962
5963
5964
5965
5966
5967
5968
5969
5970
5971
5972
5973
5974
5975
5976
5977
5978
5979
5980
5981
5982
5983
5984
5985
5986
5987
5988
5989
5990
5991
5992
5993
5994
5995
5996
5997
5998
5999
6000
6001
6002
6003
6004
6005
6006
6007
6008
6009
6010
6011
6012
6013
6014
6015
6016
6017
6018
6019
6020
6021
6022
6023
6024
6025
6026
6027
6028
6029
6030
6031
6032
6033
6034
6035
6036
6037
6038
6039
6040
6041
6042
6043
6044
6045
6046
6047
6048
6049
6050
6051
6052
6053
6054
6055
6056
6057
6058
6059
6060
6061
6062
6063
6064
6065
6066
6067
6068
6069
6070
6071
6072
6073
6074
6075
6076
6077
6078
6079
6080
6081
6082
6083
6084
6085
6086
6087
6088
6089
6090
6091
6092
6093
6094
6095
6096
6097
6098
6099
6100
6101
6102
6103
6104
6105
6106
6107
6108
6109
6110
6111
6112
6113
6114
6115
6116
6117
6118
6119
6120
6121
6122
6123
6124
6125
6126
6127
6128
6129
6130
6131
6132
6133
6134
6135
6136
6137
6138
6139
6140
6141
6142
6143
6144
6145
6146
6147
6148
6149
6150
6151
6152
6153
6154
6155
6156
6157
6158
6159
6160
6161
6162
6163
6164
6165
6166
6167
6168
6169
6170
6171
6172
6173
6174
6175
6176
6177
6178
6179
6180
6181
6182
6183
6184
6185
6186
6187
6188
6189
6190
6191
6192
6193
6194
6195
6196
6197
6198
6199
6200
6201
6202
6203
6204
6205
6206
6207
6208
6209
6210
6211
6212
6213
6214
6215
6216
6217
6218
6219
6220
6221
6222
6223
6224
6225
6226
6227
6228
6229
6230
6231
6232
6233
6234
6235
6236
6237
6238
6239
6240
6241
6242
6243
6244
6245
6246
6247
6248
6249
6250
6251
6252
6253
6254
6255
6256
6257
6258
6259
6260
6261
6262
6263
6264
6265
6266
6267
6268
6269
6270
6271
6272
6273
6274
6275
6276
6277
6278
6279
6280
6281
6282
6283
6284
6285
6286
6287
6288
6289
6290
6291
6292
6293
6294
6295
6296
6297
6298
6299
6300
6301
6302
6303
6304
6305
6306
6307
6308
6309
6310
6311
6312
6313
6314
6315
6316
6317
6318
6319
6320
6321
6322
6323
6324
6325
6326
6327
6328
6329
6330
6331
6332
6333
6334
6335
6336
6337
6338
6339
6340
6341
6342
6343
6344
6345
6346
6347
6348
6349
6350
6351
6352
6353
6354
6355
6356
6357
6358
6359
6360
6361
6362
6363
6364
6365
6366
6367
6368
6369
6370
6371
6372
6373
6374
6375
6376
6377
6378
6379
6380
6381
6382
6383
6384
6385
6386
6387
6388
6389
6390
6391
6392
6393
6394
6395
6396
6397
6398
6399
6400
6401
6402
6403
6404
6405
6406
6407
6408
6409
6410
6411
6412
6413
6414
6415
6416
6417
6418
6419
6420
6421
6422
6423
6424
6425
6426
6427
6428
6429
6430
6431
6432
6433
6434
6435
6436
6437
6438
6439
6440
6441
6442
6443
6444
6445
6446
6447
6448
6449
6450
6451
6452
6453
6454
6455
6456
6457
6458
6459
6460
6461
6462
6463
6464
6465
6466
6467
6468
6469
6470
6471
6472
6473
6474
6475
6476
6477
6478
6479
6480
6481
6482
6483
6484
6485
6486
6487
6488
6489
6490
6491
6492
6493
6494
6495
6496
6497
6498
6499
6500
6501
6502
6503
6504
6505
6506
6507
6508
6509
6510
6511
6512
6513
6514
6515
6516
6517
6518
6519
6520
6521
6522
6523
6524
6525
6526
6527
6528
6529
6530
6531
6532
6533
6534
6535
6536
6537
6538
6539
6540
6541
6542
6543
6544
6545
6546
6547
6548
6549
6550
6551
6552
6553
6554
6555
6556
6557
6558
6559
6560
6561
6562
6563
6564
6565
6566
6567
6568
6569
6570
6571
6572
6573
6574
6575
6576
6577
6578
6579
6580
6581
6582
6583
6584
6585
6586
6587
6588
6589
6590
6591
6592
6593
6594
6595
6596
6597
6598
6599
6600
6601
6602
6603
6604
6605
6606
6607
6608
6609
6610
6611
6612
6613
6614
6615
6616
6617
6618
6619
6620
6621
6622
6623
6624
6625
6626
6627
6628
6629
6630
6631
6632
6633
6634
6635
6636
6637
6638
6639
6640
6641
6642
6643
6644
6645
6646
6647
6648
6649
6650
6651
6652
6653
6654
6655
6656
6657
6658
6659
6660
6661
6662
6663
6664
6665
6666
6667
6668
6669
6670
6671
6672
6673
6674
6675
6676
6677
6678
6679
6680
6681
6682
6683
6684
6685
6686
6687
6688
6689
6690
6691
6692
6693
6694
6695
6696
6697
6698
6699
6700
6701
6702
6703
6704
6705
6706
6707
6708
6709
6710
6711
6712
6713
6714
6715
6716
6717
6718
6719
6720
6721
6722
6723
6724
6725
6726
6727
6728
6729
6730
6731
6732
6733
6734
6735
6736
6737
6738
6739
6740
6741
6742
6743
6744
6745
6746
6747
6748
6749
6750
6751
6752
6753
6754
6755
6756
6757
6758
6759
6760
6761
6762
6763
6764
6765
6766
6767
6768
6769
6770
6771
6772
6773
6774
6775
6776
6777
6778
6779
6780
6781
6782
6783
6784
6785
6786
6787
6788
6789
6790
6791
6792
6793
6794
6795
6796
6797
6798
6799
6800
6801
6802
6803
6804
6805
6806
6807
6808
6809
6810
6811
6812
6813
6814
6815
6816
6817
6818
6819
6820
6821
6822
6823
6824
6825
6826
6827
6828
6829
6830
6831
6832
6833
6834
6835
6836
6837
6838
6839
6840
6841
6842
6843
6844
6845
6846
6847
6848
6849
6850
6851
6852
6853
6854
6855
6856
6857
6858
6859
6860
6861
6862
6863
6864
6865
6866
6867
6868
6869
6870
6871
6872
6873
6874
6875
6876
6877
6878
6879
6880
6881
6882
6883
6884
6885
6886
6887
6888
6889
6890
6891
6892
6893
6894
6895
6896
6897
6898
6899
6900
6901
6902
6903
6904
6905
6906
6907
6908
6909
6910
6911
6912
6913
6914
6915
6916
6917
6918
6919
6920
6921
6922
6923
6924
6925
6926
6927
6928
6929
6930
6931
6932
6933
6934
6935
6936
6937
6938
6939
6940
6941
6942
6943
6944
6945
6946
6947
6948
6949
6950
6951
6952
6953
6954
6955
6956
6957
6958
6959
6960
6961
6962
6963
6964
6965
6966
6967
6968
6969
6970
6971
6972
6973
6974
6975
6976
6977
6978
6979
6980
6981
6982
6983
6984
6985
6986
6987
6988
6989
6990
6991
6992
6993
6994
6995
6996
6997
6998
6999
7000
7001
7002
7003
7004
7005
7006
7007
7008
7009
7010
7011
7012
7013
7014
7015
7016
7017
7018
7019
7020
7021
7022
7023
7024
7025
7026
7027
7028
7029
7030
7031
7032
7033
7034
7035
7036
7037
7038
7039
7040
7041
7042
7043
7044
7045
7046
7047
7048
7049
7050
7051
7052
7053
7054
7055
7056
7057
7058
7059
7060
7061
7062
7063
7064
7065
7066
7067
7068
7069
7070
7071
7072
7073
7074
7075
7076
7077
7078
7079
7080
7081
7082
7083
7084
7085
7086
7087
7088
7089
7090
7091
7092
7093
7094
7095
7096
7097
7098
7099
7100
7101
7102
7103
7104
7105
7106
7107
7108
7109
7110
7111
7112
7113
7114
7115
7116
7117
7118
7119
7120
7121
7122
7123
7124
7125
7126
7127
7128
7129
7130
7131
7132
7133
7134
7135
7136
7137
7138
7139
7140
7141
7142
7143
7144
7145
7146
7147
7148
7149
7150
7151
7152
7153
7154
7155
7156
7157
7158
7159
7160
7161
7162
7163
7164
7165
7166
7167
7168
7169
7170
7171
7172
7173
7174
7175
7176
7177
7178
7179
7180
7181
7182
7183
7184
7185
7186
7187
7188
7189
7190
7191
7192
7193
7194
7195
7196
7197
7198
7199
7200
7201
7202
7203
7204
7205
7206
7207
7208
7209
7210
7211
7212
7213
7214
7215
7216
7217
7218
7219
7220
7221
7222
7223
7224
7225
7226
7227
7228
7229
7230
7231
7232
7233
7234
7235
7236
7237
7238
7239
7240
7241
7242
7243
7244
7245
7246
7247
7248
7249
7250
7251
7252
7253
7254
7255
7256
7257
7258
7259
7260
7261
7262
7263
7264
7265
7266
7267
7268
7269
7270
7271
7272
7273
7274
7275
7276
7277
7278
7279
7280
7281
7282
7283
7284
7285
7286
7287
7288
7289
7290
7291
7292
7293
7294
7295
7296
7297
7298
7299
7300
7301
7302
7303
7304
7305
7306
7307
7308
7309
7310
7311
7312
7313
7314
7315
7316
7317
7318
7319
7320
7321
7322
7323
7324
7325
7326
7327
7328
7329
7330
7331
7332
7333
7334
7335
7336
7337
7338
7339
7340
7341
7342
7343
7344
7345
7346
7347
7348
7349
7350
7351
7352
7353
7354
7355
7356
7357
7358
7359
7360
7361
7362
7363
7364
7365
7366
7367
7368
7369
7370
7371
7372
7373
7374
7375
7376
7377
7378
7379
7380
7381
7382
7383
7384
7385
7386
7387
7388
7389
7390
7391
7392
7393
7394
7395
7396
7397
7398
7399
7400
7401
7402
7403
7404
7405
7406
7407
7408
7409
7410
7411
7412
7413
7414
7415
7416
7417
7418
7419
7420
7421
7422
7423
7424
7425
7426
7427
7428
7429
7430
7431
7432
7433
7434
7435
7436
7437
7438
7439
7440
7441
7442
7443
7444
7445
7446
7447
7448
7449
7450
7451
7452
7453
7454
7455
7456
7457
7458
7459
7460
7461
7462
7463
7464
7465
7466
7467
7468
7469
7470
7471
7472
7473
7474
7475
7476
7477
7478
7479
7480
7481
7482
7483
7484
7485
7486
7487
7488
7489
7490
7491
7492
7493
7494
7495
7496
7497
7498
7499
7500
7501
7502
7503
7504
7505
7506
7507
7508
7509
7510
7511
7512
7513
7514
7515
7516
7517
7518
7519
7520
7521
7522
7523
7524
7525
7526
7527
7528
7529
7530
7531
7532
7533
7534
7535
7536
7537
7538
7539
7540
7541
7542
7543
7544
7545
7546
7547
7548
7549
7550
7551
7552
7553
7554
7555
7556
7557
7558
7559
7560
7561
7562
7563
7564
7565
7566
7567
7568
7569
7570
7571
7572
7573
7574
7575
7576
7577
7578
7579
7580
7581
7582
7583
7584
7585
7586
7587
7588
7589
7590
7591
7592
7593
7594
7595
7596
7597
7598
7599
7600
7601
7602
7603
7604
7605
7606
7607
7608
7609
7610
7611
7612
7613
7614
7615
7616
7617
7618
7619
7620
7621
7622
7623
7624
7625
7626
7627
7628
7629
7630
7631
7632
7633
7634
7635
7636
7637
7638
7639
7640
7641
7642
7643
7644
7645
7646
7647
7648
7649
7650
7651
7652
7653
7654
7655
7656
7657
7658
7659
7660
7661
7662
7663
7664
7665
7666
7667
7668
7669
7670
7671
7672
7673
7674
7675
7676
7677
7678
7679
7680
7681
7682
7683
7684
7685
7686
7687
7688
7689
7690
7691
7692
7693
7694
7695
7696
7697
7698
7699
7700
7701
7702
7703
7704
7705
7706
7707
7708
7709
7710
7711
7712
7713
7714
7715
7716
7717
7718
7719
7720
7721
7722
7723
7724
7725
7726
7727
7728
7729
7730
7731
7732
7733
7734
7735
7736
7737
7738
7739
7740
7741
7742
7743
7744
7745
7746
7747
7748
7749
7750
7751
7752
7753
7754
7755
7756
7757
7758
7759
7760
7761
7762
7763
7764
7765
7766
7767
7768
7769
7770
7771
7772
7773
7774
7775
7776
7777
7778
7779
7780
7781
7782
7783
7784
7785
7786
7787
7788
7789
7790
7791
7792
7793
7794
7795
7796
7797
7798
7799
7800
7801
7802
7803
7804
7805
7806
7807
7808
7809
7810
7811
7812
7813
7814
7815
7816
7817
7818
7819
7820
7821
7822
7823
7824
7825
7826
7827
7828
7829
7830
7831
7832
7833
7834
7835
7836
7837
7838
7839
7840
7841
7842
7843
7844
7845
7846
7847
7848
7849
7850
7851
7852
7853
7854
7855
7856
7857
7858
7859
7860
7861
7862
7863
7864
7865
7866
7867
7868
7869
7870
7871
7872
7873
7874
7875
7876
7877
7878
7879
7880
7881
7882
7883
7884
7885
7886
7887
7888
7889
7890
7891
7892
7893
7894
7895
7896
7897
7898
7899
7900
7901
7902
7903
7904
7905
7906
7907
7908
7909
7910
7911
7912
7913
7914
7915
7916
7917
7918
7919
7920
7921
7922
7923
7924
7925
7926
7927
7928
7929
7930
7931
7932
7933
7934
7935
7936
7937
7938
7939
7940
7941
7942
7943
7944
7945
7946
7947
7948
7949
7950
7951
7952
7953
7954
7955
7956
7957
7958
7959
7960
7961
7962
7963
7964
7965
7966
7967
7968
7969
7970
7971
7972
7973
7974
7975
7976
7977
7978
7979
7980
7981
7982
7983
7984
7985
7986
7987
7988
7989
7990
7991
7992
7993
7994
7995
7996
7997
7998
7999
8000
8001
8002
8003
8004
8005
8006
8007
8008
8009
8010
8011
8012
8013
8014
8015
8016
8017
8018
8019
8020
8021
8022
8023
8024
8025
8026
8027
8028
8029
8030
8031
8032
8033
8034
8035
8036
8037
8038
8039
8040
8041
8042
8043
8044
8045
8046
8047
8048
8049
8050
8051
8052
8053
8054
8055
8056
8057
8058
8059
8060
8061
8062
8063
8064
8065
8066
8067
8068
8069
8070
8071
8072
8073
8074
8075
8076
8077
8078
8079
8080
8081
8082
8083
8084
8085
8086
8087
8088
8089
8090
8091
8092
8093
8094
8095
8096
8097
8098
8099
8100
8101
8102
8103
8104
8105
8106
8107
8108
8109
8110
8111
8112
8113
8114
8115
8116
8117
8118
8119
8120
8121
8122
8123
8124
8125
8126
8127
8128
8129
8130
8131
8132
8133
8134
8135
8136
8137
8138
8139
8140
8141
8142
8143
8144
8145
8146
8147
8148
8149
8150
8151
8152
8153
8154
8155
8156
8157
8158
8159
8160
8161
8162
8163
8164
8165
8166
8167
8168
8169
8170
8171
8172
8173
8174
8175
8176
8177
8178
8179
8180
8181
8182
8183
8184
8185
8186
8187
8188
8189
8190
8191
8192
8193
8194
8195
8196
8197
8198
8199
8200
8201
8202
8203
8204
8205
8206
8207
8208
8209
8210
8211
8212
8213
8214
8215
8216
8217
8218
8219
8220
8221
8222
8223
8224
8225
8226
8227
8228
8229
8230
8231
8232
8233
8234
8235
8236
8237
8238
8239
8240
8241
8242
8243
8244
8245
8246
8247
8248
8249
8250
8251
8252
8253
8254
8255
8256
8257
8258
8259
8260
8261
8262
8263
8264
8265
8266
8267
8268
8269
8270
8271
8272
8273
8274
8275
8276
8277
8278
8279
8280
8281
8282
8283
8284
8285
8286
8287
8288
8289
8290
8291
8292
8293
8294
8295
8296
8297
8298
8299
8300
8301
8302
8303
8304
8305
8306
8307
8308
8309
8310
8311
8312
8313
8314
8315
8316
8317
8318
8319
8320
8321
8322
8323
8324
8325
8326
8327
8328
8329
8330
8331
8332
8333
8334
8335
8336
8337
8338
8339
8340
8341
8342
8343
8344
8345
8346
8347
8348
8349
8350
8351
8352
8353
8354
8355
8356
8357
8358
8359
8360
8361
8362
8363
8364
8365
8366
8367
8368
8369
8370
8371
8372
8373
8374
8375
8376
8377
8378
8379
8380
8381
8382
8383
8384
8385
8386
8387
8388
8389
8390
8391
8392
8393
8394
8395
8396
8397
8398
8399
8400
8401
8402
8403
8404
8405
8406
8407
8408
8409
8410
8411
8412
8413
8414
8415
8416
8417
8418
8419
8420
8421
8422
8423
8424
8425
8426
8427
8428
8429
8430
8431
8432
8433
8434
8435
8436
8437
8438
8439
8440
8441
8442
8443
8444
8445
8446
8447
8448
8449
8450
8451
8452
8453
8454
8455
8456
8457
8458
8459
8460
8461
8462
8463
8464
8465
8466
8467
8468
8469
8470
8471
8472
8473
8474
8475
8476
8477
8478
8479
8480
8481
8482
8483
8484
8485
8486
8487
8488
8489
8490
8491
8492
8493
8494
8495
8496
8497
8498
8499
8500
8501
8502
8503
8504
8505
8506
8507
8508
8509
8510
8511
8512
8513
8514
8515
8516
8517
8518
8519
8520
8521
8522
8523
8524
8525
8526
8527
8528
8529
8530
8531
8532
8533
8534
8535
8536
8537
8538
8539
8540
8541
8542
8543
8544
8545
8546
8547
8548
8549
8550
8551
8552
8553
8554
8555
8556
8557
8558
8559
8560
8561
8562
8563
8564
8565
8566
8567
8568
8569
8570
8571
8572
8573
8574
8575
8576
8577
8578
8579
8580
8581
8582
8583
8584
8585
8586
8587
8588
8589
8590
8591
8592
8593
8594
8595
8596
8597
8598
8599
8600
8601
8602
8603
8604
8605
8606
8607
8608
8609
8610
8611
8612
8613
8614
8615
8616
8617
8618
8619
8620
8621
8622
8623
8624
8625
8626
8627
8628
8629
8630
8631
8632
8633
8634
8635
8636
8637
8638
8639
8640
8641
8642
8643
8644
8645
8646
8647
8648
8649
8650
8651
8652
8653
8654
8655
8656
8657
8658
8659
8660
8661
8662
8663
8664
8665
8666
8667
8668
8669
8670
8671
8672
8673
8674
8675
8676
8677
8678
8679
8680
8681
8682
8683
8684
8685
8686
8687
8688
8689
8690
8691
8692
8693
8694
8695
8696
8697
8698
8699
8700
8701
8702
8703
8704
8705
8706
8707
8708
8709
8710
8711
8712
8713
8714
8715
8716
8717
8718
8719
8720
8721
8722
8723
8724
8725
8726
8727
8728
8729
8730
8731
8732
8733
8734
8735
8736
8737
8738
8739
8740
8741
8742
8743
8744
8745
8746
8747
8748
8749
8750
8751
8752
8753
8754
8755
8756
8757
8758
8759
8760
8761
8762
8763
8764
8765
8766
8767
8768
8769
8770
8771
8772
8773
8774
8775
8776
8777
8778
8779
8780
8781
8782
8783
8784
8785
8786
8787
8788
8789
8790
8791
8792
8793
8794
8795
8796
8797
8798
8799
8800
8801
8802
8803
8804
8805
8806
8807
8808
8809
8810
8811
8812
8813
8814
8815
8816
8817
8818
8819
8820
8821
8822
8823
8824
8825
8826
8827
8828
8829
8830
8831
8832
8833
8834
8835
8836
8837
8838
8839
8840
8841
8842
8843
8844
8845
8846
8847
8848
8849
8850
8851
8852
8853
8854
8855
8856
8857
8858
8859
8860
8861
8862
8863
8864
8865
8866
8867
8868
8869
8870
8871
8872
8873
8874
8875
8876
8877
8878
8879
8880
8881
8882
8883
8884
8885
8886
8887
8888
8889
8890
8891
8892
8893
8894
8895
8896
8897
8898
8899
8900
8901
8902
8903
8904
8905
8906
8907
8908
8909
8910
8911
8912
8913
8914
8915
8916
8917
8918
8919
8920
8921
8922
8923
8924
8925
8926
8927
8928
8929
8930
8931
8932
8933
8934
8935
8936
8937
8938
8939
8940
8941
8942
8943
8944
8945
8946
8947
8948
8949
8950
8951
8952
8953
8954
8955
8956
8957
8958
8959
8960
8961
8962
8963
8964
8965
8966
8967
8968
8969
8970
8971
8972
8973
8974
8975
8976
8977
8978
8979
8980
8981
8982
8983
8984
8985
8986
8987
8988
8989
8990
8991
8992
8993
8994
8995
8996
8997
8998
8999
9000
9001
9002
9003
9004
9005
9006
9007
9008
9009
9010
9011
9012
9013
9014
9015
9016
9017
9018
9019
9020
9021
9022
9023
9024
9025
9026
9027
9028
9029
9030
9031
9032
9033
9034
9035
9036
9037
9038
9039
9040
9041
9042
9043
9044
9045
9046
9047
9048
9049
9050
9051
9052
9053
9054
9055
9056
9057
9058
9059
9060
9061
9062
9063
9064
9065
9066
9067
9068
9069
9070
9071
9072
9073
9074
9075
9076
9077
9078
9079
9080
9081
9082
9083
9084
9085
9086
9087
9088
9089
9090
9091
9092
9093
9094
9095
9096
9097
9098
9099
9100
9101
9102
9103
9104
9105
9106
9107
9108
9109
9110
9111
9112
9113
9114
9115
9116
9117
9118
9119
9120
9121
9122
9123
9124
9125
9126
9127
9128
9129
9130
9131
9132
9133
9134
9135
9136
9137
9138
9139
9140
9141
9142
9143
9144
9145
9146
9147
9148
9149
9150
9151
9152
9153
9154
9155
9156
9157
9158
9159
9160
9161
9162
9163
9164
9165
9166
9167
9168
9169
9170
9171
9172
9173
9174
9175
9176
9177
9178
9179
9180
9181
9182
9183
9184
9185
9186
9187
9188
9189
9190
9191
9192
9193
9194
9195
9196
9197
9198
9199
9200
9201
9202
9203
9204
9205
9206
9207
9208
9209
9210
9211
9212
9213
9214
9215
9216
9217
9218
9219
9220
9221
9222
9223
9224
9225
9226
9227
9228
9229
9230
9231
9232
9233
9234
9235
9236
9237
9238
9239
9240
9241
9242
9243
9244
9245
9246
9247
9248
9249
9250
9251
9252
9253
9254
9255
9256
9257
9258
9259
9260
9261
9262
9263
9264
9265
9266
9267
9268
9269
9270
9271
9272
9273
9274
9275
9276
9277
9278
9279
9280
9281
9282
9283
9284
9285
9286
9287
9288
9289
9290
9291
9292
9293
9294
9295
9296
9297
9298
9299
9300
9301
9302
9303
9304
9305
9306
9307
9308
9309
9310
9311
9312
9313
9314
9315
9316
9317
9318
9319
9320
9321
9322
9323
9324
9325
9326
9327
9328
9329
9330
9331
9332
9333
9334
9335
9336
9337
9338
9339
9340
9341
9342
9343
9344
9345
9346
9347
9348
9349
9350
9351
9352
9353
9354
9355
9356
9357
9358
9359
9360
9361
9362
9363
9364
9365
9366
9367
9368
9369
9370
9371
9372
9373
9374
9375
9376
9377
9378
9379
9380
9381
9382
9383
9384
9385
9386
9387
9388
9389
9390
9391
9392
9393
9394
9395
9396
9397
9398
9399
9400
9401
9402
9403
9404
9405
9406
9407
9408
9409
9410
9411
9412
9413
9414
9415
9416
9417
9418
9419
9420
9421
9422
9423
9424
9425
9426
9427
9428
9429
9430
9431
9432
9433
9434
9435
9436
9437
9438
9439
9440
9441
9442
9443
9444
9445
9446
9447
9448
9449
9450
9451
9452
9453
9454
9455
9456
9457
9458
9459
9460
9461
9462
9463
9464
9465
9466
9467
9468
9469
9470
9471
9472
9473
9474
9475
9476
9477
9478
9479
9480
9481
9482
9483
9484
9485
9486
9487
9488
9489
9490
9491
9492
9493
9494
9495
9496
9497
9498
9499
9500
9501
9502
9503
9504
9505
9506
9507
9508
9509
9510
9511
9512
9513
9514
9515
9516
9517
9518
9519
9520
9521
9522
9523
9524
9525
9526
9527
9528
9529
9530
9531
9532
9533
9534
9535
9536
9537
9538
9539
9540
9541
9542
9543
9544
9545
9546
9547
9548
9549
9550
9551
9552
9553
9554
9555
9556
9557
9558
9559
9560
9561
9562
9563
9564
9565
9566
9567
9568
9569
9570
9571
9572
9573
9574
9575
9576
9577
9578
9579
9580
9581
9582
9583
9584
9585
9586
9587
9588
9589
9590
9591
9592
9593
9594
9595
9596
9597
//! HDF5 file reader.
//!
//! Opens an HDF5 file, parses the superblock and root group, and provides
//! access to dataset metadata and raw data.
//!
//! Supports both legacy (v0/v1 superblock, v1 object headers, symbol tables)
//! and modern (v2/v3 superblock, v2 object headers, link messages) formats.

use std::path::{Path, PathBuf};

use crate::dataset::{DatasetAccess, VirtualView};
use crate::format::btree_v1::{BTreeV1Config, BTreeV1Node, ChunkBTreeV1Node};
use crate::format::bytes::read_le_uint as read_uint;
use crate::format::creation_order::CreationOrder;
use crate::format::fractal_heap::{self, FractalHeapHeader};
use crate::format::global_heap::{
    decode_vlen_reference, vlen_reference_size, GlobalHeapCollection,
};
use crate::format::local_heap::{local_heap_get_string, LocalHeapHeader};
use crate::format::messages::attr_info::AttributeInfoMessage;
use crate::format::messages::attribute::{AttributeEntry, AttributeMessage};
use crate::format::messages::data_layout::{self, DataLayoutMessage};
use crate::format::messages::dataspace::DataspaceMessage;
use crate::format::messages::datatype::{DatatypeMessage, OldReferenceKind, ReferenceEncoding};
use crate::format::messages::external_file_list::ExternalFileListMessage;
use crate::format::messages::fill_value::{
    try_tiled_fill, FillValueMessage, ALLOC_TIME_LATE, FILL_TIME_IFSET,
};
use crate::format::messages::filter::{self, FilterPipeline};
use crate::format::messages::link::LinkMessage;
use crate::format::messages::link::LinkTarget;
use crate::format::messages::link_info::LinkInfoMessage;
use crate::format::messages::shared::{MessageStorage, MSG_FLAG_SHARED};
use crate::format::messages::superblock_ext::{
    BtreeKMessage, DriverInfoMessage, FileSpaceInfoMessage, SharedMessageTableMessage,
};
use crate::format::messages::virtual_mapping::{
    parse_source_name, VirtualMapping, VirtualMappingList,
};
use crate::format::messages::*;
use crate::format::object_header::ObjectHeader;
use crate::format::reference::{
    decode_object_element, decode_region_element, decode_region_heap_object, decode_revised_body,
    decode_revised_element, DecodedReference, Reference, ReferenceTarget, RevisedElement,
};
use crate::format::selection::{
    Hyperslab, PointSelection, RegularHyperslab, ResolvedSelection, Selection,
};
use crate::format::sohm::SohmMasterTable;
use crate::format::storage_kind::{AttributeStorage, LinkStorage};
use crate::format::superblock::{
    detect_superblock_version, SuperblockV0V1, SuperblockV2V3, SymbolTableCache,
};
use crate::format::symbol_table::SymbolTableNode;
use crate::format::{BlockReader, FormatContext, UNDEF_ADDR};

use crate::format::selection::check_hyperslab;
use crate::io::file_handle::{FileHandle, ReadDst};
use crate::io::hyperslab::{compute_strides, for_each_contiguous_run};
use crate::io::locking::FileLocking;
use crate::io::{FileMeta, IoResult};

/// The version-4 chunk-index descriptor pulled from a data-layout message:
/// the index kind, its address, and the per-kind parameters the reader needs
/// to walk it. Bundled so the chunked read entry point takes one descriptor
/// instead of a long parameter list.
struct ChunkIndexDesc<'a> {
    /// The kind of chunk index (single chunk, fixed/extensible array, …).
    index_type: data_layout::ChunkIndexType,
    /// Address of the chunk index structure (or the chunk itself, for a
    /// single chunk). `UNDEF_ADDR` when unallocated.
    index_address: u64,
    /// Extensible-array parameters (present iff `index_type == ExtensibleArray`).
    earray_params: Option<&'a data_layout::EarrayParams>,
    /// Filtered single-chunk parameters (present iff `index_type ==
    /// SingleChunk` and the layout's filtered flag is set): the chunk's exact
    /// on-disk size and per-chunk filter mask.
    single_chunk_filter: Option<data_layout::SingleChunkFilter>,
}

/// One v2-B-tree chunk-index record, resolved to `(chunk address, on-disk
/// read size, scaled chunk-grid offsets, filter mask)` — see
/// [`Hdf5Reader::collect_bt2_chunk_entries`].
type Bt2ChunkEntry = (u64, usize, Vec<u64>, u32);

/// What a chunked read should produce: the whole dataset, or one hyperslab.
///
/// Threaded through every chunked reader so the index walk, raw read, and
/// filter pipeline are shared between full reads and slice reads. For a
/// `Slice`, the reader allocates a `counts`-shaped buffer, skips reading any
/// chunk that does not overlap the selection (the I/O win), and scatters only
/// the chunk∩selection intersection. `Full` reads and places every chunk.
#[derive(Clone, Copy)]
enum ChunkTarget<'a> {
    Full,
    Slice {
        starts: &'a [u64],
        counts: &'a [u64],
    },
}

impl<'a> ChunkTarget<'a> {
    /// Whether a chunk at chunk-grid `coords` (extent `chunk_dims`) intersects
    /// the target. `Full` always intersects; a `Slice` intersects iff every
    /// dimension's chunk span `[origin, origin+chunk_dims)` overlaps the
    /// selection span `[start, start+count)`.
    fn overlaps(&self, coords: &[u64], chunk_dims: &[u64]) -> bool {
        match self {
            ChunkTarget::Full => true,
            ChunkTarget::Slice { starts, counts } => coords.iter().enumerate().all(|(d, &c)| {
                let origin = c.saturating_mul(chunk_dims[d]);
                let chunk_end = origin.saturating_add(chunk_dims[d]);
                let sel_end = starts[d].saturating_add(counts[d]);
                origin < sel_end && starts[d] < chunk_end
            }),
        }
    }
}

/// What one chunked read wants of every chunk, apart from which dataset it is
/// reading: the filter pipeline the chunks were stored through, the box of the
/// dataset it fills, and the fill value every byte no chunk covers takes.
///
/// Carried as one value because all four are constant across a read and are
/// threaded unchanged from the entry point through each index type down to
/// [`place_chunk_jobs`] — so a new index reader cannot pick up the target and
/// forget the fill or the destination fact.
#[derive(Clone, Copy)]
struct ChunkReadRequest<'a> {
    pipeline: Option<&'a FilterPipeline>,
    target: ChunkTarget<'a>,
    fill_value: Option<&'a [u8]>,
    /// The caller's destination fact ([`ReadDst`]) for the output buffer,
    /// which prices the per-run direct reads that land straight in it.
    /// Whole-chunk reads into a fresh image are unaffected.
    dst: ReadDst,
}

/// The dataset extent, chunk shape and element size a chunk-placement call
/// reads through — constant across every chunk of one read, so
/// [`for_each_chunk_run`] and the two sinks built on it take this once instead
/// of the three fields separately, leaving only what actually varies per chunk
/// (its data and grid coordinates) as their own parameters.
#[derive(Clone, Copy)]
struct ChunkOutputGeometry<'a> {
    dims: &'a [u64],
    chunk_dims: &'a [u64],
    element_size: u64,
}

/// One chunked read's placement constants: the geometry above plus the box in
/// dataset coordinates the read fills.
///
/// A full read fills the whole extent, which is the box `starts = 0`,
/// `counts = dims` describes, so resolving the target once
/// ([`ChunkPlacement::resolve`]) leaves the placement code below with one
/// shape instead of a full-read case and a slice case.
#[derive(Clone, Copy)]
struct ChunkPlacement<'a> {
    geo: ChunkOutputGeometry<'a>,
    starts: &'a [u64],
    counts: &'a [u64],
}

impl ChunkOutputGeometry<'_> {
    /// Bytes one whole chunk's image holds — the product of the chunk shape,
    /// element-wide. `None` only for geometry a corrupt file can carry: a
    /// shape whose product overflows, or an empty image.
    ///
    /// The size the layout says a decoded chunk is. Both the destination a
    /// chunk decodes into and the buffer a staged chunk decodes into come from
    /// here, so neither has to discover it by growing.
    fn image_bytes(&self) -> Option<u64> {
        self.chunk_dims
            .iter()
            .copied()
            .try_fold(1u64, |a, d| a.checked_mul(d))
            .and_then(|elems| elems.checked_mul(self.element_size))
            .filter(|b| *b > 0)
    }

    /// Bytes the chunk at `coords` holds that the dataset extent reaches:
    /// [`image_bytes`](Self::image_bytes), except at an edge where the extent
    /// cuts the chunk short. A read covering all of them has taken everything
    /// that chunk has to give, which is the same clip
    /// [`ChunkOverlap::of`] applies.
    fn resident_bytes(&self, coords: &[u64]) -> u64 {
        let mut elems = 1u64;
        for (d, &c) in coords.iter().enumerate().take(self.dims.len()) {
            let origin = c.saturating_mul(self.chunk_dims[d]);
            let end = origin.saturating_add(self.chunk_dims[d]).min(self.dims[d]);
            elems = elems.saturating_mul(end.saturating_sub(origin));
        }
        elems.saturating_mul(self.element_size)
    }
}

impl<'a> ChunkPlacement<'a> {
    /// Resolve a read target against the geometry it reads through. `zeros`
    /// lends a full read the origin it fills from and does not carry itself.
    fn resolve(geo: &ChunkOutputGeometry<'a>, target: ChunkTarget<'a>, zeros: &'a [u64]) -> Self {
        let (starts, counts) = match target {
            ChunkTarget::Full => (zeros, geo.dims),
            ChunkTarget::Slice { starts, counts } => (starts, counts),
        };
        ChunkPlacement {
            geo: *geo,
            starts,
            counts,
        }
    }

    /// Whether this read leaves part of the chunk at `coords` untouched, so a
    /// later read can still want the rest — which is what makes that chunk's
    /// decoded image worth keeping.
    ///
    /// What the chunk has to give is the chunk clipped to the dataset extent,
    /// the same clip [`ChunkOverlap::of`] applies and not the chunk image: an
    /// edge chunk the extent cuts short is fully consumed by a read that takes
    /// what is inside the extent. So a whole-dataset read leaves nothing,
    /// whatever the extent does at the edges.
    fn leaves_chunk_unconsumed(&self, coords: &[u64]) -> bool {
        match ChunkOverlap::of(self, coords) {
            Some(overlap) => overlap.bytes(self.geo.element_size) < self.geo.resident_bytes(coords),
            None => false,
        }
    }
}

/// One resolved external-file slot (H5O_EFL_ID): the on-disk message
/// stores each slot's name as an offset into a local heap, so this is that
/// slot after the heap lookup, in the order the dataset's logical byte
/// range concatenates them.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct ExternalFileSegment {
    /// The external file's name, exactly as stored — relative names are
    /// resolved against `HDF5_EXTFILE_PREFIX` at read time, not here.
    pub name: String,
    /// Byte offset within the named file where this slot's reserved
    /// region begins.
    pub offset: u64,
    /// Bytes reserved for this slot. `u64::MAX` (`H5O_EFL_UNLIMITED`)
    /// marks the last slot as unlimited/growable.
    pub size: u64,
}

/// Where a dataset's stored image lies, as far as a zero-copy view of it is
/// concerned.
///
/// [`Hdf5Reader::dataset_view_source`] is the only producer; it reports what
/// the layout message says and judges nothing.
#[cfg(feature = "mmap")]
pub(crate) enum ViewStorage {
    /// One stretch of this file: `len` bytes at absolute file offset
    /// `offset`, with the userblock already added, so it indexes the map
    /// directly.
    Contiguous { offset: u64, len: u64 },
    /// No storage is allocated. Every element reads as the fill value, which
    /// is a property of the header, not bytes anywhere in the file.
    Unallocated,
    /// The image is not one stretch of this file. The phrase says what it is
    /// instead, and lands verbatim in the refusal.
    Elsewhere(&'static str),
}

/// Everything a zero-copy view of one dataset rests on, gathered from the
/// file that owns it.
#[cfg(feature = "mmap")]
pub(crate) struct DatasetViewSource {
    /// The owning file's whole-file map, or `None` when that file is read
    /// through `pread`.
    pub map: Option<std::sync::Arc<crate::io::file_handle::LockedMap>>,
    /// Where the dataset's image lies in that file.
    pub storage: ViewStorage,
    /// How one element is stored, which decides whether the stored bytes are
    /// already the host image of a `T`.
    pub datatype: DatatypeMessage,
    /// The dataset's extent, for turning a requested range into a byte run.
    pub dims: Vec<u64>,
}

/// Read-side metadata for a single dataset.
pub struct DatasetReadInfo {
    /// Dataset name (the link name in the root group).
    pub name: String,
    /// Address of the dataset's object header — what an object reference to
    /// this dataset stores.
    pub object_header_address: u64,
    /// Element datatype.
    pub datatype: DatatypeMessage,
    /// Dataspace (dimensionality).
    pub dataspace: DataspaceMessage,
    /// Data layout (contiguous, compact, or chunked).
    pub layout: DataLayoutMessage,
    /// Filter pipeline for compressed chunks (None = uncompressed).
    pub filter_pipeline: Option<FilterPipeline>,
    /// Attributes attached to this dataset.
    pub attributes: ObjectAttributes,
    /// User-defined fill value bytes (one element wide), decoded from the
    /// fill-value message when `fill_defined == 2`. `None` => default
    /// zero-fill. Applied to unallocated chunks and unwritten regions.
    pub fill_value: Option<Vec<u8>>,
    /// The fill-value message's own definedness byte
    /// (`H5D_fill_value_t`/`H5Pfill_value_defined`): 0 = explicitly
    /// undefined (no fill is ever performed), 1 = default (zero-fill, no
    /// value stored), 2 = user-defined (`fill_value` carries the bytes). A
    /// dataset with no fill-value message at all reads as 1, matching a
    /// fresh dataset creation property list (`FillValueMessage::default`).
    pub fill_defined: u8,
    /// The fill-value message's write-time byte (`H5D_fill_time_t`): 0 =
    /// `H5D_FILL_TIME_ALLOC`, 1 = `H5D_FILL_TIME_NEVER`, 2 =
    /// `H5D_FILL_TIME_IFSET`. A dataset with no fill-value message at all
    /// reads as 2, `H5D_CRT_FILL_TIME_DEF` — the same "no message" default
    /// [`fill_defined`](Self::fill_defined) uses.
    pub fill_write_time: u8,
    /// The fill-value message's space-allocation-time byte
    /// (`H5D_alloc_time_t`): 1 = `H5D_ALLOC_TIME_EARLY`, 2 =
    /// `H5D_ALLOC_TIME_LATE`, 3 = `H5D_ALLOC_TIME_INCR`. A dataset with no
    /// fill-value message at all reads as `ALLOC_TIME_LATE`, matching
    /// [`FillValueMessage::default`]'s "no message" convention.
    pub alloc_time: u8,
    /// External raw-data segments (H5O_EFL_ID). Non-empty only when this
    /// dataset's storage is an External Data Files list instead of a
    /// normal contiguous block — `layout` still reports `Contiguous` with
    /// an undefined address in that case (H5Dlayout.c overrides the
    /// layout's storage ops whenever this message is present).
    pub external_files: Vec<ExternalFileSegment>,
    /// Virtual dataset source/virtual mappings (H5D_VIRTUAL), resolved from
    /// the global heap object `layout`'s `Virtual` variant points at.
    /// `Some` only when `layout` is `DataLayoutMessage::Virtual` and it
    /// names a mapping list (`heap_index != 0`); `None` for every other
    /// layout, and for a virtual dataset that has no mappings yet.
    pub virtual_mappings: Option<VirtualMappingList>,
    /// What each mapping in [`virtual_mappings`](Self::virtual_mappings)
    /// resolved to when the file was opened, in the same order — see
    /// [`MappingResolution`] and `Hdf5Reader::resolve_virtual_extents`.
    /// `None` for every non-virtual dataset and for a virtual one with no
    /// mapping list.
    pub virtual_resolution: Option<Vec<MappingResolution>>,
    /// The extent this dataset's dataspace message stores, kept because
    /// `dataspace.dims` holds the extent the *sources* gave it once
    /// `resolve_virtual_extents` has run. `H5D__virtual_set_extent_unlim`
    /// resolves from the space freshly loaded from the object header on
    /// every `H5Dopen` (H5Dvirtual.c:1386), so a later open under different
    /// [`DatasetAccess`] must resolve from this, not from its own last
    /// answer.
    ///
    /// `Some` exactly for a virtual dataset whose extent has been resolved;
    /// `None` for every dataset whose extent is simply its stored one.
    pub virtual_stored_dims: Option<Vec<u64>>,
}

/// What one virtual-dataset mapping resolved to at open time —
/// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c), which libhdf5 runs when
/// the dataset is opened and which both the dataset's reported extent and
/// every read of it depend on.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum MappingResolution {
    /// Neither selection grows: the mapping is already concrete, and the
    /// dataset's stored extent is its extent.
    Bounded,
    /// Both selections are unlimited (`unlim_dim_virtual >= 0` and
    /// `unlim_dim_source >= 0`). `virtual_clip` is how far this mapping
    /// reaches in its unlimited virtual dimension, `source_clip` the source
    /// dataset's own extent in its unlimited source dimension — the two
    /// values `H5S_hyper_clip_unlim` clips the mapping's selections to.
    Unlimited { virtual_clip: u64, source_clip: u64 },
    /// A printf mapping (unlimited virtual selection, limited source
    /// selection, `%b` in a source name).
    ///
    /// `blocks` is upstream's `first_missing`: the scan stops at the first
    /// block whose source is absent and then looks
    /// [`DatasetAccess::virtual_printf_gap`] blocks further, so blocks
    /// `0..blocks` are the ones the extent covers. `present` lists which of
    /// them actually have a source — with a non-zero gap the others are
    /// inside the extent but read as the fill value.
    Printf { blocks: u64, present: Vec<u64> },
}

/// The class of one link record in a group: what `H5Lget_info` reports,
/// carrying the value `H5Lget_val` returns for the classes that have one.
///
/// Every link a group holds gets one of these, whether or not the object it
/// names can be opened — a listing is a listing of links, not of objects.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum LinkClass {
    /// Another name for an object in this file.
    Hard,
    /// A path inside this file, resolved when the link is traversed.
    Soft { path: String },
    /// A path inside another file. Listed and reported, but not followed.
    External { file: String, path: String },
    /// A user-defined link class this reader has no interpreter for; libhdf5
    /// needs a registered link class for these too.
    UserDefined { link_type: u8 },
}

impl LinkClass {
    pub(crate) fn from_target(target: &LinkTarget) -> Self {
        match target {
            LinkTarget::Hard { .. } => Self::Hard,
            LinkTarget::Soft { target } => Self::Soft {
                path: target.clone(),
            },
            LinkTarget::External { file, path } => Self::External {
                file: file.clone(),
                path: path.clone(),
            },
            LinkTarget::UserDefined { link_type, .. } => Self::UserDefined {
                link_type: *link_type,
            },
        }
    }
}

/// The soft link a traversal crossed, kept so a lookup that finds nothing can
/// say the link dangles instead of reporting a bare absence.
struct SoftLinkRef {
    link: String,
    target: String,
}

/// Where a path leaves this file: the external link it crosses, the file that
/// link names, and the remainder of the path inside that file.
pub(crate) struct ExternalEdge {
    pub link: String,
    pub file: String,
    pub path: String,
}

impl ExternalEdge {
    /// The error for "the link resolved to a file, but the object it names is
    /// not in it" — the external counterpart of a dangling soft link.
    fn dangling(&self) -> crate::io::IoError {
        crate::io::IoError::DanglingLink {
            link: self.link.clone(),
            target: format!("{}::{}", self.file, self.path),
        }
    }
}

/// How a reader was opened, carried from the entry point to the per-superblock
/// constructors: both facts an external link needs later (where to resolve a
/// relative target from, and under which locking policy to open it) are fixed
/// at open time and belong together.
struct Origin {
    path: PathBuf,
    locking: crate::io::locking::FileLocking,
}

/// Bound on how many external links one path resolution may cross, matching
/// libhdf5's `H5L_NUM_LINKS` (the `H5Pset_nlinks` default). Two files that
/// link to each other form a cycle whose every hop opens a fresh target, so it
/// is this count, not target identity, that terminates the walk.
const MAX_EXTERNAL_HOPS: usize = 16;

/// What path traversal produced.
enum Traversal {
    /// A path in this file, after every group hard-link alias and soft link
    /// on it was followed. `via` names the last soft link crossed, if any.
    Path {
        path: String,
        via: Option<SoftLinkRef>,
    },
    /// A component of the path is an external link, so the path leaves this
    /// file. `path` is the remainder inside `file`.
    External {
        link: String,
        file: String,
        path: String,
    },
}

/// One rewrite a traversal step can apply to the path prefix it matched.
#[derive(Clone, Copy)]
enum Rewrite<'a> {
    /// A group hard link: continue from the group's first-walked path.
    Alias(&'a str),
    /// A soft link: continue from its value.
    Soft(&'a str),
    /// An external link: stop, the rest of the path is in another file.
    External { file: &'a str, path: &'a str },
}

/// Resolve a soft link's value against the group the link lives in, the way
/// `H5G_traverse` does: a value starting with `/` is absolute, anything else
/// is relative to that group. `.` and `..` components fold. The result has no
/// leading `/`.
fn resolve_link_value(link_path: &str, value: &str) -> String {
    let mut components: Vec<&str> = Vec::new();
    if !value.starts_with('/') {
        // The link's own parent group is everything before its last component.
        if let Some(parent) = link_path.rsplit_once('/').map(|(p, _)| p) {
            components.extend(parent.split('/').filter(|c| !c.is_empty()));
        }
    }
    for component in value.split('/') {
        match component {
            "" | "." => {}
            ".." => {
                components.pop();
            }
            c => components.push(c),
        }
    }
    components.join("/")
}

/// Everything one discovery walk found: the objects, the link records that
/// name them, and the group metadata the lookup paths need. Carried as one
/// value so the walk has a single owner rather than a widening tuple, and so
/// every walk (link-message and symbol-table alike) fills the same fields.
#[derive(Default)]
struct Catalog {
    datasets: Vec<DatasetReadInfo>,
    /// Dataset-shaped objects this crate cannot read, keyed by path; the
    /// value names what stopped it. They are listed exactly like readable
    /// datasets — the name is in the file either way — and refuse typed
    /// access with that reason.
    unreadable: std::collections::BTreeMap<String, String>,
    /// Attributes on non-root groups, keyed by group path.
    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
    /// Link storage kind and link creation-order policy of non-root groups,
    /// keyed by group path.
    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
    /// Every non-root group path the walk traversed into.
    group_paths: std::collections::BTreeSet<String>,
    /// Group object header address → the first path that reached it (no
    /// leading `/`), taken from the walk's cycle guard. What turns the
    /// address an object reference stores back into a name.
    group_object_paths: std::collections::HashMap<u64, String>,
    /// Group hard-link aliases: alias path → first-walked path.
    group_aliases: std::collections::HashMap<String, String>,
    /// Every link record seen, keyed by its full path (no leading `/`).
    links: std::collections::BTreeMap<String, LinkClass>,
    /// Committed (named) datatype objects, keyed by path.
    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
    /// Committed datatype object header address → its path (no leading `/`).
    /// The third object kind needs the same address→name entry groups and
    /// datasets have, or a path that names one resolves to nothing.
    datatype_object_paths: std::collections::HashMap<u64, String>,
}

impl Catalog {
    /// The address→absolute-path catalog this walk implies, with `root_addr`
    /// named `/` whether or not the walk itself reached it.
    ///
    /// The single owner of the catalog: file open and the SWMR
    /// [`Hdf5Reader::refresh`] rescan both build it here, so a dataset that
    /// appears after open is resolvable exactly as one present at open is.
    fn object_paths(&self, root_addr: u64) -> std::collections::HashMap<u64, String> {
        let mut paths = std::collections::HashMap::new();
        paths.insert(root_addr, "/".to_string());
        for (addr, path) in &self.group_object_paths {
            paths.insert(*addr, absolute_path(path));
        }
        for ds in &self.datasets {
            paths.insert(ds.object_header_address, absolute_path(&ds.name));
        }
        for (addr, path) in &self.datatype_object_paths {
            paths.insert(*addr, absolute_path(path));
        }
        paths
    }
}

/// The state one catalog walk threads through every group it visits.
///
/// A group stores its children either as `Link` messages in its object header
/// (with dense overflow in a fractal heap) or in the legacy symbol-table
/// B-tree plus local heap — and *which* it uses is a property of that group
/// alone. One file mixes the two freely: writing a single link that the old
/// format cannot express (an external link, a creation-order-tracked group)
/// migrates just that group, leaving its parent and its children where they
/// were. Walking a group in the format its *parent* used therefore finds no
/// children at all and reports an empty group, which is why the two storages
/// share one walker here: [`CatalogWalk::group`] asks each object header what
/// it declares, and [`CatalogWalk::child`] is the single place a child is
/// classified, recorded and descended into.
struct CatalogWalk<'a> {
    handle: &'a mut FileHandle,
    meta: &'a FileMeta,
    catalog: Catalog,
    /// Object headers already descended into, keyed to the first path that
    /// reached them: a later path to the same header is a group hard link,
    /// recorded in `group_aliases` so lookups resolve through it instead of
    /// walking (and cycling) a second time.
    visited: std::collections::HashMap<u64, String>,
}

impl<'a> CatalogWalk<'a> {
    /// Bound group nesting on a hostile or corrupt file.
    const MAX_DEPTH: usize = 256;

    /// Start a walk at the root group's object header address, seeded so a
    /// hard link cycling back to the root is not descended into again.
    fn new(handle: &'a mut FileHandle, meta: &'a FileMeta, root_addr: u64) -> Self {
        let mut visited = std::collections::HashMap::new();
        visited.insert(root_addr, String::new());
        Self {
            handle,
            meta,
            catalog: Catalog::default(),
            visited,
        }
    }

    /// The address/length widths, which most of the walk needs and `meta`
    /// carries alongside the file's B-tree ranks and shared-message table.
    fn ctx(&self) -> &FormatContext {
        &self.meta.ctx
    }

    fn finish(mut self) -> Catalog {
        self.catalog.group_object_paths = self.visited;
        self.catalog
    }

    /// Enumerate one group's children, choosing the storage from what this
    /// group's own header declares.
    ///
    /// `stab` is the symbol-table scratch-pad copy from the entry that named
    /// this group (the superblock's root entry, or the parent's symbol-table
    /// entry), which is the only source of those addresses when the header
    /// itself did not decode. `header` is `None` in exactly that case.
    fn group(
        &mut self,
        header: Option<&ObjectHeader>,
        prefix: &str,
        depth: usize,
        stab: Option<(u64, u64)>,
    ) -> IoResult<()> {
        if depth > Self::MAX_DEPTH {
            return Ok(());
        }
        let link_storage = header.filter(|h| header_declares_link_storage(h));
        if let Some(h) = link_storage {
            return self.links(h, prefix, depth);
        }
        // Symbol-table storage: the scratch-pad copy wins when it is set,
        // otherwise the addresses come from the group's own `stab` message.
        let (btree_addr, heap_addr) = match stab {
            Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
            _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
                Hdf5Reader::stab_from_header(h, self.ctx())
            }),
        };
        if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
            self.btree(btree_addr, heap_addr, prefix, depth)?;
        }
        Ok(())
    }

    /// Enumerate a group that stores its children as `Link` messages.
    fn links(&mut self, header: &ObjectHeader, prefix: &str, depth: usize) -> IoResult<()> {
        // Collect every link in this group: inline `Link` messages plus, for
        // groups using dense storage, links held in a fractal heap referenced
        // by the `Link Info` message.
        //
        // A link message that does not decode has no name to report the
        // failure against, and a `Link Info` message that does not decode
        // hides a whole group's dense storage. Either way the listing would
        // come back silently short, so both are errors: a listing this
        // reader cannot complete must not present itself as complete.
        let mut links: Vec<LinkMessage> = Vec::new();
        for msg in &header.messages {
            if msg.msg_type == MSG_LINK {
                let (link, _) = LinkMessage::decode(&msg.data, self.ctx())?;
                links.push(link);
            } else if msg.msg_type == MSG_LINK_INFO {
                let (info, _) = LinkInfoMessage::decode(&msg.data, self.ctx())?;
                if info.fractal_heap_address != UNDEF_ADDR {
                    let ctx = self.meta.ctx;
                    let dense =
                        Hdf5Reader::read_dense_links(self.handle, &ctx, info.fractal_heap_address)?;
                    links.extend(dense);
                }
            }
        }

        for link in &links {
            let full_name = join_path(prefix, &link.name);
            // Every link is a listing entry whatever it points at; only a
            // hard link names an object in this file to descend into.
            self.catalog
                .links
                .insert(full_name.clone(), LinkClass::from_target(&link.target));
            let LinkTarget::Hard { address } = &link.target else {
                continue;
            };
            self.child(full_name, *address, depth, None)?;
        }
        Ok(())
    }

    /// Enumerate a group that stores its children in a symbol-table B-tree
    /// plus local heap.
    fn btree(
        &mut self,
        btree_addr: u64,
        heap_addr: u64,
        prefix: &str,
        depth: usize,
    ) -> IoResult<()> {
        let sa = self.ctx().sizeof_addr as usize;
        let ss = self.ctx().sizeof_size as usize;

        // Read the local heap header + data for this group.
        let heap_hdr_buf = self.handle.read_at_most(heap_addr, 64)?;
        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
        let heap_data = self
            .handle
            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;

        // Collect all SNOD addresses by walking the B-tree.
        let mut snod_tree_visited = std::collections::HashSet::new();
        let snod_addrs = Hdf5Reader::collect_snod_addresses(
            self.handle,
            self.meta,
            btree_addr,
            0,
            &mut snod_tree_visited,
        )?;

        // A symbol-table node is a fixed-size record sized by `sym_leaf_k`,
        // which the superblock extension may have overridden.
        let snod_size = self.meta.btree.symbol_table_node_size(sa, ss);
        for snod_addr in snod_addrs {
            let snod_buf = self.handle.read_at_most(snod_addr, snod_size)?;
            let snod =
                SymbolTableNode::decode(&snod_buf, sa, ss, self.meta.btree.sym_leaf_max_entries())?;

            for entry in &snod.entries {
                let name = local_heap_get_string(&heap_data, entry.name_offset)?;
                // Skip empty names (root group self-reference).
                if name.is_empty() {
                    continue;
                }
                let full_name = join_path(prefix, &name);

                // A `H5G_CACHED_SLINK` entry is a soft link: it names no
                // object at all, and its value string lives in this group's
                // local heap. Record the link and move on — reading its
                // undefined object-header address is what used to drop it.
                if let SymbolTableCache::SoftLink { value_offset } = entry.cache {
                    let target = local_heap_get_string(&heap_data, value_offset as u64)?;
                    self.catalog
                        .links
                        .insert(full_name, LinkClass::Soft { path: target });
                    continue;
                }
                self.catalog
                    .links
                    .insert(full_name.clone(), LinkClass::Hard);
                self.child(
                    full_name,
                    entry.obj_header_addr,
                    depth,
                    entry.cached_symbol_table(),
                )?;
            }
        }

        Ok(())
    }

    /// Record one child of a group and, when it is itself a group, descend.
    ///
    /// Both storages end here, so a child is classified, catalogued and
    /// cycle-guarded the same way whichever way its name was found.
    fn child(
        &mut self,
        full_name: String,
        addr: u64,
        depth: usize,
        stab: Option<(u64, u64)>,
    ) -> IoResult<()> {
        // The entry names an object, so the object is in the listing whatever
        // comes of reading it: a header that does not decode (a stale link
        // left by a deletion, say) is reported against this name, never
        // dropped from it.
        let header = match Hdf5Reader::read_object_header_full(self.handle, self.meta, addr) {
            Ok(h) => h,
            Err(e) => {
                self.catalog
                    .unreadable
                    .insert(full_name, format!("its object header does not decode: {e}"));
                return Ok(());
            }
        };
        match Hdf5Reader::classify_object(self.handle, &header, self.meta, &full_name, addr) {
            ObjectKind::Dataset(info) => {
                self.catalog.datasets.push(*info);
                return Ok(());
            }
            ObjectKind::UnreadableDataset(why) => {
                self.catalog.unreadable.insert(full_name, why);
                return Ok(());
            }
            // A committed (named) datatype is neither a group nor a dataset,
            // so it must not be recorded as either; the link record above
            // already carries its name.
            ObjectKind::CommittedDatatype(info) => {
                self.catalog
                    .datatype_object_paths
                    .insert(addr, full_name.clone());
                self.catalog.datatypes.insert(full_name, *info);
                return Ok(());
            }
            ObjectKind::Group => {}
        }

        // It is a group. Record its path from the actual link record — before
        // the cycle check, so a hard-link alias of an already-visited group
        // still appears — whether or not it holds datasets or attributes.
        self.catalog.group_paths.insert(full_name.clone());
        // Capture group attributes (e.g. the NeXus `NX_class` marker), keyed
        // by path. Through the shared collector, which carries a per-attribute
        // failure as an entry naming it: a group whose attribute did not
        // decode must not come back as a group with one fewer attribute.
        // Recorded unconditionally, even for a group with none: the entry
        // also carries the header's own creation-order and storage facts,
        // which exist whether or not the group currently has any attributes.
        let ctx = self.meta.ctx;
        let attrs = collect_object_attributes(self.handle, &ctx, &header);
        self.catalog
            .group_attributes
            .insert(full_name.clone(), attrs);
        // Same unconditional recording for link storage and link
        // creation-order: this group's own header answers both whether or
        // not it descends any further.
        self.catalog.group_link_storage.insert(
            full_name.clone(),
            describe_link_storage(Some(&header), &ctx, stab),
        );

        // Descend at most once per object header (cycle guard); a second path
        // to it is a group hard link — record the alias for lookups instead.
        if let Some(first) = self.visited.get(&addr) {
            let first = first.clone();
            self.catalog.group_aliases.insert(full_name, first);
            return Ok(());
        }
        self.visited.insert(addr, full_name.clone());
        self.group(Some(&header), &full_name, depth + 1, stab)
    }
}

/// Join a group path prefix and a child name, with no leading `/` on a
/// root-level name.
fn join_path(prefix: &str, name: &str) -> String {
    if prefix.is_empty() {
        name.to_string()
    } else {
        format!("{}/{}", prefix, name)
    }
}

/// Whether an object header declares link-message storage (compact or
/// dense) rather than the legacy symbol-table format — `Walk::group`'s own
/// dispatch predicate, factored out so [`describe_link_storage`] answers the
/// same question by construction rather than by keeping two checks in sync.
fn header_declares_link_storage(header: &ObjectHeader) -> bool {
    header
        .messages
        .iter()
        .any(|m| m.msg_type == MSG_LINK || m.msg_type == MSG_LINK_INFO)
}

/// A group's own link storage kind and link creation-order policy — h5py's
/// `link_storage_str` and `get_link_creation_order()`, computed together
/// because both read the same `Link Info` message.
///
/// `stab` is the symbol-table scratch-pad copy from the entry that named
/// this group, exactly as [`CatalogWalk::group`] takes it; `header` is
/// `None` only when the object header itself did not decode.
fn describe_link_storage(
    header: Option<&ObjectHeader>,
    ctx: &FormatContext,
    stab: Option<(u64, u64)>,
) -> (LinkStorage, CreationOrder) {
    if let Some(h) = header.filter(|h| header_declares_link_storage(h)) {
        // Link-message storage: the `Link Info` message, when present, gives
        // both facts at once. Its absence means compact and untracked — no
        // message exists to carry a creation index in.
        return h
            .messages
            .iter()
            .find(|m| m.msg_type == MSG_LINK_INFO)
            .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
            .map(|(info, _)| {
                let storage = if info.is_dense() {
                    LinkStorage::Dense
                } else {
                    LinkStorage::Compact
                };
                (storage, info.creation_order())
            })
            .unwrap_or((LinkStorage::Compact, CreationOrder::Untracked));
    }
    // No link-message storage in the header (or no header to check at all):
    // symbol-table format when the scratch-pad or the header's own `Symbol
    // Table` message resolves an address pair. Creation order is always
    // untracked here — the pre-1.8 format predates the feature, and 1.8+
    // never tracks creation order without also converting to link storage.
    let (btree_addr, heap_addr) = match stab {
        Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
        _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
            Hdf5Reader::stab_from_header(h, ctx)
        }),
    };
    let storage = if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
        LinkStorage::SymbolTable
    } else {
        // Neither link-message nor symbol-table storage is declared — an
        // object header this crate could not fully account for. Nothing in
        // the 92-case oracle suite reaches this path; it exists so an
        // unreadable root group still answers rather than panicking.
        LinkStorage::Compact
    };
    (storage, CreationOrder::Untracked)
}

/// What an object header describes.
///
/// libhdf5 decides an object's class from *which* messages the header holds
/// (`H5O_obj_class`) and only then reads their contents; the two questions are
/// separate, and answering them with one `Option<DatasetReadInfo>` is what let
/// a dataset whose datatype this crate cannot decode leave the catalog as if
/// the file did not contain it. Each outcome now has its own name, so no
/// caller can turn "unreadable" back into "absent".
enum ObjectKind {
    /// A dataset: it carries a datatype, a dataspace and a data layout, and
    /// every message the payload depends on decoded.
    Dataset(Box<DatasetReadInfo>),
    /// A dataset whose payload depends on a message this crate cannot decode.
    /// The string names what stopped it and reaches the caller of any typed
    /// access to the name.
    UnreadableDataset(String),
    /// A group: it carries link, link-info, symbol-table or group-info
    /// storage, or none of the messages that identify anything else.
    Group,
    /// A committed (named) datatype object: a datatype message with neither
    /// group storage nor the dataspace/layout pair a dataset needs.
    CommittedDatatype(Box<CommittedDatatypeInfo>),
}

/// Whether a header's messages say it is a committed (named) datatype: a
/// datatype message, no group storage, and not the dataspace/layout pair a
/// dataset needs.
///
/// The one authority for that question. [`Hdf5Reader::classify_object`] asks it
/// of a file being read and `ReopenWalk::plan` of one being appended to, and a
/// header that is a named datatype to one and an unclassifiable object to the
/// other is how `named_datatype_names` came to answer differently in the two
/// modes for the same file.
pub(crate) fn header_is_committed_datatype(header: &ObjectHeader) -> bool {
    let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
    let is_group = present(MSG_LINK)
        || present(MSG_LINK_INFO)
        || present(MSG_SYMBOL_TABLE)
        || present(MSG_GROUP_INFO);
    !is_group && present(MSG_DATATYPE) && !(present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT))
}

/// A committed (named) datatype as read from its own object header.
///
/// `H5Tcommit` gives a type a name and a place in the file; every dataset and
/// attribute built on it then stores a reference to this object rather than a
/// copy of the type. It is a third kind of object beside groups and datasets,
/// and classifying it as neither is what left its name in the file with
/// nothing behind it.
#[derive(Debug, Clone)]
pub struct CommittedDatatypeInfo {
    /// The type this object commits, or what stopped it from decoding. The
    /// object is in the listing either way, exactly as an unreadable dataset
    /// is: the name is in the file whether or not this crate can read what it
    /// names.
    datatype: Result<DatatypeMessage, String>,
    /// Attributes attached to the committed datatype itself.
    attributes: Vec<AttributeMessage>,
}

impl CommittedDatatypeInfo {
    /// The committed type, or the reason it cannot be read.
    pub fn datatype(&self) -> Result<&DatatypeMessage, &str> {
        self.datatype.as_ref().map_err(String::as_str)
    }

    /// The attributes attached to the committed datatype.
    pub fn attributes(&self) -> &[AttributeMessage] {
        &self.attributes
    }
}

/// Everything the superblock extension object header contributes to the
/// file-level view.
///
/// `H5Fsuper.c::H5F__super_read` opens this header immediately after decoding
/// the superblock, before any user object is reachable, so every message here
/// is in force for the first metadata decode that follows.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct SuperblockExtension {
    /// Shared Message Table message (0x000F): where the SOHM master table is.
    pub shared_message_table: Option<SharedMessageTableMessage>,
    /// v1 B-tree "K" values message (0x0013): non-default split ranks.
    pub btree_k: Option<BtreeKMessage>,
    /// Driver info message (0x0014).
    pub driver_info: Option<DriverInfoMessage>,
    /// File space info message (0x0017): allocation strategy, page size, and
    /// the persisted free-space manager addresses.
    pub file_space_info: Option<FileSpaceInfoMessage>,
}

/// The datasets the discovery walk found, in the order it found them, with
/// an index from canonical path to position.
///
/// Every open and every read resolves a dataset by name, so a plain `Vec` a
/// lookup scans made each of them cost a pass over the whole catalog — a file
/// holding thousands of datasets paid that per access. The index is derived
/// in [`DatasetTable::new`], which is the only way to build one, so it cannot
/// fall out of step with the list it indexes.
/// One dataset's chunk index, as decoded from the file: `(chunk address,
/// on-disk byte count, filter mask)` per chunk the index records, in the
/// order the index walks them, alongside each chunk's chunk-grid coordinates
/// (`coords[i * rank .. (i + 1) * rank]`).
///
/// What the file decides is `entries` and `coords`; what a read then does with
/// an entry is that read's own business — a chunk outside the selection is
/// never fetched, a filtered chunk is read to its recorded size — so nothing
/// about one call is recorded here. `images` is the exception the rest of this
/// file is built around: decompressed chunk images, which say nothing about
/// one call either (an image is the decode of stored bytes this index names)
/// and which live here precisely so that they cannot outlive the index that
/// named them — see [`ChunkImageCache`].
struct DecodedChunkIndex {
    entries: Vec<(u64, u64, u32)>,
    coords: Vec<u64>,
    images: ChunkImageCache,
}

impl DecodedChunkIndex {
    /// A freshly decoded index, with an empty image cache. The only way to
    /// build one, so no decode site can forget the cache or hand one index's
    /// images to another.
    fn new(entries: Vec<(u64, u64, u32)>, coords: Vec<u64>) -> Self {
        Self {
            entries,
            coords,
            images: ChunkImageCache::default(),
        }
    }
}

/// The stored bytes one cached chunk image was decoded from: the chunk's file
/// address, the byte count the read asked for at it, and the per-chunk filter
/// mask the pipeline ran under.
///
/// An image is the decode of exactly these three, so an entry cannot answer
/// for a chunk stored somewhere else, read at another length, or filtered
/// through another mask — the key carries the whole of what the image depends
/// on apart from the pipeline, which belongs to the catalog entry this cache
/// hangs under.
#[derive(Clone, Copy, PartialEq, Eq, Hash)]
struct ChunkImageKey {
    addr: u64,
    len: usize,
    mask: u32,
}

/// Bytes of decompressed chunk images one dataset keeps — libhdf5's
/// `H5D_CHUNK_CACHE_NBYTES_DEF`, the default `rdcc` size every dataset gets
/// there, so a comparison against it is a comparison of like for like.
const CHUNK_CACHE_BYTES: usize = 1024 * 1024;

/// Images one dataset keeps at once — libhdf5's `H5D_CHUNK_CACHE_NSLOTS_DEF`.
/// The byte budget is the real bound; this one keeps a dataset of very small
/// chunks from filling the budget with thousands of entries, which is also
/// what bounds the least-recently-used scan below.
const CHUNK_CACHE_SLOTS: usize = 521;

/// Decompressed chunk images, kept for the next read that wants the same
/// chunk.
///
/// Four consecutive 64 KiB slices of a 256 KiB deflated chunk are four reads
/// of one chunk; without a cache each inflates it whole and throws away three
/// quarters. libhdf5 gives every dataset a raw-data chunk cache for exactly
/// this (`H5D__chunk_lock`, `rdcc`), which is why a row-at-a-time walk of a
/// compressed dataset costs it one inflate per chunk rather than one per row.
///
/// MUST NOT outlive the [`DecodedChunkIndex`] the images were named by, and is
/// therefore a field of it: an image is reachable only through the index it
/// was decoded under, so everything that drops a decoded index —
/// `DatasetTable::entry_mut`, and the table rebuild a SWMR refresh does —
/// drops the images with it. There is no second lifetime to keep in step.
///
/// A read hands over an image only for a chunk it did not consume
/// ([`ChunkPlacement::leaves_chunk_unconsumed`]), so a whole-dataset read,
/// which takes every chunk entire and can never want one twice, never touches
/// the cache at all. [`ChunkImageCacheState::insert`] is the sole owner of
/// what the cache holds: what fits the budget, and what is evicted to keep it
/// fitting. Everything above it may offer an image; only it decides.
#[derive(Default)]
struct ChunkImageCache {
    /// Interior mutability because a read reaches the index through an `Arc`.
    /// A poisoned lock degrades to "no cache", never a failed read.
    state: std::sync::Mutex<ChunkImageCacheState>,
}

#[derive(Default)]
struct ChunkImageCacheState {
    /// Key → (last-use stamp, image).
    images: std::collections::HashMap<ChunkImageKey, (u64, std::sync::Arc<Vec<u8>>)>,
    /// Summed `images` lengths, against [`CHUNK_CACHE_BYTES`].
    bytes: usize,
    /// Monotonic use counter; the smallest stamp is the eviction victim.
    clock: u64,
}

impl ChunkImageCache {
    /// The image cached for `key`, if one is held.
    fn get(&self, key: &ChunkImageKey) -> Option<std::sync::Arc<Vec<u8>>> {
        let mut state = self.state.lock().ok()?;
        state.clock += 1;
        let clock = state.clock;
        let (stamp, image) = state.images.get_mut(key)?;
        *stamp = clock;
        Some(std::sync::Arc::clone(image))
    }

    /// How many images are held. A poisoned lock holds none it could serve.
    #[cfg(test)]
    fn held(&self) -> usize {
        self.state.lock().map(|s| s.images.len()).unwrap_or(0)
    }

    /// Keep `image` as the decode of `key`, and hand it back shared.
    fn keep(&self, key: ChunkImageKey, image: Vec<u8>) -> std::sync::Arc<Vec<u8>> {
        let image = std::sync::Arc::new(image);
        if let Ok(mut state) = self.state.lock() {
            state.insert(key, std::sync::Arc::clone(&image));
        }
        image
    }
}

impl ChunkImageCacheState {
    /// Add one image, then evict least-recently-used entries until the cache
    /// is inside both bounds again.
    ///
    /// An image over the byte budget is refused outright rather than evicting
    /// everything to hold it, so the entry just inserted — which carries the
    /// newest stamp and is therefore the last one eviction would reach — is
    /// never the entry evicted.
    fn insert(&mut self, key: ChunkImageKey, image: std::sync::Arc<Vec<u8>>) {
        let len = image.len();
        if len > CHUNK_CACHE_BYTES {
            return;
        }
        self.clock += 1;
        if let Some((_, old)) = self.images.insert(key, (self.clock, image)) {
            self.bytes -= old.len();
        }
        self.bytes += len;
        while self.bytes > CHUNK_CACHE_BYTES || self.images.len() > CHUNK_CACHE_SLOTS {
            let Some(victim) = self
                .images
                .iter()
                .min_by_key(|(_, (stamp, _))| *stamp)
                .map(|(k, _)| *k)
            else {
                break;
            };
            if let Some((_, old)) = self.images.remove(&victim) {
                self.bytes -= old.len();
            }
        }
    }
}

/// The reader's dataset catalog, and the one place a decoded chunk index
/// lives.
///
/// A cached [`DecodedChunkIndex`] MUST describe the file exactly as the
/// catalog entry beside it does, and MUST NOT be handed out for an index
/// address other than the one it was decoded from. The fields are private to
/// the module, so an entry is reachable read-only (`get`, `iter`, `Index`) or
/// through [`DatasetTable::entry_mut`], which drops that entry's cached index
/// before handing out the reference: changing what an entry says about the
/// file cannot leave a stale index behind it. Rebuilding the table — what a
/// SWMR refresh does — drops every cached index with it.
mod dataset_table {
    use super::{DatasetReadInfo, DecodedChunkIndex};
    use std::sync::Arc;

    pub(super) struct DatasetTable {
        list: Vec<DatasetReadInfo>,
        /// Canonical path (no leading `/`) → position in `list`. A name the
        /// walk recorded twice keeps its first position, which is the entry a
        /// scan from the front would have found.
        by_name: std::collections::HashMap<String, usize>,
        /// Per entry, the index address a chunk index was decoded from and
        /// what it decoded to. Parallel to `list`.
        chunk_index: Vec<Option<(u64, Arc<DecodedChunkIndex>)>>,
    }

    impl DatasetTable {
        pub(super) fn new(list: Vec<DatasetReadInfo>) -> Self {
            let mut by_name = std::collections::HashMap::with_capacity(list.len());
            for (i, ds) in list.iter().enumerate() {
                by_name.entry(ds.name.clone()).or_insert(i);
            }
            let chunk_index = list.iter().map(|_| None).collect();
            Self {
                list,
                by_name,
                chunk_index,
            }
        }

        /// Where the dataset named `name` sits in the list, or `None` when the
        /// catalog holds no such name.
        pub(super) fn position(&self, name: &str) -> Option<usize> {
            self.by_name.get(name).copied()
        }

        pub(super) fn get(&self, name: &str) -> Option<&DatasetReadInfo> {
            self.list.get(self.position(name)?)
        }

        pub(super) fn iter(&self) -> std::slice::Iter<'_, DatasetReadInfo> {
            self.list.iter()
        }

        /// The entry at `i`, mutably. Whatever the caller changes about it may
        /// be what a decoded chunk index was read against, so that index goes
        /// first: this is the only way to a `&mut DatasetReadInfo`, which is
        /// what keeps the cache from outliving the entry it describes.
        pub(super) fn entry_mut(&mut self, i: usize) -> &mut DatasetReadInfo {
            self.chunk_index[i] = None;
            &mut self.list[i]
        }

        /// The chunk index cached for entry `i`, if one was decoded from
        /// `index_address`. A different address means the entry now points at
        /// another structure, and the cached one does not answer for it.
        pub(super) fn chunk_index(
            &self,
            i: usize,
            index_address: u64,
        ) -> Option<&Arc<DecodedChunkIndex>> {
            match self.chunk_index.get(i)? {
                Some((addr, index)) if *addr == index_address => Some(index),
                _ => None,
            }
        }

        /// Keep `index` as entry `i`'s decoded chunk index for
        /// `index_address`, and hand it back.
        pub(super) fn cache_chunk_index(
            &mut self,
            i: usize,
            index_address: u64,
            index: DecodedChunkIndex,
        ) -> Arc<DecodedChunkIndex> {
            let index = Arc::new(index);
            self.chunk_index[i] = Some((index_address, Arc::clone(&index)));
            index
        }
    }

    impl std::ops::Index<usize> for DatasetTable {
        type Output = DatasetReadInfo;

        fn index(&self, i: usize) -> &DatasetReadInfo {
            &self.list[i]
        }
    }
}

use dataset_table::DatasetTable;

/// HDF5 file reader.
pub struct Hdf5Reader {
    handle: FileHandle,
    meta: FileMeta,
    /// Messages read from the superblock extension object header, empty when
    /// the file has no extension.
    ext: SuperblockExtension,
    /// End-of-file address from the superblock.
    _eof: u64,
    /// Superblock format version (0-3), decoded once at open time by
    /// `detect_superblock_version` and never re-derived: 0/1 is the legacy
    /// symbol-table root, 2/3 the link-message root.
    superblock_version: u8,
    datasets: DatasetTable,
    /// Dataset-shaped objects this crate cannot read, keyed by path (no
    /// leading `/`), the value naming what stopped it. They are listed with
    /// the readable datasets and refuse typed access with that reason: an
    /// object the file contains is never reported as one it does not.
    unreadable: std::collections::BTreeMap<String, String>,
    /// Attributes on the root group (file-level attributes).
    root_attributes: ObjectAttributes,
    /// Link storage kind and link creation-order policy of the root group.
    root_link_storage: (LinkStorage, CreationOrder),
    /// Attributes on non-root groups, keyed by group path (no leading `/`).
    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
    /// Link storage kind and link creation-order policy of non-root groups,
    /// keyed by group path (no leading `/`).
    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
    /// Every non-root group path the discovery walk traversed into (no
    /// leading `/`), regardless of whether the group has datasets or
    /// attributes. Built from actual link records, so empty groups,
    /// attribute-only groups, and subgroup-only groups are all included.
    group_paths: std::collections::BTreeSet<String>,
    /// Group hard links: alias path → the first-walked path of the same
    /// group object header (both without a leading `/`). The walk
    /// descends each header once, so objects under the alias are stored
    /// under the first path; lookups resolve alias prefixes through this
    /// map, as HDF5 path traversal does.
    group_aliases: std::collections::HashMap<String, String>,
    /// Every link record in the file, keyed by full path (no leading `/`).
    /// A listing is a listing of links, so this holds soft and external
    /// links as well as the hard links that name the objects above.
    links: std::collections::BTreeMap<String, LinkClass>,
    /// The path this file was opened with. An external link resolves its
    /// target relative to the directory holding it (libhdf5 keeps the same
    /// thing as `H5F_EXTPATH`), so the reader has to remember where it came
    /// from.
    path: PathBuf,
    /// The directory holding this HDF5 file, resolved once at open time.
    /// External raw-data files (H5O_EFL_ID) are named relative to it when
    /// `HDF5_EXTFILE_PREFIX` contains `${ORIGIN}` (`H5D__build_file_prefix`,
    /// H5Dint.c) — captured at open time rather than re-derived from the
    /// process's current directory at read time, matching libhdf5's own
    /// one-time capture in `H5F_t::extpath`.
    source_dir: PathBuf,
    /// The locking policy this file was opened under, reused verbatim for
    /// every external target: libhdf5 hands `H5F_prefix_open_file` the
    /// parent's file-access property list, so one `HDF5_USE_FILE_LOCKING`
    /// setting (or one `H5FileOptions::locking` call) governs every file a
    /// path touches, not just the first.
    locking: crate::io::locking::FileLocking,
    /// `H5Pset_elink_prefix` for every external link this reader crosses —
    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix).
    /// Propagated verbatim to every target this reader opens, the way a lapl
    /// reaches the next hop of a link chain upstream (measured against
    /// libhdf5 1.14.6: a two-hop chain resolves its second hop under the
    /// prefix given at the first).
    elink_prefix: Option<String>,
    /// Files opened on another file's behalf — external-link targets,
    /// external-reference targets and virtual-dataset sources — keyed by the
    /// resolved path that opened them, so N names for one file share one
    /// open handle. libhdf5 shares one open file the same way: `H5F_open`
    /// hands back the `H5F_shared_t` a path already open has rather than a
    /// second one (H5Fint.c:1906-1918).
    ///
    /// How long an entry lives is its [`CrossFileOwner`]'s business, and
    /// differs by what named it. A target's own external links are cached in
    /// that target's map, so the first reader in a chain transitively holds
    /// the whole chain open.
    external: std::collections::BTreeMap<PathBuf, CrossFileEntry>,
    /// What each virtual mapping's stored source file name resolved to,
    /// keyed by the canonical path of the virtual dataset that named it and
    /// that name exactly as the mapping holds it. Separate from
    /// [`external_resolved`](Self::external_resolved) because the two
    /// searches read different environment variables and property lists, so
    /// one name can resolve two ways depending on which named it — and keyed
    /// by the virtual dataset as well because
    /// [`DatasetAccess::virtual_prefix`] is a per-open property, so two
    /// virtual datasets naming one source file name can legitimately reach
    /// two different files.
    vds_resolved: std::collections::BTreeMap<(String, String), PathBuf>,
    /// What each external link's stored file name resolved to, keyed by that
    /// name exactly as the link holds it.
    ///
    /// The search runs once per distinct name per reader and its answer is
    /// then fixed: re-probing the filesystem on a later crossing would let one
    /// link answer differently mid-session, and would fail to find the handle
    /// it already holds once the target has been renamed or unlinked.
    external_resolved: std::collections::BTreeMap<String, PathBuf>,
    /// Committed (named) datatype objects, keyed by path (no leading `/`).
    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
    /// Object header address → absolute path, for every group and dataset the
    /// discovery walk reached plus the root group. This is what turns the
    /// address an object reference stores back into a name.
    object_paths: std::collections::HashMap<u64, String>,
    /// The dataset-access properties in force for each *open* dataset, keyed
    /// by canonical path (no leading `/`); an absent entry means
    /// [`DatasetAccess::default`].
    ///
    /// libhdf5 keeps this in the dataset's *shared* open-object info, which
    /// is why the first open of a dataset fixes it for every later one
    /// ([`apply_dataset_access`](Self::apply_dataset_access)):
    /// `H5D__virtual_init` puts the view and the printf gap into
    /// `dset->shared->layout.storage.u.virt` (H5Dvirtual.c:2178-2188), and
    /// `H5D__open_name` puts both file prefixes into `shared->extfile_prefix`
    /// and `shared->vds_prefix` (H5Dint.c:1488-1521) — for every dataset,
    /// not only a virtual one. It is the single owner of that answer: the
    /// extent resolution and the external-file read both read it and nothing
    /// else writes it, so a SWMR [`refresh`](Self::refresh) re-resolves under
    /// the same properties rather than reverting to the defaults.
    dataset_access: std::collections::BTreeMap<String, AccessInForce>,
}

/// A live open on a dataset.
///
/// The reader holds only a [`Weak`](std::sync::Weak) to it and every handle
/// an open handed out holds the strong one, so "is this dataset still open"
/// is answered by the handles themselves — nothing has to tell the reader
/// when one is dropped, and a handle's drop takes no lock on the file.
pub(crate) type DatasetOpenToken = std::sync::Arc<()>;

/// One file this reader opened on another file's behalf, and what keeps it
/// open.
struct CrossFileEntry {
    reader: Box<Hdf5Reader>,
    owner: CrossFileOwner,
}

/// What holds a [`CrossFileEntry`] open, which is the same question as how
/// long it stays open.
///
/// libhdf5 asks it the same way and gets two different answers, because
/// nothing caches these files across the open that needed them: the default
/// external file cache is *disabled* — `H5F_ACS_EFC_SIZE_DEF` is 0
/// (H5Pfapl.c:191) and `H5F_open` builds no cache below that
/// (H5Fint.c:1217-1218) — so the `H5F_efc_close` both kinds of crossing end
/// with (H5Lexternal.c:241, H5Dvirtual.c:927) falls straight through to
/// `H5F_try_close` (H5Fefc.c:420-426). What is left holding the file is
/// whatever object the crossing opened inside it.
enum CrossFileOwner {
    /// This reader, until it is dropped.
    ///
    /// An external link's target is held by the object the traversal opened
    /// in it (`H5O_open_name`, H5Lexternal.c:225), and an external
    /// reference's by `H5R__reopen_file`'s file handle; this crate's
    /// equivalent of those objects is the reader itself, which holds the
    /// target's whole catalog and answers every later name from it.
    ///
    /// This is a **deliberate difference from libhdf5**, not a match.
    /// Measured against 1.14.6 by watching `/proc/self/fd`: a link target's
    /// descriptor appears at the traversal, survives while any object opened
    /// through the link is open, and goes at that object's close — and a
    /// traversal that keeps no object (`f["ext/data"][...]`, a listing, an
    /// `H5Lget_info`) leaves nothing open at all. It is *not* cached past
    /// that: see this enum's own note on the disabled default EFC.
    ///
    /// Matching it would mean this crate re-walked a target file's whole
    /// catalog on every name that crosses the link, because a reader — not
    /// a per-object handle — is what it opens a target *as*. The delta is a
    /// descriptor-and-lock window with no effect on any byte read or
    /// written, so the reader's lifetime stands and the window is documented
    /// here rather than paid for with that redesign.
    Reader,
    /// The virtual datasets that named this file as a source, by canonical
    /// path in *this* reader. Dropped once none of them has a live open.
    ///
    /// `H5D__virtual_open_source_dset` leaves the source *dataset* open in
    /// the virtual dataset's shared layout (H5Dvirtual.c:901-902), and that
    /// is what keeps the source file open; `H5D__virtual_reset_layout` closes
    /// it at the last `H5Dclose` of the virtual dataset (H5Dvirtual.c:709-710
    /// via `H5D__virtual_reset_source_dset`, :955). Measured against
    /// libhdf5 1.14.6 by watching `/proc/self/fd`: the source appears at the
    /// read that needs it, survives a second virtual dataset naming the same
    /// file until *both* are closed, and goes at the last `H5Dclose` — not at
    /// `H5Fclose`.
    VirtualOpens(std::collections::BTreeSet<String>),
}

impl CrossFileOwner {
    /// Record that `also` now names this file too, keeping whichever
    /// ownership outlives the other. [`Reader`](Self::Reader) outlives every
    /// virtual open, so once a file is held that way it stays held.
    fn widen(&mut self, also: CrossFileOwner) {
        match (&mut *self, also) {
            (CrossFileOwner::Reader, _) => {}
            (slot, CrossFileOwner::Reader) => *slot = CrossFileOwner::Reader,
            (CrossFileOwner::VirtualOpens(have), CrossFileOwner::VirtualOpens(more)) => {
                have.extend(more)
            }
        }
    }

    /// The owner of a source file one virtual dataset named.
    fn virtual_open(vds: &str) -> Self {
        CrossFileOwner::VirtualOpens(std::iter::once(vds.to_string()).collect())
    }
}

/// The dataset-access properties one dataset was resolved under, and the
/// opens that fixed them.
struct AccessInForce {
    access: DatasetAccess,
    /// Live while at least one handle from the open that set `access` is
    /// alive. Once it is dead the properties are only a record of how the
    /// stamped extent was arrived at (what a SWMR refresh re-resolves
    /// under); the next open resolves afresh under its own.
    open: std::sync::Weak<()>,
}

/// The absolute form of a discovery-walk path (which carries no leading `/`):
/// the root group's empty path becomes `/`, `entry/data` becomes
/// `/entry/data`.
fn absolute_path(path: &str) -> String {
    format!("/{}", path.trim_start_matches('/'))
}

/// Total byte length of `dims.product() * element_size`, computed with
/// saturating arithmetic. `dims` and `element_size` are file-derived; a
/// crafted file with huge dimensions thus yields a saturated (too-large)
/// value — rejected downstream by the file-size/buffer checks — rather
/// than panicking in a debug build or wrapping in release.
fn saturating_byte_len(dims: &[u64], element_size: u64) -> u64 {
    dims.iter()
        .fold(1u64, |acc, &d| acc.saturating_mul(d))
        .saturating_mul(element_size)
}

/// A fixed-length string attribute's value, under the padding rule its
/// datatype declares.
///
/// The single owner for the attribute side of that rule: both
/// [`H5Reader::attr_string_value`] and the writer-mode fallback in
/// `H5Attribute::read_string` end here, so a space-padded attribute does not
/// read back with its padding attached on one path and not the other. Bytes
/// that are not valid UTF-8 become U+FFFD, as they always have on this path.
///
/// A datatype that is not a string at all is read as null-terminated, which is
/// what asking for the string value of, say, an integer attribute has always
/// meant here.
pub(crate) fn fixed_string_attr_value(attr: &AttributeMessage) -> IoResult<String> {
    use crate::format::messages::datatype::{fixed_string_content, DatatypeMessage};
    let padding = match attr.datatype {
        DatatypeMessage::FixedString { padding, .. } => padding,
        _ => 0,
    };
    let content = fixed_string_content(&attr.data, padding).ok_or_else(|| {
        crate::io::IoError::InvalidState(format!(
            "attribute {:?} uses string padding rule {padding}, which the format reserves",
            attr.name
        ))
    })?;
    Ok(String::from_utf8_lossy(content).to_string())
}

/// Materialize a `total`-byte fill buffer, mapping allocation failure to a
/// clean error. `total` on a read path comes from untrusted file fields, so
/// a crafted file declaring an absurd dataset size would otherwise abort the
/// process when `vec![0u8; total]` fails to allocate.
fn alloc_tiled_fill(total: usize, fill_value: Option<&[u8]>) -> IoResult<Vec<u8>> {
    try_tiled_fill(total, fill_value).map_err(|_| {
        crate::io::IoError::InvalidState(format!(
            "cannot allocate {total} bytes for dataset buffer (file may be corrupt)"
        ))
    })
}

/// Build a `Vec<T>` of `count` elements out of the bytes `define` writes.
///
/// `define` receives the vector's whole byte image — `count *
/// size_of::<T>()` bytes — and **must define every one of them** before it
/// returns `Ok`; only then are the elements claimed. That contract is what
/// lets a full-image read land in its destination directly: the buffer a read
/// fills is already the buffer the caller keeps, so a 128 MiB image is
/// touched once by the read instead of once to zero it, once to read it and
/// once to copy it into a typed vector.
///
/// [`Hdf5Reader::read_dataset_raw_into_unconverted`] is the read side of that
/// contract — it is the single owner of read-destination semantics precisely
/// because it defines every byte of the buffer it is handed — and
/// [`read_dataset_raw_into`](Hdf5Reader::read_dataset_raw_into) inherits it.
///
/// `count` on a read path comes from untrusted file fields, so the
/// reservation is fallible: a crafted file declaring an absurd dataset size
/// gets a clean error rather than an allocator abort.
pub(crate) fn read_image_into_new<T, E, F>(count: usize, define: F) -> Result<Vec<T>, E>
where
    T: crate::types::H5Type,
    F: FnOnce(&mut [u8]) -> Result<(), E>,
    E: From<crate::io::IoError>,
{
    let too_big = || {
        E::from(crate::io::IoError::InvalidState(format!(
            "cannot allocate {count} elements of {} bytes for a dataset buffer \
             (file may be corrupt)",
            std::mem::size_of::<T>()
        )))
    };
    let bytes = count
        .checked_mul(std::mem::size_of::<T>())
        .ok_or_else(too_big)?;
    let mut out: Vec<T> = Vec::new();
    out.try_reserve_exact(count).map_err(|_| too_big())?;

    // Safety: `try_reserve_exact` succeeded, so the allocation holds `bytes`
    // contiguous bytes aligned for `T`, and `as_mut_ptr` is non-null (a
    // dangling-but-aligned pointer with `bytes == 0`, which an empty slice
    // permits). The slice is the only live reference to that memory while
    // `define` runs. `define` writes every byte before returning `Ok` — its
    // documented contract — so the elements are initialized by the time
    // `set_len` claims them, and every byte pattern is a valid `T`, the same
    // property of `H5Type` implementors that a typed read reinterpreting the
    // stored image already rests on. A `define` that fails returns before
    // `set_len`, so the vector drops empty and nothing reads the bytes it
    // left undefined.
    let image = unsafe { std::slice::from_raw_parts_mut(out.as_mut_ptr().cast::<u8>(), bytes) };
    define(image)?;
    unsafe { out.set_len(count) };
    Ok(out)
}

/// Fill an existing buffer in place with the dataset's tiled fill value (or
/// zero when no fill value is set), matching [`try_tiled_fill`]'s tiling.
///
/// Used to initialize a read destination — both an internally-allocated `Vec`
/// and a caller-provided read-into buffer (whose prior contents are arbitrary)
/// — so any region a chunked read leaves untouched reads back as the fill
/// value rather than stale bytes.
fn fill_tiled_into(out: &mut [u8], fill_value: Option<&[u8]>) {
    out.fill(0);
    if let Some(fv) = fill_value {
        if !fv.is_empty() && !out.is_empty() {
            for slot in out.chunks_mut(fv.len()) {
                let n = slot.len().min(fv.len());
                slot[..n].copy_from_slice(&fv[..n]);
            }
        }
    }
}

/// Resolve the directory a raw-data file name is joined against, matching
/// libhdf5's `H5D__build_file_prefix` (H5Dint.c) for the given environment
/// variable — `HDF5_EXTFILE_PREFIX` for External Data Files
/// ([`resolve_extfile_prefix`]), `HDF5_VDS_PREFIX` for Virtual Dataset
/// sources ([`resolve_vdsfile_prefix`]); both features route through the
/// same C function, just keyed on a different variable. `prop` is the dapl
/// property the variable falls back to, `None` when the caller has none —
/// which behaves exactly as an unset or empty one would.
///
/// `${ORIGIN}` expands to `source_dir` (the directory holding the open
/// HDF5 file); any other value is used as a literal prefix; unset, empty,
/// or `"."` means "no prefix" (`H5_combine_path`'s own default), so a
/// relative name resolves against the process's current directory instead.
fn resolve_file_prefix(env_var: &str, prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
    // `H5D__build_file_prefix` reads the environment variable first and only
    // falls back to the property list when it is unset or empty
    // (H5Dint.c:1077-1082, :1085-1090) — so an environment prefix *shadows*
    // the property rather than being tried before it. What
    // `H5F_prefix_open_file` then tries before this is the raw environment
    // string, split on `:` and unexpanded, which is a different candidate
    // from the expansion built here.
    let env = std::env::var(env_var).ok().filter(|v| !v.is_empty());
    let prefix = match env.as_deref() {
        Some(v) => v,
        None => prop.filter(|v| !v.is_empty())?,
    };
    if prefix.is_empty() || prefix == "." {
        return None;
    }
    Some(match prefix.strip_prefix("${ORIGIN}") {
        Some(rest) => {
            let rest = rest.trim_start_matches(['/', '\\']);
            if rest.is_empty() {
                source_dir.to_path_buf()
            } else {
                source_dir.join(rest)
            }
        }
        None => PathBuf::from(prefix),
    })
}

/// Resolve the directory an external file list's stored names are joined
/// against — `HDF5_EXTFILE_PREFIX`, or `H5Pset_efile_prefix` when the
/// environment names none (see [`resolve_file_prefix`]).
///
/// The single owner of that rule for both directions of I/O: `H5D__efl_read`
/// and `H5D__efl_write` join against the same `shared->extfile_prefix`
/// (H5Defl.c:315-317, :429-431), which `H5D__build_file_prefix` built once
/// for the open that created the dataset's shared info.
pub(crate) fn resolve_extfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
    resolve_file_prefix("HDF5_EXTFILE_PREFIX", prop, source_dir)
}

/// Resolve the directory Virtual Dataset source file names are joined
/// against — `HDF5_VDS_PREFIX`, or `H5Pset_virtual_prefix` when the
/// environment names none (see [`resolve_file_prefix`]).
fn resolve_vdsfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
    resolve_file_prefix("HDF5_VDS_PREFIX", prop, source_dir)
}

/// Join a raw-data file `name` against a resolved prefix, matching
/// libhdf5's `H5_combine_path` (H5system.c): an absolute `name` is used
/// as-is regardless of the prefix, and no prefix means "relative to the
/// process's current directory" — both of which `Path::join` already
/// implements for an absolute joinee. Shared by External Data Files and
/// Virtual Dataset source resolution — both call the same C function.
pub(crate) fn combine_prefixed_path(prefix: Option<&Path>, name: &str) -> PathBuf {
    match prefix {
        Some(p) => p.join(name),
        None => PathBuf::from(name),
    }
}

/// Read `len` bytes starting at *dataset-relative* offset `skip` from an
/// external file list into `out`, walking slots by cumulative declared
/// size exactly like libhdf5's `H5D__efl_read` (H5Defl.c). A read past an
/// individual slot's actual on-disk length reads back as zero — the file
/// backing a slot may be shorter than the space the layout reserved in
/// it — but a read past the *total* declared size of the file list is
/// still an error, matching `H5D__efl_read`'s own "read past logical end
/// of file" check.
///
/// An `H5O_EFL_UNLIMITED` last slot needs no special case: the walk below
/// never steps past it (`skip >= u64::MAX` is never true), which is upstream's
/// `H5O_EFL_UNLIMITED == size || addr < cur + size`, and the read it then
/// takes is the whole remainder, bounded by whatever the file physically
/// holds.
fn read_external_file_bytes(
    external_files: &[ExternalFileSegment],
    extfile_prefix: Option<&Path>,
    mut skip: u64,
    out: &mut [u8],
) -> IoResult<()> {
    let mut slot_idx = 0usize;
    while slot_idx < external_files.len() && skip >= external_files[slot_idx].size {
        skip -= external_files[slot_idx].size;
        slot_idx += 1;
    }

    let mut written = 0usize;
    while written < out.len() {
        let Some(slot) = external_files.get(slot_idx) else {
            return Err(crate::io::IoError::InvalidState(
                "read past the logical end of the external file list".into(),
            ));
        };
        let full_path = combine_prefixed_path(extfile_prefix, &slot.name);
        let ext_handle = FileHandle::open_read_with_locking(&full_path, FileLocking::Disabled)
            .map_err(|e| {
                crate::io::IoError::InvalidState(format!(
                    "unable to open external raw data file {}: {e}",
                    full_path.display()
                ))
            })?;
        let avail_in_slot = slot.size.saturating_sub(skip);
        let want = (out.len() - written) as u64;
        let this_read = avail_in_slot.min(want) as usize;
        let dst = &mut out[written..written + this_read];
        // A short physical file — the reserved slot size exceeds what was
        // ever actually written to it — reads back as zero for the
        // remainder, exactly like `H5D__efl_read`.
        let at = slot.offset.checked_add(skip).ok_or_else(|| {
            crate::io::IoError::InvalidState(format!(
                "external file '{}' slot offset {} overflows {skip} bytes into the slot",
                slot.name, slot.offset
            ))
        })?;
        let got = ext_handle.read_at_most(at, this_read)?;
        dst[..got.len()].copy_from_slice(&got);
        dst[got.len()..].fill(0);

        written += this_read;
        skip = 0;
        slot_idx += 1;
    }
    Ok(())
}

/// Recursion ceiling for virtual dataset nesting (a VDS whose source is
/// itself a VDS — possibly in another file). Bounded so a crafted cyclic
/// mapping chain fails cleanly instead of recursing until the stack
/// overflows; real VDS chains do not nest anywhere near this deep.
const MAX_VIRTUAL_DEPTH: usize = 16;

/// Replace every unlimited mapping with the concrete mapping its open-time
/// resolution makes it — the clipped selections `H5D__virtual_set_extent_unlim`
/// leaves in `clipped_virtual_select` / `clipped_source_select` for a read to
/// use (`H5D__virtual_read` never sees the unclipped ones).
///
/// A mapping with no resolution recorded is passed through unchanged, which is
/// what a bounded mapping needs and what a mapping list written before this
/// pass existed reduces to.
fn concrete_virtual_mappings(
    list: &VirtualMappingList,
    resolution: &[MappingResolution],
) -> IoResult<Vec<VirtualMapping>> {
    let mut out = Vec::with_capacity(list.mappings.len());
    for (i, m) in list.mappings.iter().enumerate() {
        match resolution.get(i) {
            Some(MappingResolution::Unlimited {
                virtual_clip,
                source_clip,
            }) => out.push(VirtualMapping {
                virtual_selection: m.virtual_selection.clip_unlimited(*virtual_clip)?,
                source_selection: m.source_selection.clip_unlimited(*source_clip)?,
                ..built_names(m, 0)?
            }),
            // One printf mapping is a whole family: block `j` of the virtual
            // selection (`H5S_hyper_get_unlim_block`) is filled by the source
            // dataset whose name substitutes `j`, taking the mapping's whole
            // (limited) source selection. Only the blocks that have a source
            // become mappings — a non-zero printf gap leaves the others
            // inside the extent, reading as the fill value.
            Some(MappingResolution::Printf { present, .. }) => {
                let Some(r) = regular_hyperslab(&m.virtual_selection) else {
                    continue;
                };
                let rank = r.start.len();
                for &j in present {
                    out.push(VirtualMapping {
                        virtual_selection: Selection::Hyperslab {
                            rank,
                            form: Hyperslab::Regular(r.unlim_block(j)),
                        },
                        ..built_names(m, j)?
                    });
                }
            }
            _ => out.push(built_names(m, 0)?),
        }
    }
    Ok(out)
}

/// One mapping with both source names built for block `blockno` —
/// `H5D__virtual_build_source_name`. A mapping with no substitutions still
/// goes through this, because that is where `%%` is unescaped: upstream uses
/// the parsed name rather than the stored one for an ordinary mapping too
/// (`H5D__virtual_load_layout`).
fn built_names(m: &VirtualMapping, blockno: u64) -> IoResult<VirtualMapping> {
    Ok(VirtualMapping {
        source_file_name: parse_source_name(&m.source_file_name)?.build(blockno),
        source_dset_name: parse_source_name(&m.source_dset_name)?.build(blockno),
        ..m.clone()
    })
}

/// Recursion bound for open-time virtual-extent resolution.
///
/// Resolving a virtual dataset's extent opens the source datasets its
/// unlimited mappings name, and a source may itself be a virtual dataset in
/// another file whose own open resolves its own extent. A crafted cyclic
/// chain would otherwise recurse until the stack overflows, so the nesting is
/// counted per thread and the resolution is skipped once it reaches
/// [`MAX_VIRTUAL_DEPTH`] — a dataset that deep keeps its stored extent
/// instead of taking one from a cycle.
struct VirtualResolveDepth;

thread_local! {
    static VIRTUAL_RESOLVE_DEPTH: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
}

impl VirtualResolveDepth {
    fn enter<F: FnOnce() -> IoResult<()>>(f: F) -> IoResult<()> {
        let depth = VIRTUAL_RESOLVE_DEPTH.with(std::cell::Cell::get);
        if depth >= MAX_VIRTUAL_DEPTH {
            return Ok(());
        }
        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth + 1));
        let out = f();
        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth));
        out
    }
}

/// The regular (start, stride, count, block) form behind a selection, or
/// `None` — the only form `H5S_UNLIMITED` can appear in, so every unlimited
/// computation goes through it.
fn regular_hyperslab(sel: &Selection) -> Option<&RegularHyperslab> {
    match sel {
        Selection::Hyperslab {
            form: Hyperslab::Regular(r),
            ..
        } => Some(r),
        _ => None,
    }
}

/// Read `source` (via `read_source_box`, one call per box) and scatter it
/// into `out` at the elements `target` selects.
///
/// The two selections are paired element by element in their own linear
/// order — `H5S_select_project_intersection` (H5Sselect.c:2402) walks a VDS
/// mapping's virtual and source selections with one iterator each and
/// matches the two streams off one against one. Its only precondition is the
/// one asserted there and enforced by `H5D_virtual_check_mapping_pre` when
/// the mapping is written (H5Dvirtual.c:254-257): the two hold the same
/// number of elements. Ranks may differ, and so may the boxes each side
/// decomposes into — a source `H5S_SEL_ALL` over a 2x4 dataset legitimately
/// fills two 1x4 blocks of a virtual dataset.
///
/// Each source box is read whole, once, and the runs the pairing places from
/// it are copied out of that one buffer.
fn copy_matched_selections(
    mut read_source_box: impl FnMut(&[u64], &[u64], &mut [u8]) -> IoResult<()>,
    source: &ResolvedSelection,
    target: &ResolvedSelection,
    element_size: u64,
    out: &mut [u8],
) -> IoResult<()> {
    let (n_source, n_target) = (source.n_elements(), target.n_elements());
    if n_source != n_target {
        return Err(crate::io::IoError::InvalidState(format!(
            "virtual dataset mapping's source selection holds {n_source} elements and its              virtual selection {n_target}, which H5D_virtual_check_mapping_pre refuses"
        )));
    }
    // Pair the two element streams, splitting a run of either side wherever
    // the other side's run ends first, and file each matched segment under
    // the source box it must be read out of.
    let mut per_box: Vec<Vec<(u64, u64, u64)>> = vec![Vec::new(); source.boxes.len()];
    let (mut si, mut ti) = (0usize, 0usize);
    let (mut s_done, mut t_done) = (0u64, 0u64);
    while si < source.runs.len() && ti < target.runs.len() {
        let (s, t) = (source.runs[si], target.runs[ti]);
        let len = (s.len - s_done).min(t.len - t_done);
        per_box[s.box_index].push((s.offset_in_box + s_done, t.offset_in_extent + t_done, len));
        s_done += len;
        t_done += len;
        if s_done == s.len {
            si += 1;
            s_done = 0;
        }
        if t_done == t.len {
            ti += 1;
            t_done = 0;
        }
    }

    for (segments, (box_start, box_count)) in per_box.iter().zip(&source.boxes) {
        if segments.is_empty() {
            continue;
        }
        let nbytes = saturating_byte_len(box_count, element_size) as usize;
        let mut buf = alloc_tiled_fill(nbytes, None)?;
        read_source_box(box_start, box_count, &mut buf)?;
        for &(from, to, len) in segments {
            let (from, to, len) = (
                (from * element_size) as usize,
                (to * element_size) as usize,
                (len * element_size) as usize,
            );
            let src = buf.get(from..from + len).ok_or_else(|| {
                crate::io::IoError::InvalidState(
                    "virtual dataset mapping's source selection reaches past its source box".into(),
                )
            })?;
            let dst = out.get_mut(to..to + len).ok_or_else(|| {
                crate::io::IoError::InvalidState(
                    "virtual dataset mapping's virtual selection reaches past the dataset".into(),
                )
            })?;
            dst.copy_from_slice(src);
        }
    }
    Ok(())
}

/// Read and decode the global-heap collection at `addr`, applying the
/// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
/// signature must be present and the declared size at least `H5HG_MINSIZE`
/// (4096 bytes). There is no upper size cap — libhdf5 has none, and this
/// crate's writers put a whole write call's strings into one collection,
/// which a cap would turn into silent data loss.
///
/// A free function (not a method) so both [`Hdf5Reader::read_heap_collection`]
/// and the static dataset-open path (which only has a `&mut FileHandle`, not
/// a full `&mut Hdf5Reader`) share the one implementation.
fn read_heap_collection_from(
    handle: &mut FileHandle,
    ctx: &FormatContext,
    addr: u64,
) -> IoResult<GlobalHeapCollection> {
    let ss = ctx.sizeof_size as usize;
    let header_len = 4 + 1 + 3 + ss;
    let header_buf = handle.read_at_most(addr, header_len)?;
    if header_buf.len() < header_len || header_buf[0..4] != *b"GCOL" {
        return Err(crate::io::IoError::InvalidState(format!(
            "bad global heap collection signature at address {addr:#x}"
        )));
    }
    let declared = read_uint(&header_buf[8..], ss) as usize;
    if declared < 4096 {
        return Err(crate::io::IoError::InvalidState(format!(
            "global heap collection at address {addr:#x} declares size {declared}, \
             below the 4096-byte minimum"
        )));
    }
    let heap_buf = handle.read_at(addr, declared)?;
    let (coll, _) = GlobalHeapCollection::decode(&heap_buf, ctx)?;
    Ok(coll)
}

/// One chunk's on-disk read request, built by a read path before any I/O.
struct ChunkReadJob {
    /// Byte offset of the chunk's stored bytes.
    addr: u64,
    /// Number of bytes to read.
    len: usize,
    /// `true` → [`FileHandle::read_at_most`] (short reads near EOF are fine);
    /// `false` → [`FileHandle::read_at`] (exact, errors on a short read).
    at_most: bool,
    /// Per-chunk filter mask (ignored when the pipeline is `None`).
    mask: u32,
}

/// Read one chunk's raw bytes according to its job.
fn read_chunk_raw(handle: &FileHandle, j: &ChunkReadJob) -> IoResult<Vec<u8>> {
    if j.at_most {
        Ok(handle.read_at_most(j.addr, j.len)?)
    } else {
        Ok(handle.read_at(j.addr, j.len)?)
    }
}

/// Memory traffic one positioned read is worth: a `pread` costs about what
/// copying this many bytes costs. It is the exchange rate
/// [`read_chunk_runs_into`] weighs when a chunk's selected runs could each be
/// read on their own instead of reading the chunk whole and copying them out.
const PREAD_COST_BYTES: u64 = 16 * 1024;

/// One chunk's intersection with the read box, in global dataset indices.
///
/// The single derivation of chunk ∩ selection ∩ extent:
/// [`for_each_chunk_run`] decomposes this box into contiguous runs and
/// [`ChunkOverlap::bytes`] measures it, so what a plan is measured to cover
/// and what the same plan places can never disagree.
struct ChunkOverlap {
    lo: Vec<u64>,
    hi: Vec<u64>,
}

impl ChunkOverlap {
    /// `None` when the chunk and the read box do not meet. A corrupt index can
    /// place a chunk past the extent, so the arithmetic saturates and an empty
    /// span simply yields `None`.
    fn of(place: &ChunkPlacement, chunk_coords: &[u64]) -> Option<Self> {
        let ChunkPlacement {
            geo: ChunkOutputGeometry {
                dims, chunk_dims, ..
            },
            starts,
            counts,
        } = *place;
        let ndims = dims.len();
        if ndims == 0 {
            return None;
        }
        let mut lo = vec![0u64; ndims];
        let mut hi = vec![0u64; ndims];
        for d in 0..ndims {
            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
            let chunk_end = origin.saturating_add(chunk_dims[d]).min(dims[d]);
            let sel_end = starts[d].saturating_add(counts[d]);
            lo[d] = origin.max(starts[d]);
            hi[d] = chunk_end.min(sel_end);
            if lo[d] >= hi[d] {
                return None;
            }
        }
        Some(Self { lo, hi })
    }

    /// Output bytes this overlap covers — exactly the summed length of the
    /// runs [`for_each_chunk_run`] yields for the same chunk.
    fn bytes(&self, element_size: u64) -> u64 {
        self.lo
            .iter()
            .zip(&self.hi)
            .map(|(&l, &h)| h - l)
            .product::<u64>()
            .saturating_mul(element_size)
    }
}

/// Output bytes a chunk plan covers, counting each chunk-grid slot once.
///
/// Distinct slots own disjoint boxes of the index space and one chunk's runs
/// are disjoint from each other, so this sum is the exact volume of their
/// union — provided a slot a corrupt index names twice is counted once, which
/// is what `seen` is for. The caller compares it against the output length:
/// anything short means some output byte belongs to a chunk this plan will not
/// place, and only then does the fill value have to go down first.
fn planned_coverage(place: &ChunkPlacement, jobs: &[Option<ChunkReadJob>], coords: &[u64]) -> u64 {
    let rank = place.geo.dims.len();
    if rank == 0 {
        return 0;
    }
    let mut seen: std::collections::HashSet<&[u64]> = std::collections::HashSet::new();
    let mut covered = 0u64;
    for (i, job) in jobs.iter().enumerate() {
        if job.is_none() {
            continue;
        }
        let c = &coords[i * rank..(i + 1) * rank];
        if !seen.insert(c) {
            continue;
        }
        if let Some(overlap) = ChunkOverlap::of(place, c) {
            covered = covered.saturating_add(overlap.bytes(place.geo.element_size));
        }
    }
    covered
}

/// Walk the contiguous byte runs of one chunk's intersection with the read box
/// `[starts, starts + counts)`, as `(offset in the chunk image, offset in the
/// output, run length in bytes)`.
///
/// The last axis is innermost in both the chunk and the output, so the
/// intersection box decomposes into one contiguous run per setting of the
/// outer axes — no per-element loop. The single owner of chunk placement
/// geometry: [`copy_chunk_runs`] copies these runs out of a decoded chunk
/// image and [`read_chunk_runs_into`] reads the same runs straight out of the
/// file, so the two cannot place a byte differently. A full read passes the
/// whole extent as the box, which is why it needs no placement path of its
/// own.
fn for_each_chunk_run(
    place: &ChunkPlacement,
    chunk_coords: &[u64],
    mut f: impl FnMut(u64, u64, usize),
) {
    let ChunkPlacement {
        geo:
            ChunkOutputGeometry {
                dims,
                chunk_dims,
                element_size,
            },
        starts,
        counts,
    } = *place;
    let ndims = dims.len();
    // Global intersection box [lo, hi) of chunk ∩ selection ∩ dataset; `None`
    // covers the rank-0 case the indexing below could not survive.
    let Some(ChunkOverlap { lo, hi }) = ChunkOverlap::of(place, chunk_coords) else {
        return;
    };

    let chunk_strides = compute_strides(chunk_dims, element_size);
    let out_strides = compute_strides(counts, element_size);
    let last = ndims - 1;
    let run_bytes = ((hi[last] - lo[last]) * element_size) as usize;

    // Iterate the outer box dimensions [0, last); each position places one
    // contiguous last-axis run.
    let outer_extent: Vec<u64> = (0..last).map(|d| hi[d] - lo[d]).collect();
    let n_outer: u64 = outer_extent.iter().product(); // empty product == 1
    let mut oc = vec![0u64; last];
    for _ in 0..n_outer {
        let mut src_off = 0u64;
        let mut dst_off = 0u64;
        for d in 0..ndims {
            let g = if d < last { lo[d] + oc[d] } else { lo[last] };
            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
            src_off += (g - origin) * chunk_strides[d];
            dst_off += (g - starts[d]) * out_strides[d];
        }
        f(src_off, dst_off, run_bytes);
        for d in (0..last).rev() {
            oc[d] += 1;
            if oc[d] < outer_extent[d] {
                break;
            }
            oc[d] = 0;
        }
    }
}

/// Copy one decoded chunk image's intersection with the read box into
/// `output`.
///
/// A run reaching past the image — a chunk read short at the end of the file —
/// or past the output cannot be copied; it is appended to `skipped` as the
/// output range `(offset, length)` nothing wrote, which is what
/// [`place_chunk_jobs`] fills with the fill value.
fn copy_chunk_runs(
    chunk_data: &[u8],
    output: &mut [u8],
    place: &ChunkPlacement,
    chunk_coords: &[u64],
    skipped: &mut Vec<(usize, usize)>,
) {
    for_each_chunk_run(place, chunk_coords, |src, dst, len| {
        let (s, d) = (src as usize, dst as usize);
        if s + len <= chunk_data.len() && d + len <= output.len() {
            output[d..d + len].copy_from_slice(&chunk_data[s..s + len]);
        } else {
            push_skipped(skipped, output.len(), d, len);
        }
    });
}

/// Record the output range `[dst, dst + len)` as one nothing wrote, clamped to
/// the output so a range a corrupt geometry pushed past the end never widens
/// into a slice the fill would panic on.
fn push_skipped(skipped: &mut Vec<(usize, usize)>, out_len: usize, dst: usize, len: usize) {
    let start = dst.min(out_len);
    let end = dst.saturating_add(len).min(out_len);
    if start < end {
        skipped.push((start, end - start));
    }
}

/// Read one unfiltered chunk's intersection with the read box straight from
/// the file into `output`, never materializing the chunk.
///
/// An unfiltered chunk's stored bytes are the dataset's bytes in the dataset's
/// own order, so every run of the intersection is one positioned read — the
/// byte ranges libhdf5 reads for the same selection instead of the whole chunk
/// (`H5D__chunk_read`, H5Dchunk.c). Adjacent runs coalesce into one read.
/// `job.len` is how many bytes of the chunk the whole-chunk read would have
/// held, so the runs placed here are exactly the runs
/// [`copy_chunk_runs`] would have placed out of that image.
///
/// Returns whether the chunk's bytes landed in `output`; `false` means the
/// caller must still read the chunk whole, and the `skipped` ranges this call
/// appended are the caller's to discard (the whole-chunk path records its
/// own). That happens when the runs are too
/// small for a read each to beat one whole-chunk read plus the copy out of it,
/// and when a read fails — the whole-chunk path owns both the error and the
/// short read a truncated file gives, and re-places any run already placed
/// here from the same file offsets, so a fallback never leaves a half-written
/// output.
fn read_chunk_runs_into(
    handle: &FileHandle,
    job: &ChunkReadJob,
    place: &ChunkPlacement,
    chunk_coords: &[u64],
    output: &mut [u8],
    skipped: &mut Vec<(usize, usize)>,
    dst: ReadDst,
) -> bool {
    let (addr, image_len) = (job.addr, job.len);
    // (file offset, offset in output, length), coalesced as they are planned.
    let mut runs: Vec<(u64, usize, usize)> = Vec::new();
    let mut selected = 0u64;
    let out_len = output.len();
    for_each_chunk_run(place, chunk_coords, |src, out_off, len| {
        let (s, d) = (src as usize, out_off as usize);
        if s + len > image_len || d + len > out_len {
            push_skipped(skipped, out_len, d, len);
            return;
        }
        selected += len as u64;
        // `addr` is the index's claim: a run end that does not fit in a u64
        // simply does not coalesce, and the read at that address fails on
        // its own terms.
        let at = addr.saturating_add(src);
        if let Some(last) = runs.last_mut() {
            if last.0.checked_add(last.2 as u64) == Some(at) && last.1 + last.2 == d {
                last.2 += len;
                return;
            }
        }
        runs.push((at, d, len));
    });
    if runs.is_empty() {
        // Nothing to place, which only a chunk outside the extent or a short
        // image produces: leave it to the whole-chunk path so a chunk that
        // cannot be read at all still fails there.
        return false;
    }
    // One read per run pays off while the extra syscalls cost less than the
    // whole-chunk read and the copy out of it they replace.
    if (runs.len() as u64 - 1).saturating_mul(PREAD_COST_BYTES) > image_len as u64 + selected {
        return false;
    }
    for (offset, at, len) in runs {
        if handle
            .read_exact_at_into(offset, &mut output[at..at + len], dst)
            .is_err()
        {
            return false;
        }
    }
    true
}

/// Place every chunk a chunked read planned into `output`.
///
/// The single owner of "planned chunk → output bytes" for all five chunk index
/// types: each read path walks its own index and builds the jobs, and this
/// decides, per chunk, whether the read box's byte runs come straight out of
/// the file ([`read_chunk_runs_into`]) or whether the chunk has to be read and
/// decoded whole first ([`copy_chunk_runs`]). A filtered chunk's stored bytes
/// have no byte-range correspondence to the dataset's, so it always reads
/// whole. `coords` is the packed chunk-grid position table
/// ([`crate::io::chunk_grid::coords_table`]): job `i` sits at rank-many values
/// from `i * rank`.
///
/// It is also the single owner of the fill: every byte of `output` this
/// returns `Ok` on is either a byte some chunk placed or a byte filled with
/// the tiled fill value, so callers hand it an arbitrary (even uninitialized)
/// buffer and a chunked read that reaches no chunk at all still routes through
/// here with an empty job list. The fill is derived from the plan rather than
/// laid down blanket-first: when [`planned_coverage`] proves the plan covers
/// the whole output, a 128 MiB read never pays a 128 MiB memset it is about to
/// overwrite, and only the runs a short chunk image left unwritten are filled
/// afterwards.
///
/// `cache` is the dataset's [`ChunkImageCache`] — present for every read that
/// reached its chunks through a decoded index. It is consulted only for a
/// filtered chunk this read leaves partly unconsumed: an unfiltered chunk's
/// selected runs come straight out of the file below and never materialize an
/// image there is anything to keep, and a chunk this read takes entire is one
/// no later read can want more of.
fn place_chunk_jobs(
    handle: &FileHandle,
    mut jobs: Vec<Option<ChunkReadJob>>,
    coords: &[u64],
    req: ChunkReadRequest,
    geo: &ChunkOutputGeometry,
    cache: Option<&ChunkImageCache>,
    output: &mut [u8],
) -> IoResult<()> {
    let ChunkReadRequest {
        pipeline,
        target,
        fill_value,
        dst,
    } = req;
    let rank = geo.dims.len();
    let zeros = vec![0u64; rank];
    let place = ChunkPlacement::resolve(geo, target, &zeros);
    let at = |i: usize| &coords[i * rank..(i + 1) * rank];

    // Fill first unless the plan already accounts for every output byte; then
    // only the runs placement could not honour need filling afterwards.
    let prefilled = planned_coverage(&place, &jobs, coords) != output.len() as u64;
    if prefilled {
        fill_tiled_into(output, fill_value);
    }
    let mut skipped: Vec<(usize, usize)> = Vec::new();

    if pipeline.is_none() {
        for (i, job) in jobs.iter_mut().enumerate() {
            let Some(j) = job.as_ref() else { continue };
            let mark = skipped.len();
            if read_chunk_runs_into(handle, j, &place, at(i), output, &mut skipped, dst) {
                *job = None;
            } else {
                // The whole-chunk path re-places this chunk and records what
                // it could not place; a half-planned attempt must not leave a
                // range behind that the fill would then write over good bytes.
                skipped.truncate(mark);
            }
        }
    }

    // A filtered chunk this read only partly consumes is worth an image: the
    // next slice of it pays a copy instead of a second inflate. `hits[i]` is
    // an image the cache already held — its job is dropped, so nothing reads
    // or decodes it — and `keys[i]` marks a chunk whose image this read is to
    // hand over once it has one.
    let mut hits: Vec<Option<std::sync::Arc<Vec<u8>>>> = Vec::new();
    let mut keys: Vec<Option<ChunkImageKey>> = Vec::new();
    let cache = cache.filter(|_| pipeline.is_some());
    if let Some(cache) = cache {
        hits.resize_with(jobs.len(), || None);
        keys.resize(jobs.len(), None);
        for (i, job) in jobs.iter_mut().enumerate() {
            let Some(j) = job.as_ref() else { continue };
            if !place.leaves_chunk_unconsumed(at(i)) {
                continue;
            }
            let key = ChunkImageKey {
                addr: j.addr,
                len: j.len,
                mask: j.mask,
            };
            match cache.get(&key) {
                Some(image) => {
                    hits[i] = Some(image);
                    *job = None;
                }
                None => keys[i] = Some(key),
            }
        }
        for (i, image) in hits.iter().enumerate() {
            if let Some(image) = image {
                copy_chunk_runs(image, output, &place, at(i), &mut skipped);
            }
        }
    }

    if jobs.iter().any(Option::is_some) {
        // Every chunk whose image is a contiguous stretch of `output` decodes
        // into that stretch; the rest come back as images to scatter. The
        // borrow of `output` the sinks hold ends with the call, which returns
        // nothing that points into it.
        let decoded = {
            let sinks = carve_sinks(output, &jobs, coords, &place);
            read_and_decompress_chunks(handle, pipeline, jobs, sinks, geo.image_bytes())?
        };
        for (i, chunk) in decoded.into_iter().enumerate() {
            match chunk {
                ChunkDecoded::Absent => {}
                // A chunk marked for the cache is by construction a staged
                // one: a chunk whose image is a contiguous stretch of the
                // output is a chunk this read consumes entire, which
                // `ChunkPlacement::leaves_chunk_unconsumed` refuses.
                ChunkDecoded::Image(data) => {
                    match keys.get_mut(i).and_then(Option::take).zip(cache) {
                        Some((key, cache)) => {
                            let image = cache.keep(key, data);
                            copy_chunk_runs(&image, output, &place, at(i), &mut skipped)
                        }
                        None => copy_chunk_runs(&data, output, &place, at(i), &mut skipped),
                    }
                }
                // A chunk that decoded short of its image placed no usable run
                // — the same verdict `copy_chunk_runs` passes on a run reaching
                // past a short image — so the whole stretch reverts to fill,
                // not just the part past the image: a decode writes its output
                // in blocks and may have reached past the byte it stopped on.
                ChunkDecoded::InPlace { dst, len, bytes } if bytes < len => {
                    fill_tiled_into(&mut output[dst..dst + len], fill_value)
                }
                ChunkDecoded::InPlace { .. } => {}
            }
        }
    }
    if !prefilled {
        for (dst, len) in skipped {
            fill_tiled_into(&mut output[dst..dst + len], fill_value);
        }
    }
    Ok(())
}

/// Where one planned chunk's decoded image lands.
enum ChunkSink<'a> {
    /// The chunk's whole image is this stretch of the read's output, starting
    /// at output offset `dst`: the decoder writes it in place and nothing is
    /// copied afterwards.
    Direct { dst: usize, out: &'a mut [u8] },
    /// The image has no contiguous home in the output; it is materialized and
    /// then placed run by run.
    Staged,
}

/// What one planned chunk left for the placement step.
enum ChunkDecoded {
    /// The slot planned no chunk (`jobs[i]` was `None`).
    Absent,
    /// The image is here and still has to be placed run by run.
    Image(Vec<u8>),
    /// The image went straight into `output[dst..dst + len]`. `bytes` is the
    /// length the pipeline produced there: short of `len` only for a stored
    /// chunk that decoded to less than its image.
    InPlace {
        dst: usize,
        len: usize,
        bytes: usize,
    },
}

/// Hand every chunk whose image is one contiguous stretch of `output` that
/// stretch to decode into, and stage every other chunk.
///
/// A chunk's image *is* the output's own bytes exactly when its intersection
/// with the read box is a single run that starts at the image's first byte and
/// carries the image's whole length — the whole chunk, laid down contiguously.
/// The runs come from [`for_each_chunk_run`], the same walk that would have
/// copied the image out, so a stretch handed out here is byte-for-byte the
/// stretch the copy would have written.
///
/// Distinct chunk-grid slots own disjoint boxes of the output, so their
/// stretches never overlap; a corrupt index naming one slot twice would break
/// that, so a stretch overlapping one already handed out is staged instead.
fn carve_sinks<'a>(
    output: &'a mut [u8],
    jobs: &[Option<ChunkReadJob>],
    coords: &[u64],
    place: &ChunkPlacement,
) -> Vec<ChunkSink<'a>> {
    let mut sinks = Vec::with_capacity(jobs.len());
    sinks.resize_with(jobs.len(), || ChunkSink::Staged);
    let rank = place.geo.dims.len();
    let out_len = output.len();
    if rank == 0 {
        return sinks;
    }
    let Some(image_bytes) = place.geo.image_bytes() else {
        return sinks;
    };
    // A chunk is whole inside the read box only if the box is at least a chunk
    // wide in every dimension: a narrower selection has no direct chunk at all,
    // and testing it once here spares it the per-chunk walk.
    if place
        .counts
        .iter()
        .zip(place.geo.chunk_dims)
        .any(|(c, k)| c < k)
    {
        return sinks;
    }

    let mut wanted: Vec<(usize, usize)> = Vec::new();
    for (i, job) in jobs.iter().enumerate() {
        if job.is_none() {
            continue;
        }
        let mut runs = 0usize;
        let mut first = (0u64, 0u64, 0usize);
        for_each_chunk_run(place, &coords[i * rank..(i + 1) * rank], |src, dst, len| {
            if runs == 0 {
                first = (src, dst, len);
            }
            runs += 1;
        });
        let (src, dst, len) = first;
        if runs == 1
            && src == 0
            && len as u64 == image_bytes
            && dst.saturating_add(len as u64) <= out_len as u64
        {
            wanted.push((dst as usize, i));
        }
    }
    wanted.sort_unstable();

    let len = image_bytes as usize;
    let mut rest: &'a mut [u8] = output;
    let mut base = 0usize;
    for (dst, i) in wanted {
        if dst < base {
            continue;
        }
        let (_, tail) = std::mem::take(&mut rest).split_at_mut(dst - base);
        let (mine, tail) = tail.split_at_mut(len);
        sinks[i] = ChunkSink::Direct { dst, out: mine };
        rest = tail;
        base = dst + len;
    }
    sinks
}

/// Run the reverse filter pipeline (if any) over one chunk's raw bytes,
/// straight into `out`, returning the length of the image it produced.
///
/// The counterpart of [`decompress_chunk`] for a chunk whose image is already
/// the output's own bytes. A length below `out.len()` means the stored chunk
/// decoded short; a length above it means the surplus was discarded, which is
/// what copying `out.len()` bytes out of a materialized image does too.
fn decompress_chunk_into(
    pipeline: Option<&FilterPipeline>,
    raw: &[u8],
    mask: u32,
    out: &mut [u8],
) -> IoResult<usize> {
    match pipeline {
        Some(pl) => Ok(filter::reverse_filters_masked_into(pl, raw, mask, out)?),
        None => {
            let n = raw.len().min(out.len());
            out[..n].copy_from_slice(&raw[..n]);
            Ok(raw.len())
        }
    }
}

/// Run the reverse filter pipeline (if any) over one chunk's raw bytes into a
/// fresh image, for a chunk whose bytes have no contiguous home in the output.
///
/// `image_bytes` is what the layout says the chunk decodes to
/// ([`ChunkOutputGeometry::image_bytes`]), so the image is allocated once at
/// its real size instead of being grown to it — a filter that has to discover
/// the size re-enters its decoder once per doubling. Bytes past what the
/// pipeline produced are cut off, so a chunk that decoded short is as short
/// here as the growing spelling left it and [`copy_chunk_runs`] passes the
/// same verdict on the runs reaching past it. Bytes past `image_bytes` are cut
/// off too: no run of a chunk reaches past its own image.
fn decompress_chunk(
    pipeline: Option<&FilterPipeline>,
    raw: Vec<u8>,
    mask: u32,
    image_bytes: Option<u64>,
) -> IoResult<Vec<u8>> {
    let Some(pl) = pipeline else { return Ok(raw) };
    let Some(image_bytes) = image_bytes.and_then(|b| usize::try_from(b).ok()) else {
        // Geometry a corrupt file can carry says nothing about the size; the
        // pipeline discovers it.
        return Ok(filter::reverse_filters_masked(pl, &raw, mask)?);
    };
    let mut image = vec![0u8; image_bytes];
    let produced = filter::reverse_filters_masked_into(pl, &raw, mask, &mut image)?;
    image.truncate(produced.min(image_bytes));
    Ok(image)
}

/// Planned chunk bytes a batch must carry before the rayon pool earns its
/// entry cost.
///
/// `ThreadPool::install` injects a job and blocks on a latch; with the workers
/// parked that costs tens of microseconds, which is more than a small
/// selection's entire read. One 64 KiB slice of a chunked dataset plans one or
/// two chunks, so it decodes on the thread that planned it and never pays the
/// dispatch.
#[cfg(feature = "parallel")]
const PARALLEL_MIN_JOB_BYTES: u64 = 256 * 1024;

/// Whether a batch is worth handing to the pool: more than one chunk to read,
/// and enough bytes behind them to repay [`PARALLEL_MIN_JOB_BYTES`]. Skipped
/// chunks (`None`) count for nothing — a slice whose chunks were all placed by
/// [`read_chunk_runs_into`] leaves no work at all.
#[cfg(feature = "parallel")]
fn worth_parallel(jobs: &[Option<ChunkReadJob>]) -> bool {
    let mut n = 0usize;
    let mut bytes = 0u64;
    for j in jobs.iter().flatten() {
        n += 1;
        bytes = bytes.saturating_add(j.len as u64);
        if n > 1 && bytes >= PARALLEL_MIN_JOB_BYTES {
            return true;
        }
    }
    false
}

/// Read and decompress a batch of chunk jobs, preserving job order.
///
/// `jobs[i] == None` yields `Ok(None)` — a chunk skipped as out-of-selection
/// or unallocated. Otherwise the chunk's raw bytes are read and, when
/// `pipeline` is `Some`, run through the reverse filter pipeline with the
/// job's mask. This is the single owner of the read-then-decompress step for
/// every chunk index type; each read path only builds the jobs and scatters
/// the results.
///
/// On Unix and Windows, positioned reads at distinct offsets on a shared
/// `&File` each carry their own explicit offset and never consult a shared file
/// cursor (on Windows the cursor may move as a side effect, but nothing reads
/// it), so read + decompress run fused in one parallel pass — overlapping chunk
/// I/O across cores, which the
/// C library's default (non-MPI) path does not do. On targets with neither
/// positioned API the seek-based fallback shares the file cursor, so reads
/// stay serial there while decompression still parallelizes.
fn read_and_decompress_chunks(
    handle: &FileHandle,
    pipeline: Option<&FilterPipeline>,
    jobs: Vec<Option<ChunkReadJob>>,
    sinks: Vec<ChunkSink<'_>>,
    image_bytes: Option<u64>,
) -> IoResult<Vec<ChunkDecoded>> {
    // Decompress one chunk's raw bytes into whatever its sink says.
    let deliver = |raw: Vec<u8>, mask: u32, sink: ChunkSink<'_>| -> IoResult<ChunkDecoded> {
        match sink {
            ChunkSink::Direct { dst, out } => {
                let len = out.len();
                let bytes = decompress_chunk_into(pipeline, &raw, mask, out)?;
                Ok(ChunkDecoded::InPlace { dst, len, bytes })
            }
            ChunkSink::Staged => Ok(ChunkDecoded::Image(decompress_chunk(
                pipeline,
                raw,
                mask,
                image_bytes,
            )?)),
        }
    };
    #[cfg(all(feature = "parallel", any(unix, windows)))]
    {
        use rayon::prelude::*;
        // Fused read + decompress for one job.
        let decode =
            |(job, sink): (Option<ChunkReadJob>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
                match job {
                    Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
                    None => Ok(ChunkDecoded::Absent),
                }
            };
        // Run on rust-hdf5's private half-cores pool, not rayon's global pool;
        // fall back to serial if the pool could not be built, and for a batch
        // too small to repay entering it.
        let pool = crate::parallel::io_pool().filter(|_| worth_parallel(&jobs));
        let work: Vec<_> = jobs.into_iter().zip(sinks).collect();
        match pool {
            Some(pool) => pool.install(|| {
                work.into_par_iter()
                    .map(&decode)
                    .collect::<IoResult<Vec<_>>>()
            }),
            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
        }
    }
    #[cfg(all(feature = "parallel", not(any(unix, windows))))]
    {
        use rayon::prelude::*;
        // No positioned read API here: concurrent reads would race the shared
        // file cursor, so read serially, then parallelize decompression on
        // rust-hdf5's private half-cores pool (not rayon's global pool).
        let parallel = worth_parallel(&jobs);
        let raws: Vec<Option<(Vec<u8>, u32)>> = jobs
            .into_iter()
            .map(|job| match job {
                Some(j) => Ok(Some((read_chunk_raw(handle, &j)?, j.mask))),
                None => Ok(None),
            })
            .collect::<IoResult<Vec<_>>>()?;
        let decode =
            |(r, sink): (Option<(Vec<u8>, u32)>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
                match r {
                    Some((raw, mask)) => deliver(raw, mask, sink),
                    None => Ok(ChunkDecoded::Absent),
                }
            };
        let work: Vec<_> = raws.into_iter().zip(sinks).collect();
        // Fall back to serial if the private pool could not be built, and for
        // a batch too small to repay entering it.
        match crate::parallel::io_pool().filter(|_| parallel) {
            Some(pool) => pool.install(|| {
                work.into_par_iter()
                    .map(&decode)
                    .collect::<IoResult<Vec<_>>>()
            }),
            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
        }
    }
    #[cfg(not(feature = "parallel"))]
    {
        jobs.into_iter()
            .zip(sinks)
            .map(|(job, sink)| match job {
                Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
                None => Ok(ChunkDecoded::Absent),
            })
            .collect()
    }
}

impl Hdf5Reader {
    /// Open an existing HDF5 file in SWMR read mode using the env-var-derived
    /// locking policy.
    ///
    /// Currently identical to `open()`, but indicates intent to use
    /// `refresh()` for re-reading metadata written by a concurrent SWMR writer.
    pub fn open_swmr(path: &Path) -> IoResult<Self> {
        Self::open(path)
    }

    /// Open an existing HDF5 file in SWMR read mode with an explicit locking
    /// policy.
    pub fn open_swmr_with_locking(
        path: &Path,
        locking: crate::io::locking::FileLocking,
    ) -> IoResult<Self> {
        Self::open_with_locking(path, locking)
    }

    /// Open an existing HDF5 file for reading using the env-var-derived
    /// locking policy.
    ///
    /// Auto-detects the superblock version and uses the appropriate code path:
    /// - v0/v1: legacy format with symbol tables and B-tree v1
    /// - v2/v3: modern format with link messages
    pub fn open(path: &Path) -> IoResult<Self> {
        Self::open_with_locking(
            path,
            crate::io::locking::FileLocking::from_env_or(Default::default()),
        )
    }

    /// Open an existing HDF5 file for reading with an explicit locking policy.
    pub fn open_with_locking(
        path: &Path,
        locking: crate::io::locking::FileLocking,
    ) -> IoResult<Self> {
        let mut handle = FileHandle::open_read_with_locking(path, locking)?;

        // The superblock is not necessarily at the start of the file: a
        // userblock precedes it, and `H5FD_locate_signature` finds it by
        // probing offset 0 and then every power of two from 512 up. The offset
        // it is found at is where HDF5 addresses are measured from, so it
        // becomes the handle's base address and every later offset — including
        // the superblock read just below — is relative to it.
        let super_addr = handle
            .locate_signature()?
            .ok_or(crate::format::FormatError::InvalidSignature)?;
        handle.set_base(super_addr);

        // Read enough bytes to detect the superblock version and parse it.
        let sb_buf = handle.read_at_most(0, 1024)?;
        let version = detect_superblock_version(&sb_buf)?;

        let origin = Origin {
            path: path.to_path_buf(),
            locking,
        };
        let mut reader = match version {
            0 | 1 => Self::open_v0v1(handle, &sb_buf, origin)?,
            2 | 3 => Self::open_v2v3(handle, &sb_buf, origin)?,
            v => {
                return Err(crate::io::IoError::Format(
                    crate::format::FormatError::InvalidVersion(v),
                ))
            }
        };
        // Resolved from the path this file was opened with (not the
        // process's current directory at read time) — see `source_dir`.
        let canonical = std::fs::canonicalize(path)?;
        reader.source_dir = canonical
            .parent()
            .map(Path::to_path_buf)
            .unwrap_or_default();
        // Only now, with `source_dir` set, can a source name be resolved —
        // and a virtual dataset's extent is not final until they are.
        VirtualResolveDepth::enter(|| reader.resolve_virtual_extents())?;
        Ok(reader)
    }

    /// Open a file with v2/v3 superblock (existing code path).
    fn open_v2v3(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
        let sb = SuperblockV2V3::decode(sb_buf)?;

        let ctx = FormatContext {
            sizeof_addr: sb.sizeof_offsets,
            sizeof_size: sb.sizeof_lengths,
        };

        // A v2/v3 superblock has no room for the B-tree K values, so they are
        // the library defaults unless the extension carries the K message.
        let (meta, ext) = Self::read_extension_and_meta(
            &mut handle,
            ctx,
            BTreeV1Config::default(),
            sb.superblock_extension_address,
        )?;

        // Read root group object header, following continuation blocks.
        let root_header =
            Self::read_object_header_full(&mut handle, &meta, sb.root_group_object_header_address)?;

        // Walk the root group to discover datasets, group attributes, and
        // every group path that exists, from whichever storage each group's
        // own header declares.
        let catalog = Self::build_catalog(
            &mut handle,
            &meta,
            Some(&root_header),
            sb.root_group_object_header_address,
            None,
        )?;

        // Collect root group attributes
        let root_attributes = collect_object_attributes(&mut handle, &ctx, &root_header);
        // A v2/v3 root group is always addressed directly by the superblock,
        // never through a symbol-table scratch-pad.
        let root_link_storage = describe_link_storage(Some(&root_header), &ctx, None);

        Ok(Self {
            handle,
            meta,
            ext,
            _eof: sb.end_of_file_address,
            superblock_version: sb.version,
            object_paths: catalog.object_paths(sb.root_group_object_header_address),
            datasets: DatasetTable::new(catalog.datasets),
            unreadable: catalog.unreadable,
            root_attributes,
            root_link_storage,
            group_attributes: catalog.group_attributes,
            group_link_storage: catalog.group_link_storage,
            group_paths: catalog.group_paths,
            group_aliases: catalog.group_aliases,
            links: catalog.links,
            datatypes: catalog.datatypes,
            path: origin.path,
            locking: origin.locking,
            elink_prefix: None,
            external: Default::default(),
            external_resolved: Default::default(),
            dataset_access: Default::default(),
            vds_resolved: Default::default(),
            // Overwritten by `open_with_locking` once this returns.
            source_dir: PathBuf::new(),
        })
    }

    /// Open a file with v0/v1 superblock (legacy format).
    fn open_v0v1(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
        let sb = SuperblockV0V1::decode(sb_buf)?;

        let ctx = FormatContext {
            sizeof_addr: sb.sizeof_offsets,
            sizeof_size: sb.sizeof_lengths,
        };

        // A v0/v1 superblock carries the K values itself; a v0 superblock has
        // no chunk-tree field, so that one keeps the library default. The
        // extension's K message, when present, overrides all three.
        let sb_btree = BTreeV1Config {
            sym_leaf_k: sb.sym_leaf_k,
            snode_internal_k: sb.btree_internal_k,
            chunk_internal_k: sb
                .indexed_storage_k
                .unwrap_or(BTreeV1Config::default().chunk_internal_k),
        };
        let (meta, ext) = Self::read_extension_and_meta(
            &mut handle,
            ctx,
            sb_btree,
            sb.superblock_extension_address,
        )?;

        let ste = &sb.root_symbol_table_entry;
        let root_obj_addr = ste.obj_header_addr;
        let ste_stab = ste.cached_symbol_table();

        // Read the root group's object header (following continuations).
        let root_hdr = Self::read_object_header_full(&mut handle, &meta, root_obj_addr).ok();

        // Collect the root group's own attributes.
        let root_attributes = match root_hdr {
            Some(ref h) => collect_object_attributes(&mut handle, &ctx, h),
            None => ObjectAttributes::default(),
        };
        // The root group's own link storage and link creation-order, from
        // the same header-first/scratch-pad-fallback rule the walk below
        // uses to choose how to enumerate it.
        let root_link_storage = describe_link_storage(root_hdr.as_ref(), &ctx, ste_stab);

        // A v0/v1-superblock file whose root group has migrated to link
        // storage (more than ~8 objects, or one link the old format cannot
        // express) carries `Link` / `Link Info` messages in its object
        // header, and the superblock symbol-table scratch-pad is then stale.
        // The walk picks the storage from the header for that reason, taking
        // the scratch-pad only as the symbol-table addresses — and only for a
        // symbol-table root group (`H5G_CACHED_STAB`), which is the one cache
        // type those two addresses mean anything for.
        let catalog = Self::build_catalog(
            &mut handle,
            &meta,
            root_hdr.as_ref(),
            root_obj_addr,
            ste_stab,
        )?;

        Ok(Self {
            handle,
            meta,
            ext,
            _eof: sb.end_of_file_address,
            superblock_version: sb.version,
            object_paths: catalog.object_paths(root_obj_addr),
            datasets: DatasetTable::new(catalog.datasets),
            unreadable: catalog.unreadable,
            root_attributes,
            root_link_storage,
            group_attributes: catalog.group_attributes,
            group_link_storage: catalog.group_link_storage,
            group_paths: catalog.group_paths,
            group_aliases: catalog.group_aliases,
            links: catalog.links,
            datatypes: catalog.datatypes,
            path: origin.path,
            locking: origin.locking,
            elink_prefix: None,
            external: Default::default(),
            external_resolved: Default::default(),
            dataset_access: Default::default(),
            vds_resolved: Default::default(),
            // Overwritten by `open_with_locking` once this returns.
            source_dir: PathBuf::new(),
        })
    }

    /// Read the superblock extension object header at `ext_addr` (if any) and
    /// fold what it says into the file-level decode parameters.
    ///
    /// `sb_btree` is what the superblock alone implies; the extension's
    /// v1-B-tree-"K" message replaces all three ranks when present, exactly as
    /// `H5F__super_read` does after `H5O_msg_read(&ext_loc, H5O_BTREEK_ID)`.
    pub(crate) fn read_extension_and_meta(
        handle: &mut FileHandle,
        ctx: FormatContext,
        sb_btree: BTreeV1Config,
        ext_addr: u64,
    ) -> IoResult<(FileMeta, SuperblockExtension)> {
        let mut meta = FileMeta {
            ctx,
            btree: sb_btree,
            sohm: None,
        };
        let ext = Self::superblock_extension_at(handle, ctx, sb_btree, ext_addr)?;
        if let Some(k) = ext.btree_k {
            meta.btree = BTreeV1Config {
                sym_leaf_k: k.sym_leaf_k,
                snode_internal_k: k.snode_internal_k,
                chunk_internal_k: k.chunk_internal_k,
            };
        }
        // A zero rank would make every v1 B-tree node zero-sized and every
        // symbol-table node hold no entries; libhdf5 rejects it at creation
        // (`H5Pset_sym_k`, `H5Pset_istore_k`), so a file carrying one is
        // corrupt rather than merely unusual.
        let b = &meta.btree;
        if b.sym_leaf_k == 0 || b.snode_internal_k == 0 || b.chunk_internal_k == 0 {
            return Err(crate::io::IoError::Format(
                crate::format::FormatError::InvalidData(format!(
                    "v1 B-tree K values must be non-zero (sym_leaf={}, snode={}, chunk={})",
                    b.sym_leaf_k, b.snode_internal_k, b.chunk_internal_k
                )),
            ));
        }
        // `H5F__super_read` calls `H5SM_get_info` here, so the shared-message
        // table is in place before the root group — the first object header
        // that can hold a shared message — is opened.
        if let Some(smt) = &ext.shared_message_table {
            meta.sohm = Some(Self::read_sohm_table(handle, &meta.ctx, smt)?);
        }
        Ok((meta, ext))
    }

    /// The superblock extension's messages, for the address the superblock
    /// names. Yields the default (every field `None`) when there is no
    /// extension.
    ///
    /// The extension header is read with the pre-extension parameters: its own
    /// messages are never shared and never in a v1 B-tree, so nothing it
    /// contains is needed to decode it. This is also how the writer's append
    /// path learns what the file declares before it rewrites anything.
    pub(crate) fn superblock_extension_at(
        handle: &mut FileHandle,
        ctx: FormatContext,
        btree: BTreeV1Config,
        ext_addr: u64,
    ) -> IoResult<SuperblockExtension> {
        if ext_addr == UNDEF_ADDR || ext_addr == 0 {
            return Ok(SuperblockExtension::default());
        }
        let meta = FileMeta {
            ctx,
            btree,
            sohm: None,
        };
        Self::read_superblock_extension(handle, &meta, ext_addr)
    }

    /// Decode the messages of the superblock extension object header.
    ///
    /// Upstream reads each of these with `H5O_msg_exists` + `H5O_msg_read` and
    /// fails the open when one is present but undecodable; a message this
    /// crate does not model is skipped, as an unknown non-critical message is
    /// elsewhere.
    fn read_superblock_extension(
        handle: &mut FileHandle,
        meta: &FileMeta,
        addr: u64,
    ) -> IoResult<SuperblockExtension> {
        let header = Self::read_object_header_full(handle, meta, addr)?;
        let ctx = &meta.ctx;
        let mut ext = SuperblockExtension::default();
        for msg in &header.messages {
            match msg.msg_type {
                MSG_SHARED_MESSAGE_TABLE => {
                    ext.shared_message_table =
                        Some(SharedMessageTableMessage::decode(&msg.data, ctx)?);
                }
                MSG_BTREE_K => ext.btree_k = Some(BtreeKMessage::decode(&msg.data)?),
                MSG_DRIVER_INFO => ext.driver_info = Some(DriverInfoMessage::decode(&msg.data)?),
                MSG_FILE_SPACE_INFO => {
                    ext.file_space_info = Some(FileSpaceInfoMessage::decode(&msg.data, ctx)?);
                }
                _ => {}
            }
        }
        Ok(ext)
    }

    /// Read the SOHM master table named by the extension's shared-message
    /// table message.
    ///
    /// The table's length is not stored with it: the index count comes from
    /// the message, exactly as `H5SM__cache_table_get_final_load_size` takes it
    /// from `H5F_SOHM_NINDEXES`.
    fn read_sohm_table(
        handle: &mut FileHandle,
        ctx: &FormatContext,
        smt: &SharedMessageTableMessage,
    ) -> IoResult<SohmMasterTable> {
        if smt.table_address == UNDEF_ADDR || smt.nindexes == 0 {
            return Ok(SohmMasterTable::default());
        }
        let size = SohmMasterTable::encoded_size(ctx, smt.nindexes);
        let buf = handle.read_at(smt.table_address, size)?;
        Ok(SohmMasterTable::decode(&buf, ctx, smt.nindexes)?)
    }

    /// Extract the symbol-table message (btree_addr, heap_addr) from an
    /// already-decoded object header.
    fn stab_from_header(header: &ObjectHeader, ctx: &FormatContext) -> (u64, u64) {
        for msg in &header.messages {
            if msg.msg_type == MSG_SYMBOL_TABLE {
                let sa = ctx.sizeof_addr as usize;
                if msg.data.len() >= 2 * sa {
                    return (read_uint(&msg.data, sa), read_uint(&msg.data[sa..], sa));
                }
            }
        }
        (UNDEF_ADDR, UNDEF_ADDR)
    }

    /// Build the file catalog for a whole file, starting at its root group.
    ///
    /// `root_header` is `None` only when the root object header did not
    /// decode; `root_stab` then carries the superblock symbol-table entry's
    /// cached B-tree and local heap, which is enough to list a legacy file.
    fn build_catalog(
        handle: &mut FileHandle,
        meta: &FileMeta,
        root_header: Option<&ObjectHeader>,
        root_addr: u64,
        root_stab: Option<(u64, u64)>,
    ) -> IoResult<Catalog> {
        let mut walk = CatalogWalk::new(handle, meta, root_addr);
        walk.group(root_header, "", 0, root_stab)?;
        Ok(walk.finish())
    }

    /// Read every link stored in a group's dense (fractal-heap) link storage.
    ///
    /// The `Link Info` message gives the fractal-heap address; each managed
    /// object in the heap is an encoded `Link` message. Returns the decoded
    /// links (hard and soft).
    pub(crate) fn read_dense_links(
        handle: &mut FileHandle,
        ctx: &FormatContext,
        fractal_heap_addr: u64,
    ) -> IoResult<Vec<LinkMessage>> {
        // Read the fractal heap header. Its on-disk size depends only on the
        // address/length widths, so a generous prefix read covers it.
        let hdr_buf = handle.read_at_most(fractal_heap_addr, 512)?;
        let fh_header = FractalHeapHeader::decode(&hdr_buf, ctx)?;

        // Walk the heap's managed blocks; each block hands back a payload
        // region holding one or more packed encoded `Link` messages.
        let mut br = HandleBlockReader { handle };
        let payloads = fractal_heap::collect_managed_objects(&fh_header, ctx, &mut br)?;

        let mut links = Vec::new();
        for payload in payloads {
            // Decode packed `Link` messages sequentially. Each decode reports
            // its consumed length; stop at the first byte that is not a valid
            // link (trailing free space or an unrelated managed object).
            let mut pos = 0;
            while pos < payload.len() {
                // A v1 link message starts with version byte 1.
                if payload[pos] != 1 {
                    break;
                }
                match LinkMessage::decode(&payload[pos..], ctx) {
                    Ok((link, consumed)) if consumed > 0 => {
                        links.push(link);
                        pos += consumed;
                    }
                    _ => break,
                }
            }
        }

        // That scan stops at the first byte that does not begin a link, which
        // is how trailing free space in a direct block ends it — and would
        // equally swallow a link the scan could not read. The heap header
        // counts its managed objects, so a short scan is detectable, and a
        // group listing that is short is the loss this guards against.
        if links.len() < fh_header.man_nobjs as usize {
            return Err(crate::io::IoError::InvalidState(format!(
                "dense link storage at address {fractal_heap_addr:#x} holds {} managed \
                 objects but only {} decoded as links",
                fh_header.man_nobjs,
                links.len()
            )));
        }

        Ok(links)
    }

    /// Recursively walk a B-tree v1 to collect leaf-level SNOD addresses.
    fn collect_snod_addresses(
        handle: &mut FileHandle,
        meta: &FileMeta,
        tree_addr: u64,
        depth: usize,
        visited: &mut std::collections::HashSet<u64>,
    ) -> IoResult<Vec<u64>> {
        let sizeof_addr = meta.ctx.sizeof_addr as usize;
        let sizeof_size = meta.ctx.sizeof_size as usize;
        // A well-formed v1 B-tree's level strictly decreases with depth;
        // bound the descent so a corrupt/cyclic tree cannot recurse forever.
        // The `visited` set additionally stops a corrupt tree whose child
        // points back at an ancestor node from fanning out exponentially.
        if depth > 256 || !visited.insert(tree_addr) {
            return Ok(Vec::new());
        }
        // A v1 B-tree node is a fixed-size record whose length follows from
        // the file's K values; reading exactly that much also bounds what a
        // corrupt address can pull in.
        let node_size = meta.btree.snode_btree_node_size(sizeof_addr, sizeof_size);
        let buf = handle.read_at_most(tree_addr, node_size)?;
        let node = BTreeV1Node::decode(
            &buf,
            sizeof_addr,
            sizeof_size,
            meta.btree.snode_max_entries(),
        )?;

        if node.level == 0 {
            // Leaf level: children are SNOD addresses
            Ok(node.children.clone())
        } else {
            // Internal level: children are sub-TREE addresses
            let mut addrs = Vec::new();
            for &child_addr in &node.children {
                let child_addrs =
                    Self::collect_snod_addresses(handle, meta, child_addr, depth + 1, visited)?;
                addrs.extend(child_addrs);
            }
            Ok(addrs)
        }
    }

    /// Read the object header at `addr` with every continuation block
    /// flattened in and every stored-shared message resolved to its literal
    /// body. One owner for both halves of the crate — see
    /// [`crate::io::object_header_io`].
    fn read_object_header_full(
        handle: &mut FileHandle,
        meta: &FileMeta,
        addr: u64,
    ) -> IoResult<ObjectHeader> {
        crate::io::object_header_io::read_object_header_full(handle, meta, addr)
    }

    /// Read a committed datatype object's type and attributes from its header.
    ///
    /// A committed datatype's own message holds the type itself, but the
    /// format does not forbid it being a reference in turn, so it goes
    /// through the same resolver every other datatype message does.
    fn committed_datatype(
        handle: &mut FileHandle,
        header: &ObjectHeader,
        meta: &FileMeta,
    ) -> CommittedDatatypeInfo {
        let datatype = header
            .messages
            .iter()
            .find(|m| m.msg_type == MSG_DATATYPE)
            .cloned()
            .ok_or_else(|| "it holds no datatype message".to_string())
            .and_then(|m| {
                crate::io::object_header_io::read_datatype_message(handle, meta, &m).map_err(|e| {
                    match e {
                        crate::io::IoError::Unsupported(why) => why,
                        other => format!("its datatype message does not decode: {other}"),
                    }
                })
            });
        let attributes = header
            .messages
            .iter()
            .filter(|m| m.msg_type == MSG_ATTRIBUTE && m.flags & MSG_FLAG_SHARED == 0)
            .filter_map(|m| {
                AttributeMessage::decode(&m.data, &meta.ctx)
                    .ok()
                    .map(|(a, _)| a)
            })
            .collect();
        CommittedDatatypeInfo {
            datatype,
            attributes,
        }
    }

    /// Classify one object from its (already read) header, and decode the
    /// dataset metadata while doing so.
    ///
    /// The class comes from which messages are present, never from whether
    /// they decode: an object holding a datatype, a dataspace and a data
    /// layout is a dataset even when this crate cannot decode one of them,
    /// and it says so as [`ObjectKind::UnreadableDataset`] rather than
    /// vanishing.
    ///
    /// Only the messages the payload depends on can make a dataset
    /// unreadable. A failed *attribute* decode leaves the dataset itself
    /// readable, so it does not.
    fn classify_object(
        handle: &mut FileHandle,
        header: &ObjectHeader,
        meta: &FileMeta,
        name: &str,
        addr: u64,
    ) -> ObjectKind {
        let ctx = &meta.ctx;
        let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
        let is_group = present(MSG_LINK)
            || present(MSG_LINK_INFO)
            || present(MSG_SYMBOL_TABLE)
            || present(MSG_GROUP_INFO);
        if is_group {
            return ObjectKind::Group;
        }
        if header_is_committed_datatype(header) {
            return ObjectKind::CommittedDatatype(Box::new(Self::committed_datatype(
                handle, header, meta,
            )));
        }
        let is_dataset =
            present(MSG_DATATYPE) && present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT);
        if !is_dataset {
            return ObjectKind::Group;
        }

        let mut datatype = None;
        let mut dataspace = None;
        let mut layout = None;
        let mut filter_pipeline = None;
        let mut fill_value = None;
        // No message at all is the library default: a fresh dataset
        // creation property list starts fill_defined = 1
        // (`FillValueMessage::default`), so a dataset that never got one
        // written reads back exactly as if it had.
        let mut fill_defined: u8 = 1;
        let mut fill_write_time: u8 = FILL_TIME_IFSET;
        let mut alloc_time: u8 = ALLOC_TIME_LATE;
        // The first message that did not decode, kept verbatim: it is the
        // answer a caller gets when it asks for this dataset.
        let mut blocked: Option<String> = None;
        let mut block = |why: String| {
            if blocked.is_none() {
                blocked = Some(why);
            }
        };
        let mut external_file_list = None;

        for msg in &header.messages {
            // A shared message holds a reference to where its body lives, not
            // the body. Decoding one as a body does not fail loudly — it
            // reads the reference's version byte as the body's — so anything
            // this crate does not follow is named here instead. A datatype
            // reference is followed; an attribute reference is skipped, since
            // an attribute never blocks the dataset it hangs on.
            let shared = msg.flags & MSG_FLAG_SHARED != 0;
            if shared && !matches!(msg.msg_type, MSG_DATATYPE | MSG_ATTRIBUTE) {
                block(format!(
                    "its message of type {:#04x} is a shared-message reference, which this \
                     crate follows only for datatypes",
                    msg.msg_type
                ));
                continue;
            }
            match msg.msg_type {
                // The resolver already says whether the type failed to decode
                // or sits somewhere this crate does not follow, so its wording
                // is the reason rather than something to wrap.
                MSG_DATATYPE => {
                    match crate::io::object_header_io::read_datatype_message(handle, meta, msg) {
                        Ok(dt) => datatype = Some(dt),
                        Err(crate::io::IoError::Unsupported(why)) => block(why),
                        Err(e) => block(format!("its datatype message does not decode: {e}")),
                    }
                }
                MSG_DATASPACE => match DataspaceMessage::decode(&msg.data, ctx) {
                    Ok((ds, _)) => dataspace = Some(ds),
                    Err(e) => block(format!("its dataspace message does not decode: {e}")),
                },
                MSG_DATA_LAYOUT => match DataLayoutMessage::decode(&msg.data, ctx) {
                    Ok((dl, _)) => layout = Some(dl),
                    Err(e) => block(format!("its data layout message does not decode: {e}")),
                },
                // A filter pipeline that does not decode would leave the raw
                // chunk bytes to be handed back as if they were never
                // filtered, and an undecodable fill value would leave
                // unwritten regions reading as zeros. Both change the data a
                // read returns, so both block the dataset.
                MSG_FILTER_PIPELINE => match FilterPipeline::decode(&msg.data) {
                    Ok((fp, _)) => {
                        if !fp.filters.is_empty() {
                            filter_pipeline = Some(fp);
                        }
                    }
                    Err(e) => block(format!("its filter pipeline message does not decode: {e}")),
                },
                MSG_FILL_VALUE => match FillValueMessage::decode(&msg.data) {
                    Ok((fv, _)) => {
                        fill_defined = fv.fill_defined;
                        fill_write_time = fv.fill_write_time;
                        alloc_time = fv.alloc_time;
                        if fv.fill_defined == 2 {
                            fill_value = fv.fill_value;
                        }
                    }
                    Err(e) => block(format!("its fill value message does not decode: {e}")),
                },
                MSG_EXTERNAL_FILE_LIST => {
                    // Unlike a layout message, this one *is* the storage: a
                    // dataset with an external file list has no data address
                    // of its own (H5Dlayout.c routes storage through this
                    // message instead), so a list that does not decode must
                    // block the dataset rather than read back as zero bytes.
                    match ExternalFileListMessage::decode(&msg.data, ctx) {
                        Ok((efl, _)) => external_file_list = Some(efl),
                        Err(e) => block(format!(
                            "its external file list message does not decode: {e}"
                        )),
                    }
                }
                _ => {}
            }
        }

        // libhdf5 checks a layout against its sibling dataspace and datatype
        // as the dataset opens (`H5O__layout_decode` for the chunk rank,
        // `H5D__compact_init` for the compact size); the three decode side
        // by side here, so this is where those checks land.
        if let (Some(ds), Some(dt), Some(dl)) = (&dataspace, &datatype, &layout) {
            if let Err(e) = dl.check_against_dataset(ds, dt, ctx) {
                block(format!(
                    "its layout doesn't fit its dataspace and datatype: {e}"
                ));
            }
        }

        if let Some(why) = blocked {
            return ObjectKind::UnreadableDataset(why);
        }
        // The storage a dataset names outside its layout message. Both are
        // resolved before the dataset is registered and both block it when
        // they do not resolve, for the same reason the decode above does: a
        // `Virtual` or external-file layout carries no address of its own, so
        // a dropped mapping reads back as fill with no error at all.
        let external_files = match external_file_list {
            Some(efl) => match Self::resolve_external_file_slots(handle, ctx, &efl) {
                Ok(slots) => slots,
                Err(e) => {
                    return ObjectKind::UnreadableDataset(format!(
                        "its external file list does not resolve: {e}"
                    ))
                }
            },
            None => Vec::new(),
        };
        let virtual_mappings = match &layout {
            Some(DataLayoutMessage::Virtual {
                heap_address,
                heap_index,
                ..
            }) if *heap_index != 0 => {
                match Self::resolve_virtual_mappings(handle, ctx, *heap_address, *heap_index, name)
                {
                    Ok(list) => Some(list),
                    Err(e) => {
                        return ObjectKind::UnreadableDataset(format!(
                            "its virtual dataset mapping list does not resolve: {e}"
                        ))
                    }
                }
            }
            _ => None,
        };
        // The attribute set is collected whole, or the object says it could
        // not be: a short list here would be a dataset reporting attributes
        // the file does not agree it has.
        let attributes = collect_object_attributes(handle, ctx, header);
        match (datatype, dataspace, layout) {
            (Some(dt), Some(ds), Some(dl)) => ObjectKind::Dataset(Box::new(DatasetReadInfo {
                name: name.to_string(),
                object_header_address: addr,
                datatype: dt,
                dataspace: ds,
                layout: dl,
                filter_pipeline,
                attributes,
                fill_value,
                fill_defined,
                fill_write_time,
                alloc_time,
                external_files,
                virtual_mappings,
                // Both filled in by `resolve_virtual_extents` once the
                // reader has the directory source names resolve against; a
                // catalog on its own cannot open another file.
                virtual_resolution: None,
                virtual_stored_dims: None,
            })),
            // The three messages are present and none of them reported an
            // error, so this is unreachable; report it as unreadable rather
            // than dropping the name on an invariant this function owns.
            _ => ObjectKind::UnreadableDataset(
                "its datatype, dataspace and data layout messages decoded but did not all \
                 produce a value"
                    .into(),
            ),
        }
    }

    /// Every dataset in the file, by path (no leading `/`).
    ///
    /// A dataset this crate cannot read is still a dataset the file
    /// contains, so it is listed here alongside the readable ones and
    /// answers [`Self::unreadable_reason`]; opening it reports that reason.
    /// Resolve a virtual dataset's mapping list from the global heap object
    /// its layout message points at (`H5D__virtual_load_layout`,
    /// H5Dvirtual.c). Like the external-file-list decode above, a failure
    /// here must not fall back to silently treating the dataset as having
    /// no data: a `Virtual` layout carries no data address of its own, so a
    /// dropped mapping list would read back as all-fill with no error.
    fn resolve_virtual_mappings(
        handle: &mut FileHandle,
        ctx: &FormatContext,
        heap_address: u64,
        heap_index: u32,
        name: &str,
    ) -> IoResult<VirtualMappingList> {
        let coll = read_heap_collection_from(handle, ctx, heap_address)?;
        let idx = u16::try_from(heap_index).map_err(|_| {
            crate::io::IoError::InvalidState(format!(
                "dataset {name:?} virtual mapping heap index {heap_index} does not fit \
                 the 16-bit on-disk field"
            ))
        })?;
        let obj = coll.get_object(idx).ok_or_else(|| {
            crate::io::IoError::InvalidState(format!(
                "dataset {name:?} virtual mapping list object {idx} not found in the \
                 global heap collection at address {heap_address:#x}"
            ))
        })?;
        VirtualMappingList::decode(obj, ctx).map_err(|e| {
            crate::io::IoError::InvalidState(format!(
                "dataset {name:?} has a malformed virtual dataset mapping list: {e}"
            ))
        })
    }

    /// Resolve every external-file slot's name through the local heap the
    /// EFL message points at (H5Oefl.c decodes only the byte offset; the
    /// string itself lives in a separate on-disk local heap, exactly like a
    /// v0/v1 group's link names — see [`local_heap_get_string`]).
    pub(crate) fn resolve_external_file_slots(
        handle: &mut FileHandle,
        ctx: &FormatContext,
        efl: &ExternalFileListMessage,
    ) -> IoResult<Vec<ExternalFileSegment>> {
        let sa = ctx.sizeof_addr as usize;
        let ss = ctx.sizeof_size as usize;
        let heap_hdr_buf = handle.read_at_most(efl.heap_addr, 64)?;
        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
        let heap_data = handle.read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;

        efl.slots
            .iter()
            .map(|slot| {
                let name = local_heap_get_string(&heap_data, slot.name_offset)?;
                Ok(ExternalFileSegment {
                    name,
                    offset: slot.offset,
                    size: slot.size,
                })
            })
            .collect()
    }

    /// Return the names of all datasets in the root group.
    pub fn dataset_names(&self) -> Vec<&str> {
        let mut names: Vec<&str> = self.datasets.iter().map(|d| d.name.as_str()).collect();
        names.extend(self.unreadable.keys().map(String::as_str));
        names
    }

    /// Why the dataset at `path` (no leading `/`) cannot be read, or `None`
    /// when it can be — or does not exist.
    pub fn unreadable_reason(&mut self, path: &str) -> Option<&str> {
        if self.external_edge(path).is_some() {
            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
            let local = owner.canonical_path(&local);
            return owner.unreadable.get(&local).map(String::as_str);
        }
        let path = self.canonical_path(path);
        self.unreadable.get(&path).map(String::as_str)
    }

    /// Every link record in the file, keyed by full path (no leading `/`).
    pub fn links(&self) -> &std::collections::BTreeMap<String, LinkClass> {
        &self.links
    }

    /// The paths of every committed (named) datatype object in this file.
    ///
    /// A committed datatype is in neither [`dataset_names`](Self::dataset_names)
    /// nor the group listing — it is a third kind of object, and this is its
    /// listing.
    pub fn named_datatype_names(&self) -> Vec<&str> {
        self.datatypes.keys().map(String::as_str).collect()
    }

    /// The committed datatype at `path` (no leading `/`), following group hard
    /// links, soft links and external links the way `H5Topen` does.
    ///
    /// `NotFound` means no committed datatype of that name; a name that *is*
    /// one but whose type this crate cannot decode answers `Unsupported` with
    /// the reason, never an absence.
    pub fn named_datatype(&mut self, path: &str) -> IoResult<&DatatypeMessage> {
        self.named_datatype_info(path)?
            .datatype()
            .map_err(|why| crate::io::IoError::Unsupported(why.to_string()))
    }

    /// The attribute names of the committed datatype at `path`, in name
    /// order — matching h5py's default iteration for the (usual) case where
    /// the committed datatype does not track attribute creation order.
    /// Unlike [`Self::dataset_attr_names`] and its group/root counterparts,
    /// this path does not carry a per-attribute creation index to prefer
    /// when the object does track it: committed-datatype attributes are
    /// collected straight from compact header messages
    /// ([`Self::committed_datatype`]), without the envelope's creation index
    /// or dense-storage support the shared `AttributeEntry` collector has.
    pub fn named_datatype_attr_names(&mut self, path: &str) -> IoResult<Vec<String>> {
        let mut names: Vec<String> = self
            .named_datatype_info(path)?
            .attributes()
            .iter()
            .map(|a| a.name.clone())
            .collect();
        names.sort();
        Ok(names)
    }

    /// The committed datatype at `path`'s own object-header attribute count.
    ///
    /// Committed-datatype attributes are collected only from compact header
    /// messages ([`Self::committed_datatype`]) — this crate does not model
    /// dense attribute storage on a named datatype — so unlike
    /// [`ObjectAttributes::header_count`] this is simply the count of what
    /// [`Self::named_datatype_attr_names`] already lists, with no separate
    /// dense-index path to fall back to.
    pub fn named_datatype_header_attr_count(&mut self, path: &str) -> IoResult<u64> {
        Ok(self.named_datatype_info(path)?.attributes().len() as u64)
    }

    /// One attribute of the committed datatype at `path`, by name.
    pub fn named_datatype_attr(
        &mut self,
        path: &str,
        attr_name: &str,
    ) -> IoResult<&AttributeMessage> {
        let owned = attr_name.to_string();
        self.named_datatype_info(path)?
            .attributes()
            .iter()
            .find(|a| a.name == owned)
            .ok_or_else(|| crate::io::IoError::NotFound(format!("{path}:{attr_name}")))
    }

    /// The committed datatype object at `path`, after link traversal.
    ///
    /// The object answers here whether or not its type decodes; the reason it
    /// does not is on [`CommittedDatatypeInfo::datatype`].
    pub fn named_datatype_info(&mut self, path: &str) -> IoResult<&CommittedDatatypeInfo> {
        if self.external_edge(path).is_some() {
            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS)?;
            let local = owner.canonical_path(&local);
            return owner
                .datatypes
                .get(&local)
                .ok_or(crate::io::IoError::NotFound(local));
        }
        let local = self.canonical_path(path);
        self.datatypes
            .get(&local)
            .ok_or(crate::io::IoError::NotFound(local))
    }

    /// The class of the link at `path` (no leading `/`), or `None` when no
    /// link of that name exists. The path is traversed first, so a link
    /// reached through a group hard link, a soft link or an external link
    /// resolves — an external link's own record is found before the
    /// traversal crosses it, since a name matches its own link exactly.
    pub fn link_class(&mut self, path: &str) -> Option<&LinkClass> {
        let path = path.trim_start_matches('/');
        if self.links.contains_key(path) {
            return self.links.get(path);
        }
        if self.external_edge(path).is_some() {
            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
            let local = owner.canonical_path(&local);
            return owner.links.get(&local);
        }
        let path = self.canonical_path(path);
        self.links.get(&path)
    }

    /// Follow a path (no leading `/`) the way `H5Dopen` / `H5Gopen` do:
    /// rewrite each component that is a group hard-link alias or a soft link
    /// until nothing changes, bounded so a link cycle cannot loop forever.
    ///
    /// This is the single owner of link traversal — every lookup that takes a
    /// caller-supplied path goes through it rather than comparing the path to
    /// a catalog key directly.
    fn traverse(&self, name: &str) -> Traversal {
        // libhdf5 bounds soft-link traversal at `H5L_NLINKS_DEF`; this covers
        // that and the hard-link alias rewrites interleaved with it.
        const MAX_TRAVERSALS: usize = 64;
        let mut name = name.trim_start_matches('/').to_string();
        let mut via = None;
        for _ in 0..MAX_TRAVERSALS {
            let Some((prefix, rewrite)) = self.longest_rewrite(&name) else {
                break;
            };
            let rest = name[prefix.len()..].to_string();
            match rewrite {
                Rewrite::Alias(first) => {
                    // `first` is empty for an alias of the root group;
                    // trimming keeps the no-leading-'/' form either way.
                    name = format!("{first}{rest}").trim_start_matches('/').to_string();
                }
                Rewrite::Soft(target) => {
                    let resolved = resolve_link_value(prefix, target);
                    via = Some(SoftLinkRef {
                        link: prefix.to_string(),
                        target: target.to_string(),
                    });
                    name = format!("{resolved}{rest}")
                        .trim_start_matches('/')
                        .to_string();
                }
                Rewrite::External { file, path } => {
                    return Traversal::External {
                        link: prefix.to_string(),
                        file: file.to_string(),
                        path: format!("{path}{rest}"),
                    };
                }
            }
        }
        Traversal::Path { path: name, via }
    }

    /// The rewrite one traversal step applies to `path`, and the prefix it
    /// matched: the longest prefix of `path` that is a group hard-link alias
    /// or a soft/external link, so a nested alias wins over a shorter one
    /// that also covers the path, and an alias wins over a link naming the
    /// same prefix.
    ///
    /// A prefix covers `path` only when it *is* `path` or ends at one of its
    /// `/` boundaries, so the candidates are `path` and its own ancestors —
    /// walking those from the longest down asks the catalogs by key instead
    /// of comparing every alias and every link against the path, which is
    /// what made each traversal cost a pass over the file's whole link table.
    fn longest_rewrite<'a>(&'a self, path: &str) -> Option<(&'a str, Rewrite<'a>)> {
        let mut end = path.len();
        loop {
            let candidate = &path[..end];
            if let Some((alias, first)) = self.group_aliases.get_key_value(candidate) {
                return Some((alias.as_str(), Rewrite::Alias(first)));
            }
            match self.links.get_key_value(candidate) {
                Some((link, LinkClass::Soft { path })) => {
                    return Some((link.as_str(), Rewrite::Soft(path)))
                }
                Some((link, LinkClass::External { file, path })) => {
                    return Some((link.as_str(), Rewrite::External { file, path }))
                }
                // A hard or user-defined link rewrites nothing, and a shorter
                // prefix of the path may still rewrite it.
                _ => {}
            }
            end = candidate.rfind('/')?;
        }
    }

    /// The messages read from the superblock extension object header. All
    /// fields are `None` for a file without an extension.
    pub fn superblock_extension(&self) -> &SuperblockExtension {
        &self.ext
    }

    /// Bytes the file's on-disk free-space managers record as free —
    /// `H5Fget_freespace`, the number `h5stat -S` prints as "Amount of tracked
    /// free space".
    ///
    /// Zero for a file whose file-space info message names no manager, which
    /// includes every file written without `persist`. The strategy is not
    /// consulted: a manager's header and section-info blocks have one layout
    /// whichever strategy allocated the space they describe.
    pub fn tracked_free_space(&mut self) -> IoResult<u64> {
        let Some(info) = self.ext.file_space_info.clone() else {
            return Ok(0);
        };
        crate::io::free_space_io::tracked_free_space(&mut self.handle, &self.meta.ctx, &info)
    }

    /// Size in bytes of the userblock preceding the superblock: the offset the
    /// signature was found at, which is also the file's base address. Zero for
    /// a file without a userblock.
    pub fn userblock_size(&self) -> u64 {
        self.handle.base()
    }

    /// The superblock format version (0-3), decoded once at open time and
    /// immutable for the life of an open file — a live SWMR refresh rescans
    /// the file's contents but never its own format version.
    pub fn superblock_version(&self) -> u8 {
        self.superblock_version
    }

    /// Rewrite a path (no leading `/`) into the path of the object it reaches
    /// after link traversal. A path that leaves the file through an external
    /// link comes back unchanged — the callers that must report that case use
    /// [`Self::traverse`] directly.
    pub fn canonical_path(&self, name: &str) -> String {
        match self.traverse(name) {
            Traversal::Path { path, .. } => path,
            Traversal::External { .. } => name.trim_start_matches('/').to_string(),
        }
    }

    /// Where `name` leaves this file, or `None` when it resolves inside it.
    ///
    /// This is the one question every path-taking entry point asks before it
    /// looks anything up: a name that crosses an external link is not this
    /// file's to answer, and answering it from this file's catalog anyway is
    /// how such a name came back as a plain absence.
    pub(crate) fn external_edge(&self, name: &str) -> Option<ExternalEdge> {
        match self.traverse(name) {
            Traversal::Path { .. } => None,
            Traversal::External { link, file, path } => Some(ExternalEdge { link, file, path }),
        }
    }

    /// Candidate filesystem paths for a file named from inside this one, in
    /// the order `H5F_prefix_open_file` tries them (H5Fint.c:826-1025):
    ///
    /// 1. an absolute name exactly as given (:854-887) — and if that misses,
    ///    every later step uses its last component instead, as the C does;
    /// 2. each `:`-separated component of `env_var`, joined with that name
    ///    (:889-937);
    /// 3. `prop_prefix`, the property-list prefix (:938-950);
    /// 4. the directory of the path this file was opened by — libhdf5's
    ///    `H5F_EXTPATH` (:952-969);
    /// 5. the bare relative name, against the process's working directory
    ///    (:971-977);
    /// 6. the directory of that path *resolved* — libhdf5's
    ///    `H5F_ACTUAL_NAME`, which differs from step 4 through a symlink
    ///    (:979-1004).
    ///
    /// Both kinds of cross-file name run this one order and differ only in
    /// the two parameters: an external link is `H5F_PREFIX_ELINK` with
    /// `HDF5_EXT_PREFIX` and `H5Pset_elink_prefix` (H5Lexternal.c:210-215),
    /// a virtual dataset's source is `H5F_PREFIX_VDS` with
    /// `HDF5_VDS_PREFIX` and `H5Pset_virtual_prefix` (H5Dvirtual.c:877-882).
    fn prefix_open_candidates(
        &self,
        env_var: &str,
        prop_prefix: Option<&Path>,
        file: &str,
    ) -> Vec<PathBuf> {
        let raw = Path::new(file);
        let mut candidates = Vec::new();
        if raw.is_absolute() {
            candidates.push(raw.to_path_buf());
        }
        // Every attempt after an absolute miss uses the bare file name.
        let base: &Path = if raw.is_absolute() {
            Path::new(raw.file_name().unwrap_or(raw.as_os_str()))
        } else {
            raw
        };
        if let Ok(prefixes) = std::env::var(env_var) {
            candidates.extend(
                prefixes
                    .split(':')
                    .filter(|p| !p.is_empty())
                    .map(|p| Path::new(p).join(base)),
            );
        }
        if let Some(prefix) = prop_prefix {
            candidates.push(prefix.join(base));
        }
        if let Some(dir) = self.path.parent().filter(|d| !d.as_os_str().is_empty()) {
            candidates.push(dir.join(base));
        }
        candidates.push(base.to_path_buf());
        if !self.source_dir.as_os_str().is_empty() {
            candidates.push(self.source_dir.join(base));
        }
        candidates
    }

    /// Put an external-link prefix in force for this reader and every file
    /// it opens on another's behalf. Set once at open, before any name has
    /// been resolved, because an external link's answer is fixed the first
    /// time it is asked ([`external_resolved`](Self::external_resolved)).
    pub(crate) fn set_elink_prefix(&mut self, prefix: Option<String>) {
        self.elink_prefix = prefix;
    }

    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for an
    /// external link. The property-list step is
    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix)
    /// exactly as given: `H5L__extern_traverse` peeks
    /// `H5L_ACS_ELINK_PREFIX_NAME` and hands it straight to the search
    /// (H5Lexternal.c:210-215), so unlike a virtual dataset's prefix it goes
    /// through no `H5D__build_file_prefix` — no `${ORIGIN}` expansion, and
    /// `HDF5_EXT_PREFIX` does not shadow it.
    fn external_candidates(&self, file: &str) -> Vec<PathBuf> {
        let prop = self.elink_prefix.as_deref().map(Path::new);
        self.prefix_open_candidates("HDF5_EXT_PREFIX", prop, file)
    }

    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for a
    /// virtual dataset's source. The property-list step is whatever
    /// `H5D__build_file_prefix` puts in `dset->shared->vds_prefix`
    /// (H5Dint.c:1076-1119): `HDF5_VDS_PREFIX` if the environment names one,
    /// otherwise [`DatasetAccess::virtual_prefix`], either way with
    /// `${ORIGIN}` expanded.
    fn vds_candidates(&self, access: &DatasetAccess, file: &str) -> Vec<PathBuf> {
        let prop = resolve_vdsfile_prefix(access.virtual_prefix_value(), &self.source_dir);
        self.prefix_open_candidates("HDF5_VDS_PREFIX", prop.as_deref(), file)
    }

    /// Open one external link's target file, or hand back the handle a
    /// previous link to the same resolved path already opened.
    fn external_target(&mut self, link: &str, file: &str) -> IoResult<&mut Hdf5Reader> {
        // The search runs once per link value; after that the answer is what
        // this reader resolved it to, whatever the filesystem does next.
        let resolved = match self.external_resolved.get(file) {
            Some(resolved) => resolved.clone(),
            None => {
                let candidates = self.external_candidates(file);
                let resolved = candidates
                    .iter()
                    .find(|p| p.is_file())
                    .cloned()
                    .ok_or_else(|| crate::io::IoError::ExternalFileNotFound {
                        link: link.to_string(),
                        file: file.to_string(),
                        searched: candidates.iter().map(|p| p.display().to_string()).collect(),
                    })?;
                self.external_resolved
                    .insert(file.to_string(), resolved.clone());
                resolved
            }
        };
        self.cross_file(resolved, CrossFileOwner::Reader)
    }

    /// Open `resolved`, or hand back the handle a previous crossing to the
    /// same file already opened.
    ///
    /// The single owner of every file this reader opens on another file's
    /// behalf, so a path named by any number of external links, external
    /// references and virtual-dataset sources is opened once and read
    /// through one handle. What resolved the name to this path is the
    /// caller's business, and differs by kind: an external link and a
    /// virtual source each run `H5F_prefix_open_file`'s search order under
    /// their own prefix, a reference has no search order at all.
    ///
    /// `owner` says how long the handle stays open. One path can be reached
    /// by both kinds of crossing, and the wider ownership wins: a file an
    /// external link holds for this reader's life does not start expiring
    /// with a virtual dataset that also names it.
    fn cross_file(
        &mut self,
        resolved: PathBuf,
        owner: CrossFileOwner,
    ) -> IoResult<&mut Hdf5Reader> {
        let locking = self.locking;
        let elink_prefix = self.elink_prefix.clone();
        match self.external.entry(resolved) {
            std::collections::btree_map::Entry::Occupied(e) => {
                let e = e.into_mut();
                e.owner.widen(owner);
                Ok(&mut *e.reader)
            }
            std::collections::btree_map::Entry::Vacant(e) => {
                let mut reader = Hdf5Reader::open_with_locking(e.key(), locking)?;
                reader.elink_prefix.clone_from(&elink_prefix);
                Ok(&mut *e
                    .insert(CrossFileEntry {
                        reader: Box::new(reader),
                        owner,
                    })
                    .reader)
            }
        }
    }

    /// Close every cross-file handle whose last owning virtual-dataset open
    /// has gone, which is where `H5D__virtual_reset_layout` closes the source
    /// datasets holding libhdf5's (H5Dvirtual.c:709-710).
    ///
    /// The single releaser of a [`CrossFileOwner::VirtualOpens`] entry —
    /// nothing else removes one, so a source cannot be closed while a handle
    /// on the virtual dataset that named it is still alive. Its two callers
    /// are the two moments the owning set can be empty: a handle's drop, and
    /// an extent resolution run with no handle open at all (this crate
    /// resolves at `H5Fopen`, where libhdf5 has nothing to resolve yet).
    pub(crate) fn release_closed_virtual_sources(&mut self) {
        let dead: Vec<PathBuf> = self
            .external
            .iter()
            .filter(|(_, e)| match &e.owner {
                CrossFileOwner::Reader => false,
                CrossFileOwner::VirtualOpens(vds) => !vds.iter().any(|v| self.is_open_dataset(v)),
            })
            .map(|(p, _)| p.clone())
            .collect();
        for path in dead {
            self.external.remove(&path);
        }
    }

    /// Whether the dataset at canonical path `name` still has a live handle
    /// — libhdf5's "is this dataset in `H5FO_opened`".
    fn is_open_dataset(&self, name: &str) -> bool {
        self.dataset_access
            .get(name)
            .is_some_and(|e| e.open.strong_count() > 0)
    }

    /// Open the file a virtual mapping's source name points at, or hand back
    /// the handle a previous mapping to the same file already opened.
    ///
    /// A virtual dataset's source is not opened by any path of its own:
    /// `H5D__virtual_open_source_dset` hands the name to
    /// `H5F_prefix_open_file` (H5Dvirtual.c:877-882) against the *primary*
    /// file's external file cache and with `source_fapl`, a copy of the
    /// primary file's own file-access property list
    /// (H5Dvirtual.c:2193-2194), which carries its `use_file_locking`
    /// verbatim (H5Fint.c:389). That is the same cache and the same call
    /// external links reach, keyed by the name the open used
    /// (H5Fefc.c:245), and it is released back to it with `H5F_efc_close`
    /// (H5Dvirtual.c:925-927). So a source file is this reader's
    /// [`cross_file`](Self::cross_file) like any other target: one handle
    /// per path, under this file's locking policy — measured against
    /// libhdf5 1.14.6, reading a cross-file VDS leaves the source flock'd
    /// under the default `HDF5_USE_FILE_LOCKING` and unlocked under
    /// `HDF5_USE_FILE_LOCKING=FALSE`.
    ///
    /// What it is *not* is a target this reader holds for its own life:
    /// `vds` — the canonical path of the virtual dataset naming the source —
    /// owns the handle, and it goes when that dataset's last handle does
    /// ([`CrossFileOwner::VirtualOpens`]).
    ///
    /// `None` where the file cannot be opened at all: `H5F_prefix_open_file`
    /// is asked to *try*, and a null source file is "no data there yet",
    /// not a failure.
    fn vds_source_file(&mut self, vds: &str, file_name: &str) -> Option<&mut Hdf5Reader> {
        // A name that has resolved once stays resolved, the same way an
        // external link's does — the handle this reader already holds is the
        // answer, whatever the filesystem does next. A name that has *not*
        // resolved is searched again on the next read, which is what the C
        // does too: it re-runs `H5D__virtual_open_source_dset` whenever the
        // source dataset is still unopened (H5Dvirtual.c:1421-1423,
        // :2558-2561).
        let key = (vds.to_string(), file_name.to_string());
        let resolved = match self.vds_resolved.get(&key) {
            Some(resolved) => resolved.clone(),
            None => {
                let access = self.access_in_force(vds);
                let resolved = self
                    .vds_candidates(&access, file_name)
                    .into_iter()
                    .find(|p| p.is_file())?;
                self.vds_resolved.insert(key, resolved.clone());
                resolved
            }
        };
        self.cross_file(resolved, CrossFileOwner::virtual_open(vds))
            .ok()
    }

    /// The reader that owns `name`, the path of `name` inside it, and the last
    /// external link crossed to get there.
    ///
    /// This is the single owner of cross-file resolution: it follows external
    /// links until the remaining path resolves inside the reader it returns,
    /// so callers do exactly one delegation and never have to re-check.
    fn external_owner(
        &mut self,
        name: &str,
        hops: usize,
    ) -> IoResult<(&mut Self, String, Option<ExternalEdge>)> {
        let path = name.trim_start_matches('/').to_string();
        let Some(edge) = self.external_edge(&path) else {
            return Ok((self, path, None));
        };
        if hops == 0 {
            return Err(crate::io::IoError::InvalidState(format!(
                "resolving '{name}' crossed more than {MAX_EXTERNAL_HOPS} external links \
                 (libhdf5 stops at the same H5L_NUM_LINKS); the links may form a cycle"
            )));
        }
        let target = self.external_target(&edge.link, &edge.file)?;
        let (owner, path, deeper) = target.external_owner(&edge.path, hops - 1)?;
        Ok((owner, path, deeper.or(Some(edge))))
    }

    /// Return metadata for a dataset by name. Like `H5Dopen`, the name may
    /// pass through group hard links, soft links and external links.
    pub fn dataset_info(&mut self, name: &str) -> Option<&DatasetReadInfo> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS).ok()?;
            return owner.dataset_info_local(&path);
        }
        self.dataset_info_local(name)
    }

    /// [`dataset_info`](Self::dataset_info) restricted to this file: soft
    /// links and group hard links resolve, an external link does not. Every
    /// read path uses this, because by then the owning reader has already been
    /// selected and the path is local to it.
    fn dataset_info_local(&self, name: &str) -> Option<&DatasetReadInfo> {
        let name = self.canonical_path(name);
        self.datasets.get(&name)
    }

    /// Where `name` sits in this file's catalog, resolving the same soft
    /// links and group hard links [`dataset_info_local`](Self::dataset_info_local)
    /// does. Read paths that need the entry more than once take the position
    /// once and index with it, rather than walking the path again per lookup.
    fn dataset_position(&self, name: &str) -> IoResult<usize> {
        let canonical = self.canonical_path(name);
        self.datasets
            .position(&canonical)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
    }

    /// Open `name` as a dataset the way `H5Dopen2` does, reporting *why* it
    /// cannot be opened instead of collapsing every cause into absence: a
    /// soft link whose target does not exist is a dangling link, and a path
    /// through an external link is resolved in the file that link names.
    ///
    /// This is the gate every typed dataset access goes through. `access` is
    /// the dapl the open names; its properties are put in force for the
    /// dataset first, so the extent this returns is the one they resolve it
    /// to.
    ///
    /// Returns the token that holds the open alive alongside the extent: the
    /// caller must keep it for as long as its handle lives, because the
    /// properties this open put in force stay in force exactly that long
    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
    pub fn open_dataset_with(
        &mut self,
        name: &str,
        access: &DatasetAccess,
    ) -> IoResult<(Option<DatasetOpenToken>, &DatasetReadInfo)> {
        if self.external_edge(name).is_some() {
            let (owner, path, edge) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            let open = owner.apply_dataset_access(&path, access)?;
            return match owner.open_dataset_local(&path) {
                // The name is absent in the target file, which makes the link
                // that pointed there dangling — not the caller's path absent.
                Err(crate::io::IoError::NotFound(_)) => {
                    Err(edge.map_or_else(|| crate::io::IoError::NotFound(path), |e| e.dangling()))
                }
                other => other.map(|info| (open, info)),
            };
        }
        let open = self.apply_dataset_access(name, access)?;
        self.open_dataset_local(name).map(|info| (open, info))
    }

    /// [`open_dataset`](Self::open_dataset) restricted to this file.
    fn open_dataset_local(&self, name: &str) -> IoResult<&DatasetReadInfo> {
        let Traversal::Path { path, via } = self.traverse(name) else {
            // `external_owner` only ever returns a reader in which the
            // remaining path resolves locally, so no caller can land here.
            return Err(crate::io::IoError::NotFound(name.to_string()));
        };
        if let Some(info) = self.datasets.get(&path) {
            return Ok(info);
        }
        if let Some(why) = self.unreadable.get(&path) {
            return Err(crate::io::IoError::Unsupported(format!(
                "'{name}' is a dataset this crate cannot read: {why}"
            )));
        }
        if let Some(SoftLinkRef { link, target }) = via {
            return Err(crate::io::IoError::DanglingLink { link, target });
        }
        Err(crate::io::IoError::NotFound(name.to_string()))
    }

    /// Resolve one entry of an attribute list.
    ///
    /// The cases an attribute list can answer are kept apart here rather than
    /// at each call site: decoded, present but undecodable, absent from a set
    /// known to be whole, and absent from a set that was never read whole. A
    /// caller that collapsed any of the middle cases into the last would
    /// report an attribute the file contains as one it does not.
    fn resolve_attr<'a>(
        attrs: &'a ObjectAttributes,
        owner: &str,
        name: &str,
    ) -> IoResult<&'a AttributeMessage> {
        match attrs.entries.iter().find(|a| a.name() == name) {
            Some(entry) => entry.decoded().map_err(|reason| {
                crate::io::IoError::Unsupported(format!(
                    "attribute '{name}' on '{owner}' cannot be decoded: {reason}"
                ))
            }),
            // Not among what was read — but the part that was not read could
            // hold it, so an incomplete set cannot answer "absent".
            None => match attrs.unreadable_reason() {
                Some(reason) => Err(incomplete_error(owner, reason)),
                None => Err(crate::io::IoError::NotFound(format!("{owner}:{name}"))),
            },
        }
    }

    /// Why the attribute `name` in `attrs` cannot be read, or `None` when it
    /// can be — or is not there at all, which the accessors above report as
    /// `NotFound`.
    fn attr_reason<'a>(attrs: &'a ObjectAttributes, name: &str) -> Option<&'a str> {
        attrs
            .entries
            .iter()
            .find(|a| a.name() == name)?
            .unreadable_reason()
    }

    /// Why a dataset's attribute cannot be read, or `None` when it can be.
    pub fn dataset_attr_unreadable_reason(
        &mut self,
        ds_name: &str,
        attr_name: &str,
    ) -> Option<&str> {
        Self::attr_reason(&self.dataset_info(ds_name)?.attributes, attr_name)
    }

    /// Why a root-level attribute cannot be read, or `None` when it can be.
    pub fn root_attr_unreadable_reason(&self, name: &str) -> Option<&str> {
        Self::attr_reason(&self.root_attributes, name)
    }

    /// Why a non-root group's attribute cannot be read, or `None` when it can
    /// be.
    pub fn group_attr_unreadable_reason(&self, group_path: &str, name: &str) -> Option<&str> {
        Self::attr_reason(
            self.group_attributes
                .get(&self.canonical_path(group_path))?,
            name,
        )
    }

    /// Why a dataset's attributes cannot be listed at all, or `None` when the
    /// set is whole. Object scope, unlike
    /// [`Self::dataset_attr_unreadable_reason`]: the failure belongs to no
    /// single name.
    pub fn dataset_attrs_unreadable_reason(&mut self, ds_name: &str) -> Option<&str> {
        self.dataset_info(ds_name)?.attributes.unreadable_reason()
    }

    /// A dataset's own compact-vs-dense attribute storage.
    pub fn dataset_attr_storage(&mut self, ds_name: &str) -> IoResult<AttributeStorage> {
        Ok(self
            .dataset_info(ds_name)
            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?
            .attributes
            .storage())
    }

    /// A dataset's own object-header attribute count.
    pub fn dataset_header_attr_count(&mut self, ds_name: &str) -> IoResult<u64> {
        let info = self
            .dataset_info(ds_name)
            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
        info.attributes.header_count(ds_name)
    }

    /// Why the root group's attributes cannot be listed at all, or `None` when
    /// the set is whole.
    pub fn root_attrs_unreadable_reason(&self) -> Option<&str> {
        self.root_attributes.unreadable_reason()
    }

    /// Why a non-root group's attributes cannot be listed at all, or `None`
    /// when the set is whole.
    pub fn group_attrs_unreadable_reason(&self, group_path: &str) -> Option<&str> {
        self.group_attributes
            .get(&self.canonical_path(group_path))?
            .unreadable_reason()
    }

    /// Return the attribute names of a dataset.
    ///
    /// Includes attributes this crate cannot decode: the object header carries
    /// them, so the listing does too. [`Self::dataset_attr`] says why one of
    /// those cannot be read. An object whose attribute set could not be read
    /// whole has no listing to give and returns the reason instead — see
    /// [`Self::dataset_attrs_unreadable_reason`].
    pub fn dataset_attr_names(&mut self, name: &str) -> IoResult<Vec<String>> {
        let info = self
            .dataset_info(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        info.attributes.ordered_names(name)
    }

    /// Return a specific attribute by dataset name and attribute name.
    pub fn dataset_attr(&mut self, ds_name: &str, attr_name: &str) -> IoResult<&AttributeMessage> {
        let info = self
            .dataset_info(ds_name)
            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
        Self::resolve_attr(&info.attributes, ds_name, attr_name)
    }

    /// Return the names of root-level (file) attributes, undecodable ones
    /// included — see [`Self::dataset_attr_names`].
    pub fn root_attr_names(&self) -> IoResult<Vec<String>> {
        self.root_attributes.ordered_names("/")
    }

    /// Return a root-level attribute by name.
    pub fn root_attr(&self, name: &str) -> IoResult<&AttributeMessage> {
        Self::resolve_attr(&self.root_attributes, "/", name)
    }

    /// The root group's own attribute creation-order policy.
    pub fn root_attr_creation_order(&self) -> CreationOrder {
        self.root_attributes.creation_order()
    }

    /// The root group's own compact-vs-dense attribute storage.
    pub fn root_attr_storage(&self) -> AttributeStorage {
        self.root_attributes.storage()
    }

    /// The root group's own object-header attribute count.
    pub fn root_header_attr_count(&self) -> IoResult<u64> {
        self.root_attributes.header_count("/")
    }

    /// The root group's own link creation-order policy.
    pub fn root_link_creation_order(&self) -> CreationOrder {
        self.root_link_storage.1
    }

    /// The root group's own link storage kind: symbol-table (legacy),
    /// compact link messages, or dense (fractal heap plus name index).
    pub fn root_link_storage(&self) -> LinkStorage {
        self.root_link_storage.0
    }

    /// Return the attribute names of a non-root group (path without a
    /// leading `/`, e.g. `"detector"` or `"entry/instrument"`; may pass
    /// through group hard links). Undecodable attributes included — see
    /// [`Self::dataset_attr_names`].
    pub fn group_attr_names(&mut self, group_path: &str) -> IoResult<Vec<String>> {
        if self.external_edge(group_path).is_some() {
            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
            // The empty remainder is the target file's root group, whose
            // attributes are not in the per-group map.
            if local.is_empty() {
                return owner.root_attr_names();
            }
            return owner.group_attr_names_local(&local);
        }
        self.group_attr_names_local(group_path)
    }

    fn group_attr_names_local(&self, group_path: &str) -> IoResult<Vec<String>> {
        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
            return Ok(Vec::new());
        };
        attrs.ordered_names(group_path)
    }

    /// A non-root group's own attribute creation-order policy. `Untracked`
    /// for a path the walk never reached, the same silent default
    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
    /// unknown group's attribute listing.
    pub fn group_attr_creation_order(&self, group_path: &str) -> CreationOrder {
        self.group_attributes
            .get(&self.canonical_path(group_path))
            .map(ObjectAttributes::creation_order)
            .unwrap_or_default()
    }

    /// A non-root group's own compact-vs-dense attribute storage. `Compact`
    /// — the same silent default as an empty attribute set — for a path the
    /// walk never reached.
    pub fn group_attr_storage(&self, group_path: &str) -> AttributeStorage {
        self.group_attributes
            .get(&self.canonical_path(group_path))
            .map(ObjectAttributes::storage)
            .unwrap_or_default()
    }

    /// A non-root group's own object-header attribute count. `0` for a path
    /// the walk never reached, the same silent default
    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
    /// unknown group's attribute listing.
    pub fn group_header_attr_count(&self, group_path: &str) -> IoResult<u64> {
        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
            return Ok(0);
        };
        attrs.header_count(group_path)
    }

    /// A non-root group's own link creation-order policy. `Untracked` for a
    /// path the walk never reached, the same silent default
    /// [`group_attr_creation_order`](Self::group_attr_creation_order) gives.
    pub fn group_link_creation_order(&self, group_path: &str) -> CreationOrder {
        self.group_link_storage
            .get(&self.canonical_path(group_path))
            .map_or(CreationOrder::Untracked, |(_, order)| *order)
    }

    /// A non-root group's own link storage kind. `Compact` — the same
    /// silent default as an empty link set — for a path the walk never
    /// reached.
    pub fn group_link_storage(&self, group_path: &str) -> LinkStorage {
        self.group_link_storage
            .get(&self.canonical_path(group_path))
            .map_or(LinkStorage::Compact, |(storage, _)| *storage)
    }

    /// Return a non-root group's attribute by name.
    pub fn group_attr(&mut self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
        if self.external_edge(group_path).is_some() {
            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
            if local.is_empty() {
                return owner.root_attr(name);
            }
            return owner.group_attr_local(&local, name);
        }
        self.group_attr_local(group_path, name)
    }

    fn group_attr_local(&self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
        match self.group_attributes.get(&self.canonical_path(group_path)) {
            Some(attrs) => Self::resolve_attr(attrs, group_path, name),
            // No entry at all: the walk found nothing to record on this group.
            None => Err(crate::io::IoError::NotFound(format!("{group_path}:{name}"))),
        }
    }

    /// Return every non-root group path the discovery walk traversed into
    /// (no leading `/`). Built from actual link records, so empty groups,
    /// attribute-only groups, and subgroup-only groups are all included.
    pub fn group_paths(&self) -> &std::collections::BTreeSet<String> {
        &self.group_paths
    }

    /// Report whether a group exists at `group_path` (no leading `/`;
    /// may pass through group hard links). The empty string denotes the
    /// root group, which always exists.
    pub fn has_group(&self, group_path: &str) -> bool {
        if group_path.is_empty() || self.group_paths.contains(group_path) {
            return true;
        }
        let canon = self.canonical_path(group_path);
        canon.is_empty() || self.group_paths.contains(&canon)
    }

    /// Read and decode the global-heap collection at `addr`, applying the
    /// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
    /// signature must be present and the declared size at least
    /// `H5HG_MINSIZE` (4096 bytes). There is no upper size cap — libhdf5
    /// has none, and this crate's writers put a whole write call's strings
    /// into one collection, which a cap would turn into silent data loss.
    fn read_heap_collection(&mut self, addr: u64) -> IoResult<GlobalHeapCollection> {
        read_heap_collection_from(&mut self.handle, &self.meta.ctx, addr)
    }

    /// Decode an attribute's value as a string, resolving a variable-length
    /// string attribute through the global heap (h5py writes string
    /// attributes as variable-length by default).
    pub fn attr_string_value(&mut self, attr: &AttributeMessage) -> IoResult<String> {
        use crate::format::messages::datatype::DatatypeMessage;
        if !matches!(attr.datatype, DatatypeMessage::VarLenString { .. }) {
            return fixed_string_attr_value(attr);
        }
        // Variable-length string: the attribute value is a global-heap
        // reference (sequence length + collection address + object index).
        if attr.data.len() < vlen_reference_size(&self.meta.ctx) {
            return Ok(String::new());
        }
        let (_seq, coll_addr, obj_index) = decode_vlen_reference(&attr.data, &self.meta.ctx)?;
        if coll_addr == UNDEF_ADDR || coll_addr == 0 {
            return Ok(String::new());
        }
        let coll = self.read_heap_collection(coll_addr)?;
        let idx = u16::try_from(obj_index).map_err(|_| {
            crate::io::IoError::InvalidState(format!(
                "global heap object index {obj_index} does not fit the 16-bit on-disk field"
            ))
        })?;
        let obj = coll.get_object(idx).ok_or_else(|| {
            crate::io::IoError::InvalidState(format!(
                "global heap object {idx} not found in the collection at address {coll_addr:#x}"
            ))
        })?;
        Ok(String::from_utf8_lossy(obj).to_string())
    }

    /// The absolute path of the object whose header sits at `addr` — what an
    /// object reference to it names — or `None` when no group or dataset the
    /// discovery walk reached lives there (a reference into a file region the
    /// walk never traversed, or a stale one).
    pub fn path_for_object(&self, addr: u64) -> Option<&str> {
        self.object_paths.get(&addr).map(String::as_str)
    }

    /// How the object header of `path` stores each message it does not hold
    /// privately, as `(message type, storage)` in header order.
    ///
    /// The observable is the message's flags byte, so this reads the raw
    /// header chain: every other read path resolves shared pointers into
    /// bodies and clears the flag on the way through, which is exactly the
    /// evidence wanted here.
    pub fn object_message_storage(&mut self, path: &str) -> IoResult<Vec<(u8, MessageStorage)>> {
        let addr = self.object_header_address(path)?;
        crate::io::object_header_io::read_header_message_storage(&mut self.handle, &self.meta, addr)
    }

    /// The flags byte of every message the object header of `path` holds, as
    /// `(message type, flags)` in header order, null and continuation
    /// messages left out.
    ///
    /// The byte `h5debug` renders as `<C>`, `<DS>`, `<S>` and the rest
    /// (`H5O__debug_real`, H5Odbg.c:409-455). It says which messages the
    /// library may cache as never-changing and which it refuses to share, and
    /// nothing else in the file records either.
    pub fn object_message_flags(&mut self, path: &str) -> IoResult<Vec<(u8, u8)>> {
        let addr = self.object_header_address(path)?;
        crate::io::object_header_io::read_header_message_flags(&mut self.handle, &self.meta, addr)
    }

    /// The class and version of every datatype message the object at `path`
    /// carries, outermost first; see
    /// [`DatatypeMessage::decode_versions`](crate::format::messages::datatype::DatatypeMessage::decode_versions).
    pub fn object_datatype_versions(
        &mut self,
        path: &str,
    ) -> IoResult<Vec<crate::format::messages::datatype::DatatypeNodeVersion>> {
        let addr = self.object_header_address(path)?;
        crate::io::object_header_io::read_header_datatype_versions(
            &mut self.handle,
            &self.meta,
            addr,
        )
    }

    /// Whether the object at `path` records its times —
    /// `H5Pget_obj_track_times` on the property list it was created with, read
    /// back from the header that answers it; see
    /// [`ObjectHeader::recorded_times`](crate::format::object_header::ObjectHeader::recorded_times).
    pub fn object_records_times(&mut self, path: &str) -> IoResult<bool> {
        let addr = self.object_header_address(path)?;
        Ok(crate::io::object_header_io::read_header_recorded_times(
            &mut self.handle,
            &self.meta,
            addr,
        )?
        .is_some())
    }

    /// The object header address `path` names.
    ///
    /// By name first, then by address: `object_paths` keeps one path per
    /// object header, so a hard link — two names, one header — is only ever
    /// found under whichever name the walk reached first.
    fn object_header_address(&mut self, path: &str) -> IoResult<u64> {
        if self.external_edge(path).is_some() {
            return Err(crate::io::IoError::NotFound(format!(
                "{path} is in another file; its header is not this file's to read"
            )));
        }
        match self.dataset_info(path) {
            Some(info) => Ok(info.object_header_address),
            None => {
                let want = absolute_path(&self.canonical_path(path));
                self.object_paths
                    .iter()
                    .find(|(_, p)| **p == want)
                    .map(|(addr, _)| *addr)
                    .ok_or_else(|| crate::io::IoError::NotFound(path.to_string()))
            }
        }
    }

    /// Read a reference dataset's elements, resolved to the objects they name.
    pub fn read_references(&mut self, name: &str) -> IoResult<Vec<Reference>> {
        let datatype = self
            .dataset_info(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?
            .datatype
            .clone();
        let raw = self.read_dataset_raw(name)?;
        self.decode_references(&datatype, &raw)
    }

    /// Read an attribute's value as reference elements.
    pub fn attr_references(&mut self, attr: &AttributeMessage) -> IoResult<Vec<Reference>> {
        self.decode_references(&attr.datatype, &attr.data)
    }

    /// The single owner of reference decoding for both carriers of reference
    /// elements — dataset payloads and attribute values.
    fn decode_references(
        &mut self,
        datatype: &DatatypeMessage,
        bytes: &[u8],
    ) -> IoResult<Vec<Reference>> {
        let DatatypeMessage::Reference { size, kind } = datatype else {
            return Err(crate::io::IoError::InvalidState(format!(
                "datatype {datatype} is not a reference"
            )));
        };
        let (size, kind) = (*size as usize, *kind);
        if size == 0 {
            // A corrupt file can declare it; `chunks_exact(0)` panics.
            return Err(crate::io::IoError::InvalidState(
                "reference datatype has zero width".into(),
            ));
        }
        let encoding = kind.encoding();

        // References written by one call share one heap collection, so read
        // each collection once rather than per element.
        let mut heaps = std::collections::HashMap::new();
        let mut out = Vec::with_capacity(bytes.len() / size);
        for elem in bytes.chunks_exact(size) {
            out.push(self.decode_reference_element(elem, encoding, &mut heaps)?);
        }
        Ok(out)
    }

    /// One reference element, resolved against the file.
    ///
    /// `heaps` caches the global-heap collections region references point
    /// into, keyed by collection address.
    fn decode_reference_element(
        &mut self,
        elem: &[u8],
        encoding: ReferenceEncoding,
        heaps: &mut std::collections::HashMap<u64, GlobalHeapCollection>,
    ) -> IoResult<Reference> {
        match encoding {
            ReferenceEncoding::Old(OldReferenceKind::Object) => {
                match decode_object_element(elem, &self.meta.ctx)? {
                    None => Ok(Reference::Null),
                    Some(address) => Ok(self.resolve_reference(DecodedReference {
                        address,
                        file: None,
                        target: ReferenceTarget::Object,
                    })),
                }
            }
            ReferenceEncoding::Old(OldReferenceKind::DatasetRegion) => {
                let Some((coll_addr, obj_index)) = decode_region_element(elem, &self.meta.ctx)?
                else {
                    return Ok(Reference::Null);
                };
                let obj = self.heap_object(coll_addr, obj_index, heaps)?;
                let (address, selection) = decode_region_heap_object(obj, &self.meta.ctx)?;
                Ok(self.resolve_reference(DecodedReference {
                    address,
                    file: None,
                    target: ReferenceTarget::Region(selection),
                }))
            }
            ReferenceEncoding::Revised => {
                let (kind, external, body) = match decode_revised_element(elem, &self.meta.ctx)? {
                    RevisedElement::Null => return Ok(Reference::Null),
                    RevisedElement::Inline { kind, body } => (kind, false, body.to_vec()),
                    RevisedElement::Heap {
                        kind,
                        external,
                        collection,
                        index,
                    } => (
                        kind,
                        external,
                        self.heap_object(collection, index, heaps)?.to_vec(),
                    ),
                };
                match decode_revised_body(kind, external, &body, &self.meta.ctx)? {
                    None => Ok(Reference::Null),
                    Some(decoded) => Ok(self.resolve_reference(decoded)),
                }
            }
        }
    }

    /// Attach the target's path to a decoded reference — the one place an
    /// address becomes a [`Reference`], so every kind resolves the same way.
    ///
    /// A reference naming another file is looked up in that file, which this
    /// opens by the name the reference carries and nothing else:
    /// `H5R__reopen_file` hands the name straight to `H5VL_file_open` with no
    /// prefix search, so it is read against the process working directory the
    /// way `H5Ropen_object` would read it (H5Rint.c:466, :487). A file that is
    /// not there leaves the path unresolved while the reference still names
    /// it, which is `H5Rget_file_name` answering from the reference alone
    /// while `H5Ropen_object` fails (H5R.c:1036-1039).
    fn resolve_reference(&mut self, decoded: DecodedReference) -> Reference {
        let DecodedReference {
            address,
            file,
            target,
        } = decoded;
        let path = match &file {
            None => self.path_for_object(address).map(str::to_string),
            Some(name) => self
                .cross_file(PathBuf::from(name), CrossFileOwner::Reader)
                .ok()
                .and_then(|target| target.path_for_object(address).map(str::to_string)),
        };
        match target {
            ReferenceTarget::Object => Reference::Object {
                address,
                file,
                path,
            },
            ReferenceTarget::Region(selection) => Reference::Region {
                address,
                file,
                path,
                selection,
            },
            ReferenceTarget::Attribute(name) => Reference::Attr {
                address,
                file,
                path,
                name,
            },
        }
    }

    /// One global-heap object, reading its collection at most once.
    fn heap_object<'h>(
        &mut self,
        collection: u64,
        index: u32,
        heaps: &'h mut std::collections::HashMap<u64, GlobalHeapCollection>,
    ) -> IoResult<&'h [u8]> {
        if let std::collections::hash_map::Entry::Vacant(slot) = heaps.entry(collection) {
            slot.insert(self.read_heap_collection(collection)?);
        }
        let idx = u16::try_from(index).map_err(|_| {
            crate::io::IoError::InvalidState(format!(
                "global heap object index {index} does not fit the 16-bit on-disk field"
            ))
        })?;
        heaps[&collection].get_object(idx).ok_or_else(|| {
            crate::io::IoError::InvalidState(format!(
                "global heap object {idx} not found in the collection at address {collection:#x}"
            ))
        })
    }

    /// Return the dimensions of a dataset.
    pub fn dataset_shape(&mut self, name: &str) -> IoResult<Vec<u64>> {
        let info = self
            .dataset_info(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        Ok(info.dataspace.dims.clone())
    }

    /// Logical byte size of a dataset's full image (`product(dims) *
    /// element_size`), with the datatype needed for the post-filter conversion.
    fn raw_size_and_datatype(&self, name: &str) -> IoResult<(DatatypeMessage, u64)> {
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        Ok((info.datatype.clone(), Self::raw_size_of(info)))
    }

    /// Logical byte size of `info`'s full image.
    ///
    /// The NULL dataspace (`dataspace.is_null()`) holds zero elements — not
    /// one, the way an empty `dims` would suggest by the same product-of-dims
    /// arithmetic a scalar dataspace uses (`dims` is empty for both).
    fn raw_size_of(info: &DatasetReadInfo) -> u64 {
        if info.dataspace.is_null() {
            0
        } else {
            saturating_byte_len(&info.dataspace.dims, info.datatype.element_size() as u64)
        }
    }

    /// Logical byte size of a dataset's full image: how many bytes
    /// [`read_dataset_raw`](Self::read_dataset_raw) returns, and how large a
    /// buffer [`read_dataset_raw_into`](Self::read_dataset_raw_into) needs.
    ///
    /// Resolved in the file that owns the dataset, so a name crossing an
    /// external link answers with the target's size rather than an absence.
    pub fn dataset_raw_size(&mut self, name: &str) -> IoResult<u64> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.dataset_raw_size(&path);
        }
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        Ok(Self::raw_size_of(info))
    }

    /// Everything a zero-copy view of `name` needs to know: the map of the
    /// file that owns the dataset, where in that file the dataset's image
    /// lies, and how its elements are stored.
    ///
    /// Facts only — whether they add up to a view `T` may be handed is
    /// [`crate::mapped::view`]'s decision, which is the single place that
    /// weighs them. Resolved in the file that owns the dataset, so a name
    /// crossing an external link answers with the target's map and the
    /// target's addresses rather than this file's.
    #[cfg(feature = "mmap")]
    pub(crate) fn dataset_view_source(&mut self, name: &str) -> IoResult<DatasetViewSource> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.dataset_view_source(&path);
        }
        let base = self.handle.base();
        let map = self.handle.map_snapshot();
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let len = Self::raw_size_of(info);
        let storage = match &info.layout {
            // An external file list overrides contiguous storage: the layout
            // still says `Contiguous`, but the bytes are in other files
            // (H5Dlayout.c swaps the storage ops out whenever the message is
            // present), so nothing in this map holds them.
            DataLayoutMessage::Contiguous { .. } if !info.external_files.is_empty() => {
                ViewStorage::Elsewhere("its raw data is in external data files")
            }
            DataLayoutMessage::Contiguous { address, .. } if *address == UNDEF_ADDR => {
                ViewStorage::Unallocated
            }
            DataLayoutMessage::Contiguous { address, .. } => {
                let offset = address.checked_add(base).ok_or_else(|| {
                    crate::io::IoError::InvalidState(format!(
                        "dataset '{name}' claims raw data at {address}, which overflows \
                         past the userblock at {base}"
                    ))
                })?;
                ViewStorage::Contiguous { offset, len }
            }
            DataLayoutMessage::Compact { .. } => {
                ViewStorage::Elsewhere("its raw data is compact, stored inside the object header")
            }
            DataLayoutMessage::ChunkedV3 { .. } | DataLayoutMessage::ChunkedV4 { .. } => {
                ViewStorage::Elsewhere("it is chunked")
            }
            DataLayoutMessage::Virtual { .. } => {
                ViewStorage::Elsewhere("it is virtual, mapped from other datasets")
            }
        };
        Ok(DatasetViewSource {
            map,
            storage,
            datatype: info.datatype.clone(),
            dims: info.dataspace.dims.clone(),
        })
    }

    /// Read the raw bytes of a dataset.
    pub fn read_dataset_raw(&mut self, name: &str) -> IoResult<Vec<u8>> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_dataset_raw(&path);
        }
        let (datatype, total) = self.raw_size_and_datatype(name)?;
        read_image_into_new(total as usize, |data| {
            self.read_dataset_raw_into_unconverted(name, data, ReadDst::Fresh)?;
            Self::apply_post_filter_conversion(data, &datatype)
        })
    }

    /// Read the full raw dataset image into a caller-provided buffer.
    ///
    /// `out.len()` must equal the dataset's logical byte size
    /// (`product(dims) * element_size`); otherwise an error is returned. This
    /// is the no-allocation counterpart of [`read_dataset_raw`](Self::read_dataset_raw):
    /// the bytes are read straight into `out`, making it the zero-copy entry
    /// point for reading directly into a pinned/registered host buffer for an
    /// H2D transfer.
    pub fn read_dataset_raw_into(&mut self, name: &str, out: &mut [u8]) -> IoResult<()> {
        self.read_dataset_raw_into_dst(name, out, ReadDst::Reused)
    }

    /// [`read_dataset_raw_into`](Self::read_dataset_raw_into) with the
    /// caller's destination fact made explicit, for internal callers whose
    /// buffer is a fresh allocation rather than a kept one (the allocating
    /// wrappers in `dataset.rs` / `swmr.rs`).
    pub(crate) fn read_dataset_raw_into_dst(
        &mut self,
        name: &str,
        out: &mut [u8],
        dst: ReadDst,
    ) -> IoResult<()> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_dataset_raw_into_dst(&path, out, dst);
        }
        let (datatype, total) = self.raw_size_and_datatype(name)?;
        if out.len() as u64 != total {
            return Err(crate::io::IoError::InvalidState(format!(
                "read_dataset_raw_into: buffer is {} bytes but dataset needs {}",
                out.len(),
                total
            )));
        }
        self.read_dataset_raw_into_unconverted(name, out, dst)?;
        Self::apply_post_filter_conversion(out, &datatype)?;
        Ok(())
    }

    /// Fill `out` with the full raw dataset image, before the post-filter
    /// datatype conversion. The single owner of read-destination semantics for
    /// full reads: it fully defines every byte of `out` (reading allocated data
    /// straight in, pre-filling chunked or never-written regions with the tiled
    /// fill value), so callers supply only a correctly-sized buffer. Both the
    /// allocating `read_dataset_raw` and the zero-copy `read_dataset_raw_into`
    /// wrap it and apply the conversion exactly once.
    ///
    /// `out.len()` must equal `product(dims) * element_size`. `dst` is the
    /// caller's destination fact ([`ReadDst`]): whether `out` is a fresh
    /// allocation or a buffer the caller keeps across reads, which decides
    /// whether a mapped contiguous read is priced at the cold or the warm row.
    fn read_dataset_raw_into_unconverted(
        &mut self,
        name: &str,
        out: &mut [u8],
        dst: ReadDst,
    ) -> IoResult<()> {
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;

        // Clone to avoid borrow conflict with &mut self in read methods.
        let layout = info.layout.clone();
        let pipeline = info.filter_pipeline.clone();
        let fill_value = info.fill_value.clone();
        let external_files = info.external_files.clone();

        match &layout {
            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
                let prefix = self.extfile_prefix_in_force(name);
                read_external_file_bytes(&external_files, prefix.as_deref(), 0, out)?;
            }
            DataLayoutMessage::Contiguous { address, .. } => {
                if *address == UNDEF_ADDR {
                    // Never-written contiguous data reads back as the fill value.
                    fill_tiled_into(out, fill_value.as_deref());
                } else {
                    // Read exactly the logical image straight into `out`.
                    self.handle.read_exact_at_into(*address, out, dst)?;
                }
            }
            DataLayoutMessage::Compact { data } => {
                let n = out.len().min(data.len());
                out[..n].copy_from_slice(&data[..n]);
                if n < out.len() {
                    fill_tiled_into(&mut out[n..], fill_value.as_deref());
                }
            }
            DataLayoutMessage::ChunkedV3 {
                chunk_dims,
                b_tree_address,
            } => {
                // The layout's chunk_dims include the element size as the
                // trailing dimension. Strip it for chunk indexing. The chunk
                // read defines every byte of `out`, filling whatever no chunk
                // covers.
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                self.read_chunked_btree_v1(
                    name,
                    real_chunk_dims,
                    *b_tree_address,
                    ChunkReadRequest {
                        pipeline: pipeline.as_ref(),
                        target: ChunkTarget::Full,
                        fill_value: fill_value.as_deref(),
                        dst,
                    },
                    out,
                )?;
            }
            DataLayoutMessage::ChunkedV4 {
                chunk_dims,
                index_address,
                index_type,
                earray_params,
                single_chunk_filter,
                ..
            } => {
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                self.read_chunked_v4(
                    name,
                    real_chunk_dims,
                    ChunkIndexDesc {
                        index_type: *index_type,
                        index_address: *index_address,
                        earray_params: earray_params.as_ref(),
                        single_chunk_filter: *single_chunk_filter,
                    },
                    ChunkReadRequest {
                        pipeline: pipeline.as_ref(),
                        target: ChunkTarget::Full,
                        fill_value: fill_value.as_deref(),
                        dst,
                    },
                    out,
                )?;
            }
            DataLayoutMessage::Virtual { .. } => {
                fill_tiled_into(out, fill_value.as_deref());
                self.read_virtual_into(name, out, 0)?;
            }
        }
        Ok(())
    }

    /// Set every virtual dataset's extent from the sources its unlimited
    /// mappings can reach — `H5D__virtual_set_extent_unlim` (H5Dvirtual.c),
    /// which libhdf5 runs when it opens such a dataset.
    ///
    /// INVARIANT: a virtual dataset's `dataspace.dims` are the extent its
    /// available sources give it. This is the single owner of that
    /// resolution: the dims a VDS with an unlimited mapping reports are not
    /// the ones its dataspace message stores, and every shape query, read and
    /// slice bound must see the same value, so the resolved extent is stamped
    /// in once — here, immediately after a catalog is built — rather than
    /// recomputed per call. The per-mapping clip sizes the extent came from
    /// are kept beside it in
    /// [`virtual_resolution`](DatasetReadInfo::virtual_resolution) so a read
    /// walks exactly the sources the extent was derived from.
    ///
    /// A source that cannot be opened contributes a clip size of 0, never an
    /// error: a virtual dataset whose sources are not written yet is legal,
    /// and reads back as the fill value (upstream's "clip_size = 0" arm when
    /// `H5D__virtual_open_source_dset` leaves the dataset closed).
    ///
    /// The default view is assumed throughout — `H5D_VDS_LAST_AVAILABLE` is
    /// `H5D_ACS_VDS_VIEW_DEF`, and `H5Pset_virtual_view` sets a *dataset
    /// access* property that is never stored in the file, so a reader opening
    /// a file it did not create always sees the default.
    fn resolve_virtual_extents(&mut self) -> IoResult<()> {
        let targets: Vec<usize> = self
            .datasets
            .iter()
            .enumerate()
            .filter(|(_, d)| d.virtual_mappings.is_some())
            .map(|(i, _)| i)
            .collect();
        for i in targets {
            // The catalog is freshly built, so `dataspace.dims` is still the
            // extent the dataspace message stores. Record it before the
            // resolution replaces it: that is what a later open under other
            // access properties has to resolve from.
            let stored = self.datasets[i].dataspace.dims.clone();
            self.datasets.entry_mut(i).virtual_stored_dims = Some(stored.clone());
            self.resolve_virtual_extent_of(i, &stored)?;
        }
        // This resolution belongs to no dataset open: libhdf5 runs its
        // equivalent from `H5D__virtual_init` at `H5Dopen` (H5Dvirtual.c:2178),
        // where the open that asked for it holds the source, while this one
        // runs at `H5Fopen` and at a SWMR refresh. Whatever it opened is
        // therefore unowned the moment it is done — and libhdf5 measured on
        // the same file has no source file open after `H5Fopen` either.
        self.release_closed_virtual_sources();
        Ok(())
    }

    /// Resolve one virtual dataset's extent from its *stored* dims under the
    /// [`DatasetAccess`] in force for it, and stamp both the extent and the
    /// per-mapping resolutions in.
    fn resolve_virtual_extent_of(&mut self, i: usize, stored: &[u64]) -> IoResult<()> {
        let Some(mappings) = self.datasets[i].virtual_mappings.clone() else {
            return Ok(());
        };
        let vds = self.datasets[i].name.clone();
        let access = self.access_in_force(&vds);
        let (resolution, dims) =
            self.resolve_one_virtual_extent(&vds, &mappings, stored, &access)?;
        let entry = self.datasets.entry_mut(i);
        entry.dataspace.dims = dims;
        entry.virtual_resolution = Some(resolution);
        Ok(())
    }

    /// The dataset-access properties in force for `name` (already canonical),
    /// libhdf5's defaults when no open has named others.
    fn access_in_force(&self, canonical: &str) -> DatasetAccess {
        self.dataset_access
            .get(canonical)
            .map(|e| e.access.clone())
            .unwrap_or_default()
    }

    /// Open `name` under `access`: put those properties in force and
    /// re-resolve its extent under them, or — when the dataset already has a
    /// live handle — join that open and drop `access` on the floor.
    ///
    /// First open wins, which is `H5D_open`'s own rule. Only the open that
    /// finds no shared info for the dataset runs `H5D__open_oid(dataset,
    /// dapl_id)` and so reaches `H5D__virtual_init`, where the view and the
    /// printf gap are read out of the dapl into the *shared* layout storage
    /// (H5Dvirtual.c:2178-2188); an open that finds the dataset in
    /// `H5FO_opened` just points at that shared info and increments its
    /// count, never looking at its own dapl at all (H5Dint.c:1496-1500,
    /// :1523-1528). The shared info goes away with the last handle, so the
    /// next open after that resolves afresh. Measured against libhdf5 1.14.6
    /// and 2.0.0: a second open of a printf-gap VDS with a different gap
    /// reports the first open's extent and reads the first open's data —
    /// even through a second `H5Fopen` of the same file — and only once
    /// every handle is closed does a new open see its own gap.
    ///
    /// The one exception to "first open wins" is the external file prefix,
    /// which the joining open is not allowed to disagree about:
    /// `H5D__open_name` compares its own expanded prefix against the open
    /// dataset's and fails the open when they differ (H5Dint.c:1533-1545).
    /// Expanded, so two opens differing only in a property
    /// `HDF5_EXTFILE_PREFIX` shadows still agree — measured under libhdf5
    /// 1.14.6 and 2.0.0: with that variable set, an open naming no prefix
    /// joins one that named a directory, and without it the same pair is
    /// refused.
    ///
    /// Returns the token that keeps the open alive; the caller hands it to
    /// the dataset handle it builds. A name no dataset in this file answers
    /// to takes nothing and returns `None`.
    fn apply_dataset_access(
        &mut self,
        name: &str,
        access: &DatasetAccess,
    ) -> IoResult<Option<DatasetOpenToken>> {
        let canonical = self.canonical_path(name);
        let Some(i) = self.datasets.position(&canonical) else {
            return Ok(None);
        };
        if let Some(open) = self
            .dataset_access
            .get(&canonical)
            .and_then(|e| e.open.upgrade())
        {
            let in_force = self.extfile_prefix_of(&self.access_in_force(&canonical));
            if in_force != self.extfile_prefix_of(access) {
                return Err(crate::io::IoError::InvalidState(format!(
                    "dataset {canonical:?} is already open under a different external file \
                     prefix, and libhdf5 refuses to join an open that disagrees about one"
                )));
            }
            return Ok(Some(open));
        }
        let token: DatasetOpenToken = std::sync::Arc::new(());
        let unchanged = &self.access_in_force(&canonical) == access;
        let access = access.clone();
        self.dataset_access.insert(
            canonical,
            AccessInForce {
                access,
                open: std::sync::Arc::downgrade(&token),
            },
        );
        if !unchanged {
            if let Some(stored) = self.datasets[i].virtual_stored_dims.clone() {
                // A source may be a virtual dataset in this same file, and the
                // access propagates to it (H5Dvirtual.c:2224-2226), so this can
                // re-enter; the depth counter is the same cycle guard the
                // open-time resolution uses.
                VirtualResolveDepth::enter(|| self.resolve_virtual_extent_of(i, &stored))?;
            }
        }
        Ok(Some(token))
    }

    /// The directory an external file list's stored names are joined against
    /// under `access` — `H5D__build_file_prefix(dset, H5F_PREFIX_EFILE)`
    /// (H5Dint.c:1084-1090), whose answer libhdf5 keeps in
    /// `dset->shared->extfile_prefix`.
    fn extfile_prefix_of(&self, access: &DatasetAccess) -> Option<PathBuf> {
        resolve_extfile_prefix(access.efile_prefix_value(), &self.source_dir)
    }

    /// The external file prefix in force for the open dataset `name` — the
    /// one the open that is still holding it named, not whatever a later
    /// caller might have asked for
    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
    fn extfile_prefix_in_force(&self, name: &str) -> Option<PathBuf> {
        let canonical = self.canonical_path(name);
        self.extfile_prefix_of(&self.access_in_force(&canonical))
    }

    /// [`resolve_virtual_extents`](Self::resolve_virtual_extents) for one
    /// dataset: the per-mapping resolutions and the extent they imply.
    fn resolve_one_virtual_extent(
        &mut self,
        vds: &str,
        mappings: &VirtualMappingList,
        curr_dims: &[u64],
        access: &DatasetAccess,
    ) -> IoResult<(Vec<MappingResolution>, Vec<u64>)> {
        let rank = curr_dims.len();
        let mut resolution = Vec::with_capacity(mappings.mappings.len());
        let mut new_dims: Vec<Option<u64>> = vec![None; rank];
        // `H5D_virtual_update_min_dims`: whatever the unlimited dimension
        // resolves to, the extent must still hold every bounded mapping.
        let mut min_dims = vec![0u64; rank];
        // `H5S_hyper_get_clip_extent_match`'s `incl_trail`: a
        // `H5D_VDS_FIRST_MISSING` view stops where the trailing partial
        // block would begin (H5Dvirtual.c:1447-1451).
        let incl_trail = access.view() == VirtualView::FirstMissing;
        // Where two mappings disagree about the unlimited dimension,
        // `H5D_VDS_FIRST_MISSING` takes the smallest clip and
        // `H5D_VDS_LAST_AVAILABLE` the largest (H5Dvirtual.c:1662-1667).
        let take_clip = |slot: &mut Option<u64>, clip: u64| {
            if slot.is_none_or(|d| if incl_trail { clip < d } else { clip > d }) {
                *slot = Some(clip);
            }
        };

        for m in &mappings.mappings {
            let unlim_virtual = m.virtual_selection.unlim_dim();
            let res = match (unlim_virtual, m.source_selection.unlim_dim()) {
                (Some(vd), Some(sd)) => {
                    let source_clip = self
                        .virtual_source_dims(vds, m, access)
                        .ok()
                        .flatten()
                        .and_then(|d| d.get(sd).copied())
                        .unwrap_or(0);
                    let virtual_clip = match (
                        regular_hyperslab(&m.virtual_selection),
                        regular_hyperslab(&m.source_selection),
                    ) {
                        // `H5S_hyper_get_clip_extent_match`: how many slices
                        // the source supplies, then the virtual extent that
                        // covers exactly that many. Its `incl_trail`
                        // argument is `view == H5D_VDS_FIRST_MISSING`
                        // (H5Dvirtual.c:1447-1451).
                        (Some(v), Some(sr)) => {
                            v.clip_extent(sr.num_slices(source_clip), incl_trail)
                        }
                        _ => 0,
                    };
                    take_clip(&mut new_dims[vd], virtual_clip);
                    MappingResolution::Unlimited {
                        virtual_clip,
                        source_clip,
                    }
                }
                // Unlimited virtual selection, limited source selection:
                // the printf shape, where the successive blocks of the
                // virtual selection come from successively-named source
                // datasets.
                (Some(vd), None) => {
                    let (blocks, present) = self.printf_blocks_present(vds, m, access);
                    let virtual_clip = match (blocks, regular_hyperslab(&m.virtual_selection)) {
                        // `H5D__virtual_set_extent_unlim`'s "check for no
                        // datasets" arm, which is 0 under either view
                        // (H5Dvirtual.c:1623-1626).
                        (0, _) | (_, None) => 0,
                        // The extent ends just past the last block that has
                        // a source under `H5D_VDS_LAST_AVAILABLE`, and where
                        // the first missing block starts under
                        // `H5D_VDS_FIRST_MISSING` (H5Dvirtual.c:1630-1653).
                        (n, Some(r)) => match access.view() {
                            VirtualView::LastAvailable => {
                                let last = r.unlim_block(n - 1);
                                last.start[vd] + last.block[vd]
                            }
                            VirtualView::FirstMissing => r.unlim_block(n).start[vd],
                        },
                    };
                    take_clip(&mut new_dims[vd], virtual_clip);
                    MappingResolution::Printf { blocks, present }
                }
                _ => MappingResolution::Bounded,
            };
            if let Some((_, hi)) = m.virtual_selection.bounds() {
                for (d, &e) in hi.iter().enumerate().take(rank) {
                    if Some(d) != unlim_virtual && e + 1 > min_dims[d] {
                        min_dims[d] = e + 1;
                    }
                }
            }
            resolution.push(res);
        }

        let dims = (0..rank)
            .map(|d| match new_dims[d] {
                Some(v) => v.max(min_dims[d]),
                None => curr_dims[d],
            })
            .collect();
        Ok((resolution, dims))
    }

    /// The extent of the source dataset one mapping names, or `None` when it
    /// cannot be reached — `H5D__virtual_open_source_dset` leaving the source
    /// closed, which upstream reads as "no data there yet" rather than an
    /// error.
    fn virtual_source_dims(
        &mut self,
        vds: &str,
        m: &VirtualMapping,
        access: &DatasetAccess,
    ) -> IoResult<Option<Vec<u64>>> {
        let m = built_names(m, 0)?;
        Ok(self.source_dims(vds, &m.source_file_name, &m.source_dset_name, access))
    }

    /// A printf mapping's `first_missing` and the blocks below it that
    /// actually have a source — upstream's search loop in
    /// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c:1519-1614), which stops
    /// at the first block whose source cannot be opened and looks
    /// [`DatasetAccess::virtual_printf_gap`] blocks past it before giving up.
    ///
    /// The loop bound is upstream's `j <= printf_gap + first_missing`
    /// rearranged so a large gap cannot overflow the sum: `first_missing` is
    /// never above `j` when the test runs, because it only ever becomes the
    /// *previous* `j` plus one.
    fn printf_blocks_present(
        &mut self,
        vds: &str,
        m: &VirtualMapping,
        access: &DatasetAccess,
    ) -> (u64, Vec<u64>) {
        let gap = access.effective_printf_gap();
        let mut first_missing = 0u64;
        let mut present = Vec::new();
        let mut j = 0u64;
        while j - first_missing <= gap {
            let Ok(built) = built_names(m, j) else {
                break;
            };
            if self
                .source_dims(
                    vds,
                    &built.source_file_name,
                    &built.source_dset_name,
                    access,
                )
                .is_some()
            {
                first_missing = j + 1;
                present.push(j);
            }
            j += 1;
        }
        (first_missing, present)
    }

    /// The extent of one named source dataset, or `None` when the file or
    /// the dataset in it cannot be opened.
    ///
    /// `access` is the virtual dataset's own: `H5D__virtual_init` copies the
    /// dapl into the layout as `source_dapl` (H5Dvirtual.c:2224-2226) and
    /// every source is opened with it (H5Dvirtual.c:901-902), so a source
    /// that is itself a virtual dataset resolves under the same view and
    /// printf gap.
    fn source_dims(
        &mut self,
        vds: &str,
        file_name: &str,
        dset_name: &str,
        access: &DatasetAccess,
    ) -> Option<Vec<u64>> {
        let dset_name = dset_name.trim_start_matches('/');
        if file_name == "." {
            self.apply_dataset_access(dset_name, access).ok()?;
            return self
                .dataset_info_local(dset_name)
                .map(|i| i.dataspace.dims.clone());
        }
        let reader = self.vds_source_file(vds, file_name)?;
        reader.apply_dataset_access(dset_name, access).ok()?;
        reader
            .dataset_info(dset_name)
            .map(|i| i.dataspace.dims.clone())
    }

    /// Fill `out` (shaped like the virtual dataset's own extent) by
    /// stitching each mapping's source bytes in order (`H5D__virtual_read`,
    /// H5Dvirtual.c). `out` must already be pre-filled with the tiled fill
    /// value — every element no mapping covers is left exactly as the
    /// caller filled it. Mappings apply in list order, so a later mapping's
    /// bytes win over an earlier one's on overlap, exactly like the C
    /// reader; an unlimited or printf mapping has already been replaced by
    /// the concrete mappings its open-time resolution made it
    /// (`H5D_VDS_LAST_AVAILABLE`, the default view — see
    /// [`Hdf5Reader::resolve_virtual_extents`]).
    ///
    /// A mapping whose source cannot be opened — the file is absent, or the
    /// dataset is not in it — contributes nothing and leaves its virtual
    /// region at the fill value, rather than failing the read.
    /// `H5D__virtual_open_source_dset` treats both as "no data there yet":
    /// it asks `H5F_prefix_open_file` to *try* the file and accepts a null
    /// one, and clears the error stack when the dataset is missing
    /// (H5Dvirtual.c:877-909); `H5D__virtual_read_one` then performs I/O
    /// "only ... if there is a projected memory space, otherwise there were
    /// no elements in the projection or the source dataset could not be
    /// opened" (H5Dvirtual.c:2661-2665).
    ///
    /// `depth` counts virtual-dataset nesting — a mapping whose source is
    /// itself a virtual dataset, possibly in another file — so a crafted
    /// cyclic mapping chain fails cleanly instead of recursing until the
    /// stack overflows.
    fn read_virtual_into(&mut self, name: &str, out: &mut [u8], depth: usize) -> IoResult<()> {
        if depth >= MAX_VIRTUAL_DEPTH {
            return Err(crate::io::IoError::InvalidState(format!(
                "dataset {name:?}: virtual dataset mapping nests {MAX_VIRTUAL_DEPTH} levels \
                 deep, aborting (possible cyclic mapping)"
            )));
        }
        let info = self
            .dataset_info(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let Some(mappings) = info.virtual_mappings.clone() else {
            // No mapping list written yet: every element is unmapped, and
            // `out` is already the fill value the caller pre-filled it with.
            return Ok(());
        };
        // Every unlimited mapping is replaced by the concrete one its
        // open-time resolution makes it, so the walk below only ever sees
        // bounded selections.
        let resolution = info.virtual_resolution.clone().unwrap_or_default();
        let mappings = concrete_virtual_mappings(&mappings, &resolution)?;
        // The same properties the extent resolved under reach every source
        // (H5Dvirtual.c:2224-2226, :901-902), so a source that is itself a
        // virtual dataset is read the same way this one is.
        let canonical = self.canonical_path(name);
        let access = self.access_in_force(&canonical);

        for mapping in &mappings {
            let virtual_sel = mapping.virtual_selection.resolve(&dims).map_err(|e| {
                crate::io::IoError::InvalidState(format!(
                    "dataset {name:?}: virtual mapping's virtual selection is not \
                     supported: {e}"
                ))
            })?;
            if virtual_sel.runs.is_empty() {
                continue;
            }

            let source_name = mapping.source_dset_name.trim_start_matches('/');

            if mapping.source_file_name == "." {
                // `H5D__virtual_open_source_dset` opens the source for the
                // read and `H5D__virtual_reset_source_dset` closes it again,
                // so this open holds nothing past the mapping — dropping the
                // token is what that close does.
                self.apply_dataset_access(source_name, &access)?;
                let Some(src_dims) = self
                    .dataset_info(source_name)
                    .map(|i| i.dataspace.dims.clone())
                else {
                    continue;
                };
                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
                    crate::io::IoError::InvalidState(format!(
                        "dataset {name:?}: virtual mapping's source selection is not \
                         supported: {e}"
                    ))
                })?;
                copy_matched_selections(
                    |s, c, buf| {
                        // `buf` is `copy_matched_selections`' fresh per-box
                        // buffer, never the virtual dataset's own `out`.
                        self.read_slice_into_unconverted(
                            source_name,
                            s,
                            c,
                            buf,
                            depth + 1,
                            ReadDst::Fresh,
                        )
                    },
                    &source_sel,
                    &virtual_sel,
                    element_size,
                    out,
                )?;
            } else {
                let Some(src_reader) = self.vds_source_file(&canonical, &mapping.source_file_name)
                else {
                    continue;
                };
                src_reader.apply_dataset_access(source_name, &access)?;
                let Some(src_dims) = src_reader
                    .dataset_info(source_name)
                    .map(|i| i.dataspace.dims.clone())
                else {
                    continue;
                };
                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
                    crate::io::IoError::InvalidState(format!(
                        "dataset {name:?}: virtual mapping's source selection is not \
                         supported: {e}"
                    ))
                })?;
                copy_matched_selections(
                    |s, c, buf| {
                        // As above: `buf` is a fresh per-box buffer.
                        src_reader.read_slice_into_unconverted(
                            source_name,
                            s,
                            c,
                            buf,
                            depth + 1,
                            ReadDst::Fresh,
                        )
                    },
                    &source_sel,
                    &virtual_sel,
                    element_size,
                    out,
                )?;
            }
        }
        Ok(())
    }

    /// Apply the post-filter datatype conversion (libhdf5's `H5T_convert`
    /// step) to a fully-decoded output buffer.
    ///
    /// For N-bit / reduced-precision `FixedPoint` datatypes the filter
    /// pipeline leaves the significant value occupying `bit_precision` bits
    /// at `bit_offset` within each element, zero-filled and not
    /// sign-extended. This rewrites every element so the value occupies the
    /// whole element at bit offset 0, sign-extended when signed. It is a
    /// no-op for ordinary full-width datatypes.
    fn apply_post_filter_conversion(buffer: &mut [u8], datatype: &DatatypeMessage) -> IoResult<()> {
        use crate::format::nbit_scaleoffset::{
            apply_datatype_conversion, datatype_needs_bit_conversion,
        };
        if datatype_needs_bit_conversion(datatype) {
            apply_datatype_conversion(buffer, datatype)?;
        }
        Ok(())
    }

    /// Re-read the superblock and dataset metadata for SWMR.
    ///
    /// Call this periodically to pick up new data written by a concurrent
    /// SWMR writer. The superblock is re-read to get the latest EOF, then
    /// the root group is re-scanned for updated dataset headers (which may
    /// contain updated dataspace dimensions and chunk index addresses).
    pub fn refresh(&mut self) -> IoResult<()> {
        // Whatever the handle reads from must cover the file as the SWMR
        // writer has left it: a memory map taken at open ends where the file
        // ended then, so it is retaken before a byte of the new metadata is
        // decoded. Nothing happens for a handle reading through `pread`.
        self.handle.refresh_read_source();

        // Re-read superblock to get latest EOF and root group address.
        let sb_buf = self.handle.read_at_most(0, 256)?;

        // Only v2/v3 superblocks support SWMR refresh
        let sb = SuperblockV2V3::decode(&sb_buf)?;

        let ctx = FormatContext {
            sizeof_addr: sb.sizeof_offsets,
            sizeof_size: sb.sizeof_lengths,
        };

        // The superblock extension can also have changed under SWMR (a new
        // free-space or shared-message table), so re-read it before the walk.
        let (meta, ext) = Self::read_extension_and_meta(
            &mut self.handle,
            ctx,
            self.meta.btree,
            sb.superblock_extension_address,
        )?;

        // Re-read root group object header, following continuation blocks.
        let root_header = Self::read_object_header_full(
            &mut self.handle,
            &meta,
            sb.root_group_object_header_address,
        )?;

        // Re-scan datasets, group attributes, group paths, and link records.
        let catalog = Self::build_catalog(
            &mut self.handle,
            &meta,
            Some(&root_header),
            sb.root_group_object_header_address,
            None,
        )?;

        // Root link storage from the freshly re-read header, the same way
        // `open_v2v3` derives it at open time — SWMR refresh is v2/v3-only,
        // so there is no symbol-table scratch-pad to fall back to here either.
        let root_link_storage = describe_link_storage(Some(&root_header), &meta.ctx, None);

        self._eof = sb.end_of_file_address;
        self.meta = meta;
        self.ext = ext;
        self.object_paths = catalog.object_paths(sb.root_group_object_header_address);
        self.datasets = DatasetTable::new(catalog.datasets);
        self.unreadable = catalog.unreadable;
        self.root_link_storage = root_link_storage;
        self.group_attributes = catalog.group_attributes;
        self.group_link_storage = catalog.group_link_storage;
        self.group_paths = catalog.group_paths;
        self.group_aliases = catalog.group_aliases;
        self.links = catalog.links;
        self.datatypes = catalog.datatypes;
        // The catalog is freshly built, so every virtual dataset's resolved
        // extent went with the old one — a SWMR refresh is exactly when a
        // source may have grown.
        self.resolve_virtual_extents()?;

        Ok(())
    }

    /// The dataset at `pos`'s chunk index, from the cache when it already
    /// holds one decoded from `index_address`, otherwise by running `decode`
    /// once and keeping what it returns.
    ///
    /// The single owner of the chunk-index cache: nothing else reads or
    /// writes it. Every chunked read of a dataset asks the same question of
    /// the same on-disk structure — a thousand small slice reads re-walked
    /// the fixed array a thousand times — and only a change to the catalog
    /// entry can change the answer, which drops the entry's cache
    /// (`DatasetTable::entry_mut`, and a SWMR refresh rebuilding the table).
    fn decoded_chunk_index<F>(
        &mut self,
        pos: usize,
        index_address: u64,
        decode: F,
    ) -> IoResult<std::sync::Arc<DecodedChunkIndex>>
    where
        F: FnOnce(&mut Self) -> IoResult<DecodedChunkIndex>,
    {
        if let Some(hit) = self.datasets.chunk_index(pos, index_address) {
            return Ok(std::sync::Arc::clone(hit));
        }
        let decoded = decode(self)?;
        Ok(self.datasets.cache_chunk_index(pos, index_address, decoded))
    }

    /// Read chunked dataset data by walking the chunk index.
    ///
    /// `desc` bundles the version-4 chunk-index descriptor extracted from the
    /// data-layout message (kind, address, and per-kind parameters), so this
    /// entry point takes one descriptor rather than a long parameter list.
    ///
    /// Scatters only; `output` must already be sized to the target extent.
    /// Every byte of it is defined before this returns `Ok`: what no chunk
    /// covers is filled with the tiled fill value by
    /// [`place_chunk_jobs`], the one exit every branch here takes.
    fn read_chunked_v4(
        &mut self,
        name: &str,
        chunk_dims: &[u64],
        desc: ChunkIndexDesc<'_>,
        req: ChunkReadRequest,
        output: &mut [u8],
    ) -> IoResult<()> {
        let ChunkReadRequest {
            pipeline, target, ..
        } = req;
        let ChunkIndexDesc {
            index_type,
            index_address,
            earray_params,
            single_chunk_filter,
        } = desc;
        let pos = self.dataset_position(name)?;
        let info = &self.datasets[pos];
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;

        match index_type {
            data_layout::ChunkIndexType::SingleChunk => {
                // Single chunk: the index_address IS the chunk address
                let total_size: u64 = saturating_byte_len(&dims, element_size);
                let geo = ChunkOutputGeometry {
                    dims: &dims,
                    chunk_dims,
                    element_size,
                };
                if index_address == UNDEF_ADDR || total_size == 0 {
                    // Unallocated single chunk: nothing to read, so the empty
                    // plan makes the whole output fill.
                    return place_chunk_jobs(
                        &self.handle,
                        Vec::new(),
                        &[],
                        req,
                        &geo,
                        None,
                        output,
                    );
                }
                // A filtered single chunk records its exact on-disk size and
                // per-chunk filter mask in the layout message. Use them to
                // read precisely the stored bytes and to skip the filters the
                // mask marks as not applied. Without those params
                // (older/edge layouts) fall back to the read-extra-and-inflate
                // heuristic with the full pipeline.
                let job = match (pipeline, single_chunk_filter) {
                    (Some(_), Some(scf)) => ChunkReadJob {
                        addr: index_address,
                        len: scf.nbytes as usize,
                        at_most: false,
                        mask: scf.filter_mask,
                    },
                    (Some(_), None) => ChunkReadJob {
                        addr: index_address,
                        len: total_size.saturating_mul(2) as usize,
                        at_most: true,
                        mask: 0,
                    },
                    (None, _) => ChunkReadJob {
                        addr: index_address,
                        len: total_size as usize,
                        at_most: false,
                        mask: 0,
                    },
                };
                // The lone chunk spans the whole dataset; place it respecting
                // the dataset extent (Full) or the selection (Slice). This
                // index type has no on-disk structure to decode — the layout
                // message is the index — but it is still recorded through the
                // one owner, so that its chunk's image reaches the cache by
                // the same route every other index type's chunks do.
                let index = self.decoded_chunk_index(pos, index_address, |_| {
                    Ok(DecodedChunkIndex::new(
                        vec![(job.addr, job.len as u64, job.mask)],
                        vec![0u64; dims.len()],
                    ))
                })?;
                place_chunk_jobs(
                    &self.handle,
                    vec![Some(job)],
                    &index.coords,
                    req,
                    &geo,
                    Some(&index.images),
                    output,
                )
            }
            data_layout::ChunkIndexType::Implicit => {
                self.read_chunked_implicit(name, chunk_dims, index_address, req, output)
            }
            data_layout::ChunkIndexType::FixedArray => {
                self.read_chunked_fixed_array(name, chunk_dims, index_address, req, output)
            }
            data_layout::ChunkIndexType::BTreeV2 => {
                self.read_chunked_btree_v2(name, chunk_dims, index_address, req, output)
            }
            data_layout::ChunkIndexType::ExtensibleArray => {
                let params = earray_params.ok_or_else(|| {
                    crate::io::IoError::InvalidState("missing earray params".into())
                })?;

                if index_address == UNDEF_ADDR {
                    // Unallocated: the empty plan makes the whole output fill.
                    let geo = ChunkOutputGeometry {
                        dims: &dims,
                        chunk_dims,
                        element_size,
                    };
                    return place_chunk_jobs(
                        &self.handle,
                        Vec::new(),
                        &[],
                        req,
                        &geo,
                        None,
                        output,
                    );
                }

                // Total slot count of the index grid. The maximum extent
                // decides the multipliers (libhdf5 max_down_chunks); an
                // unlimited dimension 0 is bounded by the current extent for
                // this read — a slot beyond it (written before a shrink) is
                // not visible.
                let max_dims = self.datasets[pos].dataspace.max_dims.clone();

                // Chunks are placed N-dimensionally: each slot decodes
                // (row-major, against the index grid) to chunk-grid
                // coordinates, so sub-frame chunks (a chunk smaller than a
                // full frame) land correctly.
                let rank = dims.len();
                let index = self.decoded_chunk_index(pos, index_address, |reader| {
                    let grid =
                        crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
                    let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
                    let mut entries = reader.collect_ea_chunk_entries(
                        index_address,
                        params,
                        &dims,
                        max_dims.as_deref(),
                        chunk_dims,
                        element_size,
                    )?;
                    entries.truncate(std::cmp::min(chunks_total as usize, entries.len()));
                    let coords = crate::io::chunk_grid::coords_table(
                        &dims,
                        max_dims.as_deref(),
                        chunk_dims,
                        entries.len(),
                    )?;
                    Ok(DecodedChunkIndex::new(entries, coords))
                })?;
                let chunk_entries = &index.entries;
                let slot_coords = &index.coords;
                let chunk_coords = |i: usize| -> &[u64] { &slot_coords[i * rank..(i + 1) * rank] };

                // Build one read job per chunk (no I/O yet), then read +
                // decompress them together (in parallel where positioned reads
                // are race-free), then scatter serially. Filtered chunks record
                // their exact on-disk size, so read exactly that; unfiltered
                // chunks read at-most since the entry size can exceed the file
                // tail. Skip conditions differ between the two, so build jobs
                // per branch.
                let jobs: Vec<Option<ChunkReadJob>> = if pipeline.is_some() {
                    let file_size = self.handle.file_size()?;
                    chunk_entries
                        .iter()
                        .enumerate()
                        .map(|(i, &(addr, nbytes, mask))| {
                            if addr == UNDEF_ADDR
                                || nbytes == 0
                                || addr >= file_size
                                || nbytes > file_size
                                || !target.overlaps(chunk_coords(i), chunk_dims)
                            {
                                None
                            } else {
                                Some(ChunkReadJob {
                                    addr,
                                    len: nbytes as usize,
                                    at_most: false,
                                    mask,
                                })
                            }
                        })
                        .collect()
                } else {
                    chunk_entries
                        .iter()
                        .enumerate()
                        .map(|(i, &(addr, nbytes, _))| {
                            if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(i), chunk_dims) {
                                None
                            } else {
                                Some(ChunkReadJob {
                                    addr,
                                    len: nbytes as usize,
                                    at_most: true,
                                    mask: 0,
                                })
                            }
                        })
                        .collect()
                };

                let geo = ChunkOutputGeometry {
                    dims: &dims,
                    chunk_dims,
                    element_size,
                };
                place_chunk_jobs(
                    &self.handle,
                    jobs,
                    slot_coords,
                    req,
                    &geo,
                    Some(&index.images),
                    output,
                )
            }
        }
    }

    /// Collect a fixed-array dataset's per-chunk `(address, on-disk byte
    /// count, filter mask)` entries, indexed by index-grid linear slot
    /// ([`crate::io::chunk_grid`]). Empty when the index or its data block is
    /// unallocated.
    ///
    /// Shared by the full/slice chunked reader
    /// ([`read_chunked_fixed_array`](Self::read_chunked_fixed_array)) and the
    /// direct single-chunk read
    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the fixed-array
    /// wire format has one decoder.
    fn collect_fa_chunk_entries(
        &mut self,
        chunk_dims: &[u64],
        ndims: usize,
        element_size: u64,
        index_address: u64,
    ) -> IoResult<Vec<(u64, u64, u32)>> {
        use crate::format::chunk_index::fixed_array::*;

        if index_address == UNDEF_ADDR {
            // Unallocated: no chunks recorded.
            return Ok(Vec::new());
        }

        // Read FA header
        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
        let fa_hdr = FixedArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;

        if fa_hdr.data_blk_addr == UNDEF_ADDR {
            // Unallocated data block: no chunks recorded.
            return Ok(Vec::new());
        }

        // The chunk shape (from the layout message) must match the
        // dataspace rank; otherwise the chunk-grid indexing panics.
        if chunk_dims.len() != ndims {
            return Err(crate::io::IoError::InvalidState(format!(
                "fixed-array dataset rank {} does not match chunk rank {}",
                ndims,
                chunk_dims.len()
            )));
        }

        let is_filtered = fa_hdr.client_id == FA_CLIENT_FILT_CHUNK;
        let sizeof_addr = self.meta.ctx.sizeof_addr as usize;
        // chunk_size_len = element_size - sizeof_addr - filter_mask(4)
        let chunk_size_len = if is_filtered {
            (fa_hdr.element_size as usize)
                .checked_sub(sizeof_addr + 4)
                .ok_or_else(|| {
                    crate::io::IoError::InvalidState(
                        "fixed array filtered element_size too small".into(),
                    )
                })?
        } else {
            0
        };
        // The compressed-size field is read into a u64; reject a width that
        // would overflow the read_size helper.
        if chunk_size_len > 8 {
            return Err(crate::io::IoError::InvalidState(format!(
                "fixed array filtered chunk-size width {chunk_size_len} exceeds 8 bytes"
            )));
        }

        // Compute chunk byte size
        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);

        // Bytes one element takes in the data block (or a page of it).
        let elem_size = if is_filtered {
            sizeof_addr + chunk_size_len + 4
        } else {
            sizeof_addr
        };
        // The element count is the header's claim, and everything below is
        // sized from it: refuse one whose elements could not all be in the
        // file before it sizes a reservation, a read, or a page count.
        let file_size = self.handle.file_size()?;
        let num_elmts = usize::try_from(fa_hdr.num_elmts)
            .ok()
            .filter(|&n| {
                (n as u64)
                    .checked_mul(elem_size as u64)
                    .is_some_and(|bytes| bytes <= file_size)
            })
            .ok_or_else(|| {
                crate::io::IoError::InvalidState(format!(
                    "fixed array declares {} elements of {elem_size} bytes, more than the \
                     {file_size}-byte file holds",
                    fa_hdr.num_elmts
                ))
            })?;

        // Collect per-chunk (address, compressed_size). compressed_size is the
        // exact on-disk byte count for filtered chunks, or chunk_bytes when
        // unfiltered.
        // (chunk address, on-disk byte count, filter mask). The mask is the
        // per-chunk filter mask for filtered chunks, 0 when unfiltered.
        let mut chunk_entries: Vec<(u64, u64, u32)> = Vec::with_capacity(num_elmts);

        if fa_hdr.is_paged() {
            // Paged data block: prefix (with page-init bitmap) followed by pages.
            let npages = fa_hdr.npages();
            let dblk_page_nelmts = fa_hdr.dblk_page_nelmts();
            let prefix_len = 4 + 1 + 1 + sizeof_addr + (npages as usize).div_ceil(8) + 4;
            let prefix_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, prefix_len)?;
            let prefix = FixedArrayPagedPrefix::decode(&prefix_buf, &self.meta.ctx, npages)?;

            // All pages have the same on-disk stride; only the last page holds
            // fewer elements (libhdf5: dblk_page_size is constant).
            let page_stride = dblk_page_nelmts as usize * elem_size + 4;
            let pages_base = fa_hdr.data_blk_addr + prefix.prefix_size as u64;

            for p in 0..npages as usize {
                // Elements on this page (last page may be short).
                let page_nelmts = if p + 1 == npages as usize {
                    let rem = fa_hdr.num_elmts % dblk_page_nelmts;
                    if rem == 0 {
                        dblk_page_nelmts
                    } else {
                        rem
                    }
                } else {
                    dblk_page_nelmts
                } as usize;

                if !prefix.page_initialized(p) {
                    // Uninitialized page: all chunk entries are undefined.
                    chunk_entries
                        .extend(std::iter::repeat_n((UNDEF_ADDR, 0u64, 0u32), page_nelmts));
                    continue;
                }

                let page_addr = pages_base + (p as u64) * page_stride as u64;
                let page_size = page_nelmts * elem_size + 4;
                let page_buf = self.handle.read_at_most(page_addr, page_size)?;

                if is_filtered {
                    let elems = decode_filtered_page(
                        &page_buf,
                        &self.meta.ctx,
                        page_nelmts,
                        chunk_size_len,
                    )?;
                    for e in elems {
                        chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
                    }
                } else {
                    let addrs = decode_unfiltered_page(&page_buf, &self.meta.ctx, page_nelmts)?;
                    for addr in addrs {
                        chunk_entries.push((addr, chunk_bytes, 0));
                    }
                }
            }
        } else {
            // Non-paged data block: all elements live inline in the data block.
            let dblk_size = 4 + 1 + 1 + sizeof_addr + num_elmts * elem_size + 4;
            let dblk_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, dblk_size)?;

            if is_filtered {
                let fa_dblk = FixedArrayDataBlock::decode_filtered(
                    &dblk_buf,
                    &self.meta.ctx,
                    num_elmts,
                    chunk_size_len,
                )?;
                for e in &fa_dblk.filtered_elements {
                    chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
                }
            } else {
                let fa_dblk =
                    FixedArrayDataBlock::decode_unfiltered(&dblk_buf, &self.meta.ctx, num_elmts)?;
                for &addr in &fa_dblk.elements {
                    chunk_entries.push((addr, chunk_bytes, 0));
                }
            }
        }

        Ok(chunk_entries)
    }

    /// Read a dataset indexed by a fixed array.
    ///
    /// Scatters only; `output` must already be sized to the target extent.
    /// Every byte of it is defined before this returns `Ok`: what no chunk
    /// covers is filled with the tiled fill value by
    /// [`place_chunk_jobs`], the one exit every branch here takes.
    fn read_chunked_fixed_array(
        &mut self,
        name: &str,
        chunk_dims: &[u64],
        index_address: u64,
        req: ChunkReadRequest,
        output: &mut [u8],
    ) -> IoResult<()> {
        let ChunkReadRequest {
            pipeline, target, ..
        } = req;
        let pos = self.dataset_position(name)?;
        let info = &self.datasets[pos];
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let max_dims = info.dataspace.max_dims.clone();
        let ndims = dims.len();
        let geo = ChunkOutputGeometry {
            dims: &dims,
            chunk_dims,
            element_size,
        };

        // Index-grid slot -> chunk-grid coordinates (row-major, against the
        // maximum extent — the array was sized from its chunk grid, so a slot
        // beyond the current extent still decodes to its true position and
        // then simply falls outside the read target). A zero chunk dimension
        // from a malformed layout message is rejected inside.
        let index = self.decoded_chunk_index(pos, index_address, |reader| {
            let entries =
                reader.collect_fa_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
            let coords = crate::io::chunk_grid::coords_table(
                &dims,
                max_dims.as_deref(),
                chunk_dims,
                entries.len(),
            )?;
            Ok(DecodedChunkIndex::new(entries, coords))
        })?;
        if index.entries.is_empty() {
            // Unallocated index/data block: the empty plan makes the whole
            // output fill.
            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
        }
        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
        let chunk_coords = |i: usize| -> &[u64] { &index.coords[i * ndims..(i + 1) * ndims] };

        // Build one read job per chunk (no I/O yet). Filtered chunks carry
        // their exact compressed size (read at-most, since a zero size means
        // "unknown" and falls back to a generous estimate); unfiltered chunks
        // read the exact chunk byte count. For a slice, chunks outside the
        // selection become None and are never read.
        let jobs: Vec<Option<ChunkReadJob>> = index
            .entries
            .iter()
            .enumerate()
            .map(|(linear_idx, &(addr, comp_size, mask))| {
                if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(linear_idx), chunk_dims) {
                    None
                } else if pipeline.is_some() {
                    let read_len = if comp_size > 0 {
                        comp_size as usize
                    } else {
                        chunk_bytes as usize * 2
                    };
                    Some(ChunkReadJob {
                        addr,
                        len: read_len,
                        at_most: true,
                        mask,
                    })
                } else {
                    Some(ChunkReadJob {
                        addr,
                        len: chunk_bytes as usize,
                        at_most: false,
                        mask,
                    })
                }
            })
            .collect();

        // Read and place each chunk: the selected byte runs straight out of
        // the file where that beats reading the chunk whole.
        place_chunk_jobs(
            &self.handle,
            jobs,
            &index.coords,
            req,
            &geo,
            Some(&index.images),
            output,
        )
    }

    /// Read a dataset indexed by the implicit ("none") chunk index.
    ///
    /// There is no on-disk index structure at all (`H5Dnone.c`): every chunk
    /// slot in the maximum-extent grid is allocated in one block at dataset
    /// creation, so a chunk's address is purely arithmetic — `index_address +
    /// slot * chunk_bytes`, where `slot` is its row-major position in the
    /// same maximum-extent grid the fixed/extensible-array/v2-B-tree indexes
    /// use (`H5D__chunk_set_info_real`'s `max_down_chunks`). This index type
    /// is only ever selected for a fixed (non-unlimited) chunked dataset with
    /// early allocation and no filters, so there is no per-chunk allocation
    /// flag, compressed size, or filter mask to track.
    ///
    /// Scatters only; `output` must already be sized to the target extent.
    /// Every byte of it is defined before this returns `Ok`: what no chunk
    /// covers is filled with the tiled fill value by
    /// [`place_chunk_jobs`], the one exit every branch here takes.
    fn read_chunked_implicit(
        &mut self,
        name: &str,
        chunk_dims: &[u64],
        index_address: u64,
        req: ChunkReadRequest,
        output: &mut [u8],
    ) -> IoResult<()> {
        let target = req.target;
        let pos = self.dataset_position(name)?;
        let info = &self.datasets[pos];
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let ndims = dims.len();

        if index_address == UNDEF_ADDR {
            // Unallocated: the empty plan makes the whole output fill.
            let geo = ChunkOutputGeometry {
                dims: &dims,
                chunk_dims,
                element_size,
            };
            return place_chunk_jobs(
                &self.handle,
                Vec::new(),
                &[],
                ChunkReadRequest {
                    pipeline: None,
                    ..req
                },
                &geo,
                None,
                output,
            );
        }

        if chunk_dims.len() != ndims {
            return Err(crate::io::IoError::InvalidState(format!(
                "implicit-index dataset rank {} does not match chunk rank {}",
                ndims,
                chunk_dims.len()
            )));
        }

        let max_dims = self.datasets[pos].dataspace.max_dims.clone();
        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);

        // Every slot's coordinates and address are computed directly, not
        // read from disk, so there is no "unallocated chunk" case here the
        // way a sparse index has one: `at_most: false` because
        // `H5D__none_idx_create` guarantees the whole block is present.
        let index = self.decoded_chunk_index(pos, index_address, |_| {
            let grid = crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
            let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
            let coords = crate::io::chunk_grid::coords_table(
                &dims,
                max_dims.as_deref(),
                chunk_dims,
                chunks_total as usize,
            )?;
            let entries = (0..chunks_total)
                .map(|i| (index_address + i * chunk_bytes, chunk_bytes, 0))
                .collect();
            Ok(DecodedChunkIndex::new(entries, coords))
        })?;
        let slot_coords = &index.coords;

        let jobs: Vec<Option<ChunkReadJob>> = index
            .entries
            .iter()
            .enumerate()
            .map(|(i, &(addr, len, _))| {
                let coords = &slot_coords[i * ndims..(i + 1) * ndims];
                if !target.overlaps(coords, chunk_dims) {
                    None
                } else {
                    Some(ChunkReadJob {
                        addr,
                        len: len as usize,
                        at_most: false,
                        mask: 0,
                    })
                }
            })
            .collect();

        let geo = ChunkOutputGeometry {
            dims: &dims,
            chunk_dims,
            element_size,
        };
        place_chunk_jobs(
            &self.handle,
            jobs,
            slot_coords,
            ChunkReadRequest {
                pipeline: None,
                ..req
            },
            &geo,
            Some(&index.images),
            output,
        )
    }

    /// Collect a v2-B-tree-indexed dataset's per-chunk `(address, read
    /// size, scaled chunk-grid offsets, filter mask)` entries, walking the
    /// tree once. `read_size` is the compressed size for a filtered chunk,
    /// the full chunk size otherwise. Empty when the index has no records.
    ///
    /// Shared by the full/slice chunked reader
    /// ([`read_chunked_btree_v2`](Self::read_chunked_btree_v2)) and the
    /// direct single-chunk read
    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the v2 B-tree
    /// record format has one decoder.
    fn collect_bt2_chunk_entries(
        &mut self,
        chunk_dims: &[u64],
        ndims: usize,
        element_size: u64,
        index_address: u64,
    ) -> IoResult<Vec<Bt2ChunkEntry>> {
        use crate::format::chunk_index::btree_v2::*;

        if index_address == UNDEF_ADDR {
            // Unallocated: no chunks recorded.
            return Ok(Vec::new());
        }

        // Read BT2 header
        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
        let bt2_hdr = Bt2Header::decode(&hdr_buf, &self.meta.ctx)?;

        if bt2_hdr.root_node_addr == UNDEF_ADDR || bt2_hdr.total_num_records == 0 {
            // No records.
            return Ok(Vec::new());
        }

        // Walk the B-tree to any depth, collecting every record's raw bytes
        // from the internal nodes and leaves.
        let ctx = self.meta.ctx;
        let record_bytes = collect_btree_v2_records(
            &bt2_hdr,
            &ctx,
            &mut HandleBlockReader {
                handle: &mut self.handle,
            },
        )?;
        let total_records = if bt2_hdr.record_size > 0 {
            record_bytes.len() / bt2_hdr.record_size as usize
        } else {
            0
        };

        // Decode records
        // Compute chunk byte size
        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);

        // Unify filtered and unfiltered records into (address, read_size,
        // scaled offsets, filter mask). read_size is the compressed size for
        // filtered chunks, the full chunk size otherwise; the mask is 0 for
        // unfiltered records.
        let entries: Vec<Bt2ChunkEntry> = if bt2_hdr.record_type == BT2_TYPE_CHUNK_UNFILT {
            Bt2ChunkIndex::decode_unfiltered_records(
                &record_bytes,
                total_records,
                ndims,
                &self.meta.ctx,
            )?
            .into_iter()
            .map(|r| (r.chunk_address, chunk_bytes as usize, r.scaled_offsets, 0))
            .collect()
        } else {
            Bt2ChunkIndex::decode_filtered_records(
                &record_bytes,
                total_records,
                ndims,
                bt2_hdr.record_size,
                &self.meta.ctx,
            )?
            .into_iter()
            .map(|r| {
                (
                    r.chunk_address,
                    r.chunk_size as usize,
                    r.scaled_offsets,
                    r.filter_mask,
                )
            })
            .collect()
        };

        Ok(entries)
    }

    /// Read a dataset indexed by a B-tree v2.
    ///
    /// Scatters only; `output` must already be sized to the target extent.
    /// Every byte of it is defined before this returns `Ok`: what no chunk
    /// covers is filled with the tiled fill value by
    /// [`place_chunk_jobs`], the one exit every branch here takes.
    fn read_chunked_btree_v2(
        &mut self,
        name: &str,
        chunk_dims: &[u64],
        index_address: u64,
        req: ChunkReadRequest,
        output: &mut [u8],
    ) -> IoResult<()> {
        let target = req.target;
        let pos = self.dataset_position(name)?;
        let info = &self.datasets[pos];
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let ndims = dims.len();
        let geo = ChunkOutputGeometry {
            dims: &dims,
            chunk_dims,
            element_size,
        };

        // A record carries its own scaled (chunk-grid) offsets, always `ndims`
        // of them, so flattening them into the coordinate table loses nothing.
        let index = self.decoded_chunk_index(pos, index_address, |reader| {
            let records =
                reader.collect_bt2_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
            let mut entries = Vec::with_capacity(records.len());
            let mut coords = Vec::with_capacity(records.len() * ndims);
            for (addr, read_size, scaled, mask) in records {
                entries.push((addr, read_size as u64, mask));
                coords.extend_from_slice(&scaled);
            }
            Ok(DecodedChunkIndex::new(entries, coords))
        })?;
        if index.entries.is_empty() {
            // Unallocated index or no records: the empty plan makes the whole
            // output fill.
            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
        }

        // Build one read job per chunk (no I/O yet), placing each by its
        // scaled offsets. For a slice, chunks outside the selection become
        // None and are never read.
        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
        for (i, &(addr, read_size, mask)) in index.entries.iter().enumerate() {
            let scaled = &index.coords[i * ndims..(i + 1) * ndims];
            if addr == UNDEF_ADDR || read_size == 0 || !target.overlaps(scaled, chunk_dims) {
                jobs.push(None);
            } else {
                jobs.push(Some(ChunkReadJob {
                    addr,
                    len: read_size as usize,
                    at_most: false,
                    mask,
                }));
            }
        }

        // Read and place each chunk N-dimensionally by its scaled offsets.
        place_chunk_jobs(
            &self.handle,
            jobs,
            &index.coords,
            req,
            &geo,
            Some(&index.images),
            output,
        )
    }

    /// Read a chunked dataset indexed by a version-1 B-tree (layout
    /// message version 3, class 2 "chunked").
    ///
    /// `chunk_dims` excludes the trailing element-size dimension.
    ///
    /// Scatters only; `output` must already be sized to the target extent.
    /// Every byte of it is defined before this returns `Ok`: what no chunk
    /// covers is filled with the tiled fill value by
    /// [`place_chunk_jobs`], the one exit every branch here takes.
    fn read_chunked_btree_v1(
        &mut self,
        name: &str,
        chunk_dims: &[u64],
        b_tree_address: u64,
        req: ChunkReadRequest,
        output: &mut [u8],
    ) -> IoResult<()> {
        let ChunkReadRequest {
            pipeline, target, ..
        } = req;
        let pos = self.dataset_position(name)?;
        let info = &self.datasets[pos];
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let ndims = dims.len();

        // The chunk shape must match the dataspace rank or the chunk-grid
        // indexing below panics.
        if chunk_dims.len() != ndims {
            return Err(crate::io::IoError::InvalidState(format!(
                "B-tree-v1 dataset rank {} does not match chunk rank {}",
                ndims,
                chunk_dims.len()
            )));
        }

        let total_size: u64 = saturating_byte_len(&dims, element_size);
        if b_tree_address == UNDEF_ADDR || total_size == 0 {
            // Unallocated: the empty plan makes the whole output fill.
            let geo = ChunkOutputGeometry {
                dims: &dims,
                chunk_dims,
                element_size,
            };
            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
        }

        // Walk the B-tree, collecting every leaf entry as
        // (element_offsets, chunk_address, chunk_size, filter_mask), and turn
        // each key's element offsets into chunk-grid coordinates. The keys
        // carry rank + 1 offsets; the trailing element-size dimension offset
        // is always 0 and is dropped.
        let file_size = self.handle.file_size()?;
        let index = self.decoded_chunk_index(pos, b_tree_address, |reader| {
            let mut leaves: Vec<(Vec<u64>, u64, u32, u32)> = Vec::new();
            reader.collect_btree_v1_chunks(b_tree_address, ndims, file_size, 0, &mut leaves)?;
            let mut entries = Vec::with_capacity(leaves.len());
            let mut coords = Vec::with_capacity(leaves.len() * ndims);
            for (offsets, addr, chunk_size, mask) in leaves {
                for (d, &cd) in chunk_dims.iter().enumerate().take(ndims) {
                    coords.push(offsets[d].checked_div(cd).unwrap_or(0));
                }
                entries.push((addr, chunk_size as u64, mask));
            }
            Ok(DecodedChunkIndex::new(entries, coords))
        })?;

        // The uncompressed byte size of a full chunk.
        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);

        // Build one read job per chunk (no I/O yet), placing each by its
        // scaled coordinates.
        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
        for (i, &(addr, chunk_size, mask)) in index.entries.iter().enumerate() {
            let skip = addr == UNDEF_ADDR
                || chunk_size == 0
                || addr >= file_size
                || chunk_size > file_size
                || !target.overlaps(&index.coords[i * ndims..(i + 1) * ndims], chunk_dims);
            jobs.push(if skip {
                None
            } else {
                Some(ChunkReadJob {
                    addr,
                    len: chunk_size as usize,
                    at_most: false,
                    mask,
                })
            });
        }

        // Read and place each chunk N-dimensionally by its scaled offsets.
        let geo = ChunkOutputGeometry {
            dims: &dims,
            chunk_dims,
            element_size,
        };
        place_chunk_jobs(
            &self.handle,
            jobs,
            &index.coords,
            req,
            &geo,
            Some(&index.images),
            output,
        )?;

        // libhdf5 stores raw byte sizes; verify the uncompressed chunk
        // size is consistent for unfiltered datasets so a corrupt index
        // surfaces instead of silently producing garbage.
        if pipeline.is_none() {
            for &(addr, chunk_size, _) in &index.entries {
                if addr != UNDEF_ADDR && chunk_size != chunk_bytes && chunk_size != 0 {
                    return Err(crate::io::IoError::InvalidState(format!(
                        "chunk B-tree v1: unfiltered chunk size {} != expected {}",
                        chunk_size, chunk_bytes
                    )));
                }
            }
        }

        Ok(())
    }

    /// Recursively walk a version-1 raw-data-chunk B-tree, collecting every
    /// leaf entry as `(element_offsets, chunk_address, chunk_size,
    /// filter_mask)`.
    ///
    /// `rank` is the chunk rank excluding the trailing element-size
    /// dimension. Recursion is bounded by the node level read from disk:
    /// each recursive step descends to a strictly lower level, and the
    /// `depth` counter caps the descent at the 1-byte level field's range.
    fn collect_btree_v1_chunks(
        &mut self,
        addr: u64,
        rank: usize,
        file_size: u64,
        depth: u32,
        out: &mut Vec<(Vec<u64>, u64, u32, u32)>,
    ) -> IoResult<()> {
        // A v1 B-tree node level fits in one byte, so the tree can be at
        // most 256 levels deep; this also stops cyclic/corrupt indices.
        if depth > 256 {
            return Err(crate::io::IoError::InvalidState(
                "chunk B-tree v1 exceeds maximum depth".into(),
            ));
        }
        if addr == UNDEF_ADDR || addr >= file_size {
            return Ok(());
        }

        // A node is a fixed-size record: the header (8 + 2*sizeof_addr) plus
        // `2 * chunk_internal_k` interleaved keys/children.
        let sa = self.meta.ctx.sizeof_addr as usize;
        let node_size = self.meta.btree.chunk_btree_node_size(sa, rank);
        let buf = self.handle.read_at_most(addr, node_size)?;
        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.meta.btree.chunk_max_entries())?;

        if node.level == 0 {
            // Leaf node: each child points at chunk data.
            for (i, &child_addr) in node.children.iter().enumerate() {
                let key = &node.keys[i];
                out.push((
                    key.offsets[..rank].to_vec(),
                    child_addr,
                    key.chunk_size,
                    key.filter_mask,
                ));
            }
        } else {
            // Internal node: each child points at a sub-TREE node one
            // level below. `node.level` is read from disk and decreases on
            // every descent, so it also bounds the recursion.
            let children: Vec<u64> = node.children.clone();
            for child_addr in children {
                if child_addr == UNDEF_ADDR || child_addr >= file_size {
                    continue;
                }
                self.collect_btree_v1_chunks(child_addr, rank, file_size, depth + 1, out)?;
            }
        }
        Ok(())
    }

    /// Read variable-length string data from a dataset.
    ///
    /// h5py stores vlen strings as global heap references. Each element
    /// in the raw data is a (collection_address, object_index) pair that
    /// points to a string blob in a global heap collection.
    ///
    /// Returns a Vec<String> with one entry per element.
    pub fn read_vlen_strings(&mut self, name: &str) -> IoResult<Vec<String>> {
        // A vlen string is a vlen sequence of `u8` reinterpreted as UTF-8.
        // The global-heap walk is identical; decode the raw object bytes.
        Ok(self
            .read_vlen_objects(name)?
            .into_iter()
            .map(|bytes| String::from_utf8_lossy(&bytes).to_string())
            .collect())
    }

    /// Read a 1-D variable-length byte-array dataset (vlen sequence of `u8`).
    ///
    /// Returns a `Vec<Vec<u8>>` with one byte array per element. Missing or
    /// undefined references yield an empty `Vec`.
    pub fn read_vlen_bytes(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
        self.read_vlen_objects(name)
    }

    /// Shared owner of the global-heap walk for variable-length datasets.
    ///
    /// Each element of the raw data is a vlen reference (collection address +
    /// object index) into a global heap collection. Returns the raw object
    /// bytes for each element, with an empty `Vec` for undefined/missing
    /// references. Both `read_vlen_strings` (UTF-8 view) and `read_vlen_bytes`
    /// (raw view) layer on top of this.
    fn read_vlen_objects(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_vlen_objects(&path);
        }
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let layout = info.layout.clone();
        let external_files = info.external_files.clone();
        let total_elements: u64 = dims.iter().fold(1u64, |acc, &d| acc.saturating_mul(d));

        let raw = match &layout {
            DataLayoutMessage::Contiguous { size, .. } if !external_files.is_empty() => {
                let prefix = self.extfile_prefix_in_force(name);
                let mut buf = vec![0u8; *size as usize];
                read_external_file_bytes(&external_files, prefix.as_deref(), 0, &mut buf)?;
                buf
            }
            DataLayoutMessage::Contiguous { address, size } => {
                if *address == UNDEF_ADDR {
                    return Ok(vec![]);
                }
                self.handle.read_at(*address, *size as usize)?
            }
            DataLayoutMessage::Compact { data } => data.clone(),
            _ => {
                // For chunked, read the full dataset first
                self.read_dataset_raw(name)?
            }
        };

        let ref_size = vlen_reference_size(&self.meta.ctx);
        // The extent is the file's claim; the loop below stops at the image
        // it actually read, so the reservation is bounded by that image.
        let mut items: Vec<Vec<u8>> =
            Vec::with_capacity((total_elements as usize).min(raw.len() / ref_size));

        // Cache global heap collections to avoid re-reading.
        // Store as (collection, index→offset lookup) for O(1) object access.
        let mut heap_cache: std::collections::HashMap<
            u64,
            (GlobalHeapCollection, std::collections::HashMap<u16, usize>),
        > = std::collections::HashMap::new();

        for i in 0..total_elements as usize {
            let offset = i * ref_size;
            if offset + ref_size > raw.len() {
                break;
            }

            let (_seq_len, collection_addr, obj_index) =
                decode_vlen_reference(&raw[offset..], &self.meta.ctx)?;

            if collection_addr == UNDEF_ADDR || collection_addr == 0 {
                items.push(Vec::new());
                continue;
            }

            // Read or get cached global heap collection
            if let std::collections::hash_map::Entry::Vacant(e) = heap_cache.entry(collection_addr)
            {
                let coll = self.read_heap_collection(collection_addr)?;
                let lookup: std::collections::HashMap<u16, usize> = coll
                    .objects
                    .iter()
                    .enumerate()
                    .map(|(i, o)| (o.index, i))
                    .collect();
                e.insert((coll, lookup));
            }

            let idx = u16::try_from(obj_index).map_err(|_| {
                crate::io::IoError::InvalidState(format!(
                    "global heap object index {obj_index} does not fit the 16-bit on-disk field \
                     (element {i} of \"{name}\")"
                ))
            })?;
            let (coll, lookup) = &heap_cache[&collection_addr];
            let &oi = lookup.get(&idx).ok_or_else(|| {
                crate::io::IoError::InvalidState(format!(
                    "global heap object {idx} not found in the collection at address \
                     {collection_addr:#x} (element {i} of \"{name}\")"
                ))
            })?;
            items.push(coll.objects[oi].data.clone());
        }

        Ok(items)
    }

    /// Collect chunk (address, size) entries from an EA index.
    /// Returns a vector indexed by chunk linear index.
    fn collect_ea_chunk_entries(
        &mut self,
        index_address: u64,
        params: &data_layout::EarrayParams,
        dims: &[u64],
        max_dims: Option<&[u64]>,
        chunk_dims: &[u64],
        element_size: u64,
    ) -> IoResult<Vec<(u64, u64, u32)>> {
        use crate::format::chunk_index::extensible_array::{self as ea, *};

        if index_address == UNDEF_ADDR {
            return Ok(vec![]);
        }
        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
        let ea_hdr = ExtensibleArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;
        if ea_hdr.idx_blk_addr == UNDEF_ADDR {
            return Ok(vec![]);
        }

        // Slot count of the index grid, bounding the collection walk. The
        // maximum extent decides the multipliers (sub-frame chunks make this
        // larger than the dim-0 chunk count alone); the unlimited dimension 0
        // is bounded by the current extent.
        let chunks_dim0: usize = crate::io::chunk_grid::index_grid(dims, max_dims, chunk_dims)?
            .iter()
            .fold(1usize, |acc, &n| acc.saturating_mul(n as usize));
        let geo = EaGeometry::new(
            params.idx_blk_elmts,
            params.data_blk_min_elmts,
            params.sup_blk_min_data_ptrs,
            params.max_nelmts_bits,
            params.max_dblk_page_nelmts_bits,
        )?;
        let chunk_bytes = saturating_byte_len(chunk_dims, element_size);
        let is_filtered = ea_hdr.class_id == ea::EA_CLS_FILT_CHUNK;
        let chunk_size_len = if is_filtered {
            ea_hdr.raw_elmt_size - self.meta.ctx.sizeof_addr - 4
        } else {
            0
        };
        let max_nelmts_bits = params.max_nelmts_bits;
        // Each entry is (chunk address, on-disk byte count, filter mask). The
        // mask is the per-chunk filter mask for filtered datasets (0 for an
        // unfiltered index, where it is meaningless).
        let mut entries: Vec<(u64, u64, u32)> = Vec::new();

        // Read the index block: direct elements + the data-block / super-block
        // address arrays (the address arrays are filter-agnostic).
        let (dblk_addrs, sblk_addrs): (Vec<u64>, Vec<u64>) = if is_filtered {
            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
            let fiblk = ea::FilteredIndexBlock::decode(
                &buf,
                &self.meta.ctx,
                params.idx_blk_elmts as usize,
                geo.ndblk_addrs,
                geo.nsblk_addrs,
                chunk_size_len,
            )?;
            for e in &fiblk.elements {
                entries.push((e.addr, e.nbytes, e.filter_mask));
            }
            (fiblk.dblk_addrs, fiblk.sblk_addrs)
        } else {
            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
            let iblk = ExtensibleArrayIndexBlock::decode(
                &buf,
                &self.meta.ctx,
                params.idx_blk_elmts as usize,
                geo.ndblk_addrs,
                geo.nsblk_addrs,
            )?;
            for &addr in &iblk.elements {
                entries.push((addr, chunk_bytes, 0));
            }
            (iblk.dblk_addrs, iblk.sblk_addrs)
        };

        // Walk super blocks in order, collecting each data block's entries.
        let sa = self.meta.ctx.sizeof_addr as usize;
        let raw_elmt_size = if is_filtered {
            ea::FilteredChunkEntry::raw_size(self.meta.ctx.sizeof_addr, chunk_size_len) as usize
        } else {
            sa
        };
        'outer: for (u, s) in geo.sblk.iter().enumerate() {
            if entries.len() >= chunks_dim0 {
                break;
            }
            let dblk_nelmts = s.dblk_nelmts as usize;
            let paged = geo.is_sblk_paged(u);

            // This super block's data-block addresses, plus its page-init
            // bitmap region (empty unless the super block is paged).
            let (this_dblk_addrs, page_init): (Vec<u64>, Vec<u8>) = if u < geo.iblock_nsblks {
                let start = s.start_dblk as usize;
                (
                    (0..s.ndblks as usize)
                        .map(|d| dblk_addrs.get(start + d).copied().unwrap_or(UNDEF_ADDR))
                        .collect(),
                    Vec::new(),
                )
            } else {
                let sblk_addr = sblk_addrs
                    .get(u - geo.iblock_nsblks)
                    .copied()
                    .unwrap_or(UNDEF_ADDR);
                if sblk_addr == UNDEF_ADDR {
                    (vec![UNDEF_ADDR; s.ndblks as usize], Vec::new())
                } else {
                    let page_init_total = if paged {
                        s.ndblks as usize * geo.dblk_page_init_size(u)
                    } else {
                        0
                    };
                    // Size the read from the super block's geometry rather
                    // than a fixed cap: signature+version+class+header_addr
                    // + block_offset(<=8) + page-init bitmaps
                    // + ndblks data-block addresses + checksum.
                    let sblk_size =
                        4 + 1 + 1 + sa + 8 + page_init_total + s.ndblks as usize * sa + 4;
                    let buf = self.handle.read_at_most(sblk_addr, sblk_size)?;
                    let sb = ExtensibleArraySuperBlock::decode(
                        &buf,
                        &self.meta.ctx,
                        max_nelmts_bits,
                        s.ndblks as usize,
                        page_init_total,
                    )?;
                    (sb.dblk_addrs, sb.page_init)
                }
            };

            let npages = geo.npages(u) as usize;
            let page_size = geo.dblk_page_size(raw_elmt_size);
            let prefix = geo.dblk_prefix_size(self.meta.ctx.sizeof_addr, max_nelmts_bits);

            for (d, &dblk_addr) in this_dblk_addrs.iter().enumerate() {
                if dblk_addr == UNDEF_ADDR {
                    entries.extend(std::iter::repeat_n((UNDEF_ADDR, 0, 0), dblk_nelmts));
                } else if paged {
                    // Paged data block: only a prefix lives at `dblk_addr`;
                    // the elements live in `npages` page structures that
                    // follow it on disk. The super block's page-init bitmap
                    // is one flat MSB-first bitmap (H5VM bit ops) indexed by
                    // `dblk_idx * npages + page_idx` (H5EA.c), not a series
                    // of per-data-block sub-bitmaps.
                    for p in 0..npages {
                        let bit = d * npages + p;
                        let initialized = page_init[bit / 8] & (0x80u8 >> (bit % 8)) != 0;
                        if !initialized {
                            entries.extend(std::iter::repeat_n(
                                (UNDEF_ADDR, 0, 0),
                                geo.dblk_page_nelmts as usize,
                            ));
                            continue;
                        }
                        let page_addr = dblk_addr + prefix as u64 + (p as u64) * page_size as u64;
                        let page = self.handle.read_at(page_addr, page_size)?;
                        for k in 0..geo.dblk_page_nelmts as usize {
                            let off = k * raw_elmt_size;
                            if is_filtered {
                                let e = ea::FilteredChunkEntry::decode(
                                    &page[off..],
                                    sa,
                                    chunk_size_len as usize,
                                );
                                entries.push((e.addr, e.nbytes, e.filter_mask));
                            } else {
                                entries.push((read_addr(&page[off..], sa), chunk_bytes, 0));
                            }
                        }
                    }
                } else if is_filtered {
                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
                    let dblk = ea::FilteredDataBlock::decode(
                        &buf,
                        &self.meta.ctx,
                        max_nelmts_bits,
                        dblk_nelmts,
                        chunk_size_len,
                    )?;
                    for e in &dblk.elements {
                        entries.push((e.addr, e.nbytes, e.filter_mask));
                    }
                } else {
                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
                    let dblk = ExtensibleArrayDataBlock::decode(
                        &buf,
                        &self.meta.ctx,
                        max_nelmts_bits,
                        dblk_nelmts,
                    )?;
                    for &addr in &dblk.elements {
                        entries.push((addr, chunk_bytes, 0));
                    }
                }
                if entries.len() >= chunks_dim0 {
                    break 'outer;
                }
            }
        }
        Ok(entries)
    }

    /// Read a slice (hyperslab) of a dataset, whatever its layout.
    ///
    /// Contiguous and compact datasets are read run by run; a chunked one
    /// reads only the chunks the selection overlaps, and any gap the writer
    /// never filled comes back as the fill value.
    ///
    /// `starts` and `counts` define the N-dimensional selection:
    /// starts[d] is the first index along dim d, counts[d] is how many.
    /// Returns the selected data in row-major order.
    pub fn read_slice(&mut self, name: &str, starts: &[u64], counts: &[u64]) -> IoResult<Vec<u8>> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_slice(&path, starts, counts);
        }
        let (datatype, out_bytes) = self.slice_size_and_datatype(name, starts, counts)?;
        // The selection lands in the vector this returns:
        // `read_slice_into_unconverted` defines every byte of the buffer it is
        // handed, so there is nothing to zero first and nothing to copy after.
        read_image_into_new::<u8, _, _>(out_bytes as usize, |image| {
            self.read_slice_into_unconverted(name, starts, counts, image, 0, ReadDst::Fresh)?;
            Self::apply_post_filter_conversion(image, &datatype)
        })
    }

    /// Read a hyperslab straight into a caller-provided buffer (no allocation).
    ///
    /// `out.len()` must equal `product(counts) * element_size`; otherwise an
    /// error is returned. The no-allocation counterpart of
    /// [`read_slice`](Self::read_slice) and the zero-copy entry point for
    /// reading a selection directly into a pinned/registered host buffer for an
    /// H2D transfer.
    pub fn read_slice_into(
        &mut self,
        name: &str,
        starts: &[u64],
        counts: &[u64],
        out: &mut [u8],
    ) -> IoResult<()> {
        self.read_slice_into_dst(name, starts, counts, out, ReadDst::Reused)
    }

    /// [`read_slice_into`](Self::read_slice_into) with the caller's
    /// destination fact made explicit, for internal callers whose buffer is a
    /// fresh allocation rather than a kept one (the allocating wrappers in
    /// `dataset.rs`).
    pub(crate) fn read_slice_into_dst(
        &mut self,
        name: &str,
        starts: &[u64],
        counts: &[u64],
        out: &mut [u8],
        dst: ReadDst,
    ) -> IoResult<()> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_slice_into_dst(&path, starts, counts, out, dst);
        }
        let (datatype, out_bytes) = self.slice_size_and_datatype(name, starts, counts)?;
        if out.len() as u64 != out_bytes {
            return Err(crate::io::IoError::InvalidState(format!(
                "read_slice_into: buffer is {} bytes but selection needs {}",
                out.len(),
                out_bytes
            )));
        }
        self.read_slice_into_unconverted(name, starts, counts, out, 0, dst)?;
        Self::apply_post_filter_conversion(out, &datatype)?;
        Ok(())
    }

    /// Logical byte size of a hyperslab (`product(counts) * element_size`) with
    /// the datatype needed for the post-filter conversion.
    fn slice_size_and_datatype(
        &self,
        name: &str,
        starts: &[u64],
        counts: &[u64],
    ) -> IoResult<(DatatypeMessage, u64)> {
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        // Bounds first: a selection the extent does not admit is refused
        // before its byte size is computed, so an oversized count is reported
        // as such rather than as a failed allocation.
        check_hyperslab(&info.dataspace.dims, starts, counts)?;
        let out_bytes = saturating_byte_len(counts, info.datatype.element_size() as u64);
        Ok((info.datatype.clone(), out_bytes))
    }

    /// Fill `out` with a hyperslab selection, before the post-filter datatype
    /// conversion. The single owner of read-destination semantics for slice
    /// reads (mirrors [`read_dataset_raw_into_unconverted`](Self::read_dataset_raw_into_unconverted)):
    /// it validates the selection, then fully defines every byte of `out`
    /// (contiguous/compact runs cover the whole selection; chunked layouts
    /// pre-fill with the tiled fill value and scatter only overlapping chunks).
    /// Both [`read_slice`](Self::read_slice) and [`read_slice_into`](Self::read_slice_into)
    /// wrap it and apply the conversion exactly once.
    ///
    /// `out.len()` must equal `product(counts) * element_size`. `depth`
    /// counts virtual-dataset nesting for a caller reached through
    /// [`read_virtual_into`](Self::read_virtual_into); pass `0` for a
    /// top-level call. `dst` is the caller's destination fact ([`ReadDst`]):
    /// whether `out` is a fresh allocation or a buffer the caller keeps
    /// across reads, which decides whether a mapped contiguous read is priced
    /// at the cold or the warm row.
    fn read_slice_into_unconverted(
        &mut self,
        name: &str,
        starts: &[u64],
        counts: &[u64],
        out: &mut [u8],
        depth: usize,
        dst: ReadDst,
    ) -> IoResult<()> {
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let layout = info.layout.clone();
        let pipeline = info.filter_pipeline.clone();
        let fill_value = info.fill_value.clone();
        let external_files = info.external_files.clone();
        let ndims = dims.len();

        check_hyperslab(&dims, starts, counts)?;
        if ndims == 0 {
            return Err(crate::io::IoError::InvalidState(
                "read_slice does not support scalar datasets; use read_dataset_raw".into(),
            ));
        }

        match &layout {
            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
                // Same coalesced run geometry as the normal contiguous case
                // below, but each run is read through the external file
                // list instead of straight from this file (H5D__efl_read,
                // H5Defl.c) — `src_off` is already dataset-relative, which
                // is exactly what `read_external_file_bytes` walks slots by.
                let prefix = self.extfile_prefix_in_force(name);
                for_each_contiguous_run(
                    &dims,
                    starts,
                    counts,
                    element_size,
                    |src_off, out_off, len| {
                        read_external_file_bytes(
                            &external_files,
                            prefix.as_deref(),
                            src_off,
                            &mut out[out_off..out_off + len],
                        )
                    },
                )?;
            }
            DataLayoutMessage::Contiguous { address, .. } => {
                if *address == UNDEF_ADDR {
                    // Never-written: the selection reads back as the fill value.
                    fill_tiled_into(out, fill_value.as_deref());
                } else {
                    // Read each maximal contiguous run straight into `out`.
                    // Trailing full-selected dimensions coalesce, so a slice
                    // like `[:, r0:r1, :]` of `[nproj, nz, nx]` becomes `nproj`
                    // reads of `(r1-r0)*nx` elements instead of `nproj*(r1-r0)`
                    // per-`nx`-row reads. The 1-D case folds to a single run.
                    // The runs cover the whole selection, so every byte of `out`
                    // is written.
                    let base = *address;
                    for_each_contiguous_run(
                        &dims,
                        starts,
                        counts,
                        element_size,
                        |src_off, out_off, len| {
                            // `base` is the file's claim; a sum that wraps
                            // would land on unrelated bytes, so it is an
                            // error, not an offset.
                            let at = base.checked_add(src_off).ok_or_else(|| {
                                crate::io::IoError::InvalidState(format!(
                                    "dataset '{name}' claims raw data at {base}, which \
                                     overflows {src_off} bytes into the selection"
                                ))
                            })?;
                            self.handle
                                .read_exact_at_into(at, &mut out[out_off..out_off + len], dst)
                                .map_err(Into::into)
                        },
                    )?;
                }
            }
            DataLayoutMessage::Compact { data } => {
                // Same coalesced geometry, copying from the in-memory full
                // dataset instead of reading from the file.
                for_each_contiguous_run(
                    &dims,
                    starts,
                    counts,
                    element_size,
                    |src_off, out_off, len| {
                        let src = src_off as usize;
                        out[out_off..out_off + len].copy_from_slice(&data[src..src + len]);
                        Ok(())
                    },
                )?;
            }
            DataLayoutMessage::ChunkedV3 {
                chunk_dims,
                b_tree_address,
            } => {
                // Walk the v1 B-tree index reading only chunks that overlap the
                // selection, scattering each chunk∩selection into the slice
                // buffer; whatever no chunk covers is filled there. The
                // unconverted read keeps the post-filter conversion to exactly
                // once, in the wrappers.
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                self.read_chunked_btree_v1(
                    name,
                    real_chunk_dims,
                    *b_tree_address,
                    ChunkReadRequest {
                        pipeline: pipeline.as_ref(),
                        target: ChunkTarget::Slice { starts, counts },
                        fill_value: fill_value.as_deref(),
                        dst,
                    },
                    out,
                )?;
            }
            DataLayoutMessage::ChunkedV4 {
                chunk_dims,
                index_address,
                index_type,
                earray_params,
                single_chunk_filter,
                ..
            } => {
                // Same selection-aware chunk read for every v4 index kind
                // (single chunk, fixed/extensible array, B-tree v2): only
                // overlapping chunks are read and only their intersection with
                // the selection is scattered into the slice output.
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                self.read_chunked_v4(
                    name,
                    real_chunk_dims,
                    ChunkIndexDesc {
                        index_type: *index_type,
                        index_address: *index_address,
                        earray_params: earray_params.as_ref(),
                        single_chunk_filter: *single_chunk_filter,
                    },
                    ChunkReadRequest {
                        pipeline: pipeline.as_ref(),
                        target: ChunkTarget::Slice { starts, counts },
                        fill_value: fill_value.as_deref(),
                        dst,
                    },
                    out,
                )?;
            }
            DataLayoutMessage::Virtual { .. } => {
                // Stitch the full virtual image, then extract the
                // requested region from it. This reads more than the
                // selection strictly needs (no per-mapping intersection
                // against the caller's box), but a virtual dataset's data
                // is composed from other datasets rather than stored
                // contiguously, so there is no cheaper selective path
                // without duplicating `read_virtual_into`'s mapping walk
                // for a bounded region — correctness, not I/O pruning, is
                // what a VDS slice read needs here. The extraction below
                // writes every byte of `out`, so no pre-fill is needed on top
                // of the full image's own.
                let total = saturating_byte_len(&dims, element_size) as usize;
                let mut full = alloc_tiled_fill(total, fill_value.as_deref())?;
                self.read_virtual_into(name, &mut full, depth)?;
                for_each_contiguous_run(
                    &dims,
                    starts,
                    counts,
                    element_size,
                    |src_off, out_off, len| {
                        let src_off = src_off as usize;
                        out[out_off..out_off + len].copy_from_slice(&full[src_off..src_off + len]);
                        Ok(())
                    },
                )?;
            }
        }
        Ok(())
    }

    /// Read a strided hyperslab — h5py's stepped slicing (`ds[a:b:s]`) or
    /// the general `start`/`stride`/`count`/`block` form of
    /// `H5Sselect_hyperslab` — into a caller-provided buffer.
    ///
    /// One tuple per dimension: `start[d]` is the first index, `stride[d]`
    /// the spacing between selected blocks (`1` = the classic contiguous
    /// selection [`read_slice`](Self::read_slice) reads), `count[d]` how
    /// many blocks, and `block[d]` how many contiguous elements each block
    /// covers. `out` is row-major over `count[d] * block[d]` per dimension —
    /// exactly the shape h5py's stepped slicing produces — and `out.len()`
    /// must equal that times the element size.
    ///
    /// Built on the same selection-decomposition primitives the virtual
    /// dataset reader uses for its per-mapping scatter
    /// ([`Selection::resolve`], [`copy_matched_selections`]) rather than a
    /// second box walker: the requested selection and a densely-packed
    /// "output" selection sharing the same `count` hold the same elements in
    /// the same order, so pairing the two element streams places each one
    /// where h5py's stepped slicing puts it, and each source box is read with
    /// the ordinary per-layout selective read
    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
    ///
    /// The single owner of read-destination semantics for stepped selections:
    /// it defines every byte of `out` before returning `Ok`, so a typed caller
    /// reads straight into the vector it keeps rather than into a byte buffer
    /// it then copies.
    pub fn read_hyperslab_into(
        &mut self,
        name: &str,
        start: &[u64],
        stride: &[u64],
        count: &[u64],
        block: &[u64],
        out: &mut [u8],
    ) -> IoResult<()> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_hyperslab_into(&path, start, stride, count, block, out);
        }
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let datatype = info.datatype.clone();
        let element_size = datatype.element_size() as u64;
        let rank = dims.len();

        if start.len() != rank || stride.len() != rank || count.len() != rank || block.len() != rank
        {
            return Err(crate::io::IoError::InvalidState(
                "start/stride/count/block length must match dataset rank".into(),
            ));
        }
        if stride.contains(&0) {
            return Err(crate::io::IoError::InvalidState(
                "hyperslab stride must be nonzero in every dimension".into(),
            ));
        }
        let out_bytes = saturating_byte_len(
            &(0..rank)
                .map(|d| count[d].saturating_mul(block[d]))
                .collect::<Vec<u64>>(),
            element_size,
        );
        if out.len() as u64 != out_bytes {
            return Err(crate::io::IoError::InvalidState(format!(
                "read_hyperslab_into: buffer is {} bytes but selection needs {out_bytes}",
                out.len(),
            )));
        }

        let src_sel = Selection::Hyperslab {
            rank,
            form: Hyperslab::Regular(RegularHyperslab {
                start: start.to_vec(),
                stride: stride.to_vec(),
                count: count.to_vec(),
                block: block.to_vec(),
            }),
        };
        let out_dims: Vec<u64> = (0..rank)
            .map(|d| count[d].saturating_mul(block[d]))
            .collect();
        // A densely-packed selection sharing the same `count`: it holds the
        // same elements as `src_sel` in the same order, and its extent is the
        // output buffer, so pairing the two element streams lands every
        // source element at its h5py stepped-slicing position.
        let dst_sel = Selection::Hyperslab {
            rank,
            form: Hyperslab::Regular(RegularHyperslab {
                start: vec![0u64; rank],
                stride: block.to_vec(),
                count: count.to_vec(),
                block: block.to_vec(),
            }),
        };

        // `copy_matched_selections` is shared with the virtual-dataset
        // scatter, where a target selection may legitimately leave elements
        // untouched, so it carries no whole-output guarantee of its own: zero
        // first, exactly as the allocating form always did.
        fill_tiled_into(out, None);
        copy_matched_selections(
            |bstart, bcount, buf| {
                // `buf` is `copy_matched_selections`' fresh per-box buffer,
                // not the caller's `out`.
                self.read_slice_into_unconverted(name, bstart, bcount, buf, 0, ReadDst::Fresh)
            },
            &src_sel.resolve(&dims)?,
            &dst_sel.resolve(&out_dims)?,
            element_size,
            out,
        )?;
        Self::apply_post_filter_conversion(out, &datatype)
    }

    /// Read a list of coordinates in one call — h5py fancy indexing with a
    /// coordinate list (`H5S_SEL_POINTS`) — into a caller-provided buffer.
    ///
    /// `points[i]` is a `rank`-length coordinate; `out` holds one element per
    /// point, `element_size` bytes each, in the same order as `points` (point
    /// selection order is significant, see [`PointSelection`]), so `out.len()`
    /// must equal `points.len() * element_size`. Backed by
    /// [`Selection::Points`] and [`Selection::to_boxes`] — the same
    /// decomposition [`read_hyperslab_into`](Self::read_hyperslab_into) and
    /// the virtual dataset reader use — each point's 1-element box is read
    /// with the ordinary per-layout selective read
    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
    /// A 1-element box is already a flat `element_size`-byte run, so
    /// placing it needs no further run-decomposition
    /// ([`for_each_dual_run`] would degenerate to exactly this copy).
    ///
    /// One element-sized box per point covers the buffer exactly, so every
    /// byte of `out` is defined before it returns `Ok` and a typed caller
    /// reads straight into the vector it keeps.
    /// `dst` is the caller's destination fact ([`ReadDst`]) for `out` — a
    /// required argument rather than a defaulted wrapper, since unlike the
    /// raw/slice pair this read has no kept-buffer caller to assert it for.
    pub fn read_points_into(
        &mut self,
        name: &str,
        points: &[Vec<u64>],
        out: &mut [u8],
        dst: ReadDst,
    ) -> IoResult<()> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_points_into(&path, points, out, dst);
        }
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let datatype = info.datatype.clone();
        let element_size = datatype.element_size() as u64;
        let rank = dims.len();

        for p in points {
            if p.len() != rank {
                return Err(crate::io::IoError::InvalidState(format!(
                    "point coordinate has {} entries but the dataset has {} dimensions",
                    p.len(),
                    rank
                )));
            }
        }

        let sel = Selection::Points(PointSelection {
            rank,
            points: points.to_vec(),
        });
        let boxes = sel.to_boxes(&dims)?;

        let es = element_size as usize;
        if out.len() != points.len() * es {
            return Err(crate::io::IoError::InvalidState(format!(
                "read_points_into: buffer is {} bytes but {} points need {}",
                out.len(),
                points.len(),
                points.len() * es
            )));
        }
        for (i, (bstart, bcount)) in boxes.iter().enumerate() {
            self.read_slice_into_unconverted(
                name,
                bstart,
                bcount,
                &mut out[i * es..(i + 1) * es],
                0,
                dst,
            )?;
        }
        Self::apply_post_filter_conversion(out, &datatype)
    }

    /// Read one chunk's raw (still-filtered) bytes and its filter mask —
    /// the read half of `H5Dread_chunk` (h5py:
    /// `Dataset.id.read_direct_chunk`).
    ///
    /// `chunk_coords` is the chunk's position in the chunk grid, one
    /// coordinate per dimension counted in chunks (not elements) — the same
    /// addressing [`write_chunk_raw_at`](crate::Dataset::write_chunk_raw_at)
    /// uses on the write side. The bytes returned are exactly what is
    /// stored on disk: filtered/compressed if the dataset has a filter
    /// pipeline, with no decompression applied — the caller runs the
    /// pipeline itself (honoring the returned mask, which marks any filter
    /// this particular chunk skipped) if it wants decoded data.
    ///
    /// Resolved through whichever chunk index the dataset uses, reusing the
    /// same per-index decoders the full/slice chunked reader walks
    /// ([`collect_fa_chunk_entries`](Self::collect_fa_chunk_entries),
    /// [`collect_ea_chunk_entries`](Self::collect_ea_chunk_entries),
    /// [`collect_bt2_chunk_entries`](Self::collect_bt2_chunk_entries),
    /// [`collect_btree_v1_chunks`](Self::collect_btree_v1_chunks)) rather
    /// than a new index walker.
    ///
    /// `Err` when the dataset is not chunked, `chunk_coords` has the wrong
    /// rank, or the chunk at those coordinates has never been written.
    pub fn read_chunk_raw_at(
        &mut self,
        name: &str,
        chunk_coords: &[u64],
    ) -> IoResult<(Vec<u8>, u32)> {
        if self.external_edge(name).is_some() {
            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
            return owner.read_chunk_raw_at(&path, chunk_coords);
        }
        let info = self
            .dataset_info_local(name)
            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
        let dims = info.dataspace.dims.clone();
        let max_dims = info.dataspace.max_dims.clone();
        let element_size = info.datatype.element_size() as u64;
        let layout = info.layout.clone();
        let ndims = dims.len();

        if chunk_coords.len() != ndims {
            return Err(crate::io::IoError::InvalidState(format!(
                "chunk_coords has {} entries but the dataset has {} dimensions",
                chunk_coords.len(),
                ndims
            )));
        }

        let not_written = || {
            crate::io::IoError::InvalidState(format!(
                "chunk at coordinates {chunk_coords:?} has not been written"
            ))
        };

        match &layout {
            DataLayoutMessage::ChunkedV3 {
                chunk_dims,
                b_tree_address,
            } => {
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                if *b_tree_address == UNDEF_ADDR {
                    return Err(not_written());
                }
                let file_size = self.handle.file_size()?;
                let mut entries = Vec::new();
                self.collect_btree_v1_chunks(*b_tree_address, ndims, file_size, 0, &mut entries)?;
                for (offsets, addr, chunk_size, mask) in &entries {
                    if *addr == UNDEF_ADDR {
                        continue;
                    }
                    let mut scaled = Vec::with_capacity(ndims);
                    for d in 0..ndims {
                        scaled.push(offsets[d].checked_div(real_chunk_dims[d]).unwrap_or(0));
                    }
                    if scaled.as_slice() == chunk_coords {
                        return Ok((self.handle.read_at(*addr, *chunk_size as usize)?, *mask));
                    }
                }
                Err(not_written())
            }
            DataLayoutMessage::ChunkedV4 {
                chunk_dims,
                index_type,
                index_address,
                earray_params,
                single_chunk_filter,
                ..
            } => {
                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
                match index_type {
                    data_layout::ChunkIndexType::SingleChunk => {
                        if chunk_coords.iter().any(|&c| c != 0) {
                            return Err(crate::io::IoError::InvalidState(format!(
                                "chunk coordinates {chunk_coords:?} are outside the chunk \
                                 grid (0..1): this dataset has a single-chunk index"
                            )));
                        }
                        if *index_address == UNDEF_ADDR {
                            return Err(not_written());
                        }
                        match single_chunk_filter {
                            Some(scf) => Ok((
                                self.handle.read_at(*index_address, scf.nbytes as usize)?,
                                scf.filter_mask,
                            )),
                            None => {
                                let total = saturating_byte_len(&dims, element_size);
                                Ok((self.handle.read_at(*index_address, total as usize)?, 0))
                            }
                        }
                    }
                    data_layout::ChunkIndexType::Implicit => {
                        if *index_address == UNDEF_ADDR {
                            return Err(not_written());
                        }
                        let linear = crate::io::chunk_grid::linear_index(
                            &dims,
                            max_dims.as_deref(),
                            real_chunk_dims,
                            chunk_coords,
                        )?;
                        let chunk_bytes = saturating_byte_len(real_chunk_dims, element_size);
                        let addr = index_address.saturating_add(linear.saturating_mul(chunk_bytes));
                        Ok((self.handle.read_at(addr, chunk_bytes as usize)?, 0))
                    }
                    data_layout::ChunkIndexType::FixedArray => {
                        let linear = crate::io::chunk_grid::linear_index(
                            &dims,
                            max_dims.as_deref(),
                            real_chunk_dims,
                            chunk_coords,
                        )?;
                        let entries = self.collect_fa_chunk_entries(
                            real_chunk_dims,
                            ndims,
                            element_size,
                            *index_address,
                        )?;
                        match entries.get(linear as usize) {
                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
                                Ok((self.handle.read_at(addr, size as usize)?, mask))
                            }
                            _ => Err(not_written()),
                        }
                    }
                    data_layout::ChunkIndexType::ExtensibleArray => {
                        let params = earray_params.as_ref().ok_or_else(|| {
                            crate::io::IoError::InvalidState("missing earray params".into())
                        })?;
                        let linear = crate::io::chunk_grid::linear_index(
                            &dims,
                            max_dims.as_deref(),
                            real_chunk_dims,
                            chunk_coords,
                        )?;
                        let entries = self.collect_ea_chunk_entries(
                            *index_address,
                            params,
                            &dims,
                            max_dims.as_deref(),
                            real_chunk_dims,
                            element_size,
                        )?;
                        match entries.get(linear as usize) {
                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
                                Ok((self.handle.read_at(addr, size as usize)?, mask))
                            }
                            _ => Err(not_written()),
                        }
                    }
                    data_layout::ChunkIndexType::BTreeV2 => {
                        let entries = self.collect_bt2_chunk_entries(
                            real_chunk_dims,
                            ndims,
                            element_size,
                            *index_address,
                        )?;
                        match entries
                            .iter()
                            .find(|(_, _, scaled, _)| scaled.as_slice() == chunk_coords)
                        {
                            Some(&(addr, size, _, mask)) if addr != UNDEF_ADDR => {
                                Ok((self.handle.read_at(addr, size)?, mask))
                            }
                            _ => Err(not_written()),
                        }
                    }
                }
            }
            _ => Err(crate::io::IoError::InvalidState(
                "read_chunk_raw_at is only for chunked datasets".into(),
            )),
        }
    }
}

/// Adapts a `FileHandle` to the `BlockReader` trait used by the fractal-heap
/// walker, so heap blocks can be fetched from the open file.
///
/// The handle is shared, not exclusive: every read goes through
/// `FileHandle::read_at_most`, which takes `&self`, so the writer can walk a
/// structure it is about to free while holding only `&self` itself.
pub(crate) struct HandleBlockReader<'a> {
    pub(crate) handle: &'a FileHandle,
}

/// One object's attribute set, or the reason it could not be read whole.
///
/// [`AttributeEntry`] carries a per-attribute failure, which needs a name to
/// hang on. Two failures have none. A dense set is indexed by name *hash*, so
/// a heap or index that will not read yields no names at all; and an attribute
/// message too damaged to yield its own name cannot be listed under one
/// either. The only listing that can report those honestly is the object's, so
/// this type carries the object-scope reason beside the entries, and every
/// accessor that would present the set as whole returns the reason instead.
///
/// The entries are deliberately unreachable while the set is incomplete: the
/// only way out is [`Self::into_complete`], which refuses. That is what keeps
/// the writer from rebuilding an object header out of a partial set and
/// deleting the attributes it never saw.
#[derive(Debug, Clone, Default, PartialEq)]
pub struct ObjectAttributes {
    entries: Vec<AttributeEntry>,
    incomplete: Option<String>,
    /// This object's own attribute creation-order policy, from the header
    /// flag bits `H5Pget_attr_creation_order` reads back
    /// (`attribute_creation_order`) — a structural fact about the header,
    /// known even when the entries above are `incomplete`.
    creation_order: CreationOrder,
    /// This object's own compact-vs-dense attribute storage, from the
    /// attribute info message's heap address (or its absence) — likewise a
    /// structural fact known even when the entries are `incomplete`.
    storage: AttributeStorage,
}

impl ObjectAttributes {
    /// Record an attribute this collector could name.
    fn push(&mut self, entry: AttributeEntry) {
        self.entries.push(entry);
    }

    /// Record that part of the set could not be read. The first reason stands:
    /// it is the one that explains the earliest missing attributes.
    fn mark_incomplete(&mut self, reason: String) {
        if self.incomplete.is_none() {
            self.incomplete = Some(reason);
        }
    }

    /// Why this object's attributes cannot be listed, or `None` when the set
    /// is whole. Individual entries in a whole set may still be undecodable —
    /// [`AttributeEntry::unreadable_reason`] answers for those.
    pub fn unreadable_reason(&self) -> Option<&str> {
        self.incomplete.as_deref()
    }

    /// This object's own attribute creation-order policy —
    /// `H5Pget_attr_creation_order`'s answer, read off the object header's
    /// own flag bits rather than derived from the entries. Available even
    /// when the set is [`incomplete`](Self::unreadable_reason): it names
    /// nothing that failed to decode.
    pub fn creation_order(&self) -> CreationOrder {
        self.creation_order
    }

    /// This object's own compact-vs-dense attribute storage — h5py's
    /// `h5o.get_info(...).meta_size.attr.index_size` check, read off the
    /// attribute info message's heap address rather than derived from the
    /// entries. Available even when the set is
    /// [`incomplete`](Self::unreadable_reason).
    pub fn storage(&self) -> AttributeStorage {
        self.storage
    }

    /// This object's own attribute count as `H5Oget_info().num_attrs`
    /// reports it — the object header's count, not necessarily the same
    /// enumeration path as [`ordered_names`](Self::ordered_names).
    ///
    /// `H5O__attr_count_real` derives this from the attribute info message
    /// when the header carries one (the dense name-index record count, or
    /// the compact message count `H5O__attr_open_by_idx` already counted
    /// while building it) and from the raw attribute-message envelope count
    /// otherwise. Both reduce to the number of attributes this collector
    /// successfully names: a conformant writer creates the info message
    /// exactly when it has attributes to report through it, so a whole set's
    /// length already equals what libhdf5's header-count algorithm answers,
    /// without replaying its v1-header/v2-header branch here.
    pub fn header_count(&self, owner: &str) -> IoResult<u64> {
        Ok(self.complete(owner)?.len() as u64)
    }

    /// The entries, once the set is known to be whole.
    ///
    /// The sole route from an `ObjectAttributes` to an owned entry list. A
    /// caller that rewrites the object header — the append path — must take
    /// this route, so an unread set stops the rewrite instead of erasing the
    /// attributes behind it.
    pub(crate) fn into_complete(self, owner: &str) -> IoResult<Vec<AttributeEntry>> {
        match self.incomplete {
            Some(reason) => Err(incomplete_error(owner, &reason)),
            None => Ok(self.entries),
        }
    }

    /// The entries, once the set is known to be whole, borrowed.
    fn complete(&self, owner: &str) -> IoResult<&[AttributeEntry]> {
        match &self.incomplete {
            Some(reason) => Err(incomplete_error(owner, reason)),
            None => Ok(&self.entries),
        }
    }

    /// This object's attribute names, once the set is known to be whole, in
    /// the order h5py's default iteration produces them: creation order when
    /// the object tracks it, name order otherwise
    /// (`H5A__compact_cmp_corder`/`H5A__compact_cmp_name` for compact
    /// storage, the matching v2 B-tree index for dense) — never the physical
    /// order the entries happen to sit in, which `entries` otherwise
    /// preserves for the writer's rewrite path.
    pub(crate) fn ordered_names(&self, owner: &str) -> IoResult<Vec<String>> {
        let mut ordered: Vec<&AttributeEntry> = self.complete(owner)?.iter().collect();
        if !ordered.is_empty() && ordered.iter().all(|e| e.creation_index().is_some()) {
            ordered.sort_by_key(|e| e.creation_index());
        } else {
            ordered.sort_by(|a, b| a.name().cmp(b.name()));
        }
        Ok(ordered.into_iter().map(|e| e.name().to_string()).collect())
    }
}

/// The one wording for "this object's attributes are not all here".
///
/// `Unsupported`, the same variant an undecodable dataset message raises: the
/// name is in the listing and the content is out of reach, which is what the
/// variant is for. Wrapping a `FormatError` here instead would put the same
/// condition behind two different public variants.
fn incomplete_error(owner: &str, reason: &str) -> crate::io::IoError {
    crate::io::IoError::Unsupported(format!(
        "attributes of '{owner}' cannot be read whole: {reason}"
    ))
}

/// Every attribute attached to an object, whichever storage it uses.
///
/// This is the only place attributes are pulled off an object header — reader
/// and writer alike. Compact storage keeps them as `Attribute` messages in the
/// header itself; once an object crosses the phase-change threshold libhdf5
/// moves *all* of them into a fractal heap named by the `Attribute Info`
/// message and leaves no attribute message behind
/// (`H5Oattribute.c::H5O__attr_create`). Scanning only the messages therefore
/// reports zero attributes for a dense object: a silent loss on read, and a
/// silent deletion when the writer rebuilds that object's header from what it
/// collected.
///
/// An attribute this crate cannot decode is kept, named, with the reason
/// attached: a listing that omitted it would report a file that does not
/// contain it. What cannot be named at all — a damaged attribute message, an
/// attribute info message that will not decode, a dense set whose heap or name
/// index will not read — marks the whole set incomplete, so the object reports
/// the failure rather than a short list.
pub(crate) fn collect_object_attributes(
    handle: &mut FileHandle,
    ctx: &FormatContext,
    header: &ObjectHeader,
) -> ObjectAttributes {
    let mut attrs = ObjectAttributes {
        creation_order: header.attribute_creation_order(),
        ..ObjectAttributes::default()
    };
    // The message envelope carries a creation index only when the header says
    // the object tracks one; the field is not even encoded otherwise
    // (`H5O_SIZEOF_MSGHDR_OH`), so reading it as an index would report zero
    // for every attribute of an untracked object.
    let tracked = header.has_creation_order();
    for msg in &header.messages {
        match msg.msg_type {
            MSG_ATTRIBUTE => match AttributeEntry::parse(&msg.data, ctx) {
                Ok(entry) => {
                    attrs.push(entry.with_creation_index(tracked.then_some(msg.creation_index)))
                }
                Err(e) => attrs.mark_incomplete(format!("an attribute message is unreadable: {e}")),
            },
            MSG_ATTR_INFO => match AttributeInfoMessage::decode(&msg.data, ctx) {
                Ok((info, _)) => {
                    attrs.storage = if info.is_dense() {
                        AttributeStorage::Dense
                    } else {
                        AttributeStorage::Compact
                    };
                    let mut br = HandleBlockReader { handle };
                    match crate::format::dense_attr::read_dense_attributes(&info, ctx, &mut br) {
                        Ok(dense) => attrs.entries.extend(dense),
                        Err(e) => attrs
                            .mark_incomplete(format!("dense attribute storage is unreadable: {e}")),
                    }
                }
                Err(e) => {
                    attrs.mark_incomplete(format!("the attribute info message is unreadable: {e}"))
                }
            },
            _ => {}
        }
    }
    attrs
}

impl BlockReader for HandleBlockReader<'_> {
    fn read_block(&mut self, offset: u64, len: usize) -> crate::format::FormatResult<Vec<u8>> {
        // `read_at_most`, not `read_at`: a metadata block allocated at the end
        // of the file can be shorter on disk than its nominal size, and every
        // decoder re-checks the length it needs.
        self.handle.read_at_most(offset, len).map_err(|e| {
            crate::format::FormatError::InvalidData(format!(
                "metadata block read failed at {:#x}: {}",
                offset, e
            ))
        })
    }
}

#[cfg(test)]
mod tests {
    use super::*;
    use std::io::Write;

    /// Per-call unique temp path. PID + atomic counter avoids
    /// path collisions across concurrent cargo invocations and
    /// kernel-side flock release races.
    fn temp_path(name: &str) -> std::path::PathBuf {
        use std::sync::atomic::{AtomicU64, Ordering};
        static COUNTER: AtomicU64 = AtomicU64::new(0);
        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
        std::env::temp_dir().join(format!(
            "rust_hdf5_reader_test_{}_{}_{}.h5",
            name,
            std::process::id(),
            n
        ))
    }

    /// Write `bytes` to a fresh temp file and open a read handle on it.
    fn handle_over(name: &str, bytes: &[u8]) -> (std::path::PathBuf, FileHandle) {
        let path = temp_path(name);
        std::fs::File::create(&path)
            .unwrap()
            .write_all(bytes)
            .unwrap();
        let handle =
            FileHandle::open_read_with_locking(&path, crate::io::locking::FileLocking::Disabled)
                .unwrap();
        (path, handle)
    }

    /// The chunk-index cache. A decoded index MUST describe the file exactly
    /// as the catalog entry beside it does, and MUST NOT be served for an
    /// index address other than the one it was decoded from.
    /// [`Hdf5Reader::decoded_chunk_index`] is the single owner — the only
    /// reader and writer of the cache — and `DatasetTable` is what keeps a
    /// stale index unreachable: `entry_mut` is the one way to a mutable
    /// entry and it drops that entry's index first, where the `IndexMut` it
    /// replaced would have left it standing.
    mod chunk_index_cache {
        use super::*;

        /// `d` = `[0, 1, .., 7]` in two chunks of four, fixed extent, so the
        /// layout points at a fixed array a read has to walk.
        fn chunked_file(name: &str) -> (std::path::PathBuf, Hdf5Reader) {
            let path = temp_path(name);
            {
                let file = crate::H5File::create(&path).unwrap();
                let ds = file
                    .new_dataset::<u8>()
                    .shape([8usize])
                    .chunk(&[4])
                    .create("d")
                    .unwrap();
                ds.write_raw(&(0u8..8).collect::<Vec<u8>>()).unwrap();
                file.close().unwrap();
            }
            let reader = Hdf5Reader::open(&path).unwrap();
            (path, reader)
        }

        /// The address of the chunk index `d`'s layout points at.
        fn index_address(reader: &Hdf5Reader, pos: usize) -> u64 {
            match &reader.datasets[pos].layout {
                DataLayoutMessage::ChunkedV4 { index_address, .. } => *index_address,
                _ => panic!("expected a version-4 chunked layout"),
            }
        }

        #[test]
        fn a_read_leaves_its_decoded_index_for_the_next_read() {
            let (path, mut r) = chunked_file("index_cache_hit");
            let pos = r.dataset_position("d").unwrap();
            let addr = index_address(&r, pos);
            assert!(
                r.datasets.chunk_index(pos, addr).is_none(),
                "a reader that has read nothing holds a decoded index"
            );

            assert_eq!(r.read_slice("d", &[0], &[4]).unwrap(), vec![0u8, 1, 2, 3]);
            let cached = r
                .datasets
                .chunk_index(pos, addr)
                .expect("the read did not keep the index it decoded");
            assert_eq!(cached.entries.len(), 2, "one entry per chunk");
            assert_eq!(cached.coords, vec![0, 1], "slot 0 and slot 1 of the grid");

            // The second read answers from it and reaches the same bytes.
            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
            let _ = std::fs::remove_file(&path);
        }

        #[test]
        fn an_index_is_never_served_for_another_address() {
            let (path, mut r) = chunked_file("index_cache_addr");
            let pos = r.dataset_position("d").unwrap();
            let addr = index_address(&r, pos);
            r.read_slice("d", &[0], &[4]).unwrap();

            assert!(
                r.datasets.chunk_index(pos, addr.wrapping_add(1)).is_none(),
                "an index decoded from {addr:#x} answered for another address"
            );
            let _ = std::fs::remove_file(&path);
        }

        #[test]
        fn changing_an_entry_drops_the_index_decoded_against_it() {
            let (path, mut r) = chunked_file("index_cache_entry_mut");
            let pos = r.dataset_position("d").unwrap();
            let addr = index_address(&r, pos);
            r.read_slice("d", &[0], &[4]).unwrap();
            assert!(r.datasets.chunk_index(pos, addr).is_some());

            // What a caller reaches an entry mutably for — the virtual extent
            // resolution rewrites `dataspace.dims` — is what the coordinates
            // in a decoded index were built against.
            let _ = r.datasets.entry_mut(pos);
            assert!(
                r.datasets.chunk_index(pos, addr).is_none(),
                "a mutable entry left its decoded index behind"
            );
            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
            let _ = std::fs::remove_file(&path);
        }
    }

    /// The decompressed-chunk cache. An image MUST be served only for the
    /// stored bytes it was decoded from, MUST be kept only for a chunk the
    /// read left partly unconsumed, and MUST NOT outlive the decoded index
    /// that named its chunk — which it cannot, being a field of it, so the
    /// invalidation these tests exercise is the index's own (`entry_mut`, and
    /// the table rebuild a SWMR refresh does).
    mod chunk_image_cache {
        use super::*;

        /// `d` = a ramp of `n` f64 in chunks of `chunk`, deflated or not.
        #[cfg(feature = "deflate")]
        fn dataset(name: &str, n: usize, chunk: usize, deflate: bool) -> std::path::PathBuf {
            let path = temp_path(name);
            let data: Vec<f64> = (0..n).map(|i| (i % 251) as f64).collect();
            let file = crate::H5File::create(&path).unwrap();
            let mut b = file.new_dataset::<f64>().shape([n]).chunk(&[chunk]);
            if deflate {
                b = b.deflate(6);
            }
            b.create("d").unwrap().write_raw(&data).unwrap();
            file.close().unwrap();
            path
        }

        /// `d` = a ramp of `rows * cols` f64 in `chunk`-shaped chunks. A
        /// two-dimensional chunk is never one contiguous stretch of a read's
        /// output, so every chunk of it is staged and materialized — which is
        /// what leaves the cache free to keep it, where a one-dimensional
        /// whole-chunk read would have decoded straight into the output.
        fn dataset_2d(
            name: &str,
            rows: usize,
            cols: usize,
            chunk: [usize; 2],
            deflate: bool,
        ) -> std::path::PathBuf {
            let path = temp_path(name);
            let data: Vec<f64> = (0..rows * cols).map(|i| (i % 251) as f64).collect();
            let file = crate::H5File::create(&path).unwrap();
            let mut b = file
                .new_dataset::<f64>()
                .shape([rows, cols])
                .chunk(&chunk[..]);
            if deflate {
                b = b.deflate(6);
            }
            b.create("d").unwrap().write_raw(&data).unwrap();
            file.close().unwrap();
            path
        }

        /// How many images the reader holds for `d`.
        fn held(reader: &Hdf5Reader) -> usize {
            let pos = reader.dataset_position("d").unwrap();
            let DataLayoutMessage::ChunkedV4 { index_address, .. } = &reader.datasets[pos].layout
            else {
                panic!("expected a version-4 chunked layout");
            };
            match reader.datasets.chunk_index(pos, *index_address) {
                Some(index) => index.images.held(),
                None => 0,
            }
        }

        /// Four consecutive quarter-chunk slices are one chunk read four
        /// times: the first decodes it, the other three place it out of the
        /// image it left, and all four read what the file holds.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_partial_read_keeps_the_chunk_it_did_not_finish() {
            let path = dataset("images_partial", 4096, 1024, true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            let whole = {
                let mut fresh = Hdf5Reader::open(&path).unwrap();
                fresh.read_dataset_raw("d").unwrap()
            };

            for q in 0..4u64 {
                let got = r.read_slice("d", &[q * 256], &[256]).unwrap();
                let from = (q as usize) * 256 * 8;
                assert_eq!(got, whole[from..from + 256 * 8], "quarter {q}");
                assert_eq!(held(&r), 1, "quarter {q} left the wrong image count");
            }
            let _ = std::fs::remove_file(&path);
        }

        /// A whole-dataset read takes every chunk entire, so there is nothing
        /// a later read could want more of: it keeps nothing, and never pays
        /// to copy an image it is about to throw away.
        ///
        /// Two-dimensional on purpose. A one-dimensional full read decodes
        /// each chunk straight into the output and so never has an image to
        /// offer in the first place; here every chunk really is materialized,
        /// and only "the read consumed it" keeps it out of the cache.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_whole_dataset_read_keeps_nothing() {
            let path = dataset_2d("images_full", 256, 256, [64, 64], true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            r.read_dataset_raw("d").unwrap();
            assert_eq!(held(&r), 0, "a full read kept a chunk image");
            let _ = std::fs::remove_file(&path);
        }

        /// The extent cutting the last chunk short does not make a full read
        /// look partial: what that chunk has to give is what the extent
        /// reaches, and the read took all of it.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_whole_read_of_a_ragged_extent_keeps_nothing() {
            let path = dataset("images_ragged", 3000, 1024, true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            r.read_dataset_raw("d").unwrap();
            assert_eq!(held(&r), 0, "the ragged edge chunk was kept");
            let _ = std::fs::remove_file(&path);
        }

        /// An unfiltered chunk never reaches the cache, whatever a read does
        /// with it: its stored bytes are the dataset's bytes, so a partial
        /// read has nothing to inflate and nothing to save by inflating once.
        ///
        /// A single-column selection is the case that bites. Its runs are one
        /// element each, far too many for a positioned read apiece, so
        /// `read_chunk_runs_into` declines and every chunk really is read and
        /// materialized whole — leaving an image the cache would take if it
        /// were offered one.
        #[test]
        fn an_unfiltered_partial_read_keeps_nothing() {
            let path = dataset_2d("images_plain", 256, 256, [64, 64], false);
            let mut r = Hdf5Reader::open(&path).unwrap();
            let column = r.read_slice("d", &[0, 0], &[256, 1]).unwrap();
            assert_eq!(column.len(), 256 * 8);
            assert_eq!(held(&r), 0, "an unfiltered chunk reached the cache");
            let _ = std::fs::remove_file(&path);
        }

        /// The same single-column walk over a *filtered* dataset is the
        /// region-of-interest case the cache exists for: each column of chunks
        /// inflates once however many columns are read out of it.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_filtered_column_walk_reuses_each_chunk() {
            let path = dataset_2d("images_column", 256, 256, [64, 64], true);
            let mut warm = Hdf5Reader::open(&path).unwrap();
            for c in 0..4u64 {
                let mut cold = Hdf5Reader::open(&path).unwrap();
                assert_eq!(
                    warm.read_slice("d", &[0, c], &[256, 1]).unwrap(),
                    cold.read_slice("d", &[0, c], &[256, 1]).unwrap(),
                    "column {c} differed once served from the cache"
                );
            }
            // The four chunks of the first chunk-column, still held after the
            // fourth read of them.
            assert_eq!(held(&warm), 4, "the column walk re-inflated its chunks");
            let _ = std::fs::remove_file(&path);
        }

        /// A chunk whose image is larger than the whole budget would evict
        /// everything to hold one entry, so it is never kept.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_chunk_over_the_budget_is_never_kept() {
            let over = CHUNK_CACHE_BYTES / 8 + 1;
            let path = dataset("images_oversize", over * 2, over, true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            r.read_slice("d", &[0], &[1024]).unwrap();
            assert_eq!(held(&r), 0, "a chunk over the budget was kept");
            let _ = std::fs::remove_file(&path);
        }

        /// The budget's boundary, on the state machine itself: an image of
        /// exactly the budget is kept, and one byte more is refused — refused
        /// rather than admitted and then evicted, so that what the cache
        /// already holds survives an image that was never going to fit.
        #[test]
        fn an_image_over_the_budget_never_evicts_the_ones_under_it() {
            let cache = ChunkImageCache::default();
            let key = |addr| ChunkImageKey {
                addr,
                len: 1,
                mask: 0,
            };
            for i in 0..3 {
                cache.keep(key(i), vec![0u8; CHUNK_CACHE_BYTES / 4]);
            }
            assert_eq!(cache.held(), 3, "three quarter-budget images did not fit");

            cache.keep(key(9), vec![0u8; CHUNK_CACHE_BYTES + 1]);
            assert_eq!(
                cache.held(),
                3,
                "an image that cannot fit evicted the ones that did"
            );

            cache.keep(key(8), vec![0u8; CHUNK_CACHE_BYTES]);
            assert_eq!(
                cache.held(),
                1,
                "an image of exactly the budget was refused"
            );
        }

        /// The budget is bytes: eight partial reads of eight 256 KiB chunks
        /// leave the four the budget pays for, not eight.
        #[test]
        #[cfg(feature = "deflate")]
        fn the_cache_holds_no_more_than_its_byte_budget() {
            let chunk = 32 * 1024; // f64 -> 256 KiB
            let path = dataset("images_budget", chunk * 8, chunk, true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            for c in 0..8u64 {
                r.read_slice("d", &[c * chunk as u64], &[1024]).unwrap();
            }
            assert_eq!(
                held(&r),
                CHUNK_CACHE_BYTES / (chunk * 8),
                "the cache outgrew its byte budget"
            );
            let _ = std::fs::remove_file(&path);
        }

        /// Reaching an entry mutably drops the index decoded against it, and
        /// the images live in that index — so what a read cached against the
        /// old entry cannot be served against the new one.
        #[test]
        #[cfg(feature = "deflate")]
        fn changing_an_entry_drops_the_chunk_images() {
            let path = dataset("images_entry_mut", 4096, 1024, true);
            let mut r = Hdf5Reader::open(&path).unwrap();
            r.read_slice("d", &[0], &[256]).unwrap();
            assert_eq!(held(&r), 1, "the read kept no image to invalidate");

            let pos = r.dataset_position("d").unwrap();
            let _ = r.datasets.entry_mut(pos);
            assert_eq!(held(&r), 0, "a mutable entry left its chunk images behind");
            let _ = std::fs::remove_file(&path);
        }

        /// Every slice a cache-holding reader returns is the slice a reader
        /// that never cached anything returns, over slices that start inside
        /// a chunk, end inside one, and span several.
        #[test]
        #[cfg(feature = "deflate")]
        fn a_cached_chunk_places_what_a_cold_read_places() {
            let path = dataset("images_differential", 4096, 1024, true);
            let mut warm = Hdf5Reader::open(&path).unwrap();
            for &(start, count) in &[
                (0u64, 100u64),
                (100, 100),
                (900, 300),
                (1024, 1),
                (1500, 2000),
                (3000, 1096),
                (4095, 1),
                (0, 4096),
            ] {
                let mut cold = Hdf5Reader::open(&path).unwrap();
                assert_eq!(
                    warm.read_slice("d", &[start], &[count]).unwrap(),
                    cold.read_slice("d", &[start], &[count]).unwrap(),
                    "slice {start}+{count} differed once served from the cache"
                );
            }
            let _ = std::fs::remove_file(&path);
        }
    }

    /// `place_chunk_jobs` is the single owner of "every byte of a chunked
    /// read's output is defined": it skips the blanket fill only when
    /// [`planned_coverage`] proves the plan reaches every output byte, and
    /// fills whatever a plan or a short image leaves behind. The buffer these
    /// hand it is poisoned, so a byte no one defines shows up as `0xAA`.
    mod chunk_fill_plan {
        use super::*;

        /// dims [8] of 1-byte elements in chunks of 4: slot 0 holds `ABCD` at
        /// file offset 0, slot 1 holds `EFGH` at offset 4.
        const DIMS: [u64; 1] = [8];
        const CHUNKS: [u64; 1] = [4];
        const FILL: [u8; 1] = [0x7E];

        fn place(
            name: &str,
            jobs: Vec<Option<ChunkReadJob>>,
            coords: &[u64],
        ) -> (std::path::PathBuf, Vec<u8>) {
            let (path, handle) = handle_over(name, b"ABCDEFGH");
            let geo = ChunkOutputGeometry {
                dims: &DIMS,
                chunk_dims: &CHUNKS,
                element_size: 1,
            };
            let mut out = vec![0xAAu8; 8];
            place_chunk_jobs(
                &handle,
                jobs,
                coords,
                ChunkReadRequest {
                    pipeline: None,
                    target: ChunkTarget::Full,
                    fill_value: Some(&FILL),
                    dst: ReadDst::Fresh,
                },
                &geo,
                None,
                &mut out,
            )
            .unwrap();
            (path, out)
        }

        fn job(addr: u64, len: usize) -> Option<ChunkReadJob> {
            Some(ChunkReadJob {
                addr,
                len,
                at_most: true,
                mask: 0,
            })
        }

        /// The owner path: a plan that tiles the output places every byte, so
        /// nothing is filled and nothing is left poisoned.
        #[test]
        fn a_plan_that_covers_the_output_leaves_no_byte_to_fill() {
            let geo = ChunkOutputGeometry {
                dims: &DIMS,
                chunk_dims: &CHUNKS,
                element_size: 1,
            };
            let zeros = [0u64];
            let placement = ChunkPlacement::resolve(&geo, ChunkTarget::Full, &zeros);
            let jobs = vec![job(0, 4), job(4, 4)];
            assert_eq!(
                planned_coverage(&placement, &jobs, &[0, 1]),
                8,
                "a tiling plan must measure as covering the whole output"
            );
            let (path, out) = place("cover_full", jobs, &[0, 1]);
            assert_eq!(out, b"ABCDEFGH");
            let _ = std::fs::remove_file(path);
        }

        /// A slot the plan never reaches reads back as the tiled fill value,
        /// which is what the blanket pre-fill used to guarantee.
        #[test]
        fn a_slot_the_plan_skips_reads_back_as_fill() {
            let (path, out) = place("cover_gap", vec![job(0, 4), None], &[0, 1]);
            assert_eq!(out, b"ABCD~~~~");
            let _ = std::fs::remove_file(path);
        }

        /// A corrupt index naming one chunk-grid slot twice must not be
        /// mistaken for a plan that covers two slots: counting the duplicate
        /// once leaves the coverage short, so the fill still goes down and the
        /// slot no entry named reads as fill rather than as poison.
        #[test]
        fn duplicate_chunk_coordinates_still_leave_defined_output() {
            let (path, out) = place("cover_dup", vec![job(0, 4), job(0, 4)], &[0, 0]);
            assert_eq!(out, b"ABCD~~~~");
            let _ = std::fs::remove_file(path);
        }

        /// A chunk whose stored image is shorter than the read box needs it to
        /// be places no run at all; the plan counted it, so the shortfall is
        /// filled after placement instead of before it.
        #[test]
        fn a_chunk_read_short_has_its_runs_filled_after_placement() {
            let (path, out) = place("cover_short", vec![job(0, 4), job(4, 2)], &[0, 1]);
            assert_eq!(out, b"ABCD~~~~");
            let _ = std::fs::remove_file(path);
        }

        /// A chunk the index places a few bytes short of `u64::MAX` has runs
        /// whose ends do not fit in a `u64`. Two rows of one chunk are the
        /// second run that asks whether it continues the first; the answer
        /// comes from the read at that address (fill for a read that may
        /// come up short, an error for one that may not), not from the
        /// arithmetic.
        #[test]
        fn a_chunk_address_near_u64_max_fails_the_read_not_the_arithmetic() {
            let (path, handle) = handle_over("cover_wrap", b"ABEFCDGH");
            let dims = [2u64, 4];
            let chunks = [2u64, 2];
            let geo = ChunkOutputGeometry {
                dims: &dims,
                chunk_dims: &chunks,
                element_size: 1,
            };
            let run = |at_most: bool, out: &mut [u8]| {
                place_chunk_jobs(
                    &handle,
                    vec![
                        job(0, 4),
                        Some(ChunkReadJob {
                            addr: u64::MAX - 1,
                            len: 4,
                            at_most,
                            mask: 0,
                        }),
                    ],
                    &[0, 0, 0, 1],
                    ChunkReadRequest {
                        pipeline: None,
                        target: ChunkTarget::Full,
                        fill_value: Some(&FILL),
                        dst: ReadDst::Fresh,
                    },
                    &geo,
                    None,
                    out,
                )
            };
            let mut out = vec![0xAAu8; 8];
            run(true, &mut out).unwrap();
            assert_eq!(out, b"AB~~EF~~");
            let mut out = vec![0xAAu8; 8];
            run(false, &mut out).expect_err("an exact read at u64::MAX - 1 cannot succeed");
            let _ = std::fs::remove_file(path);
        }
    }

    /// Helper: write a little-endian u64 truncated to `n` bytes.
    fn write_le(buf: &mut Vec<u8>, value: u64, n: usize) {
        buf.extend_from_slice(&value.to_le_bytes()[..n]);
    }

    /// Build a minimal v0 HDF5 file in memory with one dataset containing
    /// `dataset_data`. Returns the complete file bytes.
    ///
    /// The file structure is:
    /// - Superblock v0 with root group STE
    /// - Root group object header (v1) with symbol table message
    /// - Local heap (header + data) with dataset name
    /// - B-tree v1 (group, leaf) pointing to one SNOD
    /// - SNOD with one entry for the dataset
    /// - Dataset object header (v1) with dataspace, datatype, layout messages
    /// - Raw dataset data (contiguous)
    fn build_v0_file(dataset_name: &str, dims: &[u64], data: &[u8]) -> Vec<u8> {
        let sa: usize = 8; // sizeof_addr
        let ss: usize = 8; // sizeof_size
        let ndims = dims.len();
        let element_size = data.len() as u64 / dims.iter().product::<u64>();

        // We'll lay out the file regions in order, computing offsets as we go.
        let mut file = Vec::new();

        // ---- Plan layout offsets ----
        // We need to know the addresses before writing, so let's compute them.
        // Superblock: starts at 0
        let sb_size = 8 + 8 + 4 + 4 * sa + (ss + sa + 4 + 4 + 16); // sig + header + flags + 4 addrs + STE
                                                                   // Pad to 8-byte alignment
        let sb_size_aligned = (sb_size + 7) & !7;

        // Root group object header (v1): after superblock
        let root_ohdr_addr = sb_size_aligned as u64;
        // The root ohdr contains a symbol table message (type 0x11):
        //   btree_addr(8) + heap_addr(8) = 16 bytes
        // v1 message wire format: type(2) + size(2) + flags(1) + reserved(3) + data
        let stab_msg_data_size = 2 * sa; // btree + heap addr
        let stab_msg_wire = 8 + stab_msg_data_size;
        let stab_msg_wire_aligned = (stab_msg_wire + 7) & !7;
        let root_ohdr_data_size = stab_msg_wire_aligned;
        let root_ohdr_total = 16 + root_ohdr_data_size; // v1 16-byte prefix + messages
        let root_ohdr_total_aligned = (root_ohdr_total + 7) & !7;

        // Local heap header: after root ohdr
        let heap_hdr_addr = root_ohdr_addr + root_ohdr_total_aligned as u64;
        let heap_hdr_size = 4 + 1 + 3 + ss + ss + sa;
        let heap_hdr_size_aligned = (heap_hdr_size + 7) & !7;

        // Local heap data: after heap header
        let heap_data_addr = heap_hdr_addr + heap_hdr_size_aligned as u64;
        // Data: empty string at offset 0 (for root), then dataset_name at offset 1
        let name_bytes = dataset_name.as_bytes();
        let heap_data_content_size = 1 + name_bytes.len() + 1; // \0 + name + \0
        let heap_data_size = (heap_data_content_size + 7) & !7;

        // B-tree v1 node: after heap data
        let btree_addr = heap_data_addr + heap_data_size as u64;
        // B-tree header: TREE(4) + type(1) + level(1) + entries_used(2) + left(sa) + right(sa)
        // Plus interleaved keys/children: key[0](ss), child[0](sa), key[1](ss)
        let btree_size = 4 + 1 + 1 + 2 + 2 * sa + 2 * ss + sa;
        let btree_size_aligned = (btree_size + 7) & !7;

        // SNOD: after B-tree
        let snod_addr = btree_addr + btree_size_aligned as u64;
        // SNOD header: SNOD(4) + version(1) + reserved(1) + num_symbols(2)
        // + 1 entry: name_offset(ss) + obj_header_addr(sa) + cache_type(4) + reserved(4) + scratch(16)
        let entry_size = ss + sa + 4 + 4 + 16;
        let snod_size = 8 + entry_size;
        let snod_size_aligned = (snod_size + 7) & !7;

        // Dataset object header (v1): after SNOD
        let ds_ohdr_addr = snod_addr + snod_size_aligned as u64;
        // Messages: dataspace(0x01), datatype(0x03), data_layout(0x08)

        // Dataspace v1: version(1) + ndims(1) + flags(1) + reserved(1) + reserved(4) + ndims*ss
        let ds_msg_data_size = 8 + ndims * ss;
        let ds_msg_wire = 8 + ds_msg_data_size;
        let ds_msg_wire_aligned = (ds_msg_wire + 7) & !7;

        // Datatype: for integer types, 12 bytes
        // Use i32: class=0, version=1, size=4, bit_offset=0, bit_precision=32, signed
        let dt_msg_data_size = 12;
        let dt_msg_wire = 8 + dt_msg_data_size;
        let dt_msg_wire_aligned = (dt_msg_wire + 7) & !7;

        // Data layout v3 contiguous: version(1) + class(1) + addr(sa) + size(ss)
        let dl_msg_data_size = 2 + sa + ss;
        let dl_msg_wire = 8 + dl_msg_data_size;
        let dl_msg_wire_aligned = (dl_msg_wire + 7) & !7;

        let ds_ohdr_data_size = ds_msg_wire_aligned + dt_msg_wire_aligned + dl_msg_wire_aligned;
        let ds_ohdr_total = 16 + ds_ohdr_data_size; // v1 16-byte prefix
        let ds_ohdr_total_aligned = (ds_ohdr_total + 7) & !7;

        // Raw data: after dataset object header
        let raw_data_addr = ds_ohdr_addr + ds_ohdr_total_aligned as u64;
        let raw_data_size = data.len();

        let eof = raw_data_addr + raw_data_size as u64;

        // ---- Write the file ----

        // 1. Superblock v0
        let sig: [u8; 8] = [0x89, 0x48, 0x44, 0x46, 0x0d, 0x0a, 0x1a, 0x0a];
        file.extend_from_slice(&sig);
        file.push(0); // version 0
        file.push(0); // free-space version
        file.push(0); // root group STE version
        file.push(0); // reserved
        file.push(0); // shared header version
        file.push(sa as u8); // sizeof_addr
        file.push(ss as u8); // sizeof_size
        file.push(0); // reserved
        file.extend_from_slice(&4u16.to_le_bytes()); // sym_leaf_k
        file.extend_from_slice(&32u16.to_le_bytes()); // btree_internal_k
        file.extend_from_slice(&0u32.to_le_bytes()); // file_consistency_flags
                                                     // base_addr
        write_le(&mut file, 0, sa);
        // extension_addr = UNDEF
        write_le(&mut file, UNDEF_ADDR, sa);
        // eof_addr
        write_le(&mut file, eof, sa);
        // driver_info_addr = UNDEF
        write_le(&mut file, UNDEF_ADDR, sa);
        // Root group STE:
        write_le(&mut file, 0, ss); // name_offset
        write_le(&mut file, root_ohdr_addr, sa); // obj_header_addr
        file.extend_from_slice(&1u32.to_le_bytes()); // cache_type = 1 (stab)
        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
                                                     // scratch pad: btree_addr + heap_addr
        write_le(&mut file, btree_addr, sa);
        write_le(&mut file, heap_hdr_addr, sa);
        // Pad superblock
        while file.len() < sb_size_aligned {
            file.push(0);
        }

        // 2. Root group object header (v1, 16-byte prefix)
        assert_eq!(file.len(), root_ohdr_addr as usize);
        file.push(1); // version
        file.push(0); // reserved
        file.extend_from_slice(&1u16.to_le_bytes()); // num_messages = 1
        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
        file.extend_from_slice(&(root_ohdr_data_size as u32).to_le_bytes());
        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)
                                           // Symbol table message (type 0x0011)
        file.extend_from_slice(&0x0011u16.to_le_bytes()); // type
        file.extend_from_slice(&(stab_msg_data_size as u16).to_le_bytes()); // size
        file.push(0); // flags
        file.extend_from_slice(&[0u8; 3]); // reserved
        write_le(&mut file, btree_addr, sa);
        write_le(&mut file, heap_hdr_addr, sa);
        // Pad
        while file.len() < (root_ohdr_addr as usize + root_ohdr_total_aligned) {
            file.push(0);
        }

        // 3. Local heap header
        assert_eq!(file.len(), heap_hdr_addr as usize);
        file.extend_from_slice(b"HEAP");
        file.push(0); // version
        file.extend_from_slice(&[0u8; 3]); // reserved
        write_le(&mut file, heap_data_size as u64, ss); // data_size
        write_le(&mut file, u64::MAX, ss); // free_list_offset (none)
        write_le(&mut file, heap_data_addr, sa); // data_addr
        while file.len() < (heap_hdr_addr as usize + heap_hdr_size_aligned) {
            file.push(0);
        }

        // 4. Local heap data
        assert_eq!(file.len(), heap_data_addr as usize);
        file.push(0); // offset 0: empty string (root self-reference)
        file.extend_from_slice(name_bytes); // offset 1: dataset name
        file.push(0); // null terminator
        while file.len() < (heap_data_addr as usize + heap_data_size) {
            file.push(0);
        }

        // 5. B-tree v1 (leaf, 1 entry)
        assert_eq!(file.len(), btree_addr as usize);
        file.extend_from_slice(b"TREE");
        file.push(0); // type = group
        file.push(0); // level = leaf
        file.extend_from_slice(&1u16.to_le_bytes()); // entries_used = 1
        write_le(&mut file, UNDEF_ADDR, sa); // left sibling
        write_le(&mut file, UNDEF_ADDR, sa); // right sibling
                                             // key[0] = 0 (first name offset)
        write_le(&mut file, 0, ss);
        // child[0] = snod_addr
        write_le(&mut file, snod_addr, sa);
        // key[1] = dataset name offset (after root)
        write_le(&mut file, 1, ss);
        while file.len() < (btree_addr as usize + btree_size_aligned) {
            file.push(0);
        }

        // 6. SNOD with 1 entry
        assert_eq!(file.len(), snod_addr as usize);
        file.extend_from_slice(b"SNOD");
        file.push(1); // version
        file.push(0); // reserved
        file.extend_from_slice(&1u16.to_le_bytes()); // num_symbols = 1
                                                     // Entry: dataset
        write_le(&mut file, 1, ss); // name_offset = 1 (index into local heap)
        write_le(&mut file, ds_ohdr_addr, sa); // obj_header_addr
        file.extend_from_slice(&0u32.to_le_bytes()); // cache_type = 0 (not a group)
        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
        file.extend_from_slice(&[0u8; 16]); // scratch pad (unused)
        while file.len() < (snod_addr as usize + snod_size_aligned) {
            file.push(0);
        }

        // 7. Dataset object header (v1, 16-byte prefix)
        assert_eq!(file.len(), ds_ohdr_addr as usize);
        file.push(1); // version
        file.push(0); // reserved
        file.extend_from_slice(&3u16.to_le_bytes()); // num_messages = 3
        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
        file.extend_from_slice(&(ds_ohdr_data_size as u32).to_le_bytes());
        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)

        // Message 1: Dataspace (type 0x01) - version 1
        file.extend_from_slice(&0x0001u16.to_le_bytes());
        file.extend_from_slice(&(ds_msg_data_size as u16).to_le_bytes());
        file.push(0); // flags
        file.extend_from_slice(&[0u8; 3]); // reserved
                                           // Dataspace v1 payload:
        file.push(1); // version = 1
        file.push(ndims as u8);
        file.push(0); // flags (no max dims)
        file.push(0); // reserved
        file.extend_from_slice(&[0u8; 4]); // reserved (4 bytes)
        for &d in dims {
            write_le(&mut file, d, ss);
        }
        // Pad message
        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned;
        while file.len() < target {
            file.push(0);
        }

        // Message 2: Datatype (type 0x03) - i32
        file.extend_from_slice(&0x0003u16.to_le_bytes());
        file.extend_from_slice(&(dt_msg_data_size as u16).to_le_bytes());
        file.push(0); // flags
        file.extend_from_slice(&[0u8; 3]); // reserved
                                           // Datatype payload: class=0 (fixed point), version=1
        file.push(0x10); // class(0) | version(1)<<4
        file.push(0x08); // byte_order=LE, signed=true (bit 3)
        file.push(0); // flags byte 1
        file.push(0); // flags byte 2
        file.extend_from_slice(&(element_size as u32).to_le_bytes()); // element size
        file.extend_from_slice(&0u16.to_le_bytes()); // bit_offset
        file.extend_from_slice(&((element_size * 8) as u16).to_le_bytes()); // bit_precision
        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned + dt_msg_wire_aligned;
        while file.len() < target {
            file.push(0);
        }

        // Message 3: Data Layout (type 0x08) - contiguous v3
        file.extend_from_slice(&0x0008u16.to_le_bytes());
        file.extend_from_slice(&(dl_msg_data_size as u16).to_le_bytes());
        file.push(0); // flags
        file.extend_from_slice(&[0u8; 3]); // reserved
                                           // Data layout payload:
        file.push(3); // version = 3
        file.push(1); // class = contiguous
        write_le(&mut file, raw_data_addr, sa); // address
        write_le(&mut file, raw_data_size as u64, ss); // size
        let target = ds_ohdr_addr as usize + ds_ohdr_total_aligned;
        while file.len() < target {
            file.push(0);
        }

        // 8. Raw data
        assert_eq!(file.len(), raw_data_addr as usize);
        file.extend_from_slice(data);

        assert_eq!(file.len(), eof as usize);
        file
    }

    #[test]
    fn test_read_v0_file_with_one_dataset() {
        let dims = [3u64, 4];
        let values: Vec<i32> = (0..12).collect();
        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();

        let file_bytes = build_v0_file("my_dataset", &dims, &raw_data);

        // Write to a temp file
        let path = temp_path("v0_reader");
        {
            let mut f = std::fs::File::create(&path).unwrap();
            f.write_all(&file_bytes).unwrap();
            f.sync_all().unwrap();
        }

        // Read it back
        let mut reader = Hdf5Reader::open(&path).unwrap();
        let names = reader.dataset_names();
        assert_eq!(names, vec!["my_dataset"]);

        let shape = reader.dataset_shape("my_dataset").unwrap();
        assert_eq!(shape, vec![3, 4]);

        let data = reader.read_dataset_raw("my_dataset").unwrap();
        assert_eq!(data, raw_data);

        // Verify the values
        let read_values: Vec<i32> = data
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(read_values, values);

        std::fs::remove_file(&path).ok();
    }

    #[test]
    fn test_read_v0_file_1d_dataset() {
        let dims = [5u64];
        let values: Vec<i32> = vec![100, 200, 300, 400, 500];
        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();

        let file_bytes = build_v0_file("data_1d", &dims, &raw_data);

        let path = temp_path("v0_1d");
        {
            let mut f = std::fs::File::create(&path).unwrap();
            f.write_all(&file_bytes).unwrap();
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();
        assert_eq!(reader.dataset_names(), vec!["data_1d"]);
        assert_eq!(reader.dataset_shape("data_1d").unwrap(), vec![5]);

        let data = reader.read_dataset_raw("data_1d").unwrap();
        let read_values: Vec<i32> = data
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(read_values, values);

        std::fs::remove_file(&path).ok();
    }

    #[test]
    fn test_detect_v2v3_still_works() {
        // Verify that opening a v3 file written by our writer still works
        let path = temp_path("detect_v3");
        {
            use crate::io::writer::Hdf5Writer;
            let writer = Hdf5Writer::create(&path).unwrap();
            let datatype = crate::format::messages::datatype::DatatypeMessage::i32_type();
            let idx = writer.create_dataset("test", datatype, &[4]).unwrap();
            let data = [1i32, 2, 3, 4];
            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
            writer.write_dataset_raw(idx, &raw).unwrap();
            writer.close().unwrap();
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();
        assert_eq!(reader.dataset_names(), vec!["test"]);
        let shape = reader.dataset_shape("test").unwrap();
        assert_eq!(shape, vec![4]);

        let data = reader.read_dataset_raw("test").unwrap();
        let vals: Vec<i32> = data
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(vals, vec![1, 2, 3, 4]);

        std::fs::remove_file(&path).ok();
    }

    /// Collect the (src_off, out_off, len) runs the coalescer emits.
    fn collect_runs(
        dims: &[u64],
        starts: &[u64],
        counts: &[u64],
        es: u64,
    ) -> Vec<(u64, usize, usize)> {
        let mut v = Vec::new();
        for_each_contiguous_run(dims, starts, counts, es, |s, o, l| {
            v.push((s, o, l));
            Ok(())
        })
        .unwrap();
        v
    }

    #[test]
    fn coalesce_1d_is_single_run() {
        // 1-D selection is always one contiguous run.
        assert_eq!(collect_runs(&[10], &[2], &[3], 1), vec![(2, 0, 3)]);
        // element_size scales offsets and length.
        assert_eq!(collect_runs(&[10], &[2], &[3], 4), vec![(8, 0, 12)]);
    }

    #[test]
    fn coalesce_full_last_dim_merges_into_one_run() {
        // 2-D, full last dim => the whole [r0:r1, :] block is one run.
        // dims=[4,5], select rows 1..3, all 5 columns.
        assert_eq!(collect_runs(&[4, 5], &[1, 0], &[2, 5], 1), vec![(5, 0, 10)]);
    }

    #[test]
    fn coalesce_partial_last_dim_keeps_one_run_per_row() {
        // 2-D, partial last dim => no merge; one run per selected row.
        // dims=[4,5] strides=[5,1]; select rows 1..3, cols 1..4.
        assert_eq!(
            collect_runs(&[4, 5], &[1, 1], &[2, 3], 1),
            vec![(6, 0, 3), (11, 3, 3)]
        );
        // Same shape, element_size=4: strides=[20,4], inner_base=4.
        assert_eq!(
            collect_runs(&[4, 5], &[1, 1], &[2, 3], 4),
            vec![(24, 0, 12), (44, 12, 12)]
        );
    }

    #[test]
    fn coalesce_3d_reported_case_one_run_per_outer_index() {
        // The reported workload: [:, r0:r1, :] of [nproj, nz, nx].
        // dims=[3,4,5] strides=[20,5,1]; select all of dim0, rows 1..3 of
        // dim1, all of dim2. Last dim full => merge dim1+dim2; dim1 partial
        // => one run per dim0 index (3 runs, not 3*2=6 rows).
        assert_eq!(
            collect_runs(&[3, 4, 5], &[0, 1, 0], &[3, 2, 5], 1),
            vec![(5, 0, 10), (25, 10, 10), (45, 20, 10)]
        );
    }

    #[test]
    fn coalesce_3d_full_inner_dims_is_single_run() {
        // [r0:r1, :, :] => both inner dims full => one contiguous run.
        // dims=[3,4,5] strides=[20,5,1]; select rows 1..3 of dim0.
        assert_eq!(
            collect_runs(&[3, 4, 5], &[1, 0, 0], &[2, 4, 5], 1),
            vec![(20, 0, 40)]
        );
    }

    /// Build a contiguous i32 dataset and verify `read_slice` returns the
    /// correct bytes for both coalesced and non-coalesced selections.
    #[test]
    fn read_slice_contiguous_3d_matches_naive_extraction() {
        let dims = [3u64, 4, 5];
        let total: usize = (dims[0] * dims[1] * dims[2]) as usize;
        let values: Vec<i32> = (0..total as i32).collect();
        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
        let file_bytes = build_v0_file("vol", &dims, &raw_data);

        let path = temp_path("slice_3d_contig");
        {
            let mut f = std::fs::File::create(&path).unwrap();
            f.write_all(&file_bytes).unwrap();
            f.sync_all().unwrap();
        }
        let mut reader = Hdf5Reader::open(&path).unwrap();

        // Naive row-major extraction for an arbitrary [starts, counts).
        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
            let mut out = Vec::new();
            for i in 0..counts[0] {
                for j in 0..counts[1] {
                    for k in 0..counts[2] {
                        let gi = starts[0] + i;
                        let gj = starts[1] + j;
                        let gk = starts[2] + k;
                        out.push(values[(gi * dims[1] * dims[2] + gj * dims[2] + gk) as usize]);
                    }
                }
            }
            out
        };
        let decode = |raw: Vec<u8>| -> Vec<i32> {
            raw.as_chunks::<4>()
                .0
                .iter()
                .map(|c| i32::from_le_bytes(*c))
                .collect()
        };

        // Mix of coalesced and non-coalesced selections.
        let cases: &[([u64; 3], [u64; 3])] = &[
            ([0, 1, 0], [3, 2, 5]), // [:, 1:3, :]  -> coalesced (3 runs)
            ([1, 0, 0], [2, 4, 5]), // [1:3, :, :]  -> single run
            ([0, 0, 1], [3, 4, 3]), // [:, :, 1:4]  -> partial last dim, no merge
            ([1, 2, 1], [2, 2, 4]), // interior block, partial all dims
            ([0, 0, 0], [3, 4, 5]), // whole dataset -> single run
            ([2, 3, 4], [1, 1, 1]), // single element
        ];
        for &(starts, counts) in cases {
            let got = decode(reader.read_slice("vol", &starts, &counts).unwrap());
            assert_eq!(
                got,
                expect(starts, counts),
                "slice starts={starts:?} counts={counts:?}"
            );
        }

        let _ = std::fs::remove_file(&path);
    }

    /// Run the collector against a header built by hand. The handle is only
    /// touched when a message sends the collector to the heap, which none of
    /// these do, so an empty file is enough of a file.
    fn collect_from(
        messages: Vec<crate::format::object_header::ObjectHeaderMessage>,
    ) -> Result<Vec<String>, String> {
        let path = temp_path("collect");
        std::fs::File::create(&path).unwrap();
        let mut handle = FileHandle::open_read(&path).unwrap();
        let ctx = FormatContext {
            sizeof_addr: 8,
            sizeof_size: 8,
        };
        let header = ObjectHeader {
            flags: 0x02,
            times: None,
            messages,
        };
        let attrs = collect_object_attributes(&mut handle, &ctx, &header);
        drop(handle);
        let _ = std::fs::remove_file(&path);
        attrs
            .complete("obj")
            .map(|e| e.iter().map(|a| a.name().to_string()).collect())
            .map_err(|e| e.to_string())
    }

    fn msg(msg_type: u8, data: Vec<u8>) -> crate::format::object_header::ObjectHeaderMessage {
        crate::format::object_header::ObjectHeaderMessage {
            msg_type,
            flags: 0,
            data,
            creation_index: 0,
        }
    }

    /// An attribute info message that will not decode takes the object's whole
    /// attribute set with it — the dense storage it names is where the
    /// attributes are. The listing must say so rather than come back short.
    #[test]
    fn an_undecodable_attribute_info_message_fails_the_listing() {
        // Version 9: `H5O_AINFO_VERSION_0` is the only one that exists.
        let err = collect_from(vec![msg(MSG_ATTR_INFO, vec![9, 0])]).unwrap_err();
        assert!(
            err.contains("attributes of 'obj' cannot be read whole")
                && err.contains("attribute info message"),
            "{err}"
        );
    }

    /// An attribute message damaged past its own name has no name to be listed
    /// under, so it too is an object-level failure — the one case
    /// [`AttributeEntry::parse`] cannot name.
    #[test]
    fn an_unnameable_attribute_message_fails_the_listing() {
        let err = collect_from(vec![msg(MSG_ATTRIBUTE, vec![1, 0, 0])]).unwrap_err();
        assert!(
            err.contains("attributes of 'obj' cannot be read whole")
                && err.contains("attribute message"),
            "{err}"
        );
    }

    /// The failure is the object's, not one name's: a set that reads whole
    /// still lists, and a compact attribute info message is not dense storage.
    #[test]
    fn a_whole_attribute_set_still_lists() {
        let ctx = FormatContext {
            sizeof_addr: 8,
            sizeof_size: 8,
        };
        let ainfo = crate::format::messages::attr_info::AttributeInfoMessage::compact();
        let names = collect_from(vec![msg(MSG_ATTR_INFO, ainfo.encode(&ctx))]).unwrap();
        assert!(names.is_empty(), "{names:?}");
    }

    /// A space-padded fixed-length string attribute reads back without its
    /// padding: `H5T__conv_s_s` ends the value after the last non-space byte,
    /// and nothing else in the element marks where it stops.
    #[test]
    fn fixed_string_attr_value_honors_the_declared_pad() {
        use crate::format::messages::dataspace::DataspaceMessage;
        use crate::format::messages::datatype::DatatypeMessage;

        let attr = |padding: u8, data: &[u8]| AttributeMessage {
            name: "units".to_string(),
            datatype: DatatypeMessage::FixedString {
                size: data.len() as u32,
                padding,
                charset: 0,
            },
            dataspace: DataspaceMessage::scalar(),
            data: data.to_vec(),
        };

        // Space padded: no NUL anywhere, so truncating at the first NUL kept
        // the padding.
        assert_eq!(
            fixed_string_attr_value(&attr(2, b"volt    ")).unwrap(),
            "volt"
        );
        // Null terminated and null padded end at the first NUL, so trailing
        // spaces before it are content.
        assert_eq!(
            fixed_string_attr_value(&attr(0, b"volt  \0\0")).unwrap(),
            "volt  "
        );
        assert_eq!(
            fixed_string_attr_value(&attr(1, b"volt\0\0\0\0")).unwrap(),
            "volt"
        );
        // A reserved rule is named rather than guessed at.
        let err = fixed_string_attr_value(&attr(7, b"volt    ")).unwrap_err();
        assert!(
            err.to_string().contains("padding rule 7"),
            "unexpected error: {err}"
        );
    }
}

#[cfg(test)]
mod h5py_debug_tests {
    use super::*;

    #[test]
    fn debug_read_h5py() {
        let path = std::path::Path::new("/tmp/test_h5py_default.h5");
        if !path.exists() {
            return;
        }

        let handle = FileHandle::open_read(path).unwrap();
        let sb_buf = handle.read_at_most(0, 1024).unwrap();
        let version = detect_superblock_version(&sb_buf).unwrap();
        eprintln!("Superblock version: {}", version);

        let sb = SuperblockV0V1::decode(&sb_buf).unwrap();
        eprintln!(
            "sizeof_addr={}, sizeof_size={}",
            sb.sizeof_offsets, sb.sizeof_lengths
        );
        let (ste_btree, ste_heap) = sb
            .root_symbol_table_entry
            .cached_symbol_table()
            .unwrap_or((UNDEF_ADDR, UNDEF_ADDR));
        eprintln!(
            "STE: obj_header={}, cache={:?}, btree={}, heap={}",
            sb.root_symbol_table_entry.obj_header_addr,
            sb.root_symbol_table_entry.cache,
            ste_btree,
            ste_heap
        );

        let ctx = FormatContext {
            sizeof_addr: sb.sizeof_offsets,
            sizeof_size: sb.sizeof_lengths,
        };

        // Read local heap
        let heap_buf = handle.read_at_most(ste_heap, 128).unwrap();
        let heap_hdr = LocalHeapHeader::decode(
            &heap_buf,
            ctx.sizeof_addr as usize,
            ctx.sizeof_size as usize,
        )
        .unwrap();
        eprintln!(
            "Heap data_addr={}, data_size={}",
            heap_hdr.data_addr, heap_hdr.data_size
        );

        let heap_data = handle
            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)
            .unwrap();
        eprintln!(
            "Heap data bytes: {:?}",
            &heap_data[..std::cmp::min(64, heap_data.len())]
        );

        // Read btree
        let btree_buf = handle.read_at_most(ste_btree, 8192).unwrap();
        let btree = BTreeV1Node::decode(
            &btree_buf,
            ctx.sizeof_addr as usize,
            ctx.sizeof_size as usize,
            BTreeV1Config::default().snode_max_entries(),
        )
        .unwrap();
        eprintln!(
            "BTree: type={}, level={}, entries={}, children={:?}",
            btree.node_type, btree.level, btree.entries_used, btree.children
        );

        // Read SNOD
        for &child in &btree.children {
            let snod_buf = handle.read_at_most(child, 8192).unwrap();
            let snod = SymbolTableNode::decode(
                &snod_buf,
                ctx.sizeof_addr as usize,
                ctx.sizeof_size as usize,
                BTreeV1Config::default().sym_leaf_max_entries(),
            )
            .unwrap();
            eprintln!("SNOD at {}: {} entries", child, snod.entries.len());
            for entry in &snod.entries {
                let name = local_heap_get_string(&heap_data, entry.name_offset).unwrap();
                eprintln!(
                    "  entry: name='{}' (offset={}), obj_header={}, cache={:?}",
                    name, entry.name_offset, entry.obj_header_addr, entry.cache
                );
            }
        }

        // Try full open
        let reader = Hdf5Reader::open(path).unwrap();
        eprintln!("Datasets found: {:?}", reader.dataset_names());
    }

    // ====================================================================
    // Group/link discovery: continuation blocks, dense links, v0/v1 groups.
    //
    // These tests generate HDF5 fixtures with h5py (HDF5 2.0.0). If the
    // pinned Python interpreter is not present, the test skips so the suite
    // still runs in environments without it.
    // ====================================================================

    const TEST_PYTHON: &str = "/Users/stevek/mamba/envs/bs2026.1/bin/python";

    /// Per-call unique temp path (PID + atomic counter) to avoid collisions
    /// across concurrent test runs.
    fn temp_path(name: &str) -> std::path::PathBuf {
        use std::sync::atomic::{AtomicU64, Ordering};
        static COUNTER: AtomicU64 = AtomicU64::new(0);
        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
        std::env::temp_dir().join(format!(
            "rust_hdf5_gap_test_{}_{}_{}.h5",
            name,
            std::process::id(),
            n
        ))
    }

    /// Run a Python snippet to generate a fixture; returns false if Python
    /// is unavailable so the caller can skip the test.
    fn gen_fixture(script: &str) -> bool {
        if !std::path::Path::new(TEST_PYTHON).exists() {
            return false;
        }
        let status = std::process::Command::new(TEST_PYTHON)
            .arg("-c")
            .arg(script)
            .status();
        matches!(status, Ok(s) if s.success())
    }

    #[test]
    fn gap1_v2_root_continuation_block() {
        let path = temp_path("gap1_cont");
        let p = path.display().to_string();
        // ~6 datasets in a v2 root group forces an object-header
        // continuation block.
        let script = format!(
            "import h5py,numpy as np\n\
             f=h5py.File(r'{p}','w',libver='latest')\n\
             [f.create_dataset('ds_%d'%i,data=np.arange(i*10,i*10+10,dtype='int32')) for i in range(6)]\n\
             f.close()"
        );
        if !gen_fixture(&script) {
            eprintln!("skipping gap1: python unavailable");
            return;
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();
        let mut names = reader.dataset_names();
        names.sort();
        assert_eq!(
            names,
            vec!["ds_0", "ds_1", "ds_2", "ds_3", "ds_4", "ds_5"],
            "all 6 datasets must be found across the continuation block"
        );
        // Element-exact read of one dataset.
        let raw = reader.read_dataset_raw("ds_3").unwrap();
        let vals: Vec<i32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(vals, (30..40).collect::<Vec<i32>>());
        let _ = std::fs::remove_file(&path);
    }

    #[test]
    fn gap2_v2_dense_fractal_heap_links() {
        let path = temp_path("gap2_dense");
        let p = path.display().to_string();
        // 14 datasets in one v2 group forces dense (fractal-heap) link
        // storage.
        let script = format!(
            "import h5py,numpy as np\n\
             f=h5py.File(r'{p}','w',libver='latest')\n\
             g=f.create_group('dense')\n\
             [g.create_dataset('d%02d'%i,data=np.full(4,i,dtype='float64')) for i in range(14)]\n\
             f.close()"
        );
        if !gen_fixture(&script) {
            eprintln!("skipping gap2: python unavailable");
            return;
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();
        let mut names = reader.dataset_names();
        names.sort();
        let expected: Vec<String> = (0..14).map(|i| format!("dense/d{:02}", i)).collect();
        assert_eq!(
            names, expected,
            "all 14 dense-stored links must be recovered from the fractal heap"
        );
        // Element-exact read of one dense-stored dataset.
        let raw = reader.read_dataset_raw("dense/d07").unwrap();
        let vals: Vec<f64> = raw
            .as_chunks::<8>()
            .0
            .iter()
            .map(|c| f64::from_le_bytes(*c))
            .collect();
        assert_eq!(vals, vec![7.0; 4]);
        let _ = std::fs::remove_file(&path);
    }

    #[test]
    fn gap3_v0v1_legacy_subgroups() {
        let path = temp_path("gap3_legacy");
        let p = path.display().to_string();
        // libver='earliest' => v0 superblock, symbol-table groups; datasets
        // nested inside subgroups.
        let script = format!(
            "import h5py,numpy as np\n\
             f=h5py.File(r'{p}','w',libver='earliest')\n\
             g1=f.create_group('grp1')\n\
             g1.create_dataset('a',data=np.arange(5,dtype='int16'))\n\
             g2=g1.create_group('sub')\n\
             g2.create_dataset('b',data=np.arange(7,dtype='int64'))\n\
             f.create_dataset('top',data=np.arange(3,dtype='int32'))\n\
             f.close()"
        );
        if !gen_fixture(&script) {
            eprintln!("skipping gap3: python unavailable");
            return;
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();
        let mut names = reader.dataset_names();
        names.sort();
        assert_eq!(
            names,
            vec!["grp1/a", "grp1/sub/b", "top"],
            "datasets nested in legacy symbol-table subgroups must be found"
        );
        // Element-exact read of a doubly-nested dataset.
        let raw = reader.read_dataset_raw("grp1/sub/b").unwrap();
        let vals: Vec<i64> = raw
            .as_chunks::<8>()
            .0
            .iter()
            .map(|c| i64::from_le_bytes(*c))
            .collect();
        assert_eq!(vals, (0..7).collect::<Vec<i64>>());
        let _ = std::fs::remove_file(&path);
    }

    /// N-bit chunked datasets with non-zero bit offset and signed types with
    /// negative values must read back element-exact through the crate's
    /// chunked readers. The post-filter datatype conversion shifts/masks/
    /// sign-extends each element after the filter pipeline.
    #[test]
    fn nbit_chunked_post_filter_conversion() {
        let path = temp_path("nbit_conv");
        let p = path.display().to_string();
        // Build N-bit datasets with reduced precision + non-zero offset via
        // h5py's low-level filter API (h5py has no high-level N-bit knob).
        let script = format!(
            "import h5py,numpy as np\n\
             from h5py import h5t,h5p,h5s,h5d,h5f,h5z\n\
             fid=h5f.create(r'{p}'.encode())\n\
             def mk(name,bt,prec,off,npd,vals,chunk):\n\
            \x20dt=bt.copy();dt.set_precision(prec);dt.set_offset(off)\n\
            \x20arr=np.ascontiguousarray(np.asarray(vals,dtype=npd))\n\
            \x20sp=h5s.create_simple(arr.shape)\n\
            \x20dc=h5p.create(h5p.DATASET_CREATE);dc.set_chunk(chunk)\n\
            \x20dc.set_filter(h5z.FILTER_NBIT,h5z.FLAG_OPTIONAL,())\n\
            \x20ds=h5d.create(fid,name.encode(),dt,sp,dc)\n\
            \x20ds.write(h5s.ALL,h5s.ALL,arr);ds.close()\n\
             mk('u4_p17_o3',h5t.STD_U32LE,17,3,'u4',[0,1,1000,65535,131071,70000,42,99999],(4,))\n\
             mk('i4_p13_o5',h5t.STD_I32LE,13,5,'i4',[-5,-1,0,1,7,-4096,4095,-77,42,100,-100,3],(4,))\n\
             mk('i2_p9_o4',h5t.STD_I16LE,9,4,'i2',[-256,-1,0,1,255,-7,7,-200],(3,))\n\
             mk('i4_2d_p11_o6',h5t.STD_I32LE,11,6,'i4',np.array([[-1024,-1,0,5],[1023,-77,88,-3]],dtype='i4'),(1,4))\n\
             fid.close()"
        );
        if !gen_fixture(&script) {
            eprintln!("skipping nbit_chunked_post_filter_conversion: python unavailable");
            return;
        }

        let mut reader = Hdf5Reader::open(&path).unwrap();

        // Unsigned u4, precision 17, bit offset 3.
        let raw = reader.read_dataset_raw("u4_p17_o3").unwrap();
        let got: Vec<u32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| u32::from_le_bytes(*c))
            .collect();
        assert_eq!(
            got,
            vec![0u32, 1, 1000, 65535, 131071, 70000, 42, 99999],
            "u4 N-bit dataset must decode to exact unsigned values"
        );

        // Signed i4 with negatives, precision 13, bit offset 5.
        let raw = reader.read_dataset_raw("i4_p13_o5").unwrap();
        let got: Vec<i32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(
            got,
            vec![-5i32, -1, 0, 1, 7, -4096, 4095, -77, 42, 100, -100, 3],
            "i4 N-bit dataset must sign-extend negative values"
        );

        // Signed i2 with negatives, precision 9, bit offset 4.
        let raw = reader.read_dataset_raw("i2_p9_o4").unwrap();
        let got: Vec<i16> = raw
            .as_chunks::<2>()
            .0
            .iter()
            .map(|c| i16::from_le_bytes(*c))
            .collect();
        assert_eq!(
            got,
            vec![-256i16, -1, 0, 1, 255, -7, 7, -200],
            "i2 N-bit dataset must sign-extend negative values"
        );

        // 2D signed i4, precision 11, bit offset 6 (1-row chunks).
        let raw = reader.read_dataset_raw("i4_2d_p11_o6").unwrap();
        let got: Vec<i32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(
            got,
            vec![-1024i32, -1, 0, 5, 1023, -77, 88, -3],
            "2D i4 N-bit dataset must decode element-exact"
        );

        // read_slice path must also apply the conversion exactly once.
        let raw = reader.read_slice("i4_p13_o5", &[4], &[3]).unwrap();
        let got: Vec<i32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(got, vec![7i32, -4096, 4095], "read_slice must convert too");

        // 2D slice: second row, all columns.
        let raw = reader.read_slice("i4_2d_p11_o6", &[1, 0], &[1, 4]).unwrap();
        let got: Vec<i32> = raw
            .as_chunks::<4>()
            .0
            .iter()
            .map(|c| i32::from_le_bytes(*c))
            .collect();
        assert_eq!(
            got,
            vec![1023i32, -77, 88, -3],
            "2D read_slice must convert"
        );

        let _ = std::fs::remove_file(&path);
    }

    /// Partial-slice reads of chunked datasets must skip non-overlapping
    /// chunks yet return exactly the selected region, for every chunk index
    /// type: v1 B-tree (libver=earliest), single chunk, fixed array,
    /// extensible array (one unlimited dim), and v2 B-tree (>1 unlimited dim).
    ///
    /// The dataset is a 5×4×6 int32 `arange`, chunked 2×2×2 so the chunk grid
    /// is ragged (edge chunks) and most selections touch a strict subset of
    /// chunks. Each slice is checked against the row-major `arange` value so a
    /// dropped/misplaced chunk or a mis-sized output buffer is caught.
    #[test]
    fn read_slice_chunked_all_index_types() {
        let latest = temp_path("slice_chunk_latest");
        let earliest = temp_path("slice_chunk_earliest");
        let pl = latest.display().to_string();
        let pe = earliest.display().to_string();
        // libver=latest selects modern indices by maxshape: fixed -> Fixed
        // Array, one unlimited dim -> Extensible Array, >1 unlimited -> v2
        // B-tree, single chunk -> Single Chunk index. libver=earliest always
        // uses the v1 B-tree chunk index.
        let script = format!(
            "import h5py,numpy as np\n\
             a=np.arange(5*4*6,dtype='int32').reshape(5,4,6)\n\
             f=h5py.File(r'{pl}','w',libver='latest')\n\
             f.create_dataset('single',data=a,chunks=(5,4,6))\n\
             f.create_dataset('fa',data=a,chunks=(2,2,2))\n\
             f.create_dataset('ea',data=a,chunks=(2,2,2),maxshape=(None,4,6))\n\
             f.create_dataset('btv2',data=a,chunks=(2,2,2),maxshape=(None,None,6))\n\
             f.close()\n\
             g=h5py.File(r'{pe}','w',libver='earliest')\n\
             g.create_dataset('btv1',data=a,chunks=(2,2,2))\n\
             g.close()"
        );
        if !gen_fixture(&script) {
            eprintln!("skipping read_slice_chunked_all_index_types: python unavailable");
            return;
        }

        let dims = [5u64, 4, 6];
        // Row-major value of element (i,j,k) in the arange dataset.
        let val = |i: u64, j: u64, k: u64| (i * dims[1] * dims[2] + j * dims[2] + k) as i32;
        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
            let mut out = Vec::new();
            for i in 0..counts[0] {
                for j in 0..counts[1] {
                    for k in 0..counts[2] {
                        out.push(val(starts[0] + i, starts[1] + j, starts[2] + k));
                    }
                }
            }
            out
        };
        let decode = |raw: Vec<u8>| -> Vec<i32> {
            raw.as_chunks::<4>()
                .0
                .iter()
                .map(|c| i32::from_le_bytes(*c))
                .collect()
        };
        let cases: &[([u64; 3], [u64; 3])] = &[
            ([0, 1, 0], [5, 2, 6]), // full last dim, partial mid -> coalesced runs
            ([1, 0, 0], [3, 4, 6]), // partial dim0, inner dims full -> one run
            ([0, 0, 2], [5, 4, 3]), // partial last dim -> one run per (i,j) row
            ([2, 1, 3], [1, 2, 2]), // interior block spanning few chunks
            ([0, 0, 0], [5, 4, 6]), // whole dataset via read_slice
            ([4, 3, 5], [1, 1, 1]), // single element at the far edge chunk
            ([1, 1, 1], [3, 3, 4]), // straddles chunk boundaries on all axes
        ];

        let mut reader_l = Hdf5Reader::open(&latest).unwrap();
        for name in ["single", "fa", "ea", "btv2"] {
            // Full read sanity first, then every slice.
            let full = decode(reader_l.read_dataset_raw(name).unwrap());
            assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "{name} full read");
            for &(starts, counts) in cases {
                let got = decode(reader_l.read_slice(name, &starts, &counts).unwrap());
                assert_eq!(
                    got,
                    expect(starts, counts),
                    "{name} slice starts={starts:?} counts={counts:?}"
                );
            }
        }

        let mut reader_e = Hdf5Reader::open(&earliest).unwrap();
        let full = decode(reader_e.read_dataset_raw("btv1").unwrap());
        assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "btv1 full read");
        for &(starts, counts) in cases {
            let got = decode(reader_e.read_slice("btv1", &starts, &counts).unwrap());
            assert_eq!(
                got,
                expect(starts, counts),
                "btv1 slice starts={starts:?} counts={counts:?}"
            );
        }

        let _ = std::fs::remove_file(&latest);
        let _ = std::fs::remove_file(&earliest);
    }

    /// A slice read must place the same bytes whichever way a chunk reaches
    /// the output: read run by run straight out of the file — what unfiltered
    /// data allows, its stored bytes being the dataset's bytes — or copied out
    /// of a decoded whole-chunk image, which filtered data always needs and
    /// which runs too small to be worth a positioned read each fall back to.
    /// Both sinks are driven off one array here, so naive extraction is the
    /// shared oracle for the two of them and for every run shape in between.
    #[test]
    #[cfg(feature = "deflate")]
    fn slice_reads_agree_however_the_chunk_reaches_the_output() {
        let path = temp_path("slice_run_placement");
        let (rows, cols) = (8usize, 4096usize); // a chunk row is 32 KiB
        let data: Vec<f64> = (0..rows * cols).map(|i| i as f64).collect();
        let (rows, cols) = (rows as u64, cols as u64);
        {
            let file = crate::H5File::create(&path).unwrap();
            let ds = file
                .new_dataset::<f64>()
                .shape([rows as usize, cols as usize])
                .chunk(&[2, cols as usize])
                .create("plain")
                .unwrap();
            ds.write_raw(&data).unwrap();
            let ds = file
                .new_dataset::<f64>()
                .shape([rows as usize, cols as usize])
                .chunk(&[2, cols as usize])
                .deflate(1)
                .create("zipped")
                .unwrap();
            ds.write_raw(&data).unwrap();
            file.close().unwrap();
        }

        let expect = |starts: [u64; 2], counts: [u64; 2]| -> Vec<f64> {
            let mut out = Vec::new();
            for i in 0..counts[0] {
                for j in 0..counts[1] {
                    out.push(data[((starts[0] + i) * cols + starts[1] + j) as usize]);
                }
            }
            out
        };
        let decode = |raw: Vec<u8>| -> Vec<f64> {
            raw.as_chunks::<8>()
                .0
                .iter()
                .map(|c| f64::from_le_bytes(*c))
                .collect()
        };
        let cases: &[([u64; 2], [u64; 2])] = &[
            ([0, 0], [rows, cols]), // whole dataset: one run per chunk
            ([3, 0], [4, cols]),    // straddles chunk rows, full width
            ([1, 7], [5, 3]),       // 24-byte runs: not worth a read each
            ([2, 1000], [2, 2048]), // 16 KiB runs, off the chunk row origin
            ([7, 4095], [1, 1]),    // one element in the last chunk
        ];

        let mut reader = Hdf5Reader::open(&path).unwrap();
        for name in ["plain", "zipped"] {
            let full = decode(reader.read_dataset_raw(name).unwrap());
            assert_eq!(full, data, "{name} full read");
            for &(starts, counts) in cases {
                let got = decode(reader.read_slice(name, &starts, &counts).unwrap());
                assert_eq!(
                    got,
                    expect(starts, counts),
                    "{name} slice starts={starts:?} counts={counts:?}"
                );
            }
        }

        std::fs::remove_file(&path).ok();
    }
}