rust-hdf5 0.7.2

Pure Rust HDF5 library with full read/write and SWMR support
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
#!/usr/bin/env python3
"""The oracle's feature matrix.

Each case names one HDF5 feature, carries an h5py generator that writes the
reference file, and — when rust-hdf5's *public* API can express the same file —
the id of the matching arm in `src/bin/oracle_probe.rs`'s `write` subcommand.

The data every case writes is deliberately formulaic (`arange`-like ramps,
short literal strings) so the rust writer can reproduce it byte for byte
without the two sides sharing a data file.

Only the standard library, numpy and h5py are used.
"""

import pathlib
import shutil

import numpy as np

import h5py
from h5py import h5d, h5p, h5s, h5t

# --------------------------------------------------------------------------
# helpers
# --------------------------------------------------------------------------

N = 8  # default element count for the 1-D dtype ramps


def ramp(dtype, n=N):
    return np.arange(n, dtype=np.dtype(dtype))


def lowlevel_dataset(f, name, tid, sid, data=None, dcpl=None, mtype=None):
    """Create a dataset from a raw TypeID/SpaceID, which h5py cannot express.

    Returns the low-level DatasetID so the caller can write through it.
    """
    dsid = h5d.create(f.id, name.encode("utf-8"), tid, sid, dcpl=dcpl)
    if data is not None:
        dsid.write(h5s.ALL, h5s.ALL, np.ascontiguousarray(data), mtype=mtype)
    return dsid


def chunked_dcpl(chunk, alloc_time=None, layout=h5d.CHUNKED):
    dcpl = h5p.create(h5p.DATASET_CREATE)
    dcpl.set_layout(layout)
    if layout == h5d.CHUNKED:
        dcpl.set_chunk(tuple(chunk))
    if alloc_time is not None:
        dcpl.set_alloc_time(alloc_time)
    return dcpl


class Case:
    def __init__(self, name, group, gen, rust=None, note="", ext_files=(), access=None):
        self.name = name
        self.group = group
        self.gen = gen
        self.rust = rust
        self.note = note
        self.ext_files = ext_files
        # Dataset-access properties both sides open every dataset under —
        # `{"view": "first_missing"|"last_available", "printf_gap": int}`.
        # They are not stored in the file, so they belong to the case, not to
        # the generator: the same bytes describe differently under different
        # ones.
        self.access = access

    def __repr__(self):
        return "Case(%s)" % self.name


# --------------------------------------------------------------------------
# integer widths / signedness / endianness
# --------------------------------------------------------------------------


def _int_case(name, npdtype, rust):
    def gen(path):
        with h5py.File(path, "w") as f:
            f.create_dataset("data", data=ramp(npdtype))

    return Case(name, "dtype-int", gen, rust, "1-D ramp of %s" % npdtype)


INT_CASES = [
    _int_case("int_i8", "i1", "int_i8"),
    _int_case("int_u8", "u1", "int_u8"),
    _int_case("int_i16le", "<i2", "int_i16le"),
    _int_case("int_u16le", "<u2", "int_u16le"),
    _int_case("int_i32le", "<i4", "int_i32le"),
    _int_case("int_u32le", "<u4", "int_u32le"),
    _int_case("int_i64le", "<i8", "int_i64le"),
    _int_case("int_u64le", "<u8", "int_u64le"),
    _int_case("int_i16be", ">i2", "int_i16be"),
    _int_case("int_i32be", ">i4", "int_i32be"),
    _int_case("int_u64be", ">u8", "int_u64be"),
]


# --------------------------------------------------------------------------
# floating point
# --------------------------------------------------------------------------


def _float_case(name, npdtype, rust):
    def gen(path):
        with h5py.File(path, "w") as f:
            f.create_dataset("data", data=ramp(npdtype))

    return Case(name, "dtype-float", gen, rust, "1-D ramp of %s" % npdtype)


def gen_float_specials(path):
    bits = np.array(
        [
            0x7FF8000000000001,  # quiet NaN with a payload
            0x7FF0000000000000,  # +inf
            0xFFF0000000000000,  # -inf
            0x8000000000000000,  # -0.0
            0x0000000000000001,  # smallest denormal
            0x3FF0000000000000,  # 1.0
            0xBFF0000000000000,  # -1.0
            0x0000000000000000,  # +0.0
        ],
        dtype="<u8",
    )
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=bits.view("<f8"))


FLOAT_CASES = [
    _float_case("float_f16le", "<f2", "float_f16le"),
    _float_case("float_f32le", "<f4", "float_f32le"),
    _float_case("float_f64le", "<f8", "float_f64le"),
    _float_case("float_f64be", ">f8", "float_f64be"),
    Case(
        "float_specials",
        "dtype-float",
        gen_float_specials,
        "float_specials",
        "NaN payload, +/-inf, -0.0, denormal — bit patterns must survive",
    ),
]


# --------------------------------------------------------------------------
# strings
# --------------------------------------------------------------------------

STRINGS = ["alpha", "b", "", "delta12"]
UNISTR = ["été", "日本", "", "café"]


def gen_str_fixed_ascii(path):
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "data",
            data=np.array([s.encode("ascii") for s in STRINGS], dtype="S8"),
            dtype=h5py.string_dtype("ascii", 8),
        )


def gen_str_fixed_utf8(path):
    # libhdf5 has no ASCII -> UTF-8 conversion path, so the memory type has to
    # be the UTF-8 fixed string itself and the bytes go through verbatim.
    with h5py.File(path, "w") as f:
        tid = h5t.C_S1.copy()
        tid.set_size(16)
        tid.set_cset(h5t.CSET_UTF8)
        tid.set_strpad(h5t.STR_NULLPAD)
        sid = h5s.create_simple((len(UNISTR),))
        lowlevel_dataset(
            f,
            "data",
            tid,
            sid,
            np.array([s.encode("utf-8") for s in UNISTR], dtype="S16"),
            mtype=tid,
        )


def _fixed_str_pad(strpad):
    # libhdf5 does not re-pad on a same-size same-cset copy, so the padding
    # bytes have to be written explicitly for the declared rule to be what is
    # actually on disk.
    padbyte = b" " if strpad == h5t.STR_SPACEPAD else b"\0"

    def gen(path):
        with h5py.File(path, "w") as f:
            tid = h5t.C_S1.copy()
            tid.set_size(8)
            tid.set_cset(h5t.CSET_ASCII)
            tid.set_strpad(strpad)
            sid = h5s.create_simple((len(STRINGS),))
            data = np.array(
                [s.encode("ascii").ljust(8, padbyte) for s in STRINGS], dtype="S8"
            )
            lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)

    return gen


def gen_str_vlen_ascii(path):
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "data",
            data=np.array(STRINGS, dtype=object),
            dtype=h5py.string_dtype("ascii"),
        )


def gen_str_vlen_utf8(path):
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "data",
            data=np.array(UNISTR, dtype=object),
            dtype=h5py.string_dtype("utf-8"),
        )


STRING_CASES = [
    Case("str_fixed_ascii", "dtype-string", gen_str_fixed_ascii, "str_fixed_ascii",
         "8-byte fixed ASCII strings"),
    Case("str_fixed_utf8", "dtype-string", gen_str_fixed_utf8, "str_fixed_utf8",
         "16-byte fixed UTF-8 strings"),
    Case("str_fixed_nullpad", "dtype-string", _fixed_str_pad(h5t.STR_NULLPAD), "str_fixed_nullpad",
         "fixed string with STR_NULLPAD"),
    Case("str_fixed_spacepad", "dtype-string", _fixed_str_pad(h5t.STR_SPACEPAD), "str_fixed_spacepad",
         "fixed string with STR_SPACEPAD"),
    Case("str_vlen_ascii", "dtype-string", gen_str_vlen_ascii, "str_vlen_ascii",
         "variable-length ASCII strings via the global heap"),
    Case("str_vlen_utf8", "dtype-string", gen_str_vlen_utf8, "str_vlen_utf8",
         "variable-length UTF-8 strings via the global heap"),
]


# --------------------------------------------------------------------------
# compound / array / enum / opaque / bitfield / references / vlen
# --------------------------------------------------------------------------

COMPOUND_SIMPLE = np.dtype([("x", "<f4"), ("y", "<f4")])
COMPOUND_NESTED = np.dtype([("a", "<i4"), ("inner", [("u", "<i2"), ("v", "<i2")])])
COMPOUND_STR = np.dtype([("id", "<i4"), ("name", "S8")])
COMPOUND_PAD = np.dtype(
    {"names": ["a", "b"], "formats": ["<i2", "<i4"], "offsets": [0, 4], "itemsize": 12}
)


def gen_compound_simple(path):
    arr = np.zeros(4, dtype=COMPOUND_SIMPLE)
    arr["x"] = np.arange(4, dtype="<f4")
    arr["y"] = np.arange(100, 104, dtype="<f4")
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=arr)


def gen_compound_nested(path):
    arr = np.zeros(4, dtype=COMPOUND_NESTED)
    arr["a"] = np.arange(4, dtype="<i4")
    arr["inner"]["u"] = np.arange(10, 14, dtype="<i2")
    arr["inner"]["v"] = np.arange(20, 24, dtype="<i2")
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=arr)


def gen_compound_with_string(path):
    arr = np.zeros(3, dtype=COMPOUND_STR)
    arr["id"] = np.arange(3, dtype="<i4")
    arr["name"] = [b"aa", b"bbb", b"cccc"]
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=arr)


def gen_compound_padded(path):
    arr = np.zeros(4, dtype=COMPOUND_PAD)
    arr["a"] = np.arange(4, dtype="<i2")
    arr["b"] = np.arange(1000, 1004, dtype="<i4")
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=arr)


def gen_array_dtype(path):
    with h5py.File(path, "w") as f:
        tid = h5t.array_create(h5t.IEEE_F64LE, (2, 3))
        sid = h5s.create_simple((2,))
        data = np.arange(12, dtype="<f8").reshape(2, 2, 3)
        lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)


def gen_enum_i8(path):
    dt = h5py.enum_dtype({"RED": 0, "GREEN": 1, "BLUE": 2}, basetype="i1")
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=np.array([0, 1, 2, 1], dtype="i1"), dtype=dt)


def gen_enum_i32(path):
    dt = h5py.enum_dtype({"LOW": -1, "MID": 0, "HIGH": 1000}, basetype="<i4")
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=np.array([-1, 0, 1000, 0], dtype="<i4"), dtype=dt)


def gen_compound_dtype_v4(path):
    # A v1.12 low bound makes libhdf5 tag the compound datatype message
    # version 4 (H5O_dtype_ver_bounds); its members stay version 1. Nothing
    # else in the matrix produces a datatype message above version 3.
    #
    # Chunked on purpose: a *contiguous* dataset at a v1.10+ bound is dropped
    # from the listing by an unrelated gap (`layout_contiguous_v110`), which
    # would mask what this case is about.
    arr = np.zeros(4, dtype=COMPOUND_SIMPLE)
    arr["x"] = np.arange(4, dtype="<f4")
    arr["y"] = np.arange(100, 104, dtype="<f4")
    with h5py.File(path, "w", libver=("v112", "v112")) as f:
        ds = f.create_dataset("data", (4,), chunks=(4,), dtype=COMPOUND_SIMPLE)
        ds[...] = arr


def gen_opaque(path):
    with h5py.File(path, "w") as f:
        tid = h5t.create(h5t.OPAQUE, 4)
        tid.set_tag(b"raw4")
        sid = h5s.create_simple((3,))
        data = np.frombuffer(bytes(range(12)), dtype="V4")
        lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)


def gen_bitfield(path):
    with h5py.File(path, "w") as f:
        tid = h5t.STD_B8LE.copy()
        sid = h5s.create_simple((4,))
        lowlevel_dataset(
            f, "data", tid, sid, np.array([0x01, 0x80, 0xFF, 0x00], dtype="u1")
        )


def gen_ref_object(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("target", data=ramp("<i4"))
        g = f.create_group("grp")
        refs = np.array([f["target"].ref, g.ref], dtype=h5py.ref_dtype)
        f.create_dataset("refs", data=refs, dtype=h5py.ref_dtype)


def gen_ref_region(path):
    with h5py.File(path, "w") as f:
        t = f.create_dataset("target", data=ramp("<i4"))
        refs = np.array([t.regionref[0:3], t.regionref[4:8]],
                        dtype=h5py.regionref_dtype)
        f.create_dataset("refs", data=refs, dtype=h5py.regionref_dtype)


def gen_vlen_numeric(path):
    dt = h5py.vlen_dtype(np.dtype("<i4"))
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", (3,), dtype=dt)
        ds[0] = np.array([1, 2, 3], dtype="<i4")
        ds[1] = np.array([], dtype="<i4")
        ds[2] = np.array([-7], dtype="<i4")


def gen_vlen_bytes(path):
    dt = h5py.vlen_dtype(np.dtype("u1"))
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", (3,), dtype=dt)
        ds[0] = np.array([0, 1, 2], dtype="u1")
        ds[1] = np.array([], dtype="u1")
        ds[2] = np.array([255], dtype="u1")


def gen_named_datatype(path):
    with h5py.File(path, "w") as f:
        f["t"] = np.dtype("<i4")
        f.create_dataset("data", data=ramp("<i4"))
        # A dataset created from the committed TypeID stores a *shared*
        # datatype message pointing at /t rather than a datatype of its own.
        sid = h5s.create_simple((N,))
        lowlevel_dataset(f, "shared", f["t"].id, sid, ramp("<i4"))


COMPOSITE_CASES = [
    Case("compound_simple", "dtype-composite", gen_compound_simple, "compound_simple",
         "two f32 members, no padding"),
    Case("compound_nested", "dtype-composite", gen_compound_nested, "compound_nested",
         "compound member inside a compound"),
    Case("compound_with_string", "dtype-composite", gen_compound_with_string,
         "compound_with_string", "fixed string member"),
    Case("compound_padded", "dtype-composite", gen_compound_padded, "compound_padded",
         "member offsets with gaps and trailing padding"),
    Case("compound_dtype_v4", "dtype-composite", gen_compound_dtype_v4,
         "compound_dtype_v4", "version-4 datatype message (libver v1.12 bounds)"),
    Case("array_dtype", "dtype-composite", gen_array_dtype, "array_dtype",
         "H5T_ARRAY element type (2x3 f64)"),
    Case("enum_i8", "dtype-composite", gen_enum_i8, "enum_i8", "3-member i8 enum"),
    Case("enum_i32", "dtype-composite", gen_enum_i32, "enum_i32",
         "i32 enum with a negative member"),
    Case("opaque", "dtype-composite", gen_opaque, "opaque", "H5T_OPAQUE with a tag"),
    Case("bitfield", "dtype-composite", gen_bitfield, "bitfield",
         "H5T_BITFIELD (STD_B8LE)"),
    Case("ref_object", "dtype-composite", gen_ref_object, "ref_object",
         "object references to a dataset and a group"),
    Case("ref_region", "dtype-composite", gen_ref_region, "ref_region",
         "dataset region references"),
    Case("vlen_numeric", "dtype-composite", gen_vlen_numeric, "vlen_numeric",
         "variable-length i32 sequences"),
    Case("vlen_bytes", "dtype-composite", gen_vlen_bytes, "vlen_bytes",
         "variable-length u8 sequences"),
    Case("named_datatype", "dtype-composite", gen_named_datatype, "named_datatype",
         "committed datatype object, and a dataset that shares it"),
]


# --------------------------------------------------------------------------
# layouts and chunk indexes
# --------------------------------------------------------------------------


def gen_layout_compact(path):
    with h5py.File(path, "w") as f:
        dcpl = h5p.create(h5p.DATASET_CREATE)
        dcpl.set_layout(h5d.COMPACT)
        sid = h5s.create_simple((16,))
        lowlevel_dataset(
            f, "data", h5t.STD_I32LE, sid, ramp("<i4", 16), dcpl=dcpl
        )


def gen_layout_contiguous(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=ramp("<i4", 16))


def gen_layout_contiguous_v110(path):
    with h5py.File(path, "w", libver=("v110", "v110")) as f:
        f.create_dataset("data", data=ramp("<i4", 16))


def gen_layout_chunked_v110(path):
    with h5py.File(path, "w", libver=("v110", "v110")) as f:
        ds = f.create_dataset("data", (16,), chunks=(16,), dtype="<i4")
        ds[...] = ramp("<i4", 16)


def gen_chunkidx_btree1(path):
    with h5py.File(path, "w", libver="earliest") as f:
        ds = f.create_dataset(
            "data", (8,), maxshape=(None,), chunks=(4,), dtype="<i4"
        )
        ds[...] = ramp("<i4")


def gen_chunkidx_single(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset("data", (8,), chunks=(8,), dtype="<i4")
        ds[...] = ramp("<i4")


def gen_chunkidx_implicit(path):
    with h5py.File(path, "w", libver="latest") as f:
        dcpl = chunked_dcpl((4,), alloc_time=h5d.ALLOC_TIME_EARLY)
        sid = h5s.create_simple((16,))
        lowlevel_dataset(f, "data", h5t.STD_I32LE, sid, ramp("<i4", 16), dcpl=dcpl)


def gen_chunkidx_farray(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset("data", (16,), chunks=(4,), dtype="<i4")
        ds[...] = ramp("<i4", 16)


def gen_chunkidx_earray(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset(
            "data", (16,), maxshape=(None,), chunks=(4,), dtype="<i4"
        )
        ds[...] = ramp("<i4", 16)


def gen_chunkidx_earray_unlim_inner(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset(
            "data", (4, 4), maxshape=(4, None), chunks=(2, 2), dtype="<i4"
        )
        ds[...] = ramp("<i4", 16).reshape(4, 4)


def gen_layout_contiguous_v108(path):
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        f.create_dataset("data", data=ramp("<i4", 16))


def gen_layout_chunked_v108(path):
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        ds = f.create_dataset("data", (16,), chunks=(16,), dtype="<i4")
        ds[...] = ramp("<i4", 16)


def gen_chunkidx_earray_dim1(path):
    # The extensible dimension is the fastest-changing one, so the chunk
    # coordinate the index is keyed on is not the first.
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset(
            "data", (4, 4), maxshape=(4, None), chunks=(2, 4), dtype="<i4"
        )
        ds[...] = ramp("<i4", 16).reshape(4, 4)


def gen_external_storage(path):
    raw = path.parent / (path.stem + "_ext.raw")
    raw.write_bytes(ramp("<i4", 16).tobytes())
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "data", shape=(16,), dtype="<i4", external=[(raw.name, 0, 64)]
        )


def gen_vds(path):
    src = path.parent / (path.stem + "_src.h5")
    with h5py.File(src, "w") as g:
        g.create_dataset("src", data=ramp("<i4", 16))
    layout = h5py.VirtualLayout(shape=(16,), dtype="<i4")
    layout[...] = h5py.VirtualSource(src.name, "src", shape=(16,))
    with h5py.File(path, "w") as f:
        f.create_virtual_dataset("vds", layout)


def gen_external_unlimited(path):
    """The last EFL slot sized H5O_EFL_UNLIMITED over an unlimited dataspace.

    `H5D__efl_construct` requires exactly this pairing: a dataset whose
    dataspace can grow has no finite reservation that could cover it, so its
    last — and here only — external slot must be unlimited.
    """
    raw = path.parent / (path.stem + "_ext.raw")
    raw.write_bytes(ramp("<i4", 16).tobytes())
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "data",
            shape=(16,),
            maxshape=(None,),
            dtype="<i4",
            external=[(raw.name, 0, h5py.h5f.UNLIMITED)],
        )


def gen_vds_unlim(path):
    """A mapping unlimited on both sides: the extent follows the source.

    `H5D__virtual_set_extent_unlim` clips the virtual selection against what
    the source actually holds, so the dataset created one row tall reads back
    as tall as its source is.
    """
    src = path.parent / (path.stem + "_src.h5")
    with h5py.File(src, "w") as g:
        g.create_dataset(
            "src", data=ramp("<i4", 20).reshape(10, 2), maxshape=(None, 2),
            chunks=(5, 2),
        )
    layout = h5py.VirtualLayout(shape=(1, 2), dtype="<i4", maxshape=(None, 2))
    vsrc = h5py.VirtualSource(src.name, "src", shape=(1, 2), maxshape=(None, 2))
    layout[: h5s.UNLIMITED, :] = vsrc[: h5s.UNLIMITED, :]
    with h5py.File(path, "w") as f:
        f.create_virtual_dataset("vds", layout)


def gen_vds_printf_unlim(path):
    """A printf-pattern source name over an unlimited virtual selection.

    The two features only exist together: `H5D_virtual_check_mapping_post`
    refuses a `%b` in a source name unless the virtual selection is unlimited
    and the source selection is not, which is the mapping that says "one source
    file per block, as many as are there". Three blocks are present, so the
    extent `H5D__virtual_set_extent_unlim` gives the dataset is three rows.
    """
    stem = path.stem
    for b in range(3):
        with h5py.File(path.parent / ("%s_b%d.h5" % (stem, b)), "w") as g:
            g.create_dataset("data", data=ramp("<i4", 4) + 10 * b)
    layout = h5py.VirtualLayout(shape=(1, 4), dtype="<i4", maxshape=(None, 4))
    vsrc = h5py.VirtualSource("%s_b%%b.h5" % stem, "data", shape=(4,))
    layout[: h5s.UNLIMITED, :] = vsrc
    with h5py.File(path, "w") as f:
        f.create_virtual_dataset("vds", layout)


def gen_vds_printf_gap(path):
    """A printf mapping over blocks 0, 1 and 3 — block 2 is not written.

    What the extent is depends entirely on the dataset access properties the
    open names, and the file records none of them: with the default gap of 0
    the scan stops at block 2 and the dataset is two rows tall, and the cases
    reading this same file with a gap look past it to block 3 and see four,
    the third filled (`H5D__virtual_set_extent_unlim`, H5Dvirtual.c:1519).
    """
    stem = path.stem
    for b in (0, 1, 3):
        with h5py.File(path.parent / ("%s_b%d.h5" % (stem, b)), "w") as g:
            g.create_dataset("data", data=ramp("<i4", 4) + 10 * b)
    layout = h5py.VirtualLayout(shape=(1, 4), dtype="<i4", maxshape=(None, 4))
    vsrc = h5py.VirtualSource("%s_b%%b.h5" % stem, "data", shape=(4,))
    layout[: h5s.UNLIMITED, :] = vsrc
    with h5py.File(path, "w") as f:
        f.create_virtual_dataset("vds", layout, fillvalue=-7)


def gen_vds_view_trail(path):
    """A mapping unlimited on both sides whose stride is wider than its block.

    `H5S_hyper_get_clip_extent_match` takes `incl_trail` from the view
    (H5Dvirtual.c:1447-1451), and it only changes the answer when the last
    mapped block is followed by a gap: three source rows under stride 3,
    block 2 give a two-row extent under `H5D_VDS_LAST_AVAILABLE` and a
    three-row one under `H5D_VDS_FIRST_MISSING`.
    """
    with h5py.File(path, "w") as f:
        f.create_dataset(
            "src", data=ramp("<i4", 6).reshape(3, 2), maxshape=(None, 2),
            chunks=(1, 2),
        )
        vsid = h5s.create_simple((1, 2), (h5s.UNLIMITED, 2))
        vsid.select_hyperslab((0, 0), (h5s.UNLIMITED, 1), stride=(3, 1), block=(2, 2))
        ssid = h5s.create_simple((1, 2), (h5s.UNLIMITED, 2))
        ssid.select_hyperslab((0, 0), (h5s.UNLIMITED, 1), stride=(3, 1), block=(2, 2))
        dcpl = h5p.create(h5p.DATASET_CREATE)
        # `H5Pset_virtual` pokes the layout straight into the list
        # (H5Pdcpl.c:2146) and so leaves the allocation time at the contiguous
        # default, while `H5Pset_layout` resets it to the layout's own
        # (H5Pdcpl.c:1762-1782). h5py's VDS path always makes this call
        # (`VirtualLayout.__init__`, vds.py:174), so every virtual dataset
        # written through an API rather than by hand is incremental.
        dcpl.set_layout(h5d.VIRTUAL)
        dcpl.set_fill_value(np.array(-9, dtype="<i4"))
        dcpl.set_virtual(vsid, b".", b"/src", ssid)
        lowlevel_dataset(f, "vds", h5t.STD_I32LE, vsid, dcpl=dcpl)


def gen_vds_split(path):
    """A mapping whose two selections decompose into different box counts.

    The virtual side is two 1x4 blocks — rows 0 and 2 of a 4x4 dataset — and
    the source side one `H5S_SEL_ALL` over a 2x4 one, so there is no
    positional pairing of boxes to be had. `H5S_select_project_intersection`
    (H5Sselect.c:2402) runs one selection iterator per side and matches the
    two element streams off one against one; the only thing it asks of the
    pair is that the element counts agree, which is also the only thing
    `H5D_virtual_check_mapping_pre` checks when the mapping is created
    (H5Dvirtual.c:254-257).
    """
    with h5py.File(path, "w") as f:
        f.create_dataset("src", data=ramp("<i4", 8).reshape(2, 4))
        vsid = h5s.create_simple((4, 4))
        vsid.select_hyperslab((0, 0), (2, 1), stride=(2, 1), block=(1, 4))
        ssid = h5s.create_simple((2, 4))
        ssid.select_all()
        dcpl = h5p.create(h5p.DATASET_CREATE)
        dcpl.set_layout(h5d.VIRTUAL)
        dcpl.set_fill_value(np.array(-9, dtype="<i4"))
        dcpl.set_virtual(vsid, b".", b"/src", ssid)
        lowlevel_dataset(f, "vds", h5t.STD_I32LE, vsid, dcpl=dcpl)


def gen_chunkidx_btree2(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset(
            "data", (4, 4), maxshape=(None, None), chunks=(2, 2), dtype="<i4"
        )
        ds[...] = ramp("<i4", 16).reshape(4, 4)


LAYOUT_CASES = [
    Case("layout_compact", "layout", gen_layout_compact, "layout_compact",
         "compact layout — data inside the object header"),
    Case("layout_contiguous", "layout", gen_layout_contiguous, "layout_contiguous",
         "contiguous layout"),
    Case("layout_contiguous_v110", "layout", gen_layout_contiguous_v110,
         "layout_contiguous_v110",
         "contiguous layout under v1.10 bounds — data layout message v4"),
    Case("layout_chunked_v110", "layout", gen_layout_chunked_v110,
         "layout_chunked_v110",
         "chunked layout under v1.10 bounds — the control for the case above"),
    Case("chunkidx_btree1", "layout", gen_chunkidx_btree1, "chunkidx_btree1",
         "layout v3 + version-1 B-tree chunk index (libver earliest)"),
    Case("chunkidx_single", "layout", gen_chunkidx_single, "chunkidx_single",
         "single-chunk index"),
    Case("chunkidx_implicit", "layout", gen_chunkidx_implicit, "chunkidx_implicit",
         "implicit index — fixed shape, early allocation, no filter"),
    Case("chunkidx_farray", "layout", gen_chunkidx_farray, "chunkidx_farray",
         "fixed-array index"),
    Case("chunkidx_earray", "layout", gen_chunkidx_earray, "chunkidx_earray",
         "extensible-array index — one unlimited dimension"),
    Case("chunkidx_earray_unlim_inner", "layout", gen_chunkidx_earray_unlim_inner,
         "chunkidx_earray_unlim_inner",
         "extensible-array index — the unlimited dimension is dim 1, not dim 0"),
    Case("chunkidx_btree2", "layout", gen_chunkidx_btree2, "chunkidx_btree2",
         "version-2 B-tree index — two unlimited dimensions"),
    Case("layout_contiguous_v108", "layout", gen_layout_contiguous_v108,
         "layout_contiguous_v108",
         "contiguous layout under v1.8 bounds — the v1.10 pair's control"),
    Case("layout_chunked_v108", "layout", gen_layout_chunked_v108,
         "layout_chunked_v108", "chunked layout under v1.8 bounds"),
    Case("chunkidx_earray_dim1", "layout", gen_chunkidx_earray_dim1,
         "chunkidx_earray_dim1",
         "extensible-array index whose unlimited dimension is not the first"),
    Case("external_storage", "layout", gen_external_storage, "external_storage",
         "contiguous data held in an external raw file",
         ext_files=("_ext.raw",)),
    Case("vds", "layout", gen_vds, "vds",
         "virtual dataset mapped onto a dataset in a sibling file",
         ext_files=("_src.h5",)),
    Case("external_unlimited", "layout", gen_external_unlimited,
         "external_unlimited",
         "H5O_EFL_UNLIMITED on the last external slot of an unlimited dataset",
         ext_files=("_ext.raw",)),
    Case("vds_unlim", "layout", gen_vds_unlim, "vds_unlim",
         "virtual dataset whose mapping is unlimited on both sides — the "
         "extent comes from the source",
         ext_files=("_src.h5",)),
    Case("vds_printf_gap", "layout", gen_vds_printf_gap, "vds_printf_gap",
         "printf mapping with block 2 missing, read at the default printf "
         "gap of 0 — the extent stops at the gap"),
    Case("vds_printf_gap_1", "layout", gen_vds_printf_gap, "vds_printf_gap",
         "the same file read with H5Pset_virtual_printf_gap(1) — block 3 is "
         "reached and block 2 reads as the fill value",
         access={"printf_gap": 1}),
    Case("vds_printf_gap_first_missing", "layout", gen_vds_printf_gap, "vds_printf_gap",
         "the same file under H5D_VDS_FIRST_MISSING with a gap of 2, which "
         "H5D__virtual_init forces back to 0 (H5Dvirtual.c:2182-2188)",
         access={"view": "first_missing", "printf_gap": 2}),
    Case("vds_view_trail", "layout", gen_vds_view_trail, "vds_view_trail",
         "unlimited mapping whose stride exceeds its block, read at the "
         "default H5D_VDS_LAST_AVAILABLE view"),
    Case("vds_view_trail_first_missing", "layout", gen_vds_view_trail, "vds_view_trail",
         "the same file under H5D_VDS_FIRST_MISSING — the extent runs on to "
         "where the next block would start",
         access={"view": "first_missing"}),
    Case("vds_split", "layout", gen_vds_split, "vds_split",
         "a same-file mapping whose virtual and source selections decompose "
         "into different numbers of boxes"),
    Case("vds_printf_unlim", "layout", gen_vds_printf_unlim, "vds_printf_unlim",
         "printf-pattern source name over an unlimited virtual selection — "
         "one source file per block",
         ext_files=("_b0.h5", "_b1.h5", "_b2.h5")),
]


# --------------------------------------------------------------------------
# filters
# --------------------------------------------------------------------------


def _filter_case(name, rust, note, **kw):
    def gen(path):
        with h5py.File(path, "w", libver="latest") as f:
            ds = f.create_dataset("data", (64,), chunks=(16,), dtype="<i4", **kw)
            ds[...] = ramp("<i4", 64)

    return Case(name, "filter", gen, rust, note)


FILTER_CASES = [
    _filter_case("filter_deflate", "filter_deflate", "deflate level 6",
                 compression="gzip", compression_opts=6),
    _filter_case("filter_shuffle", "filter_shuffle", "shuffle only",
                 shuffle=True),
    _filter_case("filter_fletcher32", "filter_fletcher32", "fletcher32 checksum",
                 fletcher32=True),
    _filter_case("filter_deflate_shuffle", "filter_deflate_shuffle",
                 "shuffle then deflate",
                 compression="gzip", compression_opts=6, shuffle=True),
    _filter_case("filter_scaleoffset", "filter_scaleoffset",
                 "scale-offset, library-computed minimum bits",
                 scaleoffset=0),
    _filter_case("filter_szip_ec", "filter_szip_ec",
                 "szip entropy coding, 8 pixels per block",
                 compression="szip", compression_opts=("ec", 8)),
    _filter_case("filter_szip_nn", "filter_szip_nn",
                 "szip nearest neighbour, 16 pixels per block",
                 compression="szip", compression_opts=("nn", 16)),
]


# --------------------------------------------------------------------------
# fill values
# --------------------------------------------------------------------------


def gen_fill_default(path):
    with h5py.File(path, "w", libver="latest") as f:
        f.create_dataset("data", (16,), chunks=(4,), dtype="<i4")


def gen_fill_set_int(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset("data", (16,), chunks=(4,), dtype="<i4", fillvalue=-1)
        ds[0:4] = ramp("<i4", 4)


def gen_fill_set_float_nan(path):
    with h5py.File(path, "w", libver="latest") as f:
        f.create_dataset(
            "data", (16,), chunks=(4,), dtype="<f8", fillvalue=np.float64("nan")
        )


FILL_CASES = [
    Case("fill_default", "fillvalue", gen_fill_default, "fill_default",
         "default (zero) fill, nothing written"),
    Case("fill_set_int", "fillvalue", gen_fill_set_int, "fill_set_int",
         "user-defined integer fill, first chunk written"),
    Case("fill_set_float_nan", "fillvalue", gen_fill_set_float_nan,
         "fill_set_float_nan", "user-defined NaN fill"),
]


# --------------------------------------------------------------------------
# dataspaces
# --------------------------------------------------------------------------


def gen_space_scalar(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("data", data=np.int32(42))


def gen_space_null(path):
    with h5py.File(path, "w") as f:
        f["data"] = h5py.Empty("<i4")


def gen_space_zerosized(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("data", (0,), dtype="<i4")


def gen_space_unlimited_resized(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset("data", (4,), maxshape=(None,), chunks=(4,), dtype="<i4")
        ds[...] = ramp("<i4", 4)
        ds.resize((12,))
        ds[4:12] = ramp("<i4", 8) + 100


SPACE_CASES = [
    Case("space_scalar", "dataspace", gen_space_scalar, "space_scalar",
         "scalar (rank 0) dataspace"),
    Case("space_null", "dataspace", gen_space_null, "space_null",
         "NULL dataspace — no elements at all"),
    Case("space_zerosized", "dataspace", gen_space_zerosized, "space_zerosized",
         "simple dataspace with a zero-length dimension"),
    Case("space_unlimited_resized", "dataspace", gen_space_unlimited_resized,
         "space_unlimited_resized", "unlimited maxshape, grown after creation"),
]


# --------------------------------------------------------------------------
# groups, links
# --------------------------------------------------------------------------


def gen_groups_nested(path):
    with h5py.File(path, "w") as f:
        g = f.create_group("a")
        h = g.create_group("b")
        h.create_group("c")
        h.create_dataset("leaf", data=ramp("<i4"))
        f.create_dataset("top", data=ramp("<i4"))


def gen_link_hard(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("orig", data=ramp("<i4"))
        f["alias"] = f["orig"]


def gen_link_soft(path):
    with h5py.File(path, "w") as f:
        f.create_dataset("orig", data=ramp("<i4"))
        f["alias"] = h5py.SoftLink("/orig")


def gen_link_external(path):
    target = path.parent / (path.stem + "_ext.h5")
    with h5py.File(target, "w") as g:
        g.create_dataset("payload", data=ramp("<i4"))
    with h5py.File(path, "w") as f:
        f.create_dataset("orig", data=ramp("<i4"))
        f["ext"] = h5py.ExternalLink(target.name, "/payload")


def gen_link_external_read(path):
    """A master file whose payload lives entirely in a sibling.

    Every dataset is reached only by crossing a link, so the `resolved` field
    is the whole content check: a reader that lists the links but never opens
    the other file matches on `target` and diverges here. The two dangling
    links pin the other half — a target file that is not there and a target
    object that is not there both have to say so rather than read something.
    """
    target = path.parent / (path.stem + "_data.h5")
    with h5py.File(target, "w") as g:
        g.create_dataset("top", data=ramp("<f8"))
        g.create_group("deep").create_dataset("inner", data=ramp("<i2"))
    with h5py.File(path, "w") as f:
        f["direct"] = h5py.ExternalLink(target.name, "/top")
        f["nested"] = h5py.ExternalLink(target.name, "/deep/inner")
        f["gone_object"] = h5py.ExternalLink(target.name, "/absent")
        f["gone_file"] = h5py.ExternalLink("no_such_file.h5", "/top")


def gen_link_nonascii(path):
    """Non-ASCII link names in a file written at the earliest bound.

    h5py encodes a name it is handed as `str` to ASCII when it can and to
    UTF-8 when it cannot, and puts the result in the lcpl's character encoding
    (`CommonStateObject._e`). `H5G_obj_insert` reads that back off the link and
    converts a symbol-table group to link messages the moment it is not ASCII
    (`obj_lnk->cset != H5T_CSET_ASCII`, H5Gobj.c:514) — the same branch an
    external link takes.

    So the root here loses its symbol table over two non-ASCII names while its
    ASCII siblings come along as link messages, and every group whose own
    children are ASCII-named keeps its symbol table. The superblock stays at
    version 0 throughout: the conversion is per group, not per file.
    """
    with h5py.File(path, "w", libver="earliest") as f:
        f.create_dataset("데이터", data=ramp("<i4"))
        f.create_group("plain")
        f.create_group("그룹").create_dataset("inner", data=ramp("<i4", 4))
        f.create_group("ascii_only").create_dataset("inner", data=ramp("<i4", 4))


def gen_links_dense(path):
    # v1.8 bounds, not "latest": dense link storage needs the v1.8 group
    # format, and stopping there keeps the v1.10 layout message out of the
    # file so this case isolates link storage.
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        g = f.create_group("g", track_order=True)
        for i in range(12):
            g.create_dataset("d%02d" % i, data=np.array([i], dtype="<i4"))


def gen_track_order(path):
    # Creation-order tracking adds a second index (a v2 B-tree keyed on
    # creation order) beside the name index, on both links and attributes.
    with h5py.File(path, "w", track_order=True) as f:
        for name in ("zebra", "apple", "mango"):
            f.create_group(name)
        for i, key in enumerate(("zeta", "alpha", "mu")):
            f.attrs.create(key, np.int32(i))
        g = f.create_group("g", track_order=True)
        g.create_dataset("data", data=ramp("<i4"))
        g.attrs.create("second", np.int32(2))
        g.attrs.create("first", np.int32(1))


def gen_group_storage_modern_root(path):
    """A symbol-table group, holding children, under a link-message root.

    `track_order` migrates the group that asks for it and nothing else, so one
    h5py call writes a root using link messages over a child still using the
    legacy symbol table. A reader that walks a group the way its parent is
    stored lists `legacy` and finds none of its children.
    """
    with h5py.File(path, "w", track_order=True) as f:
        legacy = f.create_group("legacy")
        legacy.create_dataset("a", data=ramp("<i4"))
        legacy.create_group("inner").create_dataset("c", data=ramp("<i2"))


def gen_group_storage_legacy_root(path):
    """The same mismatch the other way up: a link-message group, holding
    children, under a symbol-table root."""
    with h5py.File(path, "w") as f:
        f.create_group("legacy").create_dataset("a", data=ramp("<i4"))
        modern = f.create_group("modern", track_order=True)
        modern.create_dataset("b", data=ramp("<f8"))
        modern.create_group("inner").create_dataset("c", data=ramp("<i2"))


LINK_CASES = [
    Case("groups_nested", "group", gen_groups_nested, "groups_nested",
         "three levels of nested groups plus an empty leaf group"),
    Case("link_hard", "link", gen_link_hard, "link_hard",
         "two names for one object"),
    Case("link_soft", "link", gen_link_soft, "link_soft", "soft link to /orig"),
    Case("link_external", "link", gen_link_external, "link_external",
         "external link into a sibling file",
         ext_files=("_ext.h5",)),
    Case("link_external_read", "link", gen_link_external_read,
         "link_external_read",
         "datasets read through external links, plus a dangling object and a "
         "dangling file",
         ext_files=("_data.h5",)),
    Case("link_nonascii", "link", gen_link_nonascii, "link_nonascii",
         "non-ASCII link names at the earliest bound — the root converts to "
         "link messages, the ASCII-named subgroups keep their symbol tables"),
    Case("links_dense", "link", gen_links_dense, "links_dense",
         "12 links in one group — dense link storage (fractal heap + v2 B-tree)"),
    Case("track_order", "group", gen_track_order, "track_order",
         "creation-order indices on links and attributes"),
    Case("group_storage_modern_root", "group", gen_group_storage_modern_root,
         "group_storage_modern_root",
         "symbol-table group with children under a link-message root"),
    Case("group_storage_legacy_root", "group", gen_group_storage_legacy_root,
         "group_storage_legacy_root",
         "link-message group with children under a symbol-table root"),
]


# --------------------------------------------------------------------------
# attributes
# --------------------------------------------------------------------------


def gen_attr_scalar_num(path):
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        ds.attrs.create("gain", np.float64(2.5))
        ds.attrs.create("count", np.int32(7))


def gen_attr_array_num(path):
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        ds.attrs.create("offsets", np.arange(4, dtype="<i4"))
        ds.attrs.create("matrix", np.arange(6, dtype="<f8").reshape(2, 3))


def gen_attr_string(path):
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        ds.attrs.create("units", "volt", dtype=h5py.string_dtype("utf-8"))
        f.create_group("g").attrs.create(
            "NX_class", "NXdetector", dtype=h5py.string_dtype("utf-8")
        )


def gen_attrs_dense(path):
    # v1.8 bounds for the same reason as links_dense: dense attribute storage
    # arrives with v1.8, and staying there isolates it from the v1.10 layout
    # message.
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        for i in range(12):
            ds.attrs.create("a%02d" % i, np.int32(i))


def gen_attrs_dense_group(path):
    # The same phase change on a group and on the root group, where the
    # attributes share their object header with the link messages rather than
    # with a dataset's layout.
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        g = f.create_group("g")
        for i in range(12):
            g.attrs.create("g%02d" % i, np.int32(i))
        for i in range(12):
            f.attrs.create("r%02d" % i, np.int32(i))
        f.create_dataset("data", data=ramp("<i4"))


def gen_attr_on_root(path):
    with h5py.File(path, "w") as f:
        f.attrs.create("title", "root", dtype=h5py.string_dtype("utf-8"))
        f.attrs.create("version", np.int64(3))
        f.create_dataset("data", data=ramp("<i4"))


def gen_attr_ref_object(path):
    # An attribute whose value is object references, on all three kinds of
    # object header: a dataset's, a group's and the root's. The value is part
    # of the header message, so unlike a reference dataset's elements it cannot
    # be stamped in after the header is written.
    with h5py.File(path, "w") as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        g = f.create_group("grp")
        ds.attrs.create("neighbours", np.array([ds.ref, g.ref], dtype=h5py.ref_dtype))
        g.attrs.create("owner", ds.ref, dtype=h5py.ref_dtype)
        f.attrs.create("entry", np.array([g.ref, ds.ref], dtype=h5py.ref_dtype))


def gen_attr_large(path):
    # One attribute past the 64 KiB object-header message limit: the value
    # spills to dense storage no matter how few attributes there are.
    with h5py.File(path, "w", libver=("v108", "v108")) as f:
        ds = f.create_dataset("data", data=ramp("<i4"))
        ds.attrs.create("big", np.arange(25600, dtype="<i4"))


ATTR_CASES = [
    Case("attr_scalar_num", "attribute", gen_attr_scalar_num, "attr_scalar_num",
         "scalar f64 and i32 attributes"),
    Case("attr_array_num", "attribute", gen_attr_array_num, "attr_array_num",
         "1-D and 2-D numeric attributes"),
    Case("attr_string", "attribute", gen_attr_string, "attr_string",
         "vlen UTF-8 string attributes on a dataset and a group"),
    Case("attrs_dense", "attribute", gen_attrs_dense, "attrs_dense",
         "12 attributes — dense attribute storage"),
    Case("attrs_dense_group", "attribute", gen_attrs_dense_group,
         "attrs_dense_group",
         "12 attributes on a group and on the root — dense storage"),
    Case("attr_on_root", "attribute", gen_attr_on_root, "attr_on_root",
         "attributes on the root group"),
    Case("attr_large", "attribute", gen_attr_large, "attr_large",
         "single 100 KiB attribute — dense storage forced by size, not count"),
    Case("attr_ref_object", "attribute", gen_attr_ref_object, "attr_ref_object",
         "object references stored in attributes of a dataset, a group and "
         "the root"),
]


# --------------------------------------------------------------------------
# library version bounds / superblock
# --------------------------------------------------------------------------


def _libver_case(name, libver, rust, note):
    def gen(path):
        with h5py.File(path, "w", libver=libver) as f:
            f.create_dataset("data", data=ramp("<i4"))
            f.create_group("g")

    return Case(name, "superblock", gen, rust, note)


def gen_userblock(path):
    """A 512-byte userblock in front of the superblock.

    The block is filled with text afterwards, as an application that keeps a
    script or a header there would: the reader has to find the superblock at
    512 rather than at 0, and must not mistake the block's bytes for metadata.
    """
    with h5py.File(path, "w", userblock_size=512) as f:
        f.create_dataset("data", data=ramp("<i4"))
        f.create_group("g")
    prefix = b"#!/bin/sh\n# userblock\n"
    with open(path, "r+b") as fh:
        fh.write(prefix + b"#" * (512 - len(prefix) - 1) + b"\n")


def _reopen_append_case(name, libver, rust, note):
    """Create under one bound, reopen under the default fapl, append.

    `H5F__super_init` is the only place a superblock version is decided, and
    open never re-decides it; `H5F__super_read` instead raises the file's low
    library bound to match the version it finds — v2 to `H5F_LIBVER_V18`, v3
    to `H5F_LIBVER_V110`. So the appended dataset is written in whatever
    generation the file already has, not the one the default fapl would have
    picked, and the superblock version comes out of the reopen unchanged.

    The appended dataset is chunked because that is where the two generations
    differ most visibly: a version-3 layout message has the version-1 B-tree
    and nothing else, while a version-4 one picks among the v1.10 indexes.
    """
    def gen(path):
        with h5py.File(path, "w", libver=libver) as f:
            f.create_dataset("data", data=ramp("<i4"))
            f.create_group("g")
        with h5py.File(path, "a") as f:
            f.create_dataset("appended", data=ramp("<i4", 16).reshape(4, 4),
                             chunks=(2, 4))

    return Case(name, "superblock", gen, rust, note)


LIBVER_CASES = [
    _libver_case("libver_earliest", "earliest", "libver_earliest",
                 "libver earliest — superblock v0, symbol-table groups"),
    _libver_case("libver_v108", ("v108", "v108"), "libver_v108",
                 "libver v1.8 bounds"),
    _libver_case("libver_v110", ("v110", "v110"), "libver_v110",
                 "libver v1.10 bounds"),
    _libver_case("libver_latest", "latest", "libver_latest",
                 "libver latest — superblock v3, new-style groups"),
    # Superblock v1 is not reachable from h5py 3.15: it is produced only by a
    # non-default B-tree K value (H5Pset_sym_k / H5Pset_istore_k), and neither
    # is wrapped on PropFCID. v0, v2 and v3 are covered by the four cases
    # above; the user block below is the remaining v0 variant.
    Case("userblock", "superblock", gen_userblock, "userblock",
         "512-byte userblock — the superblock, and every address, is based at 512"),
    _reopen_append_case(
        "reopen_append_earliest", "earliest", "reopen_append_earliest",
        "classic file reopened under the default fapl — stays superblock v0, "
        "the appended chunked dataset takes the version-1 B-tree"),
    _reopen_append_case(
        "reopen_append_v108", ("v108", "v108"), "reopen_append_v108",
        "v1.8 file reopened under the default fapl — stays superblock v2, "
        "the appended chunked dataset takes the version-1 B-tree"),
    _reopen_append_case(
        "reopen_append_latest", "latest", "reopen_append_latest",
        "v1.10+ file reopened under the default fapl — stays superblock v3, "
        "the appended chunked dataset takes a v1.10 index"),
]


# --------------------------------------------------------------------------
# SWMR and bulk
# --------------------------------------------------------------------------


def gen_swmr_created(path):
    with h5py.File(path, "w", libver="latest") as f:
        ds = f.create_dataset("stream", (0, 4), maxshape=(None, 4), chunks=(1, 4),
                              dtype="<f4")
        f.swmr_mode = True
        for i in range(8):
            ds.resize((i + 1, 4))
            ds[i, :] = np.arange(i * 4, i * 4 + 4, dtype="<f4")
            ds.flush()


def gen_large_multi_mb(path):
    with h5py.File(path, "w", libver="latest") as f:
        data = np.arange(512 * 512, dtype="<f8").reshape(512, 512)
        f.create_dataset("big", data=data, chunks=(64, 512))


def gen_fsm_persist(path):
    """A persisting free-space-manager file, reopened and appended to.

    `fs_persist` is what makes the free-space managers on-disk structures
    rather than in-memory bookkeeping: the close writes an H5FS header and
    section-info block per allocation type and names them in the file-space
    info message. Nothing is freed by the create alone, so the reopen and
    append are the half that matters — superseding the superblock extension
    and the root group's object header is what puts sections in a manager,
    and `#freespace` is `tracked` only if they were written back.
    """
    with h5py.File(path, "w", fs_strategy="fsm", fs_persist=True, fs_threshold=1) as f:
        f.create_dataset("data", data=ramp("<i4"))
        f.create_dataset("bulk", data=ramp("<i4", 256))
        f.create_group("g")
    with h5py.File(path, "a") as f:
        del f["bulk"]
        f.create_dataset("appended", data=ramp("<i4"))


def gen_fsm_persist_page(path):
    """[`gen_fsm_persist`] under paged aggregation.

    `H5F_FSPACE_STRATEGY_PAGE` packs everything smaller than the file-space
    page into pages of one kind and page-aligns everything else
    (`H5MF__alloc_pagefs`), so the same edits leave a different free-space
    manager set: sections that never cross a page boundary, and — under sec2,
    which declares no `H5FD_FEAT_PAGED_AGGR` — at most the three managers
    `H5MF__alloc_to_fs_type` can reach.
    """
    with h5py.File(path, "w", fs_strategy="page", fs_persist=True, fs_threshold=1) as f:
        f.create_dataset("data", data=ramp("<i4"))
        f.create_dataset("bulk", data=ramp("<i4", 256))
        f.create_group("g")
    with h5py.File(path, "a") as f:
        del f["bulk"]
        f.create_dataset("appended", data=ramp("<i4"))


def gen_fsm_page_size(path):
    """[`gen_fsm_persist_page`] on a page size that is not the library default.

    `H5Pset_file_space_page_size` takes anything from 512
    (`H5F_FILE_SPACE_PAGE_SIZE_MIN`) to 1 GiB with no power-of-two
    requirement, and the size decides which side of `H5MF__alloc_to_fs_type`
    every request falls on. At 512 the `bulk` dataset's raw data is larger
    than a page and is page-aligned as `H5F_MEM_PAGE_GENERIC`, where at the
    4096-byte default the same bytes are packed into a page — so this is a
    different manager set, not the same file with a different number in the
    message.
    """
    with h5py.File(path, "w", fs_strategy="page", fs_persist=True,
                   fs_threshold=1, fs_page_size=512) as f:
        f.create_dataset("data", data=ramp("<i4"))
        f.create_dataset("bulk", data=ramp("<i4", 256))
        f.create_group("g")
    with h5py.File(path, "a") as f:
        del f["bulk"]
        f.create_dataset("appended", data=ramp("<i4"))


MISC_CASES = [
    Case("swmr_created", "swmr", gen_swmr_created, "swmr_created",
         "file created through the SWMR writer path and appended frame by frame"),
    Case("fsm_persist", "freespace", gen_fsm_persist, "fsm_persist",
         "persisting FSM_AGGR file reopened and appended — the freed blocks "
         "must come back as free-space manager sections"),
    Case("fsm_persist_page", "freespace", gen_fsm_persist_page, "fsm_persist_page",
         "persisting PAGE file reopened and appended — the freed blocks must "
         "come back as page-shaped free-space manager sections"),
    Case("fsm_page_size", "freespace", gen_fsm_page_size, "fsm_page_size",
         "persisting PAGE file on a 512-byte file-space page — the non-default "
         "size must reach the message and shape the allocation"),
    Case("large_multi_mb", "bulk", gen_large_multi_mb, "large_multi_mb",
         "2 MiB chunked f64 dataset — payload compared by SHA-256"),
]


# --------------------------------------------------------------------------
# checked-in fixtures
#
# Some file-level features have no h5py binding at all, so the reference file
# cannot be written from Python. Those come from a C generator run against the
# pinned libhdf5 (`tests/fixtures/gen_*.sh`), are checked in, and are copied
# into the run directory here. h5py still reads them, so direction A compares
# exactly as it does for a generated case.
# --------------------------------------------------------------------------

FIXTURE_DIR = pathlib.Path(__file__).resolve().parent.parent / "tests" / "fixtures"


def _fixture_case(name, fixture, generator, group, note, rust=None):
    def gen(path):
        src = FIXTURE_DIR / fixture
        if not src.exists():
            raise FileNotFoundError(
                "%s is missing; regenerate it with tests/fixtures/%s"
                % (src, generator)
            )
        shutil.copyfile(src, path)

    # `rust` is None where the public API cannot ask for the file at all;
    # where it can, the arm mirrors the C generator rather than an h5py one.
    return Case(name, group, gen, rust, note)


def _fixture_append_case(name, fixture, generator, group, note, rust):
    """A checked-in fixture libhdf5 then reopens and appends to.

    The create is the half h5py cannot express — there is no binding for
    `H5Pset_shared_mesg_index` — but the *append* is an ordinary `'a'` open,
    so the reference is a genuine libhdf5 reopen of a file with a
    shared-message table. The rust arm creates its own equivalent and reopens
    that, which is the only shape the comparison can take: nothing on the
    Python side can hand the rust writer a file it did not create.
    """
    def gen(path):
        src = FIXTURE_DIR / fixture
        if not src.exists():
            raise FileNotFoundError(
                "%s is missing; regenerate it with tests/fixtures/%s"
                % (src, generator)
            )
        shutil.copyfile(src, path)
        with h5py.File(path, "a") as f:
            f.create_dataset("appended", data=ramp("<i4", 8))

    return Case(name, group, gen, rust, note)


FIXTURE_CASES = [
    _fixture_case(
        "sohm_list", "sohm_list.h5", "gen_sohm.sh", "sohm",
        "shared datatype/dataspace/attribute messages, list index "
        "(H5Pset_shared_mesg_index) + a committed datatype",
        rust="sohm_list",
    ),
    _fixture_case(
        "sohm_btree", "sohm_btree.h5", "gen_sohm.sh", "sohm",
        "the same file with the shared-message index forced to a v2 B-tree",
        rust="sohm_btree",
    ),
    _fixture_append_case(
        "sohm_list_append", "sohm_list.h5", "gen_sohm.sh", "sohm",
        "a file with a shared-message list index reopened and appended to — "
        "the table is laid out whole, so the append replaces it",
        rust="sohm_list_append",
    ),
    _fixture_append_case(
        "sohm_btree_append", "sohm_btree.h5", "gen_sohm.sh", "sohm",
        "the same reopen over a v2 B-tree index",
        rust="sohm_btree_append",
    ),
    _fixture_case(
        "ochk_root", "ochk_root.h5", "gen_ochk.sh", "objectheader",
        "root group whose object header spills into two continuation chunks",
        rust="ochk_root",
    ),
    _fixture_case(
        "vds_late_layout", "vds_late_layout.h5", "gen_vds_late_layout.sh",
        "layout",
        "virtual datasets built by H5Pset_virtual with and without a prior "
        "H5Pset_layout — the pairs differ only in allocation time "
        "(H5Pdcpl.c:2146 pokes the layout past H5P__set_layout's default)",
    ),
]


# --------------------------------------------------------------------------

ALL_CASES = (
    INT_CASES
    + FLOAT_CASES
    + STRING_CASES
    + COMPOSITE_CASES
    + LAYOUT_CASES
    + FILTER_CASES
    + FILL_CASES
    + SPACE_CASES
    + LINK_CASES
    + ATTR_CASES
    + LIBVER_CASES
    + MISC_CASES
    + FIXTURE_CASES
)


def by_name(name):
    for c in ALL_CASES:
        if c.name == name:
            return c
    raise KeyError(name)


if __name__ == "__main__":
    print("%d cases" % len(ALL_CASES))
    for c in ALL_CASES:
        print("  %-24s %-16s rust=%s" % (c.name, c.group, c.rust or "-"))