import pathlib
import shutil
import numpy as np
import h5py
from h5py import h5d, h5p, h5s, h5t
N = 8
def ramp(dtype, n=N):
return np.arange(n, dtype=np.dtype(dtype))
def lowlevel_dataset(f, name, tid, sid, data=None, dcpl=None, mtype=None):
dsid = h5d.create(f.id, name.encode("utf-8"), tid, sid, dcpl=dcpl)
if data is not None:
dsid.write(h5s.ALL, h5s.ALL, np.ascontiguousarray(data), mtype=mtype)
return dsid
def chunked_dcpl(chunk, alloc_time=None, layout=h5d.CHUNKED):
dcpl = h5p.create(h5p.DATASET_CREATE)
dcpl.set_layout(layout)
if layout == h5d.CHUNKED:
dcpl.set_chunk(tuple(chunk))
if alloc_time is not None:
dcpl.set_alloc_time(alloc_time)
return dcpl
class Case:
def __init__(self, name, group, gen, rust=None, note="", ext_files=(), access=None):
self.name = name
self.group = group
self.gen = gen
self.rust = rust
self.note = note
self.ext_files = ext_files
self.access = access
def __repr__(self):
return "Case(%s)" % self.name
def _int_case(name, npdtype, rust):
def gen(path):
with h5py.File(path, "w") as f:
f.create_dataset("data", data=ramp(npdtype))
return Case(name, "dtype-int", gen, rust, "1-D ramp of %s" % npdtype)
INT_CASES = [
_int_case("int_i8", "i1", "int_i8"),
_int_case("int_u8", "u1", "int_u8"),
_int_case("int_i16le", "<i2", "int_i16le"),
_int_case("int_u16le", "<u2", "int_u16le"),
_int_case("int_i32le", "<i4", "int_i32le"),
_int_case("int_u32le", "<u4", "int_u32le"),
_int_case("int_i64le", "<i8", "int_i64le"),
_int_case("int_u64le", "<u8", "int_u64le"),
_int_case("int_i16be", ">i2", "int_i16be"),
_int_case("int_i32be", ">i4", "int_i32be"),
_int_case("int_u64be", ">u8", "int_u64be"),
]
def _float_case(name, npdtype, rust):
def gen(path):
with h5py.File(path, "w") as f:
f.create_dataset("data", data=ramp(npdtype))
return Case(name, "dtype-float", gen, rust, "1-D ramp of %s" % npdtype)
def gen_float_specials(path):
bits = np.array(
[
0x7FF8000000000001, 0x7FF0000000000000, 0xFFF0000000000000, 0x8000000000000000, 0x0000000000000001, 0x3FF0000000000000, 0xBFF0000000000000, 0x0000000000000000, ],
dtype="<u8",
)
with h5py.File(path, "w") as f:
f.create_dataset("data", data=bits.view("<f8"))
FLOAT_CASES = [
_float_case("float_f16le", "<f2", "float_f16le"),
_float_case("float_f32le", "<f4", "float_f32le"),
_float_case("float_f64le", "<f8", "float_f64le"),
_float_case("float_f64be", ">f8", "float_f64be"),
Case(
"float_specials",
"dtype-float",
gen_float_specials,
"float_specials",
"NaN payload, +/-inf, -0.0, denormal — bit patterns must survive",
),
]
STRINGS = ["alpha", "b", "", "delta12"]
UNISTR = ["été", "日本", "", "café"]
def gen_str_fixed_ascii(path):
with h5py.File(path, "w") as f:
f.create_dataset(
"data",
data=np.array([s.encode("ascii") for s in STRINGS], dtype="S8"),
dtype=h5py.string_dtype("ascii", 8),
)
def gen_str_fixed_utf8(path):
with h5py.File(path, "w") as f:
tid = h5t.C_S1.copy()
tid.set_size(16)
tid.set_cset(h5t.CSET_UTF8)
tid.set_strpad(h5t.STR_NULLPAD)
sid = h5s.create_simple((len(UNISTR),))
lowlevel_dataset(
f,
"data",
tid,
sid,
np.array([s.encode("utf-8") for s in UNISTR], dtype="S16"),
mtype=tid,
)
def _fixed_str_pad(strpad):
padbyte = b" " if strpad == h5t.STR_SPACEPAD else b"\0"
def gen(path):
with h5py.File(path, "w") as f:
tid = h5t.C_S1.copy()
tid.set_size(8)
tid.set_cset(h5t.CSET_ASCII)
tid.set_strpad(strpad)
sid = h5s.create_simple((len(STRINGS),))
data = np.array(
[s.encode("ascii").ljust(8, padbyte) for s in STRINGS], dtype="S8"
)
lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)
return gen
def gen_str_vlen_ascii(path):
with h5py.File(path, "w") as f:
f.create_dataset(
"data",
data=np.array(STRINGS, dtype=object),
dtype=h5py.string_dtype("ascii"),
)
def gen_str_vlen_utf8(path):
with h5py.File(path, "w") as f:
f.create_dataset(
"data",
data=np.array(UNISTR, dtype=object),
dtype=h5py.string_dtype("utf-8"),
)
STRING_CASES = [
Case("str_fixed_ascii", "dtype-string", gen_str_fixed_ascii, "str_fixed_ascii",
"8-byte fixed ASCII strings"),
Case("str_fixed_utf8", "dtype-string", gen_str_fixed_utf8, "str_fixed_utf8",
"16-byte fixed UTF-8 strings"),
Case("str_fixed_nullpad", "dtype-string", _fixed_str_pad(h5t.STR_NULLPAD), "str_fixed_nullpad",
"fixed string with STR_NULLPAD"),
Case("str_fixed_spacepad", "dtype-string", _fixed_str_pad(h5t.STR_SPACEPAD), "str_fixed_spacepad",
"fixed string with STR_SPACEPAD"),
Case("str_vlen_ascii", "dtype-string", gen_str_vlen_ascii, "str_vlen_ascii",
"variable-length ASCII strings via the global heap"),
Case("str_vlen_utf8", "dtype-string", gen_str_vlen_utf8, "str_vlen_utf8",
"variable-length UTF-8 strings via the global heap"),
]
COMPOUND_SIMPLE = np.dtype([("x", "<f4"), ("y", "<f4")])
COMPOUND_NESTED = np.dtype([("a", "<i4"), ("inner", [("u", "<i2"), ("v", "<i2")])])
COMPOUND_STR = np.dtype([("id", "<i4"), ("name", "S8")])
COMPOUND_PAD = np.dtype(
{"names": ["a", "b"], "formats": ["<i2", "<i4"], "offsets": [0, 4], "itemsize": 12}
)
def gen_compound_simple(path):
arr = np.zeros(4, dtype=COMPOUND_SIMPLE)
arr["x"] = np.arange(4, dtype="<f4")
arr["y"] = np.arange(100, 104, dtype="<f4")
with h5py.File(path, "w") as f:
f.create_dataset("data", data=arr)
def gen_compound_nested(path):
arr = np.zeros(4, dtype=COMPOUND_NESTED)
arr["a"] = np.arange(4, dtype="<i4")
arr["inner"]["u"] = np.arange(10, 14, dtype="<i2")
arr["inner"]["v"] = np.arange(20, 24, dtype="<i2")
with h5py.File(path, "w") as f:
f.create_dataset("data", data=arr)
def gen_compound_with_string(path):
arr = np.zeros(3, dtype=COMPOUND_STR)
arr["id"] = np.arange(3, dtype="<i4")
arr["name"] = [b"aa", b"bbb", b"cccc"]
with h5py.File(path, "w") as f:
f.create_dataset("data", data=arr)
def gen_compound_padded(path):
arr = np.zeros(4, dtype=COMPOUND_PAD)
arr["a"] = np.arange(4, dtype="<i2")
arr["b"] = np.arange(1000, 1004, dtype="<i4")
with h5py.File(path, "w") as f:
f.create_dataset("data", data=arr)
def gen_array_dtype(path):
with h5py.File(path, "w") as f:
tid = h5t.array_create(h5t.IEEE_F64LE, (2, 3))
sid = h5s.create_simple((2,))
data = np.arange(12, dtype="<f8").reshape(2, 2, 3)
lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)
def gen_enum_i8(path):
dt = h5py.enum_dtype({"RED": 0, "GREEN": 1, "BLUE": 2}, basetype="i1")
with h5py.File(path, "w") as f:
f.create_dataset("data", data=np.array([0, 1, 2, 1], dtype="i1"), dtype=dt)
def gen_enum_i32(path):
dt = h5py.enum_dtype({"LOW": -1, "MID": 0, "HIGH": 1000}, basetype="<i4")
with h5py.File(path, "w") as f:
f.create_dataset("data", data=np.array([-1, 0, 1000, 0], dtype="<i4"), dtype=dt)
def gen_compound_dtype_v4(path):
arr = np.zeros(4, dtype=COMPOUND_SIMPLE)
arr["x"] = np.arange(4, dtype="<f4")
arr["y"] = np.arange(100, 104, dtype="<f4")
with h5py.File(path, "w", libver=("v112", "v112")) as f:
ds = f.create_dataset("data", (4,), chunks=(4,), dtype=COMPOUND_SIMPLE)
ds[...] = arr
def gen_opaque(path):
with h5py.File(path, "w") as f:
tid = h5t.create(h5t.OPAQUE, 4)
tid.set_tag(b"raw4")
sid = h5s.create_simple((3,))
data = np.frombuffer(bytes(range(12)), dtype="V4")
lowlevel_dataset(f, "data", tid, sid, data, mtype=tid)
def gen_bitfield(path):
with h5py.File(path, "w") as f:
tid = h5t.STD_B8LE.copy()
sid = h5s.create_simple((4,))
lowlevel_dataset(
f, "data", tid, sid, np.array([0x01, 0x80, 0xFF, 0x00], dtype="u1")
)
def gen_ref_object(path):
with h5py.File(path, "w") as f:
f.create_dataset("target", data=ramp("<i4"))
g = f.create_group("grp")
refs = np.array([f["target"].ref, g.ref], dtype=h5py.ref_dtype)
f.create_dataset("refs", data=refs, dtype=h5py.ref_dtype)
def gen_ref_region(path):
with h5py.File(path, "w") as f:
t = f.create_dataset("target", data=ramp("<i4"))
refs = np.array([t.regionref[0:3], t.regionref[4:8]],
dtype=h5py.regionref_dtype)
f.create_dataset("refs", data=refs, dtype=h5py.regionref_dtype)
def gen_vlen_numeric(path):
dt = h5py.vlen_dtype(np.dtype("<i4"))
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", (3,), dtype=dt)
ds[0] = np.array([1, 2, 3], dtype="<i4")
ds[1] = np.array([], dtype="<i4")
ds[2] = np.array([-7], dtype="<i4")
def gen_vlen_bytes(path):
dt = h5py.vlen_dtype(np.dtype("u1"))
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", (3,), dtype=dt)
ds[0] = np.array([0, 1, 2], dtype="u1")
ds[1] = np.array([], dtype="u1")
ds[2] = np.array([255], dtype="u1")
def gen_named_datatype(path):
with h5py.File(path, "w") as f:
f["t"] = np.dtype("<i4")
f.create_dataset("data", data=ramp("<i4"))
sid = h5s.create_simple((N,))
lowlevel_dataset(f, "shared", f["t"].id, sid, ramp("<i4"))
COMPOSITE_CASES = [
Case("compound_simple", "dtype-composite", gen_compound_simple, "compound_simple",
"two f32 members, no padding"),
Case("compound_nested", "dtype-composite", gen_compound_nested, "compound_nested",
"compound member inside a compound"),
Case("compound_with_string", "dtype-composite", gen_compound_with_string,
"compound_with_string", "fixed string member"),
Case("compound_padded", "dtype-composite", gen_compound_padded, "compound_padded",
"member offsets with gaps and trailing padding"),
Case("compound_dtype_v4", "dtype-composite", gen_compound_dtype_v4,
"compound_dtype_v4", "version-4 datatype message (libver v1.12 bounds)"),
Case("array_dtype", "dtype-composite", gen_array_dtype, "array_dtype",
"H5T_ARRAY element type (2x3 f64)"),
Case("enum_i8", "dtype-composite", gen_enum_i8, "enum_i8", "3-member i8 enum"),
Case("enum_i32", "dtype-composite", gen_enum_i32, "enum_i32",
"i32 enum with a negative member"),
Case("opaque", "dtype-composite", gen_opaque, "opaque", "H5T_OPAQUE with a tag"),
Case("bitfield", "dtype-composite", gen_bitfield, "bitfield",
"H5T_BITFIELD (STD_B8LE)"),
Case("ref_object", "dtype-composite", gen_ref_object, "ref_object",
"object references to a dataset and a group"),
Case("ref_region", "dtype-composite", gen_ref_region, "ref_region",
"dataset region references"),
Case("vlen_numeric", "dtype-composite", gen_vlen_numeric, "vlen_numeric",
"variable-length i32 sequences"),
Case("vlen_bytes", "dtype-composite", gen_vlen_bytes, "vlen_bytes",
"variable-length u8 sequences"),
Case("named_datatype", "dtype-composite", gen_named_datatype, "named_datatype",
"committed datatype object, and a dataset that shares it"),
]
def gen_layout_compact(path):
with h5py.File(path, "w") as f:
dcpl = h5p.create(h5p.DATASET_CREATE)
dcpl.set_layout(h5d.COMPACT)
sid = h5s.create_simple((16,))
lowlevel_dataset(
f, "data", h5t.STD_I32LE, sid, ramp("<i4", 16), dcpl=dcpl
)
def gen_layout_contiguous(path):
with h5py.File(path, "w") as f:
f.create_dataset("data", data=ramp("<i4", 16))
def gen_layout_contiguous_v110(path):
with h5py.File(path, "w", libver=("v110", "v110")) as f:
f.create_dataset("data", data=ramp("<i4", 16))
def gen_layout_chunked_v110(path):
with h5py.File(path, "w", libver=("v110", "v110")) as f:
ds = f.create_dataset("data", (16,), chunks=(16,), dtype="<i4")
ds[...] = ramp("<i4", 16)
def gen_chunkidx_btree1(path):
with h5py.File(path, "w", libver="earliest") as f:
ds = f.create_dataset(
"data", (8,), maxshape=(None,), chunks=(4,), dtype="<i4"
)
ds[...] = ramp("<i4")
def gen_chunkidx_single(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("data", (8,), chunks=(8,), dtype="<i4")
ds[...] = ramp("<i4")
def gen_chunkidx_implicit(path):
with h5py.File(path, "w", libver="latest") as f:
dcpl = chunked_dcpl((4,), alloc_time=h5d.ALLOC_TIME_EARLY)
sid = h5s.create_simple((16,))
lowlevel_dataset(f, "data", h5t.STD_I32LE, sid, ramp("<i4", 16), dcpl=dcpl)
def gen_chunkidx_farray(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("data", (16,), chunks=(4,), dtype="<i4")
ds[...] = ramp("<i4", 16)
def gen_chunkidx_earray(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset(
"data", (16,), maxshape=(None,), chunks=(4,), dtype="<i4"
)
ds[...] = ramp("<i4", 16)
def gen_chunkidx_earray_unlim_inner(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset(
"data", (4, 4), maxshape=(4, None), chunks=(2, 2), dtype="<i4"
)
ds[...] = ramp("<i4", 16).reshape(4, 4)
def gen_layout_contiguous_v108(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
f.create_dataset("data", data=ramp("<i4", 16))
def gen_layout_chunked_v108(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
ds = f.create_dataset("data", (16,), chunks=(16,), dtype="<i4")
ds[...] = ramp("<i4", 16)
def gen_chunkidx_earray_dim1(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset(
"data", (4, 4), maxshape=(4, None), chunks=(2, 4), dtype="<i4"
)
ds[...] = ramp("<i4", 16).reshape(4, 4)
def gen_external_storage(path):
raw = path.parent / (path.stem + "_ext.raw")
raw.write_bytes(ramp("<i4", 16).tobytes())
with h5py.File(path, "w") as f:
f.create_dataset(
"data", shape=(16,), dtype="<i4", external=[(raw.name, 0, 64)]
)
def gen_vds(path):
src = path.parent / (path.stem + "_src.h5")
with h5py.File(src, "w") as g:
g.create_dataset("src", data=ramp("<i4", 16))
layout = h5py.VirtualLayout(shape=(16,), dtype="<i4")
layout[...] = h5py.VirtualSource(src.name, "src", shape=(16,))
with h5py.File(path, "w") as f:
f.create_virtual_dataset("vds", layout)
def gen_external_unlimited(path):
raw = path.parent / (path.stem + "_ext.raw")
raw.write_bytes(ramp("<i4", 16).tobytes())
with h5py.File(path, "w") as f:
f.create_dataset(
"data",
shape=(16,),
maxshape=(None,),
dtype="<i4",
external=[(raw.name, 0, h5py.h5f.UNLIMITED)],
)
def gen_vds_unlim(path):
src = path.parent / (path.stem + "_src.h5")
with h5py.File(src, "w") as g:
g.create_dataset(
"src", data=ramp("<i4", 20).reshape(10, 2), maxshape=(None, 2),
chunks=(5, 2),
)
layout = h5py.VirtualLayout(shape=(1, 2), dtype="<i4", maxshape=(None, 2))
vsrc = h5py.VirtualSource(src.name, "src", shape=(1, 2), maxshape=(None, 2))
layout[: h5s.UNLIMITED, :] = vsrc[: h5s.UNLIMITED, :]
with h5py.File(path, "w") as f:
f.create_virtual_dataset("vds", layout)
def gen_vds_printf_unlim(path):
stem = path.stem
for b in range(3):
with h5py.File(path.parent / ("%s_b%d.h5" % (stem, b)), "w") as g:
g.create_dataset("data", data=ramp("<i4", 4) + 10 * b)
layout = h5py.VirtualLayout(shape=(1, 4), dtype="<i4", maxshape=(None, 4))
vsrc = h5py.VirtualSource("%s_b%%b.h5" % stem, "data", shape=(4,))
layout[: h5s.UNLIMITED, :] = vsrc
with h5py.File(path, "w") as f:
f.create_virtual_dataset("vds", layout)
def gen_vds_printf_gap(path):
stem = path.stem
for b in (0, 1, 3):
with h5py.File(path.parent / ("%s_b%d.h5" % (stem, b)), "w") as g:
g.create_dataset("data", data=ramp("<i4", 4) + 10 * b)
layout = h5py.VirtualLayout(shape=(1, 4), dtype="<i4", maxshape=(None, 4))
vsrc = h5py.VirtualSource("%s_b%%b.h5" % stem, "data", shape=(4,))
layout[: h5s.UNLIMITED, :] = vsrc
with h5py.File(path, "w") as f:
f.create_virtual_dataset("vds", layout, fillvalue=-7)
def gen_vds_view_trail(path):
with h5py.File(path, "w") as f:
f.create_dataset(
"src", data=ramp("<i4", 6).reshape(3, 2), maxshape=(None, 2),
chunks=(1, 2),
)
vsid = h5s.create_simple((1, 2), (h5s.UNLIMITED, 2))
vsid.select_hyperslab((0, 0), (h5s.UNLIMITED, 1), stride=(3, 1), block=(2, 2))
ssid = h5s.create_simple((1, 2), (h5s.UNLIMITED, 2))
ssid.select_hyperslab((0, 0), (h5s.UNLIMITED, 1), stride=(3, 1), block=(2, 2))
dcpl = h5p.create(h5p.DATASET_CREATE)
dcpl.set_layout(h5d.VIRTUAL)
dcpl.set_fill_value(np.array(-9, dtype="<i4"))
dcpl.set_virtual(vsid, b".", b"/src", ssid)
lowlevel_dataset(f, "vds", h5t.STD_I32LE, vsid, dcpl=dcpl)
def gen_vds_split(path):
with h5py.File(path, "w") as f:
f.create_dataset("src", data=ramp("<i4", 8).reshape(2, 4))
vsid = h5s.create_simple((4, 4))
vsid.select_hyperslab((0, 0), (2, 1), stride=(2, 1), block=(1, 4))
ssid = h5s.create_simple((2, 4))
ssid.select_all()
dcpl = h5p.create(h5p.DATASET_CREATE)
dcpl.set_layout(h5d.VIRTUAL)
dcpl.set_fill_value(np.array(-9, dtype="<i4"))
dcpl.set_virtual(vsid, b".", b"/src", ssid)
lowlevel_dataset(f, "vds", h5t.STD_I32LE, vsid, dcpl=dcpl)
def gen_chunkidx_btree2(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset(
"data", (4, 4), maxshape=(None, None), chunks=(2, 2), dtype="<i4"
)
ds[...] = ramp("<i4", 16).reshape(4, 4)
LAYOUT_CASES = [
Case("layout_compact", "layout", gen_layout_compact, "layout_compact",
"compact layout — data inside the object header"),
Case("layout_contiguous", "layout", gen_layout_contiguous, "layout_contiguous",
"contiguous layout"),
Case("layout_contiguous_v110", "layout", gen_layout_contiguous_v110,
"layout_contiguous_v110",
"contiguous layout under v1.10 bounds — data layout message v4"),
Case("layout_chunked_v110", "layout", gen_layout_chunked_v110,
"layout_chunked_v110",
"chunked layout under v1.10 bounds — the control for the case above"),
Case("chunkidx_btree1", "layout", gen_chunkidx_btree1, "chunkidx_btree1",
"layout v3 + version-1 B-tree chunk index (libver earliest)"),
Case("chunkidx_single", "layout", gen_chunkidx_single, "chunkidx_single",
"single-chunk index"),
Case("chunkidx_implicit", "layout", gen_chunkidx_implicit, "chunkidx_implicit",
"implicit index — fixed shape, early allocation, no filter"),
Case("chunkidx_farray", "layout", gen_chunkidx_farray, "chunkidx_farray",
"fixed-array index"),
Case("chunkidx_earray", "layout", gen_chunkidx_earray, "chunkidx_earray",
"extensible-array index — one unlimited dimension"),
Case("chunkidx_earray_unlim_inner", "layout", gen_chunkidx_earray_unlim_inner,
"chunkidx_earray_unlim_inner",
"extensible-array index — the unlimited dimension is dim 1, not dim 0"),
Case("chunkidx_btree2", "layout", gen_chunkidx_btree2, "chunkidx_btree2",
"version-2 B-tree index — two unlimited dimensions"),
Case("layout_contiguous_v108", "layout", gen_layout_contiguous_v108,
"layout_contiguous_v108",
"contiguous layout under v1.8 bounds — the v1.10 pair's control"),
Case("layout_chunked_v108", "layout", gen_layout_chunked_v108,
"layout_chunked_v108", "chunked layout under v1.8 bounds"),
Case("chunkidx_earray_dim1", "layout", gen_chunkidx_earray_dim1,
"chunkidx_earray_dim1",
"extensible-array index whose unlimited dimension is not the first"),
Case("external_storage", "layout", gen_external_storage, "external_storage",
"contiguous data held in an external raw file",
ext_files=("_ext.raw",)),
Case("vds", "layout", gen_vds, "vds",
"virtual dataset mapped onto a dataset in a sibling file",
ext_files=("_src.h5",)),
Case("external_unlimited", "layout", gen_external_unlimited,
"external_unlimited",
"H5O_EFL_UNLIMITED on the last external slot of an unlimited dataset",
ext_files=("_ext.raw",)),
Case("vds_unlim", "layout", gen_vds_unlim, "vds_unlim",
"virtual dataset whose mapping is unlimited on both sides — the "
"extent comes from the source",
ext_files=("_src.h5",)),
Case("vds_printf_gap", "layout", gen_vds_printf_gap, "vds_printf_gap",
"printf mapping with block 2 missing, read at the default printf "
"gap of 0 — the extent stops at the gap"),
Case("vds_printf_gap_1", "layout", gen_vds_printf_gap, "vds_printf_gap",
"the same file read with H5Pset_virtual_printf_gap(1) — block 3 is "
"reached and block 2 reads as the fill value",
access={"printf_gap": 1}),
Case("vds_printf_gap_first_missing", "layout", gen_vds_printf_gap, "vds_printf_gap",
"the same file under H5D_VDS_FIRST_MISSING with a gap of 2, which "
"H5D__virtual_init forces back to 0 (H5Dvirtual.c:2182-2188)",
access={"view": "first_missing", "printf_gap": 2}),
Case("vds_view_trail", "layout", gen_vds_view_trail, "vds_view_trail",
"unlimited mapping whose stride exceeds its block, read at the "
"default H5D_VDS_LAST_AVAILABLE view"),
Case("vds_view_trail_first_missing", "layout", gen_vds_view_trail, "vds_view_trail",
"the same file under H5D_VDS_FIRST_MISSING — the extent runs on to "
"where the next block would start",
access={"view": "first_missing"}),
Case("vds_split", "layout", gen_vds_split, "vds_split",
"a same-file mapping whose virtual and source selections decompose "
"into different numbers of boxes"),
Case("vds_printf_unlim", "layout", gen_vds_printf_unlim, "vds_printf_unlim",
"printf-pattern source name over an unlimited virtual selection — "
"one source file per block",
ext_files=("_b0.h5", "_b1.h5", "_b2.h5")),
]
def _filter_case(name, rust, note, **kw):
def gen(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("data", (64,), chunks=(16,), dtype="<i4", **kw)
ds[...] = ramp("<i4", 64)
return Case(name, "filter", gen, rust, note)
FILTER_CASES = [
_filter_case("filter_deflate", "filter_deflate", "deflate level 6",
compression="gzip", compression_opts=6),
_filter_case("filter_shuffle", "filter_shuffle", "shuffle only",
shuffle=True),
_filter_case("filter_fletcher32", "filter_fletcher32", "fletcher32 checksum",
fletcher32=True),
_filter_case("filter_deflate_shuffle", "filter_deflate_shuffle",
"shuffle then deflate",
compression="gzip", compression_opts=6, shuffle=True),
_filter_case("filter_scaleoffset", "filter_scaleoffset",
"scale-offset, library-computed minimum bits",
scaleoffset=0),
_filter_case("filter_szip_ec", "filter_szip_ec",
"szip entropy coding, 8 pixels per block",
compression="szip", compression_opts=("ec", 8)),
_filter_case("filter_szip_nn", "filter_szip_nn",
"szip nearest neighbour, 16 pixels per block",
compression="szip", compression_opts=("nn", 16)),
]
def gen_fill_default(path):
with h5py.File(path, "w", libver="latest") as f:
f.create_dataset("data", (16,), chunks=(4,), dtype="<i4")
def gen_fill_set_int(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("data", (16,), chunks=(4,), dtype="<i4", fillvalue=-1)
ds[0:4] = ramp("<i4", 4)
def gen_fill_set_float_nan(path):
with h5py.File(path, "w", libver="latest") as f:
f.create_dataset(
"data", (16,), chunks=(4,), dtype="<f8", fillvalue=np.float64("nan")
)
FILL_CASES = [
Case("fill_default", "fillvalue", gen_fill_default, "fill_default",
"default (zero) fill, nothing written"),
Case("fill_set_int", "fillvalue", gen_fill_set_int, "fill_set_int",
"user-defined integer fill, first chunk written"),
Case("fill_set_float_nan", "fillvalue", gen_fill_set_float_nan,
"fill_set_float_nan", "user-defined NaN fill"),
]
def gen_space_scalar(path):
with h5py.File(path, "w") as f:
f.create_dataset("data", data=np.int32(42))
def gen_space_null(path):
with h5py.File(path, "w") as f:
f["data"] = h5py.Empty("<i4")
def gen_space_zerosized(path):
with h5py.File(path, "w") as f:
f.create_dataset("data", (0,), dtype="<i4")
def gen_space_unlimited_resized(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("data", (4,), maxshape=(None,), chunks=(4,), dtype="<i4")
ds[...] = ramp("<i4", 4)
ds.resize((12,))
ds[4:12] = ramp("<i4", 8) + 100
SPACE_CASES = [
Case("space_scalar", "dataspace", gen_space_scalar, "space_scalar",
"scalar (rank 0) dataspace"),
Case("space_null", "dataspace", gen_space_null, "space_null",
"NULL dataspace — no elements at all"),
Case("space_zerosized", "dataspace", gen_space_zerosized, "space_zerosized",
"simple dataspace with a zero-length dimension"),
Case("space_unlimited_resized", "dataspace", gen_space_unlimited_resized,
"space_unlimited_resized", "unlimited maxshape, grown after creation"),
]
def gen_groups_nested(path):
with h5py.File(path, "w") as f:
g = f.create_group("a")
h = g.create_group("b")
h.create_group("c")
h.create_dataset("leaf", data=ramp("<i4"))
f.create_dataset("top", data=ramp("<i4"))
def gen_link_hard(path):
with h5py.File(path, "w") as f:
f.create_dataset("orig", data=ramp("<i4"))
f["alias"] = f["orig"]
def gen_link_soft(path):
with h5py.File(path, "w") as f:
f.create_dataset("orig", data=ramp("<i4"))
f["alias"] = h5py.SoftLink("/orig")
def gen_link_external(path):
target = path.parent / (path.stem + "_ext.h5")
with h5py.File(target, "w") as g:
g.create_dataset("payload", data=ramp("<i4"))
with h5py.File(path, "w") as f:
f.create_dataset("orig", data=ramp("<i4"))
f["ext"] = h5py.ExternalLink(target.name, "/payload")
def gen_link_external_read(path):
target = path.parent / (path.stem + "_data.h5")
with h5py.File(target, "w") as g:
g.create_dataset("top", data=ramp("<f8"))
g.create_group("deep").create_dataset("inner", data=ramp("<i2"))
with h5py.File(path, "w") as f:
f["direct"] = h5py.ExternalLink(target.name, "/top")
f["nested"] = h5py.ExternalLink(target.name, "/deep/inner")
f["gone_object"] = h5py.ExternalLink(target.name, "/absent")
f["gone_file"] = h5py.ExternalLink("no_such_file.h5", "/top")
def gen_link_nonascii(path):
with h5py.File(path, "w", libver="earliest") as f:
f.create_dataset("데이터", data=ramp("<i4"))
f.create_group("plain")
f.create_group("그룹").create_dataset("inner", data=ramp("<i4", 4))
f.create_group("ascii_only").create_dataset("inner", data=ramp("<i4", 4))
def gen_links_dense(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
g = f.create_group("g", track_order=True)
for i in range(12):
g.create_dataset("d%02d" % i, data=np.array([i], dtype="<i4"))
def gen_track_order(path):
with h5py.File(path, "w", track_order=True) as f:
for name in ("zebra", "apple", "mango"):
f.create_group(name)
for i, key in enumerate(("zeta", "alpha", "mu")):
f.attrs.create(key, np.int32(i))
g = f.create_group("g", track_order=True)
g.create_dataset("data", data=ramp("<i4"))
g.attrs.create("second", np.int32(2))
g.attrs.create("first", np.int32(1))
def gen_group_storage_modern_root(path):
with h5py.File(path, "w", track_order=True) as f:
legacy = f.create_group("legacy")
legacy.create_dataset("a", data=ramp("<i4"))
legacy.create_group("inner").create_dataset("c", data=ramp("<i2"))
def gen_group_storage_legacy_root(path):
with h5py.File(path, "w") as f:
f.create_group("legacy").create_dataset("a", data=ramp("<i4"))
modern = f.create_group("modern", track_order=True)
modern.create_dataset("b", data=ramp("<f8"))
modern.create_group("inner").create_dataset("c", data=ramp("<i2"))
LINK_CASES = [
Case("groups_nested", "group", gen_groups_nested, "groups_nested",
"three levels of nested groups plus an empty leaf group"),
Case("link_hard", "link", gen_link_hard, "link_hard",
"two names for one object"),
Case("link_soft", "link", gen_link_soft, "link_soft", "soft link to /orig"),
Case("link_external", "link", gen_link_external, "link_external",
"external link into a sibling file",
ext_files=("_ext.h5",)),
Case("link_external_read", "link", gen_link_external_read,
"link_external_read",
"datasets read through external links, plus a dangling object and a "
"dangling file",
ext_files=("_data.h5",)),
Case("link_nonascii", "link", gen_link_nonascii, "link_nonascii",
"non-ASCII link names at the earliest bound — the root converts to "
"link messages, the ASCII-named subgroups keep their symbol tables"),
Case("links_dense", "link", gen_links_dense, "links_dense",
"12 links in one group — dense link storage (fractal heap + v2 B-tree)"),
Case("track_order", "group", gen_track_order, "track_order",
"creation-order indices on links and attributes"),
Case("group_storage_modern_root", "group", gen_group_storage_modern_root,
"group_storage_modern_root",
"symbol-table group with children under a link-message root"),
Case("group_storage_legacy_root", "group", gen_group_storage_legacy_root,
"group_storage_legacy_root",
"link-message group with children under a symbol-table root"),
]
def gen_attr_scalar_num(path):
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", data=ramp("<i4"))
ds.attrs.create("gain", np.float64(2.5))
ds.attrs.create("count", np.int32(7))
def gen_attr_array_num(path):
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", data=ramp("<i4"))
ds.attrs.create("offsets", np.arange(4, dtype="<i4"))
ds.attrs.create("matrix", np.arange(6, dtype="<f8").reshape(2, 3))
def gen_attr_string(path):
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", data=ramp("<i4"))
ds.attrs.create("units", "volt", dtype=h5py.string_dtype("utf-8"))
f.create_group("g").attrs.create(
"NX_class", "NXdetector", dtype=h5py.string_dtype("utf-8")
)
def gen_attrs_dense(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
ds = f.create_dataset("data", data=ramp("<i4"))
for i in range(12):
ds.attrs.create("a%02d" % i, np.int32(i))
def gen_attrs_dense_group(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
g = f.create_group("g")
for i in range(12):
g.attrs.create("g%02d" % i, np.int32(i))
for i in range(12):
f.attrs.create("r%02d" % i, np.int32(i))
f.create_dataset("data", data=ramp("<i4"))
def gen_attr_on_root(path):
with h5py.File(path, "w") as f:
f.attrs.create("title", "root", dtype=h5py.string_dtype("utf-8"))
f.attrs.create("version", np.int64(3))
f.create_dataset("data", data=ramp("<i4"))
def gen_attr_ref_object(path):
with h5py.File(path, "w") as f:
ds = f.create_dataset("data", data=ramp("<i4"))
g = f.create_group("grp")
ds.attrs.create("neighbours", np.array([ds.ref, g.ref], dtype=h5py.ref_dtype))
g.attrs.create("owner", ds.ref, dtype=h5py.ref_dtype)
f.attrs.create("entry", np.array([g.ref, ds.ref], dtype=h5py.ref_dtype))
def gen_attr_large(path):
with h5py.File(path, "w", libver=("v108", "v108")) as f:
ds = f.create_dataset("data", data=ramp("<i4"))
ds.attrs.create("big", np.arange(25600, dtype="<i4"))
ATTR_CASES = [
Case("attr_scalar_num", "attribute", gen_attr_scalar_num, "attr_scalar_num",
"scalar f64 and i32 attributes"),
Case("attr_array_num", "attribute", gen_attr_array_num, "attr_array_num",
"1-D and 2-D numeric attributes"),
Case("attr_string", "attribute", gen_attr_string, "attr_string",
"vlen UTF-8 string attributes on a dataset and a group"),
Case("attrs_dense", "attribute", gen_attrs_dense, "attrs_dense",
"12 attributes — dense attribute storage"),
Case("attrs_dense_group", "attribute", gen_attrs_dense_group,
"attrs_dense_group",
"12 attributes on a group and on the root — dense storage"),
Case("attr_on_root", "attribute", gen_attr_on_root, "attr_on_root",
"attributes on the root group"),
Case("attr_large", "attribute", gen_attr_large, "attr_large",
"single 100 KiB attribute — dense storage forced by size, not count"),
Case("attr_ref_object", "attribute", gen_attr_ref_object, "attr_ref_object",
"object references stored in attributes of a dataset, a group and "
"the root"),
]
def _libver_case(name, libver, rust, note):
def gen(path):
with h5py.File(path, "w", libver=libver) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_group("g")
return Case(name, "superblock", gen, rust, note)
def gen_userblock(path):
with h5py.File(path, "w", userblock_size=512) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_group("g")
prefix = b"#!/bin/sh\n# userblock\n"
with open(path, "r+b") as fh:
fh.write(prefix + b"#" * (512 - len(prefix) - 1) + b"\n")
def _reopen_append_case(name, libver, rust, note):
def gen(path):
with h5py.File(path, "w", libver=libver) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_group("g")
with h5py.File(path, "a") as f:
f.create_dataset("appended", data=ramp("<i4", 16).reshape(4, 4),
chunks=(2, 4))
return Case(name, "superblock", gen, rust, note)
LIBVER_CASES = [
_libver_case("libver_earliest", "earliest", "libver_earliest",
"libver earliest — superblock v0, symbol-table groups"),
_libver_case("libver_v108", ("v108", "v108"), "libver_v108",
"libver v1.8 bounds"),
_libver_case("libver_v110", ("v110", "v110"), "libver_v110",
"libver v1.10 bounds"),
_libver_case("libver_latest", "latest", "libver_latest",
"libver latest — superblock v3, new-style groups"),
Case("userblock", "superblock", gen_userblock, "userblock",
"512-byte userblock — the superblock, and every address, is based at 512"),
_reopen_append_case(
"reopen_append_earliest", "earliest", "reopen_append_earliest",
"classic file reopened under the default fapl — stays superblock v0, "
"the appended chunked dataset takes the version-1 B-tree"),
_reopen_append_case(
"reopen_append_v108", ("v108", "v108"), "reopen_append_v108",
"v1.8 file reopened under the default fapl — stays superblock v2, "
"the appended chunked dataset takes the version-1 B-tree"),
_reopen_append_case(
"reopen_append_latest", "latest", "reopen_append_latest",
"v1.10+ file reopened under the default fapl — stays superblock v3, "
"the appended chunked dataset takes a v1.10 index"),
]
def gen_swmr_created(path):
with h5py.File(path, "w", libver="latest") as f:
ds = f.create_dataset("stream", (0, 4), maxshape=(None, 4), chunks=(1, 4),
dtype="<f4")
f.swmr_mode = True
for i in range(8):
ds.resize((i + 1, 4))
ds[i, :] = np.arange(i * 4, i * 4 + 4, dtype="<f4")
ds.flush()
def gen_large_multi_mb(path):
with h5py.File(path, "w", libver="latest") as f:
data = np.arange(512 * 512, dtype="<f8").reshape(512, 512)
f.create_dataset("big", data=data, chunks=(64, 512))
def gen_fsm_persist(path):
with h5py.File(path, "w", fs_strategy="fsm", fs_persist=True, fs_threshold=1) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_dataset("bulk", data=ramp("<i4", 256))
f.create_group("g")
with h5py.File(path, "a") as f:
del f["bulk"]
f.create_dataset("appended", data=ramp("<i4"))
def gen_fsm_persist_page(path):
with h5py.File(path, "w", fs_strategy="page", fs_persist=True, fs_threshold=1) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_dataset("bulk", data=ramp("<i4", 256))
f.create_group("g")
with h5py.File(path, "a") as f:
del f["bulk"]
f.create_dataset("appended", data=ramp("<i4"))
def gen_fsm_page_size(path):
with h5py.File(path, "w", fs_strategy="page", fs_persist=True,
fs_threshold=1, fs_page_size=512) as f:
f.create_dataset("data", data=ramp("<i4"))
f.create_dataset("bulk", data=ramp("<i4", 256))
f.create_group("g")
with h5py.File(path, "a") as f:
del f["bulk"]
f.create_dataset("appended", data=ramp("<i4"))
MISC_CASES = [
Case("swmr_created", "swmr", gen_swmr_created, "swmr_created",
"file created through the SWMR writer path and appended frame by frame"),
Case("fsm_persist", "freespace", gen_fsm_persist, "fsm_persist",
"persisting FSM_AGGR file reopened and appended — the freed blocks "
"must come back as free-space manager sections"),
Case("fsm_persist_page", "freespace", gen_fsm_persist_page, "fsm_persist_page",
"persisting PAGE file reopened and appended — the freed blocks must "
"come back as page-shaped free-space manager sections"),
Case("fsm_page_size", "freespace", gen_fsm_page_size, "fsm_page_size",
"persisting PAGE file on a 512-byte file-space page — the non-default "
"size must reach the message and shape the allocation"),
Case("large_multi_mb", "bulk", gen_large_multi_mb, "large_multi_mb",
"2 MiB chunked f64 dataset — payload compared by SHA-256"),
]
FIXTURE_DIR = pathlib.Path(__file__).resolve().parent.parent / "tests" / "fixtures"
def _fixture_case(name, fixture, generator, group, note, rust=None):
def gen(path):
src = FIXTURE_DIR / fixture
if not src.exists():
raise FileNotFoundError(
"%s is missing; regenerate it with tests/fixtures/%s"
% (src, generator)
)
shutil.copyfile(src, path)
return Case(name, group, gen, rust, note)
def _fixture_append_case(name, fixture, generator, group, note, rust):
def gen(path):
src = FIXTURE_DIR / fixture
if not src.exists():
raise FileNotFoundError(
"%s is missing; regenerate it with tests/fixtures/%s"
% (src, generator)
)
shutil.copyfile(src, path)
with h5py.File(path, "a") as f:
f.create_dataset("appended", data=ramp("<i4", 8))
return Case(name, group, gen, rust, note)
FIXTURE_CASES = [
_fixture_case(
"sohm_list", "sohm_list.h5", "gen_sohm.sh", "sohm",
"shared datatype/dataspace/attribute messages, list index "
"(H5Pset_shared_mesg_index) + a committed datatype",
rust="sohm_list",
),
_fixture_case(
"sohm_btree", "sohm_btree.h5", "gen_sohm.sh", "sohm",
"the same file with the shared-message index forced to a v2 B-tree",
rust="sohm_btree",
),
_fixture_append_case(
"sohm_list_append", "sohm_list.h5", "gen_sohm.sh", "sohm",
"a file with a shared-message list index reopened and appended to — "
"the table is laid out whole, so the append replaces it",
rust="sohm_list_append",
),
_fixture_append_case(
"sohm_btree_append", "sohm_btree.h5", "gen_sohm.sh", "sohm",
"the same reopen over a v2 B-tree index",
rust="sohm_btree_append",
),
_fixture_case(
"ochk_root", "ochk_root.h5", "gen_ochk.sh", "objectheader",
"root group whose object header spills into two continuation chunks",
rust="ochk_root",
),
_fixture_case(
"vds_late_layout", "vds_late_layout.h5", "gen_vds_late_layout.sh",
"layout",
"virtual datasets built by H5Pset_virtual with and without a prior "
"H5Pset_layout — the pairs differ only in allocation time "
"(H5Pdcpl.c:2146 pokes the layout past H5P__set_layout's default)",
),
]
ALL_CASES = (
INT_CASES
+ FLOAT_CASES
+ STRING_CASES
+ COMPOSITE_CASES
+ LAYOUT_CASES
+ FILTER_CASES
+ FILL_CASES
+ SPACE_CASES
+ LINK_CASES
+ ATTR_CASES
+ LIBVER_CASES
+ MISC_CASES
+ FIXTURE_CASES
)
def by_name(name):
for c in ALL_CASES:
if c.name == name:
return c
raise KeyError(name)
if __name__ == "__main__":
print("%d cases" % len(ALL_CASES))
for c in ALL_CASES:
print(" %-24s %-16s rust=%s" % (c.name, c.group, c.rust or "-"))