Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use std::borrow::Cow;
8
9use crate::attribute::AttrBuilder;
10use crate::error::{Hdf5Error, Result};
11use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
12use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
13use crate::format::messages::filter::Filter;
14use crate::format::messages::virtual_mapping::VirtualMapping;
15use crate::format::reference::{Reference, ReferenceTarget};
16use crate::format::selection::check_hyperslab;
17use crate::format::selection::Selection;
18use crate::format::storage_kind::AttributeStorage;
19use crate::io::file_handle::ReadDst;
20use crate::io::reader::{read_image_into_new, ExternalFileSegment};
21use crate::io::writer::ChunkIndexKind;
22use crate::types::H5Type;
23
24// ---------------------------------------------------------------------------
25// DatasetBuilder
26// ---------------------------------------------------------------------------
27
28/// A fluent builder for creating datasets.
29///
30/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
31///
32/// ```no_run
33/// # use rust_hdf5::H5File;
34/// let file = H5File::create("builder.h5").unwrap();
35/// let ds = file.new_dataset::<f32>()
36///     .shape(&[10, 20])
37///     .create("temperatures")
38///     .unwrap();
39/// ```
40pub struct DatasetBuilder<T: H5Type> {
41    file_inner: SharedInner,
42    shape: Option<Vec<usize>>,
43    is_null: bool,
44    chunk_dims: Option<Vec<usize>>,
45    max_shape: Option<Vec<Option<usize>>>,
46    is_compact: bool,
47    early_allocation: bool,
48    deflate_level: Option<u32>,
49    shuffle: bool,
50    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
51    group_path: Option<String>,
52    fill_value: Option<Vec<u8>>,
53    fill_time: Option<FillTime>,
54    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
55    committed_type: Option<String>,
56    references: Option<ReferenceElement>,
57    external: Option<Vec<(String, u64, u64)>>,
58    efile_prefix: Option<String>,
59    virtual_mappings: Vec<VirtualMapping>,
60    _marker: std::marker::PhantomData<T>,
61}
62
63/// Which reference a `*_references()` builder call asked the elements to be.
64///
65/// One field rather than a flag per kind: an element is a whole-object
66/// reference or a region reference, never both, and the width of each is only
67/// known once the file's address size is (see
68/// [`DatatypeMessage::object_reference`] and
69/// [`DatatypeMessage::region_reference`]).
70///
71/// [`DatatypeMessage::object_reference`]: crate::format::messages::datatype::DatatypeMessage::object_reference
72/// [`DatatypeMessage::region_reference`]: crate::format::messages::datatype::DatatypeMessage::region_reference
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74enum ReferenceElement {
75    /// `H5T_STD_REF_OBJ`.
76    Object,
77    /// `H5T_STD_REF_DSETREG`.
78    Region,
79    /// `H5T_STD_REF`, the 1.12 form. One datatype for all three 1.12 kinds:
80    /// the element leads with the kind it holds, so `H5T__ref_disk_getsize`
81    /// sizes every element for the widest of them and a dataset of this type
82    /// may hold objects, regions and attributes alike.
83    Revised,
84}
85
86impl ReferenceElement {
87    /// The stored datatype for this kind in a file with `ctx`'s address size.
88    fn datatype(
89        self,
90        ctx: &crate::format::FormatContext,
91    ) -> crate::format::messages::datatype::DatatypeMessage {
92        use crate::format::messages::datatype::DatatypeMessage;
93        match self {
94            Self::Object => DatatypeMessage::object_reference(ctx),
95            Self::Region => DatatypeMessage::region_reference(ctx),
96            Self::Revised => DatatypeMessage::std_object_reference(ctx),
97        }
98    }
99}
100
101impl<T: H5Type> DatasetBuilder<T> {
102    pub(crate) fn new(file_inner: SharedInner) -> Self {
103        Self {
104            file_inner,
105            shape: None,
106            is_null: false,
107            chunk_dims: None,
108            max_shape: None,
109            is_compact: false,
110            early_allocation: false,
111            deflate_level: None,
112            shuffle: false,
113            custom_pipeline: None,
114            group_path: None,
115            fill_value: None,
116            fill_time: None,
117            datatype_override: None,
118            committed_type: None,
119            references: None,
120            external: None,
121            efile_prefix: None,
122            virtual_mappings: Vec::new(),
123            _marker: std::marker::PhantomData,
124        }
125    }
126
127    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
128        Self {
129            file_inner,
130            shape: None,
131            is_null: false,
132            chunk_dims: None,
133            max_shape: None,
134            is_compact: false,
135            early_allocation: false,
136            deflate_level: None,
137            shuffle: false,
138            custom_pipeline: None,
139            group_path: Some(group_path),
140            fill_value: None,
141            fill_time: None,
142            datatype_override: None,
143            committed_type: None,
144            references: None,
145            external: None,
146            efile_prefix: None,
147            virtual_mappings: Vec::new(),
148            _marker: std::marker::PhantomData,
149        }
150    }
151
152    /// Set the dataset dimensions.
153    ///
154    /// This is required before calling [`create`](Self::create), unless
155    /// [`null`](Self::null) was called instead.
156    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
157    #[must_use]
158    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
159        self.shape = Some(dims.as_ref().to_vec());
160        self
161    }
162
163    /// Create a scalar (0-dimensional) dataset holding a single value.
164    #[must_use]
165    pub fn scalar(mut self) -> Self {
166        self.shape = Some(vec![]);
167        self
168    }
169
170    /// Create a dataset with the NULL dataspace: no elements at all.
171    ///
172    /// Distinct from [`scalar`](Self::scalar), which holds exactly one
173    /// element. A NULL dataset holds zero bytes of data and cannot be
174    /// written to — [`write_raw`](H5Dataset::write_raw) and
175    /// [`write_raw_bytes`](H5Dataset::write_raw_bytes) return an error, and
176    /// it cannot be chunked or filtered, matching h5py's `h5py.Empty`.
177    #[must_use]
178    pub fn null(mut self) -> Self {
179        self.is_null = true;
180        self
181    }
182
183    /// Set chunk dimensions for chunked storage.
184    ///
185    /// When set, the dataset uses chunked storage with the extensible array
186    /// index. You should also call [`max_shape`](Self::max_shape) or
187    /// [`resizable`](Self::resizable) to allow extending.
188    #[must_use]
189    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
190        self.chunk_dims = Some(chunk_dims.to_vec());
191        self
192    }
193
194    /// Make all dimensions unlimited (resizable).
195    ///
196    /// This sets max_dims to u64::MAX for all dimensions.
197    #[must_use]
198    pub fn resizable(mut self) -> Self {
199        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
200        self
201    }
202
203    /// Set maximum dimensions. `None` means unlimited for that dimension.
204    #[must_use]
205    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
206        self.max_shape = Some(max.to_vec());
207        self
208    }
209
210    /// Store the raw data inside the dataset's object header —
211    /// `H5Pset_layout(dcpl, H5D_COMPACT)`.
212    ///
213    /// A compact dataset costs no data block and no second seek to read, which
214    /// suits the small per-run constants an analysis file is full of. It is
215    /// bounded by what one object header message can hold
216    /// ([`MAX_COMPACT_DATA`](crate::MAX_COMPACT_DATA) bytes) and it
217    /// is fixed in size: [`chunk`](Self::chunk), a filter, and an unlimited
218    /// [`max_shape`](Self::max_shape) are all rejected at
219    /// [`create`](Self::create), as libhdf5 rejects them.
220    ///
221    /// ```no_run
222    /// # use rust_hdf5::H5File;
223    /// let file = H5File::create("compact.h5").unwrap();
224    /// let ds = file.new_dataset::<i32>()
225    ///     .shape([16])
226    ///     .compact()
227    ///     .create("data")
228    ///     .unwrap();
229    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
230    /// ```
231    #[must_use]
232    pub fn compact(mut self) -> Self {
233        self.is_compact = true;
234        self
235    }
236
237    /// Allocate the whole of a chunked dataset's storage at create —
238    /// `H5Pset_alloc_time(dcpl, H5D_ALLOC_TIME_EARLY)`, h5py's
239    /// `alloc_time=h5d.ALLOC_TIME_EARLY`.
240    ///
241    /// Every chunk exists, holding the fill value, before anything is
242    /// written, so an unwritten chunk costs a read of fill bytes rather than
243    /// a miss. On a fixed-shape unfiltered dataset that is also what lets
244    /// libhdf5 pick its cheapest chunk index — the *implicit* index, which
245    /// is no index at all: the chunks are one contiguous run in grid order
246    /// and a chunk's address is arithmetic. This builder makes the same
247    /// choice under the same conditions, so such a dataset is written with
248    /// no index structure in the file.
249    ///
250    /// Ignored by storage that has no chunk grid to allocate: contiguous,
251    /// compact and NULL-dataspace datasets.
252    ///
253    /// ```no_run
254    /// # use rust_hdf5::H5File;
255    /// let file = H5File::create("implicit.h5").unwrap();
256    /// let ds = file.new_dataset::<i32>()
257    ///     .shape([16])
258    ///     .chunk(&[4])
259    ///     .early_allocation()
260    ///     .create("data")
261    ///     .unwrap();
262    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
263    /// ```
264    #[must_use]
265    pub fn early_allocation(mut self) -> Self {
266        self.early_allocation = true;
267        self
268    }
269
270    /// Enable deflate (gzip) compression with the given level (0-9).
271    ///
272    /// Requires chunked storage (call `.chunk()` before `.create()`).
273    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
274    #[must_use]
275    pub fn deflate(mut self, level: u32) -> Self {
276        self.deflate_level = Some(level);
277        self
278    }
279
280    /// Enable the shuffle filter — `H5Pset_shuffle(dcpl)`, h5py's
281    /// `shuffle=True`.
282    ///
283    /// Shuffle reorders a chunk's bytes by their position within an element,
284    /// which typically improves how well a compressor behind it does on
285    /// numeric data. It is a permutation, not a compressor: on its own it
286    /// leaves the chunk exactly as large as it was, which is what
287    /// `H5Pset_shuffle` without a compressor writes. Combine it with
288    /// [`deflate`](Self::deflate) to compress the shuffled stream. Requires
289    /// chunked storage.
290    ///
291    /// The element width the filter records is the dataset's, so a
292    /// [`datatype`](Self::datatype) override is what it follows when the
293    /// stored element is not `T` itself.
294    #[must_use]
295    pub fn shuffle(mut self) -> Self {
296        self.shuffle = true;
297        self
298    }
299
300    /// Enable shuffle + deflate compression — the same pipeline as
301    /// `.shuffle().deflate(level)`.
302    ///
303    /// Shuffle reorders bytes by position within elements before compression,
304    /// which typically improves compression ratios for numeric data.
305    /// Requires chunked storage.
306    #[must_use]
307    pub fn shuffle_deflate(mut self, level: u32) -> Self {
308        self.shuffle = true;
309        self.deflate_level = Some(level);
310        self
311    }
312
313    /// Enable Zstandard compression with the given level (1-22, default 3).
314    ///
315    /// Requires chunked storage (call `.chunk()` before `.create()`).
316    #[must_use]
317    pub fn zstd(mut self, level: u32) -> Self {
318        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
319        self
320    }
321
322    /// Set a custom filter pipeline for compression.
323    ///
324    /// This takes precedence over [`deflate`](Self::deflate) and
325    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
326    #[must_use]
327    pub fn filter_pipeline(
328        mut self,
329        pipeline: crate::format::messages::filter::FilterPipeline,
330    ) -> Self {
331        self.custom_pipeline = Some(pipeline);
332        self
333    }
334
335    /// Override the stored element datatype.
336    ///
337    /// By default the dataset is created with the datatype derived from the
338    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
339    /// different on-disk datatype than the in-memory element type — for
340    /// example a reduced-precision fixed-point type that matches an N-bit
341    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
342    /// override must equal `T::element_size()`; the N-bit filter packs the
343    /// significant bits within that fixed footprint.
344    ///
345    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
346    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
347    #[must_use]
348    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
349        self.datatype_override = Some(dt);
350        self
351    }
352
353    /// Build the dataset on the committed (named) datatype at `path` —
354    /// h5py's `dtype=f["name"]`, `H5Dcreate2` with a committed type id.
355    ///
356    /// The dataset does not describe its type: its header stores a pointer to
357    /// that object, so the type is defined once and every dataset sharing it
358    /// is guaranteed to agree. The type comes from the committed object, so
359    /// this supersedes both `T` and [`datatype`](Self::datatype).
360    ///
361    /// The path is resolved at [`create`](Self::create), which fails when no
362    /// committed datatype is there — commit it with
363    /// [`H5File::commit_datatype`](crate::file::H5File::commit_datatype)
364    /// first.
365    ///
366    /// ```no_run
367    /// # use rust_hdf5::H5File;
368    /// # use rust_hdf5::format::messages::datatype::DatatypeMessage;
369    /// let file = H5File::create("committed.h5").unwrap();
370    /// file.commit_datatype("temperature", DatatypeMessage::f64_type()).unwrap();
371    /// file.new_dataset::<f64>()
372    ///     .committed_type("temperature")
373    ///     .shape([4])
374    ///     .create("readings")
375    ///     .unwrap();
376    /// ```
377    #[must_use]
378    pub fn committed_type(mut self, path: &str) -> Self {
379        self.committed_type = Some(path.to_string());
380        self
381    }
382
383    /// Store object references — h5py's `h5py.ref_dtype`.
384    ///
385    /// The elements are written with
386    /// [`write_object_references`](H5Dataset::write_object_references) and
387    /// name objects by path. The element width is the file's address size, so
388    /// the datatype is resolved at [`create`](Self::create) rather than here;
389    /// it overrides both `T` and any [`datatype`](Self::datatype) call.
390    ///
391    /// ```no_run
392    /// # use rust_hdf5::H5File;
393    /// let file = H5File::create("refs.h5").unwrap();
394    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
395    /// let refs = file.new_dataset::<u64>()
396    ///     .object_references()
397    ///     .shape([1])
398    ///     .create("refs")
399    ///     .unwrap();
400    /// refs.write_object_references(&["/target"]).unwrap();
401    /// file.close().unwrap();
402    /// ```
403    #[must_use]
404    pub fn object_references(mut self) -> Self {
405        self.references = Some(ReferenceElement::Object);
406        self
407    }
408
409    /// Store revised object references — the 1.12 `H5T_STD_REF`.
410    ///
411    /// Same paths and same [`write_object_references`](H5Dataset::write_object_references)
412    /// call as [`object_references`](Self::object_references); only the stored
413    /// element differs, carrying the reference's kind alongside the address so
414    /// one datatype can hold every reference kind. h5py cannot read it, so
415    /// prefer the pre-1.12 form for files h5py will open.
416    ///
417    /// ```no_run
418    /// # use rust_hdf5::H5File;
419    /// let file = H5File::create("stdrefs.h5").unwrap();
420    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
421    /// let refs = file.new_dataset::<u64>()
422    ///     .std_object_references()
423    ///     .shape([1])
424    ///     .create("refs")
425    ///     .unwrap();
426    /// refs.write_object_references(&["/target"]).unwrap();
427    /// file.close().unwrap();
428    /// ```
429    #[must_use]
430    pub fn std_object_references(mut self) -> Self {
431        self.references = Some(ReferenceElement::Revised);
432        self
433    }
434
435    /// Store revised region references — `H5R_DATASET_REGION2`, written into
436    /// the same `H5T_STD_REF` datatype
437    /// [`std_object_references`](Self::std_object_references) makes.
438    ///
439    /// The elements are written with
440    /// [`write_std_region_references`](H5Dataset::write_std_region_references).
441    /// What distinguishes them from the pre-1.12
442    /// [`region_references`](Self::region_references) is the element, not the
443    /// datatype: a 1.12 element names its own kind, so one dataset of this type
444    /// may hold object, region and attribute references together. h5py 3.15
445    /// cannot read any of them, so prefer the pre-1.12 form for files h5py will
446    /// open.
447    ///
448    /// ```no_run
449    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, LibverBound, Selection};
450    /// let file = H5File::options().libver(LibverBound::V112).create("stdregions.h5").unwrap();
451    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
452    /// let refs = file.new_dataset::<u64>()
453    ///     .std_region_references()
454    ///     .shape([1])
455    ///     .create("refs")
456    ///     .unwrap();
457    /// let rows = Selection::Hyperslab {
458    ///     rank: 1,
459    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
460    /// };
461    /// refs.write_std_region_references(&[("/target", rows)]).unwrap();
462    /// file.close().unwrap();
463    /// ```
464    #[must_use]
465    pub fn std_region_references(self) -> Self {
466        self.std_object_references()
467    }
468
469    /// Store attribute references — `H5R_ATTR`, the one reference kind with no
470    /// pre-1.12 form, in the same `H5T_STD_REF` datatype
471    /// [`std_object_references`](Self::std_object_references) makes.
472    ///
473    /// The elements are written with
474    /// [`write_attribute_references`](H5Dataset::write_attribute_references) and
475    /// name an object and one of its attributes. h5py 3.15 cannot read them.
476    ///
477    /// ```no_run
478    /// # use rust_hdf5::{H5File, LibverBound};
479    /// let file = H5File::options().libver(LibverBound::V112).create("attrrefs.h5").unwrap();
480    /// let target = file.new_dataset::<i32>().shape([4]).create("target").unwrap();
481    /// target.new_attr::<i32>().shape([3]).create("note").unwrap()
482    ///     .write_array(&[7i32, 8, 9]).unwrap();
483    /// let refs = file.new_dataset::<u64>()
484    ///     .attribute_references()
485    ///     .shape([1])
486    ///     .create("refs")
487    ///     .unwrap();
488    /// refs.write_attribute_references(&[("/target", "note")]).unwrap();
489    /// file.close().unwrap();
490    /// ```
491    #[must_use]
492    pub fn attribute_references(self) -> Self {
493        self.std_object_references()
494    }
495
496    /// Store dataset region references — h5py's `h5py.regionref_dtype`.
497    ///
498    /// The elements are written with
499    /// [`write_region_references`](H5Dataset::write_region_references) and name
500    /// a dataset plus a selection over it. The element is a global-heap id, so
501    /// its width follows the file's address size and the datatype is resolved
502    /// at [`create`](Self::create) rather than here; it overrides both `T` and
503    /// any [`datatype`](Self::datatype) call.
504    ///
505    /// ```no_run
506    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, Selection};
507    /// let file = H5File::create("regions.h5").unwrap();
508    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
509    /// let refs = file.new_dataset::<u64>()
510    ///     .region_references()
511    ///     .shape([1])
512    ///     .create("refs")
513    ///     .unwrap();
514    /// let rows = Selection::Hyperslab {
515    ///     rank: 1,
516    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
517    /// };
518    /// refs.write_region_references(&[("/target", rows)]).unwrap();
519    /// file.close().unwrap();
520    /// ```
521    #[must_use]
522    pub fn region_references(mut self) -> Self {
523        self.references = Some(ReferenceElement::Region);
524        self
525    }
526
527    /// Keep the raw data in files outside this one — `H5Pset_external`,
528    /// h5py's `external=[(name, offset, size)]`.
529    ///
530    /// Each entry is a file name, the byte offset in it where that entry's
531    /// region starts, and how many bytes of the dataset it holds; the entries
532    /// concatenate, in order, into the dataset's bytes and together must cover
533    /// them. A relative name is resolved against `HDF5_EXTFILE_PREFIX` the way
534    /// libhdf5 resolves it, so the same name reads back through this crate and
535    /// through h5py. The storage is contiguous by definition, which rules out
536    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
537    /// [`null`](Self::null) and either reference kind.
538    ///
539    /// The named files are created on first write and never truncated, so
540    /// several datasets may own disjoint ranges of one file.
541    ///
542    /// The last entry may take the unlimited size
543    /// [`external_file_list::UNLIMITED`](crate::format::messages::external_file_list::UNLIMITED)
544    /// (`H5O_EFL_UNLIMITED`), which makes it absorb the whole rest of the
545    /// dataset however far it grows. A dataset whose
546    /// [`max_shape`](Self::max_shape) is unlimited must have one, since no
547    /// finite reservation could cover it, and only the first dimension may be
548    /// extendible — both `H5D__efl_construct`'s rules.
549    ///
550    /// ```no_run
551    /// # use rust_hdf5::H5File;
552    /// let file = H5File::create("ext.h5").unwrap();
553    /// let ds = file.new_dataset::<i32>()
554    ///     .shape([16])
555    ///     .external(&[("ext.raw", 0, 64)])
556    ///     .create("data")
557    ///     .unwrap();
558    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
559    /// ```
560    #[must_use]
561    pub fn external(mut self, files: &[(&str, u64, u64)]) -> Self {
562        self.external = Some(
563            files
564                .iter()
565                .map(|&(name, offset, size)| (name.to_string(), offset, size))
566                .collect(),
567        );
568        self
569    }
570
571    /// `H5Pset_efile_prefix` on the dapl `H5Dcreate2` takes — the directory
572    /// the raw data files named by [`external`](Self::external) are created
573    /// under, and looked for under on every later write through this handle.
574    ///
575    /// `H5D__create` builds `dset->shared->extfile_prefix` from the dapl
576    /// (H5Dint.c:1318) and `H5D__efl_write` joins each slot name against it
577    /// with the same single-path `H5_combine_path` the read side uses
578    /// (H5Defl.c:429-431) — so this decides where the bytes land, and the
579    /// prefix a later reader names must agree for it to find them.
580    ///
581    /// Measured under libhdf5 1.14.6 and 2.0.0: writing through a dapl that
582    /// names a directory creates the raw data file there and nowhere else,
583    /// and a directory that does not exist fails the write outright rather
584    /// than being created.
585    ///
586    /// It shares [`DatasetAccess::efile_prefix`]'s rules, both being
587    /// `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX` shadows this outright
588    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
589    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
590    /// (:1098-1102), which leaves a stored name to resolve against the
591    /// process's current directory.
592    ///
593    /// Ignored by a dataset that names no external files, which has no slot
594    /// name to join.
595    #[must_use]
596    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
597        self.efile_prefix = Some(prefix.into());
598        self
599    }
600
601    /// Map part of this dataset onto part of a dataset in another file, making
602    /// it virtual — `H5Pset_virtual`, one `VirtualLayout[...] =
603    /// VirtualSource(...)` assignment in h5py.
604    ///
605    /// The arguments are `H5Pset_virtual`'s, in its order: which elements of
606    /// *this* dataset the mapping fills, the file and dataset the data comes
607    /// from, and which elements of that source dataset it comes from. Call it
608    /// once per mapping; they apply in the order given, which is the order
609    /// libhdf5 resolves overlapping ones in.
610    ///
611    /// The source file is named exactly as stored — resolved against
612    /// `HDF5_VDS_PREFIX`, or the virtual dataset's own directory, when the
613    /// file is read — and `"."` means this file. Nothing is opened or checked
614    /// here: a source that does not exist yet is legal, and reads of the
615    /// unmapped or unresolvable parts return the [`fill_value`](Self::fill_value).
616    ///
617    /// A virtual dataset stores nothing of its own, which rules out
618    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
619    /// [`null`](Self::null), [`external`](Self::external) and either reference
620    /// kind — and makes writing to it an error, since its elements belong to
621    /// the source datasets.
622    ///
623    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: such a
624    /// mapping grows with its source, and the dataset's extent in that
625    /// dimension is whatever the sources reachable when it is opened supply
626    /// (`H5D__virtual_set_extent_unlim`). Give it a
627    /// [`max_shape`](Self::max_shape) unlimited in the same dimension, as
628    /// libhdf5 requires of the dataspace behind one.
629    ///
630    /// A source name may carry libhdf5's `printf`-style substitutions: `%b`
631    /// is the block index and `%%` an escaped literal `%`. One such mapping
632    /// stands for the family of source datasets that fill the successive
633    /// blocks of an unlimited virtual selection, so it is legal only with an
634    /// unlimited virtual selection over a limited source selection, and the
635    /// dataset's extent stops at the first block whose source is missing.
636    ///
637    /// ```no_run
638    /// # use rust_hdf5::{H5File, Selection};
639    /// let file = H5File::create("vds.h5").unwrap();
640    /// let ds = file.new_dataset::<i32>()
641    ///     .shape([16])
642    ///     .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
643    ///     .create("vds")
644    ///     .unwrap();
645    /// ```
646    #[must_use]
647    pub fn virtual_mapping(
648        mut self,
649        virtual_selection: Selection,
650        source_file: &str,
651        source_dataset: &str,
652        source_selection: Selection,
653    ) -> Self {
654        self.virtual_mappings.push(VirtualMapping {
655            source_file_name: source_file.to_string(),
656            source_dset_name: source_dataset.to_string(),
657            source_selection,
658            virtual_selection,
659        });
660        self
661    }
662
663    /// Set a user-defined fill value for unwritten elements.
664    ///
665    /// Without this, datasets use the HDF5 default zero-fill. When set,
666    /// the value is written into the dataset's fill-value message
667    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
668    /// unwritten regions as this value rather than zero.
669    ///
670    /// ```no_run
671    /// # use rust_hdf5::H5File;
672    /// let file = H5File::create("fv.h5").unwrap();
673    /// let ds = file.new_dataset::<f32>()
674    ///     .shape(&[100])
675    ///     .fill_value(f32::NAN)
676    ///     .create("data")
677    ///     .unwrap();
678    /// ```
679    #[must_use]
680    pub fn fill_value(mut self, value: T) -> Self {
681        let es = T::element_size();
682        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
683        // well-defined byte representation; `element_size()` matches
684        // `size_of::<T>()`. The slice borrows `value` only for this call.
685        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
686        self.fill_value = Some(raw.to_vec());
687        self
688    }
689
690    /// Set when the fill value is written into allocated storage —
691    /// `H5Pset_fill_time`.
692    ///
693    /// Without this, a dataset gets [`FillTime::IfSet`]
694    /// (`H5D_CRT_FILL_TIME_DEF`), the default every dataset creation
695    /// property list carries. [`FillTime::Never`] applies to a dataset with
696    /// no fill value too: it only stops this writer's own eager tiling of
697    /// the value into newly allocated storage, not the default zero-fill
698    /// that storage already has, so its only observable effect is on a
699    /// dataset that also calls [`fill_value`](Self::fill_value).
700    ///
701    /// ```no_run
702    /// # use rust_hdf5::{FillTime, H5File};
703    /// let file = H5File::create("fv.h5").unwrap();
704    /// let ds = file.new_dataset::<f32>()
705    ///     .shape(&[100])
706    ///     .fill_value(f32::NAN)
707    ///     .fill_time(FillTime::Never)
708    ///     .create("data")
709    ///     .unwrap();
710    /// ```
711    #[must_use]
712    pub fn fill_time(mut self, time: FillTime) -> Self {
713        self.fill_time = Some(time);
714        self
715    }
716
717    /// Finalize and create the dataset with the given `name`.
718    ///
719    /// The name is the link name within the root group (e.g. `"data"` or
720    /// `"group1/data"` once nested groups are supported).
721    pub fn create(self, name: &str) -> Result<H5Dataset> {
722        // A committed type is resolved before the dataset exists and recorded
723        // after, here rather than in each storage path: every path reaches
724        // this one return, so a dataset can never be built on a committed
725        // type and then fail to say so — which would silently write the type
726        // out in full instead of pointing at the object.
727        let committed = self.resolve_committed_type()?;
728        let file_inner = clone_inner(&self.file_inner);
729        let ds = self.create_object(name, committed.as_ref().map(|(_, dt)| dt.clone()))?;
730        if let (Some((share, _)), DatasetInfo::Writer { index, .. }) = (committed, &ds.info) {
731            let inner = borrow_inner(&file_inner);
732            if let H5FileInner::Writer(writer) = &*inner {
733                writer.share_committed_type(*index, share);
734            }
735        }
736        Ok(ds)
737    }
738
739    /// The committed datatype this dataset is built on, with the type it
740    /// holds; `None` when [`committed_type`](Self::committed_type) was not
741    /// called.
742    fn resolve_committed_type(&self) -> Result<Option<(usize, DatatypeMessage)>> {
743        let Some(path) = self.committed_type.as_deref() else {
744            return Ok(None);
745        };
746        if self.references.is_some() {
747            // Both name the stored type and they cannot both be it: the
748            // pointer would say the elements are the committed type while the
749            // reference writers write addresses. True of either reference
750            // kind — an object reference is an address, a region reference is
751            // a global-heap address plus a serialized selection.
752            return Err(Hdf5Error::InvalidState(
753                "a dataset cannot be built on a committed datatype and hold references".into(),
754            ));
755        }
756        let inner = borrow_inner(&self.file_inner);
757        match &*inner {
758            H5FileInner::Writer(writer) => Ok(Some(writer.committed_datatype_for_share(path)?)),
759            H5FileInner::Reader(_) => Err(Hdf5Error::InvalidState(
760                "cannot create a dataset in read mode".into(),
761            )),
762            H5FileInner::Closed => Err(Hdf5Error::InvalidState("file is closed".into())),
763        }
764    }
765
766    /// Everything [`create`](Self::create) does apart from recording the
767    /// committed-type share; `committed` is the type that object holds.
768    fn create_object(self, name: &str, committed: Option<DatatypeMessage>) -> Result<H5Dataset> {
769        // Build the full name: if created within a group, prefix with group path
770        let full_name = if let Some(ref gp) = self.group_path {
771            if gp == "/" {
772                name.to_string()
773            } else {
774                let trimmed = gp.trim_start_matches('/');
775                format!("{}/{}", trimmed, name)
776            }
777        } else {
778            name.to_string()
779        };
780
781        let datatype = if let Some(kind) = self.references {
782            // The element is measured in file addresses, and only the writer
783            // knows how wide one is for this file.
784            let inner = borrow_inner(&self.file_inner);
785            match &*inner {
786                H5FileInner::Writer(writer) => kind.datatype(writer.ctx()),
787                H5FileInner::Reader(_) => {
788                    return Err(Hdf5Error::InvalidState(
789                        "cannot create a dataset in read mode".into(),
790                    ))
791                }
792                H5FileInner::Closed => {
793                    return Err(Hdf5Error::InvalidState("file is closed".into()))
794                }
795            }
796        } else if let Some(dt) = committed {
797            // The object header holds the type; the dataset stores a pointer
798            // to it, but every size and payload check still needs the type
799            // itself.
800            dt
801        } else {
802            self.datatype_override.clone().unwrap_or_else(T::hdf5_type)
803        };
804        // Size one element from the on-disk datatype, not the carrier `T`. For
805        // the default path this equals `T::element_size()`; when a `datatype()`
806        // override is set (N-bit, or a runtime `CompoundType`), the stored type
807        // — not `T` — defines the element width, so the dataspace, the raw
808        // allocation, and the `write_raw` length check all agree with the bytes
809        // libhdf5/h5py will read.
810        let element_size = datatype.element_size() as usize;
811        // `fill_value` took the host image of a `T`; the fill-value message
812        // holds one element in the dataset's own datatype, so it is converted
813        // here — the order is only known once the override is resolved, and
814        // the builder's calls can arrive in either order.
815        let fill_value = match self.fill_value.as_deref() {
816            Some(bytes) => Some(to_stored_byte_order(bytes, &datatype, element_size)?.into_owned()),
817            None => None,
818        };
819
820        let wants_filter =
821            self.custom_pipeline.is_some() || self.shuffle || self.deflate_level.is_some();
822
823        // External storage *is* contiguous storage: the layout message says
824        // contiguous with an undefined address, and the External File List
825        // beside it says where the bytes really are. Every other storage class
826        // names bytes of its own, so none of them can also name these.
827        if self.external.is_some() {
828            if self.chunk_dims.is_some() || wants_filter || self.is_compact || self.is_null {
829                return Err(Hdf5Error::InvalidState(
830                    "a dataset whose raw data lives in external files is contiguous, so it \
831                     cannot also be chunked, filtered, compact or NULL"
832                        .into(),
833                ));
834            }
835            if self.references.is_some() {
836                return Err(Hdf5Error::InvalidState(
837                    "object and region references are stamped into the dataset's own \
838                     contiguous block, which a dataset stored in external files has none of"
839                        .into(),
840                ));
841            }
842        }
843
844        // A virtual dataset stores nothing of its own — its elements are read
845        // out of the datasets its mappings name — so it can be none of the
846        // storage classes that do, and there is no block for a reference
847        // writer to stamp into either.
848        if !self.virtual_mappings.is_empty() {
849            if self.chunk_dims.is_some()
850                || wants_filter
851                || self.is_compact
852                || self.is_null
853                || self.external.is_some()
854            {
855                return Err(Hdf5Error::InvalidState(
856                    "a virtual dataset's elements live in the datasets its mappings name, \
857                     so it cannot also be chunked, filtered, compact, NULL or stored in \
858                     external files"
859                        .into(),
860                ));
861            }
862            if self.references.is_some() {
863                return Err(Hdf5Error::InvalidState(
864                    "object and region references are stamped into the dataset's own \
865                     contiguous block, which a virtual dataset has none of"
866                        .into(),
867                ));
868            }
869        }
870
871        if self.is_null {
872            // A NULL dataspace holds no elements at all: no chunk grid to
873            // scatter into, no raw image to put in an object header, no fill
874            // value to apply to unwritten elements (there are none), matching
875            // upstream's rejection of these combinations (`H5Dchunk.c`'s
876            // chunked-layout dataspace check).
877            if self.chunk_dims.is_some() || wants_filter || self.is_compact {
878                return Err(Hdf5Error::InvalidState(
879                    "a NULL dataspace dataset cannot be chunked, filtered or compact".into(),
880                ));
881            }
882            if fill_value.is_some() {
883                return Err(Hdf5Error::InvalidState(
884                    "a NULL dataspace dataset cannot have a fill value".into(),
885                ));
886            }
887            if self.fill_time.is_some() {
888                return Err(Hdf5Error::InvalidState(
889                    "a NULL dataspace dataset cannot have a fill time".into(),
890                ));
891            }
892
893            let index = {
894                let inner = borrow_inner(&self.file_inner);
895                match &*inner {
896                    H5FileInner::Writer(writer) => {
897                        let idx = writer.create_null_dataset(&full_name, datatype)?;
898                        if let Some(ref gp) = self.group_path {
899                            if gp != "/" {
900                                writer.assign_dataset_to_group(gp, idx)?;
901                            }
902                        }
903                        idx
904                    }
905                    H5FileInner::Reader(_) => {
906                        return Err(Hdf5Error::InvalidState(
907                            "cannot create a dataset in read mode".into(),
908                        ));
909                    }
910                    H5FileInner::Closed => {
911                        return Err(Hdf5Error::InvalidState("file is closed".into()));
912                    }
913                }
914            };
915
916            return Ok(H5Dataset {
917                file_inner: clone_inner(&self.file_inner),
918                info: DatasetInfo::Writer {
919                    index,
920                    shape: Vec::new(),
921                    element_size,
922                    chunk_index: None,
923                    is_null: true,
924                },
925                _open: None,
926            });
927        }
928
929        let shape = self.shape.ok_or_else(|| {
930            Hdf5Error::InvalidState("shape must be set before calling create()".into())
931        })?;
932        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
933
934        if self.is_compact {
935            // The raw data is the layout message, so there is no chunk grid to
936            // filter and no room to grow into: `H5D__compact_construct` refuses
937            // a max dimension above the current one, and `H5Pset_layout` and
938            // `H5Pset_chunk` overwrite each other rather than combining.
939            if self.chunk_dims.is_some() || wants_filter {
940                return Err(Hdf5Error::InvalidState(
941                    "a compact dataset stores its data in the object header, so it \
942                     cannot be chunked or filtered"
943                        .into(),
944                ));
945            }
946            if self
947                .max_shape
948                .as_ref()
949                .is_some_and(|max| max.iter().zip(&shape).any(|(m, &d)| *m != Some(d)))
950            {
951                return Err(Hdf5Error::InvalidState(
952                    "a compact dataset cannot be extendible: its maximum shape must \
953                     equal its shape"
954                        .into(),
955                ));
956            }
957
958            let index = {
959                let inner = borrow_inner(&self.file_inner);
960                match &*inner {
961                    H5FileInner::Writer(writer) => {
962                        let idx = writer.create_compact_dataset(&full_name, datatype, &dims_u64)?;
963                        // Set before the fill value: NEVER must be in place
964                        // before that call decides whether to eager-tile it.
965                        if let Some(time) = self.fill_time {
966                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
967                        }
968                        if let Some(ref fv) = fill_value {
969                            writer.set_dataset_fill_value(idx, fv.clone())?;
970                        }
971                        idx
972                    }
973                    H5FileInner::Reader(_) => {
974                        return Err(Hdf5Error::InvalidState(
975                            "cannot create a dataset in read mode".into(),
976                        ));
977                    }
978                    H5FileInner::Closed => {
979                        return Err(Hdf5Error::InvalidState("file is closed".into()));
980                    }
981                }
982            };
983
984            return Ok(H5Dataset {
985                file_inner: clone_inner(&self.file_inner),
986                info: DatasetInfo::Writer {
987                    index,
988                    shape,
989                    element_size,
990                    chunk_index: None,
991                    is_null: false,
992                },
993                _open: None,
994            });
995        }
996
997        if !self.virtual_mappings.is_empty() {
998            let index = {
999                let inner = borrow_inner(&self.file_inner);
1000                match &*inner {
1001                    H5FileInner::Writer(writer) => {
1002                        let idx = writer.create_virtual_dataset(
1003                            &full_name,
1004                            datatype,
1005                            &dims_u64,
1006                            self.max_shape
1007                                .as_ref()
1008                                .map(|max| {
1009                                    max.iter()
1010                                        .map(|m| m.map_or(u64::MAX, |v| v as u64))
1011                                        .collect::<Vec<u64>>()
1012                                })
1013                                .as_deref(),
1014                            &self.virtual_mappings,
1015                        )?;
1016                        // The fill value is what a read of an unmapped — or
1017                        // unresolvable — element returns, so it is the one
1018                        // dataset property a virtual dataset carries about its
1019                        // own elements. Nothing is tiled into storage: it has
1020                        // none.
1021                        // Set before the fill value: NEVER must be in place
1022                        // before that call decides whether to eager-tile it.
1023                        if let Some(time) = self.fill_time {
1024                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1025                        }
1026                        if let Some(ref fv) = fill_value {
1027                            writer.set_dataset_fill_value(idx, fv.clone())?;
1028                        }
1029                        idx
1030                    }
1031                    H5FileInner::Reader(_) => {
1032                        return Err(Hdf5Error::InvalidState(
1033                            "cannot create a dataset in read mode".into(),
1034                        ));
1035                    }
1036                    H5FileInner::Closed => {
1037                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1038                    }
1039                }
1040            };
1041
1042            return Ok(H5Dataset {
1043                file_inner: clone_inner(&self.file_inner),
1044                info: DatasetInfo::Writer {
1045                    index,
1046                    shape,
1047                    element_size,
1048                    chunk_index: None,
1049                    is_null: false,
1050                },
1051                _open: None,
1052            });
1053        }
1054
1055        // A filter pipeline requires chunked storage. When a filter is
1056        // requested without explicit chunk dimensions, store the whole
1057        // dataset as a single chunk instead of silently dropping the filter
1058        // on the contiguous path. (This is one whole-dataset chunk, not
1059        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
1060        // for large datasets.)
1061        let auto_chunk: Option<Vec<usize>> =
1062            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
1063                Some(shape.iter().map(|&d| d.max(1)).collect())
1064            } else {
1065                None
1066            };
1067
1068        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
1069            // Chunked dataset
1070            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
1071            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
1072                max.iter()
1073                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
1074                    .collect()
1075            } else {
1076                // Default: max = current
1077                dims_u64.clone()
1078            };
1079
1080            // The file's format settles the question before the shape gets a
1081            // say. `H5D__chunk_set_info` reaches the index-selection block
1082            // only once the data layout message is at version 4 (H5Dchunk.c:936)
1083            // — which the file's library-version bound decides, not the
1084            // dataspace — and below it the version-3 message carries a
1085            // version-1 B-tree and nothing else. The writer owns that reading
1086            // of `H5O_layout_ver_bounds`; the chunk's byte count is the one
1087            // input from here, a chunk over 4 GiB being the one thing that
1088            // forces the newer message whatever the bound says.
1089            let chunk_bytes = chunk_u64.iter().product::<u64>() * element_size as u64;
1090            let v110_indexing = match &*borrow_inner(&self.file_inner) {
1091                H5FileInner::Writer(writer) => writer.uses_v110_chunk_indexing(chunk_bytes),
1092                // Neither can create a dataset at all; the creator below
1093                // reports which of the two it is.
1094                _ => true,
1095            };
1096
1097            // Inside the block libhdf5 selects the chunk index from the
1098            // dataspace and the creation properties, in this order
1099            // (`H5D__chunk_set_info`, H5Dchunk.c:955): a v2 B-tree for two or
1100            // more unlimited dimensions, an extensible array for exactly one;
1101            // for a fixed shape, the single-chunk index takes priority —
1102            // unconditional of filter or allocation time — whenever the shape
1103            // is exactly one whole chunk, ahead of the implicit index (no
1104            // filter, and early allocation, which is what puts every chunk at
1105            // a computable address) and the fixed array (everything else).
1106            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
1107            let one_chunk = chunk_u64 == dims_u64 && max_u64 == dims_u64;
1108            let kind = if !v110_indexing {
1109                ChunkIndexKind::BtreeV1
1110            } else if n_unlimited >= 2 {
1111                ChunkIndexKind::BtreeV2
1112            } else if n_unlimited == 1 {
1113                ChunkIndexKind::ExtensibleArray
1114            } else if one_chunk {
1115                ChunkIndexKind::SingleChunk
1116            } else if self.early_allocation && !wants_filter {
1117                ChunkIndexKind::Implicit
1118            } else {
1119                ChunkIndexKind::FixedArray
1120            };
1121
1122            let index = {
1123                let inner = borrow_inner(&self.file_inner);
1124                match &*inner {
1125                    H5FileInner::Writer(writer) => {
1126                        // The requested filter pipeline, if any. Every index
1127                        // builds it from the same options, so one owner
1128                        // resolves it: a second construction site is what let
1129                        // a request naming no compressor — shuffle on its own
1130                        // — fall through to unfiltered storage.
1131                        let explicit_pipeline = || {
1132                            use crate::format::messages::filter::FilterPipeline;
1133                            if let Some(p) = self.custom_pipeline.clone() {
1134                                return p;
1135                            }
1136                            // Shuffle records the width of the element it
1137                            // permutes, which is the stored one — a `datatype`
1138                            // override moves that away from `T`.
1139                            let es = element_size as u32;
1140                            match (self.shuffle, self.deflate_level) {
1141                                (true, Some(level)) => FilterPipeline::shuffle_deflate(es, level),
1142                                (true, None) => FilterPipeline::shuffle(es),
1143                                // deflate_level (checked by wants_filter).
1144                                (false, level) => FilterPipeline::deflate(level.unwrap()),
1145                            }
1146                        };
1147                        let idx = if kind == ChunkIndexKind::BtreeV1 {
1148                            // The classic index, which takes the pipeline the
1149                            // same way the others do — and is refused with it
1150                            // in a classic file, whose filter pipeline
1151                            // message is a version this crate does not write.
1152                            let pipeline = wants_filter.then(explicit_pipeline);
1153                            writer.create_btree_v1_dataset(
1154                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1155                            )?
1156                        } else if kind == ChunkIndexKind::BtreeV2 {
1157                            // Two or more unlimited dimensions: a v2 B-tree,
1158                            // whose records carry the stored size and filter
1159                            // mask when the dataset is compressed (libhdf5
1160                            // H5D_BT2_FILT).
1161                            if wants_filter {
1162                                writer.create_btree_v2_dataset_with_pipeline(
1163                                    &full_name,
1164                                    datatype,
1165                                    &dims_u64,
1166                                    &max_u64,
1167                                    &chunk_u64,
1168                                    explicit_pipeline(),
1169                                )?
1170                            } else {
1171                                writer.create_btree_v2_dataset(
1172                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1173                                )?
1174                            }
1175                        } else if kind == ChunkIndexKind::Implicit {
1176                            // No index structure at all: every chunk of the
1177                            // grid is allocated at create in one run, so
1178                            // there is no pipeline arm — a filter is what
1179                            // makes chunks different sizes, and this index
1180                            // has no room to say so.
1181                            writer.create_implicit_dataset(
1182                                &full_name, datatype, &dims_u64, &chunk_u64,
1183                            )?
1184                        } else if kind == ChunkIndexKind::SingleChunk {
1185                            // A fixed shape covered by exactly one chunk:
1186                            // libhdf5 picks this index ahead of Implicit and
1187                            // Fixed Array regardless of filter or allocation
1188                            // time. Filtered or not, it takes the same
1189                            // explicit pipeline the other indexes do; a
1190                            // filtered chunk's stored size isn't known ahead
1191                            // of its first write, so early allocation only
1192                            // ever applies to the unfiltered form.
1193                            if wants_filter {
1194                                writer.create_single_chunk_dataset_with_pipeline(
1195                                    &full_name,
1196                                    datatype,
1197                                    &dims_u64,
1198                                    &chunk_u64,
1199                                    explicit_pipeline(),
1200                                )?
1201                            } else {
1202                                writer.create_single_chunk_dataset(
1203                                    &full_name,
1204                                    datatype,
1205                                    &dims_u64,
1206                                    &chunk_u64,
1207                                    self.early_allocation,
1208                                )?
1209                            }
1210                        } else if kind == ChunkIndexKind::FixedArray {
1211                            // A chunked dataset with no unlimited dimension
1212                            // must use the fixed-array index — libhdf5
1213                            // rejects an extensible-array index here. A
1214                            // compressed fixed-shape dataset uses a *filtered*
1215                            // fixed array (FA client id 1). The maximum shape
1216                            // sizes the array, so a finite max above the
1217                            // current shape stays growable.
1218                            let pipeline = wants_filter.then(explicit_pipeline);
1219                            writer.create_fixed_array_dataset_with_max(
1220                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1221                            )?
1222                        } else if wants_filter {
1223                            // The extensible-array index takes the pipeline
1224                            // the same way, so it goes through the one owner
1225                            // too.
1226                            writer.create_chunked_dataset_with_pipeline(
1227                                &full_name,
1228                                datatype,
1229                                &dims_u64,
1230                                &max_u64,
1231                                &chunk_u64,
1232                                explicit_pipeline(),
1233                            )?
1234                        } else {
1235                            writer.create_chunked_dataset(
1236                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1237                            )?
1238                        };
1239                        // Set before the fill value: NEVER must be in place
1240                        // before that call decides whether to eager-tile it.
1241                        if let Some(time) = self.fill_time {
1242                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1243                        }
1244                        if let Some(ref fv) = fill_value {
1245                            writer.set_dataset_fill_value(idx, fv.clone())?;
1246                        }
1247                        idx
1248                    }
1249                    H5FileInner::Reader(_) => {
1250                        return Err(Hdf5Error::InvalidState(
1251                            "cannot create a dataset in read mode".into(),
1252                        ));
1253                    }
1254                    H5FileInner::Closed => {
1255                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1256                    }
1257                }
1258            };
1259
1260            Ok(H5Dataset {
1261                file_inner: clone_inner(&self.file_inner),
1262                info: DatasetInfo::Writer {
1263                    index,
1264                    shape,
1265                    element_size,
1266                    chunk_index: Some(kind),
1267                    is_null: false,
1268                },
1269                _open: None,
1270            })
1271        } else {
1272            // Contiguous dataset (original path)
1273            let efile_access = match self.efile_prefix.as_deref() {
1274                Some(p) => DatasetAccess::new().efile_prefix(p),
1275                None => DatasetAccess::new(),
1276            };
1277            let (index, open) = {
1278                let inner = borrow_inner(&self.file_inner);
1279                match &*inner {
1280                    H5FileInner::Writer(writer) => {
1281                        let idx = match self.external.as_deref() {
1282                            Some(files) => {
1283                                let slots: Vec<(&str, u64, u64)> = files
1284                                    .iter()
1285                                    .map(|(name, offset, size)| (name.as_str(), *offset, *size))
1286                                    .collect();
1287                                writer.create_external_dataset(
1288                                    &full_name,
1289                                    datatype,
1290                                    &dims_u64,
1291                                    self.max_shape
1292                                        .as_ref()
1293                                        .map(|max| {
1294                                            max.iter()
1295                                                .map(|m| m.map_or(u64::MAX, |v| v as u64))
1296                                                .collect::<Vec<u64>>()
1297                                        })
1298                                        .as_deref(),
1299                                    &slots,
1300                                )?
1301                            }
1302                            None => writer.create_dataset(&full_name, datatype, &dims_u64)?,
1303                        };
1304                        // Before anything that can write raw bytes: the
1305                        // prefix an external dataset's slot names are joined
1306                        // against is settled by the create, as
1307                        // `H5D__build_file_prefix` settles it for
1308                        // `H5D__create` (H5Dint.c:1318).
1309                        let open = writer.bind_efile_prefix(idx, &efile_access)?;
1310                        // Set before the fill value: NEVER must be in place
1311                        // before that call decides whether to eager-tile it.
1312                        if let Some(time) = self.fill_time {
1313                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1314                        }
1315                        if let Some(ref fv) = fill_value {
1316                            writer.set_dataset_fill_value(idx, fv.clone())?;
1317                        }
1318                        (idx, open)
1319                    }
1320                    H5FileInner::Reader(_) => {
1321                        return Err(Hdf5Error::InvalidState(
1322                            "cannot create a dataset in read mode".into(),
1323                        ));
1324                    }
1325                    H5FileInner::Closed => {
1326                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1327                    }
1328                }
1329            };
1330
1331            Ok(H5Dataset {
1332                file_inner: clone_inner(&self.file_inner),
1333                info: DatasetInfo::Writer {
1334                    index,
1335                    shape,
1336                    element_size,
1337                    chunk_index: None,
1338                    is_null: false,
1339                },
1340                _open: open,
1341            })
1342        }
1343    }
1344}
1345
1346// ---------------------------------------------------------------------------
1347// DatasetInfo
1348// ---------------------------------------------------------------------------
1349
1350/// Internal metadata about a dataset handle.
1351enum DatasetInfo {
1352    /// A dataset created via `new_dataset().create()` in write mode.
1353    Writer {
1354        /// Index into the writer's dataset list.
1355        index: usize,
1356        /// Shape (current dimensions).
1357        shape: Vec<usize>,
1358        /// Size of one element in bytes.
1359        element_size: usize,
1360        /// Which chunk index this dataset uses, `None` when its storage is
1361        /// not chunked. One field rather than a flag per index: a dataset
1362        /// has exactly one chunk index, and the flags could spell
1363        /// combinations ("not chunked, but indexed by a v2 B-tree") that no
1364        /// dataset has — which is what a write path reading only some of
1365        /// them turns into a write to the wrong index.
1366        chunk_index: Option<ChunkIndexKind>,
1367        /// Whether this is a NULL dataspace (no elements at all — distinct
1368        /// from a scalar, which holds exactly one). Always `false` when
1369        /// `chunk_index` is `Some`: a NULL dataspace can never be chunked.
1370        is_null: bool,
1371    },
1372    /// A dataset opened by name in read mode.
1373    Reader {
1374        /// The link name of the dataset.
1375        name: String,
1376        /// Shape (current dimensions).
1377        shape: Vec<usize>,
1378        /// Size of one element in bytes.
1379        element_size: usize,
1380    },
1381}
1382
1383// ---------------------------------------------------------------------------
1384// H5Dataset
1385// ---------------------------------------------------------------------------
1386
1387/// A handle to an HDF5 dataset, supporting typed read and write operations.
1388///
1389/// The dataset holds a shared reference to the file's I/O backend, so it
1390/// remains valid even if the originating [`H5File`](crate::file::H5File) is
1391/// moved or dropped (they share ownership via `Rc`).
1392pub struct H5Dataset {
1393    file_inner: SharedInner,
1394    info: DatasetInfo,
1395    /// Keeps this dataset's *open* alive for as long as the handle is, so
1396    /// the reader or the writer can tell whether a later open of the same
1397    /// name joins this one or starts fresh — libhdf5's `H5FO_opened`
1398    /// shared-info count (H5Dint.c:1496-1500). `None` for a dataset with no
1399    /// per-open answer to hold: in write mode, one whose raw data is in this
1400    /// file rather than in the files an external file list names.
1401    ///
1402    /// Held, never read: its whole job is to keep the reader's `Weak` on it
1403    /// upgradable until this handle goes away.
1404    _open: Option<crate::io::reader::DatasetOpenToken>,
1405}
1406
1407impl Drop for H5Dataset {
1408    /// Closing a virtual dataset's last handle closes the source files that
1409    /// open was holding, which is what `H5D__virtual_reset_layout` does at
1410    /// the last `H5Dclose` (H5Dvirtual.c:709-710, closing each
1411    /// `source_dset->dset` at :955 and with it the file that dataset kept
1412    /// open). Nothing else in this crate can end a virtual open, so this is
1413    /// where the reader is told.
1414    ///
1415    /// The token is dropped *before* the reader is asked, so the reader's
1416    /// `Weak` already reads dead for the handle going away here. A write-mode
1417    /// handle's token belongs to the writer's own external file prefix, which
1418    /// has no source files to close, so it takes no lock either.
1419    fn drop(&mut self) {
1420        let Some(open) = self._open.take() else {
1421            return;
1422        };
1423        drop(open);
1424        if matches!(self.info, DatasetInfo::Writer { .. }) {
1425            return;
1426        }
1427        let Some(mut inner) = crate::file::try_borrow_inner_mut(&self.file_inner) else {
1428            return;
1429        };
1430        if let crate::file::H5FileInner::Reader(reader) = &mut *inner {
1431            reader.release_closed_virtual_sources();
1432        }
1433    }
1434}
1435
1436/// One chunk's bytes on the way to the file, and who filtered them.
1437///
1438/// This is what separates a normal chunk write from a direct one; everything
1439/// else about placing a chunk is identical, so the two share a single dispatch.
1440#[derive(Clone, Copy)]
1441enum ChunkBytes<'a> {
1442    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
1443    /// are stored.
1444    Unfiltered(&'a [u8]),
1445    /// Bytes already in their stored form, with `filter_mask` naming the
1446    /// filters that were skipped.
1447    Prefiltered { data: &'a [u8], filter_mask: u32 },
1448}
1449
1450/// The byte order this build reads and writes natively.
1451pub(crate) const HOST_BYTE_ORDER: ByteOrder = if cfg!(target_endian = "big") {
1452    ByteOrder::BigEndian
1453} else {
1454    ByteOrder::LittleEndian
1455};
1456
1457/// The byte order this build does not read or write natively.
1458pub(crate) const FOREIGN_BYTE_ORDER: ByteOrder = match HOST_BYTE_ORDER {
1459    ByteOrder::LittleEndian => ByteOrder::BigEndian,
1460    ByteOrder::BigEndian => ByteOrder::LittleEndian,
1461};
1462
1463/// What a typed access has to do with an element image of a given datatype.
1464#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1465enum ByteOrderAction {
1466    /// Stored order is the host's: the image is already the typed value.
1467    Keep,
1468    /// The whole element is one scalar in the foreign order: reverse it.
1469    SwapElements,
1470    /// A composite storing something in the foreign order.
1471    Refuse,
1472}
1473
1474/// Classify a datatype for a typed access of element width `width`.
1475///
1476/// The single owner of the rule; both directions ask it, so a type a read
1477/// converts is exactly a type a write converts.
1478///
1479/// A composite element cannot be swapped as a unit — its members have their
1480/// own orders and offsets — so one that touches the foreign order is refused
1481/// rather than silently passed through in the wrong order.
1482fn byte_order_action(datatype: &DatatypeMessage, width: usize) -> ByteOrderAction {
1483    match datatype.scalar_byte_order() {
1484        Some(order) if order == FOREIGN_BYTE_ORDER && width > 1 => ByteOrderAction::SwapElements,
1485        Some(_) => ByteOrderAction::Keep,
1486        None if datatype.contains_byte_order(FOREIGN_BYTE_ORDER) => ByteOrderAction::Refuse,
1487        None => ByteOrderAction::Keep,
1488    }
1489}
1490
1491/// Why the stored image of an element is not already the host image of a
1492/// value of width `width` — `None` when it is, and a copying read would only
1493/// be memcpy-ing bytes it does not touch.
1494///
1495/// The two ways a stored element can need work before it is a value are the
1496/// two conversions a copying read performs in place: a byte-order swap
1497/// ([`to_host_byte_order`]) and the n-bit/scale-offset unpacking
1498/// (`Hdf5Reader::apply_post_filter_conversion`). Asking one question of both
1499/// is what lets a zero-copy view refuse exactly the datatypes a copying read
1500/// would have had to rewrite.
1501#[cfg(feature = "mmap")]
1502pub(crate) fn stored_image_mismatch(
1503    datatype: &DatatypeMessage,
1504    width: usize,
1505) -> Option<&'static str> {
1506    match byte_order_action(datatype, width) {
1507        ByteOrderAction::SwapElements => return Some("they are stored in the foreign byte order"),
1508        ByteOrderAction::Refuse => {
1509            return Some("it is a composite storing members in the foreign byte order")
1510        }
1511        ByteOrderAction::Keep => {}
1512    }
1513    if crate::format::nbit_scaleoffset::datatype_needs_bit_conversion(datatype) {
1514        return Some("the significant bits do not fill the stored element");
1515    }
1516    None
1517}
1518
1519/// Put a raw element image into host byte order, in place, for a typed read.
1520///
1521/// Every path that reinterprets the on-disk image as `T` — `read_raw`,
1522/// `read_slice`, `read_raw_into`, `read_slice_into` and their SWMR
1523/// counterparts — passes through here. Reinterpretation only yields the
1524/// stored value when the stored order is the host's.
1525///
1526/// A refused datatype is one no reinterpretation can decode;
1527/// [`H5Dataset::read_raw_bytes`] hands over the image for the caller to
1528/// decode member by member.
1529///
1530/// `width` is the element size, already checked equal to `T::element_size()`.
1531pub(crate) fn to_host_byte_order(
1532    bytes: &mut [u8],
1533    datatype: &DatatypeMessage,
1534    width: usize,
1535) -> Result<()> {
1536    match byte_order_action(datatype, width) {
1537        ByteOrderAction::Keep => {}
1538        ByteOrderAction::SwapElements => {
1539            for elem in bytes.chunks_exact_mut(width) {
1540                elem.reverse();
1541            }
1542        }
1543        ByteOrderAction::Refuse => {
1544            return Err(Hdf5Error::TypeMismatch(format!(
1545                "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a \
1546                 typed read cannot reinterpret element by element; read_raw_bytes() returns \
1547                 the image to decode member by member"
1548            )))
1549        }
1550    }
1551    Ok(())
1552}
1553
1554/// The stored byte image of each variable-length sequence in a batch.
1555///
1556/// The vlen writers take `&[&[T]]` and store one global-heap object per
1557/// sequence, so each sequence needs the same host-image-to-stored-image step
1558/// [`to_stored_byte_order`] performs for a fixed-shape write — a `T` is
1559/// written from its host bytes, and `T::hdf5_type()` declares little-endian.
1560/// Borrows on a little-endian host, which is every machine that does not have
1561/// to swap.
1562pub(crate) fn vlen_sequence_images<'a, T: H5Type>(
1563    items: &'a [&'a [T]],
1564) -> Result<Vec<std::borrow::Cow<'a, [u8]>>> {
1565    let base = T::hdf5_type();
1566    items
1567        .iter()
1568        .map(|item| {
1569            // Safety: the same contract `write_raw` relies on — `T: Copy +
1570            // 'static` is a numeric primitive whose byte image is its value —
1571            // and the extent comes from the slice itself, so it cannot name
1572            // memory past it. The result borrows `items` and outlives nothing.
1573            let host = unsafe {
1574                std::slice::from_raw_parts(item.as_ptr() as *const u8, std::mem::size_of_val(*item))
1575            };
1576            to_stored_byte_order(host, &base, T::element_size())
1577        })
1578        .collect()
1579}
1580
1581/// Put a typed value's host-order image into the order the datatype declares.
1582///
1583/// The write-side counterpart of [`to_host_byte_order`], and the one place
1584/// every path that hands a `&[T]` to the file — `write_raw`, `write_slice`,
1585/// `append`, and the builder's fill value — turns those bytes into stored
1586/// bytes. A `T` is written from its host image, so a dataset declaring the
1587/// foreign order would otherwise hold host bytes under that declaration: a
1588/// file that is wrong by its own header.
1589///
1590/// Borrows when the declared order is the host's, which is every write that
1591/// does not set a [`datatype`](DatasetBuilder::datatype) override.
1592///
1593/// `width` is the element size, already checked equal to `T::element_size()`.
1594pub(crate) fn to_stored_byte_order<'a>(
1595    bytes: &'a [u8],
1596    datatype: &DatatypeMessage,
1597    width: usize,
1598) -> Result<std::borrow::Cow<'a, [u8]>> {
1599    match byte_order_action(datatype, width) {
1600        ByteOrderAction::Keep => Ok(std::borrow::Cow::Borrowed(bytes)),
1601        ByteOrderAction::SwapElements => {
1602            let mut owned = bytes.to_vec();
1603            for elem in owned.chunks_exact_mut(width) {
1604                elem.reverse();
1605            }
1606            Ok(std::borrow::Cow::Owned(owned))
1607        }
1608        ByteOrderAction::Refuse => Err(Hdf5Error::TypeMismatch(format!(
1609            "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a typed \
1610             write cannot lay out element by element; write_raw_bytes() takes the image the \
1611             caller encodes member by member"
1612        ))),
1613    }
1614}
1615
1616/// Strip a fixed-string element's padding, leaving the bytes that carry the
1617/// value.
1618///
1619/// The rule itself lives with the datatype message
1620/// ([`fixed_string_content`]); this adds the element index a reserved padding
1621/// rule needs to be reported against.
1622fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
1623    crate::format::messages::datatype::fixed_string_content(elem, padding).ok_or_else(|| {
1624        Hdf5Error::InvalidState(format!(
1625            "string {index} uses padding rule {padding}, which the format reserves"
1626        ))
1627    })
1628}
1629
1630/// Decode one string element's bytes under the datatype's character set.
1631///
1632/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
1633/// `index` names the element in the error otherwise.
1634fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
1635    if lossy {
1636        return Ok(String::from_utf8_lossy(bytes).into_owned());
1637    }
1638    match charset {
1639        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
1640        0 => match bytes.iter().position(|&b| b >= 0x80) {
1641            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
1642            Some(at) => Err(Hdf5Error::InvalidState(format!(
1643                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
1644                bytes[at]
1645            ))),
1646        },
1647        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
1648            Hdf5Error::InvalidState(format!(
1649                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
1650            ))
1651        }),
1652        other => Err(Hdf5Error::InvalidState(format!(
1653            "string {index} uses character set {other}, which the format reserves"
1654        ))),
1655    }
1656}
1657
1658/// A dataset's storage layout class (read mode only) — `H5Pget_layout`'s
1659/// four values.
1660///
1661/// Distinct from [`ChunkIndex`], which names the structure a `Chunked`
1662/// layout's index uses; this only says which of the four storage classes
1663/// the dataset was created with.
1664#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1665pub enum StorageLayout {
1666    /// Raw data stored inline in the object header.
1667    Compact,
1668    /// Raw data in a single contiguous block — or, when the dataset also
1669    /// carries an external file list, in one or more blocks of an outside
1670    /// file instead ([`H5Dataset::external_files`]); the layout itself
1671    /// still reports `Contiguous` either way.
1672    Contiguous,
1673    /// Raw data split into fixed-size chunks, each independently
1674    /// allocated. [`H5Dataset::chunk_dims`] gives the chunk shape,
1675    /// [`H5Dataset::chunk_index`] the index structure.
1676    Chunked,
1677    /// No raw data of its own: every element comes from another dataset,
1678    /// possibly in another file ([`H5Dataset::virtual_mappings`]).
1679    Virtual,
1680}
1681
1682/// The chunk index structure a chunked dataset uses on disk (read mode
1683/// only) — which of libhdf5's chunk-lookup structures the layout message
1684/// names.
1685///
1686/// `BtreeV1` belongs to the version-3 chunked layout message (the only
1687/// index a file whose superblock predates version 2 can carry); the other
1688/// five are what a version-4 message's index-type byte selects, per
1689/// `H5D__layout_set_latest_indexing` (H5Dlayout.c).
1690#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1691pub enum ChunkIndex {
1692    /// Version-1 B-tree — the classic chunk index, and the only one a
1693    /// version-3 chunked layout message can carry.
1694    BtreeV1,
1695    /// Version-2 B-tree — two or more unlimited dimensions.
1696    BtreeV2,
1697    /// A single index entry for a dataset whose one chunk covers the whole
1698    /// dataspace (`dims == max_dims == chunk_dims`).
1699    SingleChunk,
1700    /// No index structure at all: chunk addresses are computed
1701    /// arithmetically over a contiguous run (no filter, early allocation).
1702    Implicit,
1703    /// Fixed-size array — a fixed shape needing per-chunk bookkeeping.
1704    FixedArray,
1705    /// Extensible array — exactly one unlimited dimension.
1706    ExtensibleArray,
1707}
1708
1709/// A dataset's fill-value state (read mode only) — `H5Pfill_value_defined`'s
1710/// tri-state (`H5D_fill_value_t`).
1711#[derive(Debug, Clone, PartialEq, Eq)]
1712pub enum FillValue {
1713    /// No fill value has ever been set: unwritten elements read back
1714    /// zero-filled, and no fill-value message named an explicit value.
1715    Default,
1716    /// The fill value was explicitly disabled: unallocated storage is never
1717    /// fill-initialized.
1718    Undefined,
1719    /// An explicit fill value, one element wide.
1720    UserDefined(Vec<u8>),
1721}
1722
1723/// When a dataset's fill value is written into allocated storage —
1724/// `H5Pset_fill_time`/`H5Pget_fill_time`'s `H5D_fill_time_t`.
1725///
1726/// Distinct from [`FillValue`], which says *what* the fill value is; this
1727/// says *when* it is written. The two agree everywhere except a dataset with
1728/// no fill value of its own: there, `Alloc` writes the default fill (zeros)
1729/// at allocation and `IfSet` writes nothing into space that already reads as
1730/// zeros — indistinguishable on disk in the value itself, only in this byte.
1731#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1732pub enum FillTime {
1733    /// Fill at allocation regardless of whether a fill value was ever set.
1734    Alloc,
1735    /// Never write the fill value into allocated storage.
1736    Never,
1737    /// Fill at allocation only when a fill value was set — the default
1738    /// every dataset gets unless [`fill_time`](DatasetBuilder::fill_time)
1739    /// says otherwise.
1740    IfSet,
1741}
1742
1743impl FillTime {
1744    /// The on-disk `H5D_fill_time_t` byte this variant is — what the writer
1745    /// stores and the fill-value message's write-time field carries.
1746    fn wire_byte(self) -> u8 {
1747        match self {
1748            Self::Alloc => 0,
1749            Self::Never => 1,
1750            Self::IfSet => 2,
1751        }
1752    }
1753}
1754
1755/// When a dataset's raw-data storage is allocated —
1756/// `H5Pset_alloc_time`/`H5Pget_alloc_time`'s `H5D_alloc_time_t`, read back
1757/// from the same fill-value message [`FillTime`] is.
1758///
1759/// `H5P__set_layout` (H5Pdcpl.c) picks this from the dataset's storage
1760/// class (`H5D_ALLOC_TIME_DEFAULT` per layout — compact is `Early`,
1761/// chunked and virtual are `Incr`, contiguous is `Late`). The one override
1762/// this crate's builder offers is [`DatasetBuilder::early_allocation`],
1763/// which a chunked dataset reads back as `Early` where the writer took it
1764/// up: an implicit index, or a single unfiltered chunk.
1765/// [`H5Dataset::alloc_time`] reads back what the writer declared.
1766#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1767pub enum AllocTime {
1768    /// Space is allocated as soon as the dataset is created.
1769    Early,
1770    /// Space is allocated when data is first written.
1771    Late,
1772    /// Space is allocated incrementally, as chunks (or virtual source
1773    /// datasets) are written.
1774    Incr,
1775}
1776
1777/// Which mapped data an unlimited virtual dataset's extent covers —
1778/// libhdf5's `H5D_vds_view_t`, set with `H5Pset_virtual_view` and read back
1779/// with `H5Pget_virtual_view` (H5Pdapl.c:1067, :1102).
1780///
1781/// A *dataset access* property: it is never stored in the file, so it says
1782/// how *this* open reads a virtual dataset, not what its writer intended.
1783#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1784pub enum VirtualView {
1785    /// `H5D_VDS_LAST_AVAILABLE` — the extent reaches the end of the last
1786    /// mapped block that has a source, so a gap before it reads as the fill
1787    /// value. libhdf5's default (`H5D_ACS_VDS_VIEW_DEF`, H5Pdapl.c:62).
1788    #[default]
1789    LastAvailable,
1790    /// `H5D_VDS_FIRST_MISSING` — the extent stops where the first missing
1791    /// mapped block begins, so no unmapped block is inside it.
1792    ///
1793    /// Under this view libhdf5 ignores
1794    /// [`virtual_printf_gap`](DatasetAccess::virtual_printf_gap) entirely:
1795    /// `H5D__virtual_init` reads the gap property only for
1796    /// [`LastAvailable`](Self::LastAvailable) and forces it to 0 otherwise
1797    /// (H5Dvirtual.c:2182-2188).
1798    FirstMissing,
1799}
1800
1801/// The dataset *access* properties this crate models — libhdf5's
1802/// `H5P_DATASET_ACCESS` property list, as much of it as affects reading.
1803///
1804/// [`virtual_view`](Self::virtual_view) and
1805/// [`virtual_printf_gap`](Self::virtual_printf_gap) govern how a virtual
1806/// dataset's extent is resolved when it is opened
1807/// (`H5D__virtual_set_extent_unlim`, H5Dvirtual.c:1386);
1808/// [`virtual_prefix`](Self::virtual_prefix) and
1809/// [`efile_prefix`](Self::efile_prefix) say where the *other files* a
1810/// dataset's data lives in are looked for. None of them is stored in the
1811/// file: opening a dataset without naming them reads it exactly as
1812/// libhdf5's default dapl does.
1813///
1814/// Pass one to [`H5File::dataset_with`](crate::H5File::dataset_with).
1815///
1816/// ```no_run
1817/// use rust_hdf5::{DatasetAccess, H5File, VirtualView};
1818///
1819/// let file = H5File::open("vds.h5").unwrap();
1820/// let access = DatasetAccess::new()
1821///     .virtual_view(VirtualView::LastAvailable)
1822///     .virtual_printf_gap(2);
1823/// let ds = file.dataset_with("vds", access).unwrap();
1824/// ```
1825#[derive(Debug, Clone, PartialEq, Eq, Default)]
1826pub struct DatasetAccess {
1827    view: VirtualView,
1828    printf_gap: u64,
1829    virtual_prefix: Option<String>,
1830    efile_prefix: Option<String>,
1831}
1832
1833impl DatasetAccess {
1834    /// A property list holding libhdf5's defaults —
1835    /// [`VirtualView::LastAvailable`] and a printf gap of 0, the values
1836    /// `H5D_ACS_VDS_VIEW_DEF` and `H5D_ACS_VDS_PRINTF_GAP_DEF` register
1837    /// (H5Pdapl.c:62, :67).
1838    pub fn new() -> Self {
1839        Self::default()
1840    }
1841
1842    /// `H5Pset_virtual_view` (H5Pdapl.c:1067). The two legal values are the
1843    /// two [`VirtualView`] variants, so the "not a valid bounds option"
1844    /// argument check that call makes has nothing to reject here.
1845    pub fn virtual_view(mut self, view: VirtualView) -> Self {
1846        self.view = view;
1847        self
1848    }
1849
1850    /// `H5Pset_virtual_printf_gap` (H5Pdapl.c:1207): how many consecutive
1851    /// missing printf-named source datasets the extent resolution looks past
1852    /// before it stops. 0 — the default — stops at the first one missing.
1853    ///
1854    /// `u64::MAX` is libhdf5's `HSIZE_UNDEF`, which that call rejects as "not
1855    /// a valid printf gap size"; here the rejection surfaces from the open
1856    /// that uses the property, since a builder method has no way to report
1857    /// it.
1858    pub fn virtual_printf_gap(mut self, gap: u64) -> Self {
1859        self.printf_gap = gap;
1860        self
1861    }
1862
1863    /// `H5Pget_virtual_view` (H5Pdapl.c:1102).
1864    pub fn view(&self) -> VirtualView {
1865        self.view
1866    }
1867
1868    /// `H5Pset_virtual_prefix` (H5Pdapl.c:1478): a directory a virtual
1869    /// dataset's *source file names* are looked for under, before the
1870    /// virtual file's own directory and after `HDF5_VDS_PREFIX`.
1871    ///
1872    /// It is the third step of `H5F_prefix_open_file`'s search order
1873    /// (H5Fint.c:938-950), and it is reached only when `HDF5_VDS_PREFIX` is
1874    /// unset or empty: `H5D__build_file_prefix` reads the environment first
1875    /// and falls back to this property (H5Dint.c:1077-1082), so an
1876    /// environment prefix shadows this one outright rather than being tried
1877    /// alongside it.
1878    ///
1879    /// A leading `${ORIGIN}` stands for the directory holding the virtual
1880    /// dataset's own file (H5Dint.c:1105-1113), and `"."` or `""` means "no
1881    /// prefix" (:1096-1100), both exactly as for the environment variable.
1882    ///
1883    /// Like the other two, this is a *dataset access* property that is never
1884    /// stored in the file, and the first open of a virtual dataset fixes it
1885    /// for every open that overlaps it.
1886    pub fn virtual_prefix(mut self, prefix: impl Into<String>) -> Self {
1887        self.virtual_prefix = Some(prefix.into());
1888        self
1889    }
1890
1891    /// `H5Pset_efile_prefix` (H5Pdapl.c:1392): a directory the *raw data
1892    /// files* of a dataset stored through an external file list are looked
1893    /// for under.
1894    ///
1895    /// This one takes no search at all, unlike the other two prefixes:
1896    /// `H5D__efl_read` joins the prefix to the stored name with
1897    /// `H5_combine_path` and opens exactly that one path (H5Defl.c:315-317).
1898    /// With no prefix in force the stored name is used as written, so a
1899    /// relative one resolves against the *process's current directory* and
1900    /// not against the directory holding the HDF5 file — measured under
1901    /// libhdf5 1.14.6 and 2.0.0: a raw data file next to the HDF5 file is
1902    /// not found, while the same name under the current directory is.
1903    ///
1904    /// It shares [`virtual_prefix`](Self::virtual_prefix)'s expansion rules,
1905    /// because both are built by `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX`
1906    /// shadows this property outright rather than merely preceding it
1907    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
1908    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
1909    /// (:1098-1102).
1910    ///
1911    /// # A second open must name the same one
1912    ///
1913    /// Where a mismatched [`virtual_prefix`](Self::virtual_prefix) is
1914    /// silently ignored by the second open, a mismatched external file prefix
1915    /// is an *error*: `H5D_open` compares the expanded prefix against the one
1916    /// the already-open dataset resolved under and refuses the open when they
1917    /// differ (H5Dint.c:1533-1545). Expanded, so two opens that differ only
1918    /// in a property the environment shadows still agree. Closing every
1919    /// handle releases the answer, and the next open sets its own.
1920    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
1921        self.efile_prefix = Some(prefix.into());
1922        self
1923    }
1924
1925    /// `H5Pget_virtual_printf_gap` (H5Pdapl.c:1243) — the value set, not the
1926    /// one the extent resolution ends up using; see
1927    /// [`VirtualView::FirstMissing`].
1928    pub fn printf_gap(&self) -> u64 {
1929        self.printf_gap
1930    }
1931
1932    /// `H5Pget_virtual_prefix` (H5Pdapl.c:1510) — the property as set, before
1933    /// `HDF5_VDS_PREFIX` gets to shadow it and before `${ORIGIN}` is
1934    /// expanded. `None` is `H5D_ACS_VDS_PREFIX_DEF`, a null prefix
1935    /// (H5Pdapl.c:72).
1936    pub fn virtual_prefix_value(&self) -> Option<&str> {
1937        self.virtual_prefix.as_deref()
1938    }
1939
1940    /// `H5Pget_efile_prefix` (H5Pdapl.c:1422) — the property as set, before
1941    /// `HDF5_EXTFILE_PREFIX` gets to shadow it and before `${ORIGIN}` is
1942    /// expanded. `None` is `H5D_ACS_EFILE_PREFIX_DEF`, a null prefix
1943    /// (H5Pdapl.c:90).
1944    pub fn efile_prefix_value(&self) -> Option<&str> {
1945        self.efile_prefix.as_deref()
1946    }
1947
1948    /// The printf gap `H5D__virtual_set_extent_unlim` actually scans with:
1949    /// the property under [`VirtualView::LastAvailable`], and 0 under
1950    /// [`VirtualView::FirstMissing`], because `H5D__virtual_init` only reads
1951    /// the property in the first case (H5Dvirtual.c:2182-2188).
1952    ///
1953    /// The single owner of that rule — the resolution never reads
1954    /// [`printf_gap`](Self::printf_gap) directly.
1955    pub(crate) fn effective_printf_gap(&self) -> u64 {
1956        match self.view {
1957            VirtualView::LastAvailable => self.printf_gap,
1958            VirtualView::FirstMissing => 0,
1959        }
1960    }
1961
1962    /// Reject what `H5Pset_virtual_printf_gap` rejects, at the open that uses
1963    /// the property.
1964    pub(crate) fn validate(&self) -> Result<()> {
1965        if self.printf_gap == u64::MAX {
1966            return Err(Hdf5Error::InvalidState(
1967                "virtual_printf_gap(u64::MAX) is libhdf5's HSIZE_UNDEF, which \
1968                 H5Pset_virtual_printf_gap refuses as \"not a valid printf gap size\""
1969                    .into(),
1970            ));
1971        }
1972        Ok(())
1973    }
1974}
1975
1976/// Elements a selection of `counts` holds, refusing a product that overflows
1977/// `usize` rather than wrapping it into a small allocation.
1978fn element_count(counts: &[u64]) -> Result<usize> {
1979    counts
1980        .iter()
1981        .try_fold(1usize, |acc, &c| {
1982            usize::try_from(c).ok().and_then(|c| acc.checked_mul(c))
1983        })
1984        .ok_or_else(|| {
1985            Hdf5Error::InvalidState(format!("selection {counts:?} has more elements than usize"))
1986        })
1987}
1988
1989impl H5Dataset {
1990    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
1991    pub(crate) fn new_reader(
1992        file_inner: SharedInner,
1993        name: String,
1994        shape: Vec<usize>,
1995        element_size: usize,
1996        open: Option<crate::io::reader::DatasetOpenToken>,
1997    ) -> Self {
1998        Self {
1999            file_inner,
2000            info: DatasetInfo::Reader {
2001                name,
2002                shape,
2003                element_size,
2004            },
2005            _open: open,
2006        }
2007    }
2008
2009    /// Create a writer-mode dataset handle for an already-created dataset
2010    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
2011    ///
2012    /// Reconstructs the same handle `new_dataset().create()` returns, so the
2013    /// reopened dataset supports attribute writes and chunk appends.
2014    ///
2015    /// `is_null` is always `false` here: reopening an existing NULL-dataspace
2016    /// dataset for further writes is not a case this constructor's caller
2017    /// distinguishes (a NULL dataset has nothing to append or chunk-write in
2018    /// the first place).
2019    pub(crate) fn new_writer(
2020        file_inner: SharedInner,
2021        index: usize,
2022        parts: crate::io::writer::DatasetHandleParts,
2023    ) -> Self {
2024        Self {
2025            file_inner,
2026            info: DatasetInfo::Writer {
2027                index,
2028                shape: parts.shape,
2029                element_size: parts.element_size,
2030                chunk_index: parts.chunk_index,
2031                is_null: false,
2032            },
2033            _open: parts.open,
2034        }
2035    }
2036
2037    /// Return the dataset dimensions.
2038    pub fn shape(&self) -> Vec<usize> {
2039        match &self.info {
2040            DatasetInfo::Writer { shape, .. } => shape.clone(),
2041            DatasetInfo::Reader { shape, .. } => shape.clone(),
2042        }
2043    }
2044
2045    /// Return the number of dimensions (rank) of the dataset.
2046    pub fn ndims(&self) -> usize {
2047        match &self.info {
2048            DatasetInfo::Writer { shape, .. } => shape.len(),
2049            DatasetInfo::Reader { shape, .. } => shape.len(),
2050        }
2051    }
2052
2053    /// Return the total number of elements in the dataset.
2054    ///
2055    /// 0 for a NULL dataspace ([`is_null`](Self::is_null)) — unlike a scalar,
2056    /// whose `shape()` is the same empty `Vec` but which holds exactly one
2057    /// element, so `shape().iter().product()` cannot be used here.
2058    pub fn total_elements(&self) -> usize {
2059        if self.is_null() {
2060            return 0;
2061        }
2062        match &self.info {
2063            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
2064            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
2065        }
2066    }
2067
2068    /// Return the size of one element in bytes.
2069    pub fn element_size(&self) -> usize {
2070        match &self.info {
2071            DatasetInfo::Writer { element_size, .. } => *element_size,
2072            DatasetInfo::Reader { element_size, .. } => *element_size,
2073        }
2074    }
2075
2076    /// Return whether this dataset has the NULL dataspace: no elements at
2077    /// all, distinct from a scalar dataset (rank 0, exactly one element) —
2078    /// both report the same empty [`shape`](Self::shape). See
2079    /// [`DatasetBuilder::null`].
2080    pub fn is_null(&self) -> bool {
2081        match &self.info {
2082            DatasetInfo::Writer { is_null, .. } => *is_null,
2083            DatasetInfo::Reader { name, .. } => {
2084                let mut inner = borrow_inner_mut(&self.file_inner);
2085                match &mut *inner {
2086                    H5FileInner::Reader(reader) => reader
2087                        .dataset_info(name)
2088                        .map(|info| info.dataspace.is_null())
2089                        .unwrap_or(false),
2090                    _ => false,
2091                }
2092            }
2093        }
2094    }
2095
2096    /// Return the element datatype as parsed from the file (read mode only).
2097    ///
2098    /// Unlike [`element_size`](Self::element_size), which reports only the
2099    /// byte width, this exposes the full datatype: its class (integer vs
2100    /// floating-point vs string vs compound …), signedness, byte order and
2101    /// bit precision. Callers that must reconstruct the exact stored type —
2102    /// for example to map it to a NumPy / Arrow dtype — should use this
2103    /// instead of inferring a type from the byte width, which cannot
2104    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
2105    /// bytes).
2106    ///
2107    /// # Errors
2108    ///
2109    /// Returns an error if the file is in write mode, or if the dataset can
2110    /// no longer be found in the reader's metadata.
2111    ///
2112    /// ```no_run
2113    /// # use rust_hdf5::{H5File, DatatypeMessage};
2114    /// let file = H5File::open("data.h5").unwrap();
2115    /// let ds = file.dataset("image").unwrap();
2116    /// match ds.datatype().unwrap() {
2117    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
2118    ///         println!("integer: {} bytes, signed={}", size, signed);
2119    ///     }
2120    ///     DatatypeMessage::FloatingPoint { size, .. } => {
2121    ///         println!("float: {} bytes", size);
2122    ///     }
2123    ///     other => println!("other type: {other}"),
2124    /// }
2125    /// ```
2126    pub fn datatype(&self) -> Result<DatatypeMessage> {
2127        match &self.info {
2128            DatasetInfo::Reader { name, .. } => {
2129                let mut inner = borrow_inner_mut(&self.file_inner);
2130                match &mut *inner {
2131                    H5FileInner::Reader(reader) => reader
2132                        .dataset_info(name)
2133                        .map(|info| info.datatype.clone())
2134                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2135                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2136                }
2137            }
2138            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2139                "datatype() is only available in read mode".into(),
2140            )),
2141        }
2142    }
2143
2144    /// Return the chunk dimensions, if this is a chunked dataset.
2145    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
2146        match &self.info {
2147            DatasetInfo::Reader { name, .. } => {
2148                let mut inner = borrow_inner_mut(&self.file_inner);
2149                if let H5FileInner::Reader(reader) = &mut *inner {
2150                    if let Some(info) = reader.dataset_info(name) {
2151                        use crate::format::messages::data_layout::DataLayoutMessage;
2152                        let chunk_dims = match &info.layout {
2153                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
2154                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
2155                            _ => None,
2156                        };
2157                        if let Some(chunk_dims) = chunk_dims {
2158                            // Strip trailing element-size dimension
2159                            return Some(
2160                                chunk_dims[..chunk_dims.len() - 1]
2161                                    .iter()
2162                                    .map(|&d| d as usize)
2163                                    .collect(),
2164                            );
2165                        }
2166                    }
2167                }
2168                None
2169            }
2170            DatasetInfo::Writer { .. } => None,
2171        }
2172    }
2173
2174    /// Return whether this is a chunked dataset.
2175    pub fn is_chunked(&self) -> bool {
2176        match &self.info {
2177            DatasetInfo::Writer { chunk_index, .. } => chunk_index.is_some(),
2178            DatasetInfo::Reader { name, .. } => {
2179                let mut inner = borrow_inner_mut(&self.file_inner);
2180                match &mut *inner {
2181                    H5FileInner::Reader(reader) => {
2182                        if let Some(info) = reader.dataset_info(name) {
2183                            use crate::format::messages::data_layout::DataLayoutMessage;
2184                            matches!(
2185                                info.layout,
2186                                DataLayoutMessage::ChunkedV4 { .. }
2187                                    | DataLayoutMessage::ChunkedV3 { .. }
2188                            )
2189                        } else {
2190                            false
2191                        }
2192                    }
2193                    _ => false,
2194                }
2195            }
2196        }
2197    }
2198
2199    /// Return the dataset's storage layout class (read mode only).
2200    ///
2201    /// # Errors
2202    ///
2203    /// Returns an error if the file is in write mode, or if the dataset can
2204    /// no longer be found in the reader's metadata.
2205    pub fn storage_layout(&self) -> Result<StorageLayout> {
2206        match &self.info {
2207            DatasetInfo::Reader { name, .. } => {
2208                let mut inner = borrow_inner_mut(&self.file_inner);
2209                match &mut *inner {
2210                    H5FileInner::Reader(reader) => {
2211                        use crate::format::messages::data_layout::DataLayoutMessage;
2212                        reader
2213                            .dataset_info(name)
2214                            .map(|info| match &info.layout {
2215                                DataLayoutMessage::Compact { .. } => StorageLayout::Compact,
2216                                DataLayoutMessage::Contiguous { .. } => StorageLayout::Contiguous,
2217                                DataLayoutMessage::ChunkedV3 { .. }
2218                                | DataLayoutMessage::ChunkedV4 { .. } => StorageLayout::Chunked,
2219                                DataLayoutMessage::Virtual { .. } => StorageLayout::Virtual,
2220                            })
2221                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2222                    }
2223                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2224                }
2225            }
2226            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2227                "storage_layout() is only available in read mode".into(),
2228            )),
2229        }
2230    }
2231
2232    /// Return the chunk index structure this dataset's layout uses, or
2233    /// `None` for a dataset that is not chunked (read mode only).
2234    ///
2235    /// # Errors
2236    ///
2237    /// Returns an error if the file is in write mode, or if the dataset can
2238    /// no longer be found in the reader's metadata.
2239    pub fn chunk_index(&self) -> Result<Option<ChunkIndex>> {
2240        match &self.info {
2241            DatasetInfo::Reader { name, .. } => {
2242                let mut inner = borrow_inner_mut(&self.file_inner);
2243                match &mut *inner {
2244                    H5FileInner::Reader(reader) => {
2245                        use crate::format::messages::data_layout::{
2246                            ChunkIndexType, DataLayoutMessage,
2247                        };
2248                        reader
2249                            .dataset_info(name)
2250                            .map(|info| match &info.layout {
2251                                DataLayoutMessage::ChunkedV3 { .. } => Some(ChunkIndex::BtreeV1),
2252                                DataLayoutMessage::ChunkedV4 { index_type, .. } => {
2253                                    Some(match index_type {
2254                                        ChunkIndexType::SingleChunk => ChunkIndex::SingleChunk,
2255                                        ChunkIndexType::Implicit => ChunkIndex::Implicit,
2256                                        ChunkIndexType::FixedArray => ChunkIndex::FixedArray,
2257                                        ChunkIndexType::ExtensibleArray => {
2258                                            ChunkIndex::ExtensibleArray
2259                                        }
2260                                        ChunkIndexType::BTreeV2 => ChunkIndex::BtreeV2,
2261                                    })
2262                                }
2263                                _ => None,
2264                            })
2265                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2266                    }
2267                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2268                }
2269            }
2270            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2271                "chunk_index() is only available in read mode".into(),
2272            )),
2273        }
2274    }
2275
2276    /// Return this dataset's filter pipeline (read mode only), in
2277    /// application order. Empty when the dataset has no filter pipeline
2278    /// message at all — an unfiltered dataset, not an error.
2279    ///
2280    /// # Errors
2281    ///
2282    /// Returns an error if the file is in write mode, or if the dataset can
2283    /// no longer be found in the reader's metadata.
2284    pub fn filters(&self) -> Result<Vec<Filter>> {
2285        match &self.info {
2286            DatasetInfo::Reader { name, .. } => {
2287                let mut inner = borrow_inner_mut(&self.file_inner);
2288                match &mut *inner {
2289                    H5FileInner::Reader(reader) => reader
2290                        .dataset_info(name)
2291                        .map(|info| {
2292                            info.filter_pipeline
2293                                .as_ref()
2294                                .map(|fp| fp.filters.clone())
2295                                .unwrap_or_default()
2296                        })
2297                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2298                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2299                }
2300            }
2301            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2302                "filters() is only available in read mode".into(),
2303            )),
2304        }
2305    }
2306
2307    /// Return this dataset's fill-value state (read mode only).
2308    ///
2309    /// # Errors
2310    ///
2311    /// Returns an error if the file is in write mode, or if the dataset can
2312    /// no longer be found in the reader's metadata.
2313    pub fn fill_value(&self) -> Result<FillValue> {
2314        match &self.info {
2315            DatasetInfo::Reader { name, .. } => {
2316                let mut inner = borrow_inner_mut(&self.file_inner);
2317                match &mut *inner {
2318                    H5FileInner::Reader(reader) => reader
2319                        .dataset_info(name)
2320                        .map(|info| match info.fill_defined {
2321                            0 => FillValue::Undefined,
2322                            2 => {
2323                                FillValue::UserDefined(info.fill_value.clone().unwrap_or_default())
2324                            }
2325                            _ => FillValue::Default,
2326                        })
2327                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2328                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2329                }
2330            }
2331            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2332                "fill_value() is only available in read mode".into(),
2333            )),
2334        }
2335    }
2336
2337    /// Return when this dataset's fill value is written into allocated
2338    /// storage (read mode only) — `H5Pget_fill_time`.
2339    ///
2340    /// # Errors
2341    ///
2342    /// Returns an error if the file is in write mode, or if the dataset can
2343    /// no longer be found in the reader's metadata.
2344    pub fn fill_time(&self) -> Result<FillTime> {
2345        match &self.info {
2346            DatasetInfo::Reader { name, .. } => {
2347                let mut inner = borrow_inner_mut(&self.file_inner);
2348                match &mut *inner {
2349                    H5FileInner::Reader(reader) => reader
2350                        .dataset_info(name)
2351                        .map(|info| match info.fill_write_time {
2352                            0 => FillTime::Alloc,
2353                            1 => FillTime::Never,
2354                            _ => FillTime::IfSet,
2355                        })
2356                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2357                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2358                }
2359            }
2360            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2361                "fill_time() is only available in read mode".into(),
2362            )),
2363        }
2364    }
2365
2366    /// Return when this dataset's raw-data storage is allocated (read mode
2367    /// only) — `H5Pget_alloc_time`.
2368    ///
2369    /// # Errors
2370    ///
2371    /// Returns an error if the file is in write mode, or if the dataset can
2372    /// no longer be found in the reader's metadata.
2373    pub fn alloc_time(&self) -> Result<AllocTime> {
2374        match &self.info {
2375            DatasetInfo::Reader { name, .. } => {
2376                let mut inner = borrow_inner_mut(&self.file_inner);
2377                match &mut *inner {
2378                    H5FileInner::Reader(reader) => reader
2379                        .dataset_info(name)
2380                        .map(|info| match info.alloc_time {
2381                            1 => AllocTime::Early,
2382                            3 => AllocTime::Incr,
2383                            _ => AllocTime::Late,
2384                        })
2385                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2386                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2387                }
2388            }
2389            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2390                "alloc_time() is only available in read mode".into(),
2391            )),
2392        }
2393    }
2394
2395    /// Return this dataset's external raw-data file segments (read mode
2396    /// only), in the order the dataset's logical byte range concatenates
2397    /// them. Empty for a dataset whose data lives in this file.
2398    ///
2399    /// # Errors
2400    ///
2401    /// Returns an error if the file is in write mode, or if the dataset can
2402    /// no longer be found in the reader's metadata.
2403    pub fn external_files(&self) -> Result<Vec<ExternalFileSegment>> {
2404        match &self.info {
2405            DatasetInfo::Reader { name, .. } => {
2406                let mut inner = borrow_inner_mut(&self.file_inner);
2407                match &mut *inner {
2408                    H5FileInner::Reader(reader) => reader
2409                        .dataset_info(name)
2410                        .map(|info| info.external_files.clone())
2411                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2412                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2413                }
2414            }
2415            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2416                "external_files() is only available in read mode".into(),
2417            )),
2418        }
2419    }
2420
2421    /// Return this dataset's maximum dimension sizes (read mode only):
2422    /// `None` in a dimension marks that axis unlimited. A dataset with no
2423    /// maximum-dimensions message reports its current shape (max == current
2424    /// — the upstream convention for a fixed-extent dataset).
2425    ///
2426    /// # Errors
2427    ///
2428    /// Returns an error if the file is in write mode, or if the dataset can
2429    /// no longer be found in the reader's metadata.
2430    pub fn max_shape(&self) -> Result<Vec<Option<usize>>> {
2431        match &self.info {
2432            DatasetInfo::Reader { name, .. } => {
2433                let mut inner = borrow_inner_mut(&self.file_inner);
2434                match &mut *inner {
2435                    H5FileInner::Reader(reader) => reader
2436                        .dataset_info(name)
2437                        .map(|info| match &info.dataspace.max_dims {
2438                            Some(max_dims) => max_dims
2439                                .iter()
2440                                .map(|&d| (d != u64::MAX).then_some(d as usize))
2441                                .collect(),
2442                            None => info
2443                                .dataspace
2444                                .dims
2445                                .iter()
2446                                .map(|&d| Some(d as usize))
2447                                .collect(),
2448                        })
2449                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2450                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2451                }
2452            }
2453            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2454                "max_shape() is only available in read mode".into(),
2455            )),
2456        }
2457    }
2458
2459    /// Return this dataset's virtual-dataset source/virtual mappings (read
2460    /// mode only), in on-disk order. Empty for any dataset whose layout is
2461    /// not virtual, and for a virtual dataset that has no mappings yet.
2462    ///
2463    /// # Errors
2464    ///
2465    /// Returns an error if the file is in write mode, or if the dataset can
2466    /// no longer be found in the reader's metadata.
2467    pub fn virtual_mappings(&self) -> Result<Vec<VirtualMapping>> {
2468        match &self.info {
2469            DatasetInfo::Reader { name, .. } => {
2470                let mut inner = borrow_inner_mut(&self.file_inner);
2471                match &mut *inner {
2472                    H5FileInner::Reader(reader) => reader
2473                        .dataset_info(name)
2474                        .map(|info| {
2475                            info.virtual_mappings
2476                                .as_ref()
2477                                .map(|vml| vml.mappings.clone())
2478                                .unwrap_or_default()
2479                        })
2480                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2481                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2482                }
2483            }
2484            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2485                "virtual_mappings() is only available in read mode".into(),
2486            )),
2487        }
2488    }
2489
2490    /// Return the names of all attributes on this dataset (read mode only).
2491    pub fn attr_names(&self) -> Result<Vec<String>> {
2492        match &self.info {
2493            DatasetInfo::Reader { name, .. } => {
2494                let mut inner = borrow_inner_mut(&self.file_inner);
2495                match &mut *inner {
2496                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
2497                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2498                }
2499            }
2500            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2501                "attr_names not available in write mode".into(),
2502            )),
2503        }
2504    }
2505
2506    /// Why the attribute `attr_name` on this dataset cannot be read, or `None`
2507    /// when it can be.
2508    ///
2509    /// An attribute whose message this crate cannot decode is still listed by
2510    /// [`attr_names`](Self::attr_names) — the object header carries it — and
2511    /// this says what stands in the way. Opening it through
2512    /// [`attr`](Self::attr) fails with the same text.
2513    pub fn attr_unreadable_reason(&self, attr_name: &str) -> Result<Option<String>> {
2514        match &self.info {
2515            DatasetInfo::Reader { name, .. } => {
2516                let mut inner = borrow_inner_mut(&self.file_inner);
2517                match &mut *inner {
2518                    H5FileInner::Reader(reader) => Ok(reader
2519                        .dataset_attr_unreadable_reason(name, attr_name)
2520                        .map(str::to_string)),
2521                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2522                }
2523            }
2524            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2525                "attr_unreadable_reason not available in write mode".into(),
2526            )),
2527        }
2528    }
2529
2530    /// Why this dataset's attribute *set* cannot be listed, or `None` when it
2531    /// can be.
2532    ///
2533    /// The object-scope counterpart of
2534    /// [`attr_unreadable_reason`](Self::attr_unreadable_reason). A dense
2535    /// attribute set is indexed by name hash, so a heap or index that will not
2536    /// read yields no names to hang a per-attribute reason on;
2537    /// [`attr_names`](Self::attr_names) then returns the failure rather than a
2538    /// short list, and this reports it without an attribute name.
2539    pub fn attrs_unreadable_reason(&self) -> Result<Option<String>> {
2540        match &self.info {
2541            DatasetInfo::Reader { name, .. } => {
2542                let mut inner = borrow_inner_mut(&self.file_inner);
2543                match &mut *inner {
2544                    H5FileInner::Reader(reader) => Ok(reader
2545                        .dataset_attrs_unreadable_reason(name)
2546                        .map(str::to_string)),
2547                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2548                }
2549            }
2550            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2551                "attrs_unreadable_reason not available in write mode".into(),
2552            )),
2553        }
2554    }
2555
2556    /// This dataset's own compact-vs-dense attribute storage — the
2557    /// equivalent of `h5py.h5o.get_info(did.id).meta_size.attr.index_size`
2558    /// being nonzero (read mode only).
2559    pub fn attr_storage(&self) -> Result<AttributeStorage> {
2560        match &self.info {
2561            DatasetInfo::Reader { name, .. } => {
2562                let mut inner = borrow_inner_mut(&self.file_inner);
2563                match &mut *inner {
2564                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_storage(name)?),
2565                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2566                }
2567            }
2568            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2569                "attr_storage not available in write mode".into(),
2570            )),
2571        }
2572    }
2573
2574    /// This dataset's own object-header attribute count — the equivalent of
2575    /// `h5py.h5o.get_info(did.id).num_attrs` (read mode only).
2576    pub fn header_attr_count(&self) -> Result<u64> {
2577        match &self.info {
2578            DatasetInfo::Reader { name, .. } => {
2579                let mut inner = borrow_inner_mut(&self.file_inner);
2580                match &mut *inner {
2581                    H5FileInner::Reader(reader) => Ok(reader.dataset_header_attr_count(name)?),
2582                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2583                }
2584            }
2585            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2586                "header_attr_count not available in write mode".into(),
2587            )),
2588        }
2589    }
2590
2591    /// Open an attribute by name (read mode only).
2592    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
2593        match &self.info {
2594            DatasetInfo::Reader { name, .. } => {
2595                let mut inner = borrow_inner_mut(&self.file_inner);
2596                match &mut *inner {
2597                    H5FileInner::Reader(reader) => {
2598                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
2599                        Ok(crate::attribute::H5Attribute::new_reader(
2600                            clone_inner(&self.file_inner),
2601                            attr_msg,
2602                        ))
2603                    }
2604                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2605                }
2606            }
2607            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2608                "attr() not available in write mode".into(),
2609            )),
2610        }
2611    }
2612
2613    /// Start building a new attribute on this dataset.
2614    ///
2615    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
2616    /// and `.create("name")` to finalize.
2617    ///
2618    /// # Example
2619    ///
2620    /// ```no_run
2621    /// # use rust_hdf5::H5File;
2622    /// # use rust_hdf5::types::VarLenUnicode;
2623    /// let file = H5File::create("attr.h5").unwrap();
2624    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
2625    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
2626    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
2627    /// ```
2628    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
2629        let ds_index = match &self.info {
2630            DatasetInfo::Writer { index, .. } => *index,
2631            DatasetInfo::Reader { .. } => {
2632                // Reader mode: we'll return a builder that will error on create.
2633                // Using usize::MAX as sentinel.
2634                usize::MAX
2635            }
2636        };
2637        AttrBuilder::new(&self.file_inner, ds_index)
2638    }
2639
2640    /// Write a typed slice holding the dataset's whole image.
2641    ///
2642    /// The slice length must match the total number of elements declared by
2643    /// the dataset shape. The data is reinterpreted as raw bytes and written
2644    /// to the file: to the contiguous data block, or — for a chunked dataset —
2645    /// scattered across its chunk grid, through the filter pipeline if one is
2646    /// set. To write only part of a dataset, use
2647    /// [`write_slice`](Self::write_slice).
2648    ///
2649    /// # Errors
2650    ///
2651    /// Returns an error if:
2652    /// - The file is in read mode.
2653    /// - The data length does not match the declared shape.
2654    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
2655        match &self.info {
2656            DatasetInfo::Writer {
2657                index,
2658                shape,
2659                element_size,
2660                chunk_index,
2661                is_null,
2662            } => {
2663                if *is_null {
2664                    return Err(Hdf5Error::InvalidState(
2665                        "cannot write to a NULL dataspace dataset".into(),
2666                    ));
2667                }
2668                let total_elements: usize = shape.iter().product();
2669                if data.len() != total_elements {
2670                    return Err(Hdf5Error::InvalidState(format!(
2671                        "data length {} does not match dataset size {}",
2672                        data.len(),
2673                        total_elements,
2674                    )));
2675                }
2676
2677                // Verify element size matches
2678                if T::element_size() != *element_size {
2679                    return Err(Hdf5Error::TypeMismatch(format!(
2680                        "write type has element size {} but dataset expects {}",
2681                        T::element_size(),
2682                        element_size,
2683                    )));
2684                }
2685
2686                // Safety: T: Copy + 'static (numeric primitive) with well-defined
2687                // byte representation. The resulting slice borrows `data` and
2688                // lives only as long as this block.
2689                let byte_len = data.len() * T::element_size();
2690                let host =
2691                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
2692                let datatype = {
2693                    let inner = borrow_inner(&self.file_inner);
2694                    match &*inner {
2695                        H5FileInner::Writer(writer) => writer.dataset_datatype(*index),
2696                        _ => {
2697                            return Err(Hdf5Error::InvalidState(
2698                                "file is no longer in write mode".into(),
2699                            ))
2700                        }
2701                    }
2702                };
2703                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
2704
2705                if let Some(kind) = *chunk_index {
2706                    // A chunked dataset has no contiguous data block; scatter
2707                    // the full row-major image into its chunk grid and write
2708                    // each chunk through the dataset's filter pipeline.
2709                    return self.write_full_image_chunked(*index, kind, &stored, *element_size);
2710                }
2711
2712                let inner = borrow_inner(&self.file_inner);
2713                match &*inner {
2714                    H5FileInner::Writer(writer) => {
2715                        writer.write_dataset_raw(*index, &stored)?;
2716                        Ok(())
2717                    }
2718                    _ => Err(Hdf5Error::InvalidState(
2719                        "file is no longer in write mode".into(),
2720                    )),
2721                }
2722            }
2723            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2724                "cannot write to a dataset opened in read mode".into(),
2725            )),
2726        }
2727    }
2728
2729    /// Write the raw byte image of the whole dataset directly.
2730    ///
2731    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
2732    /// data block, or a chunk grid the image is scattered across.
2733    ///
2734    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
2735    /// `H5Type` carrier, so it works for element types that have no matching
2736    /// Rust primitive — in particular a runtime
2737    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
2738    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
2739    /// `product(shape) * element_size`, where `element_size` is taken from the
2740    /// dataset's on-disk datatype.
2741    ///
2742    /// ```no_run
2743    /// # use rust_hdf5::H5File;
2744    /// # use rust_hdf5::types::{CompoundType, H5Type};
2745    /// let file = H5File::create("c.h5").unwrap();
2746    /// let ct = CompoundType {
2747    ///     members: vec![
2748    ///         ("id".to_string(), i32::hdf5_type(), 0),
2749    ///         ("val".to_string(), f64::hdf5_type(), 4),
2750    ///     ],
2751    ///     total_size: 12,
2752    /// };
2753    /// let ds = file
2754    ///     .new_dataset::<u8>()
2755    ///     .datatype(ct.to_datatype())
2756    ///     .shape(&[2])
2757    ///     .create("records")
2758    ///     .unwrap();
2759    /// let mut bytes = Vec::new();
2760    /// bytes.extend_from_slice(&1i32.to_le_bytes());
2761    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
2762    /// bytes.extend_from_slice(&2i32.to_le_bytes());
2763    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
2764    /// ds.write_raw_bytes(&bytes).unwrap();
2765    /// ```
2766    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
2767        match &self.info {
2768            DatasetInfo::Writer {
2769                index,
2770                shape,
2771                element_size,
2772                chunk_index,
2773                is_null,
2774            } => {
2775                if *is_null {
2776                    return Err(Hdf5Error::InvalidState(
2777                        "cannot write to a NULL dataspace dataset".into(),
2778                    ));
2779                }
2780                let expected: usize = shape.iter().product::<usize>() * *element_size;
2781                if bytes.len() != expected {
2782                    return Err(Hdf5Error::InvalidState(format!(
2783                        "raw byte length {} does not match dataset size {} \
2784                         (product(shape) * element_size {})",
2785                        bytes.len(),
2786                        expected,
2787                        element_size,
2788                    )));
2789                }
2790                if let Some(kind) = *chunk_index {
2791                    // Scatter the full row-major image into the chunk grid
2792                    // (same path as write_raw, carrier-agnostic bytes).
2793                    return self.write_full_image_chunked(*index, kind, bytes, *element_size);
2794                }
2795                let inner = borrow_inner(&self.file_inner);
2796                match &*inner {
2797                    H5FileInner::Writer(writer) => {
2798                        writer.write_dataset_raw(*index, bytes)?;
2799                        Ok(())
2800                    }
2801                    _ => Err(Hdf5Error::InvalidState(
2802                        "file is no longer in write mode".into(),
2803                    )),
2804                }
2805            }
2806            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2807                "cannot write to a dataset opened in read mode".into(),
2808            )),
2809        }
2810    }
2811
2812    /// Scatter a full row-major dataset image into its chunk grid, writing
2813    /// every chunk through the dataset's filter pipeline.
2814    ///
2815    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
2816    /// — it is how [`write_raw`](Self::write_raw) and
2817    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
2818    /// (including the single auto-chunk created when a filter is set without
2819    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
2820    /// chunk footprint, exactly as libhdf5 stores them.
2821    fn write_full_image_chunked(
2822        &self,
2823        index: usize,
2824        kind: ChunkIndexKind,
2825        bytes: &[u8],
2826        element_size: usize,
2827    ) -> Result<()> {
2828        let inner = borrow_inner(&self.file_inner);
2829        let writer = match &*inner {
2830            H5FileInner::Writer(w) => w,
2831            _ => {
2832                return Err(Hdf5Error::InvalidState(
2833                    "file is no longer in write mode".into(),
2834                ))
2835            }
2836        };
2837        // Whole-operation guard: the flush, the grid snapshot and the chunk
2838        // writes below must not interleave with a concurrent same-dataset
2839        // operation.
2840        let cell = writer.ds(index);
2841        let _op = cell.op.lock();
2842        // A buffered append tail would flush over the image at close; hand
2843        // it to the chunks first, the image below overwrites everything.
2844        writer.flush_append_buffer(index)?;
2845        let chunk_dims = writer
2846            .dataset_chunk_dims(index)
2847            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
2848            .to_vec();
2849        let dims = writer.dataset_dims(index).to_vec();
2850        let rank = dims.len();
2851
2852        // Chunk grid: number of chunks along each dimension (row-major).
2853        let mut grid = vec![0u64; rank];
2854        for d in 0..rank {
2855            grid[d] = if chunk_dims[d] > 0 {
2856                dims[d].div_ceil(chunk_dims[d])
2857            } else {
2858                0
2859            };
2860        }
2861        let total_chunks: u64 = grid.iter().product();
2862
2863        // Decode the iteration counter into row-major coordinates over the
2864        // *current* image's chunk grid. This is only an odometer over the
2865        // chunks the image spans — the slot a chunk is recorded under comes
2866        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
2867        // extent decides.
2868        let coords_of = |linear: u64| -> Vec<u64> {
2869            let mut rem = linear;
2870            let mut coords = vec![0u64; rank];
2871            for d in (0..rank).rev() {
2872                coords[d] = rem % grid[d];
2873                rem /= grid[d];
2874            }
2875            coords
2876        };
2877
2878        // The batch entry points exist for one reason: to run the filter
2879        // pipeline over a window of chunks in parallel. An unfiltered dataset
2880        // has no pipeline to run, so it takes the plain per-chunk owner
2881        // whatever its index is, and only a filtered extensible or fixed array
2882        // — the two indexes with a batch entry point — takes the window below.
2883        let batched = writer.dataset_is_filtered(index)
2884            && matches!(
2885                kind,
2886                ChunkIndexKind::ExtensibleArray | ChunkIndexKind::FixedArray
2887            );
2888        if !batched {
2889            // One staging buffer for the whole image, reused chunk after
2890            // chunk: a chunk that already sits as one complete run of `bytes`
2891            // needs no staging at all and goes to the file straight out of the
2892            // caller's slice, so only an n-D interleave or a short edge pays
2893            // for a gather.
2894            let mut staging = Vec::new();
2895            for linear in 0..total_chunks {
2896                let coords = coords_of(linear);
2897                let chunk =
2898                    match Self::contiguous_chunk_span(&dims, &chunk_dims, &coords, element_size) {
2899                        Some(span) => &bytes[span],
2900                        None => {
2901                            Self::gather_chunk_into(
2902                                &mut staging,
2903                                bytes,
2904                                &dims,
2905                                &chunk_dims,
2906                                &coords,
2907                                element_size,
2908                            );
2909                            &staging[..]
2910                        }
2911                    };
2912                writer.write_chunk_at_coords(index, &coords, chunk)?;
2913            }
2914        } else {
2915            // Hand the pipeline a window of chunks so it compresses them in
2916            // parallel (with the `parallel` feature). A fixed-size window
2917            // bounds peak memory instead of materializing every chunk at once;
2918            // 256 keeps every rayon worker fed while capping the transient
2919            // buffers to window * chunk bytes. The compressors read the window
2920            // concurrently, so a gathered chunk here cannot share one reused
2921            // buffer the way the sequential path above does — but a chunk that
2922            // is already a complete run of `bytes` is borrowed, not copied.
2923            // The two indexes differ only in how a chunk is addressed: EA by
2924            // its linear grid index, FA by grid coordinates.
2925            const BATCH_WINDOW: u64 = 256;
2926            let mut start = 0u64;
2927            while start < total_chunks {
2928                let end = (start + BATCH_WINDOW).min(total_chunks);
2929                let items: Vec<(Vec<u64>, Cow<'_, [u8]>)> = (start..end)
2930                    .map(|counter| {
2931                        let coords = coords_of(counter);
2932                        let data = match Self::contiguous_chunk_span(
2933                            &dims,
2934                            &chunk_dims,
2935                            &coords,
2936                            element_size,
2937                        ) {
2938                            Some(span) => Cow::Borrowed(&bytes[span]),
2939                            None => {
2940                                let mut buf = Vec::new();
2941                                Self::gather_chunk_into(
2942                                    &mut buf,
2943                                    bytes,
2944                                    &dims,
2945                                    &chunk_dims,
2946                                    &coords,
2947                                    element_size,
2948                                );
2949                                Cow::Owned(buf)
2950                            }
2951                        };
2952                        (coords, data)
2953                    })
2954                    .collect();
2955                if kind == ChunkIndexKind::FixedArray {
2956                    let pairs: Vec<(&[u64], &[u8])> = items
2957                        .iter()
2958                        .map(|(c, d)| (c.as_slice(), d.as_ref()))
2959                        .collect();
2960                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
2961                } else {
2962                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
2963                    for (c, d) in &items {
2964                        pairs.push((writer.chunk_slot(index, c)?, d.as_ref()));
2965                    }
2966                    writer.write_chunks_batch_inner(index, &pairs)?;
2967                }
2968                start = end;
2969            }
2970        }
2971        Ok(())
2972    }
2973
2974    /// The byte range one chunk occupies in a row-major full-dataset image,
2975    /// for a chunk that needs no gather at all: its elements are one
2976    /// contiguous run of `source` *and* they fill the chunk shape exactly, so
2977    /// the bytes that go to the file are already sitting in the caller's
2978    /// buffer.
2979    ///
2980    /// Both halves hold when every dimension after the first spans the whole
2981    /// dataset (`chunk_dims[d] == dims[d]`, leaving nothing interleaved and no
2982    /// padding along those axes) and the chunk does not hang off the far edge
2983    /// of the first — which is every full chunk of a 1-D dataset. `None` means
2984    /// the chunk has to be gathered.
2985    fn contiguous_chunk_span(
2986        dims: &[u64],
2987        chunk_dims: &[u64],
2988        coords: &[u64],
2989        element_size: usize,
2990    ) -> Option<std::ops::Range<usize>> {
2991        let rank = dims.len();
2992        if rank == 0 || chunk_dims[1..] != dims[1..] {
2993            return None;
2994        }
2995        if (coords[0] + 1) * chunk_dims[0] > dims[0] {
2996            return None;
2997        }
2998        let plane: u64 = dims[1..].iter().product::<u64>() * element_size as u64;
2999        let start = usize::try_from(coords[0] * chunk_dims[0] * plane).ok()?;
3000        let len = usize::try_from(chunk_dims[0] * plane).ok()?;
3001        Some(start..start.checked_add(len)?)
3002    }
3003
3004    /// Gather one chunk's bytes from a row-major full-dataset image into
3005    /// `out`, replacing whatever it held.
3006    ///
3007    /// `coords` are the chunk's grid coordinates. `out` is left exactly
3008    /// `product(chunk_dims) * element_size` bytes long, holding the chunk's
3009    /// elements and zero where the chunk extends past the dataset edge — so a
3010    /// caller may hand the same buffer to one chunk after another.
3011    fn gather_chunk_into(
3012        out: &mut Vec<u8>,
3013        source: &[u8],
3014        dims: &[u64],
3015        chunk_dims: &[u64],
3016        coords: &[u64],
3017        element_size: usize,
3018    ) {
3019        let rank = dims.len();
3020        let chunk_elems: u64 = chunk_dims.iter().product();
3021        let chunk_bytes = chunk_elems as usize * element_size;
3022        if rank == 0 {
3023            // Scalar dataset: a single element, no chunking dimension.
3024            out.clear();
3025            out.resize(chunk_bytes, 0);
3026            if source.len() >= element_size {
3027                out[..element_size].copy_from_slice(&source[..element_size]);
3028            }
3029            return;
3030        }
3031
3032        // Actual extent of this chunk along each dimension (edge chunks are
3033        // smaller than the nominal chunk shape).
3034        let mut extent = vec![0u64; rank];
3035        for d in 0..rank {
3036            let start = coords[d] * chunk_dims[d];
3037            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
3038            extent[d] = end.saturating_sub(start);
3039        }
3040        // Size the buffer, then zero it only when this chunk leaves part of
3041        // its shape uncovered: a full chunk has every byte overwritten below,
3042        // while an edge chunk's padding must read as zero even though a
3043        // reused buffer still holds the previous chunk's bytes.
3044        if out.len() != chunk_bytes {
3045            out.clear();
3046            out.resize(chunk_bytes, 0);
3047        } else if extent != chunk_dims {
3048            out.fill(0);
3049        }
3050        if extent.contains(&0) {
3051            return; // nothing of the dataset falls in this chunk
3052        }
3053
3054        // Row-major strides (in elements) for the source (over `dims`) and the
3055        // destination chunk buffer (over `chunk_dims`).
3056        let mut src_stride = vec![1u64; rank];
3057        let mut dst_stride = vec![1u64; rank];
3058        for d in (0..rank - 1).rev() {
3059            src_stride[d] = src_stride[d + 1] * dims[d + 1];
3060            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
3061        }
3062
3063        // Copy one contiguous run along the last axis per outer multi-index.
3064        let last = rank - 1;
3065        let run = extent[last] as usize * element_size;
3066        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
3067        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
3068        for _ in 0..outer {
3069            let mut src_off = 0u64;
3070            let mut dst_off = 0u64;
3071            for d in 0..rank {
3072                let global = coords[d] * chunk_dims[d] + idx[d];
3073                src_off += global * src_stride[d];
3074                dst_off += idx[d] * dst_stride[d];
3075            }
3076            let s = src_off as usize * element_size;
3077            let dpos = dst_off as usize * element_size;
3078            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
3079
3080            // Advance the multi-index over axes [0..last); the last axis is the
3081            // contiguous run handled above.
3082            let mut d = last;
3083            while d > 0 {
3084                d -= 1;
3085                idx[d] += 1;
3086                if idx[d] < extent[d] {
3087                    break;
3088                }
3089                idx[d] = 0;
3090            }
3091        }
3092    }
3093
3094    /// Write a single chunk to a chunked dataset.
3095    ///
3096    /// `chunk_idx` is the linear chunk index (typically the frame number for
3097    /// streaming datasets). `data` is the raw byte data for one chunk.
3098    ///
3099    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
3100    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
3101    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
3102        match &self.info {
3103            DatasetInfo::Writer {
3104                index, chunk_index, ..
3105            } => {
3106                let Some(kind) = *chunk_index else {
3107                    return Err(Hdf5Error::InvalidState(
3108                        "write_chunk is only for chunked datasets".into(),
3109                    ));
3110                };
3111                if kind == ChunkIndexKind::BtreeV2 {
3112                    return Err(Hdf5Error::InvalidState(
3113                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
3114                         with the chunk's grid coordinates"
3115                            .into(),
3116                    ));
3117                }
3118
3119                let inner = borrow_inner(&self.file_inner);
3120                match &*inner {
3121                    H5FileInner::Writer(writer) => {
3122                        // One op: the slot decode and the write see the same
3123                        // extents.
3124                        let cell = writer.ds(*index);
3125                        let _op = cell.op.lock();
3126                        match kind {
3127                            // All four address a chunk by its grid
3128                            // coordinates, so the linear slot is decoded back
3129                            // into them.
3130                            ChunkIndexKind::FixedArray
3131                            | ChunkIndexKind::Implicit
3132                            | ChunkIndexKind::SingleChunk
3133                            | ChunkIndexKind::BtreeV1 => {
3134                                let coords =
3135                                    writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3136                                writer.write_chunk_at_coords(*index, &coords, data)?;
3137                            }
3138                            _ => writer.write_chunk_inner(*index, chunk_idx as u64, data)?,
3139                        }
3140                        Ok(())
3141                    }
3142                    _ => Err(Hdf5Error::InvalidState(
3143                        "file is no longer in write mode".into(),
3144                    )),
3145                }
3146            }
3147            DatasetInfo::Reader { .. } => {
3148                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3149            }
3150        }
3151    }
3152
3153    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
3154    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
3155    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
3156    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
3157    ///
3158    /// `chunk_idx` is the linear chunk index (the frame number for streaming
3159    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
3160    /// the already-filtered bytes of one chunk — its length is the *stored*
3161    /// (compressed) size, not the uncompressed chunk size.
3162    ///
3163    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
3164    /// dataset's pipeline was **not** applied to this chunk and must be skipped
3165    /// on read. Pass 0 when the full pipeline was already applied upstream (the
3166    /// common case: a codec plugin handed you compressed frames).
3167    ///
3168    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3169    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
3170    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
3171    /// against, so address its chunks with
3172    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
3173    ///
3174    /// # Reading back
3175    ///
3176    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
3177    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
3178    /// the reader skipping exactly the filters the mask marks as not applied.
3179    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
3180        match &self.info {
3181            DatasetInfo::Writer {
3182                index, chunk_index, ..
3183            } => {
3184                let Some(kind) = *chunk_index else {
3185                    return Err(Hdf5Error::InvalidState(
3186                        "write_chunk_raw is only for chunked datasets".into(),
3187                    ));
3188                };
3189                if kind == ChunkIndexKind::BtreeV2 {
3190                    return Err(Hdf5Error::InvalidState(
3191                        "this dataset uses a v2 B-tree chunk index; use \
3192                         write_chunk_raw_at with the chunk's grid coordinates"
3193                            .into(),
3194                    ));
3195                }
3196                if kind == ChunkIndexKind::Implicit {
3197                    return Err(Hdf5Error::InvalidState(
3198                        "this dataset uses the implicit chunk index, which stores \
3199                         every chunk at its full unfiltered size and has nowhere to \
3200                         record a stored size or a filter mask"
3201                            .into(),
3202                    ));
3203                }
3204
3205                let inner = borrow_inner(&self.file_inner);
3206                match &*inner {
3207                    H5FileInner::Writer(writer) => {
3208                        // One op: the slot decode and the write see the same
3209                        // extents.
3210                        let cell = writer.ds(*index);
3211                        let _op = cell.op.lock();
3212                        if kind == ChunkIndexKind::FixedArray {
3213                            // Fixed-array dataset: decode the index-grid slot
3214                            // into row-major grid coordinates.
3215                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3216                            writer.write_compressed_chunk_fixed_array_inner(
3217                                *index,
3218                                &coords,
3219                                data,
3220                                filter_mask,
3221                            )?;
3222                        } else if kind == ChunkIndexKind::BtreeV1 {
3223                            // Same for the classic index, whose key carries a
3224                            // stored size and a filter mask of its own.
3225                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3226                            writer.write_compressed_chunk_btree_v1_inner(
3227                                *index,
3228                                &coords,
3229                                data,
3230                                filter_mask,
3231                            )?;
3232                        } else if kind == ChunkIndexKind::SingleChunk {
3233                            // Same again for the single-chunk index, whose
3234                            // layout message carries the stored size and mask
3235                            // inline.
3236                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3237                            writer.write_compressed_chunk_single_chunk_inner(
3238                                *index,
3239                                &coords,
3240                                data,
3241                                filter_mask,
3242                            )?;
3243                        } else {
3244                            writer.write_compressed_chunk_inner(
3245                                *index,
3246                                chunk_idx as u64,
3247                                data,
3248                                filter_mask,
3249                            )?;
3250                        }
3251                        Ok(())
3252                    }
3253                    _ => Err(Hdf5Error::InvalidState(
3254                        "file is no longer in write mode".into(),
3255                    )),
3256                }
3257            }
3258            DatasetInfo::Reader { .. } => {
3259                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3260            }
3261        }
3262    }
3263
3264    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
3265    /// chunk-grid coordinates (one per dimension).
3266    ///
3267    /// This is the entry point for datasets with two or more unlimited
3268    /// dimensions. The dataset's logical dimensions are extended to cover
3269    /// the written chunk. `data` is the raw bytes of one full chunk.
3270    ///
3271    /// ```no_run
3272    /// # use rust_hdf5::H5File;
3273    /// let file = H5File::create("bt2.h5").unwrap();
3274    /// let ds = file.new_dataset::<i32>()
3275    ///     .shape(&[0, 0])
3276    ///     .chunk(&[2, 2])
3277    ///     .max_shape(&[None, None])
3278    ///     .create("grid")
3279    ///     .unwrap();
3280    /// let chunk = [0i32, 1, 2, 3];
3281    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
3282    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3283    /// ```
3284    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
3285        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
3286    }
3287
3288    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
3289    /// addressed by its chunk-grid coordinates.
3290    ///
3291    /// The coordinate-addressed twin of
3292    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
3293    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
3294    /// there is no fixed chunk grid for a linear index to mean anything against.
3295    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
3296    /// to cover the written chunk.
3297    ///
3298    /// `data` is the already-filtered bytes of one chunk — its length is the
3299    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
3300    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
3301    /// full pipeline already ran upstream.
3302    ///
3303    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3304    /// has no slot to record a stored size or mask.
3305    pub fn write_chunk_raw_at(
3306        &self,
3307        chunk_coords: &[usize],
3308        data: &[u8],
3309        filter_mask: u32,
3310    ) -> Result<()> {
3311        self.write_chunk_at_inner(
3312            chunk_coords,
3313            ChunkBytes::Prefiltered { data, filter_mask },
3314            "write_chunk_raw_at",
3315        )
3316    }
3317
3318    /// The single owner of coordinate-addressed chunk writes: validates the
3319    /// coordinates, grows the dataspace to cover them, and routes the bytes to
3320    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
3321    /// here or already ran upstream is carried by `bytes`, not by a second copy
3322    /// of this dispatch.
3323    fn write_chunk_at_inner(
3324        &self,
3325        chunk_coords: &[usize],
3326        bytes: ChunkBytes<'_>,
3327        what: &str,
3328    ) -> Result<()> {
3329        match &self.info {
3330            DatasetInfo::Writer {
3331                index, chunk_index, ..
3332            } => {
3333                let Some(kind) = *chunk_index else {
3334                    return Err(Hdf5Error::InvalidState(format!(
3335                        "{what} is only for chunked datasets"
3336                    )));
3337                };
3338                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3339                let inner = borrow_inner(&self.file_inner);
3340                let writer = match &*inner {
3341                    H5FileInner::Writer(w) => w,
3342                    _ => {
3343                        return Err(Hdf5Error::InvalidState(
3344                            "file is no longer in write mode".into(),
3345                        ))
3346                    }
3347                };
3348                // Whole-operation guard: the dims snapshot, the chunk write
3349                // and the extend below must not interleave with a concurrent
3350                // same-dataset operation.
3351                let cell = writer.ds(*index);
3352                let _op = cell.op.lock();
3353                let chunk_dims = writer
3354                    .dataset_chunk_dims(*index)
3355                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3356                    .to_vec();
3357                let dims = writer.dataset_dims(*index).to_vec();
3358                if coords.len() != dims.len() {
3359                    return Err(Hdf5Error::InvalidState(format!(
3360                        "chunk_coords has {} entries but the dataset has {} dimensions",
3361                        coords.len(),
3362                        dims.len()
3363                    )));
3364                }
3365                if chunk_dims.len() != dims.len() {
3366                    return Err(Hdf5Error::InvalidState(format!(
3367                        "dataset chunk shape has {} dimensions but the dataspace has {}",
3368                        chunk_dims.len(),
3369                        dims.len()
3370                    )));
3371                }
3372
3373                if kind == ChunkIndexKind::FixedArray {
3374                    // Fixed-array (fixed-shape) dataset: no dimension growth.
3375                    match bytes {
3376                        ChunkBytes::Unfiltered(data) => {
3377                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
3378                        }
3379                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3380                            .write_compressed_chunk_fixed_array_inner(
3381                                *index,
3382                                &coords,
3383                                data,
3384                                filter_mask,
3385                            )?,
3386                    }
3387                    return Ok(());
3388                }
3389
3390                if kind == ChunkIndexKind::Implicit {
3391                    // Implicit index: fixed shape, so no dimension growth
3392                    // either, and no slot to record a stored size in.
3393                    match bytes {
3394                        ChunkBytes::Unfiltered(data) => {
3395                            writer.write_chunk_implicit_inner(*index, &coords, data)?
3396                        }
3397                        ChunkBytes::Prefiltered { .. } => {
3398                            return Err(Hdf5Error::InvalidState(
3399                                "this dataset uses the implicit chunk index, which stores \
3400                                 every chunk at its full unfiltered size and has nowhere \
3401                                 to record a stored size or a filter mask"
3402                                    .into(),
3403                            ))
3404                        }
3405                    }
3406                    return Ok(());
3407                }
3408
3409                if kind == ChunkIndexKind::SingleChunk {
3410                    // Single-chunk index: fixed shape covered by exactly one
3411                    // chunk, so no dimension growth either.
3412                    match bytes {
3413                        ChunkBytes::Unfiltered(data) => {
3414                            writer.write_chunk_single_chunk_inner(*index, &coords, data)?
3415                        }
3416                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3417                            .write_compressed_chunk_single_chunk_inner(
3418                                *index,
3419                                &coords,
3420                                data,
3421                                filter_mask,
3422                            )?,
3423                    }
3424                    return Ok(());
3425                }
3426
3427                // The remaining indexes (v2 B-tree, v1 B-tree, extensible
3428                // array) can all grow: validate the coordinates and compute
3429                // the grown dimensions up-front, before any chunk is
3430                // written, so an overflowing coordinate cannot leave an
3431                // orphaned chunk in the file.
3432                //
3433                // The last chunk of a dimension usually hangs past the extent
3434                // — a length of 10 in chunks of 4 ends at 12 — so the growth
3435                // is capped at the declared maximum, which is what the chunk
3436                // still covers. Without the cap a legal edge chunk would be
3437                // written and then rejected by the extend below.
3438                let max_dims = writer.dataset_max_dims(*index);
3439                let mut new_dims = dims.clone();
3440                for d in 0..dims.len() {
3441                    let needed = coords[d]
3442                        .checked_add(1)
3443                        .and_then(|c| c.checked_mul(chunk_dims[d]))
3444                        .ok_or_else(|| {
3445                            Hdf5Error::InvalidState(format!(
3446                                "chunk coordinate {} in dimension {} is too large",
3447                                coords[d], d
3448                            ))
3449                        })?;
3450                    let needed = needed.min(max_dims[d]);
3451                    if needed > new_dims[d] {
3452                        new_dims[d] = needed;
3453                    }
3454                }
3455
3456                if kind == ChunkIndexKind::BtreeV2 {
3457                    match bytes {
3458                        ChunkBytes::Unfiltered(data) => {
3459                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
3460                        }
3461                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3462                            .write_compressed_chunk_btree_v2_inner(
3463                                *index,
3464                                &coords,
3465                                data,
3466                                filter_mask,
3467                            )?,
3468                    }
3469                } else if kind == ChunkIndexKind::BtreeV1 {
3470                    // The classic index takes any shape, fixed or unlimited,
3471                    // so it grows the dataspace with the chunk the way the v2
3472                    // B-tree does — bounded below by the maximum extent.
3473                    match bytes {
3474                        ChunkBytes::Unfiltered(data) => {
3475                            writer.write_chunk_btree_v1_inner(*index, &coords, data)?
3476                        }
3477                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3478                            .write_compressed_chunk_btree_v1_inner(
3479                                *index,
3480                                &coords,
3481                                data,
3482                                filter_mask,
3483                            )?,
3484                    }
3485                } else {
3486                    // Extensible array: the chunk's index-grid slot (row-major
3487                    // against the maximum extent).
3488                    let linear = writer.chunk_slot(*index, &coords)?;
3489                    match bytes {
3490                        ChunkBytes::Unfiltered(data) => {
3491                            writer.write_chunk_inner(*index, linear, data)?
3492                        }
3493                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3494                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
3495                    }
3496                }
3497
3498                if new_dims != dims {
3499                    writer.extend_dataset_inner(*index, &new_dims)?;
3500                }
3501                Ok(())
3502            }
3503            DatasetInfo::Reader { .. } => {
3504                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3505            }
3506        }
3507    }
3508
3509    /// Write multiple chunks in a batch, optionally compressing in parallel.
3510    ///
3511    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
3512    /// pipeline is configured and the `parallel` feature is enabled, all
3513    /// chunks are compressed concurrently via rayon.
3514    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
3515        match &self.info {
3516            DatasetInfo::Writer {
3517                index, chunk_index, ..
3518            } => {
3519                if chunk_index.is_none() {
3520                    return Err(Hdf5Error::InvalidState(
3521                        "write_chunks_batch is only for chunked datasets".into(),
3522                    ));
3523                }
3524                let pairs: Vec<(u64, &[u8])> = chunks
3525                    .iter()
3526                    .map(|(idx, data)| (*idx as u64, *data))
3527                    .collect();
3528                let inner = borrow_inner(&self.file_inner);
3529                match &*inner {
3530                    H5FileInner::Writer(writer) => {
3531                        writer.write_chunks_batch(*index, &pairs)?;
3532                        Ok(())
3533                    }
3534                    _ => Err(Hdf5Error::InvalidState(
3535                        "file is no longer in write mode".into(),
3536                    )),
3537                }
3538            }
3539            DatasetInfo::Reader { .. } => {
3540                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3541            }
3542        }
3543    }
3544
3545    /// Append data along the first dimension of a chunked dataset.
3546    ///
3547    /// `data` must contain a whole number of "frames" — slices along
3548    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
3549    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
3550    /// multiple of `H * W`.
3551    ///
3552    /// This method writes the necessary chunks and extends the dataset
3553    /// shape automatically.
3554    ///
3555    /// ```no_run
3556    /// # use rust_hdf5::H5File;
3557    /// let file = H5File::create("append.h5").unwrap();
3558    /// let ds = file.new_dataset::<f64>()
3559    ///     .shape(&[0, 3])
3560    ///     .chunk(&[1, 3])
3561    ///     .max_shape(&[None, Some(3)])
3562    ///     .create("data")
3563    ///     .unwrap();
3564    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
3565    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
3566    /// ```
3567    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
3568        match &self.info {
3569            DatasetInfo::Writer {
3570                index,
3571                element_size,
3572                chunk_index,
3573                ..
3574            } => {
3575                if chunk_index.is_none() {
3576                    return Err(Hdf5Error::InvalidState(
3577                        "append is only for chunked datasets".into(),
3578                    ));
3579                }
3580                if T::element_size() != *element_size {
3581                    return Err(Hdf5Error::TypeMismatch(format!(
3582                        "append type has element size {} but dataset expects {}",
3583                        T::element_size(),
3584                        element_size,
3585                    )));
3586                }
3587
3588                let ds_index = *index;
3589                let es = *element_size;
3590
3591                let inner = borrow_inner(&self.file_inner);
3592                let writer = match &*inner {
3593                    H5FileInner::Writer(w) => w,
3594                    _ => {
3595                        return Err(Hdf5Error::InvalidState(
3596                            "file is no longer in write mode".into(),
3597                        ))
3598                    }
3599                };
3600
3601                // Whole-operation guard: the buffer take, the frame writes,
3602                // the re-buffer and the extend below are separate slot
3603                // acquisitions that a concurrent same-dataset append must not
3604                // interleave with.
3605                let cell = writer.ds(ds_index);
3606                let _op = cell.op.lock();
3607
3608                let chunk_dims = writer
3609                    .dataset_chunk_dims(ds_index)
3610                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3611                    .to_vec();
3612                let dims = writer.dataset_dims(ds_index).to_vec();
3613
3614                // Frame size = product of dims[1..]
3615                let frame_elems: usize = if dims.len() > 1 {
3616                    dims[1..].iter().map(|&d| d as usize).product()
3617                } else {
3618                    1
3619                };
3620
3621                if frame_elems == 0 {
3622                    return Err(Hdf5Error::InvalidState(
3623                        "cannot append to dataset with zero-size trailing dimensions".into(),
3624                    ));
3625                }
3626
3627                if !data.len().is_multiple_of(frame_elems) {
3628                    return Err(Hdf5Error::InvalidState(format!(
3629                        "data length {} is not a multiple of frame size {}",
3630                        data.len(),
3631                        frame_elems,
3632                    )));
3633                }
3634
3635                let n_new_frames = data.len() / frame_elems;
3636                let current_dim0 = dims[0] as usize;
3637
3638                // Chunk size along first dimension
3639                let chunk_dim0 = chunk_dims[0] as usize;
3640                let frame_bytes = frame_elems * es;
3641
3642                let host = unsafe {
3643                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
3644                };
3645                let datatype = writer.dataset_datatype(ds_index);
3646                let raw = to_stored_byte_order(host, &datatype, es)?;
3647
3648                // Merge the buffer with the new frames when it is the
3649                // dataset's tail; a buffer left mid-extent (the extent moved
3650                // past it) keeps its recorded place — flush it and start
3651                // fresh at the current end.
3652                let taken = { writer.ds(ds_index).lock().append.take() };
3653                let (base_dim0, buffered_frames, mut combined) = match taken {
3654                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
3655                        (b.base as usize, b.frames as usize, b.bytes)
3656                    }
3657                    Some(b) => {
3658                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
3659                        (current_dim0, 0, Vec::new())
3660                    }
3661                    None => (current_dim0, 0, Vec::new()),
3662                };
3663                combined.extend_from_slice(&raw);
3664
3665                let total_frames = buffered_frames + n_new_frames;
3666
3667                // Rows up to the last chunk boundary are written now; the
3668                // tail that does not complete a chunk goes back in the
3669                // buffer for the next append (or the flush at close). The
3670                // boundary can precede `base_dim0` — a reopened file's
3671                // flushed partial chunk leaves the base mid-chunk — in
3672                // which case everything is tail.
3673                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
3674                let write_frames = last_boundary.saturating_sub(base_dim0);
3675                let tail_frames = total_frames - write_frames;
3676                if write_frames > 0 {
3677                    writer.write_append_frames(
3678                        ds_index,
3679                        base_dim0 as u64,
3680                        write_frames as u64,
3681                        &combined[..write_frames * frame_bytes],
3682                    )?;
3683                }
3684                if tail_frames > 0 {
3685                    let ds = writer.ds(ds_index);
3686                    let mut m = ds.lock();
3687                    m.append = Some(crate::io::writer::AppendBuffer {
3688                        base: (base_dim0 + write_frames) as u64,
3689                        frames: tail_frames as u64,
3690                        bytes: combined[write_frames * frame_bytes..].to_vec(),
3691                    });
3692                }
3693
3694                // Extend dims to include all frames (buffered + new)
3695                let logical_dim0 = base_dim0 + total_frames;
3696                let mut new_dims: Vec<u64> = dims;
3697                new_dims[0] = logical_dim0 as u64;
3698                writer.extend_dataset_inner(ds_index, &new_dims)?;
3699
3700                Ok(())
3701            }
3702            DatasetInfo::Reader { .. } => {
3703                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
3704            }
3705        }
3706    }
3707
3708    /// Extend the dimensions of a chunked dataset.
3709    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
3710        match &self.info {
3711            DatasetInfo::Writer {
3712                index, chunk_index, ..
3713            } => {
3714                if chunk_index.is_none() {
3715                    return Err(Hdf5Error::InvalidState(
3716                        "extend is only for chunked datasets".into(),
3717                    ));
3718                }
3719
3720                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3721                let inner = borrow_inner(&self.file_inner);
3722                match &*inner {
3723                    H5FileInner::Writer(writer) => {
3724                        writer.extend_dataset(*index, &dims_u64)?;
3725                        Ok(())
3726                    }
3727                    _ => Err(Hdf5Error::InvalidState(
3728                        "file is no longer in write mode".into(),
3729                    )),
3730                }
3731            }
3732            DatasetInfo::Reader { .. } => {
3733                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
3734            }
3735        }
3736    }
3737
3738    /// Set the logical extent of a chunked dataset, growing **or
3739    /// shrinking** any dimension.
3740    ///
3741    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
3742    /// dimension — for example to correct an over-extended frame count
3743    /// after writing a partial multi-frame chunk. Shrinking prunes the
3744    /// stored chunks the way libhdf5's `H5Dset_extent` does: a chunk
3745    /// entirely beyond the new extent is removed from the chunk index and
3746    /// its storage freed for reuse, and a chunk the new extent cuts
3747    /// through has its out-of-extent region overwritten with the fill
3748    /// value — so growing the extent back exposes fill values, not the
3749    /// old data. The new extent must not exceed the dataset's maximum
3750    /// dimensions.
3751    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
3752        match &self.info {
3753            DatasetInfo::Writer { index, .. } => {
3754                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3755                let inner = borrow_inner(&self.file_inner);
3756                match &*inner {
3757                    H5FileInner::Writer(writer) => {
3758                        writer.set_dataset_extent(*index, &dims_u64)?;
3759                        Ok(())
3760                    }
3761                    _ => Err(Hdf5Error::InvalidState(
3762                        "file is no longer in write mode".into(),
3763                    )),
3764                }
3765            }
3766            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
3767                "cannot set extent in read mode".into(),
3768            )),
3769        }
3770    }
3771
3772    /// Flush a chunked dataset's index structures to disk.
3773    pub fn flush(&self) -> Result<()> {
3774        match &self.info {
3775            DatasetInfo::Writer { index, .. } => {
3776                let inner = borrow_inner(&self.file_inner);
3777                match &*inner {
3778                    H5FileInner::Writer(writer) => {
3779                        writer.flush_dataset(*index)?;
3780                        Ok(())
3781                    }
3782                    _ => Ok(()),
3783                }
3784            }
3785            DatasetInfo::Reader { .. } => Ok(()),
3786        }
3787    }
3788
3789    /// Read a slice (hyperslab) of the dataset as a typed vector.
3790    ///
3791    /// `starts` and `counts` define the N-dimensional selection:
3792    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
3793    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
3794        match &self.info {
3795            DatasetInfo::Reader {
3796                name,
3797                shape,
3798                element_size,
3799            } => {
3800                if T::element_size() != *element_size {
3801                    return Err(Hdf5Error::TypeMismatch(format!(
3802                        "read type has element size {} but dataset has element size {}",
3803                        T::element_size(),
3804                        element_size,
3805                    )));
3806                }
3807                let datatype = self.datatype()?;
3808                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
3809                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
3810
3811                // Bounds before sizing: the destination is allocated here,
3812                // ahead of the reader's own check, so a selection the extent
3813                // does not admit must be refused before its size is computed.
3814                let dims: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
3815                check_hyperslab(&dims, &starts_u64, &counts_u64)?;
3816                let count = element_count(&counts_u64)?;
3817                let mut inner = borrow_inner_mut(&self.file_inner);
3818                let H5FileInner::Reader(reader) = &mut *inner else {
3819                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3820                };
3821                // The selection lands in the vector this returns, so its bytes
3822                // are touched once instead of being read into a byte buffer and
3823                // copied into a second one of the same size.
3824                read_image_into_new(count, |image| {
3825                    reader.read_slice_into_dst(
3826                        name,
3827                        &starts_u64,
3828                        &counts_u64,
3829                        image,
3830                        ReadDst::Fresh,
3831                    )?;
3832                    to_host_byte_order(image, &datatype, T::element_size())
3833                })
3834            }
3835            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3836                "cannot read_slice from a dataset in write mode".into(),
3837            )),
3838        }
3839    }
3840
3841    /// Read a strided hyperslab as a typed vector — h5py's stepped slicing
3842    /// (`ds[a:b:s]`) or the general `start`/`stride`/`count`/`block` form of
3843    /// `H5Sselect_hyperslab`.
3844    ///
3845    /// One entry per dimension: `start[d]` is the first index, `stride[d]`
3846    /// the spacing between selected blocks (all-`1` is the same selection
3847    /// [`read_slice`](Self::read_slice) reads), `count[d]` how many blocks,
3848    /// and `block[d]` how many contiguous elements each block covers. The
3849    /// returned vector is row-major over `count[d] * block[d]` per
3850    /// dimension — exactly the shape h5py's stepped slicing produces.
3851    ///
3852    /// ```no_run
3853    /// # use rust_hdf5::H5File;
3854    /// let file = H5File::open("data.h5").unwrap();
3855    /// let ds = file.dataset("series").unwrap(); // shape [100]
3856    /// // Python: ds[0:100:2] — every other element.
3857    /// let evens: Vec<f64> = ds.read_hyperslab(&[0], &[2], &[50], &[1]).unwrap();
3858    /// ```
3859    pub fn read_hyperslab<T: H5Type>(
3860        &self,
3861        start: &[usize],
3862        stride: &[usize],
3863        count: &[usize],
3864        block: &[usize],
3865    ) -> Result<Vec<T>> {
3866        match &self.info {
3867            DatasetInfo::Reader {
3868                name, element_size, ..
3869            } => {
3870                if T::element_size() != *element_size {
3871                    return Err(Hdf5Error::TypeMismatch(format!(
3872                        "read type has element size {} but dataset has element size {}",
3873                        T::element_size(),
3874                        element_size,
3875                    )));
3876                }
3877                let datatype = self.datatype()?;
3878                let start_u64: Vec<u64> = start.iter().map(|&s| s as u64).collect();
3879                let stride_u64: Vec<u64> = stride.iter().map(|&s| s as u64).collect();
3880                let count_u64: Vec<u64> = count.iter().map(|&c| c as u64).collect();
3881                let block_u64: Vec<u64> = block.iter().map(|&b| b as u64).collect();
3882
3883                let selected: Vec<u64> = count_u64
3884                    .iter()
3885                    .zip(&block_u64)
3886                    .map(|(&c, &b)| c.saturating_mul(b))
3887                    .collect();
3888                let n = element_count(&selected)?;
3889                let mut inner = borrow_inner_mut(&self.file_inner);
3890                let H5FileInner::Reader(reader) = &mut *inner else {
3891                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3892                };
3893                read_image_into_new(n, |image| {
3894                    reader.read_hyperslab_into(
3895                        name,
3896                        &start_u64,
3897                        &stride_u64,
3898                        &count_u64,
3899                        &block_u64,
3900                        image,
3901                    )?;
3902                    to_host_byte_order(image, &datatype, T::element_size())
3903                })
3904            }
3905            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3906                "cannot read_hyperslab from a dataset in write mode".into(),
3907            )),
3908        }
3909    }
3910
3911    /// Read a list of coordinates in one call, as a typed vector — h5py
3912    /// fancy indexing with a coordinate list.
3913    ///
3914    /// `points[i]` is a coordinate with one entry per dimension. The
3915    /// returned vector holds one element per point, in the same order as
3916    /// `points`, regardless of the dataset's rank.
3917    ///
3918    /// ```no_run
3919    /// # use rust_hdf5::H5File;
3920    /// let file = H5File::open("data.h5").unwrap();
3921    /// let ds = file.dataset("grid").unwrap(); // shape [10, 10]
3922    /// // Python: ds[np.array([[0, 0], [3, 4], [9, 9]])]
3923    /// let picked: Vec<f64> = ds.read_points(&[vec![0, 0], vec![3, 4], vec![9, 9]]).unwrap();
3924    /// ```
3925    pub fn read_points<T: H5Type>(&self, points: &[Vec<usize>]) -> Result<Vec<T>> {
3926        match &self.info {
3927            DatasetInfo::Reader {
3928                name, element_size, ..
3929            } => {
3930                if T::element_size() != *element_size {
3931                    return Err(Hdf5Error::TypeMismatch(format!(
3932                        "read type has element size {} but dataset has element size {}",
3933                        T::element_size(),
3934                        element_size,
3935                    )));
3936                }
3937                let datatype = self.datatype()?;
3938                let points_u64: Vec<Vec<u64>> = points
3939                    .iter()
3940                    .map(|p| p.iter().map(|&c| c as u64).collect())
3941                    .collect();
3942
3943                let mut inner = borrow_inner_mut(&self.file_inner);
3944                let H5FileInner::Reader(reader) = &mut *inner else {
3945                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3946                };
3947                read_image_into_new(points_u64.len(), |image| {
3948                    reader.read_points_into(name, &points_u64, image, ReadDst::Fresh)?;
3949                    to_host_byte_order(image, &datatype, T::element_size())
3950                })
3951            }
3952            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3953                "cannot read_points from a dataset in write mode".into(),
3954            )),
3955        }
3956    }
3957
3958    /// Read one chunk's raw (still-filtered) bytes and its filter mask,
3959    /// addressed by chunk-grid coordinates — the read half of
3960    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) and the HDF5 "direct
3961    /// chunk read" (`H5Dread_chunk`, formerly `H5DOread_chunk`; h5py's
3962    /// `Dataset.id.read_direct_chunk`).
3963    ///
3964    /// The bytes are exactly what is stored on disk: filtered/compressed if
3965    /// the dataset has a filter pipeline, with no decompression applied. The
3966    /// returned `u32` is the chunk's filter mask: bit *i* set means filter
3967    /// *i* of the pipeline was **not** applied to this particular chunk and
3968    /// must be skipped when reversing it.
3969    ///
3970    /// `Err` if the dataset is not chunked, `chunk_coords` has the wrong
3971    /// rank, or the chunk at those coordinates has never been written.
3972    ///
3973    /// ```no_run
3974    /// # use rust_hdf5::H5File;
3975    /// let file = H5File::open("data.h5").unwrap();
3976    /// let ds = file.dataset("frames").unwrap();
3977    /// let (raw, filter_mask) = ds.read_chunk_raw_at(&[0, 0]).unwrap();
3978    /// ```
3979    pub fn read_chunk_raw_at(&self, chunk_coords: &[usize]) -> Result<(Vec<u8>, u32)> {
3980        match &self.info {
3981            DatasetInfo::Reader { name, .. } => {
3982                let coords_u64: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3983                let mut inner = borrow_inner_mut(&self.file_inner);
3984                match &mut *inner {
3985                    H5FileInner::Reader(reader) => Ok(reader.read_chunk_raw_at(name, &coords_u64)?),
3986                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
3987                }
3988            }
3989            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3990                "cannot read_chunk_raw_at from a dataset in write mode".into(),
3991            )),
3992        }
3993    }
3994
3995    /// Write a typed slice to a sub-region of the dataset.
3996    ///
3997    /// `starts` and `counts` define the N-dimensional selection, which must lie
3998    /// inside the dataset's current extent.
3999    ///
4000    /// Works for both contiguous and chunked datasets. For a chunked dataset
4001    /// only the chunks the selection touches are rewritten — a partially
4002    /// covered chunk is read back, patched, and written again, so updating one
4003    /// row of an appendable dataset costs the chunks that row crosses rather
4004    /// than the whole dataset. Elements of a touched chunk that the selection
4005    /// does not cover keep their stored value, or the dataset's fill value if
4006    /// the chunk did not exist yet.
4007    pub fn write_slice<T: H5Type>(
4008        &self,
4009        starts: &[usize],
4010        counts: &[usize],
4011        data: &[T],
4012    ) -> Result<()> {
4013        match &self.info {
4014            DatasetInfo::Writer {
4015                index,
4016                element_size,
4017                ..
4018            } => {
4019                if T::element_size() != *element_size {
4020                    return Err(Hdf5Error::TypeMismatch(format!(
4021                        "write type has element size {} but dataset expects {}",
4022                        T::element_size(),
4023                        element_size,
4024                    )));
4025                }
4026
4027                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4028                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4029
4030                let inner = borrow_inner(&self.file_inner);
4031                let H5FileInner::Writer(writer) = &*inner else {
4032                    return Err(Hdf5Error::InvalidState(
4033                        "file is no longer in write mode".into(),
4034                    ));
4035                };
4036
4037                // Bounds before sizing, against the extent the writer holds
4038                // now (an extend since this handle was taken counts): a
4039                // selection the extent does not admit is refused for that
4040                // reason, not for the size it would have had.
4041                check_hyperslab(&writer.dataset_dims(*index), &starts_u64, &counts_u64)?;
4042                let expected = element_count(&counts_u64)?;
4043                if data.len() != expected {
4044                    return Err(Hdf5Error::InvalidState(format!(
4045                        "data length {} does not match slice size {}",
4046                        data.len(),
4047                        expected,
4048                    )));
4049                }
4050
4051                let byte_len = data.len() * T::element_size();
4052                let host =
4053                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
4054
4055                let datatype = writer.dataset_datatype(*index);
4056                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
4057                writer.write_slice(*index, &starts_u64, &counts_u64, &stored)?;
4058                Ok(())
4059            }
4060            DatasetInfo::Reader { .. } => {
4061                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4062            }
4063        }
4064    }
4065
4066    /// Replace elements `start .. start + strings.len()` of a 1-D
4067    /// variable-length string dataset.
4068    ///
4069    /// The extent and every element outside the range are left alone, and the
4070    /// cost is the new strings plus the chunks holding their references — not
4071    /// the column. The dataset's character set is enforced: a non-ASCII
4072    /// replacement in a dataset that declares ASCII is rejected rather than
4073    /// stored under a datatype that misdescribes it.
4074    ///
4075    /// The global heap objects the replaced references pointed at are freed —
4076    /// the same reclaim libhdf5 performs on an overwrite — so updating one
4077    /// element repeatedly reuses space rather than growing the file. A
4078    /// collection emptied by the update returns its block to the allocator.
4079    /// Under SWMR nothing is freed, because a reader may still be following
4080    /// those references.
4081    ///
4082    /// ```no_run
4083    /// # use rust_hdf5::H5File;
4084    /// let file = H5File::open_rw("meta.h5").unwrap();
4085    /// let ds = file.dataset_writer("notes").unwrap();
4086    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
4087    /// file.close().unwrap();
4088    /// ```
4089    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
4090        match &self.info {
4091            DatasetInfo::Writer { index, .. } => {
4092                let inner = borrow_inner(&self.file_inner);
4093                match &*inner {
4094                    H5FileInner::Writer(writer) => {
4095                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
4096                        Ok(())
4097                    }
4098                    _ => Err(Hdf5Error::InvalidState(
4099                        "file is no longer in write mode".into(),
4100                    )),
4101                }
4102            }
4103            DatasetInfo::Reader { .. } => {
4104                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4105            }
4106        }
4107    }
4108
4109    /// Read variable-length strings from a dataset.
4110    ///
4111    /// This handles h5py-style vlen string datasets that store strings
4112    /// as global heap references. Returns one String per element.
4113    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
4114        match &self.info {
4115            DatasetInfo::Reader { name, .. } => {
4116                let mut inner = borrow_inner_mut(&self.file_inner);
4117                match &mut *inner {
4118                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
4119                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4120                }
4121            }
4122            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4123                "cannot read vlen strings from a dataset in write mode".into(),
4124            )),
4125        }
4126    }
4127
4128    /// Read variable-length byte arrays from a dataset.
4129    ///
4130    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
4131    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
4132    /// that store each element as a global heap reference. Returns one
4133    /// `Vec<u8>` per element.
4134    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
4135        match &self.info {
4136            DatasetInfo::Reader { name, .. } => {
4137                let mut inner = borrow_inner_mut(&self.file_inner);
4138                match &mut *inner {
4139                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
4140                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4141                }
4142            }
4143            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4144                "cannot read vlen bytes from a dataset in write mode".into(),
4145            )),
4146        }
4147    }
4148
4149    /// Write object references naming `paths` into elements `0..paths.len()`
4150    /// — h5py's `refs[i] = f['/target'].ref`.
4151    ///
4152    /// The dataset must have been created with
4153    /// [`object_references`](DatasetBuilder::object_references). A path names
4154    /// a dataset or a group (`/` is the root group) and must already exist;
4155    /// what reaches the file is the target's object header address, which is
4156    /// assigned when the file is finalized. Elements left unwritten read back
4157    /// as null references.
4158    pub fn write_object_references(&self, paths: &[&str]) -> Result<()> {
4159        match &self.info {
4160            DatasetInfo::Writer { index, .. } => {
4161                let inner = borrow_inner(&self.file_inner);
4162                match &*inner {
4163                    H5FileInner::Writer(writer) => {
4164                        writer.write_object_references(*index, 0, paths)?;
4165                        Ok(())
4166                    }
4167                    _ => Err(Hdf5Error::InvalidState(
4168                        "file is no longer in write mode".into(),
4169                    )),
4170                }
4171            }
4172            DatasetInfo::Reader { .. } => {
4173                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4174            }
4175        }
4176    }
4177
4178    /// Write region references over `targets` into elements
4179    /// `0..targets.len()` — h5py's `refs[i] = f['/target'].regionref[0:3]`.
4180    ///
4181    /// The dataset must have been created with
4182    /// [`region_references`](DatasetBuilder::region_references). Each target is
4183    /// the path of an existing *dataset* and a [`Selection`] over it, which
4184    /// must fit that dataset's extent — the rule `H5Rcreate` applies. What
4185    /// reaches the file is a global-heap object holding the target's object
4186    /// header address (assigned when the file is finalized) and the serialized
4187    /// selection. Elements left unwritten read back as null references.
4188    ///
4189    /// ```no_run
4190    /// # use rust_hdf5::{H5File, PointSelection, Selection};
4191    /// let file = H5File::create("regions.h5").unwrap();
4192    /// file.new_dataset::<i32>().shape([4, 6]).create("m").unwrap();
4193    /// let refs = file.new_dataset::<u64>()
4194    ///     .region_references()
4195    ///     .shape([1])
4196    ///     .create("refs")
4197    ///     .unwrap();
4198    /// let points = Selection::Points(PointSelection {
4199    ///     rank: 2,
4200    ///     points: vec![vec![0, 1], vec![3, 5]],
4201    /// });
4202    /// refs.write_region_references(&[("/m", points)]).unwrap();
4203    /// file.close().unwrap();
4204    /// ```
4205    pub fn write_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4206        match &self.info {
4207            DatasetInfo::Writer { index, .. } => {
4208                let inner = borrow_inner(&self.file_inner);
4209                match &*inner {
4210                    H5FileInner::Writer(writer) => {
4211                        writer.write_region_references(*index, 0, targets)?;
4212                        Ok(())
4213                    }
4214                    _ => Err(Hdf5Error::InvalidState(
4215                        "file is no longer in write mode".into(),
4216                    )),
4217                }
4218            }
4219            DatasetInfo::Reader { .. } => {
4220                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4221            }
4222        }
4223    }
4224
4225    /// Write revised region references over `targets` into elements
4226    /// `0..targets.len()` — `H5Rcreate_region` plus `H5Dwrite` of an
4227    /// `H5T_STD_REF` dataset.
4228    ///
4229    /// The dataset must have been created with
4230    /// [`std_region_references`](DatasetBuilder::std_region_references) or one
4231    /// of its two siblings, which make the same datatype. Each target is the
4232    /// path of an existing *dataset* and a [`Selection`] over it, which must
4233    /// fit that dataset's extent. What reaches the file is a global-heap blob
4234    /// holding the target's object header address (assigned when the file is
4235    /// finalized) and the serialized selection, and an element carrying the
4236    /// blob's id and its byte count. Elements left unwritten read back as null
4237    /// references.
4238    pub fn write_std_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4239        let targets: Vec<(&str, ReferenceTarget)> = targets
4240            .iter()
4241            .map(|(path, selection)| (*path, ReferenceTarget::Region(selection.clone())))
4242            .collect();
4243        self.write_revised_references(&targets)
4244    }
4245
4246    /// Write attribute references naming `targets` into elements
4247    /// `0..targets.len()` — `H5Rcreate_attr` plus `H5Dwrite` of an
4248    /// `H5T_STD_REF` dataset.
4249    ///
4250    /// Each target is the path of an existing object — a dataset, a group, or
4251    /// `/` for the root group — and the name of an attribute it already
4252    /// carries. There is no pre-1.12 form of this reference kind, so the
4253    /// dataset must have been created with
4254    /// [`attribute_references`](DatasetBuilder::attribute_references) or one of
4255    /// its two siblings. Elements left unwritten read back as null references.
4256    pub fn write_attribute_references(&self, targets: &[(&str, &str)]) -> Result<()> {
4257        let targets: Vec<(&str, ReferenceTarget)> = targets
4258            .iter()
4259            .map(|(path, name)| (*path, ReferenceTarget::Attribute((*name).to_string())))
4260            .collect();
4261        self.write_revised_references(&targets)
4262    }
4263
4264    /// Store `targets` as 1.12 reference elements, whatever mix of kinds they
4265    /// are: the one path both revised-reference writers take.
4266    fn write_revised_references(&self, targets: &[(&str, ReferenceTarget)]) -> Result<()> {
4267        match &self.info {
4268            DatasetInfo::Writer { index, .. } => {
4269                let inner = borrow_inner(&self.file_inner);
4270                match &*inner {
4271                    H5FileInner::Writer(writer) => {
4272                        writer.write_revised_references(*index, 0, targets)?;
4273                        Ok(())
4274                    }
4275                    _ => Err(Hdf5Error::InvalidState(
4276                        "file is no longer in write mode".into(),
4277                    )),
4278                }
4279            }
4280            DatasetInfo::Reader { .. } => {
4281                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4282            }
4283        }
4284    }
4285
4286    /// Read a reference dataset's elements, each resolved to the object it
4287    /// names.
4288    ///
4289    /// Every reference kind is read: the pre-1.12 pair h5py writes —
4290    /// `Reference` (an object header address) and `RegionReference` (a heap id
4291    /// whose heap object holds the target plus a serialized selection) — and
4292    /// the 1.12 `H5T_STD_REF` trio, `H5R_OBJECT2`, `H5R_DATASET_REGION2` and
4293    /// `H5R_ATTR`. An object reference comes back as [`Reference::Object`]
4294    /// carrying the target's path, a region reference as
4295    /// [`Reference::Region`], whose [`bounds`](Reference::bounds) is the
4296    /// selection's bounding box — libhdf5's `H5Sget_select_bounds` — and an
4297    /// attribute reference as [`Reference::Attr`], which adds the attribute's
4298    /// name.
4299    ///
4300    /// A 1.12 reference written into a file other than its target's carries
4301    /// that file's name, and [`Reference::file`] reports it; the path is then
4302    /// a path inside that file, resolved by opening it under the name the
4303    /// reference carries, and `None` when nothing is there.
4304    ///
4305    /// ```no_run
4306    /// # use rust_hdf5::H5File;
4307    /// let file = H5File::open("refs.h5").unwrap();
4308    /// for r in file.dataset("refs").unwrap().read_references().unwrap() {
4309    ///     println!("{:?} {:?}", r.path(), r.bounds());
4310    /// }
4311    /// ```
4312    pub fn read_references(&self) -> Result<Vec<Reference>> {
4313        match &self.info {
4314            DatasetInfo::Reader { name, .. } => {
4315                let mut inner = borrow_inner_mut(&self.file_inner);
4316                match &mut *inner {
4317                    H5FileInner::Reader(reader) => Ok(reader.read_references(name)?),
4318                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4319                }
4320            }
4321            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4322                "cannot read references from a dataset in write mode".into(),
4323            )),
4324        }
4325    }
4326
4327    /// Read a string dataset, fixed-width or variable-length, as one `String`
4328    /// per element.
4329    ///
4330    /// The width of a `FixedString` dataset is whatever the file says, so a
4331    /// 24-byte label column and a 100-byte one are read by the same call. The
4332    /// padding rule the datatype declares decides where each element ends —
4333    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
4334    /// character set decides how the remaining bytes are decoded: ASCII (0)
4335    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
4336    /// violates either is an error naming the element, not a silent
4337    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
4338    /// call that accepts such a file, replacing what it cannot decode.
4339    ///
4340    /// ```no_run
4341    /// # use rust_hdf5::H5File;
4342    /// let file = H5File::open("labels.h5").unwrap();
4343    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
4344    /// ```
4345    pub fn read_strings(&self) -> Result<Vec<String>> {
4346        self.read_strings_inner(false)
4347    }
4348
4349    /// [`read_strings`](Self::read_strings), but bytes that do not decode
4350    /// under the dataset's character set become U+FFFD instead of an error.
4351    ///
4352    /// Producers do mislabel the character set — a file that declares ASCII
4353    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
4354    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
4355        self.read_strings_inner(true)
4356    }
4357
4358    /// The single owner of string decoding for both string datatypes: the
4359    /// element bytes are found differently, the padding and character-set
4360    /// rules that turn them into a `String` are the same.
4361    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
4362        if matches!(self.info, DatasetInfo::Writer { .. }) {
4363            return Err(Hdf5Error::InvalidState(
4364                "cannot read strings from a dataset in write mode".into(),
4365            ));
4366        }
4367        match self.datatype()? {
4368            DatatypeMessage::VarLenString { charset, .. } => self
4369                .read_vlen_bytes()?
4370                .iter()
4371                .enumerate()
4372                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
4373                .collect(),
4374            DatatypeMessage::FixedString {
4375                size,
4376                padding,
4377                charset,
4378            } => {
4379                let width = size as usize;
4380                if width == 0 {
4381                    // A corrupt file can declare it; `chunks_exact(0)` panics.
4382                    return Err(Hdf5Error::InvalidState(
4383                        "fixed-string datatype has zero width".into(),
4384                    ));
4385                }
4386                // `read_raw_bytes` returns `product(dims) * width` bytes, so
4387                // `chunks_exact` leaves no remainder.
4388                let raw = self.read_raw_bytes()?;
4389                raw.chunks_exact(width)
4390                    .enumerate()
4391                    .map(|(i, elem)| {
4392                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
4393                    })
4394                    .collect()
4395            }
4396            other => Err(Hdf5Error::InvalidState(format!(
4397                "read_strings is only for string datasets, this one is {other:?}"
4398            ))),
4399        }
4400    }
4401
4402    /// Read the entire dataset as a typed vector.
4403    ///
4404    /// The raw bytes are read from the file and reinterpreted as `T`. The
4405    /// caller must ensure that `T` matches the datatype used when the dataset
4406    /// was written.
4407    ///
4408    /// # Errors
4409    ///
4410    /// Returns an error if:
4411    /// - The file is in write mode.
4412    /// - The raw data size is not a multiple of `T::element_size()`.
4413    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
4414        match &self.info {
4415            DatasetInfo::Reader {
4416                name, element_size, ..
4417            } => {
4418                if T::element_size() != *element_size {
4419                    return Err(Hdf5Error::TypeMismatch(format!(
4420                        "read type has element size {} but dataset has element size {}",
4421                        T::element_size(),
4422                        element_size,
4423                    )));
4424                }
4425
4426                let datatype = self.datatype()?;
4427                let mut inner = borrow_inner_mut(&self.file_inner);
4428                let H5FileInner::Reader(reader) = &mut *inner else {
4429                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4430                };
4431                let total = reader.dataset_raw_size(name)? as usize;
4432                if !total.is_multiple_of(T::element_size()) {
4433                    return Err(Hdf5Error::TypeMismatch(format!(
4434                        "raw data size {total} is not a multiple of element size {}",
4435                        T::element_size(),
4436                    )));
4437                }
4438
4439                // The image is read into the vector this returns, so the
4440                // bytes are touched once rather than being zeroed, read, and
4441                // then copied into a second buffer of the same size.
4442                read_image_into_new(total / T::element_size(), |image| {
4443                    reader.read_dataset_raw_into_dst(name, image, ReadDst::Fresh)?;
4444                    to_host_byte_order(image, &datatype, T::element_size())
4445                })
4446            }
4447            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4448                "cannot read from a dataset in write mode".into(),
4449            )),
4450        }
4451    }
4452
4453    /// View the entire dataset as `&[T]` pointing straight into the file's
4454    /// memory map — no read, no copy, no allocation.
4455    ///
4456    /// The returned [`MappedView<T>`](crate::MappedView) dereferences to
4457    /// `&[T]` holding exactly what [`read_raw`](Self::read_raw) would have
4458    /// returned, bit for bit.
4459    ///
4460    /// # When it works
4461    ///
4462    /// The file must be open read-only and mapped (which a read-only open
4463    /// does whenever the OS allows), the dataset's raw data must be one
4464    /// contiguous stretch of that file, and its stored elements must already
4465    /// be the host image of a `T` — same width, host byte order, significant
4466    /// bits filling the element. That is the ordinary case for an
4467    /// uncompressed, non-chunked numeric dataset written by either this crate
4468    /// or libhdf5.
4469    ///
4470    /// # When it refuses
4471    ///
4472    /// Zero-copy is a contract, not an optimization: when the bytes cannot be
4473    /// handed over as they lie, this returns
4474    /// [`Hdf5Error::NotViewable`] naming the
4475    /// reason and never quietly falls back to copying. Every
4476    /// [`ViewRefusal`](crate::ViewRefusal) is a case where
4477    /// [`read_raw`](Self::read_raw) still works: the file is not mapped (an
4478    /// open that holds no shared lock never maps), the layout is chunked,
4479    /// compact, virtual, or external, no storage is
4480    /// allocated (the dataset reads as its fill value), `T` is the wrong
4481    /// width, the stored elements need a byte-order swap or bit unpacking,
4482    /// the data lands at an offset `T`'s alignment does not permit, or the
4483    /// image runs past the end of the map.
4484    ///
4485    /// # Snapshot semantics
4486    ///
4487    /// The view owns a share of the map rather than borrowing the file
4488    /// handle, so it stays readable after the dataset and the file are
4489    /// dropped, and after a SWMR refresh has retaken the map — a live view
4490    /// keeps showing the file as it was when *its* map was taken, while the
4491    /// refreshed handle reads the new one. Nothing about a view is
4492    /// invalidated by anything this process does. The share carries the
4493    /// shared file lock the map was taken under, so for as long as any view
4494    /// is alive a writer that honours locks cannot open the file, whether or
4495    /// not the reader that took the map is still open.
4496    ///
4497    /// # Truncation
4498    ///
4499    /// The pages are the file's own. Another process writing the file in
4500    /// place is seen through the view, and one *truncating* it under the map
4501    /// faults with `SIGBUS` on the pages that went away. The shared lock the
4502    /// view keeps is what stands between the map and such a writer; one that
4503    /// waives locks ([`FileLocking::Disabled`](crate::FileLocking::Disabled),
4504    /// or a filesystem without them) is outside what any guard inside this
4505    /// process can see.
4506    ///
4507    /// ```no_run
4508    /// # use rust_hdf5::H5File;
4509    /// let file = H5File::open("data.h5")?;
4510    /// let ds = file.dataset("matrix")?;
4511    /// let view = ds.read_mapped::<f64>()?;
4512    /// let total: f64 = view.iter().sum();
4513    /// # Ok::<(), rust_hdf5::Hdf5Error>(())
4514    /// ```
4515    #[cfg(feature = "mmap")]
4516    pub fn read_mapped<T: H5Type>(&self) -> Result<crate::mapped::MappedView<T>> {
4517        self.mapped_view(crate::mapped::ViewRange::Whole)
4518    }
4519
4520    /// View a contiguous sub-range of the dataset as `&[T]` pointing straight
4521    /// into the file's memory map.
4522    ///
4523    /// `starts` and `counts` name the same N-dimensional selection
4524    /// [`read_slice`](Self::read_slice) takes, and the view holds exactly what
4525    /// that call would have returned — but only when the selection is one
4526    /// contiguous run of the stored image: a trailing group of dimensions
4527    /// taken whole, the dimension before it taken as one span, and a single
4528    /// index along every dimension before that. Anything else steps over
4529    /// elements a single slice cannot skip, and is refused with
4530    /// [`ViewRefusal::Range`](crate::ViewRefusal::Range) rather than gathered
4531    /// into a copy.
4532    ///
4533    /// Everything [`read_mapped`](Self::read_mapped) documents about when a
4534    /// dataset can be viewed, snapshot semantics, and truncation applies here
4535    /// unchanged.
4536    #[cfg(feature = "mmap")]
4537    pub fn read_mapped_slice<T: H5Type>(
4538        &self,
4539        starts: &[usize],
4540        counts: &[usize],
4541    ) -> Result<crate::mapped::MappedView<T>> {
4542        self.mapped_view(crate::mapped::ViewRange::Slab { starts, counts })
4543    }
4544
4545    /// The one route from a dataset handle to the file's map: ask the reader
4546    /// that owns the dataset for the facts, and hand them to
4547    /// [`crate::mapped::view`], which is the only thing that can turn them
4548    /// into a view.
4549    #[cfg(feature = "mmap")]
4550    fn mapped_view<T: H5Type>(
4551        &self,
4552        range: crate::mapped::ViewRange<'_>,
4553    ) -> Result<crate::mapped::MappedView<T>> {
4554        let DatasetInfo::Reader { name, .. } = &self.info else {
4555            return Err(Hdf5Error::InvalidState(
4556                "cannot read from a dataset in write mode".into(),
4557            ));
4558        };
4559        let mut inner = borrow_inner_mut(&self.file_inner);
4560        let H5FileInner::Reader(reader) = &mut *inner else {
4561            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4562        };
4563        let src = reader.dataset_view_source(name)?;
4564        crate::mapped::view::<T>(&src, range).map_err(Hdf5Error::NotViewable)
4565    }
4566
4567    /// Read the raw byte image of a dataset without an `H5Type` carrier.
4568    ///
4569    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
4570    /// the element bytes verbatim regardless of the on-disk element type, so a
4571    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
4572    /// no matching Rust primitive can be read back and decoded by the caller.
4573    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
4574        match &self.info {
4575            DatasetInfo::Reader { name, .. } => {
4576                let mut inner = borrow_inner_mut(&self.file_inner);
4577                match &mut *inner {
4578                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
4579                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4580                }
4581            }
4582            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4583                "cannot read from a dataset in write mode".into(),
4584            )),
4585        }
4586    }
4587
4588    /// Read a numeric dataset as `T`, converting each element from the
4589    /// on-disk datatype.
4590    ///
4591    /// Unlike [`read_raw`](Self::read_raw), which requires `T`'s size to match
4592    /// the stored element size exactly, this inspects the dataset's datatype
4593    /// message — class, signedness, byte order, width — and converts per
4594    /// element:
4595    ///
4596    /// - integer → integer: checked; a stored value that does not fit in `T`
4597    ///   is an error naming the element index and value, never a silent wrap.
4598    /// - `f32` source → `f64`: exact widening.
4599    /// - `f64` source → `f32`, float → integer, and integer → float are
4600    ///   rejected as [`TypeMismatch`](Hdf5Error::TypeMismatch).
4601    ///
4602    /// Big-endian sources are decoded according to the datatype's byte order,
4603    /// which [`read_raw`](Self::read_raw)'s size-only check would misread.
4604    ///
4605    /// ```no_run
4606    /// # use rust_hdf5::H5File;
4607    /// let file = H5File::open("data.h5").unwrap();
4608    /// let ds = file.dataset("counts").unwrap(); // stored as e.g. i16
4609    /// let counts = ds.read_numeric_as::<i64>().unwrap();
4610    /// ```
4611    pub fn read_numeric_as<T: ReadNumeric>(&self) -> Result<Vec<T>> {
4612        match &self.info {
4613            DatasetInfo::Reader { name, .. } => {
4614                let (kind, raw) = {
4615                    let mut inner = borrow_inner_mut(&self.file_inner);
4616                    match &mut *inner {
4617                        H5FileInner::Reader(reader) => {
4618                            let info = reader
4619                                .dataset_info(name)
4620                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4621                            let kind = numeric::classify(&info.datatype)?;
4622                            (kind, reader.read_dataset_raw(name)?)
4623                        }
4624                        _ => {
4625                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4626                        }
4627                    }
4628                };
4629                numeric::convert(kind, &raw)
4630            }
4631            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4632                "cannot read from a dataset in write mode".into(),
4633            )),
4634        }
4635    }
4636
4637    /// Read a slice (hyperslab) of a numeric dataset as `T`, with the same
4638    /// per-element datatype conversion as
4639    /// [`read_numeric_as`](Self::read_numeric_as).
4640    ///
4641    /// `starts` and `counts` define the N-dimensional selection exactly as in
4642    /// [`read_slice`](Self::read_slice).
4643    pub fn read_numeric_slice_as<T: ReadNumeric>(
4644        &self,
4645        starts: &[usize],
4646        counts: &[usize],
4647    ) -> Result<Vec<T>> {
4648        match &self.info {
4649            DatasetInfo::Reader { name, .. } => {
4650                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4651                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4652                let (kind, raw) = {
4653                    let mut inner = borrow_inner_mut(&self.file_inner);
4654                    match &mut *inner {
4655                        H5FileInner::Reader(reader) => {
4656                            let info = reader
4657                                .dataset_info(name)
4658                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4659                            let kind = numeric::classify(&info.datatype)?;
4660                            (kind, reader.read_slice(name, &starts_u64, &counts_u64)?)
4661                        }
4662                        _ => {
4663                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4664                        }
4665                    }
4666                };
4667                numeric::convert(kind, &raw)
4668            }
4669            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4670                "cannot read_slice from a dataset in write mode".into(),
4671            )),
4672        }
4673    }
4674
4675    /// Read the whole dataset into a caller-provided buffer, with no allocation.
4676    ///
4677    /// `out` must have exactly `product(dims)` elements (the dataset's element
4678    /// count) and `T::element_size()` must match the dataset's on-disk element
4679    /// size, otherwise an error is returned and `out` is left unspecified. The
4680    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
4681    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
4682    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
4683    /// without the extra staging copy a `read_raw` + copy-into-pinned would
4684    /// incur. Works for every layout (contiguous, compact, and chunked under
4685    /// any index); for chunked data each decoded chunk is scattered directly
4686    /// into `out`.
4687    ///
4688    /// ```no_run
4689    /// # use rust_hdf5::H5File;
4690    /// let file = H5File::open("data.h5").unwrap();
4691    /// let ds = file.dataset("frames").unwrap();
4692    /// let n: usize = ds.shape().iter().product();
4693    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
4694    /// ds.read_raw_into(&mut buf).unwrap();
4695    /// ```
4696    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
4697        match &self.info {
4698            DatasetInfo::Reader {
4699                name, element_size, ..
4700            } => {
4701                if T::element_size() != *element_size {
4702                    return Err(Hdf5Error::TypeMismatch(format!(
4703                        "read type has element size {} but dataset has element size {}",
4704                        T::element_size(),
4705                        element_size,
4706                    )));
4707                }
4708                let datatype = self.datatype()?;
4709                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
4710                // byte representation; every bit pattern the read writes is a
4711                // valid `T`. The byte view borrows `out` exclusively for this
4712                // call, and `out.len() * element_size` cannot overflow because
4713                // it is the byte length of an existing slice (<= isize::MAX).
4714                let byte_len = out.len() * T::element_size();
4715                let bytes = unsafe {
4716                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4717                };
4718                {
4719                    let mut inner = borrow_inner_mut(&self.file_inner);
4720                    match &mut *inner {
4721                        H5FileInner::Reader(reader) => reader.read_dataset_raw_into(name, bytes)?,
4722                        _ => {
4723                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4724                        }
4725                    }
4726                }
4727                to_host_byte_order(bytes, &datatype, T::element_size())
4728            }
4729            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4730                "cannot read from a dataset in write mode".into(),
4731            )),
4732        }
4733    }
4734
4735    /// Read a hyperslab into a caller-provided buffer, with no allocation.
4736    ///
4737    /// `out` must have exactly `product(counts)` elements and
4738    /// `T::element_size()` must match the dataset's element size. The zero-copy
4739    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
4740    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
4741    /// selection are read, and the selected bytes land directly in `out` — the
4742    /// entry point for reading one frame / block straight into a pinned host
4743    /// buffer for an H2D transfer.
4744    ///
4745    /// ```no_run
4746    /// # use rust_hdf5::H5File;
4747    /// let file = H5File::open("vol.h5").unwrap();
4748    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
4749    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
4750    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
4751    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
4752    /// ```
4753    pub fn read_slice_into<T: H5Type>(
4754        &self,
4755        out: &mut [T],
4756        starts: &[usize],
4757        counts: &[usize],
4758    ) -> Result<()> {
4759        match &self.info {
4760            DatasetInfo::Reader {
4761                name, element_size, ..
4762            } => {
4763                if T::element_size() != *element_size {
4764                    return Err(Hdf5Error::TypeMismatch(format!(
4765                        "read type has element size {} but dataset has element size {}",
4766                        T::element_size(),
4767                        element_size,
4768                    )));
4769                }
4770                let datatype = self.datatype()?;
4771                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4772                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4773                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
4774                // borrow of `out`, byte length within bounds.
4775                let byte_len = out.len() * T::element_size();
4776                let bytes = unsafe {
4777                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4778                };
4779                {
4780                    let mut inner = borrow_inner_mut(&self.file_inner);
4781                    match &mut *inner {
4782                        H5FileInner::Reader(reader) => {
4783                            reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?
4784                        }
4785                        _ => {
4786                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4787                        }
4788                    }
4789                }
4790                to_host_byte_order(bytes, &datatype, T::element_size())
4791            }
4792            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4793                "cannot read from a dataset in write mode".into(),
4794            )),
4795        }
4796    }
4797}
4798
4799// ---------------------------------------------------------------------------
4800// Datatype-aware numeric conversion (read_numeric_as)
4801// ---------------------------------------------------------------------------
4802
4803/// Marker trait for the Rust types [`H5Dataset::read_numeric_as`] can convert
4804/// into: the integer primitives (checked, never wrapping) plus `f32`/`f64`
4805/// (widening only).
4806///
4807/// Sealed — the conversion policy is part of the library contract, so the
4808/// trait cannot be implemented outside this crate.
4809pub trait ReadNumeric: numeric::Sealed {}
4810impl<T: numeric::Sealed> ReadNumeric for T {}
4811
4812pub(crate) mod numeric {
4813    //! Per-element decode + checked conversion for `read_numeric_as` (and the
4814    //! attribute counterpart `H5Attribute::read_numeric_as`).
4815
4816    use crate::error::{Hdf5Error, Result};
4817    use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, IeeeFormat};
4818
4819    /// The on-disk element shape `classify` accepted.
4820    #[derive(Clone, Copy)]
4821    pub enum SourceKind {
4822        Int {
4823            size: usize,
4824            signed: bool,
4825            byte_order: ByteOrder,
4826        },
4827        F16(ByteOrder),
4828        F32(ByteOrder),
4829        F64(ByteOrder),
4830    }
4831
4832    impl SourceKind {
4833        fn element_size(self) -> usize {
4834            match self {
4835                SourceKind::Int { size, .. } => size,
4836                SourceKind::F16(_) => 2,
4837                SourceKind::F32(_) => 4,
4838                SourceKind::F64(_) => 8,
4839            }
4840        }
4841    }
4842
4843    /// Widen an IEEE 754 binary16 bit pattern to `f32`, which represents every
4844    /// half — including subnormals, infinities and NaN payloads — exactly.
4845    ///
4846    /// Rust has no stable `f16` to convert through.
4847    fn f16_bits_to_f32(bits: u16) -> f32 {
4848        let sign = u32::from(bits >> 15);
4849        let exponent = u32::from((bits >> 10) & 0x1f);
4850        let mantissa = u32::from(bits & 0x03ff);
4851        if exponent == 0 {
4852            // Zero and subnormals: the value is mantissa * 2^-24, which is a
4853            // normal f32 for every mantissa, so the multiply is exact. Going
4854            // through the sign separately keeps -0.0.
4855            let magnitude = mantissa as f32 * (1.0 / 16_777_216.0);
4856            return if sign == 1 { -magnitude } else { magnitude };
4857        }
4858        let out = if exponent == 0x1f {
4859            // Infinity and NaN; shifting the mantissa maps the quiet bit onto
4860            // f32's quiet bit and preserves the rest of the payload.
4861            (sign << 31) | 0x7f80_0000 | (mantissa << 13)
4862        } else {
4863            // Normal: rebias the exponent (127 - 15) and left-align the
4864            // mantissa.
4865            (sign << 31) | ((exponent + 112) << 23) | (mantissa << 13)
4866        };
4867        f32::from_bits(out)
4868    }
4869
4870    /// Map a datatype message to a supported numeric source shape.
4871    ///
4872    /// Accepts standard-width integers (1/2/4/8 bytes, full precision, zero
4873    /// bit offset) and IEEE binary32/binary64 floats; everything else is a
4874    /// `TypeMismatch` naming what was found.
4875    pub fn classify(dt: &DatatypeMessage) -> Result<SourceKind> {
4876        match *dt {
4877            DatatypeMessage::FixedPoint {
4878                size,
4879                byte_order,
4880                signed,
4881                bit_offset,
4882                bit_precision,
4883            } => {
4884                if !matches!(size, 1 | 2 | 4 | 8)
4885                    || bit_offset != 0
4886                    || u32::from(bit_precision) != size * 8
4887                {
4888                    return Err(Hdf5Error::TypeMismatch(format!(
4889                        "fixed-point datatype (size {size}, bit offset {bit_offset}, \
4890                         precision {bit_precision}) is not a standard-width integer",
4891                    )));
4892                }
4893                Ok(SourceKind::Int {
4894                    size: size as usize,
4895                    signed,
4896                    byte_order,
4897                })
4898            }
4899            DatatypeMessage::BitField {
4900                size,
4901                byte_order,
4902                bit_offset,
4903                bit_precision,
4904            } => {
4905                // A bit field has no signed form; a full-width one is the
4906                // unsigned integer of the stored width. A narrower one would
4907                // need a shift-and-mask conversion this path does not model.
4908                if !matches!(size, 1 | 2 | 4 | 8)
4909                    || bit_offset != 0
4910                    || u32::from(bit_precision) != size * 8
4911                {
4912                    return Err(Hdf5Error::TypeMismatch(format!(
4913                        "bit-field datatype (size {size}, bit offset {bit_offset}, \
4914                         precision {bit_precision}) is not a whole-width bit field",
4915                    )));
4916                }
4917                Ok(SourceKind::Int {
4918                    size: size as usize,
4919                    signed: false,
4920                    byte_order,
4921                })
4922            }
4923            DatatypeMessage::FloatingPoint {
4924                size,
4925                byte_order,
4926                exponent_size,
4927                mantissa_size,
4928                ..
4929            } => match dt.ieee_format() {
4930                Some(IeeeFormat::Binary16) => Ok(SourceKind::F16(byte_order)),
4931                Some(IeeeFormat::Binary32) => Ok(SourceKind::F32(byte_order)),
4932                Some(IeeeFormat::Binary64) => Ok(SourceKind::F64(byte_order)),
4933                None => Err(Hdf5Error::TypeMismatch(format!(
4934                    "floating-point datatype (size {size}, exponent {exponent_size} bits, \
4935                     mantissa {mantissa_size} bits) is not an IEEE 754 interchange format",
4936                ))),
4937            },
4938            ref other => Err(Hdf5Error::TypeMismatch(format!(
4939                "dataset datatype '{other}' is not numeric",
4940            ))),
4941        }
4942    }
4943
4944    /// Decode and convert every element of `raw` into `T`.
4945    ///
4946    /// One loop per source width and lane type, so an element is a
4947    /// fixed-size load, a sign extension and a range check, and the
4948    /// `TypeMismatch` for an element that does not fit is built only when
4949    /// one is found. An empty input converts to an empty vector whatever
4950    /// the classes.
4951    pub fn convert<T: Sealed>(kind: SourceKind, raw: &[u8]) -> Result<Vec<T>> {
4952        let size = kind.element_size();
4953        if !raw.len().is_multiple_of(size) {
4954            return Err(Hdf5Error::TypeMismatch(format!(
4955                "raw data size {} is not a multiple of element size {size}",
4956                raw.len(),
4957            )));
4958        }
4959        let mut out = vec![T::default(); raw.len() / size];
4960        match kind {
4961            SourceKind::Int {
4962                size,
4963                signed,
4964                byte_order,
4965            } => {
4966                let le = byte_order == ByteOrder::LittleEndian;
4967                match (size, signed) {
4968                    (1, true) => ints::<1, i64, T>(raw, le, &mut out),
4969                    (1, false) => ints::<1, u64, T>(raw, le, &mut out),
4970                    (2, true) => ints::<2, i64, T>(raw, le, &mut out),
4971                    (2, false) => ints::<2, u64, T>(raw, le, &mut out),
4972                    (4, true) => ints::<4, i64, T>(raw, le, &mut out),
4973                    (4, false) => ints::<4, u64, T>(raw, le, &mut out),
4974                    (8, true) => ints::<8, i64, T>(raw, le, &mut out),
4975                    (8, false) => ints::<8, u64, T>(raw, le, &mut out),
4976                    _ => unreachable!("classify admits 1-, 2-, 4- and 8-byte integers only"),
4977                }
4978            }
4979            SourceKind::F16(order) => floats::<2, T>(raw, order, &mut out, |b| {
4980                T::from_f32(f16_bits_to_f32(u16::from_le_bytes(b)))
4981            }),
4982            SourceKind::F32(order) => {
4983                floats::<4, T>(raw, order, &mut out, |b| T::from_f32(f32::from_le_bytes(b)))
4984            }
4985            SourceKind::F64(order) => {
4986                floats::<8, T>(raw, order, &mut out, |b| T::from_f64(f64::from_le_bytes(b)))
4987            }
4988        }?;
4989        Ok(out)
4990    }
4991
4992    /// The integer elements of `raw`, `N` bytes each, through lane `L`.
4993    fn ints<const N: usize, L: IntLane, T: Sealed>(
4994        raw: &[u8],
4995        le: bool,
4996        out: &mut [T],
4997    ) -> Result<()> {
4998        let (chunks, _) = raw.as_chunks::<N>();
4999        for (index, (o, &chunk)) in out.iter_mut().zip(chunks).enumerate() {
5000            let mut bytes = chunk;
5001            if !le {
5002                bytes.reverse();
5003            }
5004            *o = L::load(bytes).into_target(index)?;
5005        }
5006        Ok(())
5007    }
5008
5009    /// The float elements of `raw`, `N` bytes each, `decode` taking each
5010    /// one's little-endian bytes.
5011    fn floats<const N: usize, T: Sealed>(
5012        raw: &[u8],
5013        order: ByteOrder,
5014        out: &mut [T],
5015        decode: impl Fn([u8; N]) -> Result<T>,
5016    ) -> Result<()> {
5017        let (chunks, _) = raw.as_chunks::<N>();
5018        for (o, &chunk) in out.iter_mut().zip(chunks) {
5019            let mut bytes = chunk;
5020            if order == ByteOrder::BigEndian {
5021                bytes.reverse();
5022            }
5023            *o = decode(bytes)?;
5024        }
5025        Ok(())
5026    }
5027
5028    /// The lane an integer source decodes to: `i64` for a signed source,
5029    /// `u64` for an unsigned one, so every standard width — u64::MAX
5030    /// included — is held without loss.
5031    pub trait IntLane: Copy {
5032        /// The value of an element's little-endian `bytes`, `N` in 1..=8.
5033        fn load<const N: usize>(bytes: [u8; N]) -> Self;
5034        fn into_target<T: Sealed>(self, index: usize) -> Result<T>;
5035    }
5036
5037    impl IntLane for u64 {
5038        fn load<const N: usize>(bytes: [u8; N]) -> Self {
5039            let mut padded = [0u8; 8];
5040            padded[..N].copy_from_slice(&bytes);
5041            u64::from_le_bytes(padded)
5042        }
5043        fn into_target<T: Sealed>(self, index: usize) -> Result<T> {
5044            T::from_u64(self, index)
5045        }
5046    }
5047
5048    impl IntLane for i64 {
5049        fn load<const N: usize>(bytes: [u8; N]) -> Self {
5050            // Arithmetic right shift sign-extends the low `N` bytes.
5051            let shift = 64 - 8 * N as u32;
5052            ((u64::load(bytes) << shift) as i64) >> shift
5053        }
5054        fn into_target<T: Sealed>(self, index: usize) -> Result<T> {
5055            T::from_i64(self, index)
5056        }
5057    }
5058
5059    /// The sealed half of `ReadNumeric`: how one source element becomes a
5060    /// `Self`, or a `TypeMismatch` explaining why it cannot. `index` is the
5061    /// element's position, for the message.
5062    pub trait Sealed: Copy + Default {
5063        fn from_i64(v: i64, index: usize) -> Result<Self>;
5064        fn from_u64(v: u64, index: usize) -> Result<Self>;
5065        fn from_f32(v: f32) -> Result<Self>;
5066        fn from_f64(v: f64) -> Result<Self>;
5067    }
5068
5069    macro_rules! int_targets {
5070        ($($t:ty),* $(,)?) => {$(
5071            impl Sealed for $t {
5072                fn from_i64(v: i64, index: usize) -> Result<Self> {
5073                    <$t>::try_from(v).map_err(|_| int_overflow::<$t>(v, index))
5074                }
5075                fn from_u64(v: u64, index: usize) -> Result<Self> {
5076                    <$t>::try_from(v).map_err(|_| int_overflow::<$t>(v, index))
5077                }
5078                fn from_f32(_: f32) -> Result<Self> {
5079                    Err(float_as_int::<$t>())
5080                }
5081                fn from_f64(_: f64) -> Result<Self> {
5082                    Err(float_as_int::<$t>())
5083                }
5084            }
5085        )*};
5086    }
5087    int_targets!(i8, i16, i32, i64, u8, u16, u32, u64);
5088
5089    // Not in the macro: `i128::try_from` and `u128::try_from(u64)` are
5090    // infallible, which trips clippy::unnecessary_fallible_conversions.
5091    impl Sealed for i128 {
5092        fn from_i64(v: i64, _index: usize) -> Result<Self> {
5093            Ok(Self::from(v))
5094        }
5095        fn from_u64(v: u64, _index: usize) -> Result<Self> {
5096            Ok(Self::from(v))
5097        }
5098        fn from_f32(_: f32) -> Result<Self> {
5099            Err(float_as_int::<i128>())
5100        }
5101        fn from_f64(_: f64) -> Result<Self> {
5102            Err(float_as_int::<i128>())
5103        }
5104    }
5105
5106    impl Sealed for u128 {
5107        fn from_i64(v: i64, index: usize) -> Result<Self> {
5108            Self::try_from(v).map_err(|_| int_overflow::<u128>(v, index))
5109        }
5110        fn from_u64(v: u64, _index: usize) -> Result<Self> {
5111            Ok(Self::from(v))
5112        }
5113        fn from_f32(_: f32) -> Result<Self> {
5114            Err(float_as_int::<u128>())
5115        }
5116        fn from_f64(_: f64) -> Result<Self> {
5117            Err(float_as_int::<u128>())
5118        }
5119    }
5120
5121    fn int_overflow<T>(v: impl std::fmt::Display, index: usize) -> Hdf5Error {
5122        Hdf5Error::TypeMismatch(format!(
5123            "value {v} at element {index} does not fit in {}",
5124            std::any::type_name::<T>()
5125        ))
5126    }
5127
5128    fn float_as_int<T>() -> Hdf5Error {
5129        Hdf5Error::TypeMismatch(format!(
5130            "cannot read a floating-point dataset as {}; read as f64 and convert explicitly",
5131            std::any::type_name::<T>()
5132        ))
5133    }
5134
5135    impl Sealed for f32 {
5136        fn from_i64(_: i64, _index: usize) -> Result<Self> {
5137            Err(int_as_f32())
5138        }
5139        fn from_u64(_: u64, _index: usize) -> Result<Self> {
5140            Err(int_as_f32())
5141        }
5142        fn from_f32(v: f32) -> Result<Self> {
5143            Ok(v)
5144        }
5145        fn from_f64(_: f64) -> Result<Self> {
5146            Err(Hdf5Error::TypeMismatch(
5147                "narrowing an f64 dataset to f32 loses precision; read as f64".into(),
5148            ))
5149        }
5150    }
5151
5152    fn int_as_f32() -> Hdf5Error {
5153        Hdf5Error::TypeMismatch(
5154            "cannot read an integer dataset as f32; read as an integer type and convert \
5155             explicitly"
5156                .into(),
5157        )
5158    }
5159
5160    impl Sealed for f64 {
5161        fn from_i64(_: i64, _index: usize) -> Result<Self> {
5162            Err(int_as_f64())
5163        }
5164        fn from_u64(_: u64, _index: usize) -> Result<Self> {
5165            Err(int_as_f64())
5166        }
5167        // Every f32 is exactly representable as f64.
5168        fn from_f32(v: f32) -> Result<Self> {
5169            Ok(f64::from(v))
5170        }
5171        fn from_f64(v: f64) -> Result<Self> {
5172            Ok(v)
5173        }
5174    }
5175
5176    fn int_as_f64() -> Hdf5Error {
5177        Hdf5Error::TypeMismatch(
5178            "cannot read an integer dataset as f64; integers above 2^53 lose precision — \
5179             read as an integer type and convert explicitly"
5180                .into(),
5181        )
5182    }
5183
5184    #[cfg(test)]
5185    mod tests {
5186        use super::*;
5187
5188        /// Every binary16 bit pattern class widens to the f32 with the same
5189        /// value: zeros keep their sign, subnormals stay exact, infinities and
5190        /// NaN payloads survive.
5191        #[test]
5192        fn f16_widening_is_exact() {
5193            let cases: [(u16, f32); 10] = [
5194                (0x0000, 0.0),
5195                (0x3c00, 1.0),
5196                (0xc000, -2.0),
5197                (0x3555, 0.333_251_95),   // nearest half to 1/3
5198                (0x0001, 5.960_464_5e-8), // smallest subnormal, 2^-24
5199                (0x03ff, 6.097_555e-5),   // largest subnormal
5200                (0x0400, 6.103_515_6e-5), // smallest normal
5201                (0x7bff, 65504.0),        // largest finite
5202                (0x7c00, f32::INFINITY),
5203                (0xfc00, f32::NEG_INFINITY),
5204            ];
5205            for (bits, expected) in cases {
5206                let got = f16_bits_to_f32(bits);
5207                assert_eq!(got, expected, "0x{bits:04x} widened to {got}");
5208            }
5209
5210            let neg_zero = f16_bits_to_f32(0x8000);
5211            assert_eq!(neg_zero, 0.0);
5212            assert!(neg_zero.is_sign_negative(), "-0.0 lost its sign");
5213
5214            let nan = f16_bits_to_f32(0x7e01);
5215            assert!(nan.is_nan());
5216            // The quiet bit and the payload land in f32's mantissa.
5217            assert_eq!(nan.to_bits(), 0x7fc0_2000);
5218        }
5219
5220        #[test]
5221        fn f16_source_converts_and_honors_byte_order() {
5222            let kind = classify(&DatatypeMessage::f16_type()).unwrap();
5223            // 1.0, -2.0, 0.333..., 65504
5224            let raw = [0x00, 0x3c, 0x00, 0xc0, 0x55, 0x35, 0xff, 0x7b];
5225            assert_eq!(
5226                convert::<f32>(kind, &raw).unwrap(),
5227                vec![1.0, -2.0, 0.333_251_95, 65504.0]
5228            );
5229            assert_eq!(
5230                convert::<f64>(kind, &raw).unwrap(),
5231                vec![1.0, -2.0, 0.333_251_953_125, 65504.0]
5232            );
5233
5234            let DatatypeMessage::FloatingPoint { .. } = DatatypeMessage::f16_type() else {
5235                unreachable!()
5236            };
5237            let mut be = DatatypeMessage::f16_type();
5238            if let DatatypeMessage::FloatingPoint { byte_order, .. } = &mut be {
5239                *byte_order = ByteOrder::BigEndian;
5240            }
5241            let be_kind = classify(&be).unwrap();
5242            assert_eq!(convert::<f32>(be_kind, &[0x3c, 0x00]).unwrap(), vec![1.0]);
5243        }
5244
5245        /// A float whose layout is not an interchange format is refused, not
5246        /// reinterpreted.
5247        #[test]
5248        fn non_ieee_float_is_refused() {
5249            let mut odd = DatatypeMessage::f32_type();
5250            if let DatatypeMessage::FloatingPoint { exponent_bias, .. } = &mut odd {
5251                *exponent_bias = 63;
5252            }
5253            assert!(odd.ieee_format().is_none());
5254            let err = classify(&odd).err().expect("non-IEEE float was accepted");
5255            assert!(
5256                err.to_string().contains("IEEE 754 interchange format"),
5257                "unexpected error: {err}"
5258            );
5259        }
5260    }
5261}
5262
5263#[cfg(test)]
5264mod tests {
5265    use crate::H5File;
5266    use std::path::PathBuf;
5267
5268    fn temp_path(name: &str) -> PathBuf {
5269        // Include PID + a per-call atomic counter so that concurrent
5270        // cargo invocations and any kernel-level "lock not yet
5271        // released" races between sequential opens cannot collide.
5272        use std::sync::atomic::{AtomicU64, Ordering};
5273        static COUNTER: AtomicU64 = AtomicU64::new(0);
5274        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
5275        std::env::temp_dir().join(format!(
5276            "hdf5_dataset_test_{}_{}_{}.h5",
5277            name,
5278            std::process::id(),
5279            n
5280        ))
5281    }
5282
5283    #[test]
5284    fn runtime_compound_via_datatype_override_and_raw_bytes() {
5285        use crate::format::messages::datatype::DatatypeMessage;
5286        use crate::types::{CompoundType, H5Type};
5287
5288        let path = temp_path("compound_raw");
5289        // A 12-byte packed compound with NO matching Rust primitive carrier,
5290        // so it can only be written through the datatype() override +
5291        // write_raw_bytes path (the runtime-CompoundType use case).
5292        let ct = CompoundType {
5293            members: vec![
5294                ("id".to_string(), i32::hdf5_type(), 0),
5295                ("val".to_string(), f64::hdf5_type(), 4),
5296            ],
5297            total_size: 12,
5298        };
5299        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
5300        let mut bytes = Vec::new();
5301        for (id, val) in recs {
5302            bytes.extend_from_slice(&id.to_le_bytes());
5303            bytes.extend_from_slice(&val.to_le_bytes());
5304        }
5305
5306        {
5307            let file = H5File::create(&path).unwrap();
5308            let ds = file
5309                .new_dataset::<u8>()
5310                .datatype(ct.to_datatype())
5311                .shape([recs.len()])
5312                .create("records")
5313                .unwrap();
5314            ds.write_raw_bytes(&bytes).unwrap();
5315            file.close().unwrap();
5316        }
5317        {
5318            let file = H5File::open(&path).unwrap();
5319            let ds = file.dataset("records").unwrap();
5320            // The on-disk element type is the compound we specified (size 12),
5321            // not the u8 carrier.
5322            match ds.datatype().unwrap() {
5323                DatatypeMessage::Compound { size, members } => {
5324                    assert_eq!(size, 12);
5325                    assert_eq!(members.len(), 2);
5326                    assert_eq!(members[0].name, "id");
5327                    assert_eq!(members[0].offset, 0);
5328                    assert_eq!(members[1].name, "val");
5329                    assert_eq!(members[1].offset, 4);
5330                }
5331                other => panic!("expected compound datatype, got {other:?}"),
5332            }
5333            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
5334        }
5335        std::fs::remove_file(&path).ok();
5336    }
5337
5338    #[test]
5339    fn builder_requires_shape() {
5340        let path = temp_path("no_shape");
5341        let file = H5File::create(&path).unwrap();
5342        let result = file.new_dataset::<u8>().create("data");
5343        assert!(result.is_err());
5344        std::fs::remove_file(&path).ok();
5345    }
5346
5347    // The last chunk along a *fixed* dimension covers more elements than the
5348    // extent has, so growing the dataspace to the chunk's far edge asks for
5349    // more than the declared maximum. Before the clamp the chunk was written
5350    // and then the call failed on that extend, leaving the bytes in the file
5351    // and the caller an error.
5352    #[test]
5353    fn a_partial_edge_chunk_does_not_grow_past_the_declared_maximum() {
5354        let path = temp_path("edge_chunk_extent");
5355        // Extensible array: dimension 1 is unlimited, dimension 0 is fixed at
5356        // 10 and not a multiple of the chunk's 4.
5357        let file = H5File::create(&path).unwrap();
5358        let ds = file
5359            .new_dataset::<i32>()
5360            .shape([10usize, 4])
5361            .max_shape(&[Some(10), None])
5362            .chunk(&[4, 4])
5363            .create("grid")
5364            .unwrap();
5365        let chunk: Vec<u8> = (0i32..16).flat_map(|v| v.to_le_bytes()).collect();
5366        // Chunk row 2 spans elements 8..12 of a dimension that stops at 10.
5367        ds.write_chunk_at(&[2, 0], &chunk).unwrap();
5368        assert_eq!(ds.shape(), vec![10, 4]);
5369        file.close().unwrap();
5370
5371        let file = H5File::open(&path).unwrap();
5372        let back = file.dataset("grid").unwrap().read_raw::<i32>().unwrap();
5373        assert_eq!(back.len(), 40);
5374        assert_eq!(&back[32..40], &[0, 1, 2, 3, 4, 5, 6, 7]);
5375        drop(file);
5376        std::fs::remove_file(&path).ok();
5377    }
5378
5379    // The cap is in `write_chunk_at_inner`, so it belongs to every chunk index
5380    // whose write reaches the extend below it — the v2 B-tree as much as the
5381    // extensible array. A rank-3 dataset with two unlimited dimensions gets
5382    // that index, and its third, fixed dimension is where the last chunk
5383    // overhangs. (The fixed array and the implicit index return before the
5384    // extend: their shape cannot grow at all. The version-1 B-tree does reach
5385    // it — `tests/legacy_append.rs` carries that case, which needs a classic
5386    // file.)
5387    #[test]
5388    fn the_edge_write_cap_holds_for_the_v2_btree_index() {
5389        let path = temp_path("edge_chunk_bt2");
5390        let file = H5File::create(&path).unwrap();
5391        let ds = file
5392            .new_dataset::<i32>()
5393            .shape([10usize, 4, 4])
5394            .max_shape(&[Some(10), None, None])
5395            .chunk(&[4, 4, 4])
5396            .create("cube")
5397            .unwrap();
5398        let chunk: Vec<u8> = (0i32..64).flat_map(|v| v.to_le_bytes()).collect();
5399        // Chunk plane 2 spans elements 8..12 of a dimension that stops at 10.
5400        ds.write_chunk_at(&[2, 0, 0], &chunk).unwrap();
5401        assert_eq!(ds.shape(), vec![10, 4, 4]);
5402        file.close().unwrap();
5403
5404        let file = H5File::open(&path).unwrap();
5405        let back = file.dataset("cube").unwrap().read_raw::<i32>().unwrap();
5406        assert_eq!(back.len(), 160);
5407        // Rows 8 and 9 of the written plane, 16 elements each.
5408        assert_eq!(&back[128..160], &(0i32..32).collect::<Vec<_>>()[..]);
5409        drop(file);
5410        std::fs::remove_file(&path).ok();
5411    }
5412
5413    // The cap guards a chunk-coordinate write, and neither an externally
5414    // stored nor a virtual dataset has chunk coordinates to guard: both are
5415    // contiguous storage classes, refused at build together with chunked
5416    // storage, and `write_chunk_at` refuses what is not chunked. So the path
5417    // the case above exercises cannot be entered for either — asserted here
5418    // rather than left to inspection, since both classes route their raw bytes
5419    // through the same writer as the chunk grid does.
5420    #[test]
5421    fn an_external_or_virtual_dataset_never_reaches_the_edge_write_cap() {
5422        use crate::Selection;
5423        let dir = std::env::temp_dir().join(format!(
5424            "rust_hdf5_edge_cap_{}_{}",
5425            std::process::id(),
5426            temp_path("x").file_name().unwrap().to_string_lossy()
5427        ));
5428        std::fs::create_dir_all(&dir).unwrap();
5429        let path = dir.join("edge_cap.h5");
5430        let file = H5File::create(&path).unwrap();
5431        let payload = dir.join("payload.raw");
5432
5433        // Chunked storage and these two are mutually exclusive at build.
5434        for (which, res) in [
5435            (
5436                "external",
5437                file.new_dataset::<i32>()
5438                    .shape([10usize])
5439                    .external(&[(payload.to_str().unwrap(), 0, 40)])
5440                    .chunk(&[4])
5441                    .create("a"),
5442            ),
5443            (
5444                "virtual",
5445                file.new_dataset::<i32>()
5446                    .shape([10usize])
5447                    .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5448                    .chunk(&[4])
5449                    .create("b"),
5450            ),
5451        ] {
5452            match res {
5453                Ok(_) => panic!("a {which} dataset cannot also be chunked"),
5454                Err(e) => assert!(e.to_string().contains("chunked"), "{which}: {e}"),
5455            }
5456        }
5457
5458        // And the coordinate write itself is refused on both, with the extent
5459        // left exactly where it was.
5460        let ext = file
5461            .new_dataset::<i32>()
5462            .shape([10usize])
5463            .external(&[(payload.to_str().unwrap(), 0, 40)])
5464            .create("outside")
5465            .unwrap();
5466        let vds = file
5467            .new_dataset::<i32>()
5468            .shape([10usize])
5469            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5470            .create("elsewhere")
5471            .unwrap();
5472        let chunk: Vec<u8> = (0i32..4).flat_map(|v| v.to_le_bytes()).collect();
5473        for (which, ds) in [("external", &ext), ("virtual", &vds)] {
5474            let err = ds.write_chunk_at(&[2], &chunk).unwrap_err().to_string();
5475            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5476            let err = ds
5477                .write_chunk_raw_at(&[2], &chunk, 0)
5478                .unwrap_err()
5479                .to_string();
5480            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5481            assert_eq!(ds.shape(), vec![10], "{which}");
5482        }
5483        file.close().unwrap();
5484        std::fs::remove_dir_all(&dir).ok();
5485    }
5486
5487    #[test]
5488    fn write_raw_size_mismatch() {
5489        let path = temp_path("size_mismatch");
5490        let file = H5File::create(&path).unwrap();
5491        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
5492        // Provide 3 elements instead of 4
5493        let result = ds.write_raw(&[1u8, 2, 3]);
5494        assert!(result.is_err());
5495        std::fs::remove_file(&path).ok();
5496    }
5497
5498    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
5499    // dataset = one chunk) rather than silently drop the filter on the
5500    // contiguous path. write_raw then populates that single chunk.
5501    #[cfg(feature = "deflate")]
5502    #[test]
5503    fn filter_without_chunk_autochunks_and_roundtrips() {
5504        let path = temp_path("autochunk_filter");
5505        let data: Vec<i32> = (0..8).collect();
5506        {
5507            let file = H5File::create(&path).unwrap();
5508            let ds = file
5509                .new_dataset::<i32>()
5510                .deflate(6)
5511                .shape([8])
5512                .create("seq")
5513                .unwrap();
5514            ds.write_raw(&data).unwrap();
5515            file.close().unwrap();
5516        }
5517        {
5518            let file = H5File::open(&path).unwrap();
5519            let ds = file.dataset("seq").unwrap();
5520            // The filter forced chunked storage: a single whole-dataset chunk.
5521            assert!(
5522                ds.is_chunked(),
5523                "auto-chunk did not produce chunked storage"
5524            );
5525            assert_eq!(ds.chunk_dims(), Some(vec![8]));
5526            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5527        }
5528        std::fs::remove_file(&path).ok();
5529    }
5530
5531    // B: write_raw on an explicitly chunked + compressed dataset scatters the
5532    // full row-major image across a multi-chunk grid, including edge chunks
5533    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
5534    #[cfg(feature = "deflate")]
5535    #[test]
5536    fn an_edge_chunk_pads_with_zero_not_the_chunk_written_before_it() {
5537        // 4x5 over a 2x3 chunk: the second chunk of each row covers only two
5538        // of its three columns, so a third of it is padding. The image write
5539        // stages every chunk of a shape like this through one reused buffer,
5540        // and the padding has to reach the file as zero rather than as
5541        // whatever the chunk before it left in that buffer.
5542        let path = temp_path("edge_chunk_padding");
5543        let data: Vec<i32> = (1..=20).collect(); // no zeros of its own
5544        {
5545            let file = H5File::create(&path).unwrap();
5546            let ds = file
5547                .new_dataset::<i32>()
5548                .shape([4, 5])
5549                .chunk(&[2, 3])
5550                .create("grid")
5551                .unwrap();
5552            ds.write_raw(&data).unwrap();
5553            file.close().unwrap();
5554        }
5555        {
5556            let file = H5File::open(&path).unwrap();
5557            let ds = file.dataset("grid").unwrap();
5558            let as_i32 = |bytes: Vec<u8>| -> Vec<i32> {
5559                bytes
5560                    .as_chunks::<4>()
5561                    .0
5562                    .iter()
5563                    .map(|b| i32::from_le_bytes(*b))
5564                    .collect()
5565            };
5566            // The chunk that precedes each edge chunk is full, so a leak would
5567            // show as its 3rd and 6th elements (3 and 8, then 13 and 18).
5568            assert_eq!(
5569                as_i32(ds.read_chunk_raw_at(&[0, 0]).unwrap().0),
5570                [1, 2, 3, 6, 7, 8]
5571            );
5572            assert_eq!(
5573                as_i32(ds.read_chunk_raw_at(&[0, 1]).unwrap().0),
5574                [4, 5, 0, 9, 10, 0]
5575            );
5576            assert_eq!(
5577                as_i32(ds.read_chunk_raw_at(&[1, 1]).unwrap().0),
5578                [14, 15, 0, 19, 20, 0]
5579            );
5580            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5581        }
5582        std::fs::remove_file(&path).ok();
5583    }
5584
5585    #[test]
5586    #[cfg(feature = "deflate")]
5587    fn write_raw_multichunk_edge_roundtrips() {
5588        let path = temp_path("multichunk_edge");
5589        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
5590        {
5591            let file = H5File::create(&path).unwrap();
5592            let ds = file
5593                .new_dataset::<i32>()
5594                .shape([7, 5])
5595                .chunk(&[3, 2])
5596                .deflate(4)
5597                .create("grid")
5598                .unwrap();
5599            ds.write_raw(&data).unwrap();
5600            file.close().unwrap();
5601        }
5602        {
5603            let file = H5File::open(&path).unwrap();
5604            let ds = file.dataset("grid").unwrap();
5605            assert_eq!(ds.shape(), vec![7, 5]);
5606            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5607            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5608        }
5609        std::fs::remove_file(&path).ok();
5610    }
5611
5612    // B: write_raw on an unfiltered chunked dataset (previously rejected with
5613    // "use write_chunk for chunked datasets") now gathers and round-trips.
5614    #[test]
5615    fn write_raw_unfiltered_chunked_roundtrips() {
5616        let path = temp_path("chunked_unfiltered");
5617        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
5618        {
5619            let file = H5File::create(&path).unwrap();
5620            let ds = file
5621                .new_dataset::<f64>()
5622                .shape([4, 3])
5623                .chunk(&[2, 2])
5624                .create("m")
5625                .unwrap();
5626            ds.write_raw(&data).unwrap();
5627            file.close().unwrap();
5628        }
5629        {
5630            let file = H5File::open(&path).unwrap();
5631            let ds = file.dataset("m").unwrap();
5632            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
5633            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
5634        }
5635        std::fs::remove_file(&path).ok();
5636    }
5637
5638    // write_raw on an extensible-array (unlimited first dim) compressed dataset
5639    // drives write_full_image_chunked's EA branch, which gathers chunks and
5640    // compresses them through the windowed batch path. Round-trips the full
5641    // image, including a partial edge chunk along the unlimited dimension.
5642    #[cfg(feature = "deflate")]
5643    #[test]
5644    fn write_raw_ea_compressed_roundtrips() {
5645        let path = temp_path("write_raw_ea_deflate");
5646        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
5647        {
5648            let file = H5File::create(&path).unwrap();
5649            let ds = file
5650                .new_dataset::<i32>()
5651                .shape([5, 4])
5652                .chunk(&[2, 4])
5653                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
5654                .deflate(5)
5655                .create("stream")
5656                .unwrap();
5657            ds.write_raw(&data).unwrap();
5658            file.close().unwrap();
5659        }
5660        {
5661            let file = H5File::open(&path).unwrap();
5662            let ds = file.dataset("stream").unwrap();
5663            assert_eq!(ds.shape(), vec![5, 4]);
5664            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5665        }
5666        std::fs::remove_file(&path).ok();
5667    }
5668
5669    // 3D chunked Full read with partial-edge chunks in every dimension. This
5670    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
5671    // dimensions, exercising the nested outer-coordinate carry and the
5672    // last-axis edge clamp (chunks hang off the high edge in all three axes).
5673    #[cfg(feature = "deflate")]
5674    #[test]
5675    fn read_full_3d_chunked_edge_roundtrips() {
5676        let path = temp_path("full_3d_chunked_edge");
5677        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
5678        // along each axis is partial (1, 1, and 1 element respectively).
5679        let total: usize = 5 * 4 * 3;
5680        let data: Vec<i32> = (0..total as i32).collect();
5681        {
5682            let file = H5File::create(&path).unwrap();
5683            let ds = file
5684                .new_dataset::<i32>()
5685                .shape([5, 4, 3])
5686                .chunk(&[2, 3, 2])
5687                .deflate(4)
5688                .create("vol")
5689                .unwrap();
5690            ds.write_raw(&data).unwrap();
5691            file.close().unwrap();
5692        }
5693        {
5694            let file = H5File::open(&path).unwrap();
5695            let ds = file.dataset("vol").unwrap();
5696            assert_eq!(ds.shape(), vec![5, 4, 3]);
5697            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
5698            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5699        }
5700        std::fs::remove_file(&path).ok();
5701    }
5702
5703    #[test]
5704    fn roundtrip_u8_1d() {
5705        let path = temp_path("rt_u8_1d");
5706        let data: Vec<u8> = (0..10).collect();
5707
5708        {
5709            let file = H5File::create(&path).unwrap();
5710            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
5711            ds.write_raw(&data).unwrap();
5712            file.close().unwrap();
5713        }
5714
5715        {
5716            let file = H5File::open(&path).unwrap();
5717            let ds = file.dataset("seq").unwrap();
5718            assert_eq!(ds.shape(), vec![10]);
5719            let readback = ds.read_raw::<u8>().unwrap();
5720            assert_eq!(readback, data);
5721        }
5722
5723        std::fs::remove_file(&path).ok();
5724    }
5725
5726    #[test]
5727    fn roundtrip_i32_2d() {
5728        let path = temp_path("rt_i32_2d");
5729        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
5730
5731        {
5732            let file = H5File::create(&path).unwrap();
5733            let ds = file
5734                .new_dataset::<i32>()
5735                .shape([2, 3])
5736                .create("matrix")
5737                .unwrap();
5738            ds.write_raw(&data).unwrap();
5739            file.close().unwrap();
5740        }
5741
5742        {
5743            let file = H5File::open(&path).unwrap();
5744            let ds = file.dataset("matrix").unwrap();
5745            assert_eq!(ds.shape(), vec![2, 3]);
5746            let readback = ds.read_raw::<i32>().unwrap();
5747            assert_eq!(readback, data);
5748        }
5749
5750        std::fs::remove_file(&path).ok();
5751    }
5752
5753    #[test]
5754    fn roundtrip_f64_3d() {
5755        let path = temp_path("rt_f64_3d");
5756        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
5757
5758        {
5759            let file = H5File::create(&path).unwrap();
5760            let ds = file
5761                .new_dataset::<f64>()
5762                .shape([2, 3, 4])
5763                .create("cube")
5764                .unwrap();
5765            ds.write_raw(&data).unwrap();
5766            file.close().unwrap();
5767        }
5768
5769        {
5770            let file = H5File::open(&path).unwrap();
5771            let ds = file.dataset("cube").unwrap();
5772            assert_eq!(ds.shape(), vec![2, 3, 4]);
5773            let readback = ds.read_raw::<f64>().unwrap();
5774            assert_eq!(readback, data);
5775        }
5776
5777        std::fs::remove_file(&path).ok();
5778    }
5779
5780    #[test]
5781    fn cannot_read_in_write_mode() {
5782        let path = temp_path("no_read_write");
5783        let file = H5File::create(&path).unwrap();
5784        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5785        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5786        let result = ds.read_raw::<u8>();
5787        assert!(result.is_err());
5788        std::fs::remove_file(&path).ok();
5789    }
5790
5791    #[test]
5792    fn cannot_write_in_read_mode() {
5793        let path = temp_path("no_write_read");
5794
5795        {
5796            let file = H5File::create(&path).unwrap();
5797            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5798            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5799            file.close().unwrap();
5800        }
5801
5802        {
5803            let file = H5File::open(&path).unwrap();
5804            let ds = file.dataset("x").unwrap();
5805            let result = ds.write_raw(&[5u8, 6, 7, 8]);
5806            assert!(result.is_err());
5807        }
5808
5809        std::fs::remove_file(&path).ok();
5810    }
5811
5812    #[test]
5813    fn numeric_attr_roundtrip() {
5814        let path = temp_path("num_attr");
5815        {
5816            let file = H5File::create(&path).unwrap();
5817            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5818            ds.write_raw(&[1.0f32; 4]).unwrap();
5819
5820            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5821            a1.write_numeric(&1.2345f64).unwrap();
5822
5823            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
5824            a2.write_numeric(&42i32).unwrap();
5825
5826            file.close().unwrap();
5827        }
5828        {
5829            let file = H5File::open(&path).unwrap();
5830            let ds = file.dataset("data").unwrap();
5831
5832            let scale = ds.attr("scale").unwrap();
5833            let val: f64 = scale.read_numeric().unwrap();
5834            assert!((val - 1.2345).abs() < 1e-10);
5835
5836            let count = ds.attr("count").unwrap();
5837            let val: i32 = count.read_numeric().unwrap();
5838            assert_eq!(val, 42);
5839        }
5840        std::fs::remove_file(&path).ok();
5841    }
5842
5843    #[test]
5844    fn array_attr_roundtrip() {
5845        let path = temp_path("array_attr");
5846        let offsets = [10i32, -20, 30];
5847        {
5848            let file = H5File::create(&path).unwrap();
5849            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5850            ds.write_raw(&[1.0f32; 4]).unwrap();
5851
5852            // 1-D int32 array attribute (NDArrayDimOffset-style).
5853            let a = ds
5854                .new_attr::<i32>()
5855                .shape([3])
5856                .create("dim_offset")
5857                .unwrap();
5858            a.write_array(&offsets).unwrap();
5859
5860            // Wrong element count is rejected.
5861            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
5862            assert!(bad.write_array(&[1i32, 2]).is_err());
5863
5864            file.close().unwrap();
5865        }
5866        {
5867            let file = H5File::open(&path).unwrap();
5868            let ds = file.dataset("data").unwrap();
5869            let a = ds.attr("dim_offset").unwrap();
5870            let raw = a.read_raw().unwrap();
5871            assert_eq!(raw.len(), 3 * 4);
5872            let got: Vec<i32> = raw
5873                .as_chunks::<4>()
5874                .0
5875                .iter()
5876                .map(|b| i32::from_le_bytes(*b))
5877                .collect();
5878            assert_eq!(got, offsets);
5879        }
5880        std::fs::remove_file(&path).ok();
5881    }
5882
5883    #[test]
5884    fn attr_datatype_exposes_class_and_sign() {
5885        // H5Attribute::datatype() must report the stored datatype class and
5886        // signedness so a generic attr->metadata mapper need not infer it from
5887        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
5888        use crate::format::messages::datatype::DatatypeMessage;
5889
5890        let path = temp_path("attr_datatype");
5891        {
5892            let file = H5File::create(&path).unwrap();
5893            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5894            ds.new_attr::<f64>()
5895                .shape(())
5896                .create("scale")
5897                .unwrap()
5898                .write_numeric(&1.5f64)
5899                .unwrap();
5900            ds.new_attr::<i32>()
5901                .shape(())
5902                .create("count")
5903                .unwrap()
5904                .write_numeric(&7i32)
5905                .unwrap();
5906            file.close().unwrap();
5907        }
5908        {
5909            let file = H5File::open(&path).unwrap();
5910            let ds = file.dataset("data").unwrap();
5911
5912            match ds.attr("scale").unwrap().datatype().unwrap() {
5913                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
5914                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
5915            }
5916
5917            match ds.attr("count").unwrap().datatype().unwrap() {
5918                DatatypeMessage::FixedPoint { size, signed, .. } => {
5919                    assert_eq!(size, 4);
5920                    assert!(signed, "i32 attr must be signed");
5921                }
5922                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
5923            }
5924        }
5925        std::fs::remove_file(&path).ok();
5926    }
5927
5928    #[test]
5929    fn attr_datatype_in_write_mode_errors() {
5930        let path = temp_path("attr_datatype_write_mode");
5931        let file = H5File::create(&path).unwrap();
5932        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5933        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5934        assert!(attr.datatype().is_err());
5935        std::fs::remove_file(&path).ok();
5936    }
5937
5938    #[test]
5939    fn cannot_create_dataset_in_read_mode() {
5940        let path = temp_path("no_create_read");
5941
5942        {
5943            let _file = H5File::create(&path).unwrap();
5944        }
5945
5946        {
5947            let file = H5File::open(&path).unwrap();
5948            let result = file.new_dataset::<u8>().shape([4]).create("x");
5949            assert!(result.is_err());
5950        }
5951
5952        std::fs::remove_file(&path).ok();
5953    }
5954
5955    #[test]
5956    fn shape_accessor() {
5957        let path = temp_path("shape_acc");
5958
5959        let file = H5File::create(&path).unwrap();
5960        let ds = file
5961            .new_dataset::<f32>()
5962            .shape([5, 10, 3])
5963            .create("tensor")
5964            .unwrap();
5965        assert_eq!(ds.shape(), vec![5, 10, 3]);
5966
5967        std::fs::remove_file(&path).ok();
5968    }
5969
5970    #[test]
5971    fn slice_roundtrip_2d() {
5972        let path = temp_path("slice_2d");
5973
5974        // Create a 4x5 dataset, write full, then read a slice
5975        let data: Vec<i32> = (0..20).collect();
5976        {
5977            let file = H5File::create(&path).unwrap();
5978            let ds = file
5979                .new_dataset::<i32>()
5980                .shape([4, 5])
5981                .create("mat")
5982                .unwrap();
5983            ds.write_raw(&data).unwrap();
5984            file.close().unwrap();
5985        }
5986        {
5987            let file = H5File::open(&path).unwrap();
5988            let ds = file.dataset("mat").unwrap();
5989            // Read rows 1..3, cols 2..4 (2x2 slice)
5990            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
5991            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
5992            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
5993            assert_eq!(slice, vec![7, 8, 12, 13]);
5994        }
5995
5996        std::fs::remove_file(&path).ok();
5997    }
5998
5999    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
6000    // caller-provided buffer and MUST produce byte-for-byte the same data as
6001    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
6002    // creatable layout, since both now share one buffer-filling core.
6003    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
6004    where
6005        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
6006    {
6007        let n: usize = ds.shape().iter().product();
6008        let want_full = ds.read_raw::<T>().unwrap();
6009        let mut got_full = vec![T::default(); n];
6010        ds.read_raw_into::<T>(&mut got_full).unwrap();
6011        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
6012
6013        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
6014        let sn: usize = counts.iter().product();
6015        let mut got_slice = vec![T::default(); sn];
6016        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
6017            .unwrap();
6018        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
6019    }
6020
6021    #[test]
6022    fn read_into_matches_vec_contiguous() {
6023        let path = temp_path("into_contig");
6024        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
6025        {
6026            let file = H5File::create(&path).unwrap();
6027            let ds = file
6028                .new_dataset::<i32>()
6029                .shape([4, 5])
6030                .create("mat")
6031                .unwrap();
6032            ds.write_raw(&data).unwrap();
6033            file.close().unwrap();
6034        }
6035        {
6036            let file = H5File::open(&path).unwrap();
6037            let ds = file.dataset("mat").unwrap();
6038            assert_eq!(ds.chunk_dims(), None);
6039            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
6040        }
6041        std::fs::remove_file(&path).ok();
6042    }
6043
6044    #[test]
6045    fn read_into_matches_vec_chunked_unfiltered() {
6046        let path = temp_path("into_chunk");
6047        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
6048        {
6049            let file = H5File::create(&path).unwrap();
6050            let ds = file
6051                .new_dataset::<f64>()
6052                .shape([7, 5])
6053                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
6054                .create("grid")
6055                .unwrap();
6056            ds.write_raw(&data).unwrap();
6057            file.close().unwrap();
6058        }
6059        {
6060            let file = H5File::open(&path).unwrap();
6061            let ds = file.dataset("grid").unwrap();
6062            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6063            // Slice spans multiple chunks (rows 2..5, cols 1..4).
6064            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
6065        }
6066        std::fs::remove_file(&path).ok();
6067    }
6068
6069    #[test]
6070    fn read_into_matches_vec_single_chunk() {
6071        let path = temp_path("into_single_chunk");
6072        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
6073        {
6074            let file = H5File::create(&path).unwrap();
6075            let ds = file
6076                .new_dataset::<i32>()
6077                .shape([3, 4])
6078                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
6079                .create("g")
6080                .unwrap();
6081            ds.write_raw(&data).unwrap();
6082            file.close().unwrap();
6083        }
6084        {
6085            let file = H5File::open(&path).unwrap();
6086            let ds = file.dataset("g").unwrap();
6087            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
6088            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
6089        }
6090        std::fs::remove_file(&path).ok();
6091    }
6092
6093    #[cfg(feature = "deflate")]
6094    #[test]
6095    fn read_into_matches_vec_chunked_deflate() {
6096        let path = temp_path("into_chunk_deflate");
6097        let data: Vec<i32> = (0..35).collect(); // 7 x 5
6098        {
6099            let file = H5File::create(&path).unwrap();
6100            let ds = file
6101                .new_dataset::<i32>()
6102                .shape([7, 5])
6103                .chunk(&[3, 2])
6104                .deflate(4)
6105                .create("grid")
6106                .unwrap();
6107            ds.write_raw(&data).unwrap();
6108            file.close().unwrap();
6109        }
6110        {
6111            let file = H5File::open(&path).unwrap();
6112            let ds = file.dataset("grid").unwrap();
6113            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6114            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
6115        }
6116        std::fs::remove_file(&path).ok();
6117    }
6118
6119    #[test]
6120    fn read_into_wrong_buffer_size_rejected() {
6121        let path = temp_path("into_badlen");
6122        let data: Vec<i32> = (0..20).collect(); // 4 x 5
6123        {
6124            let file = H5File::create(&path).unwrap();
6125            let ds = file
6126                .new_dataset::<i32>()
6127                .shape([4, 5])
6128                .create("mat")
6129                .unwrap();
6130            ds.write_raw(&data).unwrap();
6131            file.close().unwrap();
6132        }
6133        {
6134            let file = H5File::open(&path).unwrap();
6135            let ds = file.dataset("mat").unwrap();
6136
6137            // Too small / too large full-read buffers are both rejected.
6138            let mut small = vec![0i32; 19];
6139            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
6140            let mut large = vec![0i32; 21];
6141            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
6142
6143            // Slice buffer must be exactly product(counts) = 4.
6144            let mut bad_slice = vec![0i32; 3];
6145            assert!(ds
6146                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
6147                .is_err());
6148            // The correctly sized slice buffer succeeds.
6149            let mut ok_slice = vec![0i32; 4];
6150            assert!(ds
6151                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
6152                .is_ok());
6153        }
6154        std::fs::remove_file(&path).ok();
6155    }
6156
6157    #[test]
6158    fn read_into_wrong_element_size_rejected() {
6159        let path = temp_path("into_badtype");
6160        let data: Vec<i32> = (0..20).collect(); // element size 4
6161        {
6162            let file = H5File::create(&path).unwrap();
6163            let ds = file
6164                .new_dataset::<i32>()
6165                .shape([4, 5])
6166                .create("mat")
6167                .unwrap();
6168            ds.write_raw(&data).unwrap();
6169            file.close().unwrap();
6170        }
6171        {
6172            let file = H5File::open(&path).unwrap();
6173            let ds = file.dataset("mat").unwrap();
6174            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
6175            // element size -> TypeMismatch, even with a "correctly sized" Vec.
6176            let mut as_u8 = vec![0u8; 20];
6177            assert!(matches!(
6178                ds.read_raw_into::<u8>(&mut as_u8),
6179                Err(crate::Hdf5Error::TypeMismatch(_))
6180            ));
6181            let mut as_i64 = vec![0i64; 20];
6182            assert!(matches!(
6183                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
6184                Err(crate::Hdf5Error::TypeMismatch(_))
6185            ));
6186        }
6187        std::fs::remove_file(&path).ok();
6188    }
6189
6190    #[test]
6191    fn write_slice_2d() {
6192        let path = temp_path("write_slice_2d");
6193
6194        {
6195            let file = H5File::create(&path).unwrap();
6196            let ds = file
6197                .new_dataset::<f32>()
6198                .shape([3, 4])
6199                .create("data")
6200                .unwrap();
6201            ds.write_raw(&[0.0f32; 12]).unwrap();
6202            // Overwrite a 2x2 sub-region
6203            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
6204                .unwrap();
6205            file.close().unwrap();
6206        }
6207        {
6208            let file = H5File::open(&path).unwrap();
6209            let ds = file.dataset("data").unwrap();
6210            let full = ds.read_raw::<f32>().unwrap();
6211            // Row 0: [0,0,0,0]
6212            // Row 1: [0,10,20,0]
6213            // Row 2: [0,30,40,0]
6214            assert_eq!(
6215                full,
6216                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
6217            );
6218        }
6219
6220        std::fs::remove_file(&path).ok();
6221    }
6222
6223    /// One 2x4 i32 chunk whose every element is `v`.
6224    fn chunk_of(v: i32) -> Vec<u8> {
6225        (0..8).flat_map(|_| v.to_le_bytes()).collect()
6226    }
6227
6228    /// Write chunk (0,0) `rewrites` times — each time with a different value,
6229    /// so no write can be skipped — and return the closed file's size along
6230    /// with what the chunk reads back as.
6231    fn rewrite_chunk(
6232        tag: &str,
6233        rewrites: i32,
6234        build: impl Fn(&H5File) -> crate::H5Dataset,
6235    ) -> (u64, i32) {
6236        let path = temp_path(tag);
6237        {
6238            let file = H5File::create(&path).unwrap();
6239            let ds = build(&file);
6240            for v in 1..=rewrites {
6241                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
6242            }
6243            file.close().unwrap();
6244        }
6245        let size = std::fs::metadata(&path).unwrap().len();
6246        let first = {
6247            let file = H5File::open(&path).unwrap();
6248            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
6249        };
6250        std::fs::remove_file(&path).ok();
6251        (size, first)
6252    }
6253
6254    // An unfiltered chunk's stored size is fixed by the chunk shape, so
6255    // rewriting it must overwrite the block it already occupies rather than
6256    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
6257    // leaves must_alloc false for exactly this case). The file must therefore
6258    // be byte-identical in size no matter how many times the chunk is written.
6259    #[test]
6260    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
6261        let build = |f: &H5File| {
6262            f.new_dataset::<i32>()
6263                .shape([2, 4])
6264                .chunk(&[2, 4])
6265                .max_shape(&[None, Some(4)])
6266                .create("d")
6267                .unwrap()
6268        };
6269        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
6270        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
6271        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6272        assert_eq!(last, 8, "the last write must be the one that survives");
6273    }
6274
6275    #[test]
6276    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
6277        let build = |f: &H5File| {
6278            f.new_dataset::<i32>()
6279                .shape([2, 4])
6280                .chunk(&[2, 4])
6281                .create("d")
6282                .unwrap()
6283        };
6284        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
6285        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
6286        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6287        assert_eq!(last, 8);
6288    }
6289
6290    #[test]
6291    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
6292        let build = |f: &H5File| {
6293            f.new_dataset::<i32>()
6294                .shape([2, 4])
6295                .chunk(&[2, 4])
6296                .max_shape(&[None, None])
6297                .create("d")
6298                .unwrap()
6299        };
6300        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
6301        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
6302        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6303        assert_eq!(last, 8);
6304    }
6305
6306    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
6307    // pool. Every node is the same size, so the blocks already on disk are
6308    // reused and repeated flushes cost nothing; sizing the root to its record
6309    // count instead would relocate it each time and orphan the block it left.
6310    #[test]
6311    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
6312        let flush_n = |label: &str, flushes: usize| -> u64 {
6313            let path = temp_path(label);
6314            {
6315                let file = H5File::create(&path).unwrap();
6316                let ds = file
6317                    .new_dataset::<i32>()
6318                    .shape([2, 4])
6319                    .chunk(&[2, 4])
6320                    .max_shape(&[None, None])
6321                    .create("d")
6322                    .unwrap();
6323                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
6324                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
6325                for _ in 0..flushes {
6326                    ds.flush().unwrap();
6327                }
6328                file.close().unwrap();
6329            }
6330            let size = std::fs::metadata(&path).unwrap().len();
6331            // The data must survive every rewrite of the index.
6332            {
6333                let file = H5File::open(&path).unwrap();
6334                let ds = file.dataset("d").unwrap();
6335                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
6336            }
6337            std::fs::remove_file(&path).ok();
6338            size
6339        };
6340        assert_eq!(
6341            flush_n("bt2_flush_8", 8),
6342            flush_n("bt2_flush_1", 1),
6343            "8 index flushes grew the file past a single one"
6344        );
6345    }
6346
6347    // A filtered chunk whose compressed size changes cannot stay put, so it
6348    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
6349    // H5MF_xfree). Alternating between two payloads of different compressed
6350    // size must therefore keep reusing the same two blocks instead of
6351    // appending a fresh one each time.
6352    #[cfg(feature = "deflate")]
6353    #[test]
6354    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
6355        // All-equal elements deflate to far fewer bytes than a varied payload,
6356        // so the two writes below land at different stored sizes.
6357        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
6358        let varied: Vec<u8> = (0..8i32)
6359            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
6360            .collect();
6361
6362        let sizes: Vec<u64> = [1usize, 8]
6363            .iter()
6364            .map(|&rounds| {
6365                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
6366                {
6367                    let file = H5File::create(&path).unwrap();
6368                    let ds = file
6369                        .new_dataset::<i32>()
6370                        .shape([2, 4])
6371                        .chunk(&[2, 4])
6372                        .max_shape(&[None, Some(4)])
6373                        .deflate(6)
6374                        .create("d")
6375                        .unwrap();
6376                    for _ in 0..rounds {
6377                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
6378                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
6379                    }
6380                    file.close().unwrap();
6381                }
6382                let size = std::fs::metadata(&path).unwrap().len();
6383                {
6384                    let file = H5File::open(&path).unwrap();
6385                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
6386                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
6387                    assert_eq!(got, want, "the last write must survive the round trip");
6388                }
6389                std::fs::remove_file(&path).ok();
6390                size
6391            })
6392            .collect();
6393
6394        assert_eq!(
6395            sizes[1], sizes[0],
6396            "8 alternating rewrites grew the file past a single pair"
6397        );
6398    }
6399
6400    #[test]
6401    fn write_slice_out_of_bounds_rejected() {
6402        let path = temp_path("write_slice_oob");
6403        let file = H5File::create(&path).unwrap();
6404        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
6405        ds.write_raw(&[0i32; 4]).unwrap();
6406        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
6407        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
6408        // An in-bounds slice still works.
6409        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
6410        std::fs::remove_file(&path).ok();
6411    }
6412
6413    #[test]
6414    fn duplicate_dataset_name_rejected() {
6415        let path = temp_path("dup_name");
6416        let file = H5File::create(&path).unwrap();
6417        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
6418        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
6419        std::fs::remove_file(&path).ok();
6420    }
6421
6422    #[test]
6423    fn extend_cannot_shrink() {
6424        let path = temp_path("extend_shrink");
6425        let file = H5File::create(&path).unwrap();
6426        let ds = file
6427            .new_dataset::<i32>()
6428            .shape([0])
6429            .chunk(&[2])
6430            .max_shape(&[None])
6431            .create("d")
6432            .unwrap();
6433        ds.append(&[1i32, 2, 3, 4]).unwrap();
6434        // Shrinking below the written extent must be rejected.
6435        assert!(ds.extend(&[2]).is_err());
6436        // Growing is fine.
6437        assert!(ds.extend(&[6]).is_ok());
6438        std::fs::remove_file(&path).ok();
6439    }
6440
6441    #[test]
6442    fn attr_read_roundtrip() {
6443        use crate::types::VarLenUnicode;
6444        let path = temp_path("attr_read");
6445
6446        {
6447            let file = H5File::create(&path).unwrap();
6448            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6449            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6450            let a1 = ds
6451                .new_attr::<VarLenUnicode>()
6452                .shape(())
6453                .create("units")
6454                .unwrap();
6455            a1.write_string("meters").unwrap();
6456            let a2 = ds
6457                .new_attr::<VarLenUnicode>()
6458                .shape(())
6459                .create("desc")
6460                .unwrap();
6461            a2.write_string("test data").unwrap();
6462            file.close().unwrap();
6463        }
6464        {
6465            let file = H5File::open(&path).unwrap();
6466            let ds = file.dataset("data").unwrap();
6467
6468            let names = ds.attr_names().unwrap();
6469            assert!(names.contains(&"units".to_string()));
6470            assert!(names.contains(&"desc".to_string()));
6471
6472            let units = ds.attr("units").unwrap();
6473            assert_eq!(units.read_string().unwrap(), "meters");
6474
6475            let desc = ds.attr("desc").unwrap();
6476            assert_eq!(desc.read_string().unwrap(), "test data");
6477        }
6478
6479        std::fs::remove_file(&path).ok();
6480    }
6481
6482    #[test]
6483    fn type_mismatch_element_size() {
6484        let path = temp_path("type_mismatch");
6485
6486        {
6487            let file = H5File::create(&path).unwrap();
6488            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
6489            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
6490            file.close().unwrap();
6491        }
6492
6493        {
6494            let file = H5File::open(&path).unwrap();
6495            let ds = file.dataset("data").unwrap();
6496            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
6497            let result = ds.read_raw::<u8>();
6498            assert!(result.is_err());
6499        }
6500
6501        std::fs::remove_file(&path).ok();
6502    }
6503
6504    #[test]
6505    fn dataset_survives_file_move() {
6506        let path = temp_path("ds_survives");
6507
6508        let ds = {
6509            let file = H5File::create(&path).unwrap();
6510            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
6511        };
6512        // file is dropped here, but ds still holds Rc to the inner state
6513        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6514        // The writer will finalize on drop of the last Rc
6515
6516        std::fs::remove_file(&path).ok();
6517    }
6518
6519    #[test]
6520    fn new_attr_scalar_string() {
6521        use crate::types::VarLenUnicode;
6522
6523        let path = temp_path("attr_scalar_string");
6524        {
6525            let file = H5File::create(&path).unwrap();
6526            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6527            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6528
6529            let attr = ds
6530                .new_attr::<VarLenUnicode>()
6531                .shape(())
6532                .create("name")
6533                .unwrap();
6534            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
6535                .unwrap();
6536
6537            file.close().unwrap();
6538        }
6539
6540        // Verify the file is still valid and readable
6541        {
6542            use crate::format::messages::datatype::DatatypeMessage;
6543            let file = H5File::open(&path).unwrap();
6544            let ds = file.dataset("data").unwrap();
6545            assert_eq!(ds.shape(), vec![4]);
6546            let readback = ds.read_raw::<u8>().unwrap();
6547            assert_eq!(readback, vec![1u8, 2, 3, 4]);
6548
6549            // The string attribute is stored as a true variable-length string
6550            // (not fixed-length) and round-trips its value.
6551            let attr = ds.attr("name").unwrap();
6552            assert!(
6553                matches!(
6554                    attr.datatype().unwrap(),
6555                    DatatypeMessage::VarLenString { .. }
6556                ),
6557                "string attribute should have a variable-length string datatype"
6558            );
6559            assert_eq!(attr.read_string().unwrap(), "test_value");
6560        }
6561
6562        std::fs::remove_file(&path).ok();
6563    }
6564
6565    #[test]
6566    fn all_numeric_types_roundtrip() {
6567        let path = temp_path("all_types");
6568
6569        {
6570            let file = H5File::create(&path).unwrap();
6571
6572            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
6573            ds.write_raw(&[1u8, 2]).unwrap();
6574
6575            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
6576            ds.write_raw(&[-1i8, 1]).unwrap();
6577
6578            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
6579            ds.write_raw(&[100u16, 200]).unwrap();
6580
6581            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
6582            ds.write_raw(&[-100i16, 100]).unwrap();
6583
6584            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
6585            ds.write_raw(&[1000u32, 2000]).unwrap();
6586
6587            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
6588            ds.write_raw(&[-1000i32, 1000]).unwrap();
6589
6590            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
6591            ds.write_raw(&[10000u64, 20000]).unwrap();
6592
6593            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
6594            ds.write_raw(&[-10000i64, 10000]).unwrap();
6595
6596            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
6597            ds.write_raw(&[1.5f32, 2.5]).unwrap();
6598
6599            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
6600            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
6601
6602            file.close().unwrap();
6603        }
6604
6605        {
6606            let file = H5File::open(&path).unwrap();
6607
6608            assert_eq!(
6609                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
6610                vec![1u8, 2]
6611            );
6612            assert_eq!(
6613                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
6614                vec![-1i8, 1]
6615            );
6616            assert_eq!(
6617                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
6618                vec![100u16, 200]
6619            );
6620            assert_eq!(
6621                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
6622                vec![-100i16, 100]
6623            );
6624            assert_eq!(
6625                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
6626                vec![1000u32, 2000]
6627            );
6628            assert_eq!(
6629                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
6630                vec![-1000i32, 1000]
6631            );
6632            assert_eq!(
6633                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
6634                vec![10000u64, 20000]
6635            );
6636            assert_eq!(
6637                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
6638                vec![-10000i64, 10000]
6639            );
6640            assert_eq!(
6641                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
6642                vec![1.5f32, 2.5]
6643            );
6644            assert_eq!(
6645                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
6646                vec![1.23456f64, 7.89012]
6647            );
6648        }
6649
6650        std::fs::remove_file(&path).ok();
6651    }
6652
6653    #[test]
6654    fn append_chunked_roundtrip() {
6655        let path = temp_path("append_chunked");
6656
6657        {
6658            let file = H5File::create(&path).unwrap();
6659            let ds = file
6660                .new_dataset::<f64>()
6661                .shape([0, 3])
6662                .chunk(&[1, 3])
6663                .max_shape(&[None, Some(3)])
6664                .create("data")
6665                .unwrap();
6666
6667            // Append one frame
6668            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
6669            // Append two frames at once
6670            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
6671
6672            file.close().unwrap();
6673        }
6674
6675        {
6676            let file = H5File::open(&path).unwrap();
6677            let ds = file.dataset("data").unwrap();
6678            assert_eq!(ds.shape(), vec![3, 3]);
6679            let all = ds.read_raw::<f64>().unwrap();
6680            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
6681        }
6682
6683        std::fs::remove_file(&path).ok();
6684    }
6685
6686    #[test]
6687    fn append_1d_chunked() {
6688        let path = temp_path("append_1d");
6689
6690        {
6691            let file = H5File::create(&path).unwrap();
6692            let ds = file
6693                .new_dataset::<i32>()
6694                .shape([0])
6695                .chunk(&[4])
6696                .max_shape(&[None])
6697                .create("values")
6698                .unwrap();
6699
6700            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
6701            ds.append(&[40i32]).unwrap(); // fills chunk boundary
6702            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
6703
6704            file.close().unwrap();
6705        }
6706
6707        {
6708            let file = H5File::open(&path).unwrap();
6709            let ds = file.dataset("values").unwrap();
6710            assert_eq!(ds.shape(), vec![8]);
6711            let all = ds.read_raw::<i32>().unwrap();
6712            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
6713        }
6714
6715        std::fs::remove_file(&path).ok();
6716    }
6717
6718    #[test]
6719    fn append_partial_chunk_flushed_on_close() {
6720        let path = temp_path("append_partial_close");
6721
6722        {
6723            let file = H5File::create(&path).unwrap();
6724            let ds = file
6725                .new_dataset::<f64>()
6726                .shape([0])
6727                .chunk(&[4])
6728                .max_shape(&[None])
6729                .create("vals")
6730                .unwrap();
6731
6732            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
6733            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
6734            file.close().unwrap();
6735        }
6736
6737        {
6738            let file = H5File::open(&path).unwrap();
6739            let ds = file.dataset("vals").unwrap();
6740            assert_eq!(ds.shape(), vec![5]);
6741            let all = ds.read_raw::<f64>().unwrap();
6742            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
6743            // read_raw reads total shape elements
6744            assert_eq!(all.len(), 5);
6745            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
6746        }
6747
6748        std::fs::remove_file(&path).ok();
6749    }
6750
6751    /// An append that leaves its chunk partial is buffered until close, and
6752    /// the flush has to keep the frames that chunk already holds. It built a
6753    /// fresh fill-value chunk around the buffered frame instead, so reopening
6754    /// a file and appending one row erased every earlier row of that chunk
6755    /// (issue #3). Four sessions: the second lands beside an existing row, the
6756    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
6757    /// the third left in chunk 1.
6758    #[test]
6759    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
6760        let path = temp_path("append_reopen_partial");
6761
6762        {
6763            let file = H5File::create(&path).unwrap();
6764            let ds = file
6765                .new_dataset::<i32>()
6766                .shape([0, 3])
6767                .chunk(&[4, 3])
6768                .max_shape(&[None, Some(3)])
6769                .create("values")
6770                .unwrap();
6771            ds.append(&[1, 2, 3]).unwrap();
6772            file.close().unwrap();
6773        }
6774        for rows in [
6775            vec![4, 5, 6],
6776            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
6777            vec![16, 17, 18],
6778        ] {
6779            let file = H5File::open_rw(&path).unwrap();
6780            file.dataset_writer("values")
6781                .unwrap()
6782                .append(&rows)
6783                .unwrap();
6784            file.close().unwrap();
6785        }
6786
6787        let file = H5File::open(&path).unwrap();
6788        let ds = file.dataset("values").unwrap();
6789        assert_eq!(ds.shape(), vec![6, 3]);
6790        assert_eq!(
6791            ds.read_raw::<i32>().unwrap(),
6792            (1..=18).collect::<Vec<i32>>()
6793        );
6794        std::fs::remove_file(&path).ok();
6795    }
6796
6797    #[cfg(feature = "deflate")]
6798    #[test]
6799    fn vlen_append_after_reopen_filtered() {
6800        // Reopen + append into a partially-written *compressed* vlen chunk
6801        // (index-block chunk). Exercises filtered-index-block reconstruction
6802        // in open_append plus filtered read-modify-write.
6803        let path = temp_path("vlen_reopen_filtered");
6804        {
6805            let file = H5File::create(&path).unwrap();
6806            file.create_appendable_vlen_dataset(
6807                "strs",
6808                4,
6809                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6810            )
6811            .unwrap();
6812            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
6813                .unwrap();
6814            file.close().unwrap();
6815        }
6816        {
6817            let file = H5File::open_rw(&path).unwrap();
6818            file.append_vlen_strings("strs", &["delta"]).unwrap();
6819            file.close().unwrap();
6820        }
6821        {
6822            let file = H5File::open(&path).unwrap();
6823            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6824            assert_eq!(
6825                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
6826                vec!["alpha", "beta", "gamma", "delta"]
6827            );
6828        }
6829        std::fs::remove_file(&path).ok();
6830    }
6831
6832    #[test]
6833    fn vlen_append_after_reopen_data_block() {
6834        // Reopen + append into a partial chunk that lives in an extensible-
6835        // array *data block* (chunk index >= idx_blk_elmts). Exercises
6836        // data-block resolution in read_chunk_if_present and write_chunk.
6837        let path = temp_path("vlen_reopen_datablk");
6838        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
6839        {
6840            let file = H5File::create(&path).unwrap();
6841            file.create_appendable_vlen_dataset("strs", 2, None)
6842                .unwrap();
6843            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6844            file.append_vlen_strings("strs", &refs).unwrap();
6845            file.close().unwrap();
6846        }
6847        {
6848            let file = H5File::open_rw(&path).unwrap();
6849            file.append_vlen_strings("strs", &["s9"]).unwrap();
6850            file.close().unwrap();
6851        }
6852        {
6853            let file = H5File::open(&path).unwrap();
6854            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6855            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
6856            assert_eq!(got, want);
6857        }
6858        std::fs::remove_file(&path).ok();
6859    }
6860
6861    #[test]
6862    fn vlen_append_after_reopen_super_block() {
6863        // Reopen + append into a partial chunk whose index falls in an
6864        // extensible-array *super block* (chunk index 244 with the default
6865        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
6866        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
6867        // index block or its direct data blocks, so chunk 244 is reached
6868        // via a super block read from disk). Exercises the ViaSblk branch
6869        // of read_chunk_if_present.
6870        let path = temp_path("vlen_reopen_super");
6871        // 489 strings, chunk size 2 -> chunk 244 holds one string only
6872        // (partially filled) and is flushed to disk on close.
6873        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
6874        {
6875            let file = H5File::create(&path).unwrap();
6876            file.create_appendable_vlen_dataset("strs", 2, None)
6877                .unwrap();
6878            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6879            file.append_vlen_strings("strs", &refs).unwrap();
6880            file.close().unwrap();
6881        }
6882        {
6883            let file = H5File::open_rw(&path).unwrap();
6884            file.append_vlen_strings("strs", &["v489"]).unwrap();
6885            file.close().unwrap();
6886        }
6887        {
6888            let file = H5File::open(&path).unwrap();
6889            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6890            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
6891            assert_eq!(got, want);
6892        }
6893        std::fs::remove_file(&path).ok();
6894    }
6895
6896    #[cfg(feature = "deflate")]
6897    #[test]
6898    fn vlen_append_after_reopen_filtered_data_block() {
6899        // The hardest path: compressed + chunk in a data block + partial
6900        // read-modify-write across a reopen.
6901        let path = temp_path("vlen_reopen_filt_datablk");
6902        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
6903        {
6904            let file = H5File::create(&path).unwrap();
6905            file.create_appendable_vlen_dataset(
6906                "strs",
6907                2,
6908                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6909            )
6910            .unwrap();
6911            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6912            file.append_vlen_strings("strs", &refs).unwrap();
6913            file.close().unwrap();
6914        }
6915        {
6916            let file = H5File::open_rw(&path).unwrap();
6917            file.append_vlen_strings("strs", &["item09"]).unwrap();
6918            file.close().unwrap();
6919        }
6920        {
6921            let file = H5File::open(&path).unwrap();
6922            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6923            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
6924            assert_eq!(got, want);
6925        }
6926        std::fs::remove_file(&path).ok();
6927    }
6928
6929    #[test]
6930    fn group_nx_class_attribute_roundtrip() {
6931        // Non-root groups carry attributes (NeXus `NX_class`) in their
6932        // own object header, and the reader reads them back by path.
6933        let path = temp_path("group_nx_class");
6934        {
6935            let file = H5File::create(&path).unwrap();
6936            let entry = file.create_group("entry").unwrap();
6937            entry.set_attr_string("NX_class", "NXentry").unwrap();
6938            let det = entry.create_group("detector").unwrap();
6939            det.set_attr_string("NX_class", "NXdetector").unwrap();
6940            det.set_attr_numeric("frame_count", &7i32).unwrap();
6941            det.new_dataset::<f32>()
6942                .shape([4])
6943                .create("data")
6944                .unwrap()
6945                .write_raw(&[1.0f32; 4])
6946                .unwrap();
6947            file.close().unwrap();
6948        }
6949        {
6950            let file = H5File::open(&path).unwrap();
6951            let entry = file.root_group().group("entry").unwrap();
6952            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
6953            let det = entry.group("detector").unwrap();
6954            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
6955            let names = det.attr_names().unwrap();
6956            assert!(names.contains(&"NX_class".to_string()));
6957            assert!(names.contains(&"frame_count".to_string()));
6958        }
6959        std::fs::remove_file(&path).ok();
6960    }
6961
6962    #[test]
6963    fn ea_super_block_roundtrip() {
6964        // 2000 chunks span several extensible-array super blocks. Before
6965        // super-block support the writer errored at chunk index 228.
6966        let path = temp_path("ea_super_rt");
6967        {
6968            let file = H5File::create(&path).unwrap();
6969            let ds = file
6970                .new_dataset::<i32>()
6971                .shape([0])
6972                .chunk(&[1])
6973                .max_shape(&[None])
6974                .create("v")
6975                .unwrap();
6976            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
6977            file.close().unwrap();
6978        }
6979        {
6980            let file = H5File::open(&path).unwrap();
6981            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6982            assert_eq!(v.len(), 2000);
6983            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
6984        }
6985        std::fs::remove_file(&path).ok();
6986    }
6987
6988    #[cfg(feature = "deflate")]
6989    #[test]
6990    fn ea_filtered_super_block_roundtrip() {
6991        // Compressed chunks across super blocks.
6992        let path = temp_path("ea_filt_super");
6993        {
6994            let file = H5File::create(&path).unwrap();
6995            let ds = file
6996                .new_dataset::<i32>()
6997                .shape([0])
6998                .chunk(&[1])
6999                .max_shape(&[None])
7000                .deflate(4)
7001                .create("v")
7002                .unwrap();
7003            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
7004            file.close().unwrap();
7005        }
7006        {
7007            let file = H5File::open(&path).unwrap();
7008            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7009            assert_eq!(v, (0..600).collect::<Vec<i32>>());
7010        }
7011        std::fs::remove_file(&path).ok();
7012    }
7013
7014    #[test]
7015    fn ea_super_block_open_append() {
7016        // Reopen a dataset and append chunks that fall in super blocks.
7017        let path = temp_path("ea_super_append");
7018        {
7019            let file = H5File::create(&path).unwrap();
7020            let ds = file
7021                .new_dataset::<i32>()
7022                .shape([0])
7023                .chunk(&[1])
7024                .max_shape(&[None])
7025                .create("v")
7026                .unwrap();
7027            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
7028            file.close().unwrap();
7029        }
7030        {
7031            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7032            let idx = w.dataset_index("v").unwrap();
7033            for c in 300..900u64 {
7034                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
7035            }
7036            w.extend_dataset(idx, &[900]).unwrap();
7037            w.close().unwrap();
7038        }
7039        {
7040            let file = H5File::open(&path).unwrap();
7041            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7042            assert_eq!(v.len(), 900);
7043            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
7044        }
7045        std::fs::remove_file(&path).ok();
7046    }
7047
7048    // Two or more unlimited dimensions select the v2 B-tree index; with a
7049    // filter its records become type 11, carrying each chunk's stored size and
7050    // mask. The payload is highly compressible, so the chunks really are
7051    // stored smaller than the extent — the file would be at least
7052    // 6*8*4 = 192 bytes of raw chunk data otherwise.
7053    #[cfg(feature = "deflate")]
7054    #[test]
7055    fn compressed_multi_unlimited_dataset_roundtrips() {
7056        let path = temp_path("bt2_filtered");
7057        {
7058            let file = H5File::create(&path).unwrap();
7059            let ds = file
7060                .new_dataset::<i32>()
7061                .shape([6, 8])
7062                .chunk(&[2, 4])
7063                .max_shape(&[None, None])
7064                .deflate(6)
7065                .create("d")
7066                .unwrap();
7067            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
7068            // A partial write forces a decompress-patch-recompress of one
7069            // chunk, whose new compressed size may not fit its old block.
7070            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
7071            file.close().unwrap();
7072        }
7073        {
7074            let file = H5File::open(&path).unwrap();
7075            let ds = file.dataset("d").unwrap();
7076            assert_eq!(ds.shape(), vec![6, 8]);
7077            let mut want = vec![7i32; 48];
7078            want[9] = 1;
7079            want[10] = 2;
7080            want[17] = 3;
7081            want[18] = 4;
7082            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
7083        }
7084        std::fs::remove_file(&path).ok();
7085    }
7086
7087    #[test]
7088    fn btree_v2_multi_unlimited_roundtrip() {
7089        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
7090        // index; chunks are written by grid coordinates with write_chunk_at.
7091        let path = temp_path("bt2_multi");
7092        {
7093            let file = H5File::create(&path).unwrap();
7094            let ds = file
7095                .new_dataset::<i32>()
7096                .shape([0, 0])
7097                .chunk(&[2, 2])
7098                .max_shape(&[None, None])
7099                .create("grid")
7100                .unwrap();
7101            assert!(ds.is_chunked());
7102            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
7103            for cr in 0..2usize {
7104                for cc in 0..2usize {
7105                    let mut bytes = Vec::new();
7106                    for i in 0..2usize {
7107                        for j in 0..2usize {
7108                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
7109                            bytes.extend_from_slice(&v.to_le_bytes());
7110                        }
7111                    }
7112                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
7113                }
7114            }
7115            file.close().unwrap();
7116        }
7117        {
7118            let file = H5File::open(&path).unwrap();
7119            let ds = file.dataset("grid").unwrap();
7120            assert_eq!(ds.shape(), vec![4, 4]);
7121            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
7122        }
7123        std::fs::remove_file(&path).ok();
7124    }
7125
7126    #[test]
7127    fn subframe_chunking_roundtrip() {
7128        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
7129        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
7130        // the chunk-grid coordinates.
7131        let path = temp_path("subframe");
7132        {
7133            let file = H5File::create(&path).unwrap();
7134            let ds = file
7135                .new_dataset::<i32>()
7136                .shape([0, 8, 8])
7137                .chunk(&[1, 4, 4])
7138                .max_shape(&[None, Some(8), Some(8)])
7139                .create("v")
7140                .unwrap();
7141            for f in 0..3usize {
7142                for cr in 0..2usize {
7143                    for cc in 0..2usize {
7144                        let mut bytes = Vec::new();
7145                        for i in 0..4usize {
7146                            for j in 0..4usize {
7147                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
7148                                bytes.extend_from_slice(&v.to_le_bytes());
7149                            }
7150                        }
7151                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
7152                    }
7153                }
7154            }
7155            file.close().unwrap();
7156        }
7157        {
7158            let file = H5File::open(&path).unwrap();
7159            let ds = file.dataset("v").unwrap();
7160            assert_eq!(ds.shape(), vec![3, 8, 8]);
7161            assert_eq!(
7162                ds.read_raw::<i32>().unwrap(),
7163                (0..192).collect::<Vec<i32>>()
7164            );
7165        }
7166        std::fs::remove_file(&path).ok();
7167    }
7168
7169    #[test]
7170    fn fill_value_contiguous_roundtrip() {
7171        let path = temp_path("fill_value_contig");
7172        {
7173            let file = H5File::create(&path).unwrap();
7174            let ds = file
7175                .new_dataset::<f32>()
7176                .shape([4])
7177                .fill_value(2.5f32)
7178                .create("data")
7179                .unwrap();
7180            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
7181            file.close().unwrap();
7182        }
7183        // open_append decodes the fill-value message back from the header.
7184        {
7185            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7186            let idx = writer.dataset_index("data").unwrap();
7187            assert_eq!(
7188                writer.ds(idx).lock().fill_value,
7189                Some(2.5f32.to_le_bytes().to_vec())
7190            );
7191        }
7192        // Data still reads back correctly.
7193        {
7194            let file = H5File::open(&path).unwrap();
7195            let ds = file.dataset("data").unwrap();
7196            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
7197        }
7198        std::fs::remove_file(&path).ok();
7199    }
7200
7201    /// Early allocation on a fixed unfiltered shape selects the implicit
7202    /// index, and "implicit" is literal: the file holds no index structure
7203    /// at all, only a version-4 layout message of index type 2 pointing at
7204    /// the run of chunk space the create allocated.
7205    #[test]
7206    fn early_allocation_writes_the_implicit_index() {
7207        let path = temp_path("implicit_index");
7208        {
7209            let file = H5File::create(&path).unwrap();
7210            let ds = file
7211                .new_dataset::<i32>()
7212                .shape([16])
7213                .chunk(&[4])
7214                .early_allocation()
7215                .create("data")
7216                .unwrap();
7217            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7218            file.close().unwrap();
7219        }
7220        let bytes = std::fs::read(&path).unwrap();
7221        for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7222            assert!(
7223                !bytes.windows(4).any(|w| w == magic),
7224                "{} appears in a file whose chunk index is supposed to be no \
7225                 structure at all",
7226                String::from_utf8_lossy(magic)
7227            );
7228        }
7229        {
7230            let file = H5File::open(&path).unwrap();
7231            let ds = file.dataset("data").unwrap();
7232            assert_eq!(
7233                ds.read_raw::<i32>().unwrap(),
7234                (0..16i32).collect::<Vec<_>>()
7235            );
7236        }
7237        std::fs::remove_file(&path).ok();
7238    }
7239
7240    /// Two of the conditions are conditions: an unlimited dimension or a
7241    /// filter each send the dataset to the index libhdf5 would pick
7242    /// instead, early allocation or not. (The third — one whole-dataset
7243    /// chunk — sends it to the single-chunk index instead of Fixed Array;
7244    /// see `one_whole_dataset_chunk_writes_the_single_chunk_index`.)
7245    #[test]
7246    #[cfg(feature = "deflate")]
7247    fn early_allocation_only_picks_implicit_where_libhdf5_does() {
7248        // Every case writes and reads back its data, so a mis-selected index
7249        // shows up as wrong bytes and not just as a different structure.
7250        for (which, magic) in [("unlimited", b"EAHD"), ("filtered", b"FAHD")] {
7251            let path = temp_path("implicit_not");
7252            {
7253                let file = H5File::create(&path).unwrap();
7254                let builder = file.new_dataset::<i32>().shape([16]);
7255                let builder = match which {
7256                    "unlimited" => builder.chunk(&[4]).max_shape(&[None]),
7257                    _ => builder.chunk(&[4]).deflate(6),
7258                };
7259                let ds = builder.early_allocation().create("data").unwrap();
7260                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7261                file.close().unwrap();
7262            }
7263            let bytes = std::fs::read(&path).unwrap();
7264            assert!(
7265                bytes.windows(4).any(|w| w == magic),
7266                "{which}: expected a {} index",
7267                String::from_utf8_lossy(magic)
7268            );
7269            let file = H5File::open(&path).unwrap();
7270            assert_eq!(
7271                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7272                (0..16i32).collect::<Vec<_>>(),
7273                "{which}"
7274            );
7275            std::fs::remove_file(&path).ok();
7276        }
7277    }
7278
7279    /// One whole-dataset chunk always selects the single-chunk index —
7280    /// ahead of Fixed Array, and ahead of Implicit too, whether or not early
7281    /// allocation was requested (`H5D__layout_set_latest_indexing` checks it
7282    /// unconditionally). Like Implicit, "single chunk" is literal: no index
7283    /// structure at all, just the one chunk's address — and, unfiltered and
7284    /// early-allocated, that address exists before anything is written — in
7285    /// the layout message directly.
7286    #[test]
7287    fn one_whole_dataset_chunk_writes_the_single_chunk_index() {
7288        for early in [false, true] {
7289            let path = temp_path("single_chunk_index");
7290            {
7291                let file = H5File::create(&path).unwrap();
7292                let builder = file.new_dataset::<i32>().shape([16]).chunk(&[16]);
7293                let builder = if early {
7294                    builder.early_allocation()
7295                } else {
7296                    builder
7297                };
7298                let ds = builder.create("data").unwrap();
7299                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7300                file.close().unwrap();
7301            }
7302            let bytes = std::fs::read(&path).unwrap();
7303            for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7304                assert!(
7305                    !bytes.windows(4).any(|w| w == magic),
7306                    "early={early}: {} appears in a file whose chunk index is \
7307                     supposed to be no structure at all",
7308                    String::from_utf8_lossy(magic)
7309                );
7310            }
7311            let file = H5File::open(&path).unwrap();
7312            assert_eq!(
7313                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7314                (0..16i32).collect::<Vec<_>>(),
7315                "early={early}"
7316            );
7317            std::fs::remove_file(&path).ok();
7318        }
7319    }
7320
7321    /// An implicitly indexed dataset's chunks all exist from create, so an
7322    /// unwritten one reads back as the fill value — and the fill value is
7323    /// tiled over the whole run at create, not per chunk on demand.
7324    #[test]
7325    fn implicit_index_fills_every_chunk_at_create() {
7326        let path = temp_path("implicit_fill");
7327        {
7328            let file = H5File::create(&path).unwrap();
7329            let ds = file
7330                .new_dataset::<i32>()
7331                .shape([8])
7332                .chunk(&[4])
7333                .early_allocation()
7334                .fill_value(-3i32)
7335                .create("data")
7336                .unwrap();
7337            // Only chunk 0.
7338            let chunk: Vec<u8> = [1i32, 2, 3, 4]
7339                .iter()
7340                .flat_map(|v| v.to_le_bytes())
7341                .collect();
7342            ds.write_chunk(0, &chunk).unwrap();
7343            file.close().unwrap();
7344        }
7345        let file = H5File::open(&path).unwrap();
7346        let ds = file.dataset("data").unwrap();
7347        assert_eq!(
7348            ds.read_raw::<i32>().unwrap(),
7349            vec![1, 2, 3, 4, -3, -3, -3, -3]
7350        );
7351        std::fs::remove_file(&path).ok();
7352    }
7353
7354    /// A reopen has to reconstruct the run's address *and* its length from
7355    /// the layout message alone — there is no index structure to read it
7356    /// back from — or the close would rewrite the dataset as unallocated
7357    /// contiguous storage and drop every byte.
7358    #[test]
7359    fn implicit_index_survives_a_reopen() {
7360        let path = temp_path("implicit_reopen");
7361        {
7362            let file = H5File::create(&path).unwrap();
7363            file.new_dataset::<i32>()
7364                .shape([8])
7365                .chunk(&[4])
7366                .early_allocation()
7367                .create("data")
7368                .unwrap()
7369                .write_raw(&[0i32, 1, 2, 3, 4, 5, 6, 7])
7370                .unwrap();
7371            file.close().unwrap();
7372        }
7373        {
7374            let file = H5File::open_rw(&path).unwrap();
7375            let ds = file.dataset_writer("data").unwrap();
7376            let chunk: Vec<u8> = [10i32, 11, 12, 13]
7377                .iter()
7378                .flat_map(|v| v.to_le_bytes())
7379                .collect();
7380            ds.write_chunk(1, &chunk).unwrap();
7381            file.close().unwrap();
7382        }
7383        let file = H5File::open(&path).unwrap();
7384        assert_eq!(
7385            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7386            vec![0, 1, 2, 3, 10, 11, 12, 13]
7387        );
7388        std::fs::remove_file(&path).ok();
7389    }
7390
7391    #[test]
7392    fn fill_value_chunked_roundtrip() {
7393        let path = temp_path("fill_value_chunked");
7394        {
7395            let file = H5File::create(&path).unwrap();
7396            let ds = file
7397                .new_dataset::<i32>()
7398                .shape([0])
7399                .chunk(&[4])
7400                .max_shape(&[None])
7401                .fill_value(-7i32)
7402                .create("vals")
7403                .unwrap();
7404            ds.append(&[1i32, 2, 3, 4]).unwrap();
7405            file.close().unwrap();
7406        }
7407        {
7408            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7409            let idx = writer.dataset_index("vals").unwrap();
7410            assert_eq!(
7411                writer.ds(idx).lock().fill_value,
7412                Some((-7i32).to_le_bytes().to_vec())
7413            );
7414        }
7415        std::fs::remove_file(&path).ok();
7416    }
7417
7418    #[test]
7419    fn fill_value_read_missing_chunks() {
7420        // A chunked dataset with chunk 1 left unwritten must read that
7421        // gap back as the user-defined fill value, not zero.
7422        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
7423            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
7424        }
7425        let path = temp_path("fill_value_read_missing");
7426        {
7427            let file = H5File::create(&path).unwrap();
7428            let ds = file
7429                .new_dataset::<i32>()
7430                .shape([0])
7431                .chunk(&[2])
7432                .max_shape(&[None])
7433                .fill_value(-1i32)
7434                .create("vals")
7435                .unwrap();
7436            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
7437            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
7438            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
7439            ds.extend(&[6]).unwrap();
7440            file.close().unwrap();
7441        }
7442        {
7443            let file = H5File::open(&path).unwrap();
7444            let ds = file.dataset("vals").unwrap();
7445            let all = ds.read_raw::<i32>().unwrap();
7446            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
7447        }
7448        std::fs::remove_file(&path).ok();
7449    }
7450
7451    #[test]
7452    fn fill_value_partial_chunk_padded_with_fill() {
7453        // A partial trailing chunk flushed at close must pad its unwritten
7454        // tail with the fill value. That pad sits beyond the logical shape,
7455        // so it is verified by scanning the on-disk chunk bytes directly.
7456        let path = temp_path("fill_value_partial_pad");
7457        {
7458            let file = H5File::create(&path).unwrap();
7459            let ds = file
7460                .new_dataset::<i32>()
7461                .shape([0])
7462                .chunk(&[4])
7463                .max_shape(&[None])
7464                .fill_value(-9i32)
7465                .create("vals")
7466                .unwrap();
7467            // 3 of 4 frames -> flushed as a partial chunk on close.
7468            ds.append(&[1i32, 2, 3]).unwrap();
7469            file.close().unwrap();
7470        }
7471        let bytes = std::fs::read(&path).unwrap();
7472        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
7473        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
7474        let pos = bytes
7475            .windows(needle.len())
7476            .position(|w| w == needle)
7477            .expect("chunk data [1,2,3] not found in file");
7478        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
7479        assert_eq!(
7480            pad,
7481            &(-9i32).to_le_bytes(),
7482            "partial chunk tail must be padded with fill value -9, got {:?}",
7483            pad
7484        );
7485        std::fs::remove_file(&path).ok();
7486    }
7487
7488    #[test]
7489    fn vlen_append_after_reopen_preserves_existing() {
7490        // Reopening and appending into a partially-written vlen chunk must
7491        // read-modify-write: the strings already on disk must survive.
7492        let path = temp_path("vlen_append_reopen");
7493        {
7494            let file = H5File::create(&path).unwrap();
7495            file.create_appendable_vlen_dataset("strs", 4, None)
7496                .unwrap();
7497            // 3 of 4 frames -> flushed as a partial chunk on close.
7498            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
7499            file.close().unwrap();
7500        }
7501        {
7502            // Append a 4th string -> partial-chunk write into chunk 0.
7503            let file = H5File::open_rw(&path).unwrap();
7504            file.append_vlen_strings("strs", &["d"]).unwrap();
7505            file.close().unwrap();
7506        }
7507        {
7508            let file = H5File::open(&path).unwrap();
7509            let ds = file.dataset("strs").unwrap();
7510            let got = ds.read_vlen_strings().unwrap();
7511            assert_eq!(
7512                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
7513                vec!["a", "b", "c", "d"]
7514            );
7515        }
7516        std::fs::remove_file(&path).ok();
7517    }
7518
7519    #[test]
7520    fn fill_value_size_mismatch_errors() {
7521        let path = temp_path("fill_value_mismatch");
7522        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
7523        let dt = <f64 as crate::types::H5Type>::hdf5_type();
7524        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
7525        // f64 element size is 8; a 4-byte fill value must be rejected.
7526        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
7527        // The correct width succeeds.
7528        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
7529        writer.close().unwrap();
7530        std::fs::remove_file(&path).ok();
7531    }
7532
7533    #[test]
7534    fn datatype_exposes_class_sign_and_byteorder() {
7535        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
7536        // from f32 (both 4 bytes). datatype() must report the real class and
7537        // signedness so a reader does not have to guess from element_size.
7538        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
7539
7540        let path = temp_path("datatype_accessor");
7541        {
7542            let file = H5File::create(&path).unwrap();
7543            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
7544            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
7545            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
7546            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
7547            file.close().unwrap();
7548        }
7549
7550        let file = H5File::open(&path).unwrap();
7551
7552        match file.dataset("u8d").unwrap().datatype().unwrap() {
7553            DatatypeMessage::FixedPoint {
7554                size,
7555                signed,
7556                byte_order,
7557                ..
7558            } => {
7559                assert_eq!(size, 1);
7560                assert!(!signed, "u8 must be unsigned");
7561                assert_eq!(byte_order, ByteOrder::LittleEndian);
7562            }
7563            other => panic!("expected FixedPoint for u8, got {other:?}"),
7564        }
7565
7566        match file.dataset("i8d").unwrap().datatype().unwrap() {
7567            DatatypeMessage::FixedPoint { size, signed, .. } => {
7568                assert_eq!(size, 1);
7569                assert!(signed, "i8 must be signed");
7570            }
7571            other => panic!("expected FixedPoint for i8, got {other:?}"),
7572        }
7573
7574        match file.dataset("i32d").unwrap().datatype().unwrap() {
7575            DatatypeMessage::FixedPoint { size, signed, .. } => {
7576                assert_eq!(size, 4);
7577                assert!(signed, "i32 must be signed");
7578            }
7579            other => panic!("expected FixedPoint for i32, got {other:?}"),
7580        }
7581
7582        match file.dataset("f32d").unwrap().datatype().unwrap() {
7583            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
7584            other => panic!("expected FloatingPoint for f32, got {other:?}"),
7585        }
7586
7587        std::fs::remove_file(&path).ok();
7588    }
7589
7590    #[test]
7591    fn datatype_in_write_mode_errors() {
7592        let path = temp_path("datatype_write_mode");
7593        let file = H5File::create(&path).unwrap();
7594        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
7595        assert!(ds.datatype().is_err());
7596        std::fs::remove_file(&path).ok();
7597    }
7598
7599    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
7600
7601    /// Extensible-array path: pre-compress with the dataset's pipeline, write
7602    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
7603    /// the data round-trips through the reader unchanged.
7604    #[cfg(feature = "deflate")]
7605    #[test]
7606    fn write_chunk_raw_ea_roundtrip_mask0() {
7607        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7608        let path = temp_path("wcr_ea_mask0");
7609        let original: Vec<i32> = (0..12).collect();
7610        {
7611            let file = H5File::create(&path).unwrap();
7612            let ds = file
7613                .new_dataset::<i32>()
7614                .shape([0])
7615                .chunk(&[4])
7616                .max_shape(&[None])
7617                .deflate(4)
7618                .create("v")
7619                .unwrap();
7620            assert!(ds.is_chunked());
7621            let pipeline = FilterPipeline::deflate(4);
7622            for c in 0..3usize {
7623                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7624                    .iter()
7625                    .flat_map(|v| v.to_le_bytes())
7626                    .collect();
7627                let compressed = apply_filters(&pipeline, &raw).unwrap();
7628                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7629            }
7630            ds.set_extent(&[12]).unwrap();
7631            file.close().unwrap();
7632        }
7633        {
7634            let file = H5File::open(&path).unwrap();
7635            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7636            assert_eq!(v, original);
7637        }
7638        std::fs::remove_file(&path).ok();
7639    }
7640
7641    /// Fixed-array path (all dimensions bounded): same verbatim write through
7642    /// the linear-index dispatch, round-tripped through the reader.
7643    #[cfg(feature = "deflate")]
7644    #[test]
7645    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
7646        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7647        let path = temp_path("wcr_fa_mask0");
7648        let original: Vec<i32> = (0..12).collect();
7649        {
7650            let file = H5File::create(&path).unwrap();
7651            let ds = file
7652                .new_dataset::<i32>()
7653                .shape([12])
7654                .chunk(&[4])
7655                .deflate(4)
7656                .create("v")
7657                .unwrap();
7658            assert!(ds.is_chunked());
7659            let pipeline = FilterPipeline::deflate(4);
7660            for c in 0..3usize {
7661                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7662                    .iter()
7663                    .flat_map(|v| v.to_le_bytes())
7664                    .collect();
7665                let compressed = apply_filters(&pipeline, &raw).unwrap();
7666                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7667            }
7668            file.close().unwrap();
7669        }
7670        {
7671            let file = H5File::open(&path).unwrap();
7672            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7673            assert_eq!(v, original);
7674        }
7675        std::fs::remove_file(&path).ok();
7676    }
7677
7678    /// The caller-supplied filter_mask must reach the on-disk filtered index
7679    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
7680    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
7681    /// the extensible-array filtered entry to read the mask back at the format
7682    /// level (independent of the data reader's mask handling).
7683    #[cfg(feature = "deflate")]
7684    #[test]
7685    fn write_chunk_raw_records_filter_mask() {
7686        let path = temp_path("wcr_records_mask");
7687        let raw: Vec<u8> = [10i32, 20, 30, 40]
7688            .iter()
7689            .flat_map(|v| v.to_le_bytes())
7690            .collect();
7691        assert_eq!(raw.len(), 16);
7692        {
7693            let file = H5File::create(&path).unwrap();
7694            let ds = file
7695                .new_dataset::<i32>()
7696                .shape([0])
7697                .chunk(&[4])
7698                .max_shape(&[None])
7699                .deflate(4)
7700                .create("v")
7701                .unwrap();
7702            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
7703            // chunk is stored uncompressed (its raw bytes).
7704            ds.write_chunk_raw(0, &raw, 1).unwrap();
7705            ds.set_extent(&[4]).unwrap();
7706            file.close().unwrap();
7707        }
7708        // Reopen the writer; open_append decodes the filtered index block from
7709        // disk, so the entry reflects exactly what was committed.
7710        {
7711            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7712            let idx = w.dataset_index("v").unwrap();
7713            let ds = w.ds(idx);
7714            let m = ds.lock();
7715            let entry = &m
7716                .chunked
7717                .as_ref()
7718                .unwrap()
7719                .filt_iblk
7720                .as_ref()
7721                .unwrap()
7722                .elements[0];
7723            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
7724            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
7725        }
7726        std::fs::remove_file(&path).ok();
7727    }
7728
7729    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
7730    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
7731    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
7732    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
7733    /// A chunk whose stored stream decodes to less than its image places no
7734    /// run at all: the whole chunk reads as the fill value, whether the read
7735    /// laid the fill down first (a plan that leaves output uncovered — here the
7736    /// unallocated middle chunk) or fills only what nothing wrote. The decode
7737    /// writes into the output image itself, so the bytes a short image leaves
7738    /// behind are the ones this covers.
7739    #[cfg(feature = "deflate")]
7740    #[test]
7741    fn a_chunk_that_decodes_short_of_its_image_reads_as_fill() {
7742        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7743        let path = temp_path("short_chunk_image_is_fill");
7744        let pipeline = FilterPipeline::deflate(4);
7745        {
7746            let file = H5File::create(&path).unwrap();
7747            let ds = file
7748                .new_dataset::<i32>()
7749                .shape([0])
7750                .chunk(&[4])
7751                .max_shape(&[None])
7752                .fill_value(-1i32)
7753                .deflate(4)
7754                .create("v")
7755                .unwrap();
7756            // Chunk 0 carries two elements where its image wants four.
7757            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7758            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7759                .unwrap();
7760            // Chunk 1 is never written; chunk 2 is whole.
7761            let whole: Vec<u8> = [9i32, 10, 11, 12]
7762                .iter()
7763                .flat_map(|v| v.to_le_bytes())
7764                .collect();
7765            ds.write_chunk_raw(2, &apply_filters(&pipeline, &whole).unwrap(), 0)
7766                .unwrap();
7767            ds.set_extent(&[12]).unwrap();
7768            file.close().unwrap();
7769        }
7770        {
7771            let file = H5File::open(&path).unwrap();
7772            let ds = file.dataset("v").unwrap();
7773            assert_eq!(
7774                ds.read_raw::<i32>().unwrap(),
7775                vec![-1, -1, -1, -1, -1, -1, -1, -1, 9, 10, 11, 12]
7776            );
7777            // The same verdict when the plan covers every output byte, so no
7778            // fill goes down first: chunks 0 and 2 alone.
7779            assert_eq!(ds.read_slice::<i32>(&[0], &[4]).unwrap(), vec![-1; 4]);
7780            assert_eq!(
7781                ds.read_slice::<i32>(&[8], &[4]).unwrap(),
7782                vec![9, 10, 11, 12]
7783            );
7784        }
7785        std::fs::remove_file(&path).ok();
7786    }
7787
7788    /// The staged spelling of the case above: a selection that takes only part
7789    /// of the short chunk decodes it into a buffer sized from the layout, and
7790    /// what that buffer holds past the stream is cut off rather than kept — a
7791    /// run inside the decoded bytes is real data, a run reaching past them is
7792    /// fill.
7793    #[cfg(feature = "deflate")]
7794    #[test]
7795    fn a_staged_chunk_carries_only_what_its_stream_decoded() {
7796        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7797        let path = temp_path("short_chunk_image_staged");
7798        let pipeline = FilterPipeline::deflate(4);
7799        {
7800            let file = H5File::create(&path).unwrap();
7801            let ds = file
7802                .new_dataset::<i32>()
7803                .shape([0])
7804                .chunk(&[4])
7805                .max_shape(&[None])
7806                .fill_value(-1i32)
7807                .deflate(4)
7808                .create("v")
7809                .unwrap();
7810            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7811            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7812                .unwrap();
7813            ds.set_extent(&[4]).unwrap();
7814            file.close().unwrap();
7815        }
7816        {
7817            let file = H5File::open(&path).unwrap();
7818            let ds = file.dataset("v").unwrap();
7819            // Inside the decoded bytes.
7820            assert_eq!(ds.read_slice::<i32>(&[0], &[2]).unwrap(), vec![7, 8]);
7821            // Straddling their end: the run is not placed at all.
7822            assert_eq!(ds.read_slice::<i32>(&[1], &[2]).unwrap(), vec![-1, -1]);
7823        }
7824        std::fs::remove_file(&path).ok();
7825    }
7826
7827    #[cfg(feature = "deflate")]
7828    #[test]
7829    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
7830        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7831        let path = temp_path("wcr_ea_per_chunk_mask");
7832        let original: Vec<i32> = (0..8).collect();
7833        let pipeline = FilterPipeline::deflate(4);
7834        {
7835            let file = H5File::create(&path).unwrap();
7836            let ds = file
7837                .new_dataset::<i32>()
7838                .shape([0])
7839                .chunk(&[4])
7840                .max_shape(&[None])
7841                .deflate(4)
7842                .create("v")
7843                .unwrap();
7844            let raw0: Vec<u8> = original[0..4]
7845                .iter()
7846                .flat_map(|v| v.to_le_bytes())
7847                .collect();
7848            // chunk 0: compressed through the pipeline, mask 0.
7849            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7850                .unwrap();
7851            let raw1: Vec<u8> = original[4..8]
7852                .iter()
7853                .flat_map(|v| v.to_le_bytes())
7854                .collect();
7855            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
7856            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7857            ds.set_extent(&[8]).unwrap();
7858            file.close().unwrap();
7859        }
7860        {
7861            let file = H5File::open(&path).unwrap();
7862            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7863            assert_eq!(v, original);
7864        }
7865        std::fs::remove_file(&path).ok();
7866    }
7867
7868    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
7869    /// compressed/raw chunks as the EA case, through the fixed-array index.
7870    #[cfg(feature = "deflate")]
7871    #[test]
7872    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
7873        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7874        let path = temp_path("wcr_fa_per_chunk_mask");
7875        let original: Vec<i32> = (0..8).collect();
7876        let pipeline = FilterPipeline::deflate(4);
7877        {
7878            let file = H5File::create(&path).unwrap();
7879            let ds = file
7880                .new_dataset::<i32>()
7881                .shape([8])
7882                .chunk(&[4])
7883                .deflate(4)
7884                .create("v")
7885                .unwrap();
7886            let raw0: Vec<u8> = original[0..4]
7887                .iter()
7888                .flat_map(|v| v.to_le_bytes())
7889                .collect();
7890            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7891                .unwrap();
7892            let raw1: Vec<u8> = original[4..8]
7893                .iter()
7894                .flat_map(|v| v.to_le_bytes())
7895                .collect();
7896            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7897            file.close().unwrap();
7898        }
7899        {
7900            let file = H5File::open(&path).unwrap();
7901            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7902            assert_eq!(v, original);
7903        }
7904        std::fs::remove_file(&path).ok();
7905    }
7906
7907    /// An unfiltered chunk index has no slot for a stored size or mask, so a
7908    /// direct chunk write must be rejected rather than silently dropping them.
7909    #[test]
7910    fn write_chunk_raw_rejects_unfiltered() {
7911        let path = temp_path("wcr_unfiltered");
7912        let file = H5File::create(&path).unwrap();
7913        let ds = file
7914            .new_dataset::<i32>()
7915            .shape([0])
7916            .chunk(&[4])
7917            .max_shape(&[None])
7918            .create("v")
7919            .unwrap();
7920        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7921        assert!(
7922            err.to_string().contains("filtered dataset"),
7923            "expected a filtered-dataset error, got: {err}"
7924        );
7925        std::fs::remove_file(&path).ok();
7926    }
7927
7928    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
7929    /// index to mean anything against, so the linear entry point points the
7930    /// caller at the coordinate-addressed one rather than guessing a grid.
7931    #[test]
7932    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
7933        let path = temp_path("wcr_btree2");
7934        let file = H5File::create(&path).unwrap();
7935        let ds = file
7936            .new_dataset::<i32>()
7937            .shape([0, 0])
7938            .chunk(&[2, 2])
7939            .max_shape(&[None, None])
7940            .create("grid")
7941            .unwrap();
7942        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7943        assert!(
7944            err.to_string().contains("write_chunk_raw_at"),
7945            "expected a pointer to the coordinate form, got: {err}"
7946        );
7947        std::fs::remove_file(&path).ok();
7948    }
7949
7950    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
7951    /// and the type-11 record carries their size and the caller's mask, so a
7952    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
7953    /// bytes while one written compressed (mask 0) is decompressed.
7954    #[cfg(feature = "deflate")]
7955    #[test]
7956    fn write_chunk_raw_at_round_trips_on_btree_v2() {
7957        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7958
7959        let path = temp_path("wcr_at_btree2");
7960        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
7961        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
7962        {
7963            let file = H5File::create(&path).unwrap();
7964            let ds = file
7965                .new_dataset::<i32>()
7966                .shape([0, 0])
7967                .chunk(&[2, 2])
7968                .max_shape(&[None, None])
7969                .deflate(6)
7970                .create("grid")
7971                .unwrap();
7972            let pipeline = FilterPipeline::deflate(6);
7973            // Chunk (0,0): pipeline already applied upstream, mask 0.
7974            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
7975                .unwrap();
7976            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
7977            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
7978            file.close().unwrap();
7979        }
7980        let file = H5File::open(&path).unwrap();
7981        let ds = file.dataset("grid").unwrap();
7982        assert_eq!(ds.shape(), vec![4, 4]);
7983        let all = ds.read_raw::<i32>().unwrap();
7984        // Chunk (0,0) occupies rows 0..2, columns 0..2.
7985        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
7986        // Chunk (1,1) occupies rows 2..4, columns 2..4.
7987        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
7988        drop(file);
7989        std::fs::remove_file(&path).ok();
7990    }
7991
7992    /// The coordinate form is not BT2-only: it addresses an extensible- or
7993    /// fixed-array dataset's grid just as well, and records the same mask.
7994    #[cfg(feature = "deflate")]
7995    #[test]
7996    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
7997        for (label, max_shape) in [
7998            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
7999            ("wcr_at_fa", None),
8000        ] {
8001            let path = temp_path(label);
8002            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
8003            {
8004                let file = H5File::create(&path).unwrap();
8005                let mut b = file
8006                    .new_dataset::<i32>()
8007                    .shape([4usize, 4])
8008                    .chunk(&[2, 4])
8009                    .deflate(6);
8010                if let Some(ref ms) = max_shape {
8011                    b = b.max_shape(ms);
8012                }
8013                let ds = b.create("grid").unwrap();
8014                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
8015                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
8016                file.close().unwrap();
8017            }
8018            let file = H5File::open(&path).unwrap();
8019            let ds = file.dataset("grid").unwrap();
8020            let all = ds.read_raw::<i32>().unwrap();
8021            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
8022            drop(file);
8023            std::fs::remove_file(&path).ok();
8024        }
8025    }
8026
8027    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
8028    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
8029    /// chunk gives chunk_size_len = 2 (max 65535).
8030    #[cfg(feature = "deflate")]
8031    #[test]
8032    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
8033        let path = temp_path("wcr_at_oversized");
8034        let file = H5File::create(&path).unwrap();
8035        let ds = file
8036            .new_dataset::<i32>()
8037            .shape([0, 0])
8038            .chunk(&[1, 1])
8039            .max_shape(&[None, None])
8040            .deflate(4)
8041            .create("grid")
8042            .unwrap();
8043        let err = ds
8044            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
8045            .unwrap_err();
8046        assert!(
8047            err.to_string().contains("does not fit"),
8048            "expected a chunk-size-field overflow error, got: {err}"
8049        );
8050        std::fs::remove_file(&path).ok();
8051    }
8052
8053    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
8054    /// the same reason the array indexes reject a direct write.
8055    #[test]
8056    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
8057        let path = temp_path("wcr_at_unfiltered");
8058        let file = H5File::create(&path).unwrap();
8059        let ds = file
8060            .new_dataset::<i32>()
8061            .shape([0, 0])
8062            .chunk(&[2, 2])
8063            .max_shape(&[None, None])
8064            .create("grid")
8065            .unwrap();
8066        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
8067        assert!(
8068            err.to_string().contains("filtered dataset"),
8069            "expected a filtered-dataset error, got: {err}"
8070        );
8071        std::fs::remove_file(&path).ok();
8072    }
8073
8074    /// A stored size that does not fit the index's chunk-size field must error
8075    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
8076    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
8077    /// a 70000-byte stored chunk overflows it.
8078    #[cfg(feature = "deflate")]
8079    #[test]
8080    fn write_chunk_raw_rejects_oversized_chunk() {
8081        let path = temp_path("wcr_oversized");
8082        let file = H5File::create(&path).unwrap();
8083        let ds = file
8084            .new_dataset::<i32>()
8085            .shape([0])
8086            .chunk(&[1])
8087            .max_shape(&[None])
8088            .deflate(4)
8089            .create("v")
8090            .unwrap();
8091        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
8092        assert!(
8093            err.to_string().contains("does not fit"),
8094            "expected a chunk-size-field overflow error, got: {err}"
8095        );
8096        std::fs::remove_file(&path).ok();
8097    }
8098
8099    // ---- issue #5: runtime-width fixed-string reading ----------------------
8100
8101    use crate::format::messages::datatype::{CompoundMember, DatatypeMessage};
8102
8103    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
8104    /// element images, optionally chunked and deflated.
8105    fn write_fixed_string_dataset(
8106        path: &std::path::Path,
8107        dt: DatatypeMessage,
8108        width: usize,
8109        elems: &[&[u8]],
8110        compressed: bool,
8111    ) {
8112        let mut raw = Vec::with_capacity(elems.len() * width);
8113        for e in elems {
8114            assert!(e.len() <= width);
8115            raw.extend_from_slice(e);
8116            raw.resize(raw.len() + (width - e.len()), 0);
8117        }
8118        let file = H5File::create(path).unwrap();
8119        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
8120        if compressed {
8121            b = b.chunk(&[2]).deflate(6);
8122        }
8123        let ds = b.create("labels").unwrap();
8124        ds.write_raw_bytes(&raw).unwrap();
8125        file.close().unwrap();
8126    }
8127
8128    /// The width is whatever the file says, so one call reads a 24-byte label
8129    /// column and a 100-byte one. Producers like VASP pick it per dataset.
8130    #[test]
8131    fn read_strings_handles_any_fixed_width() {
8132        for width in [4usize, 24, 100] {
8133            let path = temp_path(&format!("fixed_str_{width}"));
8134            write_fixed_string_dataset(
8135                &path,
8136                DatatypeMessage::fixed_string(width as u32),
8137                width,
8138                &[b"ab", b"cde", b""],
8139                false,
8140            );
8141            let file = H5File::open(&path).unwrap();
8142            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8143            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
8144            std::fs::remove_file(&path).ok();
8145        }
8146    }
8147
8148    /// Each padding rule decides where the value ends. Null-terminated and
8149    /// null-padded both stop at the first NUL and ignore the bytes after it;
8150    /// space-padded strips only a tail of spaces, so an embedded NUL is
8151    /// content there.
8152    ///
8153    /// Checked against libhdf5 1.14.6: reading this same `"ab\0X\0\0"`
8154    /// null-padded element into a wider null-terminated destination gives
8155    /// `"ab"`, and reading a space-padded `"a\0b     "` gives `"a\0b"` —
8156    /// `H5T__conv_s_s` runs the same `!s[nchars]` loop for both null rules.
8157    #[test]
8158    fn read_strings_honors_every_padding_rule() {
8159        // "ab" then a NUL then trailing junk that both null rules must drop.
8160        let elem: &[u8] = b"ab\0X\0\0";
8161        for (padding, want) in [(0u8, "ab"), (1, "ab")] {
8162            let path = temp_path(&format!("fixed_pad_{padding}"));
8163            write_fixed_string_dataset(
8164                &path,
8165                DatatypeMessage::FixedString {
8166                    size: 6,
8167                    padding,
8168                    charset: 0,
8169                },
8170                6,
8171                &[elem],
8172                false,
8173            );
8174            let file = H5File::open(&path).unwrap();
8175            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8176            assert_eq!(got, vec![want.to_string()], "padding {padding}");
8177            std::fs::remove_file(&path).ok();
8178        }
8179        // Space-padded keeps interior spaces and strips only the tail.
8180        let path = temp_path("fixed_pad_2");
8181        write_fixed_string_dataset(
8182            &path,
8183            DatatypeMessage::FixedString {
8184                size: 8,
8185                padding: 2,
8186                charset: 0,
8187            },
8188            8,
8189            &[b"a b     "],
8190            false,
8191        );
8192        let file = H5File::open(&path).unwrap();
8193        assert_eq!(
8194            file.dataset("labels").unwrap().read_strings().unwrap(),
8195            vec!["a b".to_string()]
8196        );
8197        std::fs::remove_file(&path).ok();
8198
8199        // ... and an embedded NUL, which no space rule marks as an end.
8200        let path = temp_path("fixed_pad_2_nul");
8201        write_fixed_string_dataset(
8202            &path,
8203            DatatypeMessage::FixedString {
8204                size: 8,
8205                padding: 2,
8206                charset: 0,
8207            },
8208            8,
8209            &[b"a\0b     "],
8210            false,
8211        );
8212        let file = H5File::open(&path).unwrap();
8213        assert_eq!(
8214            file.dataset("labels").unwrap().read_strings().unwrap(),
8215            vec!["a\0b".to_string()]
8216        );
8217        std::fs::remove_file(&path).ok();
8218    }
8219
8220    /// A reserved padding or character-set code is an error naming the element,
8221    /// not a guess.
8222    #[test]
8223    fn read_strings_rejects_reserved_datatype_codes() {
8224        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
8225            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
8226            write_fixed_string_dataset(
8227                &path,
8228                DatatypeMessage::FixedString {
8229                    size: 4,
8230                    padding,
8231                    charset,
8232                },
8233                4,
8234                &[b"ab"],
8235                false,
8236            );
8237            let file = H5File::open(&path).unwrap();
8238            let err = file
8239                .dataset("labels")
8240                .unwrap()
8241                .read_strings()
8242                .unwrap_err()
8243                .to_string();
8244            assert!(err.contains(want), "got: {err}");
8245            std::fs::remove_file(&path).ok();
8246        }
8247    }
8248
8249    /// The typed read paths reinterpret the element image, so the stored order
8250    /// has to be the host's first. A scalar is swapped; a composite cannot be
8251    /// (its members have their own orders and offsets) and is refused.
8252    #[test]
8253    fn to_host_byte_order_converts_scalars_and_refuses_composites() {
8254        use crate::dataset::{to_host_byte_order, HOST_BYTE_ORDER};
8255        use crate::format::messages::datatype::ByteOrder;
8256
8257        let foreign = match HOST_BYTE_ORDER {
8258            ByteOrder::LittleEndian => ByteOrder::BigEndian,
8259            ByteOrder::BigEndian => ByteOrder::LittleEndian,
8260        };
8261        let int = |order, size| DatatypeMessage::FixedPoint {
8262            size,
8263            byte_order: order,
8264            signed: false,
8265            bit_offset: 0,
8266            bit_precision: (size * 8) as u16,
8267        };
8268
8269        // Foreign order: each element is reversed, elementwise.
8270        let mut buf = [1u8, 2, 3, 4, 5, 6, 7, 8];
8271        to_host_byte_order(&mut buf, &int(foreign, 4), 4).unwrap();
8272        assert_eq!(buf, [4, 3, 2, 1, 8, 7, 6, 5]);
8273
8274        // Host order: untouched.
8275        let mut buf = [1u8, 2, 3, 4];
8276        to_host_byte_order(&mut buf, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8277        assert_eq!(buf, [1, 2, 3, 4]);
8278
8279        // One byte wide: no order to convert.
8280        let mut buf = [1u8, 2, 3, 4];
8281        to_host_byte_order(&mut buf, &int(foreign, 1), 1).unwrap();
8282        assert_eq!(buf, [1, 2, 3, 4]);
8283
8284        // An enum stores its values in its base type's order.
8285        let mut buf = [1u8, 2];
8286        let enumeration = DatatypeMessage::Enum {
8287            base: Box::new(int(foreign, 2)),
8288            members: Vec::new(),
8289        };
8290        to_host_byte_order(&mut buf, &enumeration, 2).unwrap();
8291        assert_eq!(buf, [2, 1]);
8292
8293        // A string has no byte order at all.
8294        let mut buf = *b"abcd";
8295        to_host_byte_order(&mut buf, &DatatypeMessage::fixed_string(4), 4).unwrap();
8296        assert_eq!(&buf, b"abcd");
8297
8298        // A compound whose members are all host-order is reinterpretable.
8299        let compound = |order| DatatypeMessage::Compound {
8300            size: 4,
8301            members: vec![CompoundMember {
8302                name: "x".into(),
8303                offset: 0,
8304                datatype: int(order, 4),
8305            }],
8306        };
8307        let mut buf = [1u8, 2, 3, 4];
8308        to_host_byte_order(&mut buf, &compound(HOST_BYTE_ORDER), 4).unwrap();
8309        assert_eq!(buf, [1, 2, 3, 4]);
8310
8311        // One that is not says so, rather than handing back the raw bytes.
8312        let mut buf = [1u8, 2, 3, 4];
8313        let err = to_host_byte_order(&mut buf, &compound(foreign), 4)
8314            .expect_err("a foreign-order compound was reinterpreted")
8315            .to_string();
8316        assert!(err.contains("read_raw_bytes"), "got: {err}");
8317        assert_eq!(buf, [1, 2, 3, 4], "the refused image is left alone");
8318    }
8319
8320    /// The write direction answers for exactly the types the read direction
8321    /// does — same classifier — and borrows the caller's bytes whenever the
8322    /// declared order is already the host's.
8323    #[test]
8324    fn to_stored_byte_order_converts_scalars_and_refuses_composites() {
8325        use crate::dataset::{to_stored_byte_order, FOREIGN_BYTE_ORDER, HOST_BYTE_ORDER};
8326        use std::borrow::Cow;
8327
8328        let int = |order, size| DatatypeMessage::FixedPoint {
8329            size,
8330            byte_order: order,
8331            signed: false,
8332            bit_offset: 0,
8333            bit_precision: (size * 8) as u16,
8334        };
8335
8336        // Declared foreign: each element is reversed on the way out.
8337        let host = [1u8, 2, 3, 4, 5, 6, 7, 8];
8338        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 4), 4).unwrap();
8339        assert_eq!(&*stored, &[4, 3, 2, 1, 8, 7, 6, 5]);
8340        assert!(matches!(stored, Cow::Owned(_)), "a swap needs its own copy");
8341
8342        // Declared host order: handed through without a copy.
8343        let stored = to_stored_byte_order(&host, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8344        assert!(matches!(stored, Cow::Borrowed(_)), "no copy without a swap");
8345        assert_eq!(&*stored, &host);
8346
8347        // One byte wide: no order to lay out.
8348        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 1), 1).unwrap();
8349        assert_eq!(&*stored, &host);
8350
8351        // An enum stores its values in its base type's order.
8352        let enumeration = DatatypeMessage::Enum {
8353            base: Box::new(int(FOREIGN_BYTE_ORDER, 2)),
8354            members: Vec::new(),
8355        };
8356        let stored = to_stored_byte_order(&[1u8, 2], &enumeration, 2).unwrap();
8357        assert_eq!(&*stored, &[2, 1]);
8358
8359        // A compound cannot be laid out as a unit; one that declares the
8360        // foreign order for a member is refused, not written host-order.
8361        let compound = |order| DatatypeMessage::Compound {
8362            size: 4,
8363            members: vec![CompoundMember {
8364                name: "x".into(),
8365                offset: 0,
8366                datatype: int(order, 4),
8367            }],
8368        };
8369        let stored = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(HOST_BYTE_ORDER), 4).unwrap();
8370        assert_eq!(&*stored, &[1, 2, 3, 4]);
8371        let err = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(FOREIGN_BYTE_ORDER), 4)
8372            .expect_err("a foreign-order compound was written from host bytes")
8373            .to_string();
8374        assert!(err.contains("write_raw_bytes"), "got: {err}");
8375    }
8376
8377    /// The declared character set is enforced: a byte that cannot be decoded is
8378    /// an error naming the element, and the lossy call is what accepts the file
8379    /// instead of a silent substitution here.
8380    #[test]
8381    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
8382        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
8383        // in one that declares UTF-8.
8384        for (charset, bytes, want) in [
8385            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
8386            (1, b"a\xff".as_slice(), "not valid UTF-8"),
8387        ] {
8388            let path = temp_path(&format!("fixed_charset_{charset}"));
8389            write_fixed_string_dataset(
8390                &path,
8391                DatatypeMessage::FixedString {
8392                    size: 6,
8393                    padding: 1,
8394                    charset,
8395                },
8396                6,
8397                &[b"ok", bytes],
8398                false,
8399            );
8400            let file = H5File::open(&path).unwrap();
8401            let ds = file.dataset("labels").unwrap();
8402            let err = ds.read_strings().unwrap_err().to_string();
8403            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
8404            let lossy = ds.read_strings_lossy().unwrap();
8405            assert_eq!(lossy[0], "ok");
8406            assert_eq!(
8407                lossy[1].chars().next().unwrap(),
8408                if charset == 0 { 'c' } else { 'a' }
8409            );
8410            std::fs::remove_file(&path).ok();
8411        }
8412    }
8413
8414    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
8415    /// split a character.
8416    #[test]
8417    fn read_strings_reads_utf8_fixed_strings() {
8418        let path = temp_path("fixed_utf8");
8419        write_fixed_string_dataset(
8420            &path,
8421            DatatypeMessage::fixed_string_utf8(12),
8422            12,
8423            &["héllo".as_bytes(), "안녕".as_bytes()],
8424            false,
8425        );
8426        let file = H5File::open(&path).unwrap();
8427        assert_eq!(
8428            file.dataset("labels").unwrap().read_strings().unwrap(),
8429            vec!["héllo".to_string(), "안녕".to_string()]
8430        );
8431        std::fs::remove_file(&path).ok();
8432    }
8433
8434    /// The decode sits on the decoded raw-data path, so a chunked and deflated
8435    /// dataset reads the same as a contiguous one.
8436    #[cfg(feature = "deflate")]
8437    #[test]
8438    fn read_strings_reads_a_compressed_fixed_string_dataset() {
8439        let path = temp_path("fixed_str_deflate");
8440        write_fixed_string_dataset(
8441            &path,
8442            DatatypeMessage::fixed_string(16),
8443            16,
8444            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
8445            true,
8446        );
8447        let file = H5File::open(&path).unwrap();
8448        assert_eq!(
8449            file.dataset("labels").unwrap().read_strings().unwrap(),
8450            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
8451        );
8452        std::fs::remove_file(&path).ok();
8453    }
8454
8455    /// One call covers both string datatypes, so a caller need not branch on
8456    /// which one the file used.
8457    #[test]
8458    fn read_strings_also_reads_variable_length_strings() {
8459        let path = temp_path("read_strings_vlen");
8460        {
8461            let file = H5File::create(&path).unwrap();
8462            file.write_vlen_strings("names", &["alpha", "", "안녕"])
8463                .unwrap();
8464            file.close().unwrap();
8465        }
8466        let file = H5File::open(&path).unwrap();
8467        assert_eq!(
8468            file.dataset("names").unwrap().read_strings().unwrap(),
8469            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
8470        );
8471        std::fs::remove_file(&path).ok();
8472    }
8473
8474    /// A file declaring a zero-width fixed string is an error, not the panic
8475    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
8476    /// test patches the width in the encoded datatype message down to zero and
8477    /// re-stamps the object header's checksum over the result.
8478    #[test]
8479    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
8480        use crate::format::checksum::checksum_metadata;
8481        use crate::format::object_header::OHDR_SIGNATURE;
8482
8483        let path = temp_path("fixed_str_zero_width");
8484        write_fixed_string_dataset(
8485            &path,
8486            DatatypeMessage::fixed_string(37),
8487            37,
8488            &[b"ab", b"cd"],
8489            false,
8490        );
8491
8492        // Version 1 string datatype: class|version, padding|charset, two
8493        // reserved bytes, then the width as a little-endian u32. The width is
8494        // 37 so the eight bytes occur once in the file.
8495        let mut bytes = std::fs::read(&path).unwrap();
8496        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
8497        let at = bytes
8498            .windows(needle.len())
8499            .position(|w| w == needle)
8500            .expect("encoded fixed-string datatype message");
8501        assert!(
8502            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
8503            "the datatype message pattern is not unique in the file"
8504        );
8505
8506        // The enclosing v2 object header ends in a checksum over everything
8507        // from its signature onwards; find the offset where the stored value
8508        // still agrees, so the patched header can be re-stamped there.
8509        let ohdr = bytes[..at]
8510            .windows(4)
8511            .rposition(|w| w == OHDR_SIGNATURE)
8512            .expect("enclosing object header");
8513        let cksum_at = (at + needle.len()..bytes.len() - 4)
8514            .find(|&e| {
8515                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
8516                    == checksum_metadata(&bytes[ohdr..e])
8517            })
8518            .expect("object header checksum");
8519
8520        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
8521        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
8522        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
8523        std::fs::write(&path, &bytes).unwrap();
8524
8525        let file = H5File::open(&path).unwrap();
8526        let err = file
8527            .dataset("labels")
8528            .unwrap()
8529            .read_strings()
8530            .unwrap_err()
8531            .to_string();
8532        assert!(err.contains("zero width"), "got: {err}");
8533        std::fs::remove_file(&path).ok();
8534    }
8535
8536    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
8537    #[test]
8538    fn read_strings_rejects_a_non_string_dataset() {
8539        let path = temp_path("read_strings_numeric");
8540        {
8541            let file = H5File::create(&path).unwrap();
8542            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8543            ds.write_raw(&[1i32, 2, 3]).unwrap();
8544            file.close().unwrap();
8545        }
8546        let file = H5File::open(&path).unwrap();
8547        let err = file
8548            .dataset("nums")
8549            .unwrap()
8550            .read_strings()
8551            .unwrap_err()
8552            .to_string();
8553        assert!(err.contains("only for string datasets"), "got: {err}");
8554        std::fs::remove_file(&path).ok();
8555    }
8556
8557    // ---- issue #6: random updates to vlen string datasets ------------------
8558
8559    /// One element changes; the extent and every other element stay as they
8560    /// were, on a contiguous vlen dataset.
8561    #[test]
8562    fn write_vlen_strings_slice_replaces_one_element() {
8563        let path = temp_path("vlen_slice_contig");
8564        {
8565            let file = H5File::create(&path).unwrap();
8566            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8567                .unwrap();
8568            file.close().unwrap();
8569        }
8570        {
8571            let file = H5File::open_rw(&path).unwrap();
8572            file.dataset_writer("notes")
8573                .unwrap()
8574                .write_vlen_strings_slice(1, &["replacement"])
8575                .unwrap();
8576            file.close().unwrap();
8577        }
8578        let file = H5File::open(&path).unwrap();
8579        let ds = file.dataset("notes").unwrap();
8580        assert_eq!(ds.shape(), vec![4]);
8581        assert_eq!(
8582            ds.read_vlen_strings().unwrap(),
8583            vec!["a", "replacement", "c", "d"]
8584        );
8585        std::fs::remove_file(&path).ok();
8586    }
8587
8588    /// The same on an appendable chunked dataset, across a reopen, over a range
8589    /// that spans a chunk boundary.
8590    #[test]
8591    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
8592        let path = temp_path("vlen_slice_chunked");
8593        {
8594            let file = H5File::create(&path).unwrap();
8595            file.create_appendable_vlen_dataset("notes", 2, None)
8596                .unwrap();
8597            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
8598            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
8599            file.append_vlen_strings("notes", &refs).unwrap();
8600            file.close().unwrap();
8601        }
8602        {
8603            // Elements 1..4 cross the 2-element chunk boundary twice.
8604            let file = H5File::open_rw(&path).unwrap();
8605            file.dataset_writer("notes")
8606                .unwrap()
8607                .write_vlen_strings_slice(1, &["x", "y", "z"])
8608                .unwrap();
8609            file.close().unwrap();
8610        }
8611        let file = H5File::open(&path).unwrap();
8612        let ds = file.dataset("notes").unwrap();
8613        assert_eq!(ds.shape(), vec![6]);
8614        assert_eq!(
8615            ds.read_vlen_strings().unwrap(),
8616            vec!["v0", "x", "y", "z", "v4", "v5"]
8617        );
8618        std::fs::remove_file(&path).ok();
8619    }
8620
8621    /// Elements the append buffer still holds are not on disk yet; the
8622    /// update flushes them to their chunks first, so the flush at close has
8623    /// nothing left to write the pre-update reference over.
8624    #[test]
8625    fn write_vlen_strings_slice_updates_buffered_elements() {
8626        let path = temp_path("vlen_slice_buffered");
8627        {
8628            let file = H5File::create(&path).unwrap();
8629            file.create_appendable_vlen_dataset("notes", 4, None)
8630                .unwrap();
8631            // 3 of a 4-element chunk: all three stay in the append buffer.
8632            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8633            file.dataset_writer("notes")
8634                .unwrap()
8635                .write_vlen_strings_slice(1, &["patched"])
8636                .unwrap();
8637            file.close().unwrap();
8638        }
8639        let file = H5File::open(&path).unwrap();
8640        assert_eq!(
8641            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8642            vec!["a", "patched", "c"]
8643        );
8644        std::fs::remove_file(&path).ok();
8645    }
8646
8647    /// A range past the end is rejected before anything is written, and an
8648    /// empty batch costs the file nothing — without the early return it would
8649    /// still allocate and write an empty global-heap collection.
8650    #[test]
8651    fn write_vlen_strings_slice_checks_its_range() {
8652        let build = |name: &str, empty_call: bool| {
8653            let path = temp_path(name);
8654            let file = H5File::create(&path).unwrap();
8655            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8656            let ds = file.dataset_writer("notes").unwrap();
8657            let err = ds
8658                .write_vlen_strings_slice(1, &["x", "y"])
8659                .unwrap_err()
8660                .to_string();
8661            assert!(
8662                err.contains("outside the dataset's 2 elements"),
8663                "got: {err}"
8664            );
8665            if empty_call {
8666                ds.write_vlen_strings_slice(0, &[]).unwrap();
8667            }
8668            file.close().unwrap();
8669            path
8670        };
8671
8672        let with_empty = build("vlen_slice_range", true);
8673        let control = build("vlen_slice_range_control", false);
8674        assert_eq!(
8675            std::fs::metadata(&with_empty).unwrap().len(),
8676            std::fs::metadata(&control).unwrap().len(),
8677            "the rejected and empty calls must leave the file untouched"
8678        );
8679
8680        let file = H5File::open(&with_empty).unwrap();
8681        assert_eq!(
8682            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8683            vec!["a", "b"]
8684        );
8685        std::fs::remove_file(&with_empty).ok();
8686        std::fs::remove_file(&control).ok();
8687    }
8688
8689    /// The element offset is one-dimensional, so a multi-dimensional dataset is
8690    /// rejected rather than silently indexed along the first axis.
8691    #[test]
8692    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
8693        let path = temp_path("vlen_slice_2d");
8694        let file = H5File::create(&path).unwrap();
8695        let ds = file
8696            .new_dataset::<u8>()
8697            .datatype(DatatypeMessage::vlen_string_utf8())
8698            .shape([2, 3])
8699            .create("grid")
8700            .unwrap();
8701        let err = ds
8702            .write_vlen_strings_slice(0, &["x"])
8703            .unwrap_err()
8704            .to_string();
8705        assert!(err.contains("1-dimension datasets"), "got: {err}");
8706        file.close().unwrap();
8707        std::fs::remove_file(&path).ok();
8708    }
8709
8710    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
8711    /// declares the ASCII character set would mislabel the bytes.
8712    #[test]
8713    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
8714        let path = temp_path("vlen_slice_ascii");
8715        let file = H5File::create(&path).unwrap();
8716        let ds = file
8717            .new_dataset::<u8>()
8718            .datatype(DatatypeMessage::vlen_string_ascii())
8719            .shape([3])
8720            .create("notes")
8721            .unwrap();
8722        let err = ds
8723            .write_vlen_strings_slice(0, &["ok", "안녕"])
8724            .unwrap_err()
8725            .to_string();
8726        assert!(
8727            err.contains("string 1") && err.contains("is not ASCII"),
8728            "got: {err}"
8729        );
8730        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
8731        file.close().unwrap();
8732        std::fs::remove_file(&path).ok();
8733    }
8734
8735    /// A numeric dataset is rejected: its elements are not vlen references and
8736    /// writing one would corrupt the column.
8737    #[test]
8738    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
8739        let path = temp_path("vlen_slice_numeric");
8740        let file = H5File::create(&path).unwrap();
8741        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8742        ds.write_raw(&[1i32, 2, 3]).unwrap();
8743        let err = ds
8744            .write_vlen_strings_slice(0, &["x"])
8745            .unwrap_err()
8746            .to_string();
8747        assert!(
8748            err.contains("only for variable-length string datasets"),
8749            "got: {err}"
8750        );
8751        file.close().unwrap();
8752        std::fs::remove_file(&path).ok();
8753    }
8754
8755    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
8756
8757    /// Repeatedly replacing the same element must not grow the file per
8758    /// update: the collection each update supersedes is freed and the next
8759    /// update's collection lands in that block. Without the release every
8760    /// update costs another `H5HG_MINALLOC` (4096) bytes.
8761    #[test]
8762    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
8763        let size_after = |updates: usize| {
8764            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
8765            let file = H5File::create(&path).unwrap();
8766            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8767            let ds = file.dataset_writer("notes").unwrap();
8768            for i in 0..updates {
8769                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
8770                    .unwrap();
8771            }
8772            file.close().unwrap();
8773            let n = std::fs::metadata(&path).unwrap().len();
8774            let read = H5File::open(&path).unwrap();
8775            assert_eq!(
8776                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8777                vec![format!("update {}", updates - 1), "b".to_string()]
8778            );
8779            drop(read);
8780            std::fs::remove_file(&path).ok();
8781            n
8782        };
8783
8784        // The allocator settles once a freed block is available to reuse, so
8785        // every count past that produces the same file.
8786        let settled = size_after(3);
8787        assert_eq!(size_after(20), settled, "20 updates against 3");
8788        assert_eq!(size_after(50), settled, "50 updates against 3");
8789    }
8790
8791    /// An empty string is stored as a real heap object under a reference whose
8792    /// sequence length is zero, so the release must go by the address, not the
8793    /// length — a length test strands the object and its collection forever.
8794    #[test]
8795    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
8796        let size_after = |updates: usize| {
8797            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
8798            let file = H5File::create(&path).unwrap();
8799            file.write_vlen_strings("notes", &["", "b"]).unwrap();
8800            let ds = file.dataset_writer("notes").unwrap();
8801            for _ in 0..updates {
8802                ds.write_vlen_strings_slice(0, &[""]).unwrap();
8803            }
8804            file.close().unwrap();
8805            let n = std::fs::metadata(&path).unwrap().len();
8806            let read = H5File::open(&path).unwrap();
8807            assert_eq!(
8808                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8809                vec!["".to_string(), "b".to_string()]
8810            );
8811            drop(read);
8812            std::fs::remove_file(&path).ok();
8813            n
8814        };
8815
8816        let settled = size_after(3);
8817        assert_eq!(size_after(20), settled, "20 empty updates against 3");
8818        assert_eq!(size_after(50), settled, "50 empty updates against 3");
8819    }
8820
8821    /// The elements the update does not name keep their strings, so freeing
8822    /// the superseded objects must not disturb the collection's survivors.
8823    #[test]
8824    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
8825        let path = temp_path("vlen_slice_heap_survivors");
8826        {
8827            let file = H5File::create(&path).unwrap();
8828            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8829                .unwrap();
8830            let ds = file.dataset_writer("notes").unwrap();
8831            // Two updates inside the one collection the create wrote, so the
8832            // second reads a collection the first already rewrote.
8833            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
8834            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
8835            file.close().unwrap();
8836        }
8837        let file = H5File::open(&path).unwrap();
8838        assert_eq!(
8839            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8840            vec!["a", "B", "c", "D"]
8841        );
8842        std::fs::remove_file(&path).ok();
8843    }
8844
8845    /// Replacing every element of a chunked dataset empties the collection the
8846    /// append wrote, and the file must still read back correctly after its
8847    /// block goes to the allocator.
8848    #[test]
8849    fn write_vlen_strings_slice_frees_an_emptied_collection() {
8850        let path = temp_path("vlen_slice_heap_emptied");
8851        {
8852            let file = H5File::create(&path).unwrap();
8853            file.create_appendable_vlen_dataset("notes", 2, None)
8854                .unwrap();
8855            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
8856                .unwrap();
8857            file.close().unwrap();
8858        }
8859        {
8860            let file = H5File::open_rw(&path).unwrap();
8861            file.dataset_writer("notes")
8862                .unwrap()
8863                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
8864                .unwrap();
8865            file.close().unwrap();
8866        }
8867        let file = H5File::open(&path).unwrap();
8868        assert_eq!(
8869            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8870            vec!["w", "x", "y", "z"]
8871        );
8872        std::fs::remove_file(&path).ok();
8873    }
8874
8875    /// A collection larger than the 4096-byte minimum must keep its size when
8876    /// an object leaves it. Re-encoding at the natural size instead shrinks
8877    /// what the header declares, so the block's tail stops being part of the
8878    /// collection and the eventual free returns less than was allocated —
8879    /// stranding the difference on every cycle.
8880    #[test]
8881    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
8882        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
8883        let size_after = |cycles: usize| {
8884            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
8885            let file = H5File::create(&path).unwrap();
8886            let seed: Vec<String> = "abcd".chars().map(big).collect();
8887            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
8888            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
8889            // is one collection well above it.
8890            file.write_vlen_strings("notes", &refs).unwrap();
8891            let ds = file.dataset_writer("notes").unwrap();
8892            for _ in 0..cycles {
8893                // Partially empty the collection, then finish it off: the
8894                // block is freed only after it has been rewritten once.
8895                let head = big('x');
8896                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
8897                let tail: Vec<String> = "yzw".chars().map(big).collect();
8898                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
8899                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
8900            }
8901            file.close().unwrap();
8902            let n = std::fs::metadata(&path).unwrap().len();
8903            let read = H5File::open(&path).unwrap();
8904            assert_eq!(
8905                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8906                vec![big('x'), big('y'), big('z'), big('w')]
8907            );
8908            drop(read);
8909            std::fs::remove_file(&path).ok();
8910            n
8911        };
8912
8913        let settled = size_after(4);
8914        assert_eq!(size_after(30), settled, "30 cycles against 4");
8915    }
8916
8917    /// An element still in the append buffer has never been on disk, so its
8918    /// superseded object has to be found in the buffer or it is stranded.
8919    #[test]
8920    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
8921        let path = temp_path("vlen_slice_heap_buffered");
8922        let file = H5File::create(&path).unwrap();
8923        file.create_appendable_vlen_dataset("notes", 4, None)
8924            .unwrap();
8925        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8926        let ds = file.dataset_writer("notes").unwrap();
8927        for i in 0..20 {
8928            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
8929                .unwrap();
8930        }
8931        file.close().unwrap();
8932
8933        let file = H5File::open(&path).unwrap();
8934        assert_eq!(
8935            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8936            vec!["a", "patch 19", "c"]
8937        );
8938        let size = std::fs::metadata(&path).unwrap().len();
8939        std::fs::remove_file(&path).ok();
8940        assert!(
8941            size < 20 * 4096,
8942            "20 buffered updates left {size} bytes, one collection per update"
8943        );
8944    }
8945
8946    /// Regression: a typed `write_slice` into rows the append buffer still
8947    /// held wrote the chunks, and the flush at close wrote the stale buffered
8948    /// rows back over it — write 99, read 50. The slice now flushes the
8949    /// buffer first, making the chunks the single authority for those rows.
8950    #[test]
8951    fn write_slice_into_the_buffered_tail_survives_close() {
8952        let path = temp_path("slice_into_buffered_tail");
8953        {
8954            let file = H5File::create(&path).unwrap();
8955            let ds = file
8956                .new_dataset::<i32>()
8957                .shape([0])
8958                .chunk(&[4])
8959                .max_shape(&[None])
8960                .create("d")
8961                .unwrap();
8962            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
8963            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
8964            ds.write_slice(&[4], &[1], &[99]).unwrap();
8965            file.close().unwrap();
8966        }
8967        {
8968            let file = H5File::open(&path).unwrap();
8969            let ds = file.dataset("d").unwrap();
8970            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
8971        }
8972        std::fs::remove_file(&path).ok();
8973    }
8974
8975    /// Extending a dataset while appends sit in the buffer must not move
8976    /// them: the buffer records the absolute row its frames belong to, so
8977    /// the flush at close lands them there, and the grown region reads as
8978    /// fill.
8979    #[test]
8980    fn extend_does_not_move_buffered_appends() {
8981        let path = temp_path("extend_keeps_buffered_rows");
8982        {
8983            let file = H5File::create(&path).unwrap();
8984            let ds = file
8985                .new_dataset::<i32>()
8986                .shape([0])
8987                .chunk(&[4])
8988                .max_shape(&[None])
8989                .create("d")
8990                .unwrap();
8991            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
8992            ds.extend(&[10]).unwrap();
8993            file.close().unwrap();
8994        }
8995        {
8996            let file = H5File::open(&path).unwrap();
8997            let ds = file.dataset("d").unwrap();
8998            assert_eq!(
8999                ds.read_raw::<i32>().unwrap(),
9000                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
9001            );
9002        }
9003        std::fs::remove_file(&path).ok();
9004    }
9005
9006    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
9007    /// dimensions) buffered fine but close() failed "not a chunked dataset"
9008    /// and lost the buffered rows — the append's chunk writes required the
9009    /// extensible-array index. They now go through the index-generic
9010    /// hyperslab engine.
9011    #[test]
9012    fn append_to_a_btree_v2_dataset_survives_close() {
9013        let path = temp_path("append_bt2_close");
9014        {
9015            let file = H5File::create(&path).unwrap();
9016            let ds = file
9017                .new_dataset::<i32>()
9018                .shape([0, 3])
9019                .chunk(&[4, 3])
9020                .max_shape(&[None, None])
9021                .create("d")
9022                .unwrap();
9023            // One buffered row, then a batch that crosses the chunk
9024            // boundary: 4 rows fill chunk band 0, one row stays buffered
9025            // for the flush at close.
9026            ds.append(&[1, 2, 3]).unwrap();
9027            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
9028            file.close().unwrap();
9029        }
9030        {
9031            let file = H5File::open(&path).unwrap();
9032            let ds = file.dataset("d").unwrap();
9033            assert_eq!(ds.shape(), vec![5, 3]);
9034            assert_eq!(
9035                ds.read_raw::<i32>().unwrap(),
9036                (1..=15).collect::<Vec<i32>>()
9037            );
9038        }
9039        std::fs::remove_file(&path).ok();
9040    }
9041
9042    /// A chunk row narrower than the frame row is legal geometry (libhdf5
9043    /// creates it); appended frames must be scattered across the row's
9044    /// tiles at the chunk stride, not packed at the frame stride.
9045    #[test]
9046    fn append_scatters_frames_across_narrow_chunk_tiles() {
9047        let path = temp_path("append_narrow_chunks");
9048        {
9049            let file = H5File::create(&path).unwrap();
9050            let ds = file
9051                .new_dataset::<i32>()
9052                .shape([0, 8])
9053                .chunk(&[2, 4])
9054                .max_shape(&[None, Some(8)])
9055                .create("d")
9056                .unwrap();
9057            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
9058            // row 2 is flushed partial at close.
9059            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
9060            file.close().unwrap();
9061        }
9062        {
9063            let file = H5File::open(&path).unwrap();
9064            let ds = file.dataset("d").unwrap();
9065            assert_eq!(ds.shape(), vec![3, 8]);
9066            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
9067        }
9068        std::fs::remove_file(&path).ok();
9069    }
9070
9071    /// A fixed-array dataset has no room to grow: appending must surface an
9072    /// error naming the chunk grid, not lose rows silently. (Before the
9073    /// index-generic append it failed as "not a chunked dataset".)
9074    #[test]
9075    fn append_to_a_full_fixed_array_dataset_errors() {
9076        let path = temp_path("append_fa_errors");
9077        let file = H5File::create(&path).unwrap();
9078        let ds = file
9079            .new_dataset::<i32>()
9080            .shape([4, 3])
9081            .chunk(&[2, 3])
9082            .create("d")
9083            .unwrap();
9084        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
9085        assert!(
9086            err.to_string().contains("chunk grid"),
9087            "unexpected error: {err}"
9088        );
9089        file.close().unwrap();
9090        std::fs::remove_file(&path).ok();
9091    }
9092
9093    /// A finite max_shape above the current shape used to be dropped on the
9094    /// fixed-array path: the array was sized from the current dims and the
9095    /// stored dataspace had no maximum, so growth failed. The array is now
9096    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
9097    /// fixed-max dataset appends up to its maximum and roundtrips.
9098    #[test]
9099    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
9100        let path = temp_path("fa_growable_dim0");
9101        {
9102            let file = H5File::create(&path).unwrap();
9103            let ds = file
9104                .new_dataset::<i32>()
9105                .shape([4, 3])
9106                .chunk(&[2, 3])
9107                .max_shape(&[Some(10), Some(3)])
9108                .create("d")
9109                .unwrap();
9110            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9111            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
9112            file.close().unwrap();
9113        }
9114        {
9115            let file = H5File::open(&path).unwrap();
9116            let ds = file.dataset("d").unwrap();
9117            assert_eq!(ds.shape(), vec![6, 3]);
9118            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
9119        }
9120        std::fs::remove_file(&path).ok();
9121    }
9122
9123    /// The multiplier-dimension boundary: growing a dimension other than 0
9124    /// changes the current chunk grid but not the index grid. Chunk slots
9125    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
9126    /// chunks written before the extend are looked up under different
9127    /// indices after it.
9128    #[test]
9129    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
9130        let path = temp_path("fa_growable_dim1");
9131        {
9132            let file = H5File::create(&path).unwrap();
9133            let ds = file
9134                .new_dataset::<i32>()
9135                .shape([4, 3])
9136                .chunk(&[2, 3])
9137                .max_shape(&[Some(4), Some(9)])
9138                .create("d")
9139                .unwrap();
9140            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9141            ds.extend(&[4, 6]).unwrap();
9142            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
9143                .unwrap();
9144            file.close().unwrap();
9145        }
9146        {
9147            let file = H5File::open(&path).unwrap();
9148            let ds = file.dataset("d").unwrap();
9149            assert_eq!(ds.shape(), vec![4, 6]);
9150            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
9151            // then [12 + r*3 ..) from the second.
9152            let mut expect = Vec::new();
9153            for r in 0i32..4 {
9154                expect.extend((r * 3)..(r * 3 + 3));
9155                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
9156            }
9157            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
9158        }
9159        std::fs::remove_file(&path).ok();
9160    }
9161
9162    /// Growth boundaries: past the stored maximum is rejected, and a dataset
9163    /// without a stored maximum is fixed at its extent (libhdf5 defaults
9164    /// maxdims to dims at creation).
9165    #[test]
9166    fn extend_beyond_the_maximum_is_rejected() {
9167        let path = temp_path("extend_beyond_max");
9168        let file = H5File::create(&path).unwrap();
9169        let ds = file
9170            .new_dataset::<i32>()
9171            .shape([4, 3])
9172            .chunk(&[2, 3])
9173            .max_shape(&[Some(6), Some(3)])
9174            .create("d")
9175            .unwrap();
9176        ds.extend(&[6, 3]).unwrap();
9177        let err = ds.extend(&[8, 3]).unwrap_err();
9178        assert!(
9179            err.to_string().contains("exceeds the maximum"),
9180            "unexpected error: {err}"
9181        );
9182        file.close().unwrap();
9183        std::fs::remove_file(&path).ok();
9184    }
9185
9186    /// An unlimited dimension other than 0 has no fixed linear slot without
9187    /// libhdf5's extensible-array swizzling; `chunk_grid::linear_index` now
9188    /// implements that swizzle for any dimension, so this creates cleanly
9189    /// and every extend keeps writing new chunks to new slots, never
9190    /// re-addressing one already on disk.
9191    #[test]
9192    fn builder_accepts_an_unlimited_inner_dimension() {
9193        let path = temp_path("unlimited_inner_dim");
9194        let file = H5File::create(&path).unwrap();
9195        let ds = file
9196            .new_dataset::<i32>()
9197            .shape([4, 0])
9198            .chunk(&[2, 2])
9199            .max_shape(&[Some(4), None])
9200            .create("d")
9201            .unwrap();
9202        assert_eq!(ds.shape(), vec![4, 0]);
9203
9204        // Write, then extend and write again: if the linear index were
9205        // recomputed from the *current* extent instead of the maximum one,
9206        // the second extend would shift every slot number and the first
9207        // write's chunks would decode under the wrong coordinates below.
9208        ds.extend(&[4, 2]).unwrap();
9209        ds.write_slice(&[0, 0], &[4, 2], &[1, 2, 3, 4, 5, 6, 7, 8])
9210            .unwrap();
9211        ds.extend(&[4, 4]).unwrap();
9212        ds.write_slice(&[0, 2], &[4, 2], &[9, 10, 11, 12, 13, 14, 15, 16])
9213            .unwrap();
9214
9215        file.close().unwrap();
9216        let file = H5File::open(&path).unwrap();
9217        let ds = file.dataset("d").unwrap();
9218        assert_eq!(
9219            ds.read_slice::<i32>(&[0, 0], &[4, 4]).unwrap(),
9220            vec![1, 2, 9, 10, 3, 4, 11, 12, 5, 6, 13, 14, 7, 8, 15, 16]
9221        );
9222        std::fs::remove_file(&path).ok();
9223    }
9224
9225    /// Regression: a chunk wider than a fixed max dimension used to be
9226    /// accepted, and appends then packed rows at the chunk stride — writing
9227    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
9228    /// geometry at create (`H5D__chunk_construct`); so do we now.
9229    #[test]
9230    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
9231        let path = temp_path("builder_chunk_wider_than_max");
9232        let file = H5File::create(&path).unwrap();
9233        let err = match file
9234            .new_dataset::<i32>()
9235            .shape([0, 2])
9236            .chunk(&[2, 4])
9237            .max_shape(&[None, Some(2)])
9238            .create("v5")
9239        {
9240            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
9241            Err(e) => e,
9242        };
9243        assert!(
9244            err.to_string().contains("maximum dimension size"),
9245            "unexpected error: {err}"
9246        );
9247        file.close().unwrap();
9248        std::fs::remove_file(&path).ok();
9249    }
9250
9251    /// Boundary: `fits in T` vs `does not fit in T`, for both the too-large
9252    /// (u64::MAX → i64) and the negative-to-unsigned (−1 → u32) directions.
9253    #[test]
9254    fn numeric_int_checked_conversion_boundaries() {
9255        let path = temp_path("numeric_int_bounds");
9256        {
9257            let file = H5File::create(&path).unwrap();
9258            let ds = file.new_dataset::<u64>().shape([2]).create("u").unwrap();
9259            ds.write_raw(&[1u64, u64::MAX]).unwrap();
9260            let ds = file.new_dataset::<i32>().shape([2]).create("i").unwrap();
9261            ds.write_raw(&[-1i32, 5]).unwrap();
9262            file.close().unwrap();
9263        }
9264        let file = H5File::open(&path).unwrap();
9265
9266        let u = file.dataset("u").unwrap();
9267        assert_eq!(u.read_numeric_as::<u64>().unwrap(), vec![1, u64::MAX]);
9268        assert_eq!(
9269            u.read_numeric_as::<i128>().unwrap(),
9270            vec![1, i128::from(u64::MAX)]
9271        );
9272        let err = u.read_numeric_as::<i64>().unwrap_err();
9273        assert!(
9274            err.to_string()
9275                .contains("value 18446744073709551615 at element 1 does not fit in i64"),
9276            "unexpected error: {err}"
9277        );
9278
9279        let i = file.dataset("i").unwrap();
9280        assert_eq!(i.read_numeric_as::<i64>().unwrap(), vec![-1, 5]);
9281        let err = i.read_numeric_as::<u32>().unwrap_err();
9282        assert!(
9283            err.to_string()
9284                .contains("value -1 at element 0 does not fit in u32"),
9285            "unexpected error: {err}"
9286        );
9287        std::fs::remove_file(&path).ok();
9288    }
9289
9290    /// Boundary: f32 → f64 is exact widening; f64 → f32 is rejected.
9291    #[test]
9292    fn numeric_float_widening_only() {
9293        let path = temp_path("numeric_float");
9294        {
9295            let file = H5File::create(&path).unwrap();
9296            let ds = file.new_dataset::<f32>().shape([2]).create("f4").unwrap();
9297            ds.write_raw(&[1.5f32, -2.25]).unwrap();
9298            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9299            ds.write_raw(&[3.75f64]).unwrap();
9300            file.close().unwrap();
9301        }
9302        let file = H5File::open(&path).unwrap();
9303
9304        let f4 = file.dataset("f4").unwrap();
9305        assert_eq!(f4.read_numeric_as::<f32>().unwrap(), vec![1.5, -2.25]);
9306        assert_eq!(f4.read_numeric_as::<f64>().unwrap(), vec![1.5, -2.25]);
9307
9308        let f8 = file.dataset("f8").unwrap();
9309        assert_eq!(f8.read_numeric_as::<f64>().unwrap(), vec![3.75]);
9310        let err = f8.read_numeric_as::<f32>().unwrap_err();
9311        assert!(
9312            err.to_string().contains("narrowing"),
9313            "unexpected error: {err}"
9314        );
9315        std::fs::remove_file(&path).ok();
9316    }
9317
9318    /// Boundary: cross-class conversions (float ↔ integer) are rejected in
9319    /// both directions, and a non-numeric datatype is rejected at classify.
9320    #[test]
9321    fn numeric_cross_class_and_non_numeric_rejected() {
9322        let path = temp_path("numeric_cross_class");
9323        {
9324            let file = H5File::create(&path).unwrap();
9325            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9326            ds.write_raw(&[1.0f64]).unwrap();
9327            let ds = file.new_dataset::<i32>().shape([1]).create("i4").unwrap();
9328            ds.write_raw(&[7i32]).unwrap();
9329            file.write_vlen_strings("s", &["a", "b"]).unwrap();
9330            file.close().unwrap();
9331        }
9332        let file = H5File::open(&path).unwrap();
9333
9334        let err = file
9335            .dataset("f8")
9336            .unwrap()
9337            .read_numeric_as::<i64>()
9338            .unwrap_err();
9339        assert!(
9340            err.to_string().contains("floating-point dataset as i64"),
9341            "unexpected error: {err}"
9342        );
9343        let err = file
9344            .dataset("i4")
9345            .unwrap()
9346            .read_numeric_as::<f64>()
9347            .unwrap_err();
9348        assert!(
9349            err.to_string().contains("integer dataset as f64"),
9350            "unexpected error: {err}"
9351        );
9352        let err = file
9353            .dataset("s")
9354            .unwrap()
9355            .read_numeric_as::<i64>()
9356            .unwrap_err();
9357        assert!(
9358            err.to_string().contains("is not numeric"),
9359            "unexpected error: {err}"
9360        );
9361        std::fs::remove_file(&path).ok();
9362    }
9363
9364    /// Boundary: big-endian sources decode per the datatype's byte order.
9365    /// Unit-level (the writer only emits little-endian): feed `convert` a
9366    /// big-endian datatype plus big-endian bytes directly.
9367    #[test]
9368    fn numeric_big_endian_decode() {
9369        use super::numeric;
9370        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
9371
9372        let dt = DatatypeMessage::FixedPoint {
9373            size: 4,
9374            byte_order: ByteOrder::BigEndian,
9375            signed: true,
9376            bit_offset: 0,
9377            bit_precision: 32,
9378        };
9379        let mut raw = Vec::new();
9380        raw.extend_from_slice(&(-2i32).to_be_bytes());
9381        raw.extend_from_slice(&(100_000i32).to_be_bytes());
9382        let kind = numeric::classify(&dt).unwrap();
9383        assert_eq!(
9384            numeric::convert::<i64>(kind, &raw).unwrap(),
9385            vec![-2, 100_000]
9386        );
9387
9388        let dt = DatatypeMessage::FloatingPoint {
9389            size: 8,
9390            byte_order: ByteOrder::BigEndian,
9391            sign_location: 63,
9392            bit_offset: 0,
9393            bit_precision: 64,
9394            exponent_location: 52,
9395            exponent_size: 11,
9396            mantissa_location: 0,
9397            mantissa_size: 52,
9398            exponent_bias: 1023,
9399        };
9400        let raw = (-2.25f64).to_be_bytes();
9401        let kind = numeric::classify(&dt).unwrap();
9402        assert_eq!(numeric::convert::<f64>(kind, &raw).unwrap(), vec![-2.25]);
9403    }
9404
9405    /// `H5Attribute::read_numeric` validates the stored datatype before
9406    /// reinterpreting bytes: cross-width, cross-class, and non-numeric
9407    /// attributes error instead of returning bit-garbage, while the exact
9408    /// type and the HBool / complex-compound paths keep working.
9409    #[test]
9410    fn attr_read_numeric_validates_datatype() {
9411        use crate::types::{Complex64, HBool, VarLenUnicode};
9412        let path = temp_path("attr_read_numeric_validate");
9413        {
9414            let file = H5File::create(&path).unwrap();
9415            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9416            ds.write_raw(&[1.0f32; 2]).unwrap();
9417            let a = ds.new_attr::<f64>().shape(()).create("f8").unwrap();
9418            a.write_numeric(&1.5f64).unwrap();
9419            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9420            a.write_numeric(&-7i32).unwrap();
9421            let a = ds.new_attr::<HBool>().shape(()).create("b").unwrap();
9422            a.write_numeric(&HBool::from(true)).unwrap();
9423            let a = ds.new_attr::<Complex64>().shape(()).create("z").unwrap();
9424            a.write_numeric(&Complex64 { re: 1.0, im: -2.0 }).unwrap();
9425            let a = ds
9426                .new_attr::<VarLenUnicode>()
9427                .shape(())
9428                .create("s")
9429                .unwrap();
9430            a.write_scalar(&VarLenUnicode("text".into())).unwrap();
9431            file.close().unwrap();
9432        }
9433        let file = H5File::open(&path).unwrap();
9434        let ds = file.dataset("d").unwrap();
9435
9436        let f8 = ds.attr("f8").unwrap();
9437        assert_eq!(f8.read_numeric::<f64>().unwrap(), 1.5);
9438        // Previously returned the low half of the f64 image as an f32.
9439        let err = f8.read_numeric::<f32>().unwrap_err();
9440        assert!(
9441            err.to_string().contains("read_numeric_as"),
9442            "unexpected error: {err}"
9443        );
9444        assert!(f8.read_numeric::<i64>().is_err());
9445
9446        let i4 = ds.attr("i4").unwrap();
9447        assert_eq!(i4.read_numeric::<i32>().unwrap(), -7);
9448        assert!(i4.read_numeric::<u32>().is_err());
9449
9450        assert!(bool::from(
9451            ds.attr("b").unwrap().read_numeric::<HBool>().unwrap()
9452        ));
9453        let z = ds.attr("z").unwrap().read_numeric::<Complex64>().unwrap();
9454        assert_eq!((z.re, z.im), (1.0, -2.0));
9455
9456        // A vlen string attribute: read_numeric used to transmute the heap
9457        // reference bytes into the requested type.
9458        let s = ds.attr("s").unwrap();
9459        assert!(s.read_numeric::<f64>().is_err());
9460        assert!(s.read_numeric_as::<f64>().is_err());
9461        std::fs::remove_file(&path).ok();
9462    }
9463
9464    /// The attribute conversion read applies the dataset rules: checked
9465    /// int → int naming index and value on overflow, widening-only floats,
9466    /// cross-class rejected; an array attribute converts every element.
9467    #[test]
9468    fn attr_read_numeric_as_converts() {
9469        let path = temp_path("attr_read_numeric_as");
9470        {
9471            let file = H5File::create(&path).unwrap();
9472            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9473            ds.write_raw(&[1.0f32; 2]).unwrap();
9474            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9475            a.write_numeric(&-7i32).unwrap();
9476            let a = ds.new_attr::<u64>().shape(()).create("u8max").unwrap();
9477            a.write_numeric(&u64::MAX).unwrap();
9478            let a = ds.new_attr::<i16>().shape([3]).create("arr").unwrap();
9479            a.write_array(&[1i16, -2, 3]).unwrap();
9480            file.close().unwrap();
9481        }
9482        let file = H5File::open(&path).unwrap();
9483        let ds = file.dataset("d").unwrap();
9484        assert_eq!(
9485            ds.attr("i4").unwrap().read_numeric_as::<i64>().unwrap(),
9486            vec![-7]
9487        );
9488        let err = ds
9489            .attr("u8max")
9490            .unwrap()
9491            .read_numeric_as::<i64>()
9492            .unwrap_err();
9493        assert!(
9494            err.to_string().contains("does not fit in i64"),
9495            "unexpected error: {err}"
9496        );
9497        assert_eq!(
9498            ds.attr("arr").unwrap().read_numeric_as::<i32>().unwrap(),
9499            vec![1, -2, 3]
9500        );
9501        assert!(ds.attr("i4").unwrap().read_numeric_as::<f64>().is_err());
9502        std::fs::remove_file(&path).ok();
9503    }
9504
9505    /// The hyperslab variant applies the same conversion to a sub-selection.
9506    #[test]
9507    fn numeric_slice_conversion() {
9508        let path = temp_path("numeric_slice");
9509        {
9510            let file = H5File::create(&path).unwrap();
9511            let ds = file.new_dataset::<i16>().shape([2, 3]).create("m").unwrap();
9512            ds.write_raw(&[1i16, 2, 3, 4, 5, 6]).unwrap();
9513            file.close().unwrap();
9514        }
9515        let file = H5File::open(&path).unwrap();
9516        let m = file.dataset("m").unwrap();
9517        assert_eq!(
9518            m.read_numeric_slice_as::<i32>(&[0, 1], &[2, 2]).unwrap(),
9519            vec![2, 3, 5, 6]
9520        );
9521        std::fs::remove_file(&path).ok();
9522    }
9523
9524    /// A NULL dataspace round-trips through the public API as `is_null() ==
9525    /// true`, `shape() == []`, and `read_raw_bytes()` empty — and stays
9526    /// distinguishable from a scalar dataset, which shares the same empty
9527    /// `shape()` but holds exactly one element.
9528    #[test]
9529    fn null_dataspace_distinct_from_scalar() {
9530        let path = temp_path("null_vs_scalar");
9531        {
9532            let file = H5File::create(&path).unwrap();
9533            file.new_dataset::<i32>().null().create("empty").unwrap();
9534            let scalar = file.new_dataset::<i32>().scalar().create("scalar").unwrap();
9535            scalar.write_raw(&[42i32]).unwrap();
9536            file.close().unwrap();
9537        }
9538        let file = H5File::open(&path).unwrap();
9539
9540        let empty = file.dataset("empty").unwrap();
9541        assert!(empty.is_null());
9542        assert_eq!(empty.shape(), Vec::<usize>::new());
9543        assert_eq!(empty.total_elements(), 0);
9544        assert_eq!(empty.read_raw_bytes().unwrap(), Vec::<u8>::new());
9545
9546        let scalar = file.dataset("scalar").unwrap();
9547        assert!(!scalar.is_null());
9548        assert_eq!(scalar.shape(), Vec::<usize>::new());
9549        assert_eq!(scalar.total_elements(), 1);
9550        assert_eq!(scalar.read_raw::<i32>().unwrap(), vec![42]);
9551
9552        std::fs::remove_file(&path).ok();
9553    }
9554
9555    /// A NULL dataspace dataset rejects writes outright — there is nothing
9556    /// to write into — rather than silently accepting a scalar-shaped
9557    /// write against unallocated storage.
9558    #[test]
9559    fn null_dataspace_rejects_writes() {
9560        let path = temp_path("null_write_rejected");
9561        let file = H5File::create(&path).unwrap();
9562        let ds = file.new_dataset::<i32>().null().create("empty").unwrap();
9563        assert!(ds.write_raw(&[1i32]).is_err());
9564        assert!(ds.write_raw_bytes(&[0u8; 4]).is_err());
9565        file.close().unwrap();
9566        std::fs::remove_file(&path).ok();
9567    }
9568
9569    /// `.null()` combined with `.chunk()` or a fill value is rejected at
9570    /// `create()` rather than silently dropping the conflicting option —
9571    /// a NULL dataspace can never be chunked or filtered upstream.
9572    #[test]
9573    fn null_dataspace_rejects_chunking_and_fill_value() {
9574        let path = temp_path("null_chunk_rejected");
9575        let file = H5File::create(&path).unwrap();
9576        assert!(file
9577            .new_dataset::<i32>()
9578            .null()
9579            .chunk(&[4])
9580            .create("a")
9581            .is_err());
9582        assert!(file
9583            .new_dataset::<i32>()
9584            .null()
9585            .fill_value(7i32)
9586            .create("b")
9587            .is_err());
9588        file.close().unwrap();
9589        std::fs::remove_file(&path).ok();
9590    }
9591
9592    /// A committed type is resolved before the dataset is created, so a name
9593    /// that is not one — or one paired with object references, which would
9594    /// make the stored type disagree with the payload — leaves no dataset
9595    /// behind.
9596    #[test]
9597    fn a_committed_type_that_cannot_be_shared_creates_no_dataset() {
9598        use crate::format::messages::datatype::DatatypeMessage;
9599
9600        let path = temp_path("committed_refused");
9601        let file = H5File::create(&path).unwrap();
9602        file.commit_datatype("t", DatatypeMessage::i32_type())
9603            .unwrap();
9604
9605        assert!(file
9606            .new_dataset::<i32>()
9607            .committed_type("absent")
9608            .shape([2usize])
9609            .create("a")
9610            .is_err());
9611        assert!(file
9612            .new_dataset::<u64>()
9613            .committed_type("t")
9614            .object_references()
9615            .shape([2usize])
9616            .create("b")
9617            .is_err());
9618        // A dataset already exists under that name, so the type cannot take
9619        // it either.
9620        file.new_dataset::<i32>()
9621            .shape([2usize])
9622            .create("taken")
9623            .unwrap();
9624        assert!(file
9625            .commit_datatype("taken", DatatypeMessage::i32_type())
9626            .is_err());
9627        assert!(file
9628            .commit_datatype("t", DatatypeMessage::f64_type())
9629            .is_err());
9630
9631        assert_eq!(file.dataset_names(), vec!["taken".to_string()]);
9632        assert_eq!(file.named_datatype_names(), vec!["t".to_string()]);
9633        file.close().unwrap();
9634        std::fs::remove_file(&path).ok();
9635    }
9636
9637    /// Deleting the group that named a committed datatype takes the name with
9638    /// it: nothing in the file reaches the type, so it is not written, the
9639    /// name is free again, and it can no longer be shared by that name.
9640    #[test]
9641    fn deleting_a_group_takes_the_committed_datatypes_it_named() {
9642        use crate::format::messages::datatype::DatatypeMessage;
9643
9644        let path = temp_path("committed_group_deleted");
9645        {
9646            let file = H5File::create(&path).unwrap();
9647            let types = file.create_group("types").unwrap();
9648            types
9649                .commit_datatype("t", DatatypeMessage::i32_type())
9650                .unwrap();
9651            assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9652
9653            file.delete_group("types").unwrap();
9654            assert!(file.named_datatype_names().is_empty());
9655            assert!(file
9656                .new_dataset::<i32>()
9657                .committed_type("types/t")
9658                .shape([2usize])
9659                .create("d")
9660                .is_err());
9661
9662            // The name is free, so a new group may take it back.
9663            let types = file.create_group("types").unwrap();
9664            types
9665                .commit_datatype("t", DatatypeMessage::f64_type())
9666                .unwrap();
9667            file.close().unwrap();
9668        }
9669        let file = H5File::open(&path).unwrap();
9670        assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9671        assert_eq!(
9672            file.named_datatype("types/t").unwrap().datatype().unwrap(),
9673            crate::format::messages::datatype::DatatypeMessage::f64_type()
9674        );
9675        drop(file);
9676        std::fs::remove_file(&path).ok();
9677    }
9678
9679    /// The layout class the reader sees for `name`, plus the image a compact
9680    /// layout carries — what distinguishes compact storage from contiguous
9681    /// storage that happens to hold the same bytes.
9682    fn compact_image(path: &std::path::Path, name: &str) -> Option<Vec<u8>> {
9683        use crate::format::messages::data_layout::DataLayoutMessage;
9684        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9685        match &reader.dataset_info(name).unwrap().layout {
9686            DataLayoutMessage::Compact { data } => Some(data.clone()),
9687            other => panic!("{name}: expected a compact layout, got {other:?}"),
9688        }
9689    }
9690
9691    /// `.compact()` puts the raw data inside the data layout message: the
9692    /// dataset has no data block of its own, and the image the layout carries
9693    /// is what a read returns.
9694    #[test]
9695    fn a_compact_dataset_stores_its_data_in_the_layout_message() {
9696        let path = temp_path("compact_roundtrip");
9697        let values: Vec<i32> = (0..16).collect();
9698        {
9699            let file = H5File::create(&path).unwrap();
9700            file.new_dataset::<i32>()
9701                .shape([16usize])
9702                .compact()
9703                .create("d")
9704                .unwrap()
9705                .write_raw(&values)
9706                .unwrap();
9707            file.close().unwrap();
9708        }
9709
9710        let image = compact_image(&path, "d").unwrap();
9711        assert_eq!(
9712            image,
9713            values
9714                .iter()
9715                .flat_map(|v| v.to_le_bytes())
9716                .collect::<Vec<_>>()
9717        );
9718
9719        let file = H5File::open(&path).unwrap();
9720        let ds = file.dataset("d").unwrap();
9721        assert_eq!(ds.shape(), vec![16]);
9722        assert_eq!(ds.read_raw::<i32>().unwrap(), values);
9723        std::fs::remove_file(&path).ok();
9724    }
9725
9726    /// A compact dataset created inside a group is linked from that group,
9727    /// not from the root: its create path goes through the same parent
9728    /// resolution every other layout uses.
9729    #[test]
9730    fn a_compact_dataset_lands_in_its_group() {
9731        let path = temp_path("compact_in_group");
9732        {
9733            let file = H5File::create(&path).unwrap();
9734            let g = file.root_group().create_group("g").unwrap();
9735            g.new_dataset::<u8>()
9736                .shape([4usize])
9737                .compact()
9738                .create("d")
9739                .unwrap()
9740                .write_raw(&[1u8, 2, 3, 4])
9741                .unwrap();
9742            file.close().unwrap();
9743        }
9744        assert_eq!(compact_image(&path, "/g/d").unwrap(), vec![1u8, 2, 3, 4]);
9745
9746        let file = H5File::open(&path).unwrap();
9747        assert_eq!(
9748            file.dataset("/g/d").unwrap().read_raw::<u8>().unwrap(),
9749            vec![1u8, 2, 3, 4]
9750        );
9751        std::fs::remove_file(&path).ok();
9752    }
9753
9754    /// A compact dataset's storage is the image itself, so the fill value has
9755    /// to be tiled into it at create — `H5D__compact_fill`'s job. An unwritten
9756    /// element must read back as the fill value, not as zero.
9757    #[test]
9758    fn an_unwritten_compact_dataset_reads_back_as_its_fill_value() {
9759        let path = temp_path("compact_fill");
9760        {
9761            let file = H5File::create(&path).unwrap();
9762            file.new_dataset::<i32>()
9763                .shape([4usize])
9764                .compact()
9765                .fill_value(-7i32)
9766                .create("d")
9767                .unwrap();
9768            file.close().unwrap();
9769        }
9770        let file = H5File::open(&path).unwrap();
9771        assert_eq!(
9772            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9773            vec![-7i32; 4]
9774        );
9775        std::fs::remove_file(&path).ok();
9776    }
9777
9778    /// The image is the layout message, so anything that rewrites the header
9779    /// rewrites the data with it. Reopening and attaching an attribute makes
9780    /// the header stale; the rebuilt one must still carry the image rather
9781    /// than fall back to an unallocated contiguous layout.
9782    #[test]
9783    fn a_reopened_compact_dataset_keeps_its_image() {
9784        let path = temp_path("compact_reopen");
9785        let values: Vec<i32> = (100..108).collect();
9786        {
9787            let file = H5File::create(&path).unwrap();
9788            file.new_dataset::<i32>()
9789                .shape([8usize])
9790                .compact()
9791                .create("d")
9792                .unwrap()
9793                .write_raw(&values)
9794                .unwrap();
9795            file.close().unwrap();
9796        }
9797        {
9798            let file = H5File::open_rw(&path).unwrap();
9799            file.dataset_writer("d")
9800                .unwrap()
9801                .new_attr::<i32>()
9802                .shape(())
9803                .create("note")
9804                .unwrap()
9805                .write_numeric(&1i32)
9806                .unwrap();
9807            file.close().unwrap();
9808        }
9809        assert_eq!(
9810            compact_image(&path, "d").unwrap(),
9811            values
9812                .iter()
9813                .flat_map(|v| v.to_le_bytes())
9814                .collect::<Vec<_>>()
9815        );
9816
9817        let file = H5File::open(&path).unwrap();
9818        assert_eq!(
9819            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9820            values
9821        );
9822        std::fs::remove_file(&path).ok();
9823    }
9824
9825    /// The ceiling is what a data layout message can hold, so it is checked
9826    /// in bytes and names them: the largest image that fits is accepted and
9827    /// one element more is refused.
9828    #[test]
9829    fn the_compact_ceiling_is_checked_in_bytes() {
9830        let path = temp_path("compact_ceiling");
9831        let file = H5File::create(&path).unwrap();
9832
9833        let fits = crate::MAX_COMPACT_DATA / 4;
9834        file.new_dataset::<i32>()
9835            .shape([fits])
9836            .compact()
9837            .create("fits")
9838            .unwrap();
9839
9840        let over = fits + 1;
9841        let err = match file
9842            .new_dataset::<i32>()
9843            .shape([over])
9844            .compact()
9845            .create("over")
9846        {
9847            Err(e) => e.to_string(),
9848            Ok(_) => panic!("an image {} bytes wide must be refused", over * 4),
9849        };
9850        assert!(
9851            err.contains(&(over * 4).to_string())
9852                && err.contains(&crate::MAX_COMPACT_DATA.to_string()),
9853            "the error must name both sizes: {err}"
9854        );
9855
9856        file.close().unwrap();
9857        std::fs::remove_file(&path).ok();
9858    }
9859
9860    /// Compact storage has no chunk grid to filter and no room to grow, so
9861    /// each conflicting option is refused at `create()` rather than silently
9862    /// overriding the layout the way `H5Pset_chunk` does.
9863    #[test]
9864    fn compact_rejects_chunking_filters_growth_and_a_null_dataspace() {
9865        let path = temp_path("compact_rejects");
9866        let file = H5File::create(&path).unwrap();
9867        assert!(file
9868            .new_dataset::<i32>()
9869            .shape([4usize])
9870            .compact()
9871            .chunk(&[4])
9872            .create("a")
9873            .is_err());
9874        assert!(file
9875            .new_dataset::<i32>()
9876            .shape([4usize])
9877            .compact()
9878            .deflate(4)
9879            .create("b")
9880            .is_err());
9881        assert!(file
9882            .new_dataset::<i32>()
9883            .shape([4usize])
9884            .compact()
9885            .max_shape(&[None])
9886            .create("c")
9887            .is_err());
9888        assert!(file
9889            .new_dataset::<i32>()
9890            .shape([4usize])
9891            .compact()
9892            .max_shape(&[Some(8)])
9893            .create("d")
9894            .is_err());
9895        assert!(file
9896            .new_dataset::<i32>()
9897            .null()
9898            .compact()
9899            .create("e")
9900            .is_err());
9901        file.close().unwrap();
9902        std::fs::remove_file(&path).ok();
9903    }
9904
9905    /// The filter pipeline the reader decodes from `name`'s header.
9906    fn stored_pipeline(
9907        path: &std::path::Path,
9908        name: &str,
9909    ) -> crate::format::messages::filter::FilterPipeline {
9910        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9911        reader
9912            .dataset_info(name)
9913            .unwrap()
9914            .filter_pipeline
9915            .clone()
9916            .unwrap_or_else(|| panic!("{name}: no filter pipeline"))
9917    }
9918
9919    /// Shuffle is a permutation, not a compressor, so it is a pipeline on its
9920    /// own — `H5Pset_shuffle` with nothing behind it. Its stage must reach the
9921    /// header, and the reader must unpermute what it wrote.
9922    #[test]
9923    fn shuffle_alone_is_a_filter_pipeline() {
9924        use crate::format::messages::filter::{FilterPipeline, FILTER_SHUFFLE};
9925        let path = temp_path("shuffle_alone");
9926        let values: Vec<i32> = (0..64).collect();
9927        {
9928            let file = H5File::create(&path).unwrap();
9929            file.new_dataset::<i32>()
9930                .shape([64usize])
9931                .chunk(&[16])
9932                .shuffle()
9933                .create("d")
9934                .unwrap()
9935                .write_raw(&values)
9936                .unwrap();
9937            file.close().unwrap();
9938        }
9939        let pipeline = stored_pipeline(&path, "d");
9940        assert_eq!(pipeline, FilterPipeline::shuffle(4));
9941        assert_eq!(pipeline.filters[0].id, FILTER_SHUFFLE);
9942
9943        let file = H5File::open(&path).unwrap();
9944        assert_eq!(
9945            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9946            values
9947        );
9948        std::fs::remove_file(&path).ok();
9949    }
9950
9951    /// `.shuffle()` and `.deflate()` are separate stages that compose, and
9952    /// `.shuffle_deflate()` is the shorthand for both — one pipeline, built
9953    /// once, whichever way it was asked for.
9954    #[cfg(feature = "deflate")]
9955    #[test]
9956    fn shuffle_composes_with_deflate() {
9957        let path = temp_path("shuffle_then_deflate");
9958        let values: Vec<i32> = (0..64).collect();
9959        {
9960            let file = H5File::create(&path).unwrap();
9961            for (name, ds) in [
9962                ("split", file.new_dataset::<i32>().shuffle().deflate(6)),
9963                ("combined", file.new_dataset::<i32>().shuffle_deflate(6)),
9964            ] {
9965                ds.shape([64usize])
9966                    .chunk(&[16])
9967                    .create(name)
9968                    .unwrap()
9969                    .write_raw(&values)
9970                    .unwrap();
9971            }
9972            file.close().unwrap();
9973        }
9974        assert_eq!(
9975            stored_pipeline(&path, "split"),
9976            crate::format::messages::filter::FilterPipeline::shuffle_deflate(4, 6)
9977        );
9978        assert_eq!(
9979            stored_pipeline(&path, "split"),
9980            stored_pipeline(&path, "combined")
9981        );
9982
9983        let file = H5File::open(&path).unwrap();
9984        for name in ["split", "combined"] {
9985            assert_eq!(
9986                file.dataset(name).unwrap().read_raw::<i32>().unwrap(),
9987                values,
9988                "{name}"
9989            );
9990        }
9991        std::fs::remove_file(&path).ok();
9992    }
9993
9994    /// The width shuffle permutes by is the stored element's, which a
9995    /// `datatype` override moves away from the carrier type `T`: recording
9996    /// `T`'s width would permute a 4-byte element as four 1-byte ones and
9997    /// hand libhdf5 a chunk it unshuffles into different bytes.
9998    #[test]
9999    fn shuffle_records_the_stored_element_width() {
10000        use crate::format::messages::datatype::DatatypeMessage;
10001        use crate::format::messages::filter::FilterPipeline;
10002        let path = temp_path("shuffle_override_width");
10003        let bytes: Vec<u8> = (0..16u8).collect();
10004        {
10005            let file = H5File::create(&path).unwrap();
10006            file.new_dataset::<u8>()
10007                .datatype(DatatypeMessage::i32_type())
10008                .shape([4usize])
10009                .chunk(&[4])
10010                .shuffle()
10011                .create("d")
10012                .unwrap()
10013                .write_raw_bytes(&bytes)
10014                .unwrap();
10015            file.close().unwrap();
10016        }
10017        assert_eq!(stored_pipeline(&path, "d"), FilterPipeline::shuffle(4));
10018
10019        let file = H5File::open(&path).unwrap();
10020        assert_eq!(
10021            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
10022            bytes
10023                .chunks(4)
10024                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
10025                .collect::<Vec<_>>()
10026        );
10027        std::fs::remove_file(&path).ok();
10028    }
10029
10030    /// A write into a virtual dataset is refused by name rather than landing
10031    /// somewhere no reader would look: libhdf5 pushes such a write through
10032    /// the mapping into the source dataset (`H5D__virtual_write`), which this
10033    /// writer does not do.
10034    #[test]
10035    fn a_virtual_dataset_refuses_every_write() {
10036        use crate::Selection;
10037        let path = temp_path("vds_write_refused");
10038        let file = H5File::create(&path).unwrap();
10039        let ds = file
10040            .new_dataset::<i32>()
10041            .shape([16usize])
10042            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10043            .create("vds")
10044            .unwrap();
10045        for err in [
10046            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap_err(),
10047            ds.write_slice(&[0], &[2], &[1i32, 2]).unwrap_err(),
10048        ] {
10049            let msg = err.to_string();
10050            assert!(msg.contains("virtual dataset"), "{msg}");
10051        }
10052        file.close().unwrap();
10053        std::fs::remove_file(&path).ok();
10054    }
10055
10056    /// A `%b` substitution only means something when the virtual selection is
10057    /// unlimited and the source selection is not: that is the shape where
10058    /// each block draws from a different source dataset. On any other mapping
10059    /// there is only one block, so `H5D_virtual_check_mapping_post` refuses
10060    /// the specifier — and an illegal conversion is refused wherever it
10061    /// appears.
10062    #[test]
10063    fn a_printf_source_name_needs_the_mapping_shape_that_uses_it() {
10064        use crate::Selection;
10065        let path = temp_path("vds_printf");
10066        let file = H5File::create(&path).unwrap();
10067        for (f, d) in [("src_%b.h5", "src"), ("src.h5", "block_%b")] {
10068            let err = match file
10069                .new_dataset::<i32>()
10070                .shape([16usize])
10071                .virtual_mapping(Selection::All, f, d, Selection::All)
10072                .create("vds")
10073            {
10074                Ok(_) => panic!("a bounded mapping has one block, so %b names nothing"),
10075                Err(e) => e.to_string(),
10076            };
10077            assert!(err.contains("printf specifier"), "{err}");
10078        }
10079        // `%z` is not a conversion libhdf5 has, in any mapping shape.
10080        let err = match file
10081            .new_dataset::<i32>()
10082            .shape([1usize, 2])
10083            .max_shape(&[None, Some(2)])
10084            .virtual_mapping(unlimited_rows(), "src_%z.h5", "src", Selection::All)
10085            .create("vds_bad")
10086        {
10087            Ok(_) => panic!("%z is not a legal conversion"),
10088            Err(e) => e.to_string(),
10089        };
10090        assert!(err.contains("invalid format specifier"), "{err}");
10091        file.close().unwrap();
10092        std::fs::remove_file(&path).ok();
10093    }
10094
10095    /// A printf mapping stitches one source dataset per block of its
10096    /// unlimited virtual selection, and the extent stops at the first block
10097    /// with no source (`H5D__virtual_set_extent_unlim`'s printf arm, at the
10098    /// default `H5D_VDS_LAST_AVAILABLE` view and `printf_gap` 0).
10099    #[test]
10100    fn a_printf_mapping_stitches_one_source_per_block() {
10101        use crate::Selection;
10102        let path = temp_path("vds_printf_blocks");
10103        {
10104            let file = H5File::create(&path).unwrap();
10105            for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10106                file.new_dataset::<i32>()
10107                    .shape([2usize])
10108                    .create(&format!("block{b}"))
10109                    .unwrap()
10110                    .write_raw(&[base, base + 1])
10111                    .unwrap();
10112            }
10113            file.new_dataset::<i32>()
10114                .shape([1usize, 2])
10115                .max_shape(&[None, Some(2)])
10116                .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10117                .create("vds")
10118                .unwrap();
10119            file.close().unwrap();
10120        }
10121        let file = H5File::open(&path).unwrap();
10122        let ds = file.dataset("vds").unwrap();
10123        // `block2` is missing, so `block3` is never reached: two rows.
10124        assert_eq!(ds.shape(), vec![2, 2]);
10125        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10126        drop(file);
10127        std::fs::remove_file(&path).ok();
10128    }
10129
10130    /// A mapping whose source cannot be opened reads back as the fill value,
10131    /// not as an error: `H5D__virtual_open_source_dset` accepts a null source
10132    /// file and clears the error stack for a missing source dataset
10133    /// (H5Dvirtual.c:877-909), so `H5D__virtual_read_one` finds no projected
10134    /// memory space and reads nothing for it (H5Dvirtual.c:2661-2665).
10135    #[test]
10136    fn a_source_that_cannot_be_opened_reads_as_the_fill_value() {
10137        use crate::{Hyperslab, HyperslabBlock, Selection};
10138        let block = |start: u64, end: u64| Selection::Hyperslab {
10139            rank: 1,
10140            form: Hyperslab::Blocks(vec![HyperslabBlock {
10141                start: vec![start],
10142                end: vec![end],
10143            }]),
10144        };
10145        let path = temp_path("vds_absent_source");
10146        {
10147            let file = H5File::create(&path).unwrap();
10148            file.new_dataset::<i32>()
10149                .shape([4usize])
10150                .create("here")
10151                .unwrap()
10152                .write_raw(&[1i32, 2, 3, 4])
10153                .unwrap();
10154            file.new_dataset::<i32>()
10155                .shape([12usize])
10156                .fill_value(-3i32)
10157                .virtual_mapping(block(0, 3), ".", "here", block(0, 3))
10158                // A dataset that is not in this file.
10159                .virtual_mapping(block(4, 7), ".", "absent", block(0, 3))
10160                // A file that does not exist beside this one.
10161                .virtual_mapping(block(8, 11), "no_such_vds_source.h5", "src", block(0, 3))
10162                .create("vds")
10163                .unwrap();
10164            file.close().unwrap();
10165        }
10166        let file = H5File::open(&path).unwrap();
10167        let ds = file.dataset("vds").unwrap();
10168        assert_eq!(
10169            ds.read_raw::<i32>().unwrap(),
10170            vec![1, 2, 3, 4, -3, -3, -3, -3, -3, -3, -3, -3]
10171        );
10172        // The same rule on the slice path, which stitches the whole image
10173        // before extracting the region.
10174        assert_eq!(
10175            ds.read_slice::<i32>(&[2], &[6]).unwrap(),
10176            vec![3, 4, -3, -3, -3, -3]
10177        );
10178        drop(file);
10179        std::fs::remove_file(&path).ok();
10180    }
10181
10182    /// `%%` is an escaped literal `%`, not a substitution: the mapping is an
10183    /// ordinary bounded one, and the source it resolves against is the name
10184    /// with a single `%` in it.
10185    #[test]
10186    fn an_escaped_percent_is_a_literal_in_a_source_name() {
10187        use crate::Selection;
10188        let path = temp_path("vds_escaped_pct");
10189        {
10190            let file = H5File::create(&path).unwrap();
10191            file.new_dataset::<i32>()
10192                .shape([4usize])
10193                .create("od%d")
10194                .unwrap()
10195                .write_raw(&[5i32, 6, 7, 8])
10196                .unwrap();
10197            file.new_dataset::<i32>()
10198                .shape([4usize])
10199                .virtual_mapping(Selection::All, ".", "od%%d", Selection::All)
10200                .create("vds")
10201                .unwrap();
10202            file.close().unwrap();
10203        }
10204        let file = H5File::open(&path).unwrap();
10205        let ds = file.dataset("vds").unwrap();
10206        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![5, 6, 7, 8]);
10207        // The stored name keeps its escape; only the resolution unescapes.
10208        assert_eq!(ds.virtual_mappings().unwrap()[0].source_dset_name, "od%%d");
10209        drop(file);
10210        std::fs::remove_file(&path).ok();
10211    }
10212
10213    /// An unlimited mapping takes its extent from the source it names, so a
10214    /// virtual dataset written with one reports the source's rows, not the
10215    /// seed extent its dataspace message stores
10216    /// (`H5D__virtual_set_extent_unlim`). Same file, so the resolution runs
10217    /// without opening another one.
10218    #[test]
10219    fn an_unlimited_mapping_takes_its_extent_from_its_source() {
10220        let path = temp_path("vds_unlimited");
10221        {
10222            let file = H5File::create(&path).unwrap();
10223            file.new_dataset::<i32>()
10224                .shape([5usize, 2])
10225                .chunk(&[5, 2])
10226                .max_shape(&[None, Some(2)])
10227                .create("src")
10228                .unwrap()
10229                .write_raw(&(0..10i32).collect::<Vec<_>>())
10230                .unwrap();
10231            file.new_dataset::<i32>()
10232                .shape([1usize, 2])
10233                .max_shape(&[None, Some(2)])
10234                .virtual_mapping(unlimited_rows(), ".", "src", unlimited_rows())
10235                .create("vds")
10236                .unwrap();
10237            file.close().unwrap();
10238        }
10239        let file = H5File::open(&path).unwrap();
10240        let ds = file.dataset("vds").unwrap();
10241        assert_eq!(ds.shape(), vec![5, 2]);
10242        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..10).collect::<Vec<i32>>());
10243        drop(file);
10244        std::fs::remove_file(&path).ok();
10245    }
10246
10247    /// The blocks-0/1/3 printf file every dataset-access test below reads,
10248    /// laid out exactly like `a_printf_mapping_stitches_one_source_per_block`
10249    /// so the gap is the only thing that changes.
10250    fn printf_gap_file(tag: &str) -> std::path::PathBuf {
10251        use crate::Selection;
10252        let path = temp_path(tag);
10253        let file = H5File::create(&path).unwrap();
10254        for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10255            file.new_dataset::<i32>()
10256                .shape([2usize])
10257                .create(&format!("block{b}"))
10258                .unwrap()
10259                .write_raw(&[base, base + 1])
10260                .unwrap();
10261        }
10262        file.new_dataset::<i32>()
10263            .shape([1usize, 2])
10264            .max_shape(&[None, Some(2)])
10265            .fill_value(-7i32)
10266            .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10267            .create("vds")
10268            .unwrap();
10269        file.close().unwrap();
10270        path
10271    }
10272
10273    /// `H5Pset_virtual_printf_gap` lets the block scan look past a missing
10274    /// source, and the blocks it looked past stay inside the extent reading
10275    /// as the fill value (H5Dvirtual.c:1519-1614, :2661-2665). Measured
10276    /// against libhdf5 1.14.6 through h5py's `h5p.PropDAID`: gap 0 gives
10277    /// two rows, gap 1 and gap 2 both give four with row 2 filled.
10278    #[test]
10279    fn a_printf_gap_looks_past_the_missing_block() {
10280        use crate::DatasetAccess;
10281        let path = printf_gap_file("vds_printf_gap");
10282        let file = H5File::open(&path).unwrap();
10283        for (gap, shape, data) in [
10284            (0u64, vec![2usize, 2], vec![0i32, 1, 100, 101]),
10285            (1, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10286            (2, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10287        ] {
10288            let ds = file
10289                .dataset_with("vds", DatasetAccess::new().virtual_printf_gap(gap))
10290                .unwrap();
10291            assert_eq!(ds.shape(), shape, "gap {gap}");
10292            assert_eq!(ds.read_raw::<i32>().unwrap(), data, "gap {gap}");
10293        }
10294        // Back to the default: the extent follows the properties the open
10295        // names, in both directions.
10296        let ds = file.dataset("vds").unwrap();
10297        assert_eq!(ds.shape(), vec![2, 2]);
10298        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10299        drop(file);
10300        std::fs::remove_file(&path).ok();
10301    }
10302
10303    /// A relatively-named source is found next to the virtual dataset even
10304    /// when the process is somewhere else entirely: `H5F_prefix_open_file`
10305    /// tries the primary file's `H5F_EXTPATH` — the directory it was opened
10306    /// from — before the bare relative name against the working directory
10307    /// (H5Fint.c:952-977). Measured against libhdf5 1.14.6 through h5py: a
10308    /// `VirtualSource("src.h5", ...)` beside its VDS reads its data with
10309    /// `HDF5_VDS_PREFIX` unset and the working directory elsewhere; before
10310    /// the reader took that step it read back all fill value.
10311    #[test]
10312    fn a_relative_source_resolves_next_to_the_virtual_dataset() {
10313        use crate::Selection;
10314        let dir = std::env::temp_dir().join(format!(
10315            "rust_hdf5_vds_beside_{}_{:?}",
10316            std::process::id(),
10317            std::thread::current().id()
10318        ));
10319        std::fs::create_dir_all(&dir).unwrap();
10320        {
10321            let file = H5File::create(dir.join("src.h5")).unwrap();
10322            file.new_dataset::<i32>()
10323                .shape([2usize, 4])
10324                .create("data")
10325                .unwrap()
10326                .write_raw(&(0..8i32).collect::<Vec<_>>())
10327                .unwrap();
10328            file.close().unwrap();
10329        }
10330        {
10331            let file = H5File::create(dir.join("v.h5")).unwrap();
10332            file.new_dataset::<i32>()
10333                .shape([2usize, 4])
10334                .fill_value(-9i32)
10335                // Named relatively, as h5py's `VirtualSource("src.h5", ...)`
10336                // stores it — nothing in the file says where it lives.
10337                .virtual_mapping(Selection::All, "src.h5", "data", Selection::All)
10338                .create("v")
10339                .unwrap();
10340            file.close().unwrap();
10341        }
10342        // The working directory is the crate root under `cargo test`, not
10343        // `dir`, so only the extpath step can find `src.h5`.
10344        assert_ne!(std::env::current_dir().unwrap(), dir);
10345        let file = H5File::open(dir.join("v.h5")).unwrap();
10346        let ds = file.dataset("v").unwrap();
10347        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8i32).collect::<Vec<_>>());
10348        drop(file);
10349        std::fs::remove_dir_all(&dir).ok();
10350    }
10351
10352    /// The first open of a virtual dataset fixes its access properties for
10353    /// every open that overlaps it: only the open that finds no shared info
10354    /// in `H5FO_opened` runs `H5D__open_oid(dataset, dapl_id)`, and a later
10355    /// one just points at that shared info without ever reading its own dapl
10356    /// (H5Dint.c:1496-1500, :1523-1528) — the view and the gap live in the
10357    /// shared layout storage `H5D__virtual_init` filled from that first dapl
10358    /// (H5Dvirtual.c:2178-2188). Measured against libhdf5 1.14.6 and 2.0.0
10359    /// through `h5d.open(..., dapl=...)` on the printf-gap VDS below: opening
10360    /// gap 0 then gap 1 gives both handles two rows; with every handle closed,
10361    /// opening gap 1 then gap 0 gives both four rows and the gap row reads as
10362    /// fill; with every handle closed again, gap 0 alone is back to two rows.
10363    #[test]
10364    fn the_first_open_of_a_virtual_dataset_fixes_the_properties_for_later_opens() {
10365        use crate::DatasetAccess;
10366        let path = printf_gap_file("vds_printf_first_open");
10367        let file = H5File::open(&path).unwrap();
10368        let gap = |g: u64| DatasetAccess::new().virtual_printf_gap(g);
10369        {
10370            let a = file.dataset_with("vds", gap(0)).unwrap();
10371            let b = file.dataset_with("vds", gap(1)).unwrap();
10372            assert_eq!(a.shape(), vec![2, 2]);
10373            assert_eq!(b.shape(), vec![2, 2]);
10374            assert_eq!(b.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10375        }
10376        {
10377            let c = file.dataset_with("vds", gap(1)).unwrap();
10378            let d = file.dataset_with("vds", gap(0)).unwrap();
10379            assert_eq!(c.shape(), vec![4, 2]);
10380            assert_eq!(d.shape(), vec![4, 2]);
10381            assert_eq!(
10382                d.read_raw::<i32>().unwrap(),
10383                vec![0, 1, 100, 101, -7, -7, 300, 301]
10384            );
10385        }
10386        let e = file.dataset_with("vds", gap(0)).unwrap();
10387        assert_eq!(e.shape(), vec![2, 2]);
10388        assert_eq!(e.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10389        drop(e);
10390        drop(file);
10391        std::fs::remove_file(&path).ok();
10392    }
10393
10394    /// `H5D_VDS_FIRST_MISSING` ignores the printf gap: `H5D__virtual_init`
10395    /// reads the gap property only under `H5D_VDS_LAST_AVAILABLE` and forces
10396    /// it to 0 otherwise (H5Dvirtual.c:2182-2188). Measured against libhdf5
10397    /// 1.14.6: gap 2 under this view still gives two rows.
10398    #[test]
10399    fn the_first_missing_view_ignores_the_printf_gap() {
10400        use crate::{DatasetAccess, VirtualView};
10401        let path = printf_gap_file("vds_printf_first_missing");
10402        let file = H5File::open(&path).unwrap();
10403        for gap in [0u64, 2] {
10404            let ds = file
10405                .dataset_with(
10406                    "vds",
10407                    DatasetAccess::new()
10408                        .virtual_view(VirtualView::FirstMissing)
10409                        .virtual_printf_gap(gap),
10410                )
10411                .unwrap();
10412            assert_eq!(ds.shape(), vec![2, 2], "gap {gap}");
10413            assert_eq!(
10414                ds.read_raw::<i32>().unwrap(),
10415                vec![0, 1, 100, 101],
10416                "gap {gap}"
10417            );
10418        }
10419        drop(file);
10420        std::fs::remove_file(&path).ok();
10421    }
10422
10423    /// On a mapping unlimited on both sides the view is
10424    /// `H5S_hyper_get_clip_extent_match`'s `incl_trail`
10425    /// (H5Dvirtual.c:1447-1451): with a stride wider than its block, the
10426    /// extent under `H5D_VDS_LAST_AVAILABLE` ends at the last mapped row,
10427    /// and under `H5D_VDS_FIRST_MISSING` it runs on to where the next block
10428    /// would start. Measured against libhdf5 1.14.6 over a three-row source
10429    /// with stride 3 and block 2: two rows and three rows.
10430    #[test]
10431    fn the_view_decides_whether_a_trailing_gap_is_inside_the_extent() {
10432        use crate::format::selection::UNLIMITED;
10433        use crate::{DatasetAccess, Hyperslab, RegularHyperslab, Selection, VirtualView};
10434        let strided = || Selection::Hyperslab {
10435            rank: 2,
10436            form: Hyperslab::Regular(RegularHyperslab {
10437                start: vec![0, 0],
10438                stride: vec![3, 1],
10439                count: vec![UNLIMITED, 1],
10440                block: vec![2, 2],
10441            }),
10442        };
10443        let path = temp_path("vds_view_trail");
10444        {
10445            let file = H5File::create(&path).unwrap();
10446            file.new_dataset::<i32>()
10447                .shape([3usize, 2])
10448                .max_shape(&[None, Some(2)])
10449                .chunk(&[1, 2])
10450                .create("src")
10451                .unwrap()
10452                .write_raw(&(0..6i32).collect::<Vec<_>>())
10453                .unwrap();
10454            file.new_dataset::<i32>()
10455                .shape([1usize, 2])
10456                .max_shape(&[None, Some(2)])
10457                .fill_value(-9i32)
10458                .virtual_mapping(strided(), ".", "src", strided())
10459                .create("vds")
10460                .unwrap();
10461            file.close().unwrap();
10462        }
10463        let file = H5File::open(&path).unwrap();
10464        {
10465            let last = file.dataset("vds").unwrap();
10466            assert_eq!(last.shape(), vec![2, 2]);
10467            assert_eq!(last.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3]);
10468        }
10469        // The handle above is gone, so this open is the one that resolves.
10470        let first = file
10471            .dataset_with(
10472                "vds",
10473                DatasetAccess::new().virtual_view(VirtualView::FirstMissing),
10474            )
10475            .unwrap();
10476        assert_eq!(first.shape(), vec![3, 2]);
10477        assert_eq!(first.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3, -9, -9]);
10478        drop(file);
10479        std::fs::remove_file(&path).ok();
10480    }
10481
10482    /// The property list reads back what was set (`H5Pget_virtual_view`,
10483    /// `H5Pget_virtual_printf_gap`), and the gap `HSIZE_UNDEF` that
10484    /// `H5Pset_virtual_printf_gap` refuses is refused by the open that would
10485    /// have used it.
10486    #[test]
10487    fn the_access_property_list_reads_back_and_refuses_hsize_undef() {
10488        use crate::{DatasetAccess, VirtualView};
10489        let plist = DatasetAccess::new();
10490        assert_eq!(plist.view(), VirtualView::LastAvailable);
10491        assert_eq!(plist.printf_gap(), 0);
10492        let set = plist
10493            .virtual_view(VirtualView::FirstMissing)
10494            .virtual_printf_gap(4);
10495        assert_eq!(set.view(), VirtualView::FirstMissing);
10496        // The *property* keeps what was set even though the resolution under
10497        // this view scans with 0.
10498        assert_eq!(set.printf_gap(), 4);
10499
10500        let path = printf_gap_file("vds_printf_gap_undef");
10501        let file = H5File::open(&path).unwrap();
10502        let err = match file.dataset_with("vds", DatasetAccess::new().virtual_printf_gap(u64::MAX))
10503        {
10504            Ok(_) => panic!("HSIZE_UNDEF is not a valid printf gap size"),
10505            Err(e) => e.to_string(),
10506        };
10507        assert!(err.contains("HSIZE_UNDEF"), "{err}");
10508        drop(file);
10509        std::fs::remove_file(&path).ok();
10510    }
10511
10512    /// The rank-2 `count = (H5S_UNLIMITED, 1)`, `block = (1, 2)` selection
10513    /// both sides of an unlimited row-wise mapping use.
10514    fn unlimited_rows() -> crate::Selection {
10515        use crate::format::selection::UNLIMITED;
10516        use crate::{Hyperslab, RegularHyperslab, Selection};
10517        Selection::Hyperslab {
10518            rank: 2,
10519            form: Hyperslab::Regular(RegularHyperslab {
10520                start: vec![0, 0],
10521                stride: vec![1, 1],
10522                count: vec![UNLIMITED, 1],
10523                block: vec![1, 2],
10524            }),
10525        }
10526    }
10527
10528    /// An unlimited virtual selection over a *limited* source selection is
10529    /// the printf shape, and without a `%b` in a source name there is no
10530    /// second dataset to fill the second block —
10531    /// `H5D_virtual_check_mapping_post` refuses it, and so does this.
10532    #[test]
10533    fn an_unlimited_virtual_selection_over_a_limited_source_is_refused() {
10534        use crate::Selection;
10535        let path = temp_path("vds_unlim_limited_src");
10536        let file = H5File::create(&path).unwrap();
10537        let err = match file
10538            .new_dataset::<i32>()
10539            .shape([1usize, 2])
10540            .max_shape(&[None, Some(2)])
10541            .virtual_mapping(unlimited_rows(), "src.h5", "src", Selection::All)
10542            .create("vds")
10543        {
10544            Ok(_) => panic!("no printf substitution names the mapping's later blocks"),
10545            Err(e) => e.to_string(),
10546        };
10547        assert!(err.contains("printf"), "{err}");
10548        file.close().unwrap();
10549        std::fs::remove_file(&path).ok();
10550    }
10551
10552    /// A virtual dataset stores nothing of its own, so it cannot also be one
10553    /// of the storage classes that do.
10554    #[test]
10555    fn a_virtual_dataset_cannot_also_be_chunked_or_external() {
10556        use crate::Selection;
10557        let path = temp_path("vds_exclusive");
10558        let file = H5File::create(&path).unwrap();
10559        let builder = || {
10560            file.new_dataset::<i32>().shape([16usize]).virtual_mapping(
10561                Selection::All,
10562                "src.h5",
10563                "src",
10564                Selection::All,
10565            )
10566        };
10567        for (which, res) in [
10568            ("chunked", builder().chunk(&[4]).create("a")),
10569            ("compact", builder().compact().create("b")),
10570            (
10571                "external",
10572                builder().external(&[("x.raw", 0, 64)]).create("c"),
10573            ),
10574            ("references", builder().object_references().create("d")),
10575        ] {
10576            match res {
10577                Ok(_) => panic!("a virtual dataset cannot also be {which}"),
10578                Err(e) => assert!(e.to_string().contains("virtual dataset"), "{which}: {e}"),
10579            }
10580        }
10581        file.close().unwrap();
10582        std::fs::remove_file(&path).ok();
10583    }
10584
10585    /// Deleting a virtual dataset frees the global heap object its mapping
10586    /// list lived in — `H5D__virtual_delete` reaches `H5HG_remove` — so the
10587    /// next dataset's own heap object reuses that space instead of the file
10588    /// growing by a whole collection per deleted virtual dataset.
10589    #[test]
10590    fn deleting_a_virtual_dataset_frees_its_mapping_list() {
10591        use crate::Selection;
10592        let path = temp_path("vds_delete");
10593        let baseline = {
10594            let file = H5File::create(&path).unwrap();
10595            file.new_dataset::<i32>()
10596                .shape([16usize])
10597                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10598                .create("vds")
10599                .unwrap();
10600            file.delete_dataset("vds").unwrap();
10601            file.close().unwrap();
10602            std::fs::metadata(&path).unwrap().len()
10603        };
10604        std::fs::remove_file(&path).ok();
10605
10606        // Ten more create-then-delete rounds must land on the same file size:
10607        // each round's heap object is removed, its collection becomes empty
10608        // and returns to the allocator, and the next round takes it back.
10609        let file = H5File::create(&path).unwrap();
10610        for i in 0..10 {
10611            let name = format!("vds{i}");
10612            file.new_dataset::<i32>()
10613                .shape([16usize])
10614                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10615                .create(&name)
10616                .unwrap();
10617            file.delete_dataset(&name).unwrap();
10618        }
10619        file.close().unwrap();
10620        assert_eq!(std::fs::metadata(&path).unwrap().len(), baseline);
10621        std::fs::remove_file(&path).ok();
10622    }
10623
10624    /// [`H5Dataset::storage_layout`] tells the four classes apart — the
10625    /// negative case for any one class is simply that it is not another.
10626    #[test]
10627    fn storage_layout_reports_each_class() {
10628        use crate::StorageLayout;
10629        let path = temp_path("storage_layout");
10630        {
10631            let file = H5File::create(&path).unwrap();
10632            file.new_dataset::<i32>()
10633                .shape([4usize])
10634                .create("contig")
10635                .unwrap();
10636            file.new_dataset::<i32>()
10637                .shape([4usize])
10638                .compact()
10639                .create("compact")
10640                .unwrap();
10641            file.new_dataset::<i32>()
10642                .shape([8usize])
10643                .chunk(&[4])
10644                .create("chunked")
10645                .unwrap();
10646            file.close().unwrap();
10647        }
10648        let file = H5File::open(&path).unwrap();
10649        assert_eq!(
10650            file.dataset("contig").unwrap().storage_layout().unwrap(),
10651            StorageLayout::Contiguous
10652        );
10653        assert_eq!(
10654            file.dataset("compact").unwrap().storage_layout().unwrap(),
10655            StorageLayout::Compact
10656        );
10657        assert_eq!(
10658            file.dataset("chunked").unwrap().storage_layout().unwrap(),
10659            StorageLayout::Chunked
10660        );
10661    }
10662
10663    /// Read-mode-only accessor, matching `datatype()`'s own contract.
10664    #[test]
10665    fn storage_layout_errors_in_write_mode() {
10666        let path = temp_path("storage_layout_write_mode");
10667        let file = H5File::create(&path).unwrap();
10668        let ds = file
10669            .new_dataset::<i32>()
10670            .shape([4usize])
10671            .create("data")
10672            .unwrap();
10673        assert!(ds.storage_layout().is_err());
10674        file.close().unwrap();
10675    }
10676
10677    /// [`H5Dataset::chunk_index`] reports the real on-disk index kind
10678    /// (extensible array for one unlimited dimension, version-1 B-tree
10679    /// under a legacy libver bound) and `None` for an unchunked dataset —
10680    /// the negative case.
10681    #[test]
10682    fn chunk_index_reports_the_stored_kind() {
10683        use crate::ChunkIndex;
10684        let path = temp_path("chunk_index");
10685        {
10686            let file = H5File::create(&path).unwrap();
10687            file.new_dataset::<i32>()
10688                .shape([4usize])
10689                .create("contig")
10690                .unwrap();
10691            file.new_dataset::<i32>()
10692                .shape([16usize])
10693                .chunk(&[4])
10694                .max_shape(&[None])
10695                .create("earray")
10696                .unwrap();
10697            file.set_libver_latest(false).unwrap();
10698            file.new_dataset::<i32>()
10699                .shape([8usize])
10700                .chunk(&[4])
10701                .max_shape(&[None])
10702                .create("btree1")
10703                .unwrap();
10704            file.close().unwrap();
10705        }
10706        let file = H5File::open(&path).unwrap();
10707        assert_eq!(file.dataset("contig").unwrap().chunk_index().unwrap(), None);
10708        assert_eq!(
10709            file.dataset("earray").unwrap().chunk_index().unwrap(),
10710            Some(ChunkIndex::ExtensibleArray)
10711        );
10712        assert_eq!(
10713            file.dataset("btree1").unwrap().chunk_index().unwrap(),
10714            Some(ChunkIndex::BtreeV1)
10715        );
10716    }
10717
10718    /// [`H5Dataset::filters`] reports the stored pipeline in order — and
10719    /// the negative case: an unfiltered dataset reports an empty pipeline,
10720    /// not an error.
10721    #[test]
10722    fn filters_reports_the_stored_pipeline() {
10723        use crate::format::messages::filter::{FILTER_DEFLATE, FILTER_SHUFFLE, FLAG_OPTIONAL};
10724        let path = temp_path("filters");
10725        {
10726            let file = H5File::create(&path).unwrap();
10727            file.new_dataset::<i32>()
10728                .shape([16usize])
10729                .create("unfiltered")
10730                .unwrap();
10731            file.new_dataset::<i32>()
10732                .shape([16usize])
10733                .chunk(&[4])
10734                .shuffle()
10735                .deflate(6)
10736                .create("filtered")
10737                .unwrap();
10738            file.close().unwrap();
10739        }
10740        let file = H5File::open(&path).unwrap();
10741        assert_eq!(
10742            file.dataset("unfiltered").unwrap().filters().unwrap(),
10743            Vec::new()
10744        );
10745        let filters = file.dataset("filtered").unwrap().filters().unwrap();
10746        assert_eq!(filters.len(), 2);
10747        assert_eq!(filters[0].id, FILTER_SHUFFLE);
10748        assert_eq!(filters[0].flags, FLAG_OPTIONAL);
10749        assert_eq!(filters[1].id, FILTER_DEFLATE);
10750        assert_eq!(filters[1].cd_values, vec![6]);
10751    }
10752
10753    /// [`H5Dataset::fill_value`] reports the explicit bytes for a dataset
10754    /// created with `.fill_value(...)`, and the negative case: a dataset
10755    /// with no fill value set reports [`FillValue::Default`], not an error.
10756    /// `FillValue::Undefined` has no constructor on either this crate's
10757    /// writer or h5py's public API, so it is not exercised here.
10758    #[test]
10759    fn fill_value_reports_the_stored_value() {
10760        use crate::FillValue;
10761        let path = temp_path("fill_value");
10762        {
10763            let file = H5File::create(&path).unwrap();
10764            file.new_dataset::<i32>()
10765                .shape([4usize])
10766                .create("unset")
10767                .unwrap();
10768            file.new_dataset::<i32>()
10769                .shape([4usize])
10770                .fill_value(-7i32)
10771                .create("set")
10772                .unwrap();
10773            file.close().unwrap();
10774        }
10775        let file = H5File::open(&path).unwrap();
10776        assert_eq!(
10777            file.dataset("unset").unwrap().fill_value().unwrap(),
10778            FillValue::Default
10779        );
10780        assert_eq!(
10781            file.dataset("set").unwrap().fill_value().unwrap(),
10782            FillValue::UserDefined((-7i32).to_le_bytes().to_vec())
10783        );
10784    }
10785
10786    /// The three `H5D__efl_construct` / `H5Pset_external` rules an
10787    /// `H5O_EFL_UNLIMITED` slot lives inside: it may only be the last slot,
10788    /// an unlimited dataspace must have one, and only the first dimension
10789    /// may be extendible.
10790    #[test]
10791    fn the_unlimited_external_slot_keeps_its_three_rules() {
10792        use crate::format::messages::external_file_list::UNLIMITED;
10793        let path = temp_path("efl_unlim_rules");
10794        let file = H5File::create(&path).unwrap();
10795
10796        // "previous file size is unlimited": nothing behind an unlimited slot
10797        // could ever be reached.
10798        let err = match file
10799            .new_dataset::<i32>()
10800            .shape([8usize])
10801            .external(&[("a.raw", 0, UNLIMITED), ("b.raw", 0, 32)])
10802            .create("mid")
10803        {
10804            Ok(_) => panic!("an unlimited slot absorbs everything behind it"),
10805            Err(e) => e.to_string(),
10806        };
10807        assert!(err.contains("only be the last"), "{err}");
10808
10809        // "unlimited dataspace but finite storage".
10810        let err = match file
10811            .new_dataset::<i32>()
10812            .shape([8usize])
10813            .max_shape(&[None])
10814            .external(&[("a.raw", 0, 32)])
10815            .create("finite")
10816        {
10817            Ok(_) => panic!("no finite reservation covers an unlimited extent"),
10818            Err(e) => e.to_string(),
10819        };
10820        assert!(err.contains("unlimited dataspace"), "{err}");
10821
10822        // "only the first dimension can be extendible".
10823        let err = match file
10824            .new_dataset::<i32>()
10825            .shape([2usize, 4])
10826            .max_shape(&[Some(2), None])
10827            .external(&[("a.raw", 0, UNLIMITED)])
10828            .create("dim1")
10829        {
10830            Ok(_) => panic!("only the slowest-varying dimension may be extendible"),
10831            Err(e) => e.to_string(),
10832        };
10833        assert!(err.contains("only the first dimension"), "{err}");
10834
10835        // And the legal shape: an unlimited last slot under an unlimited
10836        // first dimension.
10837        file.new_dataset::<i32>()
10838            .shape([8usize])
10839            .max_shape(&[None])
10840            .external(&[("a.raw", 0, 16), ("b.raw", 0, UNLIMITED)])
10841            .create("ok")
10842            .unwrap()
10843            .write_raw(&(0..8i32).collect::<Vec<_>>())
10844            .unwrap();
10845        file.close().unwrap();
10846        std::fs::remove_file(&path).ok();
10847        for raw in ["a.raw", "b.raw"] {
10848            std::fs::remove_file(raw).ok();
10849        }
10850    }
10851
10852    /// An unlimited slot reserves nothing, so a read of it is bounded by the
10853    /// dataset's extent and by what the file physically holds: the tail past
10854    /// the end of a short raw file reads back as zero, exactly as
10855    /// `H5D__efl_read` fills it.
10856    #[test]
10857    fn an_unlimited_external_slot_reads_zero_past_the_end_of_its_file() {
10858        use crate::format::messages::external_file_list::UNLIMITED;
10859        let dir = std::env::temp_dir().join(format!("rh5_efl_short_{}", std::process::id()));
10860        std::fs::create_dir_all(&dir).unwrap();
10861        let path = dir.join("f.h5");
10862        let raw = dir.join("short.raw");
10863        // Four elements' worth of bytes for an eight-element dataset.
10864        std::fs::write(&raw, [0u8; 16]).unwrap();
10865        {
10866            let file = H5File::create(&path).unwrap();
10867            file.new_dataset::<i32>()
10868                .shape([8usize])
10869                .max_shape(&[None])
10870                .external(&[(raw.to_str().unwrap(), 0, UNLIMITED)])
10871                .create("data")
10872                .unwrap();
10873            file.close().unwrap();
10874        }
10875        let file = H5File::open(&path).unwrap();
10876        assert_eq!(
10877            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
10878            vec![0i32; 8]
10879        );
10880        drop(file);
10881        std::fs::remove_dir_all(&dir).ok();
10882    }
10883
10884    /// [`H5Dataset::external_files`] reports the stored segment list in
10885    /// order, and the negative case: a dataset whose data lives in this
10886    /// file reports an empty list, not an error.
10887    #[test]
10888    fn external_files_reports_the_stored_segments() {
10889        let path = temp_path("external_files");
10890        {
10891            let file = H5File::create(&path).unwrap();
10892            file.new_dataset::<i32>()
10893                .shape([4usize])
10894                .create("contig")
10895                .unwrap();
10896            file.new_dataset::<i32>()
10897                .shape([16usize])
10898                .external(&[("a.raw", 0, 32), ("b.raw", 8, 32)])
10899                .create("external")
10900                .unwrap();
10901            file.close().unwrap();
10902        }
10903        let file = H5File::open(&path).unwrap();
10904        assert_eq!(
10905            file.dataset("contig").unwrap().external_files().unwrap(),
10906            Vec::new()
10907        );
10908        let segments = file.dataset("external").unwrap().external_files().unwrap();
10909        assert_eq!(segments.len(), 2);
10910        assert_eq!(segments[0].name, "a.raw");
10911        assert_eq!(segments[0].offset, 0);
10912        assert_eq!(segments[0].size, 32);
10913        assert_eq!(segments[1].name, "b.raw");
10914        assert_eq!(segments[1].offset, 8);
10915        assert_eq!(segments[1].size, 32);
10916    }
10917
10918    /// [`H5Dataset::max_shape`] reports an unlimited axis as `None` and a
10919    /// fixed one as its current size — and the negative case: a dataset
10920    /// with no maximum-dimensions message reports max == current, not an
10921    /// error.
10922    #[test]
10923    fn max_shape_reports_unlimited_and_fixed_axes() {
10924        let path = temp_path("max_shape");
10925        {
10926            let file = H5File::create(&path).unwrap();
10927            file.new_dataset::<i32>()
10928                .shape([4usize, 8])
10929                .create("fixed")
10930                .unwrap();
10931            file.new_dataset::<i32>()
10932                .shape([4usize, 8])
10933                .chunk(&[2, 8])
10934                .max_shape(&[None, Some(8)])
10935                .create("unlimited")
10936                .unwrap();
10937            file.close().unwrap();
10938        }
10939        let file = H5File::open(&path).unwrap();
10940        assert_eq!(
10941            file.dataset("fixed").unwrap().max_shape().unwrap(),
10942            vec![Some(4), Some(8)]
10943        );
10944        assert_eq!(
10945            file.dataset("unlimited").unwrap().max_shape().unwrap(),
10946            vec![None, Some(8)]
10947        );
10948    }
10949
10950    /// [`H5Dataset::virtual_mappings`] reports the stored source/virtual
10951    /// mapping list in order, and the negative case: a dataset with no
10952    /// virtual layout reports an empty list, not an error.
10953    #[test]
10954    fn virtual_mappings_reports_the_stored_mappings() {
10955        use crate::Selection;
10956        let path = temp_path("virtual_mappings");
10957        {
10958            let file = H5File::create(&path).unwrap();
10959            file.new_dataset::<i32>()
10960                .shape([4usize])
10961                .create("plain")
10962                .unwrap();
10963            file.new_dataset::<i32>()
10964                .shape([16usize])
10965                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10966                .create("vds")
10967                .unwrap();
10968            file.close().unwrap();
10969        }
10970        let file = H5File::open(&path).unwrap();
10971        assert_eq!(
10972            file.dataset("plain").unwrap().virtual_mappings().unwrap(),
10973            Vec::new()
10974        );
10975        let mappings = file.dataset("vds").unwrap().virtual_mappings().unwrap();
10976        assert_eq!(mappings.len(), 1);
10977        assert_eq!(mappings[0].source_file_name, "src.h5");
10978        assert_eq!(mappings[0].source_dset_name, "src");
10979        assert_eq!(mappings[0].source_selection, Selection::All);
10980        assert_eq!(mappings[0].virtual_selection, Selection::All);
10981    }
10982}