Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use std::borrow::Cow;
8
9use crate::attribute::AttrBuilder;
10use crate::error::{Hdf5Error, Result};
11use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
12use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
13use crate::format::messages::filter::Filter;
14use crate::format::messages::virtual_mapping::VirtualMapping;
15use crate::format::reference::{Reference, ReferenceTarget};
16use crate::format::selection::check_hyperslab;
17use crate::format::selection::Selection;
18use crate::format::storage_kind::AttributeStorage;
19use crate::io::file_handle::ReadDst;
20use crate::io::reader::{read_image_into_new, ExternalFileSegment};
21use crate::io::writer::ChunkIndexKind;
22use crate::types::H5Type;
23
24// ---------------------------------------------------------------------------
25// DatasetBuilder
26// ---------------------------------------------------------------------------
27
28/// A fluent builder for creating datasets.
29///
30/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
31///
32/// ```no_run
33/// # use rust_hdf5::H5File;
34/// let file = H5File::create("builder.h5").unwrap();
35/// let ds = file.new_dataset::<f32>()
36///     .shape(&[10, 20])
37///     .create("temperatures")
38///     .unwrap();
39/// ```
40pub struct DatasetBuilder<T: H5Type> {
41    file_inner: SharedInner,
42    shape: Option<Vec<usize>>,
43    is_null: bool,
44    chunk_dims: Option<Vec<usize>>,
45    max_shape: Option<Vec<Option<usize>>>,
46    is_compact: bool,
47    early_allocation: bool,
48    deflate_level: Option<u32>,
49    shuffle: bool,
50    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
51    group_path: Option<String>,
52    fill_value: Option<Vec<u8>>,
53    fill_time: Option<FillTime>,
54    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
55    committed_type: Option<String>,
56    references: Option<ReferenceElement>,
57    external: Option<Vec<(String, u64, u64)>>,
58    efile_prefix: Option<String>,
59    virtual_mappings: Vec<VirtualMapping>,
60    _marker: std::marker::PhantomData<T>,
61}
62
63/// Which reference a `*_references()` builder call asked the elements to be.
64///
65/// One field rather than a flag per kind: an element is a whole-object
66/// reference or a region reference, never both, and the width of each is only
67/// known once the file's address size is (see
68/// [`DatatypeMessage::object_reference`] and
69/// [`DatatypeMessage::region_reference`]).
70///
71/// [`DatatypeMessage::object_reference`]: crate::format::messages::datatype::DatatypeMessage::object_reference
72/// [`DatatypeMessage::region_reference`]: crate::format::messages::datatype::DatatypeMessage::region_reference
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74enum ReferenceElement {
75    /// `H5T_STD_REF_OBJ`.
76    Object,
77    /// `H5T_STD_REF_DSETREG`.
78    Region,
79    /// `H5T_STD_REF`, the 1.12 form. One datatype for all three 1.12 kinds:
80    /// the element leads with the kind it holds, so `H5T__ref_disk_getsize`
81    /// sizes every element for the widest of them and a dataset of this type
82    /// may hold objects, regions and attributes alike.
83    Revised,
84}
85
86impl ReferenceElement {
87    /// The stored datatype for this kind in a file with `ctx`'s address size.
88    fn datatype(
89        self,
90        ctx: &crate::format::FormatContext,
91    ) -> crate::format::messages::datatype::DatatypeMessage {
92        use crate::format::messages::datatype::DatatypeMessage;
93        match self {
94            Self::Object => DatatypeMessage::object_reference(ctx),
95            Self::Region => DatatypeMessage::region_reference(ctx),
96            Self::Revised => DatatypeMessage::std_object_reference(ctx),
97        }
98    }
99}
100
101impl<T: H5Type> DatasetBuilder<T> {
102    pub(crate) fn new(file_inner: SharedInner) -> Self {
103        Self {
104            file_inner,
105            shape: None,
106            is_null: false,
107            chunk_dims: None,
108            max_shape: None,
109            is_compact: false,
110            early_allocation: false,
111            deflate_level: None,
112            shuffle: false,
113            custom_pipeline: None,
114            group_path: None,
115            fill_value: None,
116            fill_time: None,
117            datatype_override: None,
118            committed_type: None,
119            references: None,
120            external: None,
121            efile_prefix: None,
122            virtual_mappings: Vec::new(),
123            _marker: std::marker::PhantomData,
124        }
125    }
126
127    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
128        Self {
129            file_inner,
130            shape: None,
131            is_null: false,
132            chunk_dims: None,
133            max_shape: None,
134            is_compact: false,
135            early_allocation: false,
136            deflate_level: None,
137            shuffle: false,
138            custom_pipeline: None,
139            group_path: Some(group_path),
140            fill_value: None,
141            fill_time: None,
142            datatype_override: None,
143            committed_type: None,
144            references: None,
145            external: None,
146            efile_prefix: None,
147            virtual_mappings: Vec::new(),
148            _marker: std::marker::PhantomData,
149        }
150    }
151
152    /// Set the dataset dimensions.
153    ///
154    /// This is required before calling [`create`](Self::create), unless
155    /// [`null`](Self::null) was called instead.
156    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
157    #[must_use]
158    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
159        self.shape = Some(dims.as_ref().to_vec());
160        self
161    }
162
163    /// Create a scalar (0-dimensional) dataset holding a single value.
164    #[must_use]
165    pub fn scalar(mut self) -> Self {
166        self.shape = Some(vec![]);
167        self
168    }
169
170    /// Create a dataset with the NULL dataspace: no elements at all.
171    ///
172    /// Distinct from [`scalar`](Self::scalar), which holds exactly one
173    /// element. A NULL dataset holds zero bytes of data and cannot be
174    /// written to — [`write_raw`](H5Dataset::write_raw) and
175    /// [`write_raw_bytes`](H5Dataset::write_raw_bytes) return an error, and
176    /// it cannot be chunked or filtered, matching h5py's `h5py.Empty`.
177    #[must_use]
178    pub fn null(mut self) -> Self {
179        self.is_null = true;
180        self
181    }
182
183    /// Set chunk dimensions for chunked storage.
184    ///
185    /// When set, the dataset uses chunked storage with the extensible array
186    /// index. You should also call [`max_shape`](Self::max_shape) or
187    /// [`resizable`](Self::resizable) to allow extending.
188    #[must_use]
189    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
190        self.chunk_dims = Some(chunk_dims.to_vec());
191        self
192    }
193
194    /// Make all dimensions unlimited (resizable).
195    ///
196    /// This sets max_dims to u64::MAX for all dimensions.
197    #[must_use]
198    pub fn resizable(mut self) -> Self {
199        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
200        self
201    }
202
203    /// Set maximum dimensions. `None` means unlimited for that dimension.
204    #[must_use]
205    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
206        self.max_shape = Some(max.to_vec());
207        self
208    }
209
210    /// Store the raw data inside the dataset's object header —
211    /// `H5Pset_layout(dcpl, H5D_COMPACT)`.
212    ///
213    /// A compact dataset costs no data block and no second seek to read, which
214    /// suits the small per-run constants an analysis file is full of. It is
215    /// bounded by what one object header message can hold
216    /// ([`MAX_COMPACT_DATA`](crate::MAX_COMPACT_DATA) bytes) and it
217    /// is fixed in size: [`chunk`](Self::chunk), a filter, and an unlimited
218    /// [`max_shape`](Self::max_shape) are all rejected at
219    /// [`create`](Self::create), as libhdf5 rejects them.
220    ///
221    /// ```no_run
222    /// # use rust_hdf5::H5File;
223    /// let file = H5File::create("compact.h5").unwrap();
224    /// let ds = file.new_dataset::<i32>()
225    ///     .shape([16])
226    ///     .compact()
227    ///     .create("data")
228    ///     .unwrap();
229    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
230    /// ```
231    #[must_use]
232    pub fn compact(mut self) -> Self {
233        self.is_compact = true;
234        self
235    }
236
237    /// Allocate the whole of a chunked dataset's storage at create —
238    /// `H5Pset_alloc_time(dcpl, H5D_ALLOC_TIME_EARLY)`, h5py's
239    /// `alloc_time=h5d.ALLOC_TIME_EARLY`.
240    ///
241    /// Every chunk exists, holding the fill value, before anything is
242    /// written, so an unwritten chunk costs a read of fill bytes rather than
243    /// a miss. On a fixed-shape unfiltered dataset that is also what lets
244    /// libhdf5 pick its cheapest chunk index — the *implicit* index, which
245    /// is no index at all: the chunks are one contiguous run in grid order
246    /// and a chunk's address is arithmetic. This builder makes the same
247    /// choice under the same conditions, so such a dataset is written with
248    /// no index structure in the file.
249    ///
250    /// Ignored by storage that has no chunk grid to allocate: contiguous,
251    /// compact and NULL-dataspace datasets.
252    ///
253    /// ```no_run
254    /// # use rust_hdf5::H5File;
255    /// let file = H5File::create("implicit.h5").unwrap();
256    /// let ds = file.new_dataset::<i32>()
257    ///     .shape([16])
258    ///     .chunk(&[4])
259    ///     .early_allocation()
260    ///     .create("data")
261    ///     .unwrap();
262    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
263    /// ```
264    #[must_use]
265    pub fn early_allocation(mut self) -> Self {
266        self.early_allocation = true;
267        self
268    }
269
270    /// Enable deflate (gzip) compression with the given level (0-9).
271    ///
272    /// Requires chunked storage (call `.chunk()` before `.create()`).
273    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
274    #[must_use]
275    pub fn deflate(mut self, level: u32) -> Self {
276        self.deflate_level = Some(level);
277        self
278    }
279
280    /// Enable the shuffle filter — `H5Pset_shuffle(dcpl)`, h5py's
281    /// `shuffle=True`.
282    ///
283    /// Shuffle reorders a chunk's bytes by their position within an element,
284    /// which typically improves how well a compressor behind it does on
285    /// numeric data. It is a permutation, not a compressor: on its own it
286    /// leaves the chunk exactly as large as it was, which is what
287    /// `H5Pset_shuffle` without a compressor writes. Combine it with
288    /// [`deflate`](Self::deflate) to compress the shuffled stream. Requires
289    /// chunked storage.
290    ///
291    /// The element width the filter records is the dataset's, so a
292    /// [`datatype`](Self::datatype) override is what it follows when the
293    /// stored element is not `T` itself.
294    #[must_use]
295    pub fn shuffle(mut self) -> Self {
296        self.shuffle = true;
297        self
298    }
299
300    /// Enable shuffle + deflate compression — the same pipeline as
301    /// `.shuffle().deflate(level)`.
302    ///
303    /// Shuffle reorders bytes by position within elements before compression,
304    /// which typically improves compression ratios for numeric data.
305    /// Requires chunked storage.
306    #[must_use]
307    pub fn shuffle_deflate(mut self, level: u32) -> Self {
308        self.shuffle = true;
309        self.deflate_level = Some(level);
310        self
311    }
312
313    /// Enable Zstandard compression with the given level (1-22, default 3).
314    ///
315    /// Requires chunked storage (call `.chunk()` before `.create()`).
316    #[must_use]
317    pub fn zstd(mut self, level: u32) -> Self {
318        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
319        self
320    }
321
322    /// Set a custom filter pipeline for compression.
323    ///
324    /// This takes precedence over [`deflate`](Self::deflate) and
325    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
326    #[must_use]
327    pub fn filter_pipeline(
328        mut self,
329        pipeline: crate::format::messages::filter::FilterPipeline,
330    ) -> Self {
331        self.custom_pipeline = Some(pipeline);
332        self
333    }
334
335    /// Override the stored element datatype.
336    ///
337    /// By default the dataset is created with the datatype derived from the
338    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
339    /// different on-disk datatype than the in-memory element type — for
340    /// example a reduced-precision fixed-point type that matches an N-bit
341    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
342    /// override must equal `T::element_size()`; the N-bit filter packs the
343    /// significant bits within that fixed footprint.
344    ///
345    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
346    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
347    #[must_use]
348    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
349        self.datatype_override = Some(dt);
350        self
351    }
352
353    /// Build the dataset on the committed (named) datatype at `path` —
354    /// h5py's `dtype=f["name"]`, `H5Dcreate2` with a committed type id.
355    ///
356    /// The dataset does not describe its type: its header stores a pointer to
357    /// that object, so the type is defined once and every dataset sharing it
358    /// is guaranteed to agree. The type comes from the committed object, so
359    /// this supersedes both `T` and [`datatype`](Self::datatype).
360    ///
361    /// The path is resolved at [`create`](Self::create), which fails when no
362    /// committed datatype is there — commit it with
363    /// [`H5File::commit_datatype`](crate::file::H5File::commit_datatype)
364    /// first.
365    ///
366    /// ```no_run
367    /// # use rust_hdf5::H5File;
368    /// # use rust_hdf5::format::messages::datatype::DatatypeMessage;
369    /// let file = H5File::create("committed.h5").unwrap();
370    /// file.commit_datatype("temperature", DatatypeMessage::f64_type()).unwrap();
371    /// file.new_dataset::<f64>()
372    ///     .committed_type("temperature")
373    ///     .shape([4])
374    ///     .create("readings")
375    ///     .unwrap();
376    /// ```
377    #[must_use]
378    pub fn committed_type(mut self, path: &str) -> Self {
379        self.committed_type = Some(path.to_string());
380        self
381    }
382
383    /// Store object references — h5py's `h5py.ref_dtype`.
384    ///
385    /// The elements are written with
386    /// [`write_object_references`](H5Dataset::write_object_references) and
387    /// name objects by path. The element width is the file's address size, so
388    /// the datatype is resolved at [`create`](Self::create) rather than here;
389    /// it overrides both `T` and any [`datatype`](Self::datatype) call.
390    ///
391    /// ```no_run
392    /// # use rust_hdf5::H5File;
393    /// let file = H5File::create("refs.h5").unwrap();
394    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
395    /// let refs = file.new_dataset::<u64>()
396    ///     .object_references()
397    ///     .shape([1])
398    ///     .create("refs")
399    ///     .unwrap();
400    /// refs.write_object_references(&["/target"]).unwrap();
401    /// file.close().unwrap();
402    /// ```
403    #[must_use]
404    pub fn object_references(mut self) -> Self {
405        self.references = Some(ReferenceElement::Object);
406        self
407    }
408
409    /// Store revised object references — the 1.12 `H5T_STD_REF`.
410    ///
411    /// Same paths and same [`write_object_references`](H5Dataset::write_object_references)
412    /// call as [`object_references`](Self::object_references); only the stored
413    /// element differs, carrying the reference's kind alongside the address so
414    /// one datatype can hold every reference kind. h5py cannot read it, so
415    /// prefer the pre-1.12 form for files h5py will open.
416    ///
417    /// ```no_run
418    /// # use rust_hdf5::H5File;
419    /// let file = H5File::create("stdrefs.h5").unwrap();
420    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
421    /// let refs = file.new_dataset::<u64>()
422    ///     .std_object_references()
423    ///     .shape([1])
424    ///     .create("refs")
425    ///     .unwrap();
426    /// refs.write_object_references(&["/target"]).unwrap();
427    /// file.close().unwrap();
428    /// ```
429    #[must_use]
430    pub fn std_object_references(mut self) -> Self {
431        self.references = Some(ReferenceElement::Revised);
432        self
433    }
434
435    /// Store revised region references — `H5R_DATASET_REGION2`, written into
436    /// the same `H5T_STD_REF` datatype
437    /// [`std_object_references`](Self::std_object_references) makes.
438    ///
439    /// The elements are written with
440    /// [`write_std_region_references`](H5Dataset::write_std_region_references).
441    /// What distinguishes them from the pre-1.12
442    /// [`region_references`](Self::region_references) is the element, not the
443    /// datatype: a 1.12 element names its own kind, so one dataset of this type
444    /// may hold object, region and attribute references together. h5py 3.15
445    /// cannot read any of them, so prefer the pre-1.12 form for files h5py will
446    /// open.
447    ///
448    /// ```no_run
449    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, LibverBound, Selection};
450    /// let file = H5File::options().libver(LibverBound::V112).create("stdregions.h5").unwrap();
451    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
452    /// let refs = file.new_dataset::<u64>()
453    ///     .std_region_references()
454    ///     .shape([1])
455    ///     .create("refs")
456    ///     .unwrap();
457    /// let rows = Selection::Hyperslab {
458    ///     rank: 1,
459    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
460    /// };
461    /// refs.write_std_region_references(&[("/target", rows)]).unwrap();
462    /// file.close().unwrap();
463    /// ```
464    #[must_use]
465    pub fn std_region_references(self) -> Self {
466        self.std_object_references()
467    }
468
469    /// Store attribute references — `H5R_ATTR`, the one reference kind with no
470    /// pre-1.12 form, in the same `H5T_STD_REF` datatype
471    /// [`std_object_references`](Self::std_object_references) makes.
472    ///
473    /// The elements are written with
474    /// [`write_attribute_references`](H5Dataset::write_attribute_references) and
475    /// name an object and one of its attributes. h5py 3.15 cannot read them.
476    ///
477    /// ```no_run
478    /// # use rust_hdf5::{H5File, LibverBound};
479    /// let file = H5File::options().libver(LibverBound::V112).create("attrrefs.h5").unwrap();
480    /// let target = file.new_dataset::<i32>().shape([4]).create("target").unwrap();
481    /// target.new_attr::<i32>().shape([3]).create("note").unwrap()
482    ///     .write_array(&[7i32, 8, 9]).unwrap();
483    /// let refs = file.new_dataset::<u64>()
484    ///     .attribute_references()
485    ///     .shape([1])
486    ///     .create("refs")
487    ///     .unwrap();
488    /// refs.write_attribute_references(&[("/target", "note")]).unwrap();
489    /// file.close().unwrap();
490    /// ```
491    #[must_use]
492    pub fn attribute_references(self) -> Self {
493        self.std_object_references()
494    }
495
496    /// Store dataset region references — h5py's `h5py.regionref_dtype`.
497    ///
498    /// The elements are written with
499    /// [`write_region_references`](H5Dataset::write_region_references) and name
500    /// a dataset plus a selection over it. The element is a global-heap id, so
501    /// its width follows the file's address size and the datatype is resolved
502    /// at [`create`](Self::create) rather than here; it overrides both `T` and
503    /// any [`datatype`](Self::datatype) call.
504    ///
505    /// ```no_run
506    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, Selection};
507    /// let file = H5File::create("regions.h5").unwrap();
508    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
509    /// let refs = file.new_dataset::<u64>()
510    ///     .region_references()
511    ///     .shape([1])
512    ///     .create("refs")
513    ///     .unwrap();
514    /// let rows = Selection::Hyperslab {
515    ///     rank: 1,
516    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
517    /// };
518    /// refs.write_region_references(&[("/target", rows)]).unwrap();
519    /// file.close().unwrap();
520    /// ```
521    #[must_use]
522    pub fn region_references(mut self) -> Self {
523        self.references = Some(ReferenceElement::Region);
524        self
525    }
526
527    /// Keep the raw data in files outside this one — `H5Pset_external`,
528    /// h5py's `external=[(name, offset, size)]`.
529    ///
530    /// Each entry is a file name, the byte offset in it where that entry's
531    /// region starts, and how many bytes of the dataset it holds; the entries
532    /// concatenate, in order, into the dataset's bytes and together must cover
533    /// them. A relative name is resolved against `HDF5_EXTFILE_PREFIX` the way
534    /// libhdf5 resolves it, so the same name reads back through this crate and
535    /// through h5py. The storage is contiguous by definition, which rules out
536    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
537    /// [`null`](Self::null) and either reference kind.
538    ///
539    /// The named files are created on first write and never truncated, so
540    /// several datasets may own disjoint ranges of one file.
541    ///
542    /// The last entry may take the unlimited size
543    /// [`external_file_list::UNLIMITED`](crate::format::messages::external_file_list::UNLIMITED)
544    /// (`H5O_EFL_UNLIMITED`), which makes it absorb the whole rest of the
545    /// dataset however far it grows. A dataset whose
546    /// [`max_shape`](Self::max_shape) is unlimited must have one, since no
547    /// finite reservation could cover it, and only the first dimension may be
548    /// extendible — both `H5D__efl_construct`'s rules.
549    ///
550    /// ```no_run
551    /// # use rust_hdf5::H5File;
552    /// let file = H5File::create("ext.h5").unwrap();
553    /// let ds = file.new_dataset::<i32>()
554    ///     .shape([16])
555    ///     .external(&[("ext.raw", 0, 64)])
556    ///     .create("data")
557    ///     .unwrap();
558    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
559    /// ```
560    #[must_use]
561    pub fn external(mut self, files: &[(&str, u64, u64)]) -> Self {
562        self.external = Some(
563            files
564                .iter()
565                .map(|&(name, offset, size)| (name.to_string(), offset, size))
566                .collect(),
567        );
568        self
569    }
570
571    /// `H5Pset_efile_prefix` on the dapl `H5Dcreate2` takes — the directory
572    /// the raw data files named by [`external`](Self::external) are created
573    /// under, and looked for under on every later write through this handle.
574    ///
575    /// `H5D__create` builds `dset->shared->extfile_prefix` from the dapl
576    /// (H5Dint.c:1318) and `H5D__efl_write` joins each slot name against it
577    /// with the same single-path `H5_combine_path` the read side uses
578    /// (H5Defl.c:429-431) — so this decides where the bytes land, and the
579    /// prefix a later reader names must agree for it to find them.
580    ///
581    /// Measured under libhdf5 1.14.6 and 2.0.0: writing through a dapl that
582    /// names a directory creates the raw data file there and nowhere else,
583    /// and a directory that does not exist fails the write outright rather
584    /// than being created.
585    ///
586    /// It shares [`DatasetAccess::efile_prefix`]'s rules, both being
587    /// `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX` shadows this outright
588    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
589    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
590    /// (:1098-1102), which leaves a stored name to resolve against the
591    /// process's current directory.
592    ///
593    /// Ignored by a dataset that names no external files, which has no slot
594    /// name to join.
595    #[must_use]
596    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
597        self.efile_prefix = Some(prefix.into());
598        self
599    }
600
601    /// Map part of this dataset onto part of a dataset in another file, making
602    /// it virtual — `H5Pset_virtual`, one `VirtualLayout[...] =
603    /// VirtualSource(...)` assignment in h5py.
604    ///
605    /// The arguments are `H5Pset_virtual`'s, in its order: which elements of
606    /// *this* dataset the mapping fills, the file and dataset the data comes
607    /// from, and which elements of that source dataset it comes from. Call it
608    /// once per mapping; they apply in the order given, which is the order
609    /// libhdf5 resolves overlapping ones in.
610    ///
611    /// The source file is named exactly as stored — resolved against
612    /// `HDF5_VDS_PREFIX`, or the virtual dataset's own directory, when the
613    /// file is read — and `"."` means this file. Nothing is opened or checked
614    /// here: a source that does not exist yet is legal, and reads of the
615    /// unmapped or unresolvable parts return the [`fill_value`](Self::fill_value).
616    ///
617    /// A virtual dataset stores nothing of its own, which rules out
618    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
619    /// [`null`](Self::null), [`external`](Self::external) and either reference
620    /// kind — and makes writing to it an error, since its elements belong to
621    /// the source datasets.
622    ///
623    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: such a
624    /// mapping grows with its source, and the dataset's extent in that
625    /// dimension is whatever the sources reachable when it is opened supply
626    /// (`H5D__virtual_set_extent_unlim`). Give it a
627    /// [`max_shape`](Self::max_shape) unlimited in the same dimension, as
628    /// libhdf5 requires of the dataspace behind one.
629    ///
630    /// A source name may carry libhdf5's `printf`-style substitutions: `%b`
631    /// is the block index and `%%` an escaped literal `%`. One such mapping
632    /// stands for the family of source datasets that fill the successive
633    /// blocks of an unlimited virtual selection, so it is legal only with an
634    /// unlimited virtual selection over a limited source selection, and the
635    /// dataset's extent stops at the first block whose source is missing.
636    ///
637    /// ```no_run
638    /// # use rust_hdf5::{H5File, Selection};
639    /// let file = H5File::create("vds.h5").unwrap();
640    /// let ds = file.new_dataset::<i32>()
641    ///     .shape([16])
642    ///     .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
643    ///     .create("vds")
644    ///     .unwrap();
645    /// ```
646    #[must_use]
647    pub fn virtual_mapping(
648        mut self,
649        virtual_selection: Selection,
650        source_file: &str,
651        source_dataset: &str,
652        source_selection: Selection,
653    ) -> Self {
654        self.virtual_mappings.push(VirtualMapping {
655            source_file_name: source_file.to_string(),
656            source_dset_name: source_dataset.to_string(),
657            source_selection,
658            virtual_selection,
659        });
660        self
661    }
662
663    /// Set a user-defined fill value for unwritten elements.
664    ///
665    /// Without this, datasets use the HDF5 default zero-fill. When set,
666    /// the value is written into the dataset's fill-value message
667    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
668    /// unwritten regions as this value rather than zero.
669    ///
670    /// ```no_run
671    /// # use rust_hdf5::H5File;
672    /// let file = H5File::create("fv.h5").unwrap();
673    /// let ds = file.new_dataset::<f32>()
674    ///     .shape(&[100])
675    ///     .fill_value(f32::NAN)
676    ///     .create("data")
677    ///     .unwrap();
678    /// ```
679    #[must_use]
680    pub fn fill_value(mut self, value: T) -> Self {
681        let es = T::element_size();
682        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
683        // well-defined byte representation; `element_size()` matches
684        // `size_of::<T>()`. The slice borrows `value` only for this call.
685        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
686        self.fill_value = Some(raw.to_vec());
687        self
688    }
689
690    /// Set when the fill value is written into allocated storage —
691    /// `H5Pset_fill_time`.
692    ///
693    /// Without this, a dataset gets [`FillTime::IfSet`]
694    /// (`H5D_CRT_FILL_TIME_DEF`), the default every dataset creation
695    /// property list carries. [`FillTime::Never`] applies to a dataset with
696    /// no fill value too: it only stops this writer's own eager tiling of
697    /// the value into newly allocated storage, not the default zero-fill
698    /// that storage already has, so its only observable effect is on a
699    /// dataset that also calls [`fill_value`](Self::fill_value).
700    ///
701    /// ```no_run
702    /// # use rust_hdf5::{FillTime, H5File};
703    /// let file = H5File::create("fv.h5").unwrap();
704    /// let ds = file.new_dataset::<f32>()
705    ///     .shape(&[100])
706    ///     .fill_value(f32::NAN)
707    ///     .fill_time(FillTime::Never)
708    ///     .create("data")
709    ///     .unwrap();
710    /// ```
711    #[must_use]
712    pub fn fill_time(mut self, time: FillTime) -> Self {
713        self.fill_time = Some(time);
714        self
715    }
716
717    /// Finalize and create the dataset with the given `name`.
718    ///
719    /// The name is the link name within the root group (e.g. `"data"` or
720    /// `"group1/data"` once nested groups are supported).
721    pub fn create(self, name: &str) -> Result<H5Dataset> {
722        // A committed type is resolved before the dataset exists and recorded
723        // after, here rather than in each storage path: every path reaches
724        // this one return, so a dataset can never be built on a committed
725        // type and then fail to say so — which would silently write the type
726        // out in full instead of pointing at the object.
727        let committed = self.resolve_committed_type()?;
728        let file_inner = clone_inner(&self.file_inner);
729        let ds = self.create_object(name, committed.as_ref().map(|(_, dt)| dt.clone()))?;
730        if let (Some((share, _)), DatasetInfo::Writer { index, .. }) = (committed, &ds.info) {
731            let inner = borrow_inner(&file_inner);
732            if let H5FileInner::Writer(writer) = &*inner {
733                writer.share_committed_type(*index, share);
734            }
735        }
736        Ok(ds)
737    }
738
739    /// The committed datatype this dataset is built on, with the type it
740    /// holds; `None` when [`committed_type`](Self::committed_type) was not
741    /// called.
742    fn resolve_committed_type(&self) -> Result<Option<(usize, DatatypeMessage)>> {
743        let Some(path) = self.committed_type.as_deref() else {
744            return Ok(None);
745        };
746        if self.references.is_some() {
747            // Both name the stored type and they cannot both be it: the
748            // pointer would say the elements are the committed type while the
749            // reference writers write addresses. True of either reference
750            // kind — an object reference is an address, a region reference is
751            // a global-heap address plus a serialized selection.
752            return Err(Hdf5Error::InvalidState(
753                "a dataset cannot be built on a committed datatype and hold references".into(),
754            ));
755        }
756        let inner = borrow_inner(&self.file_inner);
757        match &*inner {
758            H5FileInner::Writer(writer) => Ok(Some(writer.committed_datatype_for_share(path)?)),
759            H5FileInner::Reader(_) => Err(Hdf5Error::InvalidState(
760                "cannot create a dataset in read mode".into(),
761            )),
762            H5FileInner::Closed => Err(Hdf5Error::InvalidState("file is closed".into())),
763        }
764    }
765
766    /// Everything [`create`](Self::create) does apart from recording the
767    /// committed-type share; `committed` is the type that object holds.
768    fn create_object(self, name: &str, committed: Option<DatatypeMessage>) -> Result<H5Dataset> {
769        // Build the full name: if created within a group, prefix with group path
770        let full_name = if let Some(ref gp) = self.group_path {
771            if gp == "/" {
772                name.to_string()
773            } else {
774                let trimmed = gp.trim_start_matches('/');
775                format!("{}/{}", trimmed, name)
776            }
777        } else {
778            name.to_string()
779        };
780
781        let datatype = if let Some(kind) = self.references {
782            // The element is measured in file addresses, and only the writer
783            // knows how wide one is for this file.
784            let inner = borrow_inner(&self.file_inner);
785            match &*inner {
786                H5FileInner::Writer(writer) => kind.datatype(writer.ctx()),
787                H5FileInner::Reader(_) => {
788                    return Err(Hdf5Error::InvalidState(
789                        "cannot create a dataset in read mode".into(),
790                    ))
791                }
792                H5FileInner::Closed => {
793                    return Err(Hdf5Error::InvalidState("file is closed".into()))
794                }
795            }
796        } else if let Some(dt) = committed {
797            // The object header holds the type; the dataset stores a pointer
798            // to it, but every size and payload check still needs the type
799            // itself.
800            dt
801        } else {
802            self.datatype_override.clone().unwrap_or_else(T::hdf5_type)
803        };
804        // Size one element from the on-disk datatype, not the carrier `T`. For
805        // the default path this equals `T::element_size()`; when a `datatype()`
806        // override is set (N-bit, or a runtime `CompoundType`), the stored type
807        // — not `T` — defines the element width, so the dataspace, the raw
808        // allocation, and the `write_raw` length check all agree with the bytes
809        // libhdf5/h5py will read.
810        let element_size = datatype.element_size() as usize;
811        // `fill_value` took the host image of a `T`; the fill-value message
812        // holds one element in the dataset's own datatype, so it is converted
813        // here — the order is only known once the override is resolved, and
814        // the builder's calls can arrive in either order.
815        let fill_value = match self.fill_value.as_deref() {
816            Some(bytes) => Some(to_stored_byte_order(bytes, &datatype, element_size)?.into_owned()),
817            None => None,
818        };
819
820        let wants_filter =
821            self.custom_pipeline.is_some() || self.shuffle || self.deflate_level.is_some();
822
823        // External storage *is* contiguous storage: the layout message says
824        // contiguous with an undefined address, and the External File List
825        // beside it says where the bytes really are. Every other storage class
826        // names bytes of its own, so none of them can also name these.
827        if self.external.is_some() {
828            if self.chunk_dims.is_some() || wants_filter || self.is_compact || self.is_null {
829                return Err(Hdf5Error::InvalidState(
830                    "a dataset whose raw data lives in external files is contiguous, so it \
831                     cannot also be chunked, filtered, compact or NULL"
832                        .into(),
833                ));
834            }
835            if self.references.is_some() {
836                return Err(Hdf5Error::InvalidState(
837                    "object and region references are stamped into the dataset's own \
838                     contiguous block, which a dataset stored in external files has none of"
839                        .into(),
840                ));
841            }
842        }
843
844        // A virtual dataset stores nothing of its own — its elements are read
845        // out of the datasets its mappings name — so it can be none of the
846        // storage classes that do, and there is no block for a reference
847        // writer to stamp into either.
848        if !self.virtual_mappings.is_empty() {
849            if self.chunk_dims.is_some()
850                || wants_filter
851                || self.is_compact
852                || self.is_null
853                || self.external.is_some()
854            {
855                return Err(Hdf5Error::InvalidState(
856                    "a virtual dataset's elements live in the datasets its mappings name, \
857                     so it cannot also be chunked, filtered, compact, NULL or stored in \
858                     external files"
859                        .into(),
860                ));
861            }
862            if self.references.is_some() {
863                return Err(Hdf5Error::InvalidState(
864                    "object and region references are stamped into the dataset's own \
865                     contiguous block, which a virtual dataset has none of"
866                        .into(),
867                ));
868            }
869        }
870
871        if self.is_null {
872            // A NULL dataspace holds no elements at all: no chunk grid to
873            // scatter into, no raw image to put in an object header, no fill
874            // value to apply to unwritten elements (there are none), matching
875            // upstream's rejection of these combinations (`H5Dchunk.c`'s
876            // chunked-layout dataspace check).
877            if self.chunk_dims.is_some() || wants_filter || self.is_compact {
878                return Err(Hdf5Error::InvalidState(
879                    "a NULL dataspace dataset cannot be chunked, filtered or compact".into(),
880                ));
881            }
882            if fill_value.is_some() {
883                return Err(Hdf5Error::InvalidState(
884                    "a NULL dataspace dataset cannot have a fill value".into(),
885                ));
886            }
887            if self.fill_time.is_some() {
888                return Err(Hdf5Error::InvalidState(
889                    "a NULL dataspace dataset cannot have a fill time".into(),
890                ));
891            }
892
893            let index = {
894                let inner = borrow_inner(&self.file_inner);
895                match &*inner {
896                    H5FileInner::Writer(writer) => {
897                        let idx = writer.create_null_dataset(&full_name, datatype)?;
898                        if let Some(ref gp) = self.group_path {
899                            if gp != "/" {
900                                writer.assign_dataset_to_group(gp, idx)?;
901                            }
902                        }
903                        idx
904                    }
905                    H5FileInner::Reader(_) => {
906                        return Err(Hdf5Error::InvalidState(
907                            "cannot create a dataset in read mode".into(),
908                        ));
909                    }
910                    H5FileInner::Closed => {
911                        return Err(Hdf5Error::InvalidState("file is closed".into()));
912                    }
913                }
914            };
915
916            return Ok(H5Dataset {
917                file_inner: clone_inner(&self.file_inner),
918                info: DatasetInfo::Writer {
919                    index,
920                    shape: Vec::new(),
921                    element_size,
922                    chunk_index: None,
923                    is_null: true,
924                },
925                _open: None,
926            });
927        }
928
929        let shape = self.shape.ok_or_else(|| {
930            Hdf5Error::InvalidState("shape must be set before calling create()".into())
931        })?;
932        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
933
934        if self.is_compact {
935            // The raw data is the layout message, so there is no chunk grid to
936            // filter and no room to grow into: `H5D__compact_construct` refuses
937            // a max dimension above the current one, and `H5Pset_layout` and
938            // `H5Pset_chunk` overwrite each other rather than combining.
939            if self.chunk_dims.is_some() || wants_filter {
940                return Err(Hdf5Error::InvalidState(
941                    "a compact dataset stores its data in the object header, so it \
942                     cannot be chunked or filtered"
943                        .into(),
944                ));
945            }
946            if self
947                .max_shape
948                .as_ref()
949                .is_some_and(|max| max.iter().zip(&shape).any(|(m, &d)| *m != Some(d)))
950            {
951                return Err(Hdf5Error::InvalidState(
952                    "a compact dataset cannot be extendible: its maximum shape must \
953                     equal its shape"
954                        .into(),
955                ));
956            }
957
958            let index = {
959                let inner = borrow_inner(&self.file_inner);
960                match &*inner {
961                    H5FileInner::Writer(writer) => {
962                        let idx = writer.create_compact_dataset(&full_name, datatype, &dims_u64)?;
963                        // Set before the fill value: NEVER must be in place
964                        // before that call decides whether to eager-tile it.
965                        if let Some(time) = self.fill_time {
966                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
967                        }
968                        if let Some(ref fv) = fill_value {
969                            writer.set_dataset_fill_value(idx, fv.clone())?;
970                        }
971                        idx
972                    }
973                    H5FileInner::Reader(_) => {
974                        return Err(Hdf5Error::InvalidState(
975                            "cannot create a dataset in read mode".into(),
976                        ));
977                    }
978                    H5FileInner::Closed => {
979                        return Err(Hdf5Error::InvalidState("file is closed".into()));
980                    }
981                }
982            };
983
984            return Ok(H5Dataset {
985                file_inner: clone_inner(&self.file_inner),
986                info: DatasetInfo::Writer {
987                    index,
988                    shape,
989                    element_size,
990                    chunk_index: None,
991                    is_null: false,
992                },
993                _open: None,
994            });
995        }
996
997        if !self.virtual_mappings.is_empty() {
998            let index = {
999                let inner = borrow_inner(&self.file_inner);
1000                match &*inner {
1001                    H5FileInner::Writer(writer) => {
1002                        let idx = writer.create_virtual_dataset(
1003                            &full_name,
1004                            datatype,
1005                            &dims_u64,
1006                            self.max_shape
1007                                .as_ref()
1008                                .map(|max| {
1009                                    max.iter()
1010                                        .map(|m| m.map_or(u64::MAX, |v| v as u64))
1011                                        .collect::<Vec<u64>>()
1012                                })
1013                                .as_deref(),
1014                            &self.virtual_mappings,
1015                        )?;
1016                        // The fill value is what a read of an unmapped — or
1017                        // unresolvable — element returns, so it is the one
1018                        // dataset property a virtual dataset carries about its
1019                        // own elements. Nothing is tiled into storage: it has
1020                        // none.
1021                        // Set before the fill value: NEVER must be in place
1022                        // before that call decides whether to eager-tile it.
1023                        if let Some(time) = self.fill_time {
1024                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1025                        }
1026                        if let Some(ref fv) = fill_value {
1027                            writer.set_dataset_fill_value(idx, fv.clone())?;
1028                        }
1029                        idx
1030                    }
1031                    H5FileInner::Reader(_) => {
1032                        return Err(Hdf5Error::InvalidState(
1033                            "cannot create a dataset in read mode".into(),
1034                        ));
1035                    }
1036                    H5FileInner::Closed => {
1037                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1038                    }
1039                }
1040            };
1041
1042            return Ok(H5Dataset {
1043                file_inner: clone_inner(&self.file_inner),
1044                info: DatasetInfo::Writer {
1045                    index,
1046                    shape,
1047                    element_size,
1048                    chunk_index: None,
1049                    is_null: false,
1050                },
1051                _open: None,
1052            });
1053        }
1054
1055        // A filter pipeline requires chunked storage. When a filter is
1056        // requested without explicit chunk dimensions, store the whole
1057        // dataset as a single chunk instead of silently dropping the filter
1058        // on the contiguous path. (This is one whole-dataset chunk, not
1059        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
1060        // for large datasets.)
1061        let auto_chunk: Option<Vec<usize>> =
1062            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
1063                Some(shape.iter().map(|&d| d.max(1)).collect())
1064            } else {
1065                None
1066            };
1067
1068        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
1069            // Chunked dataset
1070            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
1071            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
1072                max.iter()
1073                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
1074                    .collect()
1075            } else {
1076                // Default: max = current
1077                dims_u64.clone()
1078            };
1079
1080            // The file's format settles the question before the shape gets a
1081            // say. `H5D__chunk_set_info` reaches the index-selection block
1082            // only once the data layout message is at version 4 (H5Dchunk.c:936)
1083            // — which the file's library-version bound decides, not the
1084            // dataspace — and below it the version-3 message carries a
1085            // version-1 B-tree and nothing else. The writer owns that reading
1086            // of `H5O_layout_ver_bounds`; the chunk's byte count is the one
1087            // input from here, a chunk over 4 GiB being the one thing that
1088            // forces the newer message whatever the bound says.
1089            let chunk_bytes = chunk_u64.iter().product::<u64>() * element_size as u64;
1090            let v110_indexing = match &*borrow_inner(&self.file_inner) {
1091                H5FileInner::Writer(writer) => writer.uses_v110_chunk_indexing(chunk_bytes),
1092                // Neither can create a dataset at all; the creator below
1093                // reports which of the two it is.
1094                _ => true,
1095            };
1096
1097            // Inside the block libhdf5 selects the chunk index from the
1098            // dataspace and the creation properties, in this order
1099            // (`H5D__chunk_set_info`, H5Dchunk.c:955): a v2 B-tree for two or
1100            // more unlimited dimensions, an extensible array for exactly one;
1101            // for a fixed shape, the single-chunk index takes priority —
1102            // unconditional of filter or allocation time — whenever the shape
1103            // is exactly one whole chunk, ahead of the implicit index (no
1104            // filter, and early allocation, which is what puts every chunk at
1105            // a computable address) and the fixed array (everything else).
1106            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
1107            let one_chunk = chunk_u64 == dims_u64 && max_u64 == dims_u64;
1108            let kind = if !v110_indexing {
1109                ChunkIndexKind::BtreeV1
1110            } else if n_unlimited >= 2 {
1111                ChunkIndexKind::BtreeV2
1112            } else if n_unlimited == 1 {
1113                ChunkIndexKind::ExtensibleArray
1114            } else if one_chunk {
1115                ChunkIndexKind::SingleChunk
1116            } else if self.early_allocation && !wants_filter {
1117                ChunkIndexKind::Implicit
1118            } else {
1119                ChunkIndexKind::FixedArray
1120            };
1121
1122            let index = {
1123                let inner = borrow_inner(&self.file_inner);
1124                match &*inner {
1125                    H5FileInner::Writer(writer) => {
1126                        // The requested filter pipeline, if any. Every index
1127                        // builds it from the same options, so one owner
1128                        // resolves it: a second construction site is what let
1129                        // a request naming no compressor — shuffle on its own
1130                        // — fall through to unfiltered storage.
1131                        let explicit_pipeline = || {
1132                            use crate::format::messages::filter::FilterPipeline;
1133                            if let Some(p) = self.custom_pipeline.clone() {
1134                                return p;
1135                            }
1136                            // Shuffle records the width of the element it
1137                            // permutes, which is the stored one — a `datatype`
1138                            // override moves that away from `T`.
1139                            let es = element_size as u32;
1140                            match (self.shuffle, self.deflate_level) {
1141                                (true, Some(level)) => FilterPipeline::shuffle_deflate(es, level),
1142                                (true, None) => FilterPipeline::shuffle(es),
1143                                // deflate_level (checked by wants_filter).
1144                                (false, level) => FilterPipeline::deflate(level.unwrap()),
1145                            }
1146                        };
1147                        let idx = if kind == ChunkIndexKind::BtreeV1 {
1148                            // The classic index, which takes the pipeline the
1149                            // same way the others do — and is refused with it
1150                            // in a classic file, whose filter pipeline
1151                            // message is a version this crate does not write.
1152                            let pipeline = wants_filter.then(explicit_pipeline);
1153                            writer.create_btree_v1_dataset(
1154                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1155                            )?
1156                        } else if kind == ChunkIndexKind::BtreeV2 {
1157                            // Two or more unlimited dimensions: a v2 B-tree,
1158                            // whose records carry the stored size and filter
1159                            // mask when the dataset is compressed (libhdf5
1160                            // H5D_BT2_FILT).
1161                            if wants_filter {
1162                                writer.create_btree_v2_dataset_with_pipeline(
1163                                    &full_name,
1164                                    datatype,
1165                                    &dims_u64,
1166                                    &max_u64,
1167                                    &chunk_u64,
1168                                    explicit_pipeline(),
1169                                )?
1170                            } else {
1171                                writer.create_btree_v2_dataset(
1172                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1173                                )?
1174                            }
1175                        } else if kind == ChunkIndexKind::Implicit {
1176                            // No index structure at all: every chunk of the
1177                            // grid is allocated at create in one run, so
1178                            // there is no pipeline arm — a filter is what
1179                            // makes chunks different sizes, and this index
1180                            // has no room to say so.
1181                            writer.create_implicit_dataset(
1182                                &full_name, datatype, &dims_u64, &chunk_u64,
1183                            )?
1184                        } else if kind == ChunkIndexKind::SingleChunk {
1185                            // A fixed shape covered by exactly one chunk:
1186                            // libhdf5 picks this index ahead of Implicit and
1187                            // Fixed Array regardless of filter or allocation
1188                            // time. Filtered or not, it takes the same
1189                            // explicit pipeline the other indexes do; a
1190                            // filtered chunk's stored size isn't known ahead
1191                            // of its first write, so early allocation only
1192                            // ever applies to the unfiltered form.
1193                            if wants_filter {
1194                                writer.create_single_chunk_dataset_with_pipeline(
1195                                    &full_name,
1196                                    datatype,
1197                                    &dims_u64,
1198                                    &chunk_u64,
1199                                    explicit_pipeline(),
1200                                )?
1201                            } else {
1202                                writer.create_single_chunk_dataset(
1203                                    &full_name,
1204                                    datatype,
1205                                    &dims_u64,
1206                                    &chunk_u64,
1207                                    self.early_allocation,
1208                                )?
1209                            }
1210                        } else if kind == ChunkIndexKind::FixedArray {
1211                            // A chunked dataset with no unlimited dimension
1212                            // must use the fixed-array index — libhdf5
1213                            // rejects an extensible-array index here. A
1214                            // compressed fixed-shape dataset uses a *filtered*
1215                            // fixed array (FA client id 1). The maximum shape
1216                            // sizes the array, so a finite max above the
1217                            // current shape stays growable.
1218                            let pipeline = wants_filter.then(explicit_pipeline);
1219                            writer.create_fixed_array_dataset_with_max(
1220                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1221                            )?
1222                        } else if wants_filter {
1223                            // The extensible-array index takes the pipeline
1224                            // the same way, so it goes through the one owner
1225                            // too.
1226                            writer.create_chunked_dataset_with_pipeline(
1227                                &full_name,
1228                                datatype,
1229                                &dims_u64,
1230                                &max_u64,
1231                                &chunk_u64,
1232                                explicit_pipeline(),
1233                            )?
1234                        } else {
1235                            writer.create_chunked_dataset(
1236                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1237                            )?
1238                        };
1239                        // Set before the fill value: NEVER must be in place
1240                        // before that call decides whether to eager-tile it.
1241                        if let Some(time) = self.fill_time {
1242                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1243                        }
1244                        if let Some(ref fv) = fill_value {
1245                            writer.set_dataset_fill_value(idx, fv.clone())?;
1246                        }
1247                        idx
1248                    }
1249                    H5FileInner::Reader(_) => {
1250                        return Err(Hdf5Error::InvalidState(
1251                            "cannot create a dataset in read mode".into(),
1252                        ));
1253                    }
1254                    H5FileInner::Closed => {
1255                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1256                    }
1257                }
1258            };
1259
1260            Ok(H5Dataset {
1261                file_inner: clone_inner(&self.file_inner),
1262                info: DatasetInfo::Writer {
1263                    index,
1264                    shape,
1265                    element_size,
1266                    chunk_index: Some(kind),
1267                    is_null: false,
1268                },
1269                _open: None,
1270            })
1271        } else {
1272            // Contiguous dataset (original path)
1273            let efile_access = match self.efile_prefix.as_deref() {
1274                Some(p) => DatasetAccess::new().efile_prefix(p),
1275                None => DatasetAccess::new(),
1276            };
1277            let (index, open) = {
1278                let inner = borrow_inner(&self.file_inner);
1279                match &*inner {
1280                    H5FileInner::Writer(writer) => {
1281                        let idx = match self.external.as_deref() {
1282                            Some(files) => {
1283                                let slots: Vec<(&str, u64, u64)> = files
1284                                    .iter()
1285                                    .map(|(name, offset, size)| (name.as_str(), *offset, *size))
1286                                    .collect();
1287                                writer.create_external_dataset(
1288                                    &full_name,
1289                                    datatype,
1290                                    &dims_u64,
1291                                    self.max_shape
1292                                        .as_ref()
1293                                        .map(|max| {
1294                                            max.iter()
1295                                                .map(|m| m.map_or(u64::MAX, |v| v as u64))
1296                                                .collect::<Vec<u64>>()
1297                                        })
1298                                        .as_deref(),
1299                                    &slots,
1300                                )?
1301                            }
1302                            None => writer.create_dataset(&full_name, datatype, &dims_u64)?,
1303                        };
1304                        // Before anything that can write raw bytes: the
1305                        // prefix an external dataset's slot names are joined
1306                        // against is settled by the create, as
1307                        // `H5D__build_file_prefix` settles it for
1308                        // `H5D__create` (H5Dint.c:1318).
1309                        let open = writer.bind_efile_prefix(idx, &efile_access)?;
1310                        // Set before the fill value: NEVER must be in place
1311                        // before that call decides whether to eager-tile it.
1312                        if let Some(time) = self.fill_time {
1313                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1314                        }
1315                        if let Some(ref fv) = fill_value {
1316                            writer.set_dataset_fill_value(idx, fv.clone())?;
1317                        }
1318                        (idx, open)
1319                    }
1320                    H5FileInner::Reader(_) => {
1321                        return Err(Hdf5Error::InvalidState(
1322                            "cannot create a dataset in read mode".into(),
1323                        ));
1324                    }
1325                    H5FileInner::Closed => {
1326                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1327                    }
1328                }
1329            };
1330
1331            Ok(H5Dataset {
1332                file_inner: clone_inner(&self.file_inner),
1333                info: DatasetInfo::Writer {
1334                    index,
1335                    shape,
1336                    element_size,
1337                    chunk_index: None,
1338                    is_null: false,
1339                },
1340                _open: open,
1341            })
1342        }
1343    }
1344}
1345
1346// ---------------------------------------------------------------------------
1347// DatasetInfo
1348// ---------------------------------------------------------------------------
1349
1350/// Internal metadata about a dataset handle.
1351enum DatasetInfo {
1352    /// A dataset created via `new_dataset().create()` in write mode.
1353    Writer {
1354        /// Index into the writer's dataset list.
1355        index: usize,
1356        /// Shape (current dimensions).
1357        shape: Vec<usize>,
1358        /// Size of one element in bytes.
1359        element_size: usize,
1360        /// Which chunk index this dataset uses, `None` when its storage is
1361        /// not chunked. One field rather than a flag per index: a dataset
1362        /// has exactly one chunk index, and the flags could spell
1363        /// combinations ("not chunked, but indexed by a v2 B-tree") that no
1364        /// dataset has — which is what a write path reading only some of
1365        /// them turns into a write to the wrong index.
1366        chunk_index: Option<ChunkIndexKind>,
1367        /// Whether this is a NULL dataspace (no elements at all — distinct
1368        /// from a scalar, which holds exactly one). Always `false` when
1369        /// `chunk_index` is `Some`: a NULL dataspace can never be chunked.
1370        is_null: bool,
1371    },
1372    /// A dataset opened by name in read mode.
1373    Reader {
1374        /// The link name of the dataset.
1375        name: String,
1376        /// Shape (current dimensions).
1377        shape: Vec<usize>,
1378        /// Size of one element in bytes.
1379        element_size: usize,
1380    },
1381}
1382
1383// ---------------------------------------------------------------------------
1384// H5Dataset
1385// ---------------------------------------------------------------------------
1386
1387/// A handle to an HDF5 dataset, supporting typed read and write operations.
1388///
1389/// The dataset holds a shared reference to the file's I/O backend, so it
1390/// remains valid even if the originating [`H5File`](crate::file::H5File) is
1391/// moved or dropped (they share ownership via `Rc`).
1392pub struct H5Dataset {
1393    file_inner: SharedInner,
1394    info: DatasetInfo,
1395    /// Keeps this dataset's *open* alive for as long as the handle is, so
1396    /// the reader or the writer can tell whether a later open of the same
1397    /// name joins this one or starts fresh — libhdf5's `H5FO_opened`
1398    /// shared-info count (H5Dint.c:1496-1500). `None` for a dataset with no
1399    /// per-open answer to hold: in write mode, one whose raw data is in this
1400    /// file rather than in the files an external file list names.
1401    ///
1402    /// Held, never read: its whole job is to keep the reader's `Weak` on it
1403    /// upgradable until this handle goes away.
1404    _open: Option<crate::io::reader::DatasetOpenToken>,
1405}
1406
1407impl Drop for H5Dataset {
1408    /// Closing a virtual dataset's last handle closes the source files that
1409    /// open was holding, which is what `H5D__virtual_reset_layout` does at
1410    /// the last `H5Dclose` (H5Dvirtual.c:709-710, closing each
1411    /// `source_dset->dset` at :955 and with it the file that dataset kept
1412    /// open). Nothing else in this crate can end a virtual open, so this is
1413    /// where the reader is told.
1414    ///
1415    /// The token is dropped *before* the reader is asked, so the reader's
1416    /// `Weak` already reads dead for the handle going away here. A write-mode
1417    /// handle's token belongs to the writer's own external file prefix, which
1418    /// has no source files to close, so it takes no lock either.
1419    fn drop(&mut self) {
1420        let Some(open) = self._open.take() else {
1421            return;
1422        };
1423        drop(open);
1424        if matches!(self.info, DatasetInfo::Writer { .. }) {
1425            return;
1426        }
1427        let Some(mut inner) = crate::file::try_borrow_inner_mut(&self.file_inner) else {
1428            return;
1429        };
1430        if let crate::file::H5FileInner::Reader(reader) = &mut *inner {
1431            reader.release_closed_virtual_sources();
1432        }
1433    }
1434}
1435
1436/// One chunk's bytes on the way to the file, and who filtered them.
1437///
1438/// This is what separates a normal chunk write from a direct one; everything
1439/// else about placing a chunk is identical, so the two share a single dispatch.
1440#[derive(Clone, Copy)]
1441enum ChunkBytes<'a> {
1442    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
1443    /// are stored.
1444    Unfiltered(&'a [u8]),
1445    /// Bytes already in their stored form, with `filter_mask` naming the
1446    /// filters that were skipped.
1447    Prefiltered { data: &'a [u8], filter_mask: u32 },
1448}
1449
1450/// The byte order this build reads and writes natively.
1451pub(crate) const HOST_BYTE_ORDER: ByteOrder = if cfg!(target_endian = "big") {
1452    ByteOrder::BigEndian
1453} else {
1454    ByteOrder::LittleEndian
1455};
1456
1457/// The byte order this build does not read or write natively.
1458pub(crate) const FOREIGN_BYTE_ORDER: ByteOrder = match HOST_BYTE_ORDER {
1459    ByteOrder::LittleEndian => ByteOrder::BigEndian,
1460    ByteOrder::BigEndian => ByteOrder::LittleEndian,
1461};
1462
1463/// What a typed access has to do with an element image of a given datatype.
1464#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1465enum ByteOrderAction {
1466    /// Stored order is the host's: the image is already the typed value.
1467    Keep,
1468    /// The whole element is one scalar in the foreign order: reverse it.
1469    SwapElements,
1470    /// A composite storing something in the foreign order.
1471    Refuse,
1472}
1473
1474/// Classify a datatype for a typed access of element width `width`.
1475///
1476/// The single owner of the rule; both directions ask it, so a type a read
1477/// converts is exactly a type a write converts.
1478///
1479/// A composite element cannot be swapped as a unit — its members have their
1480/// own orders and offsets — so one that touches the foreign order is refused
1481/// rather than silently passed through in the wrong order.
1482fn byte_order_action(datatype: &DatatypeMessage, width: usize) -> ByteOrderAction {
1483    match datatype.scalar_byte_order() {
1484        Some(order) if order == FOREIGN_BYTE_ORDER && width > 1 => ByteOrderAction::SwapElements,
1485        Some(_) => ByteOrderAction::Keep,
1486        None if datatype.contains_byte_order(FOREIGN_BYTE_ORDER) => ByteOrderAction::Refuse,
1487        None => ByteOrderAction::Keep,
1488    }
1489}
1490
1491/// Why the stored image of an element is not already the host image of a
1492/// value of width `width` — `None` when it is, and a copying read would only
1493/// be memcpy-ing bytes it does not touch.
1494///
1495/// The two ways a stored element can need work before it is a value are the
1496/// two conversions a copying read performs in place: a byte-order swap
1497/// ([`to_host_byte_order`]) and the n-bit/scale-offset unpacking
1498/// (`Hdf5Reader::apply_post_filter_conversion`). Asking one question of both
1499/// is what lets a zero-copy view refuse exactly the datatypes a copying read
1500/// would have had to rewrite.
1501#[cfg(feature = "mmap")]
1502pub(crate) fn stored_image_mismatch(
1503    datatype: &DatatypeMessage,
1504    width: usize,
1505) -> Option<&'static str> {
1506    match byte_order_action(datatype, width) {
1507        ByteOrderAction::SwapElements => return Some("they are stored in the foreign byte order"),
1508        ByteOrderAction::Refuse => {
1509            return Some("it is a composite storing members in the foreign byte order")
1510        }
1511        ByteOrderAction::Keep => {}
1512    }
1513    if crate::format::nbit_scaleoffset::datatype_needs_bit_conversion(datatype) {
1514        return Some("the significant bits do not fill the stored element");
1515    }
1516    None
1517}
1518
1519/// Put a raw element image into host byte order, in place, for a typed read.
1520///
1521/// Every path that reinterprets the on-disk image as `T` — `read_raw`,
1522/// `read_slice`, `read_raw_into`, `read_slice_into` and their SWMR
1523/// counterparts — passes through here. Reinterpretation only yields the
1524/// stored value when the stored order is the host's.
1525///
1526/// A refused datatype is one no reinterpretation can decode;
1527/// [`H5Dataset::read_raw_bytes`] hands over the image for the caller to
1528/// decode member by member.
1529///
1530/// `width` is the element size, already checked equal to `T::element_size()`.
1531pub(crate) fn to_host_byte_order(
1532    bytes: &mut [u8],
1533    datatype: &DatatypeMessage,
1534    width: usize,
1535) -> Result<()> {
1536    match byte_order_action(datatype, width) {
1537        ByteOrderAction::Keep => {}
1538        ByteOrderAction::SwapElements => {
1539            for elem in bytes.chunks_exact_mut(width) {
1540                elem.reverse();
1541            }
1542        }
1543        ByteOrderAction::Refuse => {
1544            return Err(Hdf5Error::TypeMismatch(format!(
1545                "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a \
1546                 typed read cannot reinterpret element by element; read_raw_bytes() returns \
1547                 the image to decode member by member"
1548            )))
1549        }
1550    }
1551    Ok(())
1552}
1553
1554/// The stored byte image of each variable-length sequence in a batch.
1555///
1556/// The vlen writers take `&[&[T]]` and store one global-heap object per
1557/// sequence, so each sequence needs the same host-image-to-stored-image step
1558/// [`to_stored_byte_order`] performs for a fixed-shape write — a `T` is
1559/// written from its host bytes, and `T::hdf5_type()` declares little-endian.
1560/// Borrows on a little-endian host, which is every machine that does not have
1561/// to swap.
1562pub(crate) fn vlen_sequence_images<'a, T: H5Type>(
1563    items: &'a [&'a [T]],
1564) -> Result<Vec<std::borrow::Cow<'a, [u8]>>> {
1565    let base = T::hdf5_type();
1566    items
1567        .iter()
1568        .map(|item| {
1569            // Safety: the same contract `write_raw` relies on — `T: Copy +
1570            // 'static` is a numeric primitive whose byte image is its value —
1571            // and the extent comes from the slice itself, so it cannot name
1572            // memory past it. The result borrows `items` and outlives nothing.
1573            let host = unsafe {
1574                std::slice::from_raw_parts(item.as_ptr() as *const u8, std::mem::size_of_val(*item))
1575            };
1576            to_stored_byte_order(host, &base, T::element_size())
1577        })
1578        .collect()
1579}
1580
1581/// Put a typed value's host-order image into the order the datatype declares.
1582///
1583/// The write-side counterpart of [`to_host_byte_order`], and the one place
1584/// every path that hands a `&[T]` to the file — `write_raw`, `write_slice`,
1585/// `append`, and the builder's fill value — turns those bytes into stored
1586/// bytes. A `T` is written from its host image, so a dataset declaring the
1587/// foreign order would otherwise hold host bytes under that declaration: a
1588/// file that is wrong by its own header.
1589///
1590/// Borrows when the declared order is the host's, which is every write that
1591/// does not set a [`datatype`](DatasetBuilder::datatype) override.
1592///
1593/// `width` is the element size, already checked equal to `T::element_size()`.
1594pub(crate) fn to_stored_byte_order<'a>(
1595    bytes: &'a [u8],
1596    datatype: &DatatypeMessage,
1597    width: usize,
1598) -> Result<std::borrow::Cow<'a, [u8]>> {
1599    match byte_order_action(datatype, width) {
1600        ByteOrderAction::Keep => Ok(std::borrow::Cow::Borrowed(bytes)),
1601        ByteOrderAction::SwapElements => {
1602            let mut owned = bytes.to_vec();
1603            for elem in owned.chunks_exact_mut(width) {
1604                elem.reverse();
1605            }
1606            Ok(std::borrow::Cow::Owned(owned))
1607        }
1608        ByteOrderAction::Refuse => Err(Hdf5Error::TypeMismatch(format!(
1609            "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a typed \
1610             write cannot lay out element by element; write_raw_bytes() takes the image the \
1611             caller encodes member by member"
1612        ))),
1613    }
1614}
1615
1616/// Strip a fixed-string element's padding, leaving the bytes that carry the
1617/// value.
1618///
1619/// The rule itself lives with the datatype message
1620/// ([`fixed_string_content`]); this adds the element index a reserved padding
1621/// rule needs to be reported against.
1622fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
1623    crate::format::messages::datatype::fixed_string_content(elem, padding).ok_or_else(|| {
1624        Hdf5Error::InvalidState(format!(
1625            "string {index} uses padding rule {padding}, which the format reserves"
1626        ))
1627    })
1628}
1629
1630/// Decode one string element's bytes under the datatype's character set.
1631///
1632/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
1633/// `index` names the element in the error otherwise.
1634fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
1635    if lossy {
1636        return Ok(String::from_utf8_lossy(bytes).into_owned());
1637    }
1638    match charset {
1639        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
1640        0 => match bytes.iter().position(|&b| b >= 0x80) {
1641            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
1642            Some(at) => Err(Hdf5Error::InvalidState(format!(
1643                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
1644                bytes[at]
1645            ))),
1646        },
1647        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
1648            Hdf5Error::InvalidState(format!(
1649                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
1650            ))
1651        }),
1652        other => Err(Hdf5Error::InvalidState(format!(
1653            "string {index} uses character set {other}, which the format reserves"
1654        ))),
1655    }
1656}
1657
1658/// A dataset's storage layout class (read mode only) — `H5Pget_layout`'s
1659/// four values.
1660///
1661/// Distinct from [`ChunkIndex`], which names the structure a `Chunked`
1662/// layout's index uses; this only says which of the four storage classes
1663/// the dataset was created with.
1664#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1665pub enum StorageLayout {
1666    /// Raw data stored inline in the object header.
1667    Compact,
1668    /// Raw data in a single contiguous block — or, when the dataset also
1669    /// carries an external file list, in one or more blocks of an outside
1670    /// file instead ([`H5Dataset::external_files`]); the layout itself
1671    /// still reports `Contiguous` either way.
1672    Contiguous,
1673    /// Raw data split into fixed-size chunks, each independently
1674    /// allocated. [`H5Dataset::chunk_dims`] gives the chunk shape,
1675    /// [`H5Dataset::chunk_index`] the index structure.
1676    Chunked,
1677    /// No raw data of its own: every element comes from another dataset,
1678    /// possibly in another file ([`H5Dataset::virtual_mappings`]).
1679    Virtual,
1680}
1681
1682/// The chunk index structure a chunked dataset uses on disk (read mode
1683/// only) — which of libhdf5's chunk-lookup structures the layout message
1684/// names.
1685///
1686/// `BtreeV1` belongs to the version-3 chunked layout message (the only
1687/// index a file whose superblock predates version 2 can carry); the other
1688/// five are what a version-4 message's index-type byte selects, per
1689/// `H5D__layout_set_latest_indexing` (H5Dlayout.c).
1690#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1691pub enum ChunkIndex {
1692    /// Version-1 B-tree — the classic chunk index, and the only one a
1693    /// version-3 chunked layout message can carry.
1694    BtreeV1,
1695    /// Version-2 B-tree — two or more unlimited dimensions.
1696    BtreeV2,
1697    /// A single index entry for a dataset whose one chunk covers the whole
1698    /// dataspace (`dims == max_dims == chunk_dims`).
1699    SingleChunk,
1700    /// No index structure at all: chunk addresses are computed
1701    /// arithmetically over a contiguous run (no filter, early allocation).
1702    Implicit,
1703    /// Fixed-size array — a fixed shape needing per-chunk bookkeeping.
1704    FixedArray,
1705    /// Extensible array — exactly one unlimited dimension.
1706    ExtensibleArray,
1707}
1708
1709/// A dataset's fill-value state (read mode only) — `H5Pfill_value_defined`'s
1710/// tri-state (`H5D_fill_value_t`).
1711#[derive(Debug, Clone, PartialEq, Eq)]
1712pub enum FillValue {
1713    /// No fill value has ever been set: unwritten elements read back
1714    /// zero-filled, and no fill-value message named an explicit value.
1715    Default,
1716    /// The fill value was explicitly disabled: unallocated storage is never
1717    /// fill-initialized.
1718    Undefined,
1719    /// An explicit fill value, one element wide.
1720    UserDefined(Vec<u8>),
1721}
1722
1723/// When a dataset's fill value is written into allocated storage —
1724/// `H5Pset_fill_time`/`H5Pget_fill_time`'s `H5D_fill_time_t`.
1725///
1726/// Distinct from [`FillValue`], which says *what* the fill value is; this
1727/// says *when* it is written. The two agree everywhere except a dataset with
1728/// no fill value of its own: there, `Alloc` writes the default fill (zeros)
1729/// at allocation and `IfSet` writes nothing into space that already reads as
1730/// zeros — indistinguishable on disk in the value itself, only in this byte.
1731#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1732pub enum FillTime {
1733    /// Fill at allocation regardless of whether a fill value was ever set.
1734    Alloc,
1735    /// Never write the fill value into allocated storage.
1736    Never,
1737    /// Fill at allocation only when a fill value was set — the default
1738    /// every dataset gets unless [`fill_time`](DatasetBuilder::fill_time)
1739    /// says otherwise.
1740    IfSet,
1741}
1742
1743impl FillTime {
1744    /// The on-disk `H5D_fill_time_t` byte this variant is — what the writer
1745    /// stores and the fill-value message's write-time field carries.
1746    fn wire_byte(self) -> u8 {
1747        match self {
1748            Self::Alloc => 0,
1749            Self::Never => 1,
1750            Self::IfSet => 2,
1751        }
1752    }
1753}
1754
1755/// When a dataset's raw-data storage is allocated —
1756/// `H5Pset_alloc_time`/`H5Pget_alloc_time`'s `H5D_alloc_time_t`, read back
1757/// from the same fill-value message [`FillTime`] is.
1758///
1759/// `H5P__set_layout` (H5Pdcpl.c) picks this from the dataset's storage
1760/// class (`H5D_ALLOC_TIME_DEFAULT` per layout — compact is `Early`,
1761/// chunked and virtual are `Incr`, contiguous is `Late`). The one override
1762/// this crate's builder offers is [`DatasetBuilder::early_allocation`],
1763/// which a chunked dataset reads back as `Early` where the writer took it
1764/// up: an implicit index, or a single unfiltered chunk.
1765/// [`H5Dataset::alloc_time`] reads back what the writer declared.
1766#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1767pub enum AllocTime {
1768    /// Space is allocated as soon as the dataset is created.
1769    Early,
1770    /// Space is allocated when data is first written.
1771    Late,
1772    /// Space is allocated incrementally, as chunks (or virtual source
1773    /// datasets) are written.
1774    Incr,
1775}
1776
1777/// Which mapped data an unlimited virtual dataset's extent covers —
1778/// libhdf5's `H5D_vds_view_t`, set with `H5Pset_virtual_view` and read back
1779/// with `H5Pget_virtual_view` (H5Pdapl.c:1067, :1102).
1780///
1781/// A *dataset access* property: it is never stored in the file, so it says
1782/// how *this* open reads a virtual dataset, not what its writer intended.
1783#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1784pub enum VirtualView {
1785    /// `H5D_VDS_LAST_AVAILABLE` — the extent reaches the end of the last
1786    /// mapped block that has a source, so a gap before it reads as the fill
1787    /// value. libhdf5's default (`H5D_ACS_VDS_VIEW_DEF`, H5Pdapl.c:62).
1788    #[default]
1789    LastAvailable,
1790    /// `H5D_VDS_FIRST_MISSING` — the extent stops where the first missing
1791    /// mapped block begins, so no unmapped block is inside it.
1792    ///
1793    /// Under this view libhdf5 ignores
1794    /// [`virtual_printf_gap`](DatasetAccess::virtual_printf_gap) entirely:
1795    /// `H5D__virtual_init` reads the gap property only for
1796    /// [`LastAvailable`](Self::LastAvailable) and forces it to 0 otherwise
1797    /// (H5Dvirtual.c:2182-2188).
1798    FirstMissing,
1799}
1800
1801/// The dataset *access* properties this crate models — libhdf5's
1802/// `H5P_DATASET_ACCESS` property list, as much of it as affects reading.
1803///
1804/// [`virtual_view`](Self::virtual_view) and
1805/// [`virtual_printf_gap`](Self::virtual_printf_gap) govern how a virtual
1806/// dataset's extent is resolved when it is opened
1807/// (`H5D__virtual_set_extent_unlim`, H5Dvirtual.c:1386);
1808/// [`virtual_prefix`](Self::virtual_prefix) and
1809/// [`efile_prefix`](Self::efile_prefix) say where the *other files* a
1810/// dataset's data lives in are looked for. None of them is stored in the
1811/// file: opening a dataset without naming them reads it exactly as
1812/// libhdf5's default dapl does.
1813///
1814/// Pass one to [`H5File::dataset_with`](crate::H5File::dataset_with).
1815///
1816/// ```no_run
1817/// use rust_hdf5::{DatasetAccess, H5File, VirtualView};
1818///
1819/// let file = H5File::open("vds.h5").unwrap();
1820/// let access = DatasetAccess::new()
1821///     .virtual_view(VirtualView::LastAvailable)
1822///     .virtual_printf_gap(2);
1823/// let ds = file.dataset_with("vds", access).unwrap();
1824/// ```
1825#[derive(Debug, Clone, PartialEq, Eq, Default)]
1826pub struct DatasetAccess {
1827    view: VirtualView,
1828    printf_gap: u64,
1829    virtual_prefix: Option<String>,
1830    efile_prefix: Option<String>,
1831}
1832
1833impl DatasetAccess {
1834    /// A property list holding libhdf5's defaults —
1835    /// [`VirtualView::LastAvailable`] and a printf gap of 0, the values
1836    /// `H5D_ACS_VDS_VIEW_DEF` and `H5D_ACS_VDS_PRINTF_GAP_DEF` register
1837    /// (H5Pdapl.c:62, :67).
1838    pub fn new() -> Self {
1839        Self::default()
1840    }
1841
1842    /// `H5Pset_virtual_view` (H5Pdapl.c:1067). The two legal values are the
1843    /// two [`VirtualView`] variants, so the "not a valid bounds option"
1844    /// argument check that call makes has nothing to reject here.
1845    pub fn virtual_view(mut self, view: VirtualView) -> Self {
1846        self.view = view;
1847        self
1848    }
1849
1850    /// `H5Pset_virtual_printf_gap` (H5Pdapl.c:1207): how many consecutive
1851    /// missing printf-named source datasets the extent resolution looks past
1852    /// before it stops. 0 — the default — stops at the first one missing.
1853    ///
1854    /// `u64::MAX` is libhdf5's `HSIZE_UNDEF`, which that call rejects as "not
1855    /// a valid printf gap size"; here the rejection surfaces from the open
1856    /// that uses the property, since a builder method has no way to report
1857    /// it.
1858    pub fn virtual_printf_gap(mut self, gap: u64) -> Self {
1859        self.printf_gap = gap;
1860        self
1861    }
1862
1863    /// `H5Pget_virtual_view` (H5Pdapl.c:1102).
1864    pub fn view(&self) -> VirtualView {
1865        self.view
1866    }
1867
1868    /// `H5Pset_virtual_prefix` (H5Pdapl.c:1478): a directory a virtual
1869    /// dataset's *source file names* are looked for under, before the
1870    /// virtual file's own directory and after `HDF5_VDS_PREFIX`.
1871    ///
1872    /// It is the third step of `H5F_prefix_open_file`'s search order
1873    /// (H5Fint.c:938-950), and it is reached only when `HDF5_VDS_PREFIX` is
1874    /// unset or empty: `H5D__build_file_prefix` reads the environment first
1875    /// and falls back to this property (H5Dint.c:1077-1082), so an
1876    /// environment prefix shadows this one outright rather than being tried
1877    /// alongside it.
1878    ///
1879    /// A leading `${ORIGIN}` stands for the directory holding the virtual
1880    /// dataset's own file (H5Dint.c:1105-1113), and `"."` or `""` means "no
1881    /// prefix" (:1096-1100), both exactly as for the environment variable.
1882    ///
1883    /// Like the other two, this is a *dataset access* property that is never
1884    /// stored in the file, and the first open of a virtual dataset fixes it
1885    /// for every open that overlaps it.
1886    pub fn virtual_prefix(mut self, prefix: impl Into<String>) -> Self {
1887        self.virtual_prefix = Some(prefix.into());
1888        self
1889    }
1890
1891    /// `H5Pset_efile_prefix` (H5Pdapl.c:1392): a directory the *raw data
1892    /// files* of a dataset stored through an external file list are looked
1893    /// for under.
1894    ///
1895    /// This one takes no search at all, unlike the other two prefixes:
1896    /// `H5D__efl_read` joins the prefix to the stored name with
1897    /// `H5_combine_path` and opens exactly that one path (H5Defl.c:315-317).
1898    /// With no prefix in force the stored name is used as written, so a
1899    /// relative one resolves against the *process's current directory* and
1900    /// not against the directory holding the HDF5 file — measured under
1901    /// libhdf5 1.14.6 and 2.0.0: a raw data file next to the HDF5 file is
1902    /// not found, while the same name under the current directory is.
1903    ///
1904    /// It shares [`virtual_prefix`](Self::virtual_prefix)'s expansion rules,
1905    /// because both are built by `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX`
1906    /// shadows this property outright rather than merely preceding it
1907    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
1908    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
1909    /// (:1098-1102).
1910    ///
1911    /// # A second open must name the same one
1912    ///
1913    /// Where a mismatched [`virtual_prefix`](Self::virtual_prefix) is
1914    /// silently ignored by the second open, a mismatched external file prefix
1915    /// is an *error*: `H5D_open` compares the expanded prefix against the one
1916    /// the already-open dataset resolved under and refuses the open when they
1917    /// differ (H5Dint.c:1533-1545). Expanded, so two opens that differ only
1918    /// in a property the environment shadows still agree. Closing every
1919    /// handle releases the answer, and the next open sets its own.
1920    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
1921        self.efile_prefix = Some(prefix.into());
1922        self
1923    }
1924
1925    /// `H5Pget_virtual_printf_gap` (H5Pdapl.c:1243) — the value set, not the
1926    /// one the extent resolution ends up using; see
1927    /// [`VirtualView::FirstMissing`].
1928    pub fn printf_gap(&self) -> u64 {
1929        self.printf_gap
1930    }
1931
1932    /// `H5Pget_virtual_prefix` (H5Pdapl.c:1510) — the property as set, before
1933    /// `HDF5_VDS_PREFIX` gets to shadow it and before `${ORIGIN}` is
1934    /// expanded. `None` is `H5D_ACS_VDS_PREFIX_DEF`, a null prefix
1935    /// (H5Pdapl.c:72).
1936    pub fn virtual_prefix_value(&self) -> Option<&str> {
1937        self.virtual_prefix.as_deref()
1938    }
1939
1940    /// `H5Pget_efile_prefix` (H5Pdapl.c:1422) — the property as set, before
1941    /// `HDF5_EXTFILE_PREFIX` gets to shadow it and before `${ORIGIN}` is
1942    /// expanded. `None` is `H5D_ACS_EFILE_PREFIX_DEF`, a null prefix
1943    /// (H5Pdapl.c:90).
1944    pub fn efile_prefix_value(&self) -> Option<&str> {
1945        self.efile_prefix.as_deref()
1946    }
1947
1948    /// The printf gap `H5D__virtual_set_extent_unlim` actually scans with:
1949    /// the property under [`VirtualView::LastAvailable`], and 0 under
1950    /// [`VirtualView::FirstMissing`], because `H5D__virtual_init` only reads
1951    /// the property in the first case (H5Dvirtual.c:2182-2188).
1952    ///
1953    /// The single owner of that rule — the resolution never reads
1954    /// [`printf_gap`](Self::printf_gap) directly.
1955    pub(crate) fn effective_printf_gap(&self) -> u64 {
1956        match self.view {
1957            VirtualView::LastAvailable => self.printf_gap,
1958            VirtualView::FirstMissing => 0,
1959        }
1960    }
1961
1962    /// Reject what `H5Pset_virtual_printf_gap` rejects, at the open that uses
1963    /// the property.
1964    pub(crate) fn validate(&self) -> Result<()> {
1965        if self.printf_gap == u64::MAX {
1966            return Err(Hdf5Error::InvalidState(
1967                "virtual_printf_gap(u64::MAX) is libhdf5's HSIZE_UNDEF, which \
1968                 H5Pset_virtual_printf_gap refuses as \"not a valid printf gap size\""
1969                    .into(),
1970            ));
1971        }
1972        Ok(())
1973    }
1974}
1975
1976/// Elements a selection of `counts` holds, refusing a product that overflows
1977/// `usize` rather than wrapping it into a small allocation.
1978fn element_count(counts: &[u64]) -> Result<usize> {
1979    counts
1980        .iter()
1981        .try_fold(1usize, |acc, &c| {
1982            usize::try_from(c).ok().and_then(|c| acc.checked_mul(c))
1983        })
1984        .ok_or_else(|| {
1985            Hdf5Error::InvalidState(format!("selection {counts:?} has more elements than usize"))
1986        })
1987}
1988
1989impl H5Dataset {
1990    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
1991    pub(crate) fn new_reader(
1992        file_inner: SharedInner,
1993        name: String,
1994        shape: Vec<usize>,
1995        element_size: usize,
1996        open: Option<crate::io::reader::DatasetOpenToken>,
1997    ) -> Self {
1998        Self {
1999            file_inner,
2000            info: DatasetInfo::Reader {
2001                name,
2002                shape,
2003                element_size,
2004            },
2005            _open: open,
2006        }
2007    }
2008
2009    /// Create a writer-mode dataset handle for an already-created dataset
2010    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
2011    ///
2012    /// Reconstructs the same handle `new_dataset().create()` returns, so the
2013    /// reopened dataset supports attribute writes and chunk appends.
2014    ///
2015    /// `is_null` is always `false` here: reopening an existing NULL-dataspace
2016    /// dataset for further writes is not a case this constructor's caller
2017    /// distinguishes (a NULL dataset has nothing to append or chunk-write in
2018    /// the first place).
2019    pub(crate) fn new_writer(
2020        file_inner: SharedInner,
2021        index: usize,
2022        parts: crate::io::writer::DatasetHandleParts,
2023    ) -> Self {
2024        Self {
2025            file_inner,
2026            info: DatasetInfo::Writer {
2027                index,
2028                shape: parts.shape,
2029                element_size: parts.element_size,
2030                chunk_index: parts.chunk_index,
2031                is_null: false,
2032            },
2033            _open: parts.open,
2034        }
2035    }
2036
2037    /// Return the dataset dimensions.
2038    pub fn shape(&self) -> Vec<usize> {
2039        match &self.info {
2040            DatasetInfo::Writer { shape, .. } => shape.clone(),
2041            DatasetInfo::Reader { shape, .. } => shape.clone(),
2042        }
2043    }
2044
2045    /// Return the number of dimensions (rank) of the dataset.
2046    pub fn ndims(&self) -> usize {
2047        match &self.info {
2048            DatasetInfo::Writer { shape, .. } => shape.len(),
2049            DatasetInfo::Reader { shape, .. } => shape.len(),
2050        }
2051    }
2052
2053    /// Return the total number of elements in the dataset.
2054    ///
2055    /// 0 for a NULL dataspace ([`is_null`](Self::is_null)) — unlike a scalar,
2056    /// whose `shape()` is the same empty `Vec` but which holds exactly one
2057    /// element, so `shape().iter().product()` cannot be used here.
2058    pub fn total_elements(&self) -> usize {
2059        if self.is_null() {
2060            return 0;
2061        }
2062        match &self.info {
2063            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
2064            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
2065        }
2066    }
2067
2068    /// Return the size of one element in bytes.
2069    pub fn element_size(&self) -> usize {
2070        match &self.info {
2071            DatasetInfo::Writer { element_size, .. } => *element_size,
2072            DatasetInfo::Reader { element_size, .. } => *element_size,
2073        }
2074    }
2075
2076    /// Return whether this dataset has the NULL dataspace: no elements at
2077    /// all, distinct from a scalar dataset (rank 0, exactly one element) —
2078    /// both report the same empty [`shape`](Self::shape). See
2079    /// [`DatasetBuilder::null`].
2080    pub fn is_null(&self) -> bool {
2081        match &self.info {
2082            DatasetInfo::Writer { is_null, .. } => *is_null,
2083            DatasetInfo::Reader { name, .. } => {
2084                let mut inner = borrow_inner_mut(&self.file_inner);
2085                match &mut *inner {
2086                    H5FileInner::Reader(reader) => reader
2087                        .dataset_info(name)
2088                        .map(|info| info.dataspace.is_null())
2089                        .unwrap_or(false),
2090                    _ => false,
2091                }
2092            }
2093        }
2094    }
2095
2096    /// Return the element datatype as parsed from the file (read mode only).
2097    ///
2098    /// Unlike [`element_size`](Self::element_size), which reports only the
2099    /// byte width, this exposes the full datatype: its class (integer vs
2100    /// floating-point vs string vs compound …), signedness, byte order and
2101    /// bit precision. Callers that must reconstruct the exact stored type —
2102    /// for example to map it to a NumPy / Arrow dtype — should use this
2103    /// instead of inferring a type from the byte width, which cannot
2104    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
2105    /// bytes).
2106    ///
2107    /// # Errors
2108    ///
2109    /// Returns an error if the file is in write mode, or if the dataset can
2110    /// no longer be found in the reader's metadata.
2111    ///
2112    /// ```no_run
2113    /// # use rust_hdf5::{H5File, DatatypeMessage};
2114    /// let file = H5File::open("data.h5").unwrap();
2115    /// let ds = file.dataset("image").unwrap();
2116    /// match ds.datatype().unwrap() {
2117    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
2118    ///         println!("integer: {} bytes, signed={}", size, signed);
2119    ///     }
2120    ///     DatatypeMessage::FloatingPoint { size, .. } => {
2121    ///         println!("float: {} bytes", size);
2122    ///     }
2123    ///     other => println!("other type: {other}"),
2124    /// }
2125    /// ```
2126    pub fn datatype(&self) -> Result<DatatypeMessage> {
2127        match &self.info {
2128            DatasetInfo::Reader { name, .. } => {
2129                let mut inner = borrow_inner_mut(&self.file_inner);
2130                match &mut *inner {
2131                    H5FileInner::Reader(reader) => reader
2132                        .dataset_info(name)
2133                        .map(|info| info.datatype.clone())
2134                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2135                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2136                }
2137            }
2138            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2139                "datatype() is only available in read mode".into(),
2140            )),
2141        }
2142    }
2143
2144    /// Return the chunk dimensions, if this is a chunked dataset.
2145    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
2146        match &self.info {
2147            DatasetInfo::Reader { name, .. } => {
2148                let mut inner = borrow_inner_mut(&self.file_inner);
2149                if let H5FileInner::Reader(reader) = &mut *inner {
2150                    if let Some(info) = reader.dataset_info(name) {
2151                        use crate::format::messages::data_layout::DataLayoutMessage;
2152                        let chunk_dims = match &info.layout {
2153                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
2154                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
2155                            _ => None,
2156                        };
2157                        if let Some(chunk_dims) = chunk_dims {
2158                            // Strip trailing element-size dimension
2159                            return Some(
2160                                chunk_dims[..chunk_dims.len() - 1]
2161                                    .iter()
2162                                    .map(|&d| d as usize)
2163                                    .collect(),
2164                            );
2165                        }
2166                    }
2167                }
2168                None
2169            }
2170            DatasetInfo::Writer { .. } => None,
2171        }
2172    }
2173
2174    /// Return whether this is a chunked dataset.
2175    pub fn is_chunked(&self) -> bool {
2176        match &self.info {
2177            DatasetInfo::Writer { chunk_index, .. } => chunk_index.is_some(),
2178            DatasetInfo::Reader { name, .. } => {
2179                let mut inner = borrow_inner_mut(&self.file_inner);
2180                match &mut *inner {
2181                    H5FileInner::Reader(reader) => {
2182                        if let Some(info) = reader.dataset_info(name) {
2183                            use crate::format::messages::data_layout::DataLayoutMessage;
2184                            matches!(
2185                                info.layout,
2186                                DataLayoutMessage::ChunkedV4 { .. }
2187                                    | DataLayoutMessage::ChunkedV3 { .. }
2188                            )
2189                        } else {
2190                            false
2191                        }
2192                    }
2193                    _ => false,
2194                }
2195            }
2196        }
2197    }
2198
2199    /// Return the dataset's storage layout class (read mode only).
2200    ///
2201    /// # Errors
2202    ///
2203    /// Returns an error if the file is in write mode, or if the dataset can
2204    /// no longer be found in the reader's metadata.
2205    pub fn storage_layout(&self) -> Result<StorageLayout> {
2206        match &self.info {
2207            DatasetInfo::Reader { name, .. } => {
2208                let mut inner = borrow_inner_mut(&self.file_inner);
2209                match &mut *inner {
2210                    H5FileInner::Reader(reader) => {
2211                        use crate::format::messages::data_layout::DataLayoutMessage;
2212                        reader
2213                            .dataset_info(name)
2214                            .map(|info| match &info.layout {
2215                                DataLayoutMessage::Compact { .. } => StorageLayout::Compact,
2216                                DataLayoutMessage::Contiguous { .. } => StorageLayout::Contiguous,
2217                                DataLayoutMessage::ChunkedV3 { .. }
2218                                | DataLayoutMessage::ChunkedV4 { .. } => StorageLayout::Chunked,
2219                                DataLayoutMessage::Virtual { .. } => StorageLayout::Virtual,
2220                            })
2221                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2222                    }
2223                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2224                }
2225            }
2226            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2227                "storage_layout() is only available in read mode".into(),
2228            )),
2229        }
2230    }
2231
2232    /// Return the chunk index structure this dataset's layout uses, or
2233    /// `None` for a dataset that is not chunked (read mode only).
2234    ///
2235    /// # Errors
2236    ///
2237    /// Returns an error if the file is in write mode, or if the dataset can
2238    /// no longer be found in the reader's metadata.
2239    pub fn chunk_index(&self) -> Result<Option<ChunkIndex>> {
2240        match &self.info {
2241            DatasetInfo::Reader { name, .. } => {
2242                let mut inner = borrow_inner_mut(&self.file_inner);
2243                match &mut *inner {
2244                    H5FileInner::Reader(reader) => {
2245                        use crate::format::messages::data_layout::{
2246                            ChunkIndexType, DataLayoutMessage,
2247                        };
2248                        reader
2249                            .dataset_info(name)
2250                            .map(|info| match &info.layout {
2251                                DataLayoutMessage::ChunkedV3 { .. } => Some(ChunkIndex::BtreeV1),
2252                                DataLayoutMessage::ChunkedV4 { index_type, .. } => {
2253                                    Some(match index_type {
2254                                        ChunkIndexType::SingleChunk => ChunkIndex::SingleChunk,
2255                                        ChunkIndexType::Implicit => ChunkIndex::Implicit,
2256                                        ChunkIndexType::FixedArray => ChunkIndex::FixedArray,
2257                                        ChunkIndexType::ExtensibleArray => {
2258                                            ChunkIndex::ExtensibleArray
2259                                        }
2260                                        ChunkIndexType::BTreeV2 => ChunkIndex::BtreeV2,
2261                                    })
2262                                }
2263                                _ => None,
2264                            })
2265                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2266                    }
2267                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2268                }
2269            }
2270            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2271                "chunk_index() is only available in read mode".into(),
2272            )),
2273        }
2274    }
2275
2276    /// Return this dataset's filter pipeline (read mode only), in
2277    /// application order. Empty when the dataset has no filter pipeline
2278    /// message at all — an unfiltered dataset, not an error.
2279    ///
2280    /// # Errors
2281    ///
2282    /// Returns an error if the file is in write mode, or if the dataset can
2283    /// no longer be found in the reader's metadata.
2284    pub fn filters(&self) -> Result<Vec<Filter>> {
2285        match &self.info {
2286            DatasetInfo::Reader { name, .. } => {
2287                let mut inner = borrow_inner_mut(&self.file_inner);
2288                match &mut *inner {
2289                    H5FileInner::Reader(reader) => reader
2290                        .dataset_info(name)
2291                        .map(|info| {
2292                            info.filter_pipeline
2293                                .as_ref()
2294                                .map(|fp| fp.filters.clone())
2295                                .unwrap_or_default()
2296                        })
2297                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2298                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2299                }
2300            }
2301            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2302                "filters() is only available in read mode".into(),
2303            )),
2304        }
2305    }
2306
2307    /// Return this dataset's fill-value state (read mode only).
2308    ///
2309    /// # Errors
2310    ///
2311    /// Returns an error if the file is in write mode, or if the dataset can
2312    /// no longer be found in the reader's metadata.
2313    pub fn fill_value(&self) -> Result<FillValue> {
2314        match &self.info {
2315            DatasetInfo::Reader { name, .. } => {
2316                let mut inner = borrow_inner_mut(&self.file_inner);
2317                match &mut *inner {
2318                    H5FileInner::Reader(reader) => reader
2319                        .dataset_info(name)
2320                        .map(|info| match info.fill_defined {
2321                            0 => FillValue::Undefined,
2322                            2 => {
2323                                FillValue::UserDefined(info.fill_value.clone().unwrap_or_default())
2324                            }
2325                            _ => FillValue::Default,
2326                        })
2327                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2328                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2329                }
2330            }
2331            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2332                "fill_value() is only available in read mode".into(),
2333            )),
2334        }
2335    }
2336
2337    /// Return when this dataset's fill value is written into allocated
2338    /// storage (read mode only) — `H5Pget_fill_time`.
2339    ///
2340    /// # Errors
2341    ///
2342    /// Returns an error if the file is in write mode, or if the dataset can
2343    /// no longer be found in the reader's metadata.
2344    pub fn fill_time(&self) -> Result<FillTime> {
2345        match &self.info {
2346            DatasetInfo::Reader { name, .. } => {
2347                let mut inner = borrow_inner_mut(&self.file_inner);
2348                match &mut *inner {
2349                    H5FileInner::Reader(reader) => reader
2350                        .dataset_info(name)
2351                        .map(|info| match info.fill_write_time {
2352                            0 => FillTime::Alloc,
2353                            1 => FillTime::Never,
2354                            _ => FillTime::IfSet,
2355                        })
2356                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2357                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2358                }
2359            }
2360            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2361                "fill_time() is only available in read mode".into(),
2362            )),
2363        }
2364    }
2365
2366    /// Return when this dataset's raw-data storage is allocated (read mode
2367    /// only) — `H5Pget_alloc_time`.
2368    ///
2369    /// # Errors
2370    ///
2371    /// Returns an error if the file is in write mode, or if the dataset can
2372    /// no longer be found in the reader's metadata.
2373    pub fn alloc_time(&self) -> Result<AllocTime> {
2374        match &self.info {
2375            DatasetInfo::Reader { name, .. } => {
2376                let mut inner = borrow_inner_mut(&self.file_inner);
2377                match &mut *inner {
2378                    H5FileInner::Reader(reader) => reader
2379                        .dataset_info(name)
2380                        .map(|info| match info.alloc_time {
2381                            1 => AllocTime::Early,
2382                            3 => AllocTime::Incr,
2383                            _ => AllocTime::Late,
2384                        })
2385                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2386                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2387                }
2388            }
2389            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2390                "alloc_time() is only available in read mode".into(),
2391            )),
2392        }
2393    }
2394
2395    /// Return this dataset's external raw-data file segments (read mode
2396    /// only), in the order the dataset's logical byte range concatenates
2397    /// them. Empty for a dataset whose data lives in this file.
2398    ///
2399    /// # Errors
2400    ///
2401    /// Returns an error if the file is in write mode, or if the dataset can
2402    /// no longer be found in the reader's metadata.
2403    pub fn external_files(&self) -> Result<Vec<ExternalFileSegment>> {
2404        match &self.info {
2405            DatasetInfo::Reader { name, .. } => {
2406                let mut inner = borrow_inner_mut(&self.file_inner);
2407                match &mut *inner {
2408                    H5FileInner::Reader(reader) => reader
2409                        .dataset_info(name)
2410                        .map(|info| info.external_files.clone())
2411                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2412                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2413                }
2414            }
2415            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2416                "external_files() is only available in read mode".into(),
2417            )),
2418        }
2419    }
2420
2421    /// Return this dataset's maximum dimension sizes (read mode only):
2422    /// `None` in a dimension marks that axis unlimited. A dataset with no
2423    /// maximum-dimensions message reports its current shape (max == current
2424    /// — the upstream convention for a fixed-extent dataset).
2425    ///
2426    /// # Errors
2427    ///
2428    /// Returns an error if the file is in write mode, or if the dataset can
2429    /// no longer be found in the reader's metadata.
2430    pub fn max_shape(&self) -> Result<Vec<Option<usize>>> {
2431        match &self.info {
2432            DatasetInfo::Reader { name, .. } => {
2433                let mut inner = borrow_inner_mut(&self.file_inner);
2434                match &mut *inner {
2435                    H5FileInner::Reader(reader) => reader
2436                        .dataset_info(name)
2437                        .map(|info| match &info.dataspace.max_dims {
2438                            Some(max_dims) => max_dims
2439                                .iter()
2440                                .map(|&d| (d != u64::MAX).then_some(d as usize))
2441                                .collect(),
2442                            None => info
2443                                .dataspace
2444                                .dims
2445                                .iter()
2446                                .map(|&d| Some(d as usize))
2447                                .collect(),
2448                        })
2449                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2450                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2451                }
2452            }
2453            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2454                "max_shape() is only available in read mode".into(),
2455            )),
2456        }
2457    }
2458
2459    /// Return this dataset's virtual-dataset source/virtual mappings (read
2460    /// mode only), in on-disk order. Empty for any dataset whose layout is
2461    /// not virtual, and for a virtual dataset that has no mappings yet.
2462    ///
2463    /// # Errors
2464    ///
2465    /// Returns an error if the file is in write mode, or if the dataset can
2466    /// no longer be found in the reader's metadata.
2467    pub fn virtual_mappings(&self) -> Result<Vec<VirtualMapping>> {
2468        match &self.info {
2469            DatasetInfo::Reader { name, .. } => {
2470                let mut inner = borrow_inner_mut(&self.file_inner);
2471                match &mut *inner {
2472                    H5FileInner::Reader(reader) => reader
2473                        .dataset_info(name)
2474                        .map(|info| {
2475                            info.virtual_mappings
2476                                .as_ref()
2477                                .map(|vml| vml.mappings.clone())
2478                                .unwrap_or_default()
2479                        })
2480                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2481                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2482                }
2483            }
2484            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2485                "virtual_mappings() is only available in read mode".into(),
2486            )),
2487        }
2488    }
2489
2490    /// Return the names of all attributes on this dataset (read mode only).
2491    pub fn attr_names(&self) -> Result<Vec<String>> {
2492        match &self.info {
2493            DatasetInfo::Reader { name, .. } => {
2494                let mut inner = borrow_inner_mut(&self.file_inner);
2495                match &mut *inner {
2496                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
2497                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2498                }
2499            }
2500            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2501                "attr_names not available in write mode".into(),
2502            )),
2503        }
2504    }
2505
2506    /// Why the attribute `attr_name` on this dataset cannot be read, or `None`
2507    /// when it can be.
2508    ///
2509    /// An attribute whose message this crate cannot decode is still listed by
2510    /// [`attr_names`](Self::attr_names) — the object header carries it — and
2511    /// this says what stands in the way. Opening it through
2512    /// [`attr`](Self::attr) fails with the same text.
2513    pub fn attr_unreadable_reason(&self, attr_name: &str) -> Result<Option<String>> {
2514        match &self.info {
2515            DatasetInfo::Reader { name, .. } => {
2516                let mut inner = borrow_inner_mut(&self.file_inner);
2517                match &mut *inner {
2518                    H5FileInner::Reader(reader) => Ok(reader
2519                        .dataset_attr_unreadable_reason(name, attr_name)
2520                        .map(str::to_string)),
2521                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2522                }
2523            }
2524            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2525                "attr_unreadable_reason not available in write mode".into(),
2526            )),
2527        }
2528    }
2529
2530    /// Why this dataset's attribute *set* cannot be listed, or `None` when it
2531    /// can be.
2532    ///
2533    /// The object-scope counterpart of
2534    /// [`attr_unreadable_reason`](Self::attr_unreadable_reason). A dense
2535    /// attribute set is indexed by name hash, so a heap or index that will not
2536    /// read yields no names to hang a per-attribute reason on;
2537    /// [`attr_names`](Self::attr_names) then returns the failure rather than a
2538    /// short list, and this reports it without an attribute name.
2539    pub fn attrs_unreadable_reason(&self) -> Result<Option<String>> {
2540        match &self.info {
2541            DatasetInfo::Reader { name, .. } => {
2542                let mut inner = borrow_inner_mut(&self.file_inner);
2543                match &mut *inner {
2544                    H5FileInner::Reader(reader) => Ok(reader
2545                        .dataset_attrs_unreadable_reason(name)
2546                        .map(str::to_string)),
2547                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2548                }
2549            }
2550            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2551                "attrs_unreadable_reason not available in write mode".into(),
2552            )),
2553        }
2554    }
2555
2556    /// This dataset's own compact-vs-dense attribute storage — the
2557    /// equivalent of `h5py.h5o.get_info(did.id).meta_size.attr.index_size`
2558    /// being nonzero (read mode only).
2559    pub fn attr_storage(&self) -> Result<AttributeStorage> {
2560        match &self.info {
2561            DatasetInfo::Reader { name, .. } => {
2562                let mut inner = borrow_inner_mut(&self.file_inner);
2563                match &mut *inner {
2564                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_storage(name)?),
2565                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2566                }
2567            }
2568            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2569                "attr_storage not available in write mode".into(),
2570            )),
2571        }
2572    }
2573
2574    /// This dataset's own object-header attribute count — the equivalent of
2575    /// `h5py.h5o.get_info(did.id).num_attrs` (read mode only).
2576    pub fn header_attr_count(&self) -> Result<u64> {
2577        match &self.info {
2578            DatasetInfo::Reader { name, .. } => {
2579                let mut inner = borrow_inner_mut(&self.file_inner);
2580                match &mut *inner {
2581                    H5FileInner::Reader(reader) => Ok(reader.dataset_header_attr_count(name)?),
2582                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2583                }
2584            }
2585            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2586                "header_attr_count not available in write mode".into(),
2587            )),
2588        }
2589    }
2590
2591    /// Open an attribute by name (read mode only).
2592    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
2593        match &self.info {
2594            DatasetInfo::Reader { name, .. } => {
2595                let mut inner = borrow_inner_mut(&self.file_inner);
2596                match &mut *inner {
2597                    H5FileInner::Reader(reader) => {
2598                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
2599                        Ok(crate::attribute::H5Attribute::new_reader(
2600                            clone_inner(&self.file_inner),
2601                            attr_msg,
2602                        ))
2603                    }
2604                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2605                }
2606            }
2607            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2608                "attr() not available in write mode".into(),
2609            )),
2610        }
2611    }
2612
2613    /// Start building a new attribute on this dataset.
2614    ///
2615    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
2616    /// and `.create("name")` to finalize.
2617    ///
2618    /// # Example
2619    ///
2620    /// ```no_run
2621    /// # use rust_hdf5::H5File;
2622    /// # use rust_hdf5::types::VarLenUnicode;
2623    /// let file = H5File::create("attr.h5").unwrap();
2624    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
2625    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
2626    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
2627    /// ```
2628    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
2629        let ds_index = match &self.info {
2630            DatasetInfo::Writer { index, .. } => *index,
2631            DatasetInfo::Reader { .. } => {
2632                // Reader mode: we'll return a builder that will error on create.
2633                // Using usize::MAX as sentinel.
2634                usize::MAX
2635            }
2636        };
2637        AttrBuilder::new(&self.file_inner, ds_index)
2638    }
2639
2640    /// Write a typed slice holding the dataset's whole image.
2641    ///
2642    /// The slice length must match the total number of elements declared by
2643    /// the dataset shape. The data is reinterpreted as raw bytes and written
2644    /// to the file: to the contiguous data block, or — for a chunked dataset —
2645    /// scattered across its chunk grid, through the filter pipeline if one is
2646    /// set. To write only part of a dataset, use
2647    /// [`write_slice`](Self::write_slice).
2648    ///
2649    /// # Errors
2650    ///
2651    /// Returns an error if:
2652    /// - The file is in read mode.
2653    /// - The data length does not match the declared shape.
2654    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
2655        match &self.info {
2656            DatasetInfo::Writer {
2657                index,
2658                shape,
2659                element_size,
2660                chunk_index,
2661                is_null,
2662            } => {
2663                if *is_null {
2664                    return Err(Hdf5Error::InvalidState(
2665                        "cannot write to a NULL dataspace dataset".into(),
2666                    ));
2667                }
2668                let total_elements: usize = shape.iter().product();
2669                if data.len() != total_elements {
2670                    return Err(Hdf5Error::InvalidState(format!(
2671                        "data length {} does not match dataset size {}",
2672                        data.len(),
2673                        total_elements,
2674                    )));
2675                }
2676
2677                // Verify element size matches
2678                if T::element_size() != *element_size {
2679                    return Err(Hdf5Error::TypeMismatch(format!(
2680                        "write type has element size {} but dataset expects {}",
2681                        T::element_size(),
2682                        element_size,
2683                    )));
2684                }
2685
2686                // Safety: T: Copy + 'static (numeric primitive) with well-defined
2687                // byte representation. The resulting slice borrows `data` and
2688                // lives only as long as this block.
2689                let byte_len = data.len() * T::element_size();
2690                let host =
2691                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
2692                let datatype = {
2693                    let inner = borrow_inner(&self.file_inner);
2694                    match &*inner {
2695                        H5FileInner::Writer(writer) => writer.dataset_datatype(*index),
2696                        _ => {
2697                            return Err(Hdf5Error::InvalidState(
2698                                "file is no longer in write mode".into(),
2699                            ))
2700                        }
2701                    }
2702                };
2703                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
2704
2705                if let Some(kind) = *chunk_index {
2706                    // A chunked dataset has no contiguous data block; scatter
2707                    // the full row-major image into its chunk grid and write
2708                    // each chunk through the dataset's filter pipeline.
2709                    return self.write_full_image_chunked(*index, kind, &stored, *element_size);
2710                }
2711
2712                let inner = borrow_inner(&self.file_inner);
2713                match &*inner {
2714                    H5FileInner::Writer(writer) => {
2715                        writer.write_dataset_raw(*index, &stored)?;
2716                        Ok(())
2717                    }
2718                    _ => Err(Hdf5Error::InvalidState(
2719                        "file is no longer in write mode".into(),
2720                    )),
2721                }
2722            }
2723            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2724                "cannot write to a dataset opened in read mode".into(),
2725            )),
2726        }
2727    }
2728
2729    /// Write the raw byte image of the whole dataset directly.
2730    ///
2731    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
2732    /// data block, or a chunk grid the image is scattered across.
2733    ///
2734    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
2735    /// `H5Type` carrier, so it works for element types that have no matching
2736    /// Rust primitive — in particular a runtime
2737    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
2738    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
2739    /// `product(shape) * element_size`, where `element_size` is taken from the
2740    /// dataset's on-disk datatype.
2741    ///
2742    /// ```no_run
2743    /// # use rust_hdf5::H5File;
2744    /// # use rust_hdf5::types::{CompoundType, H5Type};
2745    /// let file = H5File::create("c.h5").unwrap();
2746    /// let ct = CompoundType {
2747    ///     members: vec![
2748    ///         ("id".to_string(), i32::hdf5_type(), 0),
2749    ///         ("val".to_string(), f64::hdf5_type(), 4),
2750    ///     ],
2751    ///     total_size: 12,
2752    /// };
2753    /// let ds = file
2754    ///     .new_dataset::<u8>()
2755    ///     .datatype(ct.to_datatype())
2756    ///     .shape(&[2])
2757    ///     .create("records")
2758    ///     .unwrap();
2759    /// let mut bytes = Vec::new();
2760    /// bytes.extend_from_slice(&1i32.to_le_bytes());
2761    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
2762    /// bytes.extend_from_slice(&2i32.to_le_bytes());
2763    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
2764    /// ds.write_raw_bytes(&bytes).unwrap();
2765    /// ```
2766    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
2767        match &self.info {
2768            DatasetInfo::Writer {
2769                index,
2770                shape,
2771                element_size,
2772                chunk_index,
2773                is_null,
2774            } => {
2775                if *is_null {
2776                    return Err(Hdf5Error::InvalidState(
2777                        "cannot write to a NULL dataspace dataset".into(),
2778                    ));
2779                }
2780                let expected: usize = shape.iter().product::<usize>() * *element_size;
2781                if bytes.len() != expected {
2782                    return Err(Hdf5Error::InvalidState(format!(
2783                        "raw byte length {} does not match dataset size {} \
2784                         (product(shape) * element_size {})",
2785                        bytes.len(),
2786                        expected,
2787                        element_size,
2788                    )));
2789                }
2790                if let Some(kind) = *chunk_index {
2791                    // Scatter the full row-major image into the chunk grid
2792                    // (same path as write_raw, carrier-agnostic bytes).
2793                    return self.write_full_image_chunked(*index, kind, bytes, *element_size);
2794                }
2795                let inner = borrow_inner(&self.file_inner);
2796                match &*inner {
2797                    H5FileInner::Writer(writer) => {
2798                        writer.write_dataset_raw(*index, bytes)?;
2799                        Ok(())
2800                    }
2801                    _ => Err(Hdf5Error::InvalidState(
2802                        "file is no longer in write mode".into(),
2803                    )),
2804                }
2805            }
2806            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2807                "cannot write to a dataset opened in read mode".into(),
2808            )),
2809        }
2810    }
2811
2812    /// Scatter a full row-major dataset image into its chunk grid, writing
2813    /// every chunk through the dataset's filter pipeline.
2814    ///
2815    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
2816    /// — it is how [`write_raw`](Self::write_raw) and
2817    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
2818    /// (including the single auto-chunk created when a filter is set without
2819    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
2820    /// chunk footprint, exactly as libhdf5 stores them.
2821    fn write_full_image_chunked(
2822        &self,
2823        index: usize,
2824        kind: ChunkIndexKind,
2825        bytes: &[u8],
2826        element_size: usize,
2827    ) -> Result<()> {
2828        let inner = borrow_inner(&self.file_inner);
2829        let writer = match &*inner {
2830            H5FileInner::Writer(w) => w,
2831            _ => {
2832                return Err(Hdf5Error::InvalidState(
2833                    "file is no longer in write mode".into(),
2834                ))
2835            }
2836        };
2837        // Whole-operation guard: the flush, the grid snapshot and the chunk
2838        // writes below must not interleave with a concurrent same-dataset
2839        // operation.
2840        let cell = writer.ds(index);
2841        let _op = cell.op.lock();
2842        // A buffered append tail would flush over the image at close; hand
2843        // it to the chunks first, the image below overwrites everything.
2844        writer.flush_append_buffer(index)?;
2845        let chunk_dims = writer
2846            .dataset_chunk_dims(index)
2847            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
2848            .to_vec();
2849        let dims = writer.dataset_dims(index).to_vec();
2850        let rank = dims.len();
2851
2852        // Chunk grid: number of chunks along each dimension (row-major).
2853        let mut grid = vec![0u64; rank];
2854        for d in 0..rank {
2855            grid[d] = if chunk_dims[d] > 0 {
2856                dims[d].div_ceil(chunk_dims[d])
2857            } else {
2858                0
2859            };
2860        }
2861        let total_chunks: u64 = grid.iter().product();
2862
2863        // Decode the iteration counter into row-major coordinates over the
2864        // *current* image's chunk grid. This is only an odometer over the
2865        // chunks the image spans — the slot a chunk is recorded under comes
2866        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
2867        // extent decides.
2868        let coords_of = |linear: u64| -> Vec<u64> {
2869            let mut rem = linear;
2870            let mut coords = vec![0u64; rank];
2871            for d in (0..rank).rev() {
2872                coords[d] = rem % grid[d];
2873                rem /= grid[d];
2874            }
2875            coords
2876        };
2877
2878        // The batch entry points exist for one reason: to run the filter
2879        // pipeline over a window of chunks in parallel. An unfiltered dataset
2880        // has no pipeline to run, so it takes the plain per-chunk owner
2881        // whatever its index is, and only a filtered extensible or fixed array
2882        // — the two indexes with a batch entry point — takes the window below.
2883        let batched = writer.dataset_is_filtered(index)
2884            && matches!(
2885                kind,
2886                ChunkIndexKind::ExtensibleArray | ChunkIndexKind::FixedArray
2887            );
2888        if !batched {
2889            // One staging buffer for the whole image, reused chunk after
2890            // chunk: a chunk that already sits as one complete run of `bytes`
2891            // needs no staging at all and goes to the file straight out of the
2892            // caller's slice, so only an n-D interleave or a short edge pays
2893            // for a gather.
2894            let mut staging = Vec::new();
2895            for linear in 0..total_chunks {
2896                let coords = coords_of(linear);
2897                let chunk =
2898                    match Self::contiguous_chunk_span(&dims, &chunk_dims, &coords, element_size) {
2899                        Some(span) => &bytes[span],
2900                        None => {
2901                            Self::gather_chunk_into(
2902                                &mut staging,
2903                                bytes,
2904                                &dims,
2905                                &chunk_dims,
2906                                &coords,
2907                                element_size,
2908                            );
2909                            &staging[..]
2910                        }
2911                    };
2912                writer.write_chunk_at_coords(index, &coords, chunk)?;
2913            }
2914        } else {
2915            // Hand the pipeline a window of chunks so it compresses them in
2916            // parallel (with the `parallel` feature). A fixed-size window
2917            // bounds peak memory instead of materializing every chunk at once;
2918            // 256 keeps every rayon worker fed while capping the transient
2919            // buffers to window * chunk bytes. The compressors read the window
2920            // concurrently, so a gathered chunk here cannot share one reused
2921            // buffer the way the sequential path above does — but a chunk that
2922            // is already a complete run of `bytes` is borrowed, not copied.
2923            // The two indexes differ only in how a chunk is addressed: EA by
2924            // its linear grid index, FA by grid coordinates.
2925            const BATCH_WINDOW: u64 = 256;
2926            let mut start = 0u64;
2927            while start < total_chunks {
2928                let end = (start + BATCH_WINDOW).min(total_chunks);
2929                let items: Vec<(Vec<u64>, Cow<'_, [u8]>)> = (start..end)
2930                    .map(|counter| {
2931                        let coords = coords_of(counter);
2932                        let data = match Self::contiguous_chunk_span(
2933                            &dims,
2934                            &chunk_dims,
2935                            &coords,
2936                            element_size,
2937                        ) {
2938                            Some(span) => Cow::Borrowed(&bytes[span]),
2939                            None => {
2940                                let mut buf = Vec::new();
2941                                Self::gather_chunk_into(
2942                                    &mut buf,
2943                                    bytes,
2944                                    &dims,
2945                                    &chunk_dims,
2946                                    &coords,
2947                                    element_size,
2948                                );
2949                                Cow::Owned(buf)
2950                            }
2951                        };
2952                        (coords, data)
2953                    })
2954                    .collect();
2955                if kind == ChunkIndexKind::FixedArray {
2956                    let pairs: Vec<(&[u64], &[u8])> = items
2957                        .iter()
2958                        .map(|(c, d)| (c.as_slice(), d.as_ref()))
2959                        .collect();
2960                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
2961                } else {
2962                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
2963                    for (c, d) in &items {
2964                        pairs.push((writer.chunk_slot(index, c)?, d.as_ref()));
2965                    }
2966                    writer.write_chunks_batch_inner(index, &pairs)?;
2967                }
2968                start = end;
2969            }
2970        }
2971        Ok(())
2972    }
2973
2974    /// The byte range one chunk occupies in a row-major full-dataset image,
2975    /// for a chunk that needs no gather at all: its elements are one
2976    /// contiguous run of `source` *and* they fill the chunk shape exactly, so
2977    /// the bytes that go to the file are already sitting in the caller's
2978    /// buffer.
2979    ///
2980    /// Both halves hold when every dimension after the first spans the whole
2981    /// dataset (`chunk_dims[d] == dims[d]`, leaving nothing interleaved and no
2982    /// padding along those axes) and the chunk does not hang off the far edge
2983    /// of the first — which is every full chunk of a 1-D dataset. `None` means
2984    /// the chunk has to be gathered.
2985    fn contiguous_chunk_span(
2986        dims: &[u64],
2987        chunk_dims: &[u64],
2988        coords: &[u64],
2989        element_size: usize,
2990    ) -> Option<std::ops::Range<usize>> {
2991        let rank = dims.len();
2992        if rank == 0 || chunk_dims[1..] != dims[1..] {
2993            return None;
2994        }
2995        if (coords[0] + 1) * chunk_dims[0] > dims[0] {
2996            return None;
2997        }
2998        let plane: u64 = dims[1..].iter().product::<u64>() * element_size as u64;
2999        let start = usize::try_from(coords[0] * chunk_dims[0] * plane).ok()?;
3000        let len = usize::try_from(chunk_dims[0] * plane).ok()?;
3001        Some(start..start.checked_add(len)?)
3002    }
3003
3004    /// Gather one chunk's bytes from a row-major full-dataset image into
3005    /// `out`, replacing whatever it held.
3006    ///
3007    /// `coords` are the chunk's grid coordinates. `out` is left exactly
3008    /// `product(chunk_dims) * element_size` bytes long, holding the chunk's
3009    /// elements and zero where the chunk extends past the dataset edge — so a
3010    /// caller may hand the same buffer to one chunk after another.
3011    fn gather_chunk_into(
3012        out: &mut Vec<u8>,
3013        source: &[u8],
3014        dims: &[u64],
3015        chunk_dims: &[u64],
3016        coords: &[u64],
3017        element_size: usize,
3018    ) {
3019        let rank = dims.len();
3020        let chunk_elems: u64 = chunk_dims.iter().product();
3021        let chunk_bytes = chunk_elems as usize * element_size;
3022        if rank == 0 {
3023            // Scalar dataset: a single element, no chunking dimension.
3024            out.clear();
3025            out.resize(chunk_bytes, 0);
3026            if source.len() >= element_size {
3027                out[..element_size].copy_from_slice(&source[..element_size]);
3028            }
3029            return;
3030        }
3031
3032        // Actual extent of this chunk along each dimension (edge chunks are
3033        // smaller than the nominal chunk shape).
3034        let mut extent = vec![0u64; rank];
3035        for d in 0..rank {
3036            let start = coords[d] * chunk_dims[d];
3037            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
3038            extent[d] = end.saturating_sub(start);
3039        }
3040        // Size the buffer, then zero it only when this chunk leaves part of
3041        // its shape uncovered: a full chunk has every byte overwritten below,
3042        // while an edge chunk's padding must read as zero even though a
3043        // reused buffer still holds the previous chunk's bytes.
3044        if out.len() != chunk_bytes {
3045            out.clear();
3046            out.resize(chunk_bytes, 0);
3047        } else if extent != chunk_dims {
3048            out.fill(0);
3049        }
3050        if extent.contains(&0) {
3051            return; // nothing of the dataset falls in this chunk
3052        }
3053
3054        // Row-major strides (in elements) for the source (over `dims`) and the
3055        // destination chunk buffer (over `chunk_dims`).
3056        let mut src_stride = vec![1u64; rank];
3057        let mut dst_stride = vec![1u64; rank];
3058        for d in (0..rank - 1).rev() {
3059            src_stride[d] = src_stride[d + 1] * dims[d + 1];
3060            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
3061        }
3062
3063        // Copy one contiguous run along the last axis per outer multi-index.
3064        let last = rank - 1;
3065        let run = extent[last] as usize * element_size;
3066        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
3067        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
3068        for _ in 0..outer {
3069            let mut src_off = 0u64;
3070            let mut dst_off = 0u64;
3071            for d in 0..rank {
3072                let global = coords[d] * chunk_dims[d] + idx[d];
3073                src_off += global * src_stride[d];
3074                dst_off += idx[d] * dst_stride[d];
3075            }
3076            let s = src_off as usize * element_size;
3077            let dpos = dst_off as usize * element_size;
3078            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
3079
3080            // Advance the multi-index over axes [0..last); the last axis is the
3081            // contiguous run handled above.
3082            let mut d = last;
3083            while d > 0 {
3084                d -= 1;
3085                idx[d] += 1;
3086                if idx[d] < extent[d] {
3087                    break;
3088                }
3089                idx[d] = 0;
3090            }
3091        }
3092    }
3093
3094    /// Write a single chunk to a chunked dataset.
3095    ///
3096    /// `chunk_idx` is the linear chunk index (typically the frame number for
3097    /// streaming datasets). `data` is the raw byte data for one chunk.
3098    ///
3099    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
3100    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
3101    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
3102        match &self.info {
3103            DatasetInfo::Writer {
3104                index, chunk_index, ..
3105            } => {
3106                let Some(kind) = *chunk_index else {
3107                    return Err(Hdf5Error::InvalidState(
3108                        "write_chunk is only for chunked datasets".into(),
3109                    ));
3110                };
3111                if kind == ChunkIndexKind::BtreeV2 {
3112                    return Err(Hdf5Error::InvalidState(
3113                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
3114                         with the chunk's grid coordinates"
3115                            .into(),
3116                    ));
3117                }
3118
3119                let inner = borrow_inner(&self.file_inner);
3120                match &*inner {
3121                    H5FileInner::Writer(writer) => {
3122                        // One op: the slot decode and the write see the same
3123                        // extents.
3124                        let cell = writer.ds(*index);
3125                        let _op = cell.op.lock();
3126                        match kind {
3127                            // All four address a chunk by its grid
3128                            // coordinates, so the linear slot is decoded back
3129                            // into them.
3130                            ChunkIndexKind::FixedArray
3131                            | ChunkIndexKind::Implicit
3132                            | ChunkIndexKind::SingleChunk
3133                            | ChunkIndexKind::BtreeV1 => {
3134                                let coords =
3135                                    writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3136                                writer.write_chunk_at_coords(*index, &coords, data)?;
3137                            }
3138                            _ => writer.write_chunk_inner(*index, chunk_idx as u64, data)?,
3139                        }
3140                        Ok(())
3141                    }
3142                    _ => Err(Hdf5Error::InvalidState(
3143                        "file is no longer in write mode".into(),
3144                    )),
3145                }
3146            }
3147            DatasetInfo::Reader { .. } => {
3148                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3149            }
3150        }
3151    }
3152
3153    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
3154    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
3155    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
3156    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
3157    ///
3158    /// `chunk_idx` is the linear chunk index (the frame number for streaming
3159    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
3160    /// the already-filtered bytes of one chunk — its length is the *stored*
3161    /// (compressed) size, not the uncompressed chunk size.
3162    ///
3163    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
3164    /// dataset's pipeline was **not** applied to this chunk and must be skipped
3165    /// on read. Pass 0 when the full pipeline was already applied upstream (the
3166    /// common case: a codec plugin handed you compressed frames).
3167    ///
3168    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3169    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
3170    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
3171    /// against, so address its chunks with
3172    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
3173    ///
3174    /// # Reading back
3175    ///
3176    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
3177    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
3178    /// the reader skipping exactly the filters the mask marks as not applied.
3179    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
3180        match &self.info {
3181            DatasetInfo::Writer {
3182                index, chunk_index, ..
3183            } => {
3184                let Some(kind) = *chunk_index else {
3185                    return Err(Hdf5Error::InvalidState(
3186                        "write_chunk_raw is only for chunked datasets".into(),
3187                    ));
3188                };
3189                if kind == ChunkIndexKind::BtreeV2 {
3190                    return Err(Hdf5Error::InvalidState(
3191                        "this dataset uses a v2 B-tree chunk index; use \
3192                         write_chunk_raw_at with the chunk's grid coordinates"
3193                            .into(),
3194                    ));
3195                }
3196                if kind == ChunkIndexKind::Implicit {
3197                    return Err(Hdf5Error::InvalidState(
3198                        "this dataset uses the implicit chunk index, which stores \
3199                         every chunk at its full unfiltered size and has nowhere to \
3200                         record a stored size or a filter mask"
3201                            .into(),
3202                    ));
3203                }
3204
3205                let inner = borrow_inner(&self.file_inner);
3206                match &*inner {
3207                    H5FileInner::Writer(writer) => {
3208                        // One op: the slot decode and the write see the same
3209                        // extents.
3210                        let cell = writer.ds(*index);
3211                        let _op = cell.op.lock();
3212                        if kind == ChunkIndexKind::FixedArray {
3213                            // Fixed-array dataset: decode the index-grid slot
3214                            // into row-major grid coordinates.
3215                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3216                            writer.write_compressed_chunk_fixed_array_inner(
3217                                *index,
3218                                &coords,
3219                                data,
3220                                filter_mask,
3221                            )?;
3222                        } else if kind == ChunkIndexKind::BtreeV1 {
3223                            // Same for the classic index, whose key carries a
3224                            // stored size and a filter mask of its own.
3225                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3226                            writer.write_compressed_chunk_btree_v1_inner(
3227                                *index,
3228                                &coords,
3229                                data,
3230                                filter_mask,
3231                            )?;
3232                        } else if kind == ChunkIndexKind::SingleChunk {
3233                            // Same again for the single-chunk index, whose
3234                            // layout message carries the stored size and mask
3235                            // inline.
3236                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3237                            writer.write_compressed_chunk_single_chunk_inner(
3238                                *index,
3239                                &coords,
3240                                data,
3241                                filter_mask,
3242                            )?;
3243                        } else {
3244                            writer.write_compressed_chunk_inner(
3245                                *index,
3246                                chunk_idx as u64,
3247                                data,
3248                                filter_mask,
3249                            )?;
3250                        }
3251                        Ok(())
3252                    }
3253                    _ => Err(Hdf5Error::InvalidState(
3254                        "file is no longer in write mode".into(),
3255                    )),
3256                }
3257            }
3258            DatasetInfo::Reader { .. } => {
3259                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3260            }
3261        }
3262    }
3263
3264    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
3265    /// chunk-grid coordinates (one per dimension).
3266    ///
3267    /// This is the entry point for datasets with two or more unlimited
3268    /// dimensions. The dataset's logical dimensions are extended to cover
3269    /// the written chunk. `data` is the raw bytes of one full chunk.
3270    ///
3271    /// ```no_run
3272    /// # use rust_hdf5::H5File;
3273    /// let file = H5File::create("bt2.h5").unwrap();
3274    /// let ds = file.new_dataset::<i32>()
3275    ///     .shape(&[0, 0])
3276    ///     .chunk(&[2, 2])
3277    ///     .max_shape(&[None, None])
3278    ///     .create("grid")
3279    ///     .unwrap();
3280    /// let chunk = [0i32, 1, 2, 3];
3281    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
3282    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3283    /// ```
3284    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
3285        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
3286    }
3287
3288    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
3289    /// addressed by its chunk-grid coordinates.
3290    ///
3291    /// The coordinate-addressed twin of
3292    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
3293    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
3294    /// there is no fixed chunk grid for a linear index to mean anything against.
3295    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
3296    /// to cover the written chunk.
3297    ///
3298    /// `data` is the already-filtered bytes of one chunk — its length is the
3299    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
3300    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
3301    /// full pipeline already ran upstream.
3302    ///
3303    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3304    /// has no slot to record a stored size or mask.
3305    pub fn write_chunk_raw_at(
3306        &self,
3307        chunk_coords: &[usize],
3308        data: &[u8],
3309        filter_mask: u32,
3310    ) -> Result<()> {
3311        self.write_chunk_at_inner(
3312            chunk_coords,
3313            ChunkBytes::Prefiltered { data, filter_mask },
3314            "write_chunk_raw_at",
3315        )
3316    }
3317
3318    /// The single owner of coordinate-addressed chunk writes: validates the
3319    /// coordinates, grows the dataspace to cover them, and routes the bytes to
3320    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
3321    /// here or already ran upstream is carried by `bytes`, not by a second copy
3322    /// of this dispatch.
3323    fn write_chunk_at_inner(
3324        &self,
3325        chunk_coords: &[usize],
3326        bytes: ChunkBytes<'_>,
3327        what: &str,
3328    ) -> Result<()> {
3329        match &self.info {
3330            DatasetInfo::Writer {
3331                index, chunk_index, ..
3332            } => {
3333                let Some(kind) = *chunk_index else {
3334                    return Err(Hdf5Error::InvalidState(format!(
3335                        "{what} is only for chunked datasets"
3336                    )));
3337                };
3338                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3339                let inner = borrow_inner(&self.file_inner);
3340                let writer = match &*inner {
3341                    H5FileInner::Writer(w) => w,
3342                    _ => {
3343                        return Err(Hdf5Error::InvalidState(
3344                            "file is no longer in write mode".into(),
3345                        ))
3346                    }
3347                };
3348                // Whole-operation guard: the dims snapshot, the chunk write
3349                // and the extend below must not interleave with a concurrent
3350                // same-dataset operation.
3351                let cell = writer.ds(*index);
3352                let _op = cell.op.lock();
3353                let chunk_dims = writer
3354                    .dataset_chunk_dims(*index)
3355                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3356                    .to_vec();
3357                let dims = writer.dataset_dims(*index).to_vec();
3358                if coords.len() != dims.len() {
3359                    return Err(Hdf5Error::InvalidState(format!(
3360                        "chunk_coords has {} entries but the dataset has {} dimensions",
3361                        coords.len(),
3362                        dims.len()
3363                    )));
3364                }
3365                if chunk_dims.len() != dims.len() {
3366                    return Err(Hdf5Error::InvalidState(format!(
3367                        "dataset chunk shape has {} dimensions but the dataspace has {}",
3368                        chunk_dims.len(),
3369                        dims.len()
3370                    )));
3371                }
3372
3373                if kind == ChunkIndexKind::FixedArray {
3374                    // Fixed-array (fixed-shape) dataset: no dimension growth.
3375                    match bytes {
3376                        ChunkBytes::Unfiltered(data) => {
3377                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
3378                        }
3379                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3380                            .write_compressed_chunk_fixed_array_inner(
3381                                *index,
3382                                &coords,
3383                                data,
3384                                filter_mask,
3385                            )?,
3386                    }
3387                    return Ok(());
3388                }
3389
3390                if kind == ChunkIndexKind::Implicit {
3391                    // Implicit index: fixed shape, so no dimension growth
3392                    // either, and no slot to record a stored size in.
3393                    match bytes {
3394                        ChunkBytes::Unfiltered(data) => {
3395                            writer.write_chunk_implicit_inner(*index, &coords, data)?
3396                        }
3397                        ChunkBytes::Prefiltered { .. } => {
3398                            return Err(Hdf5Error::InvalidState(
3399                                "this dataset uses the implicit chunk index, which stores \
3400                                 every chunk at its full unfiltered size and has nowhere \
3401                                 to record a stored size or a filter mask"
3402                                    .into(),
3403                            ))
3404                        }
3405                    }
3406                    return Ok(());
3407                }
3408
3409                if kind == ChunkIndexKind::SingleChunk {
3410                    // Single-chunk index: fixed shape covered by exactly one
3411                    // chunk, so no dimension growth either.
3412                    match bytes {
3413                        ChunkBytes::Unfiltered(data) => {
3414                            writer.write_chunk_single_chunk_inner(*index, &coords, data)?
3415                        }
3416                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3417                            .write_compressed_chunk_single_chunk_inner(
3418                                *index,
3419                                &coords,
3420                                data,
3421                                filter_mask,
3422                            )?,
3423                    }
3424                    return Ok(());
3425                }
3426
3427                // The remaining indexes (v2 B-tree, v1 B-tree, extensible
3428                // array) can all grow: validate the coordinates and compute
3429                // the grown dimensions up-front, before any chunk is
3430                // written, so an overflowing coordinate cannot leave an
3431                // orphaned chunk in the file.
3432                //
3433                // The last chunk of a dimension usually hangs past the extent
3434                // — a length of 10 in chunks of 4 ends at 12 — so the growth
3435                // is capped at the declared maximum, which is what the chunk
3436                // still covers. Without the cap a legal edge chunk would be
3437                // written and then rejected by the extend below.
3438                let max_dims = writer.dataset_max_dims(*index);
3439                let mut new_dims = dims.clone();
3440                for d in 0..dims.len() {
3441                    let needed = coords[d]
3442                        .checked_add(1)
3443                        .and_then(|c| c.checked_mul(chunk_dims[d]))
3444                        .ok_or_else(|| {
3445                            Hdf5Error::InvalidState(format!(
3446                                "chunk coordinate {} in dimension {} is too large",
3447                                coords[d], d
3448                            ))
3449                        })?;
3450                    let needed = needed.min(max_dims[d]);
3451                    if needed > new_dims[d] {
3452                        new_dims[d] = needed;
3453                    }
3454                }
3455
3456                if kind == ChunkIndexKind::BtreeV2 {
3457                    match bytes {
3458                        ChunkBytes::Unfiltered(data) => {
3459                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
3460                        }
3461                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3462                            .write_compressed_chunk_btree_v2_inner(
3463                                *index,
3464                                &coords,
3465                                data,
3466                                filter_mask,
3467                            )?,
3468                    }
3469                } else if kind == ChunkIndexKind::BtreeV1 {
3470                    // The classic index takes any shape, fixed or unlimited,
3471                    // so it grows the dataspace with the chunk the way the v2
3472                    // B-tree does — bounded below by the maximum extent.
3473                    match bytes {
3474                        ChunkBytes::Unfiltered(data) => {
3475                            writer.write_chunk_btree_v1_inner(*index, &coords, data)?
3476                        }
3477                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3478                            .write_compressed_chunk_btree_v1_inner(
3479                                *index,
3480                                &coords,
3481                                data,
3482                                filter_mask,
3483                            )?,
3484                    }
3485                } else {
3486                    // Extensible array: the chunk's index-grid slot (row-major
3487                    // against the maximum extent).
3488                    let linear = writer.chunk_slot(*index, &coords)?;
3489                    match bytes {
3490                        ChunkBytes::Unfiltered(data) => {
3491                            writer.write_chunk_inner(*index, linear, data)?
3492                        }
3493                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3494                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
3495                    }
3496                }
3497
3498                if new_dims != dims {
3499                    writer.extend_dataset_inner(*index, &new_dims)?;
3500                }
3501                Ok(())
3502            }
3503            DatasetInfo::Reader { .. } => {
3504                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3505            }
3506        }
3507    }
3508
3509    /// Write multiple chunks in a batch, optionally compressing in parallel.
3510    ///
3511    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
3512    /// pipeline is configured and the `parallel` feature is enabled, all
3513    /// chunks are compressed concurrently via rayon.
3514    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
3515        match &self.info {
3516            DatasetInfo::Writer {
3517                index, chunk_index, ..
3518            } => {
3519                if chunk_index.is_none() {
3520                    return Err(Hdf5Error::InvalidState(
3521                        "write_chunks_batch is only for chunked datasets".into(),
3522                    ));
3523                }
3524                let pairs: Vec<(u64, &[u8])> = chunks
3525                    .iter()
3526                    .map(|(idx, data)| (*idx as u64, *data))
3527                    .collect();
3528                let inner = borrow_inner(&self.file_inner);
3529                match &*inner {
3530                    H5FileInner::Writer(writer) => {
3531                        writer.write_chunks_batch(*index, &pairs)?;
3532                        Ok(())
3533                    }
3534                    _ => Err(Hdf5Error::InvalidState(
3535                        "file is no longer in write mode".into(),
3536                    )),
3537                }
3538            }
3539            DatasetInfo::Reader { .. } => {
3540                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3541            }
3542        }
3543    }
3544
3545    /// Append data along the first dimension of a chunked dataset.
3546    ///
3547    /// `data` must contain a whole number of "frames" — slices along
3548    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
3549    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
3550    /// multiple of `H * W`.
3551    ///
3552    /// This method writes the necessary chunks and extends the dataset
3553    /// shape automatically.
3554    ///
3555    /// ```no_run
3556    /// # use rust_hdf5::H5File;
3557    /// let file = H5File::create("append.h5").unwrap();
3558    /// let ds = file.new_dataset::<f64>()
3559    ///     .shape(&[0, 3])
3560    ///     .chunk(&[1, 3])
3561    ///     .max_shape(&[None, Some(3)])
3562    ///     .create("data")
3563    ///     .unwrap();
3564    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
3565    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
3566    /// ```
3567    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
3568        match &self.info {
3569            DatasetInfo::Writer {
3570                index,
3571                element_size,
3572                chunk_index,
3573                ..
3574            } => {
3575                if chunk_index.is_none() {
3576                    return Err(Hdf5Error::InvalidState(
3577                        "append is only for chunked datasets".into(),
3578                    ));
3579                }
3580                if T::element_size() != *element_size {
3581                    return Err(Hdf5Error::TypeMismatch(format!(
3582                        "append type has element size {} but dataset expects {}",
3583                        T::element_size(),
3584                        element_size,
3585                    )));
3586                }
3587
3588                let ds_index = *index;
3589                let es = *element_size;
3590
3591                let inner = borrow_inner(&self.file_inner);
3592                let writer = match &*inner {
3593                    H5FileInner::Writer(w) => w,
3594                    _ => {
3595                        return Err(Hdf5Error::InvalidState(
3596                            "file is no longer in write mode".into(),
3597                        ))
3598                    }
3599                };
3600
3601                // Whole-operation guard: the buffer take, the frame writes,
3602                // the re-buffer and the extend below are separate slot
3603                // acquisitions that a concurrent same-dataset append must not
3604                // interleave with.
3605                let cell = writer.ds(ds_index);
3606                let _op = cell.op.lock();
3607
3608                let chunk_dims = writer
3609                    .dataset_chunk_dims(ds_index)
3610                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3611                    .to_vec();
3612                let dims = writer.dataset_dims(ds_index).to_vec();
3613
3614                // Frame size = product of dims[1..]
3615                let frame_elems: usize = if dims.len() > 1 {
3616                    dims[1..].iter().map(|&d| d as usize).product()
3617                } else {
3618                    1
3619                };
3620
3621                if frame_elems == 0 {
3622                    return Err(Hdf5Error::InvalidState(
3623                        "cannot append to dataset with zero-size trailing dimensions".into(),
3624                    ));
3625                }
3626
3627                if !data.len().is_multiple_of(frame_elems) {
3628                    return Err(Hdf5Error::InvalidState(format!(
3629                        "data length {} is not a multiple of frame size {}",
3630                        data.len(),
3631                        frame_elems,
3632                    )));
3633                }
3634
3635                let n_new_frames = data.len() / frame_elems;
3636                let current_dim0 = dims[0] as usize;
3637
3638                // Chunk size along first dimension
3639                let chunk_dim0 = chunk_dims[0] as usize;
3640                let frame_bytes = frame_elems * es;
3641
3642                let host = unsafe {
3643                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
3644                };
3645                let datatype = writer.dataset_datatype(ds_index);
3646                let raw = to_stored_byte_order(host, &datatype, es)?;
3647
3648                // Merge the buffer with the new frames when it is the
3649                // dataset's tail; a buffer left mid-extent (the extent moved
3650                // past it) keeps its recorded place — flush it and start
3651                // fresh at the current end.
3652                let taken = { writer.ds(ds_index).lock().append.take() };
3653                let (base_dim0, buffered_frames, mut combined) = match taken {
3654                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
3655                        (b.base as usize, b.frames as usize, b.bytes)
3656                    }
3657                    Some(b) => {
3658                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
3659                        (current_dim0, 0, Vec::new())
3660                    }
3661                    None => (current_dim0, 0, Vec::new()),
3662                };
3663                combined.extend_from_slice(&raw);
3664
3665                let total_frames = buffered_frames + n_new_frames;
3666
3667                // Rows up to the last chunk boundary are written now; the
3668                // tail that does not complete a chunk goes back in the
3669                // buffer for the next append (or the flush at close). The
3670                // boundary can precede `base_dim0` — a reopened file's
3671                // flushed partial chunk leaves the base mid-chunk — in
3672                // which case everything is tail.
3673                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
3674                let write_frames = last_boundary.saturating_sub(base_dim0);
3675                let tail_frames = total_frames - write_frames;
3676                if write_frames > 0 {
3677                    writer.write_append_frames(
3678                        ds_index,
3679                        base_dim0 as u64,
3680                        write_frames as u64,
3681                        &combined[..write_frames * frame_bytes],
3682                    )?;
3683                }
3684                if tail_frames > 0 {
3685                    let ds = writer.ds(ds_index);
3686                    let mut m = ds.lock();
3687                    m.append = Some(crate::io::writer::AppendBuffer {
3688                        base: (base_dim0 + write_frames) as u64,
3689                        frames: tail_frames as u64,
3690                        bytes: combined[write_frames * frame_bytes..].to_vec(),
3691                    });
3692                }
3693
3694                // Extend dims to include all frames (buffered + new)
3695                let logical_dim0 = base_dim0 + total_frames;
3696                let mut new_dims: Vec<u64> = dims;
3697                new_dims[0] = logical_dim0 as u64;
3698                writer.extend_dataset_inner(ds_index, &new_dims)?;
3699
3700                Ok(())
3701            }
3702            DatasetInfo::Reader { .. } => {
3703                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
3704            }
3705        }
3706    }
3707
3708    /// Extend the dimensions of a chunked dataset.
3709    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
3710        match &self.info {
3711            DatasetInfo::Writer {
3712                index, chunk_index, ..
3713            } => {
3714                if chunk_index.is_none() {
3715                    return Err(Hdf5Error::InvalidState(
3716                        "extend is only for chunked datasets".into(),
3717                    ));
3718                }
3719
3720                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3721                let inner = borrow_inner(&self.file_inner);
3722                match &*inner {
3723                    H5FileInner::Writer(writer) => {
3724                        writer.extend_dataset(*index, &dims_u64)?;
3725                        Ok(())
3726                    }
3727                    _ => Err(Hdf5Error::InvalidState(
3728                        "file is no longer in write mode".into(),
3729                    )),
3730                }
3731            }
3732            DatasetInfo::Reader { .. } => {
3733                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
3734            }
3735        }
3736    }
3737
3738    /// Set the logical extent of a chunked dataset, growing **or
3739    /// shrinking** any dimension.
3740    ///
3741    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
3742    /// dimension — for example to correct an over-extended frame count
3743    /// after writing a partial multi-frame chunk. Shrinking prunes the
3744    /// stored chunks the way libhdf5's `H5Dset_extent` does: a chunk
3745    /// entirely beyond the new extent is removed from the chunk index and
3746    /// its storage freed for reuse, and a chunk the new extent cuts
3747    /// through has its out-of-extent region overwritten with the fill
3748    /// value — so growing the extent back exposes fill values, not the
3749    /// old data. The new extent must not exceed the dataset's maximum
3750    /// dimensions.
3751    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
3752        match &self.info {
3753            DatasetInfo::Writer { index, .. } => {
3754                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3755                let inner = borrow_inner(&self.file_inner);
3756                match &*inner {
3757                    H5FileInner::Writer(writer) => {
3758                        writer.set_dataset_extent(*index, &dims_u64)?;
3759                        Ok(())
3760                    }
3761                    _ => Err(Hdf5Error::InvalidState(
3762                        "file is no longer in write mode".into(),
3763                    )),
3764                }
3765            }
3766            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
3767                "cannot set extent in read mode".into(),
3768            )),
3769        }
3770    }
3771
3772    /// Flush a chunked dataset's index structures to disk.
3773    pub fn flush(&self) -> Result<()> {
3774        match &self.info {
3775            DatasetInfo::Writer { index, .. } => {
3776                let inner = borrow_inner(&self.file_inner);
3777                match &*inner {
3778                    H5FileInner::Writer(writer) => {
3779                        writer.flush_dataset(*index)?;
3780                        Ok(())
3781                    }
3782                    _ => Ok(()),
3783                }
3784            }
3785            DatasetInfo::Reader { .. } => Ok(()),
3786        }
3787    }
3788
3789    /// Read a slice (hyperslab) of the dataset as a typed vector.
3790    ///
3791    /// `starts` and `counts` define the N-dimensional selection:
3792    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
3793    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
3794        match &self.info {
3795            DatasetInfo::Reader {
3796                name,
3797                shape,
3798                element_size,
3799            } => {
3800                if T::element_size() != *element_size {
3801                    return Err(Hdf5Error::TypeMismatch(format!(
3802                        "read type has element size {} but dataset has element size {}",
3803                        T::element_size(),
3804                        element_size,
3805                    )));
3806                }
3807                let datatype = self.datatype()?;
3808                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
3809                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
3810
3811                // Bounds before sizing: the destination is allocated here,
3812                // ahead of the reader's own check, so a selection the extent
3813                // does not admit must be refused before its size is computed.
3814                let dims: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
3815                check_hyperslab(&dims, &starts_u64, &counts_u64)?;
3816                let count = element_count(&counts_u64)?;
3817                let mut inner = borrow_inner_mut(&self.file_inner);
3818                let H5FileInner::Reader(reader) = &mut *inner else {
3819                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3820                };
3821                // The selection lands in the vector this returns, so its bytes
3822                // are touched once instead of being read into a byte buffer and
3823                // copied into a second one of the same size.
3824                read_image_into_new(count, |image| {
3825                    reader.read_slice_into_dst(
3826                        name,
3827                        &starts_u64,
3828                        &counts_u64,
3829                        image,
3830                        ReadDst::Fresh,
3831                    )?;
3832                    to_host_byte_order(image, &datatype, T::element_size())
3833                })
3834            }
3835            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3836                "cannot read_slice from a dataset in write mode".into(),
3837            )),
3838        }
3839    }
3840
3841    /// Read a strided hyperslab as a typed vector — h5py's stepped slicing
3842    /// (`ds[a:b:s]`) or the general `start`/`stride`/`count`/`block` form of
3843    /// `H5Sselect_hyperslab`.
3844    ///
3845    /// One entry per dimension: `start[d]` is the first index, `stride[d]`
3846    /// the spacing between selected blocks (all-`1` is the same selection
3847    /// [`read_slice`](Self::read_slice) reads), `count[d]` how many blocks,
3848    /// and `block[d]` how many contiguous elements each block covers. The
3849    /// returned vector is row-major over `count[d] * block[d]` per
3850    /// dimension — exactly the shape h5py's stepped slicing produces.
3851    ///
3852    /// ```no_run
3853    /// # use rust_hdf5::H5File;
3854    /// let file = H5File::open("data.h5").unwrap();
3855    /// let ds = file.dataset("series").unwrap(); // shape [100]
3856    /// // Python: ds[0:100:2] — every other element.
3857    /// let evens: Vec<f64> = ds.read_hyperslab(&[0], &[2], &[50], &[1]).unwrap();
3858    /// ```
3859    pub fn read_hyperslab<T: H5Type>(
3860        &self,
3861        start: &[usize],
3862        stride: &[usize],
3863        count: &[usize],
3864        block: &[usize],
3865    ) -> Result<Vec<T>> {
3866        match &self.info {
3867            DatasetInfo::Reader {
3868                name, element_size, ..
3869            } => {
3870                if T::element_size() != *element_size {
3871                    return Err(Hdf5Error::TypeMismatch(format!(
3872                        "read type has element size {} but dataset has element size {}",
3873                        T::element_size(),
3874                        element_size,
3875                    )));
3876                }
3877                let datatype = self.datatype()?;
3878                let start_u64: Vec<u64> = start.iter().map(|&s| s as u64).collect();
3879                let stride_u64: Vec<u64> = stride.iter().map(|&s| s as u64).collect();
3880                let count_u64: Vec<u64> = count.iter().map(|&c| c as u64).collect();
3881                let block_u64: Vec<u64> = block.iter().map(|&b| b as u64).collect();
3882
3883                let selected: Vec<u64> = count_u64
3884                    .iter()
3885                    .zip(&block_u64)
3886                    .map(|(&c, &b)| c.saturating_mul(b))
3887                    .collect();
3888                let n = element_count(&selected)?;
3889                let mut inner = borrow_inner_mut(&self.file_inner);
3890                let H5FileInner::Reader(reader) = &mut *inner else {
3891                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3892                };
3893                read_image_into_new(n, |image| {
3894                    reader.read_hyperslab_into(
3895                        name,
3896                        &start_u64,
3897                        &stride_u64,
3898                        &count_u64,
3899                        &block_u64,
3900                        image,
3901                    )?;
3902                    to_host_byte_order(image, &datatype, T::element_size())
3903                })
3904            }
3905            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3906                "cannot read_hyperslab from a dataset in write mode".into(),
3907            )),
3908        }
3909    }
3910
3911    /// Read a list of coordinates in one call, as a typed vector — h5py
3912    /// fancy indexing with a coordinate list.
3913    ///
3914    /// `points[i]` is a coordinate with one entry per dimension. The
3915    /// returned vector holds one element per point, in the same order as
3916    /// `points`, regardless of the dataset's rank.
3917    ///
3918    /// ```no_run
3919    /// # use rust_hdf5::H5File;
3920    /// let file = H5File::open("data.h5").unwrap();
3921    /// let ds = file.dataset("grid").unwrap(); // shape [10, 10]
3922    /// // Python: ds[np.array([[0, 0], [3, 4], [9, 9]])]
3923    /// let picked: Vec<f64> = ds.read_points(&[vec![0, 0], vec![3, 4], vec![9, 9]]).unwrap();
3924    /// ```
3925    pub fn read_points<T: H5Type>(&self, points: &[Vec<usize>]) -> Result<Vec<T>> {
3926        match &self.info {
3927            DatasetInfo::Reader {
3928                name, element_size, ..
3929            } => {
3930                if T::element_size() != *element_size {
3931                    return Err(Hdf5Error::TypeMismatch(format!(
3932                        "read type has element size {} but dataset has element size {}",
3933                        T::element_size(),
3934                        element_size,
3935                    )));
3936                }
3937                let datatype = self.datatype()?;
3938                let points_u64: Vec<Vec<u64>> = points
3939                    .iter()
3940                    .map(|p| p.iter().map(|&c| c as u64).collect())
3941                    .collect();
3942
3943                let mut inner = borrow_inner_mut(&self.file_inner);
3944                let H5FileInner::Reader(reader) = &mut *inner else {
3945                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3946                };
3947                read_image_into_new(points_u64.len(), |image| {
3948                    reader.read_points_into(name, &points_u64, image, ReadDst::Fresh)?;
3949                    to_host_byte_order(image, &datatype, T::element_size())
3950                })
3951            }
3952            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3953                "cannot read_points from a dataset in write mode".into(),
3954            )),
3955        }
3956    }
3957
3958    /// Read one chunk's raw (still-filtered) bytes and its filter mask,
3959    /// addressed by chunk-grid coordinates — the read half of
3960    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) and the HDF5 "direct
3961    /// chunk read" (`H5Dread_chunk`, formerly `H5DOread_chunk`; h5py's
3962    /// `Dataset.id.read_direct_chunk`).
3963    ///
3964    /// The bytes are exactly what is stored on disk: filtered/compressed if
3965    /// the dataset has a filter pipeline, with no decompression applied. The
3966    /// returned `u32` is the chunk's filter mask: bit *i* set means filter
3967    /// *i* of the pipeline was **not** applied to this particular chunk and
3968    /// must be skipped when reversing it.
3969    ///
3970    /// `Err` if the dataset is not chunked, `chunk_coords` has the wrong
3971    /// rank, or the chunk at those coordinates has never been written.
3972    ///
3973    /// ```no_run
3974    /// # use rust_hdf5::H5File;
3975    /// let file = H5File::open("data.h5").unwrap();
3976    /// let ds = file.dataset("frames").unwrap();
3977    /// let (raw, filter_mask) = ds.read_chunk_raw_at(&[0, 0]).unwrap();
3978    /// ```
3979    pub fn read_chunk_raw_at(&self, chunk_coords: &[usize]) -> Result<(Vec<u8>, u32)> {
3980        match &self.info {
3981            DatasetInfo::Reader { name, .. } => {
3982                let coords_u64: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3983                let mut inner = borrow_inner_mut(&self.file_inner);
3984                match &mut *inner {
3985                    H5FileInner::Reader(reader) => Ok(reader.read_chunk_raw_at(name, &coords_u64)?),
3986                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
3987                }
3988            }
3989            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3990                "cannot read_chunk_raw_at from a dataset in write mode".into(),
3991            )),
3992        }
3993    }
3994
3995    /// Write a typed slice to a sub-region of the dataset.
3996    ///
3997    /// `starts` and `counts` define the N-dimensional selection, which must lie
3998    /// inside the dataset's current extent.
3999    ///
4000    /// Works for both contiguous and chunked datasets. For a chunked dataset
4001    /// only the chunks the selection touches are rewritten — a partially
4002    /// covered chunk is read back, patched, and written again, so updating one
4003    /// row of an appendable dataset costs the chunks that row crosses rather
4004    /// than the whole dataset. Elements of a touched chunk that the selection
4005    /// does not cover keep their stored value, or the dataset's fill value if
4006    /// the chunk did not exist yet.
4007    pub fn write_slice<T: H5Type>(
4008        &self,
4009        starts: &[usize],
4010        counts: &[usize],
4011        data: &[T],
4012    ) -> Result<()> {
4013        match &self.info {
4014            DatasetInfo::Writer {
4015                index,
4016                element_size,
4017                ..
4018            } => {
4019                if T::element_size() != *element_size {
4020                    return Err(Hdf5Error::TypeMismatch(format!(
4021                        "write type has element size {} but dataset expects {}",
4022                        T::element_size(),
4023                        element_size,
4024                    )));
4025                }
4026
4027                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4028                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4029
4030                let inner = borrow_inner(&self.file_inner);
4031                let H5FileInner::Writer(writer) = &*inner else {
4032                    return Err(Hdf5Error::InvalidState(
4033                        "file is no longer in write mode".into(),
4034                    ));
4035                };
4036
4037                // Bounds before sizing, against the extent the writer holds
4038                // now (an extend since this handle was taken counts): a
4039                // selection the extent does not admit is refused for that
4040                // reason, not for the size it would have had.
4041                check_hyperslab(&writer.dataset_dims(*index), &starts_u64, &counts_u64)?;
4042                let expected = element_count(&counts_u64)?;
4043                if data.len() != expected {
4044                    return Err(Hdf5Error::InvalidState(format!(
4045                        "data length {} does not match slice size {}",
4046                        data.len(),
4047                        expected,
4048                    )));
4049                }
4050
4051                let byte_len = data.len() * T::element_size();
4052                let host =
4053                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
4054
4055                let datatype = writer.dataset_datatype(*index);
4056                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
4057                writer.write_slice(*index, &starts_u64, &counts_u64, &stored)?;
4058                Ok(())
4059            }
4060            DatasetInfo::Reader { .. } => {
4061                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4062            }
4063        }
4064    }
4065
4066    /// Replace elements `start .. start + strings.len()` of a 1-D
4067    /// variable-length string dataset.
4068    ///
4069    /// The extent and every element outside the range are left alone, and the
4070    /// cost is the new strings plus the chunks holding their references — not
4071    /// the column. The dataset's character set is enforced: a non-ASCII
4072    /// replacement in a dataset that declares ASCII is rejected rather than
4073    /// stored under a datatype that misdescribes it.
4074    ///
4075    /// The global heap objects the replaced references pointed at are freed —
4076    /// the same reclaim libhdf5 performs on an overwrite — so updating one
4077    /// element repeatedly reuses space rather than growing the file. A
4078    /// collection emptied by the update returns its block to the allocator.
4079    /// Under SWMR nothing is freed, because a reader may still be following
4080    /// those references.
4081    ///
4082    /// ```no_run
4083    /// # use rust_hdf5::H5File;
4084    /// let file = H5File::open_rw("meta.h5").unwrap();
4085    /// let ds = file.dataset_writer("notes").unwrap();
4086    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
4087    /// file.close().unwrap();
4088    /// ```
4089    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
4090        match &self.info {
4091            DatasetInfo::Writer { index, .. } => {
4092                let inner = borrow_inner(&self.file_inner);
4093                match &*inner {
4094                    H5FileInner::Writer(writer) => {
4095                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
4096                        Ok(())
4097                    }
4098                    _ => Err(Hdf5Error::InvalidState(
4099                        "file is no longer in write mode".into(),
4100                    )),
4101                }
4102            }
4103            DatasetInfo::Reader { .. } => {
4104                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4105            }
4106        }
4107    }
4108
4109    /// Read variable-length strings from a dataset.
4110    ///
4111    /// This handles h5py-style vlen string datasets that store strings
4112    /// as global heap references. Returns one String per element.
4113    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
4114        match &self.info {
4115            DatasetInfo::Reader { name, .. } => {
4116                let mut inner = borrow_inner_mut(&self.file_inner);
4117                match &mut *inner {
4118                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
4119                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4120                }
4121            }
4122            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4123                "cannot read vlen strings from a dataset in write mode".into(),
4124            )),
4125        }
4126    }
4127
4128    /// Read variable-length byte arrays from a dataset.
4129    ///
4130    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
4131    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
4132    /// that store each element as a global heap reference. Returns one
4133    /// `Vec<u8>` per element.
4134    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
4135        match &self.info {
4136            DatasetInfo::Reader { name, .. } => {
4137                let mut inner = borrow_inner_mut(&self.file_inner);
4138                match &mut *inner {
4139                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
4140                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4141                }
4142            }
4143            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4144                "cannot read vlen bytes from a dataset in write mode".into(),
4145            )),
4146        }
4147    }
4148
4149    /// Write object references naming `paths` into elements `0..paths.len()`
4150    /// — h5py's `refs[i] = f['/target'].ref`.
4151    ///
4152    /// The dataset must have been created with
4153    /// [`object_references`](DatasetBuilder::object_references). A path names
4154    /// a dataset or a group (`/` is the root group) and must already exist;
4155    /// what reaches the file is the target's object header address, which is
4156    /// assigned when the file is finalized. Elements left unwritten read back
4157    /// as null references.
4158    pub fn write_object_references(&self, paths: &[&str]) -> Result<()> {
4159        match &self.info {
4160            DatasetInfo::Writer { index, .. } => {
4161                let inner = borrow_inner(&self.file_inner);
4162                match &*inner {
4163                    H5FileInner::Writer(writer) => {
4164                        writer.write_object_references(*index, 0, paths)?;
4165                        Ok(())
4166                    }
4167                    _ => Err(Hdf5Error::InvalidState(
4168                        "file is no longer in write mode".into(),
4169                    )),
4170                }
4171            }
4172            DatasetInfo::Reader { .. } => {
4173                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4174            }
4175        }
4176    }
4177
4178    /// Write region references over `targets` into elements
4179    /// `0..targets.len()` — h5py's `refs[i] = f['/target'].regionref[0:3]`.
4180    ///
4181    /// The dataset must have been created with
4182    /// [`region_references`](DatasetBuilder::region_references). Each target is
4183    /// the path of an existing *dataset* and a [`Selection`] over it, which
4184    /// must fit that dataset's extent — the rule `H5Rcreate` applies. What
4185    /// reaches the file is a global-heap object holding the target's object
4186    /// header address (assigned when the file is finalized) and the serialized
4187    /// selection. Elements left unwritten read back as null references.
4188    ///
4189    /// ```no_run
4190    /// # use rust_hdf5::{H5File, PointSelection, Selection};
4191    /// let file = H5File::create("regions.h5").unwrap();
4192    /// file.new_dataset::<i32>().shape([4, 6]).create("m").unwrap();
4193    /// let refs = file.new_dataset::<u64>()
4194    ///     .region_references()
4195    ///     .shape([1])
4196    ///     .create("refs")
4197    ///     .unwrap();
4198    /// let points = Selection::Points(PointSelection {
4199    ///     rank: 2,
4200    ///     points: vec![vec![0, 1], vec![3, 5]],
4201    /// });
4202    /// refs.write_region_references(&[("/m", points)]).unwrap();
4203    /// file.close().unwrap();
4204    /// ```
4205    pub fn write_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4206        match &self.info {
4207            DatasetInfo::Writer { index, .. } => {
4208                let inner = borrow_inner(&self.file_inner);
4209                match &*inner {
4210                    H5FileInner::Writer(writer) => {
4211                        writer.write_region_references(*index, 0, targets)?;
4212                        Ok(())
4213                    }
4214                    _ => Err(Hdf5Error::InvalidState(
4215                        "file is no longer in write mode".into(),
4216                    )),
4217                }
4218            }
4219            DatasetInfo::Reader { .. } => {
4220                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4221            }
4222        }
4223    }
4224
4225    /// Write revised region references over `targets` into elements
4226    /// `0..targets.len()` — `H5Rcreate_region` plus `H5Dwrite` of an
4227    /// `H5T_STD_REF` dataset.
4228    ///
4229    /// The dataset must have been created with
4230    /// [`std_region_references`](DatasetBuilder::std_region_references) or one
4231    /// of its two siblings, which make the same datatype. Each target is the
4232    /// path of an existing *dataset* and a [`Selection`] over it, which must
4233    /// fit that dataset's extent. What reaches the file is a global-heap blob
4234    /// holding the target's object header address (assigned when the file is
4235    /// finalized) and the serialized selection, and an element carrying the
4236    /// blob's id and its byte count. Elements left unwritten read back as null
4237    /// references.
4238    pub fn write_std_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4239        let targets: Vec<(&str, ReferenceTarget)> = targets
4240            .iter()
4241            .map(|(path, selection)| (*path, ReferenceTarget::Region(selection.clone())))
4242            .collect();
4243        self.write_revised_references(&targets)
4244    }
4245
4246    /// Write attribute references naming `targets` into elements
4247    /// `0..targets.len()` — `H5Rcreate_attr` plus `H5Dwrite` of an
4248    /// `H5T_STD_REF` dataset.
4249    ///
4250    /// Each target is the path of an existing object — a dataset, a group, or
4251    /// `/` for the root group — and the name of an attribute it already
4252    /// carries. There is no pre-1.12 form of this reference kind, so the
4253    /// dataset must have been created with
4254    /// [`attribute_references`](DatasetBuilder::attribute_references) or one of
4255    /// its two siblings. Elements left unwritten read back as null references.
4256    pub fn write_attribute_references(&self, targets: &[(&str, &str)]) -> Result<()> {
4257        let targets: Vec<(&str, ReferenceTarget)> = targets
4258            .iter()
4259            .map(|(path, name)| (*path, ReferenceTarget::Attribute((*name).to_string())))
4260            .collect();
4261        self.write_revised_references(&targets)
4262    }
4263
4264    /// Store `targets` as 1.12 reference elements, whatever mix of kinds they
4265    /// are: the one path both revised-reference writers take.
4266    fn write_revised_references(&self, targets: &[(&str, ReferenceTarget)]) -> Result<()> {
4267        match &self.info {
4268            DatasetInfo::Writer { index, .. } => {
4269                let inner = borrow_inner(&self.file_inner);
4270                match &*inner {
4271                    H5FileInner::Writer(writer) => {
4272                        writer.write_revised_references(*index, 0, targets)?;
4273                        Ok(())
4274                    }
4275                    _ => Err(Hdf5Error::InvalidState(
4276                        "file is no longer in write mode".into(),
4277                    )),
4278                }
4279            }
4280            DatasetInfo::Reader { .. } => {
4281                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4282            }
4283        }
4284    }
4285
4286    /// Read a reference dataset's elements, each resolved to the object it
4287    /// names.
4288    ///
4289    /// Every reference kind is read: the pre-1.12 pair h5py writes —
4290    /// `Reference` (an object header address) and `RegionReference` (a heap id
4291    /// whose heap object holds the target plus a serialized selection) — and
4292    /// the 1.12 `H5T_STD_REF` trio, `H5R_OBJECT2`, `H5R_DATASET_REGION2` and
4293    /// `H5R_ATTR`. An object reference comes back as [`Reference::Object`]
4294    /// carrying the target's path, a region reference as
4295    /// [`Reference::Region`], whose [`bounds`](Reference::bounds) is the
4296    /// selection's bounding box — libhdf5's `H5Sget_select_bounds` — and an
4297    /// attribute reference as [`Reference::Attr`], which adds the attribute's
4298    /// name.
4299    ///
4300    /// A 1.12 reference written into a file other than its target's carries
4301    /// that file's name, and [`Reference::file`] reports it; the path is then
4302    /// a path inside that file, resolved by opening it under the name the
4303    /// reference carries, and `None` when nothing is there.
4304    ///
4305    /// ```no_run
4306    /// # use rust_hdf5::H5File;
4307    /// let file = H5File::open("refs.h5").unwrap();
4308    /// for r in file.dataset("refs").unwrap().read_references().unwrap() {
4309    ///     println!("{:?} {:?}", r.path(), r.bounds());
4310    /// }
4311    /// ```
4312    pub fn read_references(&self) -> Result<Vec<Reference>> {
4313        match &self.info {
4314            DatasetInfo::Reader { name, .. } => {
4315                let mut inner = borrow_inner_mut(&self.file_inner);
4316                match &mut *inner {
4317                    H5FileInner::Reader(reader) => Ok(reader.read_references(name)?),
4318                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4319                }
4320            }
4321            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4322                "cannot read references from a dataset in write mode".into(),
4323            )),
4324        }
4325    }
4326
4327    /// Read a string dataset, fixed-width or variable-length, as one `String`
4328    /// per element.
4329    ///
4330    /// The width of a `FixedString` dataset is whatever the file says, so a
4331    /// 24-byte label column and a 100-byte one are read by the same call. The
4332    /// padding rule the datatype declares decides where each element ends —
4333    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
4334    /// character set decides how the remaining bytes are decoded: ASCII (0)
4335    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
4336    /// violates either is an error naming the element, not a silent
4337    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
4338    /// call that accepts such a file, replacing what it cannot decode.
4339    ///
4340    /// ```no_run
4341    /// # use rust_hdf5::H5File;
4342    /// let file = H5File::open("labels.h5").unwrap();
4343    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
4344    /// ```
4345    pub fn read_strings(&self) -> Result<Vec<String>> {
4346        self.read_strings_inner(false)
4347    }
4348
4349    /// [`read_strings`](Self::read_strings), but bytes that do not decode
4350    /// under the dataset's character set become U+FFFD instead of an error.
4351    ///
4352    /// Producers do mislabel the character set — a file that declares ASCII
4353    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
4354    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
4355        self.read_strings_inner(true)
4356    }
4357
4358    /// The single owner of string decoding for both string datatypes: the
4359    /// element bytes are found differently, the padding and character-set
4360    /// rules that turn them into a `String` are the same.
4361    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
4362        if matches!(self.info, DatasetInfo::Writer { .. }) {
4363            return Err(Hdf5Error::InvalidState(
4364                "cannot read strings from a dataset in write mode".into(),
4365            ));
4366        }
4367        match self.datatype()? {
4368            DatatypeMessage::VarLenString { charset, .. } => self
4369                .read_vlen_bytes()?
4370                .iter()
4371                .enumerate()
4372                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
4373                .collect(),
4374            DatatypeMessage::FixedString {
4375                size,
4376                padding,
4377                charset,
4378            } => {
4379                let width = size as usize;
4380                if width == 0 {
4381                    // A corrupt file can declare it; `chunks_exact(0)` panics.
4382                    return Err(Hdf5Error::InvalidState(
4383                        "fixed-string datatype has zero width".into(),
4384                    ));
4385                }
4386                // `read_raw_bytes` returns `product(dims) * width` bytes, so
4387                // `chunks_exact` leaves no remainder.
4388                let raw = self.read_raw_bytes()?;
4389                raw.chunks_exact(width)
4390                    .enumerate()
4391                    .map(|(i, elem)| {
4392                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
4393                    })
4394                    .collect()
4395            }
4396            other => Err(Hdf5Error::InvalidState(format!(
4397                "read_strings is only for string datasets, this one is {other:?}"
4398            ))),
4399        }
4400    }
4401
4402    /// Read the entire dataset as a typed vector.
4403    ///
4404    /// The raw bytes are read from the file and reinterpreted as `T`. The
4405    /// caller must ensure that `T` matches the datatype used when the dataset
4406    /// was written.
4407    ///
4408    /// # Errors
4409    ///
4410    /// Returns an error if:
4411    /// - The file is in write mode.
4412    /// - The raw data size is not a multiple of `T::element_size()`.
4413    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
4414        match &self.info {
4415            DatasetInfo::Reader {
4416                name, element_size, ..
4417            } => {
4418                if T::element_size() != *element_size {
4419                    return Err(Hdf5Error::TypeMismatch(format!(
4420                        "read type has element size {} but dataset has element size {}",
4421                        T::element_size(),
4422                        element_size,
4423                    )));
4424                }
4425
4426                let datatype = self.datatype()?;
4427                let mut inner = borrow_inner_mut(&self.file_inner);
4428                let H5FileInner::Reader(reader) = &mut *inner else {
4429                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4430                };
4431                let total = reader.dataset_raw_size(name)? as usize;
4432                if !total.is_multiple_of(T::element_size()) {
4433                    return Err(Hdf5Error::TypeMismatch(format!(
4434                        "raw data size {total} is not a multiple of element size {}",
4435                        T::element_size(),
4436                    )));
4437                }
4438
4439                // The image is read into the vector this returns, so the
4440                // bytes are touched once rather than being zeroed, read, and
4441                // then copied into a second buffer of the same size.
4442                read_image_into_new(total / T::element_size(), |image| {
4443                    reader.read_dataset_raw_into_dst(name, image, ReadDst::Fresh)?;
4444                    to_host_byte_order(image, &datatype, T::element_size())
4445                })
4446            }
4447            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4448                "cannot read from a dataset in write mode".into(),
4449            )),
4450        }
4451    }
4452
4453    /// View the entire dataset as `&[T]` pointing straight into the file's
4454    /// memory map — no read, no copy, no allocation.
4455    ///
4456    /// The returned [`MappedView<T>`](crate::MappedView) dereferences to
4457    /// `&[T]` holding exactly what [`read_raw`](Self::read_raw) would have
4458    /// returned, bit for bit.
4459    ///
4460    /// # When it works
4461    ///
4462    /// The file must be open read-only and mapped (which a read-only open
4463    /// does whenever the OS allows), the dataset's raw data must be one
4464    /// contiguous stretch of that file, and its stored elements must already
4465    /// be the host image of a `T` — same width, host byte order, significant
4466    /// bits filling the element. That is the ordinary case for an
4467    /// uncompressed, non-chunked numeric dataset written by either this crate
4468    /// or libhdf5.
4469    ///
4470    /// # When it refuses
4471    ///
4472    /// Zero-copy is a contract, not an optimization: when the bytes cannot be
4473    /// handed over as they lie, this returns
4474    /// [`Hdf5Error::NotViewable`] naming the
4475    /// reason and never quietly falls back to copying. Every
4476    /// [`ViewRefusal`](crate::ViewRefusal) is a case where
4477    /// [`read_raw`](Self::read_raw) still works: the file is not mapped (an
4478    /// open that holds no shared lock never maps), the layout is chunked,
4479    /// compact, virtual, or external, no storage is
4480    /// allocated (the dataset reads as its fill value), `T` is the wrong
4481    /// width, the stored elements need a byte-order swap or bit unpacking,
4482    /// the data lands at an offset `T`'s alignment does not permit, or the
4483    /// image runs past the end of the map.
4484    ///
4485    /// # Snapshot semantics
4486    ///
4487    /// The view owns a share of the map rather than borrowing the file
4488    /// handle, so it stays readable after the dataset and the file are
4489    /// dropped, and after a SWMR refresh has retaken the map — a live view
4490    /// keeps showing the file as it was when *its* map was taken, while the
4491    /// refreshed handle reads the new one. Nothing about a view is
4492    /// invalidated by anything this process does. The share carries the
4493    /// shared file lock the map was taken under, so for as long as any view
4494    /// is alive a writer that honours locks cannot open the file, whether or
4495    /// not the reader that took the map is still open.
4496    ///
4497    /// # Truncation
4498    ///
4499    /// The pages are the file's own. Another process writing the file in
4500    /// place is seen through the view, and one *truncating* it under the map
4501    /// faults with `SIGBUS` on the pages that went away. The shared lock the
4502    /// view keeps is what stands between the map and such a writer; one that
4503    /// waives locks ([`FileLocking::Disabled`](crate::FileLocking::Disabled),
4504    /// or a filesystem without them) is outside what any guard inside this
4505    /// process can see.
4506    ///
4507    /// ```no_run
4508    /// # use rust_hdf5::H5File;
4509    /// let file = H5File::open("data.h5")?;
4510    /// let ds = file.dataset("matrix")?;
4511    /// let view = ds.read_mapped::<f64>()?;
4512    /// let total: f64 = view.iter().sum();
4513    /// # Ok::<(), rust_hdf5::Hdf5Error>(())
4514    /// ```
4515    #[cfg(feature = "mmap")]
4516    pub fn read_mapped<T: H5Type>(&self) -> Result<crate::mapped::MappedView<T>> {
4517        self.mapped_view(crate::mapped::ViewRange::Whole)
4518    }
4519
4520    /// View a contiguous sub-range of the dataset as `&[T]` pointing straight
4521    /// into the file's memory map.
4522    ///
4523    /// `starts` and `counts` name the same N-dimensional selection
4524    /// [`read_slice`](Self::read_slice) takes, and the view holds exactly what
4525    /// that call would have returned — but only when the selection is one
4526    /// contiguous run of the stored image: a trailing group of dimensions
4527    /// taken whole, the dimension before it taken as one span, and a single
4528    /// index along every dimension before that. Anything else steps over
4529    /// elements a single slice cannot skip, and is refused with
4530    /// [`ViewRefusal::Range`](crate::ViewRefusal::Range) rather than gathered
4531    /// into a copy.
4532    ///
4533    /// Everything [`read_mapped`](Self::read_mapped) documents about when a
4534    /// dataset can be viewed, snapshot semantics, and truncation applies here
4535    /// unchanged.
4536    #[cfg(feature = "mmap")]
4537    pub fn read_mapped_slice<T: H5Type>(
4538        &self,
4539        starts: &[usize],
4540        counts: &[usize],
4541    ) -> Result<crate::mapped::MappedView<T>> {
4542        self.mapped_view(crate::mapped::ViewRange::Slab { starts, counts })
4543    }
4544
4545    /// The one route from a dataset handle to the file's map: ask the reader
4546    /// that owns the dataset for the facts, and hand them to
4547    /// [`crate::mapped::view`], which is the only thing that can turn them
4548    /// into a view.
4549    #[cfg(feature = "mmap")]
4550    fn mapped_view<T: H5Type>(
4551        &self,
4552        range: crate::mapped::ViewRange<'_>,
4553    ) -> Result<crate::mapped::MappedView<T>> {
4554        let DatasetInfo::Reader { name, .. } = &self.info else {
4555            return Err(Hdf5Error::InvalidState(
4556                "cannot read from a dataset in write mode".into(),
4557            ));
4558        };
4559        let mut inner = borrow_inner_mut(&self.file_inner);
4560        let H5FileInner::Reader(reader) = &mut *inner else {
4561            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4562        };
4563        let src = reader.dataset_view_source(name)?;
4564        crate::mapped::view::<T>(&src, range).map_err(Hdf5Error::NotViewable)
4565    }
4566
4567    /// Read the raw byte image of a dataset without an `H5Type` carrier.
4568    ///
4569    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
4570    /// the element bytes verbatim regardless of the on-disk element type, so a
4571    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
4572    /// no matching Rust primitive can be read back and decoded by the caller.
4573    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
4574        match &self.info {
4575            DatasetInfo::Reader { name, .. } => {
4576                let mut inner = borrow_inner_mut(&self.file_inner);
4577                match &mut *inner {
4578                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
4579                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4580                }
4581            }
4582            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4583                "cannot read from a dataset in write mode".into(),
4584            )),
4585        }
4586    }
4587
4588    /// Read a numeric dataset as `T`, converting each element from the
4589    /// on-disk datatype.
4590    ///
4591    /// Unlike [`read_raw`](Self::read_raw), which requires `T`'s size to match
4592    /// the stored element size exactly, this inspects the dataset's datatype
4593    /// message — class, signedness, byte order, width — and converts per
4594    /// element:
4595    ///
4596    /// - integer → integer: checked; a stored value that does not fit in `T`
4597    ///   is an error naming the element index and value, never a silent wrap.
4598    /// - `f32` source → `f64`: exact widening.
4599    /// - `f64` source → `f32`, float → integer, and integer → float are
4600    ///   rejected as [`TypeMismatch`](Hdf5Error::TypeMismatch).
4601    ///
4602    /// Big-endian sources are decoded according to the datatype's byte order,
4603    /// which [`read_raw`](Self::read_raw)'s size-only check would misread.
4604    ///
4605    /// ```no_run
4606    /// # use rust_hdf5::H5File;
4607    /// let file = H5File::open("data.h5").unwrap();
4608    /// let ds = file.dataset("counts").unwrap(); // stored as e.g. i16
4609    /// let counts = ds.read_numeric_as::<i64>().unwrap();
4610    /// ```
4611    pub fn read_numeric_as<T: ReadNumeric>(&self) -> Result<Vec<T>> {
4612        match &self.info {
4613            DatasetInfo::Reader { name, .. } => {
4614                let (kind, raw) = {
4615                    let mut inner = borrow_inner_mut(&self.file_inner);
4616                    match &mut *inner {
4617                        H5FileInner::Reader(reader) => {
4618                            let info = reader
4619                                .dataset_info(name)
4620                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4621                            let kind = numeric::classify(&info.datatype)?;
4622                            (kind, reader.read_dataset_raw(name)?)
4623                        }
4624                        _ => {
4625                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4626                        }
4627                    }
4628                };
4629                numeric::convert(kind, &raw)
4630            }
4631            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4632                "cannot read from a dataset in write mode".into(),
4633            )),
4634        }
4635    }
4636
4637    /// Read a slice (hyperslab) of a numeric dataset as `T`, with the same
4638    /// per-element datatype conversion as
4639    /// [`read_numeric_as`](Self::read_numeric_as).
4640    ///
4641    /// `starts` and `counts` define the N-dimensional selection exactly as in
4642    /// [`read_slice`](Self::read_slice).
4643    pub fn read_numeric_slice_as<T: ReadNumeric>(
4644        &self,
4645        starts: &[usize],
4646        counts: &[usize],
4647    ) -> Result<Vec<T>> {
4648        match &self.info {
4649            DatasetInfo::Reader { name, .. } => {
4650                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4651                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4652                let (kind, raw) = {
4653                    let mut inner = borrow_inner_mut(&self.file_inner);
4654                    match &mut *inner {
4655                        H5FileInner::Reader(reader) => {
4656                            let info = reader
4657                                .dataset_info(name)
4658                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4659                            let kind = numeric::classify(&info.datatype)?;
4660                            (kind, reader.read_slice(name, &starts_u64, &counts_u64)?)
4661                        }
4662                        _ => {
4663                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4664                        }
4665                    }
4666                };
4667                numeric::convert(kind, &raw)
4668            }
4669            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4670                "cannot read_slice from a dataset in write mode".into(),
4671            )),
4672        }
4673    }
4674
4675    /// Read the whole dataset into a caller-provided buffer, with no allocation.
4676    ///
4677    /// `out` must have exactly `product(dims)` elements (the dataset's element
4678    /// count) and `T::element_size()` must match the dataset's on-disk element
4679    /// size, otherwise an error is returned and `out` is left unspecified. The
4680    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
4681    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
4682    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
4683    /// without the extra staging copy a `read_raw` + copy-into-pinned would
4684    /// incur. Works for every layout (contiguous, compact, and chunked under
4685    /// any index); for chunked data each decoded chunk is scattered directly
4686    /// into `out`.
4687    ///
4688    /// ```no_run
4689    /// # use rust_hdf5::H5File;
4690    /// let file = H5File::open("data.h5").unwrap();
4691    /// let ds = file.dataset("frames").unwrap();
4692    /// let n: usize = ds.shape().iter().product();
4693    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
4694    /// ds.read_raw_into(&mut buf).unwrap();
4695    /// ```
4696    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
4697        match &self.info {
4698            DatasetInfo::Reader {
4699                name, element_size, ..
4700            } => {
4701                if T::element_size() != *element_size {
4702                    return Err(Hdf5Error::TypeMismatch(format!(
4703                        "read type has element size {} but dataset has element size {}",
4704                        T::element_size(),
4705                        element_size,
4706                    )));
4707                }
4708                let datatype = self.datatype()?;
4709                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
4710                // byte representation; every bit pattern the read writes is a
4711                // valid `T`. The byte view borrows `out` exclusively for this
4712                // call, and `out.len() * element_size` cannot overflow because
4713                // it is the byte length of an existing slice (<= isize::MAX).
4714                let byte_len = out.len() * T::element_size();
4715                let bytes = unsafe {
4716                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4717                };
4718                {
4719                    let mut inner = borrow_inner_mut(&self.file_inner);
4720                    match &mut *inner {
4721                        H5FileInner::Reader(reader) => reader.read_dataset_raw_into(name, bytes)?,
4722                        _ => {
4723                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4724                        }
4725                    }
4726                }
4727                to_host_byte_order(bytes, &datatype, T::element_size())
4728            }
4729            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4730                "cannot read from a dataset in write mode".into(),
4731            )),
4732        }
4733    }
4734
4735    /// Read a hyperslab into a caller-provided buffer, with no allocation.
4736    ///
4737    /// `out` must have exactly `product(counts)` elements and
4738    /// `T::element_size()` must match the dataset's element size. The zero-copy
4739    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
4740    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
4741    /// selection are read, and the selected bytes land directly in `out` — the
4742    /// entry point for reading one frame / block straight into a pinned host
4743    /// buffer for an H2D transfer.
4744    ///
4745    /// ```no_run
4746    /// # use rust_hdf5::H5File;
4747    /// let file = H5File::open("vol.h5").unwrap();
4748    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
4749    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
4750    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
4751    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
4752    /// ```
4753    pub fn read_slice_into<T: H5Type>(
4754        &self,
4755        out: &mut [T],
4756        starts: &[usize],
4757        counts: &[usize],
4758    ) -> Result<()> {
4759        match &self.info {
4760            DatasetInfo::Reader {
4761                name, element_size, ..
4762            } => {
4763                if T::element_size() != *element_size {
4764                    return Err(Hdf5Error::TypeMismatch(format!(
4765                        "read type has element size {} but dataset has element size {}",
4766                        T::element_size(),
4767                        element_size,
4768                    )));
4769                }
4770                let datatype = self.datatype()?;
4771                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4772                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4773                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
4774                // borrow of `out`, byte length within bounds.
4775                let byte_len = out.len() * T::element_size();
4776                let bytes = unsafe {
4777                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4778                };
4779                {
4780                    let mut inner = borrow_inner_mut(&self.file_inner);
4781                    match &mut *inner {
4782                        H5FileInner::Reader(reader) => {
4783                            reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?
4784                        }
4785                        _ => {
4786                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4787                        }
4788                    }
4789                }
4790                to_host_byte_order(bytes, &datatype, T::element_size())
4791            }
4792            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4793                "cannot read from a dataset in write mode".into(),
4794            )),
4795        }
4796    }
4797}
4798
4799// ---------------------------------------------------------------------------
4800// Datatype-aware numeric conversion (read_numeric_as)
4801// ---------------------------------------------------------------------------
4802
4803/// Marker trait for the Rust types [`H5Dataset::read_numeric_as`] can convert
4804/// into: the integer primitives (checked, never wrapping) plus `f32`/`f64`
4805/// (widening only).
4806///
4807/// Sealed — the conversion policy is part of the library contract, so the
4808/// trait cannot be implemented outside this crate.
4809pub trait ReadNumeric: numeric::Sealed {}
4810impl<T: numeric::Sealed> ReadNumeric for T {}
4811
4812pub(crate) mod numeric {
4813    //! Per-element decode + checked conversion for `read_numeric_as` (and the
4814    //! attribute counterpart `H5Attribute::read_numeric_as`).
4815
4816    use crate::error::{Hdf5Error, Result};
4817    use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, IeeeFormat};
4818
4819    /// A source element, normalized: every standard integer width — u64::MAX
4820    /// included — fits in `i128` without loss.
4821    pub enum NumericSource {
4822        Int(i128),
4823        F32(f32),
4824        F64(f64),
4825    }
4826
4827    /// The on-disk element shape `classify` accepted.
4828    #[derive(Clone, Copy)]
4829    pub enum SourceKind {
4830        Int {
4831            size: usize,
4832            signed: bool,
4833            byte_order: ByteOrder,
4834        },
4835        F16(ByteOrder),
4836        F32(ByteOrder),
4837        F64(ByteOrder),
4838    }
4839
4840    impl SourceKind {
4841        fn element_size(self) -> usize {
4842            match self {
4843                SourceKind::Int { size, .. } => size,
4844                SourceKind::F16(_) => 2,
4845                SourceKind::F32(_) => 4,
4846                SourceKind::F64(_) => 8,
4847            }
4848        }
4849    }
4850
4851    /// Widen an IEEE 754 binary16 bit pattern to `f32`, which represents every
4852    /// half — including subnormals, infinities and NaN payloads — exactly.
4853    ///
4854    /// Rust has no stable `f16` to convert through.
4855    fn f16_bits_to_f32(bits: u16) -> f32 {
4856        let sign = u32::from(bits >> 15);
4857        let exponent = u32::from((bits >> 10) & 0x1f);
4858        let mantissa = u32::from(bits & 0x03ff);
4859        if exponent == 0 {
4860            // Zero and subnormals: the value is mantissa * 2^-24, which is a
4861            // normal f32 for every mantissa, so the multiply is exact. Going
4862            // through the sign separately keeps -0.0.
4863            let magnitude = mantissa as f32 * (1.0 / 16_777_216.0);
4864            return if sign == 1 { -magnitude } else { magnitude };
4865        }
4866        let out = if exponent == 0x1f {
4867            // Infinity and NaN; shifting the mantissa maps the quiet bit onto
4868            // f32's quiet bit and preserves the rest of the payload.
4869            (sign << 31) | 0x7f80_0000 | (mantissa << 13)
4870        } else {
4871            // Normal: rebias the exponent (127 - 15) and left-align the
4872            // mantissa.
4873            (sign << 31) | ((exponent + 112) << 23) | (mantissa << 13)
4874        };
4875        f32::from_bits(out)
4876    }
4877
4878    /// Map a datatype message to a supported numeric source shape.
4879    ///
4880    /// Accepts standard-width integers (1/2/4/8 bytes, full precision, zero
4881    /// bit offset) and IEEE binary32/binary64 floats; everything else is a
4882    /// `TypeMismatch` naming what was found.
4883    pub fn classify(dt: &DatatypeMessage) -> Result<SourceKind> {
4884        match *dt {
4885            DatatypeMessage::FixedPoint {
4886                size,
4887                byte_order,
4888                signed,
4889                bit_offset,
4890                bit_precision,
4891            } => {
4892                if !matches!(size, 1 | 2 | 4 | 8)
4893                    || bit_offset != 0
4894                    || u32::from(bit_precision) != size * 8
4895                {
4896                    return Err(Hdf5Error::TypeMismatch(format!(
4897                        "fixed-point datatype (size {size}, bit offset {bit_offset}, \
4898                         precision {bit_precision}) is not a standard-width integer",
4899                    )));
4900                }
4901                Ok(SourceKind::Int {
4902                    size: size as usize,
4903                    signed,
4904                    byte_order,
4905                })
4906            }
4907            DatatypeMessage::BitField {
4908                size,
4909                byte_order,
4910                bit_offset,
4911                bit_precision,
4912            } => {
4913                // A bit field has no signed form; a full-width one is the
4914                // unsigned integer of the stored width. A narrower one would
4915                // need a shift-and-mask conversion this path does not model.
4916                if !matches!(size, 1 | 2 | 4 | 8)
4917                    || bit_offset != 0
4918                    || u32::from(bit_precision) != size * 8
4919                {
4920                    return Err(Hdf5Error::TypeMismatch(format!(
4921                        "bit-field datatype (size {size}, bit offset {bit_offset}, \
4922                         precision {bit_precision}) is not a whole-width bit field",
4923                    )));
4924                }
4925                Ok(SourceKind::Int {
4926                    size: size as usize,
4927                    signed: false,
4928                    byte_order,
4929                })
4930            }
4931            DatatypeMessage::FloatingPoint {
4932                size,
4933                byte_order,
4934                exponent_size,
4935                mantissa_size,
4936                ..
4937            } => match dt.ieee_format() {
4938                Some(IeeeFormat::Binary16) => Ok(SourceKind::F16(byte_order)),
4939                Some(IeeeFormat::Binary32) => Ok(SourceKind::F32(byte_order)),
4940                Some(IeeeFormat::Binary64) => Ok(SourceKind::F64(byte_order)),
4941                None => Err(Hdf5Error::TypeMismatch(format!(
4942                    "floating-point datatype (size {size}, exponent {exponent_size} bits, \
4943                     mantissa {mantissa_size} bits) is not an IEEE 754 interchange format",
4944                ))),
4945            },
4946            ref other => Err(Hdf5Error::TypeMismatch(format!(
4947                "dataset datatype '{other}' is not numeric",
4948            ))),
4949        }
4950    }
4951
4952    fn decode_element(kind: SourceKind, bytes: &[u8]) -> NumericSource {
4953        match kind {
4954            SourceKind::Int {
4955                size,
4956                signed,
4957                byte_order,
4958            } => {
4959                let mut le = [0u8; 8];
4960                match byte_order {
4961                    ByteOrder::LittleEndian => le[..size].copy_from_slice(bytes),
4962                    ByteOrder::BigEndian => {
4963                        for (dst, src) in le[..size].iter_mut().zip(bytes.iter().rev()) {
4964                            *dst = *src;
4965                        }
4966                    }
4967                }
4968                let zero_extended = u64::from_le_bytes(le);
4969                let value = if signed {
4970                    // Arithmetic right shift sign-extends the low `size` bytes.
4971                    let shift = 64 - 8 * size as u32;
4972                    i128::from(((zero_extended as i64) << shift) >> shift)
4973                } else {
4974                    i128::from(zero_extended)
4975                };
4976                NumericSource::Int(value)
4977            }
4978            SourceKind::F16(byte_order) => {
4979                let arr: [u8; 2] = bytes.try_into().unwrap();
4980                let bits = match byte_order {
4981                    ByteOrder::LittleEndian => u16::from_le_bytes(arr),
4982                    ByteOrder::BigEndian => u16::from_be_bytes(arr),
4983                };
4984                NumericSource::F32(f16_bits_to_f32(bits))
4985            }
4986            SourceKind::F32(byte_order) => {
4987                let arr: [u8; 4] = bytes.try_into().unwrap();
4988                NumericSource::F32(match byte_order {
4989                    ByteOrder::LittleEndian => f32::from_le_bytes(arr),
4990                    ByteOrder::BigEndian => f32::from_be_bytes(arr),
4991                })
4992            }
4993            SourceKind::F64(byte_order) => {
4994                let arr: [u8; 8] = bytes.try_into().unwrap();
4995                NumericSource::F64(match byte_order {
4996                    ByteOrder::LittleEndian => f64::from_le_bytes(arr),
4997                    ByteOrder::BigEndian => f64::from_be_bytes(arr),
4998                })
4999            }
5000        }
5001    }
5002
5003    /// Decode and convert every element of `raw` into `T`.
5004    pub fn convert<T: Sealed>(kind: SourceKind, raw: &[u8]) -> Result<Vec<T>> {
5005        let size = kind.element_size();
5006        if !raw.len().is_multiple_of(size) {
5007            return Err(Hdf5Error::TypeMismatch(format!(
5008                "raw data size {} is not a multiple of element size {size}",
5009                raw.len(),
5010            )));
5011        }
5012        raw.chunks_exact(size)
5013            .enumerate()
5014            .map(|(index, bytes)| T::from_source(decode_element(kind, bytes), index))
5015            .collect()
5016    }
5017
5018    /// The sealed half of `ReadNumeric`: how one normalized source element
5019    /// becomes a `Self`, or a `TypeMismatch` explaining why it cannot.
5020    pub trait Sealed: Sized {
5021        fn from_source(src: NumericSource, index: usize) -> Result<Self>;
5022    }
5023
5024    macro_rules! int_targets {
5025        ($($t:ty),* $(,)?) => {$(
5026            impl Sealed for $t {
5027                fn from_source(src: NumericSource, index: usize) -> Result<Self> {
5028                    match src {
5029                        NumericSource::Int(v) => <$t>::try_from(v).map_err(|_| {
5030                            Hdf5Error::TypeMismatch(format!(
5031                                concat!(
5032                                    "value {} at element {} does not fit in ",
5033                                    stringify!($t),
5034                                ),
5035                                v, index,
5036                            ))
5037                        }),
5038                        NumericSource::F32(_) | NumericSource::F64(_) => {
5039                            Err(Hdf5Error::TypeMismatch(
5040                                concat!(
5041                                    "cannot read a floating-point dataset as ",
5042                                    stringify!($t),
5043                                    "; read as f64 and convert explicitly",
5044                                )
5045                                .into(),
5046                            ))
5047                        }
5048                    }
5049                }
5050            }
5051        )*};
5052    }
5053    int_targets!(i8, i16, i32, i64, u8, u16, u32, u64, u128);
5054
5055    // Not in the macro: `i128::try_from(i128)` is infallible, which trips
5056    // clippy::unnecessary_fallible_conversions.
5057    impl Sealed for i128 {
5058        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5059            match src {
5060                NumericSource::Int(v) => Ok(v),
5061                NumericSource::F32(_) | NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
5062                    "cannot read a floating-point dataset as i128; read as f64 and \
5063                     convert explicitly"
5064                        .into(),
5065                )),
5066            }
5067        }
5068    }
5069
5070    impl Sealed for f32 {
5071        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5072            match src {
5073                NumericSource::F32(v) => Ok(v),
5074                NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
5075                    "narrowing an f64 dataset to f32 loses precision; read as f64".into(),
5076                )),
5077                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
5078                    "cannot read an integer dataset as f32; read as an integer type and \
5079                     convert explicitly"
5080                        .into(),
5081                )),
5082            }
5083        }
5084    }
5085
5086    impl Sealed for f64 {
5087        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5088            match src {
5089                NumericSource::F64(v) => Ok(v),
5090                // Every f32 is exactly representable as f64.
5091                NumericSource::F32(v) => Ok(f64::from(v)),
5092                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
5093                    "cannot read an integer dataset as f64; integers above 2^53 lose \
5094                     precision — read as an integer type and convert explicitly"
5095                        .into(),
5096                )),
5097            }
5098        }
5099    }
5100
5101    #[cfg(test)]
5102    mod tests {
5103        use super::*;
5104
5105        /// Every binary16 bit pattern class widens to the f32 with the same
5106        /// value: zeros keep their sign, subnormals stay exact, infinities and
5107        /// NaN payloads survive.
5108        #[test]
5109        fn f16_widening_is_exact() {
5110            let cases: [(u16, f32); 10] = [
5111                (0x0000, 0.0),
5112                (0x3c00, 1.0),
5113                (0xc000, -2.0),
5114                (0x3555, 0.333_251_95),   // nearest half to 1/3
5115                (0x0001, 5.960_464_5e-8), // smallest subnormal, 2^-24
5116                (0x03ff, 6.097_555e-5),   // largest subnormal
5117                (0x0400, 6.103_515_6e-5), // smallest normal
5118                (0x7bff, 65504.0),        // largest finite
5119                (0x7c00, f32::INFINITY),
5120                (0xfc00, f32::NEG_INFINITY),
5121            ];
5122            for (bits, expected) in cases {
5123                let got = f16_bits_to_f32(bits);
5124                assert_eq!(got, expected, "0x{bits:04x} widened to {got}");
5125            }
5126
5127            let neg_zero = f16_bits_to_f32(0x8000);
5128            assert_eq!(neg_zero, 0.0);
5129            assert!(neg_zero.is_sign_negative(), "-0.0 lost its sign");
5130
5131            let nan = f16_bits_to_f32(0x7e01);
5132            assert!(nan.is_nan());
5133            // The quiet bit and the payload land in f32's mantissa.
5134            assert_eq!(nan.to_bits(), 0x7fc0_2000);
5135        }
5136
5137        #[test]
5138        fn f16_source_converts_and_honors_byte_order() {
5139            let kind = classify(&DatatypeMessage::f16_type()).unwrap();
5140            // 1.0, -2.0, 0.333..., 65504
5141            let raw = [0x00, 0x3c, 0x00, 0xc0, 0x55, 0x35, 0xff, 0x7b];
5142            assert_eq!(
5143                convert::<f32>(kind, &raw).unwrap(),
5144                vec![1.0, -2.0, 0.333_251_95, 65504.0]
5145            );
5146            assert_eq!(
5147                convert::<f64>(kind, &raw).unwrap(),
5148                vec![1.0, -2.0, 0.333_251_953_125, 65504.0]
5149            );
5150
5151            let DatatypeMessage::FloatingPoint { .. } = DatatypeMessage::f16_type() else {
5152                unreachable!()
5153            };
5154            let mut be = DatatypeMessage::f16_type();
5155            if let DatatypeMessage::FloatingPoint { byte_order, .. } = &mut be {
5156                *byte_order = ByteOrder::BigEndian;
5157            }
5158            let be_kind = classify(&be).unwrap();
5159            assert_eq!(convert::<f32>(be_kind, &[0x3c, 0x00]).unwrap(), vec![1.0]);
5160        }
5161
5162        /// A float whose layout is not an interchange format is refused, not
5163        /// reinterpreted.
5164        #[test]
5165        fn non_ieee_float_is_refused() {
5166            let mut odd = DatatypeMessage::f32_type();
5167            if let DatatypeMessage::FloatingPoint { exponent_bias, .. } = &mut odd {
5168                *exponent_bias = 63;
5169            }
5170            assert!(odd.ieee_format().is_none());
5171            let err = classify(&odd).err().expect("non-IEEE float was accepted");
5172            assert!(
5173                err.to_string().contains("IEEE 754 interchange format"),
5174                "unexpected error: {err}"
5175            );
5176        }
5177    }
5178}
5179
5180#[cfg(test)]
5181mod tests {
5182    use crate::H5File;
5183    use std::path::PathBuf;
5184
5185    fn temp_path(name: &str) -> PathBuf {
5186        // Include PID + a per-call atomic counter so that concurrent
5187        // cargo invocations and any kernel-level "lock not yet
5188        // released" races between sequential opens cannot collide.
5189        use std::sync::atomic::{AtomicU64, Ordering};
5190        static COUNTER: AtomicU64 = AtomicU64::new(0);
5191        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
5192        std::env::temp_dir().join(format!(
5193            "hdf5_dataset_test_{}_{}_{}.h5",
5194            name,
5195            std::process::id(),
5196            n
5197        ))
5198    }
5199
5200    #[test]
5201    fn runtime_compound_via_datatype_override_and_raw_bytes() {
5202        use crate::format::messages::datatype::DatatypeMessage;
5203        use crate::types::{CompoundType, H5Type};
5204
5205        let path = temp_path("compound_raw");
5206        // A 12-byte packed compound with NO matching Rust primitive carrier,
5207        // so it can only be written through the datatype() override +
5208        // write_raw_bytes path (the runtime-CompoundType use case).
5209        let ct = CompoundType {
5210            members: vec![
5211                ("id".to_string(), i32::hdf5_type(), 0),
5212                ("val".to_string(), f64::hdf5_type(), 4),
5213            ],
5214            total_size: 12,
5215        };
5216        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
5217        let mut bytes = Vec::new();
5218        for (id, val) in recs {
5219            bytes.extend_from_slice(&id.to_le_bytes());
5220            bytes.extend_from_slice(&val.to_le_bytes());
5221        }
5222
5223        {
5224            let file = H5File::create(&path).unwrap();
5225            let ds = file
5226                .new_dataset::<u8>()
5227                .datatype(ct.to_datatype())
5228                .shape([recs.len()])
5229                .create("records")
5230                .unwrap();
5231            ds.write_raw_bytes(&bytes).unwrap();
5232            file.close().unwrap();
5233        }
5234        {
5235            let file = H5File::open(&path).unwrap();
5236            let ds = file.dataset("records").unwrap();
5237            // The on-disk element type is the compound we specified (size 12),
5238            // not the u8 carrier.
5239            match ds.datatype().unwrap() {
5240                DatatypeMessage::Compound { size, members } => {
5241                    assert_eq!(size, 12);
5242                    assert_eq!(members.len(), 2);
5243                    assert_eq!(members[0].name, "id");
5244                    assert_eq!(members[0].offset, 0);
5245                    assert_eq!(members[1].name, "val");
5246                    assert_eq!(members[1].offset, 4);
5247                }
5248                other => panic!("expected compound datatype, got {other:?}"),
5249            }
5250            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
5251        }
5252        std::fs::remove_file(&path).ok();
5253    }
5254
5255    #[test]
5256    fn builder_requires_shape() {
5257        let path = temp_path("no_shape");
5258        let file = H5File::create(&path).unwrap();
5259        let result = file.new_dataset::<u8>().create("data");
5260        assert!(result.is_err());
5261        std::fs::remove_file(&path).ok();
5262    }
5263
5264    // The last chunk along a *fixed* dimension covers more elements than the
5265    // extent has, so growing the dataspace to the chunk's far edge asks for
5266    // more than the declared maximum. Before the clamp the chunk was written
5267    // and then the call failed on that extend, leaving the bytes in the file
5268    // and the caller an error.
5269    #[test]
5270    fn a_partial_edge_chunk_does_not_grow_past_the_declared_maximum() {
5271        let path = temp_path("edge_chunk_extent");
5272        // Extensible array: dimension 1 is unlimited, dimension 0 is fixed at
5273        // 10 and not a multiple of the chunk's 4.
5274        let file = H5File::create(&path).unwrap();
5275        let ds = file
5276            .new_dataset::<i32>()
5277            .shape([10usize, 4])
5278            .max_shape(&[Some(10), None])
5279            .chunk(&[4, 4])
5280            .create("grid")
5281            .unwrap();
5282        let chunk: Vec<u8> = (0i32..16).flat_map(|v| v.to_le_bytes()).collect();
5283        // Chunk row 2 spans elements 8..12 of a dimension that stops at 10.
5284        ds.write_chunk_at(&[2, 0], &chunk).unwrap();
5285        assert_eq!(ds.shape(), vec![10, 4]);
5286        file.close().unwrap();
5287
5288        let file = H5File::open(&path).unwrap();
5289        let back = file.dataset("grid").unwrap().read_raw::<i32>().unwrap();
5290        assert_eq!(back.len(), 40);
5291        assert_eq!(&back[32..40], &[0, 1, 2, 3, 4, 5, 6, 7]);
5292        drop(file);
5293        std::fs::remove_file(&path).ok();
5294    }
5295
5296    // The cap is in `write_chunk_at_inner`, so it belongs to every chunk index
5297    // whose write reaches the extend below it — the v2 B-tree as much as the
5298    // extensible array. A rank-3 dataset with two unlimited dimensions gets
5299    // that index, and its third, fixed dimension is where the last chunk
5300    // overhangs. (The fixed array and the implicit index return before the
5301    // extend: their shape cannot grow at all. The version-1 B-tree does reach
5302    // it — `tests/legacy_append.rs` carries that case, which needs a classic
5303    // file.)
5304    #[test]
5305    fn the_edge_write_cap_holds_for_the_v2_btree_index() {
5306        let path = temp_path("edge_chunk_bt2");
5307        let file = H5File::create(&path).unwrap();
5308        let ds = file
5309            .new_dataset::<i32>()
5310            .shape([10usize, 4, 4])
5311            .max_shape(&[Some(10), None, None])
5312            .chunk(&[4, 4, 4])
5313            .create("cube")
5314            .unwrap();
5315        let chunk: Vec<u8> = (0i32..64).flat_map(|v| v.to_le_bytes()).collect();
5316        // Chunk plane 2 spans elements 8..12 of a dimension that stops at 10.
5317        ds.write_chunk_at(&[2, 0, 0], &chunk).unwrap();
5318        assert_eq!(ds.shape(), vec![10, 4, 4]);
5319        file.close().unwrap();
5320
5321        let file = H5File::open(&path).unwrap();
5322        let back = file.dataset("cube").unwrap().read_raw::<i32>().unwrap();
5323        assert_eq!(back.len(), 160);
5324        // Rows 8 and 9 of the written plane, 16 elements each.
5325        assert_eq!(&back[128..160], &(0i32..32).collect::<Vec<_>>()[..]);
5326        drop(file);
5327        std::fs::remove_file(&path).ok();
5328    }
5329
5330    // The cap guards a chunk-coordinate write, and neither an externally
5331    // stored nor a virtual dataset has chunk coordinates to guard: both are
5332    // contiguous storage classes, refused at build together with chunked
5333    // storage, and `write_chunk_at` refuses what is not chunked. So the path
5334    // the case above exercises cannot be entered for either — asserted here
5335    // rather than left to inspection, since both classes route their raw bytes
5336    // through the same writer as the chunk grid does.
5337    #[test]
5338    fn an_external_or_virtual_dataset_never_reaches_the_edge_write_cap() {
5339        use crate::Selection;
5340        let dir = std::env::temp_dir().join(format!(
5341            "rust_hdf5_edge_cap_{}_{}",
5342            std::process::id(),
5343            temp_path("x").file_name().unwrap().to_string_lossy()
5344        ));
5345        std::fs::create_dir_all(&dir).unwrap();
5346        let path = dir.join("edge_cap.h5");
5347        let file = H5File::create(&path).unwrap();
5348        let payload = dir.join("payload.raw");
5349
5350        // Chunked storage and these two are mutually exclusive at build.
5351        for (which, res) in [
5352            (
5353                "external",
5354                file.new_dataset::<i32>()
5355                    .shape([10usize])
5356                    .external(&[(payload.to_str().unwrap(), 0, 40)])
5357                    .chunk(&[4])
5358                    .create("a"),
5359            ),
5360            (
5361                "virtual",
5362                file.new_dataset::<i32>()
5363                    .shape([10usize])
5364                    .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5365                    .chunk(&[4])
5366                    .create("b"),
5367            ),
5368        ] {
5369            match res {
5370                Ok(_) => panic!("a {which} dataset cannot also be chunked"),
5371                Err(e) => assert!(e.to_string().contains("chunked"), "{which}: {e}"),
5372            }
5373        }
5374
5375        // And the coordinate write itself is refused on both, with the extent
5376        // left exactly where it was.
5377        let ext = file
5378            .new_dataset::<i32>()
5379            .shape([10usize])
5380            .external(&[(payload.to_str().unwrap(), 0, 40)])
5381            .create("outside")
5382            .unwrap();
5383        let vds = file
5384            .new_dataset::<i32>()
5385            .shape([10usize])
5386            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5387            .create("elsewhere")
5388            .unwrap();
5389        let chunk: Vec<u8> = (0i32..4).flat_map(|v| v.to_le_bytes()).collect();
5390        for (which, ds) in [("external", &ext), ("virtual", &vds)] {
5391            let err = ds.write_chunk_at(&[2], &chunk).unwrap_err().to_string();
5392            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5393            let err = ds
5394                .write_chunk_raw_at(&[2], &chunk, 0)
5395                .unwrap_err()
5396                .to_string();
5397            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5398            assert_eq!(ds.shape(), vec![10], "{which}");
5399        }
5400        file.close().unwrap();
5401        std::fs::remove_dir_all(&dir).ok();
5402    }
5403
5404    #[test]
5405    fn write_raw_size_mismatch() {
5406        let path = temp_path("size_mismatch");
5407        let file = H5File::create(&path).unwrap();
5408        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
5409        // Provide 3 elements instead of 4
5410        let result = ds.write_raw(&[1u8, 2, 3]);
5411        assert!(result.is_err());
5412        std::fs::remove_file(&path).ok();
5413    }
5414
5415    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
5416    // dataset = one chunk) rather than silently drop the filter on the
5417    // contiguous path. write_raw then populates that single chunk.
5418    #[cfg(feature = "deflate")]
5419    #[test]
5420    fn filter_without_chunk_autochunks_and_roundtrips() {
5421        let path = temp_path("autochunk_filter");
5422        let data: Vec<i32> = (0..8).collect();
5423        {
5424            let file = H5File::create(&path).unwrap();
5425            let ds = file
5426                .new_dataset::<i32>()
5427                .deflate(6)
5428                .shape([8])
5429                .create("seq")
5430                .unwrap();
5431            ds.write_raw(&data).unwrap();
5432            file.close().unwrap();
5433        }
5434        {
5435            let file = H5File::open(&path).unwrap();
5436            let ds = file.dataset("seq").unwrap();
5437            // The filter forced chunked storage: a single whole-dataset chunk.
5438            assert!(
5439                ds.is_chunked(),
5440                "auto-chunk did not produce chunked storage"
5441            );
5442            assert_eq!(ds.chunk_dims(), Some(vec![8]));
5443            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5444        }
5445        std::fs::remove_file(&path).ok();
5446    }
5447
5448    // B: write_raw on an explicitly chunked + compressed dataset scatters the
5449    // full row-major image across a multi-chunk grid, including edge chunks
5450    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
5451    #[cfg(feature = "deflate")]
5452    #[test]
5453    fn an_edge_chunk_pads_with_zero_not_the_chunk_written_before_it() {
5454        // 4x5 over a 2x3 chunk: the second chunk of each row covers only two
5455        // of its three columns, so a third of it is padding. The image write
5456        // stages every chunk of a shape like this through one reused buffer,
5457        // and the padding has to reach the file as zero rather than as
5458        // whatever the chunk before it left in that buffer.
5459        let path = temp_path("edge_chunk_padding");
5460        let data: Vec<i32> = (1..=20).collect(); // no zeros of its own
5461        {
5462            let file = H5File::create(&path).unwrap();
5463            let ds = file
5464                .new_dataset::<i32>()
5465                .shape([4, 5])
5466                .chunk(&[2, 3])
5467                .create("grid")
5468                .unwrap();
5469            ds.write_raw(&data).unwrap();
5470            file.close().unwrap();
5471        }
5472        {
5473            let file = H5File::open(&path).unwrap();
5474            let ds = file.dataset("grid").unwrap();
5475            let as_i32 = |bytes: Vec<u8>| -> Vec<i32> {
5476                bytes
5477                    .as_chunks::<4>()
5478                    .0
5479                    .iter()
5480                    .map(|b| i32::from_le_bytes(*b))
5481                    .collect()
5482            };
5483            // The chunk that precedes each edge chunk is full, so a leak would
5484            // show as its 3rd and 6th elements (3 and 8, then 13 and 18).
5485            assert_eq!(
5486                as_i32(ds.read_chunk_raw_at(&[0, 0]).unwrap().0),
5487                [1, 2, 3, 6, 7, 8]
5488            );
5489            assert_eq!(
5490                as_i32(ds.read_chunk_raw_at(&[0, 1]).unwrap().0),
5491                [4, 5, 0, 9, 10, 0]
5492            );
5493            assert_eq!(
5494                as_i32(ds.read_chunk_raw_at(&[1, 1]).unwrap().0),
5495                [14, 15, 0, 19, 20, 0]
5496            );
5497            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5498        }
5499        std::fs::remove_file(&path).ok();
5500    }
5501
5502    #[test]
5503    #[cfg(feature = "deflate")]
5504    fn write_raw_multichunk_edge_roundtrips() {
5505        let path = temp_path("multichunk_edge");
5506        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
5507        {
5508            let file = H5File::create(&path).unwrap();
5509            let ds = file
5510                .new_dataset::<i32>()
5511                .shape([7, 5])
5512                .chunk(&[3, 2])
5513                .deflate(4)
5514                .create("grid")
5515                .unwrap();
5516            ds.write_raw(&data).unwrap();
5517            file.close().unwrap();
5518        }
5519        {
5520            let file = H5File::open(&path).unwrap();
5521            let ds = file.dataset("grid").unwrap();
5522            assert_eq!(ds.shape(), vec![7, 5]);
5523            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5524            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5525        }
5526        std::fs::remove_file(&path).ok();
5527    }
5528
5529    // B: write_raw on an unfiltered chunked dataset (previously rejected with
5530    // "use write_chunk for chunked datasets") now gathers and round-trips.
5531    #[test]
5532    fn write_raw_unfiltered_chunked_roundtrips() {
5533        let path = temp_path("chunked_unfiltered");
5534        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
5535        {
5536            let file = H5File::create(&path).unwrap();
5537            let ds = file
5538                .new_dataset::<f64>()
5539                .shape([4, 3])
5540                .chunk(&[2, 2])
5541                .create("m")
5542                .unwrap();
5543            ds.write_raw(&data).unwrap();
5544            file.close().unwrap();
5545        }
5546        {
5547            let file = H5File::open(&path).unwrap();
5548            let ds = file.dataset("m").unwrap();
5549            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
5550            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
5551        }
5552        std::fs::remove_file(&path).ok();
5553    }
5554
5555    // write_raw on an extensible-array (unlimited first dim) compressed dataset
5556    // drives write_full_image_chunked's EA branch, which gathers chunks and
5557    // compresses them through the windowed batch path. Round-trips the full
5558    // image, including a partial edge chunk along the unlimited dimension.
5559    #[cfg(feature = "deflate")]
5560    #[test]
5561    fn write_raw_ea_compressed_roundtrips() {
5562        let path = temp_path("write_raw_ea_deflate");
5563        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
5564        {
5565            let file = H5File::create(&path).unwrap();
5566            let ds = file
5567                .new_dataset::<i32>()
5568                .shape([5, 4])
5569                .chunk(&[2, 4])
5570                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
5571                .deflate(5)
5572                .create("stream")
5573                .unwrap();
5574            ds.write_raw(&data).unwrap();
5575            file.close().unwrap();
5576        }
5577        {
5578            let file = H5File::open(&path).unwrap();
5579            let ds = file.dataset("stream").unwrap();
5580            assert_eq!(ds.shape(), vec![5, 4]);
5581            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5582        }
5583        std::fs::remove_file(&path).ok();
5584    }
5585
5586    // 3D chunked Full read with partial-edge chunks in every dimension. This
5587    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
5588    // dimensions, exercising the nested outer-coordinate carry and the
5589    // last-axis edge clamp (chunks hang off the high edge in all three axes).
5590    #[cfg(feature = "deflate")]
5591    #[test]
5592    fn read_full_3d_chunked_edge_roundtrips() {
5593        let path = temp_path("full_3d_chunked_edge");
5594        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
5595        // along each axis is partial (1, 1, and 1 element respectively).
5596        let total: usize = 5 * 4 * 3;
5597        let data: Vec<i32> = (0..total as i32).collect();
5598        {
5599            let file = H5File::create(&path).unwrap();
5600            let ds = file
5601                .new_dataset::<i32>()
5602                .shape([5, 4, 3])
5603                .chunk(&[2, 3, 2])
5604                .deflate(4)
5605                .create("vol")
5606                .unwrap();
5607            ds.write_raw(&data).unwrap();
5608            file.close().unwrap();
5609        }
5610        {
5611            let file = H5File::open(&path).unwrap();
5612            let ds = file.dataset("vol").unwrap();
5613            assert_eq!(ds.shape(), vec![5, 4, 3]);
5614            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
5615            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5616        }
5617        std::fs::remove_file(&path).ok();
5618    }
5619
5620    #[test]
5621    fn roundtrip_u8_1d() {
5622        let path = temp_path("rt_u8_1d");
5623        let data: Vec<u8> = (0..10).collect();
5624
5625        {
5626            let file = H5File::create(&path).unwrap();
5627            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
5628            ds.write_raw(&data).unwrap();
5629            file.close().unwrap();
5630        }
5631
5632        {
5633            let file = H5File::open(&path).unwrap();
5634            let ds = file.dataset("seq").unwrap();
5635            assert_eq!(ds.shape(), vec![10]);
5636            let readback = ds.read_raw::<u8>().unwrap();
5637            assert_eq!(readback, data);
5638        }
5639
5640        std::fs::remove_file(&path).ok();
5641    }
5642
5643    #[test]
5644    fn roundtrip_i32_2d() {
5645        let path = temp_path("rt_i32_2d");
5646        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
5647
5648        {
5649            let file = H5File::create(&path).unwrap();
5650            let ds = file
5651                .new_dataset::<i32>()
5652                .shape([2, 3])
5653                .create("matrix")
5654                .unwrap();
5655            ds.write_raw(&data).unwrap();
5656            file.close().unwrap();
5657        }
5658
5659        {
5660            let file = H5File::open(&path).unwrap();
5661            let ds = file.dataset("matrix").unwrap();
5662            assert_eq!(ds.shape(), vec![2, 3]);
5663            let readback = ds.read_raw::<i32>().unwrap();
5664            assert_eq!(readback, data);
5665        }
5666
5667        std::fs::remove_file(&path).ok();
5668    }
5669
5670    #[test]
5671    fn roundtrip_f64_3d() {
5672        let path = temp_path("rt_f64_3d");
5673        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
5674
5675        {
5676            let file = H5File::create(&path).unwrap();
5677            let ds = file
5678                .new_dataset::<f64>()
5679                .shape([2, 3, 4])
5680                .create("cube")
5681                .unwrap();
5682            ds.write_raw(&data).unwrap();
5683            file.close().unwrap();
5684        }
5685
5686        {
5687            let file = H5File::open(&path).unwrap();
5688            let ds = file.dataset("cube").unwrap();
5689            assert_eq!(ds.shape(), vec![2, 3, 4]);
5690            let readback = ds.read_raw::<f64>().unwrap();
5691            assert_eq!(readback, data);
5692        }
5693
5694        std::fs::remove_file(&path).ok();
5695    }
5696
5697    #[test]
5698    fn cannot_read_in_write_mode() {
5699        let path = temp_path("no_read_write");
5700        let file = H5File::create(&path).unwrap();
5701        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5702        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5703        let result = ds.read_raw::<u8>();
5704        assert!(result.is_err());
5705        std::fs::remove_file(&path).ok();
5706    }
5707
5708    #[test]
5709    fn cannot_write_in_read_mode() {
5710        let path = temp_path("no_write_read");
5711
5712        {
5713            let file = H5File::create(&path).unwrap();
5714            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5715            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5716            file.close().unwrap();
5717        }
5718
5719        {
5720            let file = H5File::open(&path).unwrap();
5721            let ds = file.dataset("x").unwrap();
5722            let result = ds.write_raw(&[5u8, 6, 7, 8]);
5723            assert!(result.is_err());
5724        }
5725
5726        std::fs::remove_file(&path).ok();
5727    }
5728
5729    #[test]
5730    fn numeric_attr_roundtrip() {
5731        let path = temp_path("num_attr");
5732        {
5733            let file = H5File::create(&path).unwrap();
5734            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5735            ds.write_raw(&[1.0f32; 4]).unwrap();
5736
5737            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5738            a1.write_numeric(&1.2345f64).unwrap();
5739
5740            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
5741            a2.write_numeric(&42i32).unwrap();
5742
5743            file.close().unwrap();
5744        }
5745        {
5746            let file = H5File::open(&path).unwrap();
5747            let ds = file.dataset("data").unwrap();
5748
5749            let scale = ds.attr("scale").unwrap();
5750            let val: f64 = scale.read_numeric().unwrap();
5751            assert!((val - 1.2345).abs() < 1e-10);
5752
5753            let count = ds.attr("count").unwrap();
5754            let val: i32 = count.read_numeric().unwrap();
5755            assert_eq!(val, 42);
5756        }
5757        std::fs::remove_file(&path).ok();
5758    }
5759
5760    #[test]
5761    fn array_attr_roundtrip() {
5762        let path = temp_path("array_attr");
5763        let offsets = [10i32, -20, 30];
5764        {
5765            let file = H5File::create(&path).unwrap();
5766            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5767            ds.write_raw(&[1.0f32; 4]).unwrap();
5768
5769            // 1-D int32 array attribute (NDArrayDimOffset-style).
5770            let a = ds
5771                .new_attr::<i32>()
5772                .shape([3])
5773                .create("dim_offset")
5774                .unwrap();
5775            a.write_array(&offsets).unwrap();
5776
5777            // Wrong element count is rejected.
5778            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
5779            assert!(bad.write_array(&[1i32, 2]).is_err());
5780
5781            file.close().unwrap();
5782        }
5783        {
5784            let file = H5File::open(&path).unwrap();
5785            let ds = file.dataset("data").unwrap();
5786            let a = ds.attr("dim_offset").unwrap();
5787            let raw = a.read_raw().unwrap();
5788            assert_eq!(raw.len(), 3 * 4);
5789            let got: Vec<i32> = raw
5790                .as_chunks::<4>()
5791                .0
5792                .iter()
5793                .map(|b| i32::from_le_bytes(*b))
5794                .collect();
5795            assert_eq!(got, offsets);
5796        }
5797        std::fs::remove_file(&path).ok();
5798    }
5799
5800    #[test]
5801    fn attr_datatype_exposes_class_and_sign() {
5802        // H5Attribute::datatype() must report the stored datatype class and
5803        // signedness so a generic attr->metadata mapper need not infer it from
5804        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
5805        use crate::format::messages::datatype::DatatypeMessage;
5806
5807        let path = temp_path("attr_datatype");
5808        {
5809            let file = H5File::create(&path).unwrap();
5810            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5811            ds.new_attr::<f64>()
5812                .shape(())
5813                .create("scale")
5814                .unwrap()
5815                .write_numeric(&1.5f64)
5816                .unwrap();
5817            ds.new_attr::<i32>()
5818                .shape(())
5819                .create("count")
5820                .unwrap()
5821                .write_numeric(&7i32)
5822                .unwrap();
5823            file.close().unwrap();
5824        }
5825        {
5826            let file = H5File::open(&path).unwrap();
5827            let ds = file.dataset("data").unwrap();
5828
5829            match ds.attr("scale").unwrap().datatype().unwrap() {
5830                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
5831                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
5832            }
5833
5834            match ds.attr("count").unwrap().datatype().unwrap() {
5835                DatatypeMessage::FixedPoint { size, signed, .. } => {
5836                    assert_eq!(size, 4);
5837                    assert!(signed, "i32 attr must be signed");
5838                }
5839                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
5840            }
5841        }
5842        std::fs::remove_file(&path).ok();
5843    }
5844
5845    #[test]
5846    fn attr_datatype_in_write_mode_errors() {
5847        let path = temp_path("attr_datatype_write_mode");
5848        let file = H5File::create(&path).unwrap();
5849        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5850        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5851        assert!(attr.datatype().is_err());
5852        std::fs::remove_file(&path).ok();
5853    }
5854
5855    #[test]
5856    fn cannot_create_dataset_in_read_mode() {
5857        let path = temp_path("no_create_read");
5858
5859        {
5860            let _file = H5File::create(&path).unwrap();
5861        }
5862
5863        {
5864            let file = H5File::open(&path).unwrap();
5865            let result = file.new_dataset::<u8>().shape([4]).create("x");
5866            assert!(result.is_err());
5867        }
5868
5869        std::fs::remove_file(&path).ok();
5870    }
5871
5872    #[test]
5873    fn shape_accessor() {
5874        let path = temp_path("shape_acc");
5875
5876        let file = H5File::create(&path).unwrap();
5877        let ds = file
5878            .new_dataset::<f32>()
5879            .shape([5, 10, 3])
5880            .create("tensor")
5881            .unwrap();
5882        assert_eq!(ds.shape(), vec![5, 10, 3]);
5883
5884        std::fs::remove_file(&path).ok();
5885    }
5886
5887    #[test]
5888    fn slice_roundtrip_2d() {
5889        let path = temp_path("slice_2d");
5890
5891        // Create a 4x5 dataset, write full, then read a slice
5892        let data: Vec<i32> = (0..20).collect();
5893        {
5894            let file = H5File::create(&path).unwrap();
5895            let ds = file
5896                .new_dataset::<i32>()
5897                .shape([4, 5])
5898                .create("mat")
5899                .unwrap();
5900            ds.write_raw(&data).unwrap();
5901            file.close().unwrap();
5902        }
5903        {
5904            let file = H5File::open(&path).unwrap();
5905            let ds = file.dataset("mat").unwrap();
5906            // Read rows 1..3, cols 2..4 (2x2 slice)
5907            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
5908            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
5909            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
5910            assert_eq!(slice, vec![7, 8, 12, 13]);
5911        }
5912
5913        std::fs::remove_file(&path).ok();
5914    }
5915
5916    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
5917    // caller-provided buffer and MUST produce byte-for-byte the same data as
5918    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
5919    // creatable layout, since both now share one buffer-filling core.
5920    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
5921    where
5922        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
5923    {
5924        let n: usize = ds.shape().iter().product();
5925        let want_full = ds.read_raw::<T>().unwrap();
5926        let mut got_full = vec![T::default(); n];
5927        ds.read_raw_into::<T>(&mut got_full).unwrap();
5928        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
5929
5930        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
5931        let sn: usize = counts.iter().product();
5932        let mut got_slice = vec![T::default(); sn];
5933        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
5934            .unwrap();
5935        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
5936    }
5937
5938    #[test]
5939    fn read_into_matches_vec_contiguous() {
5940        let path = temp_path("into_contig");
5941        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
5942        {
5943            let file = H5File::create(&path).unwrap();
5944            let ds = file
5945                .new_dataset::<i32>()
5946                .shape([4, 5])
5947                .create("mat")
5948                .unwrap();
5949            ds.write_raw(&data).unwrap();
5950            file.close().unwrap();
5951        }
5952        {
5953            let file = H5File::open(&path).unwrap();
5954            let ds = file.dataset("mat").unwrap();
5955            assert_eq!(ds.chunk_dims(), None);
5956            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
5957        }
5958        std::fs::remove_file(&path).ok();
5959    }
5960
5961    #[test]
5962    fn read_into_matches_vec_chunked_unfiltered() {
5963        let path = temp_path("into_chunk");
5964        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
5965        {
5966            let file = H5File::create(&path).unwrap();
5967            let ds = file
5968                .new_dataset::<f64>()
5969                .shape([7, 5])
5970                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
5971                .create("grid")
5972                .unwrap();
5973            ds.write_raw(&data).unwrap();
5974            file.close().unwrap();
5975        }
5976        {
5977            let file = H5File::open(&path).unwrap();
5978            let ds = file.dataset("grid").unwrap();
5979            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5980            // Slice spans multiple chunks (rows 2..5, cols 1..4).
5981            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
5982        }
5983        std::fs::remove_file(&path).ok();
5984    }
5985
5986    #[test]
5987    fn read_into_matches_vec_single_chunk() {
5988        let path = temp_path("into_single_chunk");
5989        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
5990        {
5991            let file = H5File::create(&path).unwrap();
5992            let ds = file
5993                .new_dataset::<i32>()
5994                .shape([3, 4])
5995                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
5996                .create("g")
5997                .unwrap();
5998            ds.write_raw(&data).unwrap();
5999            file.close().unwrap();
6000        }
6001        {
6002            let file = H5File::open(&path).unwrap();
6003            let ds = file.dataset("g").unwrap();
6004            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
6005            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
6006        }
6007        std::fs::remove_file(&path).ok();
6008    }
6009
6010    #[cfg(feature = "deflate")]
6011    #[test]
6012    fn read_into_matches_vec_chunked_deflate() {
6013        let path = temp_path("into_chunk_deflate");
6014        let data: Vec<i32> = (0..35).collect(); // 7 x 5
6015        {
6016            let file = H5File::create(&path).unwrap();
6017            let ds = file
6018                .new_dataset::<i32>()
6019                .shape([7, 5])
6020                .chunk(&[3, 2])
6021                .deflate(4)
6022                .create("grid")
6023                .unwrap();
6024            ds.write_raw(&data).unwrap();
6025            file.close().unwrap();
6026        }
6027        {
6028            let file = H5File::open(&path).unwrap();
6029            let ds = file.dataset("grid").unwrap();
6030            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6031            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
6032        }
6033        std::fs::remove_file(&path).ok();
6034    }
6035
6036    #[test]
6037    fn read_into_wrong_buffer_size_rejected() {
6038        let path = temp_path("into_badlen");
6039        let data: Vec<i32> = (0..20).collect(); // 4 x 5
6040        {
6041            let file = H5File::create(&path).unwrap();
6042            let ds = file
6043                .new_dataset::<i32>()
6044                .shape([4, 5])
6045                .create("mat")
6046                .unwrap();
6047            ds.write_raw(&data).unwrap();
6048            file.close().unwrap();
6049        }
6050        {
6051            let file = H5File::open(&path).unwrap();
6052            let ds = file.dataset("mat").unwrap();
6053
6054            // Too small / too large full-read buffers are both rejected.
6055            let mut small = vec![0i32; 19];
6056            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
6057            let mut large = vec![0i32; 21];
6058            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
6059
6060            // Slice buffer must be exactly product(counts) = 4.
6061            let mut bad_slice = vec![0i32; 3];
6062            assert!(ds
6063                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
6064                .is_err());
6065            // The correctly sized slice buffer succeeds.
6066            let mut ok_slice = vec![0i32; 4];
6067            assert!(ds
6068                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
6069                .is_ok());
6070        }
6071        std::fs::remove_file(&path).ok();
6072    }
6073
6074    #[test]
6075    fn read_into_wrong_element_size_rejected() {
6076        let path = temp_path("into_badtype");
6077        let data: Vec<i32> = (0..20).collect(); // element size 4
6078        {
6079            let file = H5File::create(&path).unwrap();
6080            let ds = file
6081                .new_dataset::<i32>()
6082                .shape([4, 5])
6083                .create("mat")
6084                .unwrap();
6085            ds.write_raw(&data).unwrap();
6086            file.close().unwrap();
6087        }
6088        {
6089            let file = H5File::open(&path).unwrap();
6090            let ds = file.dataset("mat").unwrap();
6091            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
6092            // element size -> TypeMismatch, even with a "correctly sized" Vec.
6093            let mut as_u8 = vec![0u8; 20];
6094            assert!(matches!(
6095                ds.read_raw_into::<u8>(&mut as_u8),
6096                Err(crate::Hdf5Error::TypeMismatch(_))
6097            ));
6098            let mut as_i64 = vec![0i64; 20];
6099            assert!(matches!(
6100                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
6101                Err(crate::Hdf5Error::TypeMismatch(_))
6102            ));
6103        }
6104        std::fs::remove_file(&path).ok();
6105    }
6106
6107    #[test]
6108    fn write_slice_2d() {
6109        let path = temp_path("write_slice_2d");
6110
6111        {
6112            let file = H5File::create(&path).unwrap();
6113            let ds = file
6114                .new_dataset::<f32>()
6115                .shape([3, 4])
6116                .create("data")
6117                .unwrap();
6118            ds.write_raw(&[0.0f32; 12]).unwrap();
6119            // Overwrite a 2x2 sub-region
6120            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
6121                .unwrap();
6122            file.close().unwrap();
6123        }
6124        {
6125            let file = H5File::open(&path).unwrap();
6126            let ds = file.dataset("data").unwrap();
6127            let full = ds.read_raw::<f32>().unwrap();
6128            // Row 0: [0,0,0,0]
6129            // Row 1: [0,10,20,0]
6130            // Row 2: [0,30,40,0]
6131            assert_eq!(
6132                full,
6133                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
6134            );
6135        }
6136
6137        std::fs::remove_file(&path).ok();
6138    }
6139
6140    /// One 2x4 i32 chunk whose every element is `v`.
6141    fn chunk_of(v: i32) -> Vec<u8> {
6142        (0..8).flat_map(|_| v.to_le_bytes()).collect()
6143    }
6144
6145    /// Write chunk (0,0) `rewrites` times — each time with a different value,
6146    /// so no write can be skipped — and return the closed file's size along
6147    /// with what the chunk reads back as.
6148    fn rewrite_chunk(
6149        tag: &str,
6150        rewrites: i32,
6151        build: impl Fn(&H5File) -> crate::H5Dataset,
6152    ) -> (u64, i32) {
6153        let path = temp_path(tag);
6154        {
6155            let file = H5File::create(&path).unwrap();
6156            let ds = build(&file);
6157            for v in 1..=rewrites {
6158                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
6159            }
6160            file.close().unwrap();
6161        }
6162        let size = std::fs::metadata(&path).unwrap().len();
6163        let first = {
6164            let file = H5File::open(&path).unwrap();
6165            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
6166        };
6167        std::fs::remove_file(&path).ok();
6168        (size, first)
6169    }
6170
6171    // An unfiltered chunk's stored size is fixed by the chunk shape, so
6172    // rewriting it must overwrite the block it already occupies rather than
6173    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
6174    // leaves must_alloc false for exactly this case). The file must therefore
6175    // be byte-identical in size no matter how many times the chunk is written.
6176    #[test]
6177    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
6178        let build = |f: &H5File| {
6179            f.new_dataset::<i32>()
6180                .shape([2, 4])
6181                .chunk(&[2, 4])
6182                .max_shape(&[None, Some(4)])
6183                .create("d")
6184                .unwrap()
6185        };
6186        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
6187        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
6188        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6189        assert_eq!(last, 8, "the last write must be the one that survives");
6190    }
6191
6192    #[test]
6193    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
6194        let build = |f: &H5File| {
6195            f.new_dataset::<i32>()
6196                .shape([2, 4])
6197                .chunk(&[2, 4])
6198                .create("d")
6199                .unwrap()
6200        };
6201        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
6202        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
6203        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6204        assert_eq!(last, 8);
6205    }
6206
6207    #[test]
6208    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
6209        let build = |f: &H5File| {
6210            f.new_dataset::<i32>()
6211                .shape([2, 4])
6212                .chunk(&[2, 4])
6213                .max_shape(&[None, None])
6214                .create("d")
6215                .unwrap()
6216        };
6217        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
6218        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
6219        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6220        assert_eq!(last, 8);
6221    }
6222
6223    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
6224    // pool. Every node is the same size, so the blocks already on disk are
6225    // reused and repeated flushes cost nothing; sizing the root to its record
6226    // count instead would relocate it each time and orphan the block it left.
6227    #[test]
6228    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
6229        let flush_n = |label: &str, flushes: usize| -> u64 {
6230            let path = temp_path(label);
6231            {
6232                let file = H5File::create(&path).unwrap();
6233                let ds = file
6234                    .new_dataset::<i32>()
6235                    .shape([2, 4])
6236                    .chunk(&[2, 4])
6237                    .max_shape(&[None, None])
6238                    .create("d")
6239                    .unwrap();
6240                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
6241                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
6242                for _ in 0..flushes {
6243                    ds.flush().unwrap();
6244                }
6245                file.close().unwrap();
6246            }
6247            let size = std::fs::metadata(&path).unwrap().len();
6248            // The data must survive every rewrite of the index.
6249            {
6250                let file = H5File::open(&path).unwrap();
6251                let ds = file.dataset("d").unwrap();
6252                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
6253            }
6254            std::fs::remove_file(&path).ok();
6255            size
6256        };
6257        assert_eq!(
6258            flush_n("bt2_flush_8", 8),
6259            flush_n("bt2_flush_1", 1),
6260            "8 index flushes grew the file past a single one"
6261        );
6262    }
6263
6264    // A filtered chunk whose compressed size changes cannot stay put, so it
6265    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
6266    // H5MF_xfree). Alternating between two payloads of different compressed
6267    // size must therefore keep reusing the same two blocks instead of
6268    // appending a fresh one each time.
6269    #[cfg(feature = "deflate")]
6270    #[test]
6271    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
6272        // All-equal elements deflate to far fewer bytes than a varied payload,
6273        // so the two writes below land at different stored sizes.
6274        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
6275        let varied: Vec<u8> = (0..8i32)
6276            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
6277            .collect();
6278
6279        let sizes: Vec<u64> = [1usize, 8]
6280            .iter()
6281            .map(|&rounds| {
6282                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
6283                {
6284                    let file = H5File::create(&path).unwrap();
6285                    let ds = file
6286                        .new_dataset::<i32>()
6287                        .shape([2, 4])
6288                        .chunk(&[2, 4])
6289                        .max_shape(&[None, Some(4)])
6290                        .deflate(6)
6291                        .create("d")
6292                        .unwrap();
6293                    for _ in 0..rounds {
6294                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
6295                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
6296                    }
6297                    file.close().unwrap();
6298                }
6299                let size = std::fs::metadata(&path).unwrap().len();
6300                {
6301                    let file = H5File::open(&path).unwrap();
6302                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
6303                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
6304                    assert_eq!(got, want, "the last write must survive the round trip");
6305                }
6306                std::fs::remove_file(&path).ok();
6307                size
6308            })
6309            .collect();
6310
6311        assert_eq!(
6312            sizes[1], sizes[0],
6313            "8 alternating rewrites grew the file past a single pair"
6314        );
6315    }
6316
6317    #[test]
6318    fn write_slice_out_of_bounds_rejected() {
6319        let path = temp_path("write_slice_oob");
6320        let file = H5File::create(&path).unwrap();
6321        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
6322        ds.write_raw(&[0i32; 4]).unwrap();
6323        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
6324        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
6325        // An in-bounds slice still works.
6326        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
6327        std::fs::remove_file(&path).ok();
6328    }
6329
6330    #[test]
6331    fn duplicate_dataset_name_rejected() {
6332        let path = temp_path("dup_name");
6333        let file = H5File::create(&path).unwrap();
6334        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
6335        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
6336        std::fs::remove_file(&path).ok();
6337    }
6338
6339    #[test]
6340    fn extend_cannot_shrink() {
6341        let path = temp_path("extend_shrink");
6342        let file = H5File::create(&path).unwrap();
6343        let ds = file
6344            .new_dataset::<i32>()
6345            .shape([0])
6346            .chunk(&[2])
6347            .max_shape(&[None])
6348            .create("d")
6349            .unwrap();
6350        ds.append(&[1i32, 2, 3, 4]).unwrap();
6351        // Shrinking below the written extent must be rejected.
6352        assert!(ds.extend(&[2]).is_err());
6353        // Growing is fine.
6354        assert!(ds.extend(&[6]).is_ok());
6355        std::fs::remove_file(&path).ok();
6356    }
6357
6358    #[test]
6359    fn attr_read_roundtrip() {
6360        use crate::types::VarLenUnicode;
6361        let path = temp_path("attr_read");
6362
6363        {
6364            let file = H5File::create(&path).unwrap();
6365            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6366            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6367            let a1 = ds
6368                .new_attr::<VarLenUnicode>()
6369                .shape(())
6370                .create("units")
6371                .unwrap();
6372            a1.write_string("meters").unwrap();
6373            let a2 = ds
6374                .new_attr::<VarLenUnicode>()
6375                .shape(())
6376                .create("desc")
6377                .unwrap();
6378            a2.write_string("test data").unwrap();
6379            file.close().unwrap();
6380        }
6381        {
6382            let file = H5File::open(&path).unwrap();
6383            let ds = file.dataset("data").unwrap();
6384
6385            let names = ds.attr_names().unwrap();
6386            assert!(names.contains(&"units".to_string()));
6387            assert!(names.contains(&"desc".to_string()));
6388
6389            let units = ds.attr("units").unwrap();
6390            assert_eq!(units.read_string().unwrap(), "meters");
6391
6392            let desc = ds.attr("desc").unwrap();
6393            assert_eq!(desc.read_string().unwrap(), "test data");
6394        }
6395
6396        std::fs::remove_file(&path).ok();
6397    }
6398
6399    #[test]
6400    fn type_mismatch_element_size() {
6401        let path = temp_path("type_mismatch");
6402
6403        {
6404            let file = H5File::create(&path).unwrap();
6405            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
6406            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
6407            file.close().unwrap();
6408        }
6409
6410        {
6411            let file = H5File::open(&path).unwrap();
6412            let ds = file.dataset("data").unwrap();
6413            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
6414            let result = ds.read_raw::<u8>();
6415            assert!(result.is_err());
6416        }
6417
6418        std::fs::remove_file(&path).ok();
6419    }
6420
6421    #[test]
6422    fn dataset_survives_file_move() {
6423        let path = temp_path("ds_survives");
6424
6425        let ds = {
6426            let file = H5File::create(&path).unwrap();
6427            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
6428        };
6429        // file is dropped here, but ds still holds Rc to the inner state
6430        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6431        // The writer will finalize on drop of the last Rc
6432
6433        std::fs::remove_file(&path).ok();
6434    }
6435
6436    #[test]
6437    fn new_attr_scalar_string() {
6438        use crate::types::VarLenUnicode;
6439
6440        let path = temp_path("attr_scalar_string");
6441        {
6442            let file = H5File::create(&path).unwrap();
6443            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6444            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6445
6446            let attr = ds
6447                .new_attr::<VarLenUnicode>()
6448                .shape(())
6449                .create("name")
6450                .unwrap();
6451            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
6452                .unwrap();
6453
6454            file.close().unwrap();
6455        }
6456
6457        // Verify the file is still valid and readable
6458        {
6459            use crate::format::messages::datatype::DatatypeMessage;
6460            let file = H5File::open(&path).unwrap();
6461            let ds = file.dataset("data").unwrap();
6462            assert_eq!(ds.shape(), vec![4]);
6463            let readback = ds.read_raw::<u8>().unwrap();
6464            assert_eq!(readback, vec![1u8, 2, 3, 4]);
6465
6466            // The string attribute is stored as a true variable-length string
6467            // (not fixed-length) and round-trips its value.
6468            let attr = ds.attr("name").unwrap();
6469            assert!(
6470                matches!(
6471                    attr.datatype().unwrap(),
6472                    DatatypeMessage::VarLenString { .. }
6473                ),
6474                "string attribute should have a variable-length string datatype"
6475            );
6476            assert_eq!(attr.read_string().unwrap(), "test_value");
6477        }
6478
6479        std::fs::remove_file(&path).ok();
6480    }
6481
6482    #[test]
6483    fn all_numeric_types_roundtrip() {
6484        let path = temp_path("all_types");
6485
6486        {
6487            let file = H5File::create(&path).unwrap();
6488
6489            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
6490            ds.write_raw(&[1u8, 2]).unwrap();
6491
6492            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
6493            ds.write_raw(&[-1i8, 1]).unwrap();
6494
6495            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
6496            ds.write_raw(&[100u16, 200]).unwrap();
6497
6498            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
6499            ds.write_raw(&[-100i16, 100]).unwrap();
6500
6501            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
6502            ds.write_raw(&[1000u32, 2000]).unwrap();
6503
6504            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
6505            ds.write_raw(&[-1000i32, 1000]).unwrap();
6506
6507            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
6508            ds.write_raw(&[10000u64, 20000]).unwrap();
6509
6510            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
6511            ds.write_raw(&[-10000i64, 10000]).unwrap();
6512
6513            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
6514            ds.write_raw(&[1.5f32, 2.5]).unwrap();
6515
6516            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
6517            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
6518
6519            file.close().unwrap();
6520        }
6521
6522        {
6523            let file = H5File::open(&path).unwrap();
6524
6525            assert_eq!(
6526                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
6527                vec![1u8, 2]
6528            );
6529            assert_eq!(
6530                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
6531                vec![-1i8, 1]
6532            );
6533            assert_eq!(
6534                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
6535                vec![100u16, 200]
6536            );
6537            assert_eq!(
6538                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
6539                vec![-100i16, 100]
6540            );
6541            assert_eq!(
6542                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
6543                vec![1000u32, 2000]
6544            );
6545            assert_eq!(
6546                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
6547                vec![-1000i32, 1000]
6548            );
6549            assert_eq!(
6550                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
6551                vec![10000u64, 20000]
6552            );
6553            assert_eq!(
6554                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
6555                vec![-10000i64, 10000]
6556            );
6557            assert_eq!(
6558                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
6559                vec![1.5f32, 2.5]
6560            );
6561            assert_eq!(
6562                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
6563                vec![1.23456f64, 7.89012]
6564            );
6565        }
6566
6567        std::fs::remove_file(&path).ok();
6568    }
6569
6570    #[test]
6571    fn append_chunked_roundtrip() {
6572        let path = temp_path("append_chunked");
6573
6574        {
6575            let file = H5File::create(&path).unwrap();
6576            let ds = file
6577                .new_dataset::<f64>()
6578                .shape([0, 3])
6579                .chunk(&[1, 3])
6580                .max_shape(&[None, Some(3)])
6581                .create("data")
6582                .unwrap();
6583
6584            // Append one frame
6585            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
6586            // Append two frames at once
6587            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
6588
6589            file.close().unwrap();
6590        }
6591
6592        {
6593            let file = H5File::open(&path).unwrap();
6594            let ds = file.dataset("data").unwrap();
6595            assert_eq!(ds.shape(), vec![3, 3]);
6596            let all = ds.read_raw::<f64>().unwrap();
6597            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
6598        }
6599
6600        std::fs::remove_file(&path).ok();
6601    }
6602
6603    #[test]
6604    fn append_1d_chunked() {
6605        let path = temp_path("append_1d");
6606
6607        {
6608            let file = H5File::create(&path).unwrap();
6609            let ds = file
6610                .new_dataset::<i32>()
6611                .shape([0])
6612                .chunk(&[4])
6613                .max_shape(&[None])
6614                .create("values")
6615                .unwrap();
6616
6617            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
6618            ds.append(&[40i32]).unwrap(); // fills chunk boundary
6619            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
6620
6621            file.close().unwrap();
6622        }
6623
6624        {
6625            let file = H5File::open(&path).unwrap();
6626            let ds = file.dataset("values").unwrap();
6627            assert_eq!(ds.shape(), vec![8]);
6628            let all = ds.read_raw::<i32>().unwrap();
6629            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
6630        }
6631
6632        std::fs::remove_file(&path).ok();
6633    }
6634
6635    #[test]
6636    fn append_partial_chunk_flushed_on_close() {
6637        let path = temp_path("append_partial_close");
6638
6639        {
6640            let file = H5File::create(&path).unwrap();
6641            let ds = file
6642                .new_dataset::<f64>()
6643                .shape([0])
6644                .chunk(&[4])
6645                .max_shape(&[None])
6646                .create("vals")
6647                .unwrap();
6648
6649            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
6650            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
6651            file.close().unwrap();
6652        }
6653
6654        {
6655            let file = H5File::open(&path).unwrap();
6656            let ds = file.dataset("vals").unwrap();
6657            assert_eq!(ds.shape(), vec![5]);
6658            let all = ds.read_raw::<f64>().unwrap();
6659            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
6660            // read_raw reads total shape elements
6661            assert_eq!(all.len(), 5);
6662            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
6663        }
6664
6665        std::fs::remove_file(&path).ok();
6666    }
6667
6668    /// An append that leaves its chunk partial is buffered until close, and
6669    /// the flush has to keep the frames that chunk already holds. It built a
6670    /// fresh fill-value chunk around the buffered frame instead, so reopening
6671    /// a file and appending one row erased every earlier row of that chunk
6672    /// (issue #3). Four sessions: the second lands beside an existing row, the
6673    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
6674    /// the third left in chunk 1.
6675    #[test]
6676    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
6677        let path = temp_path("append_reopen_partial");
6678
6679        {
6680            let file = H5File::create(&path).unwrap();
6681            let ds = file
6682                .new_dataset::<i32>()
6683                .shape([0, 3])
6684                .chunk(&[4, 3])
6685                .max_shape(&[None, Some(3)])
6686                .create("values")
6687                .unwrap();
6688            ds.append(&[1, 2, 3]).unwrap();
6689            file.close().unwrap();
6690        }
6691        for rows in [
6692            vec![4, 5, 6],
6693            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
6694            vec![16, 17, 18],
6695        ] {
6696            let file = H5File::open_rw(&path).unwrap();
6697            file.dataset_writer("values")
6698                .unwrap()
6699                .append(&rows)
6700                .unwrap();
6701            file.close().unwrap();
6702        }
6703
6704        let file = H5File::open(&path).unwrap();
6705        let ds = file.dataset("values").unwrap();
6706        assert_eq!(ds.shape(), vec![6, 3]);
6707        assert_eq!(
6708            ds.read_raw::<i32>().unwrap(),
6709            (1..=18).collect::<Vec<i32>>()
6710        );
6711        std::fs::remove_file(&path).ok();
6712    }
6713
6714    #[cfg(feature = "deflate")]
6715    #[test]
6716    fn vlen_append_after_reopen_filtered() {
6717        // Reopen + append into a partially-written *compressed* vlen chunk
6718        // (index-block chunk). Exercises filtered-index-block reconstruction
6719        // in open_append plus filtered read-modify-write.
6720        let path = temp_path("vlen_reopen_filtered");
6721        {
6722            let file = H5File::create(&path).unwrap();
6723            file.create_appendable_vlen_dataset(
6724                "strs",
6725                4,
6726                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6727            )
6728            .unwrap();
6729            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
6730                .unwrap();
6731            file.close().unwrap();
6732        }
6733        {
6734            let file = H5File::open_rw(&path).unwrap();
6735            file.append_vlen_strings("strs", &["delta"]).unwrap();
6736            file.close().unwrap();
6737        }
6738        {
6739            let file = H5File::open(&path).unwrap();
6740            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6741            assert_eq!(
6742                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
6743                vec!["alpha", "beta", "gamma", "delta"]
6744            );
6745        }
6746        std::fs::remove_file(&path).ok();
6747    }
6748
6749    #[test]
6750    fn vlen_append_after_reopen_data_block() {
6751        // Reopen + append into a partial chunk that lives in an extensible-
6752        // array *data block* (chunk index >= idx_blk_elmts). Exercises
6753        // data-block resolution in read_chunk_if_present and write_chunk.
6754        let path = temp_path("vlen_reopen_datablk");
6755        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
6756        {
6757            let file = H5File::create(&path).unwrap();
6758            file.create_appendable_vlen_dataset("strs", 2, None)
6759                .unwrap();
6760            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6761            file.append_vlen_strings("strs", &refs).unwrap();
6762            file.close().unwrap();
6763        }
6764        {
6765            let file = H5File::open_rw(&path).unwrap();
6766            file.append_vlen_strings("strs", &["s9"]).unwrap();
6767            file.close().unwrap();
6768        }
6769        {
6770            let file = H5File::open(&path).unwrap();
6771            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6772            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
6773            assert_eq!(got, want);
6774        }
6775        std::fs::remove_file(&path).ok();
6776    }
6777
6778    #[test]
6779    fn vlen_append_after_reopen_super_block() {
6780        // Reopen + append into a partial chunk whose index falls in an
6781        // extensible-array *super block* (chunk index 244 with the default
6782        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
6783        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
6784        // index block or its direct data blocks, so chunk 244 is reached
6785        // via a super block read from disk). Exercises the ViaSblk branch
6786        // of read_chunk_if_present.
6787        let path = temp_path("vlen_reopen_super");
6788        // 489 strings, chunk size 2 -> chunk 244 holds one string only
6789        // (partially filled) and is flushed to disk on close.
6790        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
6791        {
6792            let file = H5File::create(&path).unwrap();
6793            file.create_appendable_vlen_dataset("strs", 2, None)
6794                .unwrap();
6795            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6796            file.append_vlen_strings("strs", &refs).unwrap();
6797            file.close().unwrap();
6798        }
6799        {
6800            let file = H5File::open_rw(&path).unwrap();
6801            file.append_vlen_strings("strs", &["v489"]).unwrap();
6802            file.close().unwrap();
6803        }
6804        {
6805            let file = H5File::open(&path).unwrap();
6806            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6807            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
6808            assert_eq!(got, want);
6809        }
6810        std::fs::remove_file(&path).ok();
6811    }
6812
6813    #[cfg(feature = "deflate")]
6814    #[test]
6815    fn vlen_append_after_reopen_filtered_data_block() {
6816        // The hardest path: compressed + chunk in a data block + partial
6817        // read-modify-write across a reopen.
6818        let path = temp_path("vlen_reopen_filt_datablk");
6819        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
6820        {
6821            let file = H5File::create(&path).unwrap();
6822            file.create_appendable_vlen_dataset(
6823                "strs",
6824                2,
6825                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6826            )
6827            .unwrap();
6828            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6829            file.append_vlen_strings("strs", &refs).unwrap();
6830            file.close().unwrap();
6831        }
6832        {
6833            let file = H5File::open_rw(&path).unwrap();
6834            file.append_vlen_strings("strs", &["item09"]).unwrap();
6835            file.close().unwrap();
6836        }
6837        {
6838            let file = H5File::open(&path).unwrap();
6839            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6840            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
6841            assert_eq!(got, want);
6842        }
6843        std::fs::remove_file(&path).ok();
6844    }
6845
6846    #[test]
6847    fn group_nx_class_attribute_roundtrip() {
6848        // Non-root groups carry attributes (NeXus `NX_class`) in their
6849        // own object header, and the reader reads them back by path.
6850        let path = temp_path("group_nx_class");
6851        {
6852            let file = H5File::create(&path).unwrap();
6853            let entry = file.create_group("entry").unwrap();
6854            entry.set_attr_string("NX_class", "NXentry").unwrap();
6855            let det = entry.create_group("detector").unwrap();
6856            det.set_attr_string("NX_class", "NXdetector").unwrap();
6857            det.set_attr_numeric("frame_count", &7i32).unwrap();
6858            det.new_dataset::<f32>()
6859                .shape([4])
6860                .create("data")
6861                .unwrap()
6862                .write_raw(&[1.0f32; 4])
6863                .unwrap();
6864            file.close().unwrap();
6865        }
6866        {
6867            let file = H5File::open(&path).unwrap();
6868            let entry = file.root_group().group("entry").unwrap();
6869            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
6870            let det = entry.group("detector").unwrap();
6871            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
6872            let names = det.attr_names().unwrap();
6873            assert!(names.contains(&"NX_class".to_string()));
6874            assert!(names.contains(&"frame_count".to_string()));
6875        }
6876        std::fs::remove_file(&path).ok();
6877    }
6878
6879    #[test]
6880    fn ea_super_block_roundtrip() {
6881        // 2000 chunks span several extensible-array super blocks. Before
6882        // super-block support the writer errored at chunk index 228.
6883        let path = temp_path("ea_super_rt");
6884        {
6885            let file = H5File::create(&path).unwrap();
6886            let ds = file
6887                .new_dataset::<i32>()
6888                .shape([0])
6889                .chunk(&[1])
6890                .max_shape(&[None])
6891                .create("v")
6892                .unwrap();
6893            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
6894            file.close().unwrap();
6895        }
6896        {
6897            let file = H5File::open(&path).unwrap();
6898            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6899            assert_eq!(v.len(), 2000);
6900            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
6901        }
6902        std::fs::remove_file(&path).ok();
6903    }
6904
6905    #[cfg(feature = "deflate")]
6906    #[test]
6907    fn ea_filtered_super_block_roundtrip() {
6908        // Compressed chunks across super blocks.
6909        let path = temp_path("ea_filt_super");
6910        {
6911            let file = H5File::create(&path).unwrap();
6912            let ds = file
6913                .new_dataset::<i32>()
6914                .shape([0])
6915                .chunk(&[1])
6916                .max_shape(&[None])
6917                .deflate(4)
6918                .create("v")
6919                .unwrap();
6920            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
6921            file.close().unwrap();
6922        }
6923        {
6924            let file = H5File::open(&path).unwrap();
6925            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6926            assert_eq!(v, (0..600).collect::<Vec<i32>>());
6927        }
6928        std::fs::remove_file(&path).ok();
6929    }
6930
6931    #[test]
6932    fn ea_super_block_open_append() {
6933        // Reopen a dataset and append chunks that fall in super blocks.
6934        let path = temp_path("ea_super_append");
6935        {
6936            let file = H5File::create(&path).unwrap();
6937            let ds = file
6938                .new_dataset::<i32>()
6939                .shape([0])
6940                .chunk(&[1])
6941                .max_shape(&[None])
6942                .create("v")
6943                .unwrap();
6944            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
6945            file.close().unwrap();
6946        }
6947        {
6948            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
6949            let idx = w.dataset_index("v").unwrap();
6950            for c in 300..900u64 {
6951                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
6952            }
6953            w.extend_dataset(idx, &[900]).unwrap();
6954            w.close().unwrap();
6955        }
6956        {
6957            let file = H5File::open(&path).unwrap();
6958            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6959            assert_eq!(v.len(), 900);
6960            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
6961        }
6962        std::fs::remove_file(&path).ok();
6963    }
6964
6965    // Two or more unlimited dimensions select the v2 B-tree index; with a
6966    // filter its records become type 11, carrying each chunk's stored size and
6967    // mask. The payload is highly compressible, so the chunks really are
6968    // stored smaller than the extent — the file would be at least
6969    // 6*8*4 = 192 bytes of raw chunk data otherwise.
6970    #[cfg(feature = "deflate")]
6971    #[test]
6972    fn compressed_multi_unlimited_dataset_roundtrips() {
6973        let path = temp_path("bt2_filtered");
6974        {
6975            let file = H5File::create(&path).unwrap();
6976            let ds = file
6977                .new_dataset::<i32>()
6978                .shape([6, 8])
6979                .chunk(&[2, 4])
6980                .max_shape(&[None, None])
6981                .deflate(6)
6982                .create("d")
6983                .unwrap();
6984            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
6985            // A partial write forces a decompress-patch-recompress of one
6986            // chunk, whose new compressed size may not fit its old block.
6987            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
6988            file.close().unwrap();
6989        }
6990        {
6991            let file = H5File::open(&path).unwrap();
6992            let ds = file.dataset("d").unwrap();
6993            assert_eq!(ds.shape(), vec![6, 8]);
6994            let mut want = vec![7i32; 48];
6995            want[9] = 1;
6996            want[10] = 2;
6997            want[17] = 3;
6998            want[18] = 4;
6999            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
7000        }
7001        std::fs::remove_file(&path).ok();
7002    }
7003
7004    #[test]
7005    fn btree_v2_multi_unlimited_roundtrip() {
7006        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
7007        // index; chunks are written by grid coordinates with write_chunk_at.
7008        let path = temp_path("bt2_multi");
7009        {
7010            let file = H5File::create(&path).unwrap();
7011            let ds = file
7012                .new_dataset::<i32>()
7013                .shape([0, 0])
7014                .chunk(&[2, 2])
7015                .max_shape(&[None, None])
7016                .create("grid")
7017                .unwrap();
7018            assert!(ds.is_chunked());
7019            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
7020            for cr in 0..2usize {
7021                for cc in 0..2usize {
7022                    let mut bytes = Vec::new();
7023                    for i in 0..2usize {
7024                        for j in 0..2usize {
7025                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
7026                            bytes.extend_from_slice(&v.to_le_bytes());
7027                        }
7028                    }
7029                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
7030                }
7031            }
7032            file.close().unwrap();
7033        }
7034        {
7035            let file = H5File::open(&path).unwrap();
7036            let ds = file.dataset("grid").unwrap();
7037            assert_eq!(ds.shape(), vec![4, 4]);
7038            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
7039        }
7040        std::fs::remove_file(&path).ok();
7041    }
7042
7043    #[test]
7044    fn subframe_chunking_roundtrip() {
7045        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
7046        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
7047        // the chunk-grid coordinates.
7048        let path = temp_path("subframe");
7049        {
7050            let file = H5File::create(&path).unwrap();
7051            let ds = file
7052                .new_dataset::<i32>()
7053                .shape([0, 8, 8])
7054                .chunk(&[1, 4, 4])
7055                .max_shape(&[None, Some(8), Some(8)])
7056                .create("v")
7057                .unwrap();
7058            for f in 0..3usize {
7059                for cr in 0..2usize {
7060                    for cc in 0..2usize {
7061                        let mut bytes = Vec::new();
7062                        for i in 0..4usize {
7063                            for j in 0..4usize {
7064                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
7065                                bytes.extend_from_slice(&v.to_le_bytes());
7066                            }
7067                        }
7068                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
7069                    }
7070                }
7071            }
7072            file.close().unwrap();
7073        }
7074        {
7075            let file = H5File::open(&path).unwrap();
7076            let ds = file.dataset("v").unwrap();
7077            assert_eq!(ds.shape(), vec![3, 8, 8]);
7078            assert_eq!(
7079                ds.read_raw::<i32>().unwrap(),
7080                (0..192).collect::<Vec<i32>>()
7081            );
7082        }
7083        std::fs::remove_file(&path).ok();
7084    }
7085
7086    #[test]
7087    fn fill_value_contiguous_roundtrip() {
7088        let path = temp_path("fill_value_contig");
7089        {
7090            let file = H5File::create(&path).unwrap();
7091            let ds = file
7092                .new_dataset::<f32>()
7093                .shape([4])
7094                .fill_value(2.5f32)
7095                .create("data")
7096                .unwrap();
7097            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
7098            file.close().unwrap();
7099        }
7100        // open_append decodes the fill-value message back from the header.
7101        {
7102            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7103            let idx = writer.dataset_index("data").unwrap();
7104            assert_eq!(
7105                writer.ds(idx).lock().fill_value,
7106                Some(2.5f32.to_le_bytes().to_vec())
7107            );
7108        }
7109        // Data still reads back correctly.
7110        {
7111            let file = H5File::open(&path).unwrap();
7112            let ds = file.dataset("data").unwrap();
7113            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
7114        }
7115        std::fs::remove_file(&path).ok();
7116    }
7117
7118    /// Early allocation on a fixed unfiltered shape selects the implicit
7119    /// index, and "implicit" is literal: the file holds no index structure
7120    /// at all, only a version-4 layout message of index type 2 pointing at
7121    /// the run of chunk space the create allocated.
7122    #[test]
7123    fn early_allocation_writes_the_implicit_index() {
7124        let path = temp_path("implicit_index");
7125        {
7126            let file = H5File::create(&path).unwrap();
7127            let ds = file
7128                .new_dataset::<i32>()
7129                .shape([16])
7130                .chunk(&[4])
7131                .early_allocation()
7132                .create("data")
7133                .unwrap();
7134            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7135            file.close().unwrap();
7136        }
7137        let bytes = std::fs::read(&path).unwrap();
7138        for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7139            assert!(
7140                !bytes.windows(4).any(|w| w == magic),
7141                "{} appears in a file whose chunk index is supposed to be no \
7142                 structure at all",
7143                String::from_utf8_lossy(magic)
7144            );
7145        }
7146        {
7147            let file = H5File::open(&path).unwrap();
7148            let ds = file.dataset("data").unwrap();
7149            assert_eq!(
7150                ds.read_raw::<i32>().unwrap(),
7151                (0..16i32).collect::<Vec<_>>()
7152            );
7153        }
7154        std::fs::remove_file(&path).ok();
7155    }
7156
7157    /// Two of the conditions are conditions: an unlimited dimension or a
7158    /// filter each send the dataset to the index libhdf5 would pick
7159    /// instead, early allocation or not. (The third — one whole-dataset
7160    /// chunk — sends it to the single-chunk index instead of Fixed Array;
7161    /// see `one_whole_dataset_chunk_writes_the_single_chunk_index`.)
7162    #[test]
7163    #[cfg(feature = "deflate")]
7164    fn early_allocation_only_picks_implicit_where_libhdf5_does() {
7165        // Every case writes and reads back its data, so a mis-selected index
7166        // shows up as wrong bytes and not just as a different structure.
7167        for (which, magic) in [("unlimited", b"EAHD"), ("filtered", b"FAHD")] {
7168            let path = temp_path("implicit_not");
7169            {
7170                let file = H5File::create(&path).unwrap();
7171                let builder = file.new_dataset::<i32>().shape([16]);
7172                let builder = match which {
7173                    "unlimited" => builder.chunk(&[4]).max_shape(&[None]),
7174                    _ => builder.chunk(&[4]).deflate(6),
7175                };
7176                let ds = builder.early_allocation().create("data").unwrap();
7177                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7178                file.close().unwrap();
7179            }
7180            let bytes = std::fs::read(&path).unwrap();
7181            assert!(
7182                bytes.windows(4).any(|w| w == magic),
7183                "{which}: expected a {} index",
7184                String::from_utf8_lossy(magic)
7185            );
7186            let file = H5File::open(&path).unwrap();
7187            assert_eq!(
7188                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7189                (0..16i32).collect::<Vec<_>>(),
7190                "{which}"
7191            );
7192            std::fs::remove_file(&path).ok();
7193        }
7194    }
7195
7196    /// One whole-dataset chunk always selects the single-chunk index —
7197    /// ahead of Fixed Array, and ahead of Implicit too, whether or not early
7198    /// allocation was requested (`H5D__layout_set_latest_indexing` checks it
7199    /// unconditionally). Like Implicit, "single chunk" is literal: no index
7200    /// structure at all, just the one chunk's address — and, unfiltered and
7201    /// early-allocated, that address exists before anything is written — in
7202    /// the layout message directly.
7203    #[test]
7204    fn one_whole_dataset_chunk_writes_the_single_chunk_index() {
7205        for early in [false, true] {
7206            let path = temp_path("single_chunk_index");
7207            {
7208                let file = H5File::create(&path).unwrap();
7209                let builder = file.new_dataset::<i32>().shape([16]).chunk(&[16]);
7210                let builder = if early {
7211                    builder.early_allocation()
7212                } else {
7213                    builder
7214                };
7215                let ds = builder.create("data").unwrap();
7216                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7217                file.close().unwrap();
7218            }
7219            let bytes = std::fs::read(&path).unwrap();
7220            for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7221                assert!(
7222                    !bytes.windows(4).any(|w| w == magic),
7223                    "early={early}: {} appears in a file whose chunk index is \
7224                     supposed to be no structure at all",
7225                    String::from_utf8_lossy(magic)
7226                );
7227            }
7228            let file = H5File::open(&path).unwrap();
7229            assert_eq!(
7230                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7231                (0..16i32).collect::<Vec<_>>(),
7232                "early={early}"
7233            );
7234            std::fs::remove_file(&path).ok();
7235        }
7236    }
7237
7238    /// An implicitly indexed dataset's chunks all exist from create, so an
7239    /// unwritten one reads back as the fill value — and the fill value is
7240    /// tiled over the whole run at create, not per chunk on demand.
7241    #[test]
7242    fn implicit_index_fills_every_chunk_at_create() {
7243        let path = temp_path("implicit_fill");
7244        {
7245            let file = H5File::create(&path).unwrap();
7246            let ds = file
7247                .new_dataset::<i32>()
7248                .shape([8])
7249                .chunk(&[4])
7250                .early_allocation()
7251                .fill_value(-3i32)
7252                .create("data")
7253                .unwrap();
7254            // Only chunk 0.
7255            let chunk: Vec<u8> = [1i32, 2, 3, 4]
7256                .iter()
7257                .flat_map(|v| v.to_le_bytes())
7258                .collect();
7259            ds.write_chunk(0, &chunk).unwrap();
7260            file.close().unwrap();
7261        }
7262        let file = H5File::open(&path).unwrap();
7263        let ds = file.dataset("data").unwrap();
7264        assert_eq!(
7265            ds.read_raw::<i32>().unwrap(),
7266            vec![1, 2, 3, 4, -3, -3, -3, -3]
7267        );
7268        std::fs::remove_file(&path).ok();
7269    }
7270
7271    /// A reopen has to reconstruct the run's address *and* its length from
7272    /// the layout message alone — there is no index structure to read it
7273    /// back from — or the close would rewrite the dataset as unallocated
7274    /// contiguous storage and drop every byte.
7275    #[test]
7276    fn implicit_index_survives_a_reopen() {
7277        let path = temp_path("implicit_reopen");
7278        {
7279            let file = H5File::create(&path).unwrap();
7280            file.new_dataset::<i32>()
7281                .shape([8])
7282                .chunk(&[4])
7283                .early_allocation()
7284                .create("data")
7285                .unwrap()
7286                .write_raw(&[0i32, 1, 2, 3, 4, 5, 6, 7])
7287                .unwrap();
7288            file.close().unwrap();
7289        }
7290        {
7291            let file = H5File::open_rw(&path).unwrap();
7292            let ds = file.dataset_writer("data").unwrap();
7293            let chunk: Vec<u8> = [10i32, 11, 12, 13]
7294                .iter()
7295                .flat_map(|v| v.to_le_bytes())
7296                .collect();
7297            ds.write_chunk(1, &chunk).unwrap();
7298            file.close().unwrap();
7299        }
7300        let file = H5File::open(&path).unwrap();
7301        assert_eq!(
7302            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7303            vec![0, 1, 2, 3, 10, 11, 12, 13]
7304        );
7305        std::fs::remove_file(&path).ok();
7306    }
7307
7308    #[test]
7309    fn fill_value_chunked_roundtrip() {
7310        let path = temp_path("fill_value_chunked");
7311        {
7312            let file = H5File::create(&path).unwrap();
7313            let ds = file
7314                .new_dataset::<i32>()
7315                .shape([0])
7316                .chunk(&[4])
7317                .max_shape(&[None])
7318                .fill_value(-7i32)
7319                .create("vals")
7320                .unwrap();
7321            ds.append(&[1i32, 2, 3, 4]).unwrap();
7322            file.close().unwrap();
7323        }
7324        {
7325            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7326            let idx = writer.dataset_index("vals").unwrap();
7327            assert_eq!(
7328                writer.ds(idx).lock().fill_value,
7329                Some((-7i32).to_le_bytes().to_vec())
7330            );
7331        }
7332        std::fs::remove_file(&path).ok();
7333    }
7334
7335    #[test]
7336    fn fill_value_read_missing_chunks() {
7337        // A chunked dataset with chunk 1 left unwritten must read that
7338        // gap back as the user-defined fill value, not zero.
7339        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
7340            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
7341        }
7342        let path = temp_path("fill_value_read_missing");
7343        {
7344            let file = H5File::create(&path).unwrap();
7345            let ds = file
7346                .new_dataset::<i32>()
7347                .shape([0])
7348                .chunk(&[2])
7349                .max_shape(&[None])
7350                .fill_value(-1i32)
7351                .create("vals")
7352                .unwrap();
7353            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
7354            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
7355            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
7356            ds.extend(&[6]).unwrap();
7357            file.close().unwrap();
7358        }
7359        {
7360            let file = H5File::open(&path).unwrap();
7361            let ds = file.dataset("vals").unwrap();
7362            let all = ds.read_raw::<i32>().unwrap();
7363            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
7364        }
7365        std::fs::remove_file(&path).ok();
7366    }
7367
7368    #[test]
7369    fn fill_value_partial_chunk_padded_with_fill() {
7370        // A partial trailing chunk flushed at close must pad its unwritten
7371        // tail with the fill value. That pad sits beyond the logical shape,
7372        // so it is verified by scanning the on-disk chunk bytes directly.
7373        let path = temp_path("fill_value_partial_pad");
7374        {
7375            let file = H5File::create(&path).unwrap();
7376            let ds = file
7377                .new_dataset::<i32>()
7378                .shape([0])
7379                .chunk(&[4])
7380                .max_shape(&[None])
7381                .fill_value(-9i32)
7382                .create("vals")
7383                .unwrap();
7384            // 3 of 4 frames -> flushed as a partial chunk on close.
7385            ds.append(&[1i32, 2, 3]).unwrap();
7386            file.close().unwrap();
7387        }
7388        let bytes = std::fs::read(&path).unwrap();
7389        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
7390        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
7391        let pos = bytes
7392            .windows(needle.len())
7393            .position(|w| w == needle)
7394            .expect("chunk data [1,2,3] not found in file");
7395        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
7396        assert_eq!(
7397            pad,
7398            &(-9i32).to_le_bytes(),
7399            "partial chunk tail must be padded with fill value -9, got {:?}",
7400            pad
7401        );
7402        std::fs::remove_file(&path).ok();
7403    }
7404
7405    #[test]
7406    fn vlen_append_after_reopen_preserves_existing() {
7407        // Reopening and appending into a partially-written vlen chunk must
7408        // read-modify-write: the strings already on disk must survive.
7409        let path = temp_path("vlen_append_reopen");
7410        {
7411            let file = H5File::create(&path).unwrap();
7412            file.create_appendable_vlen_dataset("strs", 4, None)
7413                .unwrap();
7414            // 3 of 4 frames -> flushed as a partial chunk on close.
7415            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
7416            file.close().unwrap();
7417        }
7418        {
7419            // Append a 4th string -> partial-chunk write into chunk 0.
7420            let file = H5File::open_rw(&path).unwrap();
7421            file.append_vlen_strings("strs", &["d"]).unwrap();
7422            file.close().unwrap();
7423        }
7424        {
7425            let file = H5File::open(&path).unwrap();
7426            let ds = file.dataset("strs").unwrap();
7427            let got = ds.read_vlen_strings().unwrap();
7428            assert_eq!(
7429                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
7430                vec!["a", "b", "c", "d"]
7431            );
7432        }
7433        std::fs::remove_file(&path).ok();
7434    }
7435
7436    #[test]
7437    fn fill_value_size_mismatch_errors() {
7438        let path = temp_path("fill_value_mismatch");
7439        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
7440        let dt = <f64 as crate::types::H5Type>::hdf5_type();
7441        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
7442        // f64 element size is 8; a 4-byte fill value must be rejected.
7443        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
7444        // The correct width succeeds.
7445        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
7446        writer.close().unwrap();
7447        std::fs::remove_file(&path).ok();
7448    }
7449
7450    #[test]
7451    fn datatype_exposes_class_sign_and_byteorder() {
7452        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
7453        // from f32 (both 4 bytes). datatype() must report the real class and
7454        // signedness so a reader does not have to guess from element_size.
7455        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
7456
7457        let path = temp_path("datatype_accessor");
7458        {
7459            let file = H5File::create(&path).unwrap();
7460            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
7461            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
7462            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
7463            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
7464            file.close().unwrap();
7465        }
7466
7467        let file = H5File::open(&path).unwrap();
7468
7469        match file.dataset("u8d").unwrap().datatype().unwrap() {
7470            DatatypeMessage::FixedPoint {
7471                size,
7472                signed,
7473                byte_order,
7474                ..
7475            } => {
7476                assert_eq!(size, 1);
7477                assert!(!signed, "u8 must be unsigned");
7478                assert_eq!(byte_order, ByteOrder::LittleEndian);
7479            }
7480            other => panic!("expected FixedPoint for u8, got {other:?}"),
7481        }
7482
7483        match file.dataset("i8d").unwrap().datatype().unwrap() {
7484            DatatypeMessage::FixedPoint { size, signed, .. } => {
7485                assert_eq!(size, 1);
7486                assert!(signed, "i8 must be signed");
7487            }
7488            other => panic!("expected FixedPoint for i8, got {other:?}"),
7489        }
7490
7491        match file.dataset("i32d").unwrap().datatype().unwrap() {
7492            DatatypeMessage::FixedPoint { size, signed, .. } => {
7493                assert_eq!(size, 4);
7494                assert!(signed, "i32 must be signed");
7495            }
7496            other => panic!("expected FixedPoint for i32, got {other:?}"),
7497        }
7498
7499        match file.dataset("f32d").unwrap().datatype().unwrap() {
7500            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
7501            other => panic!("expected FloatingPoint for f32, got {other:?}"),
7502        }
7503
7504        std::fs::remove_file(&path).ok();
7505    }
7506
7507    #[test]
7508    fn datatype_in_write_mode_errors() {
7509        let path = temp_path("datatype_write_mode");
7510        let file = H5File::create(&path).unwrap();
7511        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
7512        assert!(ds.datatype().is_err());
7513        std::fs::remove_file(&path).ok();
7514    }
7515
7516    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
7517
7518    /// Extensible-array path: pre-compress with the dataset's pipeline, write
7519    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
7520    /// the data round-trips through the reader unchanged.
7521    #[cfg(feature = "deflate")]
7522    #[test]
7523    fn write_chunk_raw_ea_roundtrip_mask0() {
7524        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7525        let path = temp_path("wcr_ea_mask0");
7526        let original: Vec<i32> = (0..12).collect();
7527        {
7528            let file = H5File::create(&path).unwrap();
7529            let ds = file
7530                .new_dataset::<i32>()
7531                .shape([0])
7532                .chunk(&[4])
7533                .max_shape(&[None])
7534                .deflate(4)
7535                .create("v")
7536                .unwrap();
7537            assert!(ds.is_chunked());
7538            let pipeline = FilterPipeline::deflate(4);
7539            for c in 0..3usize {
7540                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7541                    .iter()
7542                    .flat_map(|v| v.to_le_bytes())
7543                    .collect();
7544                let compressed = apply_filters(&pipeline, &raw).unwrap();
7545                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7546            }
7547            ds.set_extent(&[12]).unwrap();
7548            file.close().unwrap();
7549        }
7550        {
7551            let file = H5File::open(&path).unwrap();
7552            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7553            assert_eq!(v, original);
7554        }
7555        std::fs::remove_file(&path).ok();
7556    }
7557
7558    /// Fixed-array path (all dimensions bounded): same verbatim write through
7559    /// the linear-index dispatch, round-tripped through the reader.
7560    #[cfg(feature = "deflate")]
7561    #[test]
7562    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
7563        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7564        let path = temp_path("wcr_fa_mask0");
7565        let original: Vec<i32> = (0..12).collect();
7566        {
7567            let file = H5File::create(&path).unwrap();
7568            let ds = file
7569                .new_dataset::<i32>()
7570                .shape([12])
7571                .chunk(&[4])
7572                .deflate(4)
7573                .create("v")
7574                .unwrap();
7575            assert!(ds.is_chunked());
7576            let pipeline = FilterPipeline::deflate(4);
7577            for c in 0..3usize {
7578                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7579                    .iter()
7580                    .flat_map(|v| v.to_le_bytes())
7581                    .collect();
7582                let compressed = apply_filters(&pipeline, &raw).unwrap();
7583                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7584            }
7585            file.close().unwrap();
7586        }
7587        {
7588            let file = H5File::open(&path).unwrap();
7589            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7590            assert_eq!(v, original);
7591        }
7592        std::fs::remove_file(&path).ok();
7593    }
7594
7595    /// The caller-supplied filter_mask must reach the on-disk filtered index
7596    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
7597    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
7598    /// the extensible-array filtered entry to read the mask back at the format
7599    /// level (independent of the data reader's mask handling).
7600    #[cfg(feature = "deflate")]
7601    #[test]
7602    fn write_chunk_raw_records_filter_mask() {
7603        let path = temp_path("wcr_records_mask");
7604        let raw: Vec<u8> = [10i32, 20, 30, 40]
7605            .iter()
7606            .flat_map(|v| v.to_le_bytes())
7607            .collect();
7608        assert_eq!(raw.len(), 16);
7609        {
7610            let file = H5File::create(&path).unwrap();
7611            let ds = file
7612                .new_dataset::<i32>()
7613                .shape([0])
7614                .chunk(&[4])
7615                .max_shape(&[None])
7616                .deflate(4)
7617                .create("v")
7618                .unwrap();
7619            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
7620            // chunk is stored uncompressed (its raw bytes).
7621            ds.write_chunk_raw(0, &raw, 1).unwrap();
7622            ds.set_extent(&[4]).unwrap();
7623            file.close().unwrap();
7624        }
7625        // Reopen the writer; open_append decodes the filtered index block from
7626        // disk, so the entry reflects exactly what was committed.
7627        {
7628            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7629            let idx = w.dataset_index("v").unwrap();
7630            let ds = w.ds(idx);
7631            let m = ds.lock();
7632            let entry = &m
7633                .chunked
7634                .as_ref()
7635                .unwrap()
7636                .filt_iblk
7637                .as_ref()
7638                .unwrap()
7639                .elements[0];
7640            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
7641            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
7642        }
7643        std::fs::remove_file(&path).ok();
7644    }
7645
7646    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
7647    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
7648    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
7649    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
7650    /// A chunk whose stored stream decodes to less than its image places no
7651    /// run at all: the whole chunk reads as the fill value, whether the read
7652    /// laid the fill down first (a plan that leaves output uncovered — here the
7653    /// unallocated middle chunk) or fills only what nothing wrote. The decode
7654    /// writes into the output image itself, so the bytes a short image leaves
7655    /// behind are the ones this covers.
7656    #[cfg(feature = "deflate")]
7657    #[test]
7658    fn a_chunk_that_decodes_short_of_its_image_reads_as_fill() {
7659        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7660        let path = temp_path("short_chunk_image_is_fill");
7661        let pipeline = FilterPipeline::deflate(4);
7662        {
7663            let file = H5File::create(&path).unwrap();
7664            let ds = file
7665                .new_dataset::<i32>()
7666                .shape([0])
7667                .chunk(&[4])
7668                .max_shape(&[None])
7669                .fill_value(-1i32)
7670                .deflate(4)
7671                .create("v")
7672                .unwrap();
7673            // Chunk 0 carries two elements where its image wants four.
7674            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7675            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7676                .unwrap();
7677            // Chunk 1 is never written; chunk 2 is whole.
7678            let whole: Vec<u8> = [9i32, 10, 11, 12]
7679                .iter()
7680                .flat_map(|v| v.to_le_bytes())
7681                .collect();
7682            ds.write_chunk_raw(2, &apply_filters(&pipeline, &whole).unwrap(), 0)
7683                .unwrap();
7684            ds.set_extent(&[12]).unwrap();
7685            file.close().unwrap();
7686        }
7687        {
7688            let file = H5File::open(&path).unwrap();
7689            let ds = file.dataset("v").unwrap();
7690            assert_eq!(
7691                ds.read_raw::<i32>().unwrap(),
7692                vec![-1, -1, -1, -1, -1, -1, -1, -1, 9, 10, 11, 12]
7693            );
7694            // The same verdict when the plan covers every output byte, so no
7695            // fill goes down first: chunks 0 and 2 alone.
7696            assert_eq!(ds.read_slice::<i32>(&[0], &[4]).unwrap(), vec![-1; 4]);
7697            assert_eq!(
7698                ds.read_slice::<i32>(&[8], &[4]).unwrap(),
7699                vec![9, 10, 11, 12]
7700            );
7701        }
7702        std::fs::remove_file(&path).ok();
7703    }
7704
7705    /// The staged spelling of the case above: a selection that takes only part
7706    /// of the short chunk decodes it into a buffer sized from the layout, and
7707    /// what that buffer holds past the stream is cut off rather than kept — a
7708    /// run inside the decoded bytes is real data, a run reaching past them is
7709    /// fill.
7710    #[cfg(feature = "deflate")]
7711    #[test]
7712    fn a_staged_chunk_carries_only_what_its_stream_decoded() {
7713        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7714        let path = temp_path("short_chunk_image_staged");
7715        let pipeline = FilterPipeline::deflate(4);
7716        {
7717            let file = H5File::create(&path).unwrap();
7718            let ds = file
7719                .new_dataset::<i32>()
7720                .shape([0])
7721                .chunk(&[4])
7722                .max_shape(&[None])
7723                .fill_value(-1i32)
7724                .deflate(4)
7725                .create("v")
7726                .unwrap();
7727            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7728            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7729                .unwrap();
7730            ds.set_extent(&[4]).unwrap();
7731            file.close().unwrap();
7732        }
7733        {
7734            let file = H5File::open(&path).unwrap();
7735            let ds = file.dataset("v").unwrap();
7736            // Inside the decoded bytes.
7737            assert_eq!(ds.read_slice::<i32>(&[0], &[2]).unwrap(), vec![7, 8]);
7738            // Straddling their end: the run is not placed at all.
7739            assert_eq!(ds.read_slice::<i32>(&[1], &[2]).unwrap(), vec![-1, -1]);
7740        }
7741        std::fs::remove_file(&path).ok();
7742    }
7743
7744    #[cfg(feature = "deflate")]
7745    #[test]
7746    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
7747        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7748        let path = temp_path("wcr_ea_per_chunk_mask");
7749        let original: Vec<i32> = (0..8).collect();
7750        let pipeline = FilterPipeline::deflate(4);
7751        {
7752            let file = H5File::create(&path).unwrap();
7753            let ds = file
7754                .new_dataset::<i32>()
7755                .shape([0])
7756                .chunk(&[4])
7757                .max_shape(&[None])
7758                .deflate(4)
7759                .create("v")
7760                .unwrap();
7761            let raw0: Vec<u8> = original[0..4]
7762                .iter()
7763                .flat_map(|v| v.to_le_bytes())
7764                .collect();
7765            // chunk 0: compressed through the pipeline, mask 0.
7766            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7767                .unwrap();
7768            let raw1: Vec<u8> = original[4..8]
7769                .iter()
7770                .flat_map(|v| v.to_le_bytes())
7771                .collect();
7772            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
7773            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7774            ds.set_extent(&[8]).unwrap();
7775            file.close().unwrap();
7776        }
7777        {
7778            let file = H5File::open(&path).unwrap();
7779            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7780            assert_eq!(v, original);
7781        }
7782        std::fs::remove_file(&path).ok();
7783    }
7784
7785    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
7786    /// compressed/raw chunks as the EA case, through the fixed-array index.
7787    #[cfg(feature = "deflate")]
7788    #[test]
7789    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
7790        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7791        let path = temp_path("wcr_fa_per_chunk_mask");
7792        let original: Vec<i32> = (0..8).collect();
7793        let pipeline = FilterPipeline::deflate(4);
7794        {
7795            let file = H5File::create(&path).unwrap();
7796            let ds = file
7797                .new_dataset::<i32>()
7798                .shape([8])
7799                .chunk(&[4])
7800                .deflate(4)
7801                .create("v")
7802                .unwrap();
7803            let raw0: Vec<u8> = original[0..4]
7804                .iter()
7805                .flat_map(|v| v.to_le_bytes())
7806                .collect();
7807            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7808                .unwrap();
7809            let raw1: Vec<u8> = original[4..8]
7810                .iter()
7811                .flat_map(|v| v.to_le_bytes())
7812                .collect();
7813            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7814            file.close().unwrap();
7815        }
7816        {
7817            let file = H5File::open(&path).unwrap();
7818            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7819            assert_eq!(v, original);
7820        }
7821        std::fs::remove_file(&path).ok();
7822    }
7823
7824    /// An unfiltered chunk index has no slot for a stored size or mask, so a
7825    /// direct chunk write must be rejected rather than silently dropping them.
7826    #[test]
7827    fn write_chunk_raw_rejects_unfiltered() {
7828        let path = temp_path("wcr_unfiltered");
7829        let file = H5File::create(&path).unwrap();
7830        let ds = file
7831            .new_dataset::<i32>()
7832            .shape([0])
7833            .chunk(&[4])
7834            .max_shape(&[None])
7835            .create("v")
7836            .unwrap();
7837        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7838        assert!(
7839            err.to_string().contains("filtered dataset"),
7840            "expected a filtered-dataset error, got: {err}"
7841        );
7842        std::fs::remove_file(&path).ok();
7843    }
7844
7845    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
7846    /// index to mean anything against, so the linear entry point points the
7847    /// caller at the coordinate-addressed one rather than guessing a grid.
7848    #[test]
7849    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
7850        let path = temp_path("wcr_btree2");
7851        let file = H5File::create(&path).unwrap();
7852        let ds = file
7853            .new_dataset::<i32>()
7854            .shape([0, 0])
7855            .chunk(&[2, 2])
7856            .max_shape(&[None, None])
7857            .create("grid")
7858            .unwrap();
7859        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7860        assert!(
7861            err.to_string().contains("write_chunk_raw_at"),
7862            "expected a pointer to the coordinate form, got: {err}"
7863        );
7864        std::fs::remove_file(&path).ok();
7865    }
7866
7867    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
7868    /// and the type-11 record carries their size and the caller's mask, so a
7869    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
7870    /// bytes while one written compressed (mask 0) is decompressed.
7871    #[cfg(feature = "deflate")]
7872    #[test]
7873    fn write_chunk_raw_at_round_trips_on_btree_v2() {
7874        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7875
7876        let path = temp_path("wcr_at_btree2");
7877        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
7878        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
7879        {
7880            let file = H5File::create(&path).unwrap();
7881            let ds = file
7882                .new_dataset::<i32>()
7883                .shape([0, 0])
7884                .chunk(&[2, 2])
7885                .max_shape(&[None, None])
7886                .deflate(6)
7887                .create("grid")
7888                .unwrap();
7889            let pipeline = FilterPipeline::deflate(6);
7890            // Chunk (0,0): pipeline already applied upstream, mask 0.
7891            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
7892                .unwrap();
7893            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
7894            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
7895            file.close().unwrap();
7896        }
7897        let file = H5File::open(&path).unwrap();
7898        let ds = file.dataset("grid").unwrap();
7899        assert_eq!(ds.shape(), vec![4, 4]);
7900        let all = ds.read_raw::<i32>().unwrap();
7901        // Chunk (0,0) occupies rows 0..2, columns 0..2.
7902        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
7903        // Chunk (1,1) occupies rows 2..4, columns 2..4.
7904        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
7905        drop(file);
7906        std::fs::remove_file(&path).ok();
7907    }
7908
7909    /// The coordinate form is not BT2-only: it addresses an extensible- or
7910    /// fixed-array dataset's grid just as well, and records the same mask.
7911    #[cfg(feature = "deflate")]
7912    #[test]
7913    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
7914        for (label, max_shape) in [
7915            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
7916            ("wcr_at_fa", None),
7917        ] {
7918            let path = temp_path(label);
7919            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
7920            {
7921                let file = H5File::create(&path).unwrap();
7922                let mut b = file
7923                    .new_dataset::<i32>()
7924                    .shape([4usize, 4])
7925                    .chunk(&[2, 4])
7926                    .deflate(6);
7927                if let Some(ref ms) = max_shape {
7928                    b = b.max_shape(ms);
7929                }
7930                let ds = b.create("grid").unwrap();
7931                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
7932                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
7933                file.close().unwrap();
7934            }
7935            let file = H5File::open(&path).unwrap();
7936            let ds = file.dataset("grid").unwrap();
7937            let all = ds.read_raw::<i32>().unwrap();
7938            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
7939            drop(file);
7940            std::fs::remove_file(&path).ok();
7941        }
7942    }
7943
7944    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
7945    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
7946    /// chunk gives chunk_size_len = 2 (max 65535).
7947    #[cfg(feature = "deflate")]
7948    #[test]
7949    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
7950        let path = temp_path("wcr_at_oversized");
7951        let file = H5File::create(&path).unwrap();
7952        let ds = file
7953            .new_dataset::<i32>()
7954            .shape([0, 0])
7955            .chunk(&[1, 1])
7956            .max_shape(&[None, None])
7957            .deflate(4)
7958            .create("grid")
7959            .unwrap();
7960        let err = ds
7961            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
7962            .unwrap_err();
7963        assert!(
7964            err.to_string().contains("does not fit"),
7965            "expected a chunk-size-field overflow error, got: {err}"
7966        );
7967        std::fs::remove_file(&path).ok();
7968    }
7969
7970    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
7971    /// the same reason the array indexes reject a direct write.
7972    #[test]
7973    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
7974        let path = temp_path("wcr_at_unfiltered");
7975        let file = H5File::create(&path).unwrap();
7976        let ds = file
7977            .new_dataset::<i32>()
7978            .shape([0, 0])
7979            .chunk(&[2, 2])
7980            .max_shape(&[None, None])
7981            .create("grid")
7982            .unwrap();
7983        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
7984        assert!(
7985            err.to_string().contains("filtered dataset"),
7986            "expected a filtered-dataset error, got: {err}"
7987        );
7988        std::fs::remove_file(&path).ok();
7989    }
7990
7991    /// A stored size that does not fit the index's chunk-size field must error
7992    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
7993    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
7994    /// a 70000-byte stored chunk overflows it.
7995    #[cfg(feature = "deflate")]
7996    #[test]
7997    fn write_chunk_raw_rejects_oversized_chunk() {
7998        let path = temp_path("wcr_oversized");
7999        let file = H5File::create(&path).unwrap();
8000        let ds = file
8001            .new_dataset::<i32>()
8002            .shape([0])
8003            .chunk(&[1])
8004            .max_shape(&[None])
8005            .deflate(4)
8006            .create("v")
8007            .unwrap();
8008        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
8009        assert!(
8010            err.to_string().contains("does not fit"),
8011            "expected a chunk-size-field overflow error, got: {err}"
8012        );
8013        std::fs::remove_file(&path).ok();
8014    }
8015
8016    // ---- issue #5: runtime-width fixed-string reading ----------------------
8017
8018    use crate::format::messages::datatype::{CompoundMember, DatatypeMessage};
8019
8020    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
8021    /// element images, optionally chunked and deflated.
8022    fn write_fixed_string_dataset(
8023        path: &std::path::Path,
8024        dt: DatatypeMessage,
8025        width: usize,
8026        elems: &[&[u8]],
8027        compressed: bool,
8028    ) {
8029        let mut raw = Vec::with_capacity(elems.len() * width);
8030        for e in elems {
8031            assert!(e.len() <= width);
8032            raw.extend_from_slice(e);
8033            raw.resize(raw.len() + (width - e.len()), 0);
8034        }
8035        let file = H5File::create(path).unwrap();
8036        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
8037        if compressed {
8038            b = b.chunk(&[2]).deflate(6);
8039        }
8040        let ds = b.create("labels").unwrap();
8041        ds.write_raw_bytes(&raw).unwrap();
8042        file.close().unwrap();
8043    }
8044
8045    /// The width is whatever the file says, so one call reads a 24-byte label
8046    /// column and a 100-byte one. Producers like VASP pick it per dataset.
8047    #[test]
8048    fn read_strings_handles_any_fixed_width() {
8049        for width in [4usize, 24, 100] {
8050            let path = temp_path(&format!("fixed_str_{width}"));
8051            write_fixed_string_dataset(
8052                &path,
8053                DatatypeMessage::fixed_string(width as u32),
8054                width,
8055                &[b"ab", b"cde", b""],
8056                false,
8057            );
8058            let file = H5File::open(&path).unwrap();
8059            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8060            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
8061            std::fs::remove_file(&path).ok();
8062        }
8063    }
8064
8065    /// Each padding rule decides where the value ends. Null-terminated and
8066    /// null-padded both stop at the first NUL and ignore the bytes after it;
8067    /// space-padded strips only a tail of spaces, so an embedded NUL is
8068    /// content there.
8069    ///
8070    /// Checked against libhdf5 1.14.6: reading this same `"ab\0X\0\0"`
8071    /// null-padded element into a wider null-terminated destination gives
8072    /// `"ab"`, and reading a space-padded `"a\0b     "` gives `"a\0b"` —
8073    /// `H5T__conv_s_s` runs the same `!s[nchars]` loop for both null rules.
8074    #[test]
8075    fn read_strings_honors_every_padding_rule() {
8076        // "ab" then a NUL then trailing junk that both null rules must drop.
8077        let elem: &[u8] = b"ab\0X\0\0";
8078        for (padding, want) in [(0u8, "ab"), (1, "ab")] {
8079            let path = temp_path(&format!("fixed_pad_{padding}"));
8080            write_fixed_string_dataset(
8081                &path,
8082                DatatypeMessage::FixedString {
8083                    size: 6,
8084                    padding,
8085                    charset: 0,
8086                },
8087                6,
8088                &[elem],
8089                false,
8090            );
8091            let file = H5File::open(&path).unwrap();
8092            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8093            assert_eq!(got, vec![want.to_string()], "padding {padding}");
8094            std::fs::remove_file(&path).ok();
8095        }
8096        // Space-padded keeps interior spaces and strips only the tail.
8097        let path = temp_path("fixed_pad_2");
8098        write_fixed_string_dataset(
8099            &path,
8100            DatatypeMessage::FixedString {
8101                size: 8,
8102                padding: 2,
8103                charset: 0,
8104            },
8105            8,
8106            &[b"a b     "],
8107            false,
8108        );
8109        let file = H5File::open(&path).unwrap();
8110        assert_eq!(
8111            file.dataset("labels").unwrap().read_strings().unwrap(),
8112            vec!["a b".to_string()]
8113        );
8114        std::fs::remove_file(&path).ok();
8115
8116        // ... and an embedded NUL, which no space rule marks as an end.
8117        let path = temp_path("fixed_pad_2_nul");
8118        write_fixed_string_dataset(
8119            &path,
8120            DatatypeMessage::FixedString {
8121                size: 8,
8122                padding: 2,
8123                charset: 0,
8124            },
8125            8,
8126            &[b"a\0b     "],
8127            false,
8128        );
8129        let file = H5File::open(&path).unwrap();
8130        assert_eq!(
8131            file.dataset("labels").unwrap().read_strings().unwrap(),
8132            vec!["a\0b".to_string()]
8133        );
8134        std::fs::remove_file(&path).ok();
8135    }
8136
8137    /// A reserved padding or character-set code is an error naming the element,
8138    /// not a guess.
8139    #[test]
8140    fn read_strings_rejects_reserved_datatype_codes() {
8141        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
8142            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
8143            write_fixed_string_dataset(
8144                &path,
8145                DatatypeMessage::FixedString {
8146                    size: 4,
8147                    padding,
8148                    charset,
8149                },
8150                4,
8151                &[b"ab"],
8152                false,
8153            );
8154            let file = H5File::open(&path).unwrap();
8155            let err = file
8156                .dataset("labels")
8157                .unwrap()
8158                .read_strings()
8159                .unwrap_err()
8160                .to_string();
8161            assert!(err.contains(want), "got: {err}");
8162            std::fs::remove_file(&path).ok();
8163        }
8164    }
8165
8166    /// The typed read paths reinterpret the element image, so the stored order
8167    /// has to be the host's first. A scalar is swapped; a composite cannot be
8168    /// (its members have their own orders and offsets) and is refused.
8169    #[test]
8170    fn to_host_byte_order_converts_scalars_and_refuses_composites() {
8171        use crate::dataset::{to_host_byte_order, HOST_BYTE_ORDER};
8172        use crate::format::messages::datatype::ByteOrder;
8173
8174        let foreign = match HOST_BYTE_ORDER {
8175            ByteOrder::LittleEndian => ByteOrder::BigEndian,
8176            ByteOrder::BigEndian => ByteOrder::LittleEndian,
8177        };
8178        let int = |order, size| DatatypeMessage::FixedPoint {
8179            size,
8180            byte_order: order,
8181            signed: false,
8182            bit_offset: 0,
8183            bit_precision: (size * 8) as u16,
8184        };
8185
8186        // Foreign order: each element is reversed, elementwise.
8187        let mut buf = [1u8, 2, 3, 4, 5, 6, 7, 8];
8188        to_host_byte_order(&mut buf, &int(foreign, 4), 4).unwrap();
8189        assert_eq!(buf, [4, 3, 2, 1, 8, 7, 6, 5]);
8190
8191        // Host order: untouched.
8192        let mut buf = [1u8, 2, 3, 4];
8193        to_host_byte_order(&mut buf, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8194        assert_eq!(buf, [1, 2, 3, 4]);
8195
8196        // One byte wide: no order to convert.
8197        let mut buf = [1u8, 2, 3, 4];
8198        to_host_byte_order(&mut buf, &int(foreign, 1), 1).unwrap();
8199        assert_eq!(buf, [1, 2, 3, 4]);
8200
8201        // An enum stores its values in its base type's order.
8202        let mut buf = [1u8, 2];
8203        let enumeration = DatatypeMessage::Enum {
8204            base: Box::new(int(foreign, 2)),
8205            members: Vec::new(),
8206        };
8207        to_host_byte_order(&mut buf, &enumeration, 2).unwrap();
8208        assert_eq!(buf, [2, 1]);
8209
8210        // A string has no byte order at all.
8211        let mut buf = *b"abcd";
8212        to_host_byte_order(&mut buf, &DatatypeMessage::fixed_string(4), 4).unwrap();
8213        assert_eq!(&buf, b"abcd");
8214
8215        // A compound whose members are all host-order is reinterpretable.
8216        let compound = |order| DatatypeMessage::Compound {
8217            size: 4,
8218            members: vec![CompoundMember {
8219                name: "x".into(),
8220                offset: 0,
8221                datatype: int(order, 4),
8222            }],
8223        };
8224        let mut buf = [1u8, 2, 3, 4];
8225        to_host_byte_order(&mut buf, &compound(HOST_BYTE_ORDER), 4).unwrap();
8226        assert_eq!(buf, [1, 2, 3, 4]);
8227
8228        // One that is not says so, rather than handing back the raw bytes.
8229        let mut buf = [1u8, 2, 3, 4];
8230        let err = to_host_byte_order(&mut buf, &compound(foreign), 4)
8231            .expect_err("a foreign-order compound was reinterpreted")
8232            .to_string();
8233        assert!(err.contains("read_raw_bytes"), "got: {err}");
8234        assert_eq!(buf, [1, 2, 3, 4], "the refused image is left alone");
8235    }
8236
8237    /// The write direction answers for exactly the types the read direction
8238    /// does — same classifier — and borrows the caller's bytes whenever the
8239    /// declared order is already the host's.
8240    #[test]
8241    fn to_stored_byte_order_converts_scalars_and_refuses_composites() {
8242        use crate::dataset::{to_stored_byte_order, FOREIGN_BYTE_ORDER, HOST_BYTE_ORDER};
8243        use std::borrow::Cow;
8244
8245        let int = |order, size| DatatypeMessage::FixedPoint {
8246            size,
8247            byte_order: order,
8248            signed: false,
8249            bit_offset: 0,
8250            bit_precision: (size * 8) as u16,
8251        };
8252
8253        // Declared foreign: each element is reversed on the way out.
8254        let host = [1u8, 2, 3, 4, 5, 6, 7, 8];
8255        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 4), 4).unwrap();
8256        assert_eq!(&*stored, &[4, 3, 2, 1, 8, 7, 6, 5]);
8257        assert!(matches!(stored, Cow::Owned(_)), "a swap needs its own copy");
8258
8259        // Declared host order: handed through without a copy.
8260        let stored = to_stored_byte_order(&host, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8261        assert!(matches!(stored, Cow::Borrowed(_)), "no copy without a swap");
8262        assert_eq!(&*stored, &host);
8263
8264        // One byte wide: no order to lay out.
8265        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 1), 1).unwrap();
8266        assert_eq!(&*stored, &host);
8267
8268        // An enum stores its values in its base type's order.
8269        let enumeration = DatatypeMessage::Enum {
8270            base: Box::new(int(FOREIGN_BYTE_ORDER, 2)),
8271            members: Vec::new(),
8272        };
8273        let stored = to_stored_byte_order(&[1u8, 2], &enumeration, 2).unwrap();
8274        assert_eq!(&*stored, &[2, 1]);
8275
8276        // A compound cannot be laid out as a unit; one that declares the
8277        // foreign order for a member is refused, not written host-order.
8278        let compound = |order| DatatypeMessage::Compound {
8279            size: 4,
8280            members: vec![CompoundMember {
8281                name: "x".into(),
8282                offset: 0,
8283                datatype: int(order, 4),
8284            }],
8285        };
8286        let stored = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(HOST_BYTE_ORDER), 4).unwrap();
8287        assert_eq!(&*stored, &[1, 2, 3, 4]);
8288        let err = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(FOREIGN_BYTE_ORDER), 4)
8289            .expect_err("a foreign-order compound was written from host bytes")
8290            .to_string();
8291        assert!(err.contains("write_raw_bytes"), "got: {err}");
8292    }
8293
8294    /// The declared character set is enforced: a byte that cannot be decoded is
8295    /// an error naming the element, and the lossy call is what accepts the file
8296    /// instead of a silent substitution here.
8297    #[test]
8298    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
8299        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
8300        // in one that declares UTF-8.
8301        for (charset, bytes, want) in [
8302            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
8303            (1, b"a\xff".as_slice(), "not valid UTF-8"),
8304        ] {
8305            let path = temp_path(&format!("fixed_charset_{charset}"));
8306            write_fixed_string_dataset(
8307                &path,
8308                DatatypeMessage::FixedString {
8309                    size: 6,
8310                    padding: 1,
8311                    charset,
8312                },
8313                6,
8314                &[b"ok", bytes],
8315                false,
8316            );
8317            let file = H5File::open(&path).unwrap();
8318            let ds = file.dataset("labels").unwrap();
8319            let err = ds.read_strings().unwrap_err().to_string();
8320            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
8321            let lossy = ds.read_strings_lossy().unwrap();
8322            assert_eq!(lossy[0], "ok");
8323            assert_eq!(
8324                lossy[1].chars().next().unwrap(),
8325                if charset == 0 { 'c' } else { 'a' }
8326            );
8327            std::fs::remove_file(&path).ok();
8328        }
8329    }
8330
8331    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
8332    /// split a character.
8333    #[test]
8334    fn read_strings_reads_utf8_fixed_strings() {
8335        let path = temp_path("fixed_utf8");
8336        write_fixed_string_dataset(
8337            &path,
8338            DatatypeMessage::fixed_string_utf8(12),
8339            12,
8340            &["héllo".as_bytes(), "안녕".as_bytes()],
8341            false,
8342        );
8343        let file = H5File::open(&path).unwrap();
8344        assert_eq!(
8345            file.dataset("labels").unwrap().read_strings().unwrap(),
8346            vec!["héllo".to_string(), "안녕".to_string()]
8347        );
8348        std::fs::remove_file(&path).ok();
8349    }
8350
8351    /// The decode sits on the decoded raw-data path, so a chunked and deflated
8352    /// dataset reads the same as a contiguous one.
8353    #[cfg(feature = "deflate")]
8354    #[test]
8355    fn read_strings_reads_a_compressed_fixed_string_dataset() {
8356        let path = temp_path("fixed_str_deflate");
8357        write_fixed_string_dataset(
8358            &path,
8359            DatatypeMessage::fixed_string(16),
8360            16,
8361            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
8362            true,
8363        );
8364        let file = H5File::open(&path).unwrap();
8365        assert_eq!(
8366            file.dataset("labels").unwrap().read_strings().unwrap(),
8367            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
8368        );
8369        std::fs::remove_file(&path).ok();
8370    }
8371
8372    /// One call covers both string datatypes, so a caller need not branch on
8373    /// which one the file used.
8374    #[test]
8375    fn read_strings_also_reads_variable_length_strings() {
8376        let path = temp_path("read_strings_vlen");
8377        {
8378            let file = H5File::create(&path).unwrap();
8379            file.write_vlen_strings("names", &["alpha", "", "안녕"])
8380                .unwrap();
8381            file.close().unwrap();
8382        }
8383        let file = H5File::open(&path).unwrap();
8384        assert_eq!(
8385            file.dataset("names").unwrap().read_strings().unwrap(),
8386            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
8387        );
8388        std::fs::remove_file(&path).ok();
8389    }
8390
8391    /// A file declaring a zero-width fixed string is an error, not the panic
8392    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
8393    /// test patches the width in the encoded datatype message down to zero and
8394    /// re-stamps the object header's checksum over the result.
8395    #[test]
8396    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
8397        use crate::format::checksum::checksum_metadata;
8398        use crate::format::object_header::OHDR_SIGNATURE;
8399
8400        let path = temp_path("fixed_str_zero_width");
8401        write_fixed_string_dataset(
8402            &path,
8403            DatatypeMessage::fixed_string(37),
8404            37,
8405            &[b"ab", b"cd"],
8406            false,
8407        );
8408
8409        // Version 1 string datatype: class|version, padding|charset, two
8410        // reserved bytes, then the width as a little-endian u32. The width is
8411        // 37 so the eight bytes occur once in the file.
8412        let mut bytes = std::fs::read(&path).unwrap();
8413        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
8414        let at = bytes
8415            .windows(needle.len())
8416            .position(|w| w == needle)
8417            .expect("encoded fixed-string datatype message");
8418        assert!(
8419            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
8420            "the datatype message pattern is not unique in the file"
8421        );
8422
8423        // The enclosing v2 object header ends in a checksum over everything
8424        // from its signature onwards; find the offset where the stored value
8425        // still agrees, so the patched header can be re-stamped there.
8426        let ohdr = bytes[..at]
8427            .windows(4)
8428            .rposition(|w| w == OHDR_SIGNATURE)
8429            .expect("enclosing object header");
8430        let cksum_at = (at + needle.len()..bytes.len() - 4)
8431            .find(|&e| {
8432                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
8433                    == checksum_metadata(&bytes[ohdr..e])
8434            })
8435            .expect("object header checksum");
8436
8437        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
8438        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
8439        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
8440        std::fs::write(&path, &bytes).unwrap();
8441
8442        let file = H5File::open(&path).unwrap();
8443        let err = file
8444            .dataset("labels")
8445            .unwrap()
8446            .read_strings()
8447            .unwrap_err()
8448            .to_string();
8449        assert!(err.contains("zero width"), "got: {err}");
8450        std::fs::remove_file(&path).ok();
8451    }
8452
8453    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
8454    #[test]
8455    fn read_strings_rejects_a_non_string_dataset() {
8456        let path = temp_path("read_strings_numeric");
8457        {
8458            let file = H5File::create(&path).unwrap();
8459            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8460            ds.write_raw(&[1i32, 2, 3]).unwrap();
8461            file.close().unwrap();
8462        }
8463        let file = H5File::open(&path).unwrap();
8464        let err = file
8465            .dataset("nums")
8466            .unwrap()
8467            .read_strings()
8468            .unwrap_err()
8469            .to_string();
8470        assert!(err.contains("only for string datasets"), "got: {err}");
8471        std::fs::remove_file(&path).ok();
8472    }
8473
8474    // ---- issue #6: random updates to vlen string datasets ------------------
8475
8476    /// One element changes; the extent and every other element stay as they
8477    /// were, on a contiguous vlen dataset.
8478    #[test]
8479    fn write_vlen_strings_slice_replaces_one_element() {
8480        let path = temp_path("vlen_slice_contig");
8481        {
8482            let file = H5File::create(&path).unwrap();
8483            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8484                .unwrap();
8485            file.close().unwrap();
8486        }
8487        {
8488            let file = H5File::open_rw(&path).unwrap();
8489            file.dataset_writer("notes")
8490                .unwrap()
8491                .write_vlen_strings_slice(1, &["replacement"])
8492                .unwrap();
8493            file.close().unwrap();
8494        }
8495        let file = H5File::open(&path).unwrap();
8496        let ds = file.dataset("notes").unwrap();
8497        assert_eq!(ds.shape(), vec![4]);
8498        assert_eq!(
8499            ds.read_vlen_strings().unwrap(),
8500            vec!["a", "replacement", "c", "d"]
8501        );
8502        std::fs::remove_file(&path).ok();
8503    }
8504
8505    /// The same on an appendable chunked dataset, across a reopen, over a range
8506    /// that spans a chunk boundary.
8507    #[test]
8508    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
8509        let path = temp_path("vlen_slice_chunked");
8510        {
8511            let file = H5File::create(&path).unwrap();
8512            file.create_appendable_vlen_dataset("notes", 2, None)
8513                .unwrap();
8514            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
8515            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
8516            file.append_vlen_strings("notes", &refs).unwrap();
8517            file.close().unwrap();
8518        }
8519        {
8520            // Elements 1..4 cross the 2-element chunk boundary twice.
8521            let file = H5File::open_rw(&path).unwrap();
8522            file.dataset_writer("notes")
8523                .unwrap()
8524                .write_vlen_strings_slice(1, &["x", "y", "z"])
8525                .unwrap();
8526            file.close().unwrap();
8527        }
8528        let file = H5File::open(&path).unwrap();
8529        let ds = file.dataset("notes").unwrap();
8530        assert_eq!(ds.shape(), vec![6]);
8531        assert_eq!(
8532            ds.read_vlen_strings().unwrap(),
8533            vec!["v0", "x", "y", "z", "v4", "v5"]
8534        );
8535        std::fs::remove_file(&path).ok();
8536    }
8537
8538    /// Elements the append buffer still holds are not on disk yet; the
8539    /// update flushes them to their chunks first, so the flush at close has
8540    /// nothing left to write the pre-update reference over.
8541    #[test]
8542    fn write_vlen_strings_slice_updates_buffered_elements() {
8543        let path = temp_path("vlen_slice_buffered");
8544        {
8545            let file = H5File::create(&path).unwrap();
8546            file.create_appendable_vlen_dataset("notes", 4, None)
8547                .unwrap();
8548            // 3 of a 4-element chunk: all three stay in the append buffer.
8549            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8550            file.dataset_writer("notes")
8551                .unwrap()
8552                .write_vlen_strings_slice(1, &["patched"])
8553                .unwrap();
8554            file.close().unwrap();
8555        }
8556        let file = H5File::open(&path).unwrap();
8557        assert_eq!(
8558            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8559            vec!["a", "patched", "c"]
8560        );
8561        std::fs::remove_file(&path).ok();
8562    }
8563
8564    /// A range past the end is rejected before anything is written, and an
8565    /// empty batch costs the file nothing — without the early return it would
8566    /// still allocate and write an empty global-heap collection.
8567    #[test]
8568    fn write_vlen_strings_slice_checks_its_range() {
8569        let build = |name: &str, empty_call: bool| {
8570            let path = temp_path(name);
8571            let file = H5File::create(&path).unwrap();
8572            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8573            let ds = file.dataset_writer("notes").unwrap();
8574            let err = ds
8575                .write_vlen_strings_slice(1, &["x", "y"])
8576                .unwrap_err()
8577                .to_string();
8578            assert!(
8579                err.contains("outside the dataset's 2 elements"),
8580                "got: {err}"
8581            );
8582            if empty_call {
8583                ds.write_vlen_strings_slice(0, &[]).unwrap();
8584            }
8585            file.close().unwrap();
8586            path
8587        };
8588
8589        let with_empty = build("vlen_slice_range", true);
8590        let control = build("vlen_slice_range_control", false);
8591        assert_eq!(
8592            std::fs::metadata(&with_empty).unwrap().len(),
8593            std::fs::metadata(&control).unwrap().len(),
8594            "the rejected and empty calls must leave the file untouched"
8595        );
8596
8597        let file = H5File::open(&with_empty).unwrap();
8598        assert_eq!(
8599            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8600            vec!["a", "b"]
8601        );
8602        std::fs::remove_file(&with_empty).ok();
8603        std::fs::remove_file(&control).ok();
8604    }
8605
8606    /// The element offset is one-dimensional, so a multi-dimensional dataset is
8607    /// rejected rather than silently indexed along the first axis.
8608    #[test]
8609    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
8610        let path = temp_path("vlen_slice_2d");
8611        let file = H5File::create(&path).unwrap();
8612        let ds = file
8613            .new_dataset::<u8>()
8614            .datatype(DatatypeMessage::vlen_string_utf8())
8615            .shape([2, 3])
8616            .create("grid")
8617            .unwrap();
8618        let err = ds
8619            .write_vlen_strings_slice(0, &["x"])
8620            .unwrap_err()
8621            .to_string();
8622        assert!(err.contains("1-dimension datasets"), "got: {err}");
8623        file.close().unwrap();
8624        std::fs::remove_file(&path).ok();
8625    }
8626
8627    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
8628    /// declares the ASCII character set would mislabel the bytes.
8629    #[test]
8630    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
8631        let path = temp_path("vlen_slice_ascii");
8632        let file = H5File::create(&path).unwrap();
8633        let ds = file
8634            .new_dataset::<u8>()
8635            .datatype(DatatypeMessage::vlen_string_ascii())
8636            .shape([3])
8637            .create("notes")
8638            .unwrap();
8639        let err = ds
8640            .write_vlen_strings_slice(0, &["ok", "안녕"])
8641            .unwrap_err()
8642            .to_string();
8643        assert!(
8644            err.contains("string 1") && err.contains("is not ASCII"),
8645            "got: {err}"
8646        );
8647        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
8648        file.close().unwrap();
8649        std::fs::remove_file(&path).ok();
8650    }
8651
8652    /// A numeric dataset is rejected: its elements are not vlen references and
8653    /// writing one would corrupt the column.
8654    #[test]
8655    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
8656        let path = temp_path("vlen_slice_numeric");
8657        let file = H5File::create(&path).unwrap();
8658        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8659        ds.write_raw(&[1i32, 2, 3]).unwrap();
8660        let err = ds
8661            .write_vlen_strings_slice(0, &["x"])
8662            .unwrap_err()
8663            .to_string();
8664        assert!(
8665            err.contains("only for variable-length string datasets"),
8666            "got: {err}"
8667        );
8668        file.close().unwrap();
8669        std::fs::remove_file(&path).ok();
8670    }
8671
8672    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
8673
8674    /// Repeatedly replacing the same element must not grow the file per
8675    /// update: the collection each update supersedes is freed and the next
8676    /// update's collection lands in that block. Without the release every
8677    /// update costs another `H5HG_MINALLOC` (4096) bytes.
8678    #[test]
8679    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
8680        let size_after = |updates: usize| {
8681            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
8682            let file = H5File::create(&path).unwrap();
8683            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8684            let ds = file.dataset_writer("notes").unwrap();
8685            for i in 0..updates {
8686                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
8687                    .unwrap();
8688            }
8689            file.close().unwrap();
8690            let n = std::fs::metadata(&path).unwrap().len();
8691            let read = H5File::open(&path).unwrap();
8692            assert_eq!(
8693                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8694                vec![format!("update {}", updates - 1), "b".to_string()]
8695            );
8696            drop(read);
8697            std::fs::remove_file(&path).ok();
8698            n
8699        };
8700
8701        // The allocator settles once a freed block is available to reuse, so
8702        // every count past that produces the same file.
8703        let settled = size_after(3);
8704        assert_eq!(size_after(20), settled, "20 updates against 3");
8705        assert_eq!(size_after(50), settled, "50 updates against 3");
8706    }
8707
8708    /// An empty string is stored as a real heap object under a reference whose
8709    /// sequence length is zero, so the release must go by the address, not the
8710    /// length — a length test strands the object and its collection forever.
8711    #[test]
8712    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
8713        let size_after = |updates: usize| {
8714            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
8715            let file = H5File::create(&path).unwrap();
8716            file.write_vlen_strings("notes", &["", "b"]).unwrap();
8717            let ds = file.dataset_writer("notes").unwrap();
8718            for _ in 0..updates {
8719                ds.write_vlen_strings_slice(0, &[""]).unwrap();
8720            }
8721            file.close().unwrap();
8722            let n = std::fs::metadata(&path).unwrap().len();
8723            let read = H5File::open(&path).unwrap();
8724            assert_eq!(
8725                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8726                vec!["".to_string(), "b".to_string()]
8727            );
8728            drop(read);
8729            std::fs::remove_file(&path).ok();
8730            n
8731        };
8732
8733        let settled = size_after(3);
8734        assert_eq!(size_after(20), settled, "20 empty updates against 3");
8735        assert_eq!(size_after(50), settled, "50 empty updates against 3");
8736    }
8737
8738    /// The elements the update does not name keep their strings, so freeing
8739    /// the superseded objects must not disturb the collection's survivors.
8740    #[test]
8741    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
8742        let path = temp_path("vlen_slice_heap_survivors");
8743        {
8744            let file = H5File::create(&path).unwrap();
8745            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8746                .unwrap();
8747            let ds = file.dataset_writer("notes").unwrap();
8748            // Two updates inside the one collection the create wrote, so the
8749            // second reads a collection the first already rewrote.
8750            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
8751            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
8752            file.close().unwrap();
8753        }
8754        let file = H5File::open(&path).unwrap();
8755        assert_eq!(
8756            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8757            vec!["a", "B", "c", "D"]
8758        );
8759        std::fs::remove_file(&path).ok();
8760    }
8761
8762    /// Replacing every element of a chunked dataset empties the collection the
8763    /// append wrote, and the file must still read back correctly after its
8764    /// block goes to the allocator.
8765    #[test]
8766    fn write_vlen_strings_slice_frees_an_emptied_collection() {
8767        let path = temp_path("vlen_slice_heap_emptied");
8768        {
8769            let file = H5File::create(&path).unwrap();
8770            file.create_appendable_vlen_dataset("notes", 2, None)
8771                .unwrap();
8772            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
8773                .unwrap();
8774            file.close().unwrap();
8775        }
8776        {
8777            let file = H5File::open_rw(&path).unwrap();
8778            file.dataset_writer("notes")
8779                .unwrap()
8780                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
8781                .unwrap();
8782            file.close().unwrap();
8783        }
8784        let file = H5File::open(&path).unwrap();
8785        assert_eq!(
8786            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8787            vec!["w", "x", "y", "z"]
8788        );
8789        std::fs::remove_file(&path).ok();
8790    }
8791
8792    /// A collection larger than the 4096-byte minimum must keep its size when
8793    /// an object leaves it. Re-encoding at the natural size instead shrinks
8794    /// what the header declares, so the block's tail stops being part of the
8795    /// collection and the eventual free returns less than was allocated —
8796    /// stranding the difference on every cycle.
8797    #[test]
8798    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
8799        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
8800        let size_after = |cycles: usize| {
8801            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
8802            let file = H5File::create(&path).unwrap();
8803            let seed: Vec<String> = "abcd".chars().map(big).collect();
8804            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
8805            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
8806            // is one collection well above it.
8807            file.write_vlen_strings("notes", &refs).unwrap();
8808            let ds = file.dataset_writer("notes").unwrap();
8809            for _ in 0..cycles {
8810                // Partially empty the collection, then finish it off: the
8811                // block is freed only after it has been rewritten once.
8812                let head = big('x');
8813                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
8814                let tail: Vec<String> = "yzw".chars().map(big).collect();
8815                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
8816                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
8817            }
8818            file.close().unwrap();
8819            let n = std::fs::metadata(&path).unwrap().len();
8820            let read = H5File::open(&path).unwrap();
8821            assert_eq!(
8822                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8823                vec![big('x'), big('y'), big('z'), big('w')]
8824            );
8825            drop(read);
8826            std::fs::remove_file(&path).ok();
8827            n
8828        };
8829
8830        let settled = size_after(4);
8831        assert_eq!(size_after(30), settled, "30 cycles against 4");
8832    }
8833
8834    /// An element still in the append buffer has never been on disk, so its
8835    /// superseded object has to be found in the buffer or it is stranded.
8836    #[test]
8837    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
8838        let path = temp_path("vlen_slice_heap_buffered");
8839        let file = H5File::create(&path).unwrap();
8840        file.create_appendable_vlen_dataset("notes", 4, None)
8841            .unwrap();
8842        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8843        let ds = file.dataset_writer("notes").unwrap();
8844        for i in 0..20 {
8845            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
8846                .unwrap();
8847        }
8848        file.close().unwrap();
8849
8850        let file = H5File::open(&path).unwrap();
8851        assert_eq!(
8852            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8853            vec!["a", "patch 19", "c"]
8854        );
8855        let size = std::fs::metadata(&path).unwrap().len();
8856        std::fs::remove_file(&path).ok();
8857        assert!(
8858            size < 20 * 4096,
8859            "20 buffered updates left {size} bytes, one collection per update"
8860        );
8861    }
8862
8863    /// Regression: a typed `write_slice` into rows the append buffer still
8864    /// held wrote the chunks, and the flush at close wrote the stale buffered
8865    /// rows back over it — write 99, read 50. The slice now flushes the
8866    /// buffer first, making the chunks the single authority for those rows.
8867    #[test]
8868    fn write_slice_into_the_buffered_tail_survives_close() {
8869        let path = temp_path("slice_into_buffered_tail");
8870        {
8871            let file = H5File::create(&path).unwrap();
8872            let ds = file
8873                .new_dataset::<i32>()
8874                .shape([0])
8875                .chunk(&[4])
8876                .max_shape(&[None])
8877                .create("d")
8878                .unwrap();
8879            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
8880            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
8881            ds.write_slice(&[4], &[1], &[99]).unwrap();
8882            file.close().unwrap();
8883        }
8884        {
8885            let file = H5File::open(&path).unwrap();
8886            let ds = file.dataset("d").unwrap();
8887            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
8888        }
8889        std::fs::remove_file(&path).ok();
8890    }
8891
8892    /// Extending a dataset while appends sit in the buffer must not move
8893    /// them: the buffer records the absolute row its frames belong to, so
8894    /// the flush at close lands them there, and the grown region reads as
8895    /// fill.
8896    #[test]
8897    fn extend_does_not_move_buffered_appends() {
8898        let path = temp_path("extend_keeps_buffered_rows");
8899        {
8900            let file = H5File::create(&path).unwrap();
8901            let ds = file
8902                .new_dataset::<i32>()
8903                .shape([0])
8904                .chunk(&[4])
8905                .max_shape(&[None])
8906                .create("d")
8907                .unwrap();
8908            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
8909            ds.extend(&[10]).unwrap();
8910            file.close().unwrap();
8911        }
8912        {
8913            let file = H5File::open(&path).unwrap();
8914            let ds = file.dataset("d").unwrap();
8915            assert_eq!(
8916                ds.read_raw::<i32>().unwrap(),
8917                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
8918            );
8919        }
8920        std::fs::remove_file(&path).ok();
8921    }
8922
8923    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
8924    /// dimensions) buffered fine but close() failed "not a chunked dataset"
8925    /// and lost the buffered rows — the append's chunk writes required the
8926    /// extensible-array index. They now go through the index-generic
8927    /// hyperslab engine.
8928    #[test]
8929    fn append_to_a_btree_v2_dataset_survives_close() {
8930        let path = temp_path("append_bt2_close");
8931        {
8932            let file = H5File::create(&path).unwrap();
8933            let ds = file
8934                .new_dataset::<i32>()
8935                .shape([0, 3])
8936                .chunk(&[4, 3])
8937                .max_shape(&[None, None])
8938                .create("d")
8939                .unwrap();
8940            // One buffered row, then a batch that crosses the chunk
8941            // boundary: 4 rows fill chunk band 0, one row stays buffered
8942            // for the flush at close.
8943            ds.append(&[1, 2, 3]).unwrap();
8944            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
8945            file.close().unwrap();
8946        }
8947        {
8948            let file = H5File::open(&path).unwrap();
8949            let ds = file.dataset("d").unwrap();
8950            assert_eq!(ds.shape(), vec![5, 3]);
8951            assert_eq!(
8952                ds.read_raw::<i32>().unwrap(),
8953                (1..=15).collect::<Vec<i32>>()
8954            );
8955        }
8956        std::fs::remove_file(&path).ok();
8957    }
8958
8959    /// A chunk row narrower than the frame row is legal geometry (libhdf5
8960    /// creates it); appended frames must be scattered across the row's
8961    /// tiles at the chunk stride, not packed at the frame stride.
8962    #[test]
8963    fn append_scatters_frames_across_narrow_chunk_tiles() {
8964        let path = temp_path("append_narrow_chunks");
8965        {
8966            let file = H5File::create(&path).unwrap();
8967            let ds = file
8968                .new_dataset::<i32>()
8969                .shape([0, 8])
8970                .chunk(&[2, 4])
8971                .max_shape(&[None, Some(8)])
8972                .create("d")
8973                .unwrap();
8974            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
8975            // row 2 is flushed partial at close.
8976            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
8977            file.close().unwrap();
8978        }
8979        {
8980            let file = H5File::open(&path).unwrap();
8981            let ds = file.dataset("d").unwrap();
8982            assert_eq!(ds.shape(), vec![3, 8]);
8983            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
8984        }
8985        std::fs::remove_file(&path).ok();
8986    }
8987
8988    /// A fixed-array dataset has no room to grow: appending must surface an
8989    /// error naming the chunk grid, not lose rows silently. (Before the
8990    /// index-generic append it failed as "not a chunked dataset".)
8991    #[test]
8992    fn append_to_a_full_fixed_array_dataset_errors() {
8993        let path = temp_path("append_fa_errors");
8994        let file = H5File::create(&path).unwrap();
8995        let ds = file
8996            .new_dataset::<i32>()
8997            .shape([4, 3])
8998            .chunk(&[2, 3])
8999            .create("d")
9000            .unwrap();
9001        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
9002        assert!(
9003            err.to_string().contains("chunk grid"),
9004            "unexpected error: {err}"
9005        );
9006        file.close().unwrap();
9007        std::fs::remove_file(&path).ok();
9008    }
9009
9010    /// A finite max_shape above the current shape used to be dropped on the
9011    /// fixed-array path: the array was sized from the current dims and the
9012    /// stored dataspace had no maximum, so growth failed. The array is now
9013    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
9014    /// fixed-max dataset appends up to its maximum and roundtrips.
9015    #[test]
9016    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
9017        let path = temp_path("fa_growable_dim0");
9018        {
9019            let file = H5File::create(&path).unwrap();
9020            let ds = file
9021                .new_dataset::<i32>()
9022                .shape([4, 3])
9023                .chunk(&[2, 3])
9024                .max_shape(&[Some(10), Some(3)])
9025                .create("d")
9026                .unwrap();
9027            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9028            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
9029            file.close().unwrap();
9030        }
9031        {
9032            let file = H5File::open(&path).unwrap();
9033            let ds = file.dataset("d").unwrap();
9034            assert_eq!(ds.shape(), vec![6, 3]);
9035            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
9036        }
9037        std::fs::remove_file(&path).ok();
9038    }
9039
9040    /// The multiplier-dimension boundary: growing a dimension other than 0
9041    /// changes the current chunk grid but not the index grid. Chunk slots
9042    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
9043    /// chunks written before the extend are looked up under different
9044    /// indices after it.
9045    #[test]
9046    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
9047        let path = temp_path("fa_growable_dim1");
9048        {
9049            let file = H5File::create(&path).unwrap();
9050            let ds = file
9051                .new_dataset::<i32>()
9052                .shape([4, 3])
9053                .chunk(&[2, 3])
9054                .max_shape(&[Some(4), Some(9)])
9055                .create("d")
9056                .unwrap();
9057            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9058            ds.extend(&[4, 6]).unwrap();
9059            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
9060                .unwrap();
9061            file.close().unwrap();
9062        }
9063        {
9064            let file = H5File::open(&path).unwrap();
9065            let ds = file.dataset("d").unwrap();
9066            assert_eq!(ds.shape(), vec![4, 6]);
9067            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
9068            // then [12 + r*3 ..) from the second.
9069            let mut expect = Vec::new();
9070            for r in 0i32..4 {
9071                expect.extend((r * 3)..(r * 3 + 3));
9072                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
9073            }
9074            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
9075        }
9076        std::fs::remove_file(&path).ok();
9077    }
9078
9079    /// Growth boundaries: past the stored maximum is rejected, and a dataset
9080    /// without a stored maximum is fixed at its extent (libhdf5 defaults
9081    /// maxdims to dims at creation).
9082    #[test]
9083    fn extend_beyond_the_maximum_is_rejected() {
9084        let path = temp_path("extend_beyond_max");
9085        let file = H5File::create(&path).unwrap();
9086        let ds = file
9087            .new_dataset::<i32>()
9088            .shape([4, 3])
9089            .chunk(&[2, 3])
9090            .max_shape(&[Some(6), Some(3)])
9091            .create("d")
9092            .unwrap();
9093        ds.extend(&[6, 3]).unwrap();
9094        let err = ds.extend(&[8, 3]).unwrap_err();
9095        assert!(
9096            err.to_string().contains("exceeds the maximum"),
9097            "unexpected error: {err}"
9098        );
9099        file.close().unwrap();
9100        std::fs::remove_file(&path).ok();
9101    }
9102
9103    /// An unlimited dimension other than 0 has no fixed linear slot without
9104    /// libhdf5's extensible-array swizzling; `chunk_grid::linear_index` now
9105    /// implements that swizzle for any dimension, so this creates cleanly
9106    /// and every extend keeps writing new chunks to new slots, never
9107    /// re-addressing one already on disk.
9108    #[test]
9109    fn builder_accepts_an_unlimited_inner_dimension() {
9110        let path = temp_path("unlimited_inner_dim");
9111        let file = H5File::create(&path).unwrap();
9112        let ds = file
9113            .new_dataset::<i32>()
9114            .shape([4, 0])
9115            .chunk(&[2, 2])
9116            .max_shape(&[Some(4), None])
9117            .create("d")
9118            .unwrap();
9119        assert_eq!(ds.shape(), vec![4, 0]);
9120
9121        // Write, then extend and write again: if the linear index were
9122        // recomputed from the *current* extent instead of the maximum one,
9123        // the second extend would shift every slot number and the first
9124        // write's chunks would decode under the wrong coordinates below.
9125        ds.extend(&[4, 2]).unwrap();
9126        ds.write_slice(&[0, 0], &[4, 2], &[1, 2, 3, 4, 5, 6, 7, 8])
9127            .unwrap();
9128        ds.extend(&[4, 4]).unwrap();
9129        ds.write_slice(&[0, 2], &[4, 2], &[9, 10, 11, 12, 13, 14, 15, 16])
9130            .unwrap();
9131
9132        file.close().unwrap();
9133        let file = H5File::open(&path).unwrap();
9134        let ds = file.dataset("d").unwrap();
9135        assert_eq!(
9136            ds.read_slice::<i32>(&[0, 0], &[4, 4]).unwrap(),
9137            vec![1, 2, 9, 10, 3, 4, 11, 12, 5, 6, 13, 14, 7, 8, 15, 16]
9138        );
9139        std::fs::remove_file(&path).ok();
9140    }
9141
9142    /// Regression: a chunk wider than a fixed max dimension used to be
9143    /// accepted, and appends then packed rows at the chunk stride — writing
9144    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
9145    /// geometry at create (`H5D__chunk_construct`); so do we now.
9146    #[test]
9147    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
9148        let path = temp_path("builder_chunk_wider_than_max");
9149        let file = H5File::create(&path).unwrap();
9150        let err = match file
9151            .new_dataset::<i32>()
9152            .shape([0, 2])
9153            .chunk(&[2, 4])
9154            .max_shape(&[None, Some(2)])
9155            .create("v5")
9156        {
9157            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
9158            Err(e) => e,
9159        };
9160        assert!(
9161            err.to_string().contains("maximum dimension size"),
9162            "unexpected error: {err}"
9163        );
9164        file.close().unwrap();
9165        std::fs::remove_file(&path).ok();
9166    }
9167
9168    /// Boundary: `fits in T` vs `does not fit in T`, for both the too-large
9169    /// (u64::MAX → i64) and the negative-to-unsigned (−1 → u32) directions.
9170    #[test]
9171    fn numeric_int_checked_conversion_boundaries() {
9172        let path = temp_path("numeric_int_bounds");
9173        {
9174            let file = H5File::create(&path).unwrap();
9175            let ds = file.new_dataset::<u64>().shape([2]).create("u").unwrap();
9176            ds.write_raw(&[1u64, u64::MAX]).unwrap();
9177            let ds = file.new_dataset::<i32>().shape([2]).create("i").unwrap();
9178            ds.write_raw(&[-1i32, 5]).unwrap();
9179            file.close().unwrap();
9180        }
9181        let file = H5File::open(&path).unwrap();
9182
9183        let u = file.dataset("u").unwrap();
9184        assert_eq!(u.read_numeric_as::<u64>().unwrap(), vec![1, u64::MAX]);
9185        assert_eq!(
9186            u.read_numeric_as::<i128>().unwrap(),
9187            vec![1, i128::from(u64::MAX)]
9188        );
9189        let err = u.read_numeric_as::<i64>().unwrap_err();
9190        assert!(
9191            err.to_string()
9192                .contains("value 18446744073709551615 at element 1 does not fit in i64"),
9193            "unexpected error: {err}"
9194        );
9195
9196        let i = file.dataset("i").unwrap();
9197        assert_eq!(i.read_numeric_as::<i64>().unwrap(), vec![-1, 5]);
9198        let err = i.read_numeric_as::<u32>().unwrap_err();
9199        assert!(
9200            err.to_string()
9201                .contains("value -1 at element 0 does not fit in u32"),
9202            "unexpected error: {err}"
9203        );
9204        std::fs::remove_file(&path).ok();
9205    }
9206
9207    /// Boundary: f32 → f64 is exact widening; f64 → f32 is rejected.
9208    #[test]
9209    fn numeric_float_widening_only() {
9210        let path = temp_path("numeric_float");
9211        {
9212            let file = H5File::create(&path).unwrap();
9213            let ds = file.new_dataset::<f32>().shape([2]).create("f4").unwrap();
9214            ds.write_raw(&[1.5f32, -2.25]).unwrap();
9215            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9216            ds.write_raw(&[3.75f64]).unwrap();
9217            file.close().unwrap();
9218        }
9219        let file = H5File::open(&path).unwrap();
9220
9221        let f4 = file.dataset("f4").unwrap();
9222        assert_eq!(f4.read_numeric_as::<f32>().unwrap(), vec![1.5, -2.25]);
9223        assert_eq!(f4.read_numeric_as::<f64>().unwrap(), vec![1.5, -2.25]);
9224
9225        let f8 = file.dataset("f8").unwrap();
9226        assert_eq!(f8.read_numeric_as::<f64>().unwrap(), vec![3.75]);
9227        let err = f8.read_numeric_as::<f32>().unwrap_err();
9228        assert!(
9229            err.to_string().contains("narrowing"),
9230            "unexpected error: {err}"
9231        );
9232        std::fs::remove_file(&path).ok();
9233    }
9234
9235    /// Boundary: cross-class conversions (float ↔ integer) are rejected in
9236    /// both directions, and a non-numeric datatype is rejected at classify.
9237    #[test]
9238    fn numeric_cross_class_and_non_numeric_rejected() {
9239        let path = temp_path("numeric_cross_class");
9240        {
9241            let file = H5File::create(&path).unwrap();
9242            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9243            ds.write_raw(&[1.0f64]).unwrap();
9244            let ds = file.new_dataset::<i32>().shape([1]).create("i4").unwrap();
9245            ds.write_raw(&[7i32]).unwrap();
9246            file.write_vlen_strings("s", &["a", "b"]).unwrap();
9247            file.close().unwrap();
9248        }
9249        let file = H5File::open(&path).unwrap();
9250
9251        let err = file
9252            .dataset("f8")
9253            .unwrap()
9254            .read_numeric_as::<i64>()
9255            .unwrap_err();
9256        assert!(
9257            err.to_string().contains("floating-point dataset as i64"),
9258            "unexpected error: {err}"
9259        );
9260        let err = file
9261            .dataset("i4")
9262            .unwrap()
9263            .read_numeric_as::<f64>()
9264            .unwrap_err();
9265        assert!(
9266            err.to_string().contains("integer dataset as f64"),
9267            "unexpected error: {err}"
9268        );
9269        let err = file
9270            .dataset("s")
9271            .unwrap()
9272            .read_numeric_as::<i64>()
9273            .unwrap_err();
9274        assert!(
9275            err.to_string().contains("is not numeric"),
9276            "unexpected error: {err}"
9277        );
9278        std::fs::remove_file(&path).ok();
9279    }
9280
9281    /// Boundary: big-endian sources decode per the datatype's byte order.
9282    /// Unit-level (the writer only emits little-endian): feed `convert` a
9283    /// big-endian datatype plus big-endian bytes directly.
9284    #[test]
9285    fn numeric_big_endian_decode() {
9286        use super::numeric;
9287        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
9288
9289        let dt = DatatypeMessage::FixedPoint {
9290            size: 4,
9291            byte_order: ByteOrder::BigEndian,
9292            signed: true,
9293            bit_offset: 0,
9294            bit_precision: 32,
9295        };
9296        let mut raw = Vec::new();
9297        raw.extend_from_slice(&(-2i32).to_be_bytes());
9298        raw.extend_from_slice(&(100_000i32).to_be_bytes());
9299        let kind = numeric::classify(&dt).unwrap();
9300        assert_eq!(
9301            numeric::convert::<i64>(kind, &raw).unwrap(),
9302            vec![-2, 100_000]
9303        );
9304
9305        let dt = DatatypeMessage::FloatingPoint {
9306            size: 8,
9307            byte_order: ByteOrder::BigEndian,
9308            sign_location: 63,
9309            bit_offset: 0,
9310            bit_precision: 64,
9311            exponent_location: 52,
9312            exponent_size: 11,
9313            mantissa_location: 0,
9314            mantissa_size: 52,
9315            exponent_bias: 1023,
9316        };
9317        let raw = (-2.25f64).to_be_bytes();
9318        let kind = numeric::classify(&dt).unwrap();
9319        assert_eq!(numeric::convert::<f64>(kind, &raw).unwrap(), vec![-2.25]);
9320    }
9321
9322    /// `H5Attribute::read_numeric` validates the stored datatype before
9323    /// reinterpreting bytes: cross-width, cross-class, and non-numeric
9324    /// attributes error instead of returning bit-garbage, while the exact
9325    /// type and the HBool / complex-compound paths keep working.
9326    #[test]
9327    fn attr_read_numeric_validates_datatype() {
9328        use crate::types::{Complex64, HBool, VarLenUnicode};
9329        let path = temp_path("attr_read_numeric_validate");
9330        {
9331            let file = H5File::create(&path).unwrap();
9332            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9333            ds.write_raw(&[1.0f32; 2]).unwrap();
9334            let a = ds.new_attr::<f64>().shape(()).create("f8").unwrap();
9335            a.write_numeric(&1.5f64).unwrap();
9336            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9337            a.write_numeric(&-7i32).unwrap();
9338            let a = ds.new_attr::<HBool>().shape(()).create("b").unwrap();
9339            a.write_numeric(&HBool::from(true)).unwrap();
9340            let a = ds.new_attr::<Complex64>().shape(()).create("z").unwrap();
9341            a.write_numeric(&Complex64 { re: 1.0, im: -2.0 }).unwrap();
9342            let a = ds
9343                .new_attr::<VarLenUnicode>()
9344                .shape(())
9345                .create("s")
9346                .unwrap();
9347            a.write_scalar(&VarLenUnicode("text".into())).unwrap();
9348            file.close().unwrap();
9349        }
9350        let file = H5File::open(&path).unwrap();
9351        let ds = file.dataset("d").unwrap();
9352
9353        let f8 = ds.attr("f8").unwrap();
9354        assert_eq!(f8.read_numeric::<f64>().unwrap(), 1.5);
9355        // Previously returned the low half of the f64 image as an f32.
9356        let err = f8.read_numeric::<f32>().unwrap_err();
9357        assert!(
9358            err.to_string().contains("read_numeric_as"),
9359            "unexpected error: {err}"
9360        );
9361        assert!(f8.read_numeric::<i64>().is_err());
9362
9363        let i4 = ds.attr("i4").unwrap();
9364        assert_eq!(i4.read_numeric::<i32>().unwrap(), -7);
9365        assert!(i4.read_numeric::<u32>().is_err());
9366
9367        assert!(bool::from(
9368            ds.attr("b").unwrap().read_numeric::<HBool>().unwrap()
9369        ));
9370        let z = ds.attr("z").unwrap().read_numeric::<Complex64>().unwrap();
9371        assert_eq!((z.re, z.im), (1.0, -2.0));
9372
9373        // A vlen string attribute: read_numeric used to transmute the heap
9374        // reference bytes into the requested type.
9375        let s = ds.attr("s").unwrap();
9376        assert!(s.read_numeric::<f64>().is_err());
9377        assert!(s.read_numeric_as::<f64>().is_err());
9378        std::fs::remove_file(&path).ok();
9379    }
9380
9381    /// The attribute conversion read applies the dataset rules: checked
9382    /// int → int naming index and value on overflow, widening-only floats,
9383    /// cross-class rejected; an array attribute converts every element.
9384    #[test]
9385    fn attr_read_numeric_as_converts() {
9386        let path = temp_path("attr_read_numeric_as");
9387        {
9388            let file = H5File::create(&path).unwrap();
9389            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9390            ds.write_raw(&[1.0f32; 2]).unwrap();
9391            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9392            a.write_numeric(&-7i32).unwrap();
9393            let a = ds.new_attr::<u64>().shape(()).create("u8max").unwrap();
9394            a.write_numeric(&u64::MAX).unwrap();
9395            let a = ds.new_attr::<i16>().shape([3]).create("arr").unwrap();
9396            a.write_array(&[1i16, -2, 3]).unwrap();
9397            file.close().unwrap();
9398        }
9399        let file = H5File::open(&path).unwrap();
9400        let ds = file.dataset("d").unwrap();
9401        assert_eq!(
9402            ds.attr("i4").unwrap().read_numeric_as::<i64>().unwrap(),
9403            vec![-7]
9404        );
9405        let err = ds
9406            .attr("u8max")
9407            .unwrap()
9408            .read_numeric_as::<i64>()
9409            .unwrap_err();
9410        assert!(
9411            err.to_string().contains("does not fit in i64"),
9412            "unexpected error: {err}"
9413        );
9414        assert_eq!(
9415            ds.attr("arr").unwrap().read_numeric_as::<i32>().unwrap(),
9416            vec![1, -2, 3]
9417        );
9418        assert!(ds.attr("i4").unwrap().read_numeric_as::<f64>().is_err());
9419        std::fs::remove_file(&path).ok();
9420    }
9421
9422    /// The hyperslab variant applies the same conversion to a sub-selection.
9423    #[test]
9424    fn numeric_slice_conversion() {
9425        let path = temp_path("numeric_slice");
9426        {
9427            let file = H5File::create(&path).unwrap();
9428            let ds = file.new_dataset::<i16>().shape([2, 3]).create("m").unwrap();
9429            ds.write_raw(&[1i16, 2, 3, 4, 5, 6]).unwrap();
9430            file.close().unwrap();
9431        }
9432        let file = H5File::open(&path).unwrap();
9433        let m = file.dataset("m").unwrap();
9434        assert_eq!(
9435            m.read_numeric_slice_as::<i32>(&[0, 1], &[2, 2]).unwrap(),
9436            vec![2, 3, 5, 6]
9437        );
9438        std::fs::remove_file(&path).ok();
9439    }
9440
9441    /// A NULL dataspace round-trips through the public API as `is_null() ==
9442    /// true`, `shape() == []`, and `read_raw_bytes()` empty — and stays
9443    /// distinguishable from a scalar dataset, which shares the same empty
9444    /// `shape()` but holds exactly one element.
9445    #[test]
9446    fn null_dataspace_distinct_from_scalar() {
9447        let path = temp_path("null_vs_scalar");
9448        {
9449            let file = H5File::create(&path).unwrap();
9450            file.new_dataset::<i32>().null().create("empty").unwrap();
9451            let scalar = file.new_dataset::<i32>().scalar().create("scalar").unwrap();
9452            scalar.write_raw(&[42i32]).unwrap();
9453            file.close().unwrap();
9454        }
9455        let file = H5File::open(&path).unwrap();
9456
9457        let empty = file.dataset("empty").unwrap();
9458        assert!(empty.is_null());
9459        assert_eq!(empty.shape(), Vec::<usize>::new());
9460        assert_eq!(empty.total_elements(), 0);
9461        assert_eq!(empty.read_raw_bytes().unwrap(), Vec::<u8>::new());
9462
9463        let scalar = file.dataset("scalar").unwrap();
9464        assert!(!scalar.is_null());
9465        assert_eq!(scalar.shape(), Vec::<usize>::new());
9466        assert_eq!(scalar.total_elements(), 1);
9467        assert_eq!(scalar.read_raw::<i32>().unwrap(), vec![42]);
9468
9469        std::fs::remove_file(&path).ok();
9470    }
9471
9472    /// A NULL dataspace dataset rejects writes outright — there is nothing
9473    /// to write into — rather than silently accepting a scalar-shaped
9474    /// write against unallocated storage.
9475    #[test]
9476    fn null_dataspace_rejects_writes() {
9477        let path = temp_path("null_write_rejected");
9478        let file = H5File::create(&path).unwrap();
9479        let ds = file.new_dataset::<i32>().null().create("empty").unwrap();
9480        assert!(ds.write_raw(&[1i32]).is_err());
9481        assert!(ds.write_raw_bytes(&[0u8; 4]).is_err());
9482        file.close().unwrap();
9483        std::fs::remove_file(&path).ok();
9484    }
9485
9486    /// `.null()` combined with `.chunk()` or a fill value is rejected at
9487    /// `create()` rather than silently dropping the conflicting option —
9488    /// a NULL dataspace can never be chunked or filtered upstream.
9489    #[test]
9490    fn null_dataspace_rejects_chunking_and_fill_value() {
9491        let path = temp_path("null_chunk_rejected");
9492        let file = H5File::create(&path).unwrap();
9493        assert!(file
9494            .new_dataset::<i32>()
9495            .null()
9496            .chunk(&[4])
9497            .create("a")
9498            .is_err());
9499        assert!(file
9500            .new_dataset::<i32>()
9501            .null()
9502            .fill_value(7i32)
9503            .create("b")
9504            .is_err());
9505        file.close().unwrap();
9506        std::fs::remove_file(&path).ok();
9507    }
9508
9509    /// A committed type is resolved before the dataset is created, so a name
9510    /// that is not one — or one paired with object references, which would
9511    /// make the stored type disagree with the payload — leaves no dataset
9512    /// behind.
9513    #[test]
9514    fn a_committed_type_that_cannot_be_shared_creates_no_dataset() {
9515        use crate::format::messages::datatype::DatatypeMessage;
9516
9517        let path = temp_path("committed_refused");
9518        let file = H5File::create(&path).unwrap();
9519        file.commit_datatype("t", DatatypeMessage::i32_type())
9520            .unwrap();
9521
9522        assert!(file
9523            .new_dataset::<i32>()
9524            .committed_type("absent")
9525            .shape([2usize])
9526            .create("a")
9527            .is_err());
9528        assert!(file
9529            .new_dataset::<u64>()
9530            .committed_type("t")
9531            .object_references()
9532            .shape([2usize])
9533            .create("b")
9534            .is_err());
9535        // A dataset already exists under that name, so the type cannot take
9536        // it either.
9537        file.new_dataset::<i32>()
9538            .shape([2usize])
9539            .create("taken")
9540            .unwrap();
9541        assert!(file
9542            .commit_datatype("taken", DatatypeMessage::i32_type())
9543            .is_err());
9544        assert!(file
9545            .commit_datatype("t", DatatypeMessage::f64_type())
9546            .is_err());
9547
9548        assert_eq!(file.dataset_names(), vec!["taken".to_string()]);
9549        assert_eq!(file.named_datatype_names(), vec!["t".to_string()]);
9550        file.close().unwrap();
9551        std::fs::remove_file(&path).ok();
9552    }
9553
9554    /// Deleting the group that named a committed datatype takes the name with
9555    /// it: nothing in the file reaches the type, so it is not written, the
9556    /// name is free again, and it can no longer be shared by that name.
9557    #[test]
9558    fn deleting_a_group_takes_the_committed_datatypes_it_named() {
9559        use crate::format::messages::datatype::DatatypeMessage;
9560
9561        let path = temp_path("committed_group_deleted");
9562        {
9563            let file = H5File::create(&path).unwrap();
9564            let types = file.create_group("types").unwrap();
9565            types
9566                .commit_datatype("t", DatatypeMessage::i32_type())
9567                .unwrap();
9568            assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9569
9570            file.delete_group("types").unwrap();
9571            assert!(file.named_datatype_names().is_empty());
9572            assert!(file
9573                .new_dataset::<i32>()
9574                .committed_type("types/t")
9575                .shape([2usize])
9576                .create("d")
9577                .is_err());
9578
9579            // The name is free, so a new group may take it back.
9580            let types = file.create_group("types").unwrap();
9581            types
9582                .commit_datatype("t", DatatypeMessage::f64_type())
9583                .unwrap();
9584            file.close().unwrap();
9585        }
9586        let file = H5File::open(&path).unwrap();
9587        assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9588        assert_eq!(
9589            file.named_datatype("types/t").unwrap().datatype().unwrap(),
9590            crate::format::messages::datatype::DatatypeMessage::f64_type()
9591        );
9592        drop(file);
9593        std::fs::remove_file(&path).ok();
9594    }
9595
9596    /// The layout class the reader sees for `name`, plus the image a compact
9597    /// layout carries — what distinguishes compact storage from contiguous
9598    /// storage that happens to hold the same bytes.
9599    fn compact_image(path: &std::path::Path, name: &str) -> Option<Vec<u8>> {
9600        use crate::format::messages::data_layout::DataLayoutMessage;
9601        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9602        match &reader.dataset_info(name).unwrap().layout {
9603            DataLayoutMessage::Compact { data } => Some(data.clone()),
9604            other => panic!("{name}: expected a compact layout, got {other:?}"),
9605        }
9606    }
9607
9608    /// `.compact()` puts the raw data inside the data layout message: the
9609    /// dataset has no data block of its own, and the image the layout carries
9610    /// is what a read returns.
9611    #[test]
9612    fn a_compact_dataset_stores_its_data_in_the_layout_message() {
9613        let path = temp_path("compact_roundtrip");
9614        let values: Vec<i32> = (0..16).collect();
9615        {
9616            let file = H5File::create(&path).unwrap();
9617            file.new_dataset::<i32>()
9618                .shape([16usize])
9619                .compact()
9620                .create("d")
9621                .unwrap()
9622                .write_raw(&values)
9623                .unwrap();
9624            file.close().unwrap();
9625        }
9626
9627        let image = compact_image(&path, "d").unwrap();
9628        assert_eq!(
9629            image,
9630            values
9631                .iter()
9632                .flat_map(|v| v.to_le_bytes())
9633                .collect::<Vec<_>>()
9634        );
9635
9636        let file = H5File::open(&path).unwrap();
9637        let ds = file.dataset("d").unwrap();
9638        assert_eq!(ds.shape(), vec![16]);
9639        assert_eq!(ds.read_raw::<i32>().unwrap(), values);
9640        std::fs::remove_file(&path).ok();
9641    }
9642
9643    /// A compact dataset created inside a group is linked from that group,
9644    /// not from the root: its create path goes through the same parent
9645    /// resolution every other layout uses.
9646    #[test]
9647    fn a_compact_dataset_lands_in_its_group() {
9648        let path = temp_path("compact_in_group");
9649        {
9650            let file = H5File::create(&path).unwrap();
9651            let g = file.root_group().create_group("g").unwrap();
9652            g.new_dataset::<u8>()
9653                .shape([4usize])
9654                .compact()
9655                .create("d")
9656                .unwrap()
9657                .write_raw(&[1u8, 2, 3, 4])
9658                .unwrap();
9659            file.close().unwrap();
9660        }
9661        assert_eq!(compact_image(&path, "/g/d").unwrap(), vec![1u8, 2, 3, 4]);
9662
9663        let file = H5File::open(&path).unwrap();
9664        assert_eq!(
9665            file.dataset("/g/d").unwrap().read_raw::<u8>().unwrap(),
9666            vec![1u8, 2, 3, 4]
9667        );
9668        std::fs::remove_file(&path).ok();
9669    }
9670
9671    /// A compact dataset's storage is the image itself, so the fill value has
9672    /// to be tiled into it at create — `H5D__compact_fill`'s job. An unwritten
9673    /// element must read back as the fill value, not as zero.
9674    #[test]
9675    fn an_unwritten_compact_dataset_reads_back_as_its_fill_value() {
9676        let path = temp_path("compact_fill");
9677        {
9678            let file = H5File::create(&path).unwrap();
9679            file.new_dataset::<i32>()
9680                .shape([4usize])
9681                .compact()
9682                .fill_value(-7i32)
9683                .create("d")
9684                .unwrap();
9685            file.close().unwrap();
9686        }
9687        let file = H5File::open(&path).unwrap();
9688        assert_eq!(
9689            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9690            vec![-7i32; 4]
9691        );
9692        std::fs::remove_file(&path).ok();
9693    }
9694
9695    /// The image is the layout message, so anything that rewrites the header
9696    /// rewrites the data with it. Reopening and attaching an attribute makes
9697    /// the header stale; the rebuilt one must still carry the image rather
9698    /// than fall back to an unallocated contiguous layout.
9699    #[test]
9700    fn a_reopened_compact_dataset_keeps_its_image() {
9701        let path = temp_path("compact_reopen");
9702        let values: Vec<i32> = (100..108).collect();
9703        {
9704            let file = H5File::create(&path).unwrap();
9705            file.new_dataset::<i32>()
9706                .shape([8usize])
9707                .compact()
9708                .create("d")
9709                .unwrap()
9710                .write_raw(&values)
9711                .unwrap();
9712            file.close().unwrap();
9713        }
9714        {
9715            let file = H5File::open_rw(&path).unwrap();
9716            file.dataset_writer("d")
9717                .unwrap()
9718                .new_attr::<i32>()
9719                .shape(())
9720                .create("note")
9721                .unwrap()
9722                .write_numeric(&1i32)
9723                .unwrap();
9724            file.close().unwrap();
9725        }
9726        assert_eq!(
9727            compact_image(&path, "d").unwrap(),
9728            values
9729                .iter()
9730                .flat_map(|v| v.to_le_bytes())
9731                .collect::<Vec<_>>()
9732        );
9733
9734        let file = H5File::open(&path).unwrap();
9735        assert_eq!(
9736            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9737            values
9738        );
9739        std::fs::remove_file(&path).ok();
9740    }
9741
9742    /// The ceiling is what a data layout message can hold, so it is checked
9743    /// in bytes and names them: the largest image that fits is accepted and
9744    /// one element more is refused.
9745    #[test]
9746    fn the_compact_ceiling_is_checked_in_bytes() {
9747        let path = temp_path("compact_ceiling");
9748        let file = H5File::create(&path).unwrap();
9749
9750        let fits = crate::MAX_COMPACT_DATA / 4;
9751        file.new_dataset::<i32>()
9752            .shape([fits])
9753            .compact()
9754            .create("fits")
9755            .unwrap();
9756
9757        let over = fits + 1;
9758        let err = match file
9759            .new_dataset::<i32>()
9760            .shape([over])
9761            .compact()
9762            .create("over")
9763        {
9764            Err(e) => e.to_string(),
9765            Ok(_) => panic!("an image {} bytes wide must be refused", over * 4),
9766        };
9767        assert!(
9768            err.contains(&(over * 4).to_string())
9769                && err.contains(&crate::MAX_COMPACT_DATA.to_string()),
9770            "the error must name both sizes: {err}"
9771        );
9772
9773        file.close().unwrap();
9774        std::fs::remove_file(&path).ok();
9775    }
9776
9777    /// Compact storage has no chunk grid to filter and no room to grow, so
9778    /// each conflicting option is refused at `create()` rather than silently
9779    /// overriding the layout the way `H5Pset_chunk` does.
9780    #[test]
9781    fn compact_rejects_chunking_filters_growth_and_a_null_dataspace() {
9782        let path = temp_path("compact_rejects");
9783        let file = H5File::create(&path).unwrap();
9784        assert!(file
9785            .new_dataset::<i32>()
9786            .shape([4usize])
9787            .compact()
9788            .chunk(&[4])
9789            .create("a")
9790            .is_err());
9791        assert!(file
9792            .new_dataset::<i32>()
9793            .shape([4usize])
9794            .compact()
9795            .deflate(4)
9796            .create("b")
9797            .is_err());
9798        assert!(file
9799            .new_dataset::<i32>()
9800            .shape([4usize])
9801            .compact()
9802            .max_shape(&[None])
9803            .create("c")
9804            .is_err());
9805        assert!(file
9806            .new_dataset::<i32>()
9807            .shape([4usize])
9808            .compact()
9809            .max_shape(&[Some(8)])
9810            .create("d")
9811            .is_err());
9812        assert!(file
9813            .new_dataset::<i32>()
9814            .null()
9815            .compact()
9816            .create("e")
9817            .is_err());
9818        file.close().unwrap();
9819        std::fs::remove_file(&path).ok();
9820    }
9821
9822    /// The filter pipeline the reader decodes from `name`'s header.
9823    fn stored_pipeline(
9824        path: &std::path::Path,
9825        name: &str,
9826    ) -> crate::format::messages::filter::FilterPipeline {
9827        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9828        reader
9829            .dataset_info(name)
9830            .unwrap()
9831            .filter_pipeline
9832            .clone()
9833            .unwrap_or_else(|| panic!("{name}: no filter pipeline"))
9834    }
9835
9836    /// Shuffle is a permutation, not a compressor, so it is a pipeline on its
9837    /// own — `H5Pset_shuffle` with nothing behind it. Its stage must reach the
9838    /// header, and the reader must unpermute what it wrote.
9839    #[test]
9840    fn shuffle_alone_is_a_filter_pipeline() {
9841        use crate::format::messages::filter::{FilterPipeline, FILTER_SHUFFLE};
9842        let path = temp_path("shuffle_alone");
9843        let values: Vec<i32> = (0..64).collect();
9844        {
9845            let file = H5File::create(&path).unwrap();
9846            file.new_dataset::<i32>()
9847                .shape([64usize])
9848                .chunk(&[16])
9849                .shuffle()
9850                .create("d")
9851                .unwrap()
9852                .write_raw(&values)
9853                .unwrap();
9854            file.close().unwrap();
9855        }
9856        let pipeline = stored_pipeline(&path, "d");
9857        assert_eq!(pipeline, FilterPipeline::shuffle(4));
9858        assert_eq!(pipeline.filters[0].id, FILTER_SHUFFLE);
9859
9860        let file = H5File::open(&path).unwrap();
9861        assert_eq!(
9862            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9863            values
9864        );
9865        std::fs::remove_file(&path).ok();
9866    }
9867
9868    /// `.shuffle()` and `.deflate()` are separate stages that compose, and
9869    /// `.shuffle_deflate()` is the shorthand for both — one pipeline, built
9870    /// once, whichever way it was asked for.
9871    #[cfg(feature = "deflate")]
9872    #[test]
9873    fn shuffle_composes_with_deflate() {
9874        let path = temp_path("shuffle_then_deflate");
9875        let values: Vec<i32> = (0..64).collect();
9876        {
9877            let file = H5File::create(&path).unwrap();
9878            for (name, ds) in [
9879                ("split", file.new_dataset::<i32>().shuffle().deflate(6)),
9880                ("combined", file.new_dataset::<i32>().shuffle_deflate(6)),
9881            ] {
9882                ds.shape([64usize])
9883                    .chunk(&[16])
9884                    .create(name)
9885                    .unwrap()
9886                    .write_raw(&values)
9887                    .unwrap();
9888            }
9889            file.close().unwrap();
9890        }
9891        assert_eq!(
9892            stored_pipeline(&path, "split"),
9893            crate::format::messages::filter::FilterPipeline::shuffle_deflate(4, 6)
9894        );
9895        assert_eq!(
9896            stored_pipeline(&path, "split"),
9897            stored_pipeline(&path, "combined")
9898        );
9899
9900        let file = H5File::open(&path).unwrap();
9901        for name in ["split", "combined"] {
9902            assert_eq!(
9903                file.dataset(name).unwrap().read_raw::<i32>().unwrap(),
9904                values,
9905                "{name}"
9906            );
9907        }
9908        std::fs::remove_file(&path).ok();
9909    }
9910
9911    /// The width shuffle permutes by is the stored element's, which a
9912    /// `datatype` override moves away from the carrier type `T`: recording
9913    /// `T`'s width would permute a 4-byte element as four 1-byte ones and
9914    /// hand libhdf5 a chunk it unshuffles into different bytes.
9915    #[test]
9916    fn shuffle_records_the_stored_element_width() {
9917        use crate::format::messages::datatype::DatatypeMessage;
9918        use crate::format::messages::filter::FilterPipeline;
9919        let path = temp_path("shuffle_override_width");
9920        let bytes: Vec<u8> = (0..16u8).collect();
9921        {
9922            let file = H5File::create(&path).unwrap();
9923            file.new_dataset::<u8>()
9924                .datatype(DatatypeMessage::i32_type())
9925                .shape([4usize])
9926                .chunk(&[4])
9927                .shuffle()
9928                .create("d")
9929                .unwrap()
9930                .write_raw_bytes(&bytes)
9931                .unwrap();
9932            file.close().unwrap();
9933        }
9934        assert_eq!(stored_pipeline(&path, "d"), FilterPipeline::shuffle(4));
9935
9936        let file = H5File::open(&path).unwrap();
9937        assert_eq!(
9938            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9939            bytes
9940                .chunks(4)
9941                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
9942                .collect::<Vec<_>>()
9943        );
9944        std::fs::remove_file(&path).ok();
9945    }
9946
9947    /// A write into a virtual dataset is refused by name rather than landing
9948    /// somewhere no reader would look: libhdf5 pushes such a write through
9949    /// the mapping into the source dataset (`H5D__virtual_write`), which this
9950    /// writer does not do.
9951    #[test]
9952    fn a_virtual_dataset_refuses_every_write() {
9953        use crate::Selection;
9954        let path = temp_path("vds_write_refused");
9955        let file = H5File::create(&path).unwrap();
9956        let ds = file
9957            .new_dataset::<i32>()
9958            .shape([16usize])
9959            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
9960            .create("vds")
9961            .unwrap();
9962        for err in [
9963            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap_err(),
9964            ds.write_slice(&[0], &[2], &[1i32, 2]).unwrap_err(),
9965        ] {
9966            let msg = err.to_string();
9967            assert!(msg.contains("virtual dataset"), "{msg}");
9968        }
9969        file.close().unwrap();
9970        std::fs::remove_file(&path).ok();
9971    }
9972
9973    /// A `%b` substitution only means something when the virtual selection is
9974    /// unlimited and the source selection is not: that is the shape where
9975    /// each block draws from a different source dataset. On any other mapping
9976    /// there is only one block, so `H5D_virtual_check_mapping_post` refuses
9977    /// the specifier — and an illegal conversion is refused wherever it
9978    /// appears.
9979    #[test]
9980    fn a_printf_source_name_needs_the_mapping_shape_that_uses_it() {
9981        use crate::Selection;
9982        let path = temp_path("vds_printf");
9983        let file = H5File::create(&path).unwrap();
9984        for (f, d) in [("src_%b.h5", "src"), ("src.h5", "block_%b")] {
9985            let err = match file
9986                .new_dataset::<i32>()
9987                .shape([16usize])
9988                .virtual_mapping(Selection::All, f, d, Selection::All)
9989                .create("vds")
9990            {
9991                Ok(_) => panic!("a bounded mapping has one block, so %b names nothing"),
9992                Err(e) => e.to_string(),
9993            };
9994            assert!(err.contains("printf specifier"), "{err}");
9995        }
9996        // `%z` is not a conversion libhdf5 has, in any mapping shape.
9997        let err = match file
9998            .new_dataset::<i32>()
9999            .shape([1usize, 2])
10000            .max_shape(&[None, Some(2)])
10001            .virtual_mapping(unlimited_rows(), "src_%z.h5", "src", Selection::All)
10002            .create("vds_bad")
10003        {
10004            Ok(_) => panic!("%z is not a legal conversion"),
10005            Err(e) => e.to_string(),
10006        };
10007        assert!(err.contains("invalid format specifier"), "{err}");
10008        file.close().unwrap();
10009        std::fs::remove_file(&path).ok();
10010    }
10011
10012    /// A printf mapping stitches one source dataset per block of its
10013    /// unlimited virtual selection, and the extent stops at the first block
10014    /// with no source (`H5D__virtual_set_extent_unlim`'s printf arm, at the
10015    /// default `H5D_VDS_LAST_AVAILABLE` view and `printf_gap` 0).
10016    #[test]
10017    fn a_printf_mapping_stitches_one_source_per_block() {
10018        use crate::Selection;
10019        let path = temp_path("vds_printf_blocks");
10020        {
10021            let file = H5File::create(&path).unwrap();
10022            for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10023                file.new_dataset::<i32>()
10024                    .shape([2usize])
10025                    .create(&format!("block{b}"))
10026                    .unwrap()
10027                    .write_raw(&[base, base + 1])
10028                    .unwrap();
10029            }
10030            file.new_dataset::<i32>()
10031                .shape([1usize, 2])
10032                .max_shape(&[None, Some(2)])
10033                .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10034                .create("vds")
10035                .unwrap();
10036            file.close().unwrap();
10037        }
10038        let file = H5File::open(&path).unwrap();
10039        let ds = file.dataset("vds").unwrap();
10040        // `block2` is missing, so `block3` is never reached: two rows.
10041        assert_eq!(ds.shape(), vec![2, 2]);
10042        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10043        drop(file);
10044        std::fs::remove_file(&path).ok();
10045    }
10046
10047    /// A mapping whose source cannot be opened reads back as the fill value,
10048    /// not as an error: `H5D__virtual_open_source_dset` accepts a null source
10049    /// file and clears the error stack for a missing source dataset
10050    /// (H5Dvirtual.c:877-909), so `H5D__virtual_read_one` finds no projected
10051    /// memory space and reads nothing for it (H5Dvirtual.c:2661-2665).
10052    #[test]
10053    fn a_source_that_cannot_be_opened_reads_as_the_fill_value() {
10054        use crate::{Hyperslab, HyperslabBlock, Selection};
10055        let block = |start: u64, end: u64| Selection::Hyperslab {
10056            rank: 1,
10057            form: Hyperslab::Blocks(vec![HyperslabBlock {
10058                start: vec![start],
10059                end: vec![end],
10060            }]),
10061        };
10062        let path = temp_path("vds_absent_source");
10063        {
10064            let file = H5File::create(&path).unwrap();
10065            file.new_dataset::<i32>()
10066                .shape([4usize])
10067                .create("here")
10068                .unwrap()
10069                .write_raw(&[1i32, 2, 3, 4])
10070                .unwrap();
10071            file.new_dataset::<i32>()
10072                .shape([12usize])
10073                .fill_value(-3i32)
10074                .virtual_mapping(block(0, 3), ".", "here", block(0, 3))
10075                // A dataset that is not in this file.
10076                .virtual_mapping(block(4, 7), ".", "absent", block(0, 3))
10077                // A file that does not exist beside this one.
10078                .virtual_mapping(block(8, 11), "no_such_vds_source.h5", "src", block(0, 3))
10079                .create("vds")
10080                .unwrap();
10081            file.close().unwrap();
10082        }
10083        let file = H5File::open(&path).unwrap();
10084        let ds = file.dataset("vds").unwrap();
10085        assert_eq!(
10086            ds.read_raw::<i32>().unwrap(),
10087            vec![1, 2, 3, 4, -3, -3, -3, -3, -3, -3, -3, -3]
10088        );
10089        // The same rule on the slice path, which stitches the whole image
10090        // before extracting the region.
10091        assert_eq!(
10092            ds.read_slice::<i32>(&[2], &[6]).unwrap(),
10093            vec![3, 4, -3, -3, -3, -3]
10094        );
10095        drop(file);
10096        std::fs::remove_file(&path).ok();
10097    }
10098
10099    /// `%%` is an escaped literal `%`, not a substitution: the mapping is an
10100    /// ordinary bounded one, and the source it resolves against is the name
10101    /// with a single `%` in it.
10102    #[test]
10103    fn an_escaped_percent_is_a_literal_in_a_source_name() {
10104        use crate::Selection;
10105        let path = temp_path("vds_escaped_pct");
10106        {
10107            let file = H5File::create(&path).unwrap();
10108            file.new_dataset::<i32>()
10109                .shape([4usize])
10110                .create("od%d")
10111                .unwrap()
10112                .write_raw(&[5i32, 6, 7, 8])
10113                .unwrap();
10114            file.new_dataset::<i32>()
10115                .shape([4usize])
10116                .virtual_mapping(Selection::All, ".", "od%%d", Selection::All)
10117                .create("vds")
10118                .unwrap();
10119            file.close().unwrap();
10120        }
10121        let file = H5File::open(&path).unwrap();
10122        let ds = file.dataset("vds").unwrap();
10123        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![5, 6, 7, 8]);
10124        // The stored name keeps its escape; only the resolution unescapes.
10125        assert_eq!(ds.virtual_mappings().unwrap()[0].source_dset_name, "od%%d");
10126        drop(file);
10127        std::fs::remove_file(&path).ok();
10128    }
10129
10130    /// An unlimited mapping takes its extent from the source it names, so a
10131    /// virtual dataset written with one reports the source's rows, not the
10132    /// seed extent its dataspace message stores
10133    /// (`H5D__virtual_set_extent_unlim`). Same file, so the resolution runs
10134    /// without opening another one.
10135    #[test]
10136    fn an_unlimited_mapping_takes_its_extent_from_its_source() {
10137        let path = temp_path("vds_unlimited");
10138        {
10139            let file = H5File::create(&path).unwrap();
10140            file.new_dataset::<i32>()
10141                .shape([5usize, 2])
10142                .chunk(&[5, 2])
10143                .max_shape(&[None, Some(2)])
10144                .create("src")
10145                .unwrap()
10146                .write_raw(&(0..10i32).collect::<Vec<_>>())
10147                .unwrap();
10148            file.new_dataset::<i32>()
10149                .shape([1usize, 2])
10150                .max_shape(&[None, Some(2)])
10151                .virtual_mapping(unlimited_rows(), ".", "src", unlimited_rows())
10152                .create("vds")
10153                .unwrap();
10154            file.close().unwrap();
10155        }
10156        let file = H5File::open(&path).unwrap();
10157        let ds = file.dataset("vds").unwrap();
10158        assert_eq!(ds.shape(), vec![5, 2]);
10159        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..10).collect::<Vec<i32>>());
10160        drop(file);
10161        std::fs::remove_file(&path).ok();
10162    }
10163
10164    /// The blocks-0/1/3 printf file every dataset-access test below reads,
10165    /// laid out exactly like `a_printf_mapping_stitches_one_source_per_block`
10166    /// so the gap is the only thing that changes.
10167    fn printf_gap_file(tag: &str) -> std::path::PathBuf {
10168        use crate::Selection;
10169        let path = temp_path(tag);
10170        let file = H5File::create(&path).unwrap();
10171        for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10172            file.new_dataset::<i32>()
10173                .shape([2usize])
10174                .create(&format!("block{b}"))
10175                .unwrap()
10176                .write_raw(&[base, base + 1])
10177                .unwrap();
10178        }
10179        file.new_dataset::<i32>()
10180            .shape([1usize, 2])
10181            .max_shape(&[None, Some(2)])
10182            .fill_value(-7i32)
10183            .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10184            .create("vds")
10185            .unwrap();
10186        file.close().unwrap();
10187        path
10188    }
10189
10190    /// `H5Pset_virtual_printf_gap` lets the block scan look past a missing
10191    /// source, and the blocks it looked past stay inside the extent reading
10192    /// as the fill value (H5Dvirtual.c:1519-1614, :2661-2665). Measured
10193    /// against libhdf5 1.14.6 through h5py's `h5p.PropDAID`: gap 0 gives
10194    /// two rows, gap 1 and gap 2 both give four with row 2 filled.
10195    #[test]
10196    fn a_printf_gap_looks_past_the_missing_block() {
10197        use crate::DatasetAccess;
10198        let path = printf_gap_file("vds_printf_gap");
10199        let file = H5File::open(&path).unwrap();
10200        for (gap, shape, data) in [
10201            (0u64, vec![2usize, 2], vec![0i32, 1, 100, 101]),
10202            (1, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10203            (2, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10204        ] {
10205            let ds = file
10206                .dataset_with("vds", DatasetAccess::new().virtual_printf_gap(gap))
10207                .unwrap();
10208            assert_eq!(ds.shape(), shape, "gap {gap}");
10209            assert_eq!(ds.read_raw::<i32>().unwrap(), data, "gap {gap}");
10210        }
10211        // Back to the default: the extent follows the properties the open
10212        // names, in both directions.
10213        let ds = file.dataset("vds").unwrap();
10214        assert_eq!(ds.shape(), vec![2, 2]);
10215        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10216        drop(file);
10217        std::fs::remove_file(&path).ok();
10218    }
10219
10220    /// A relatively-named source is found next to the virtual dataset even
10221    /// when the process is somewhere else entirely: `H5F_prefix_open_file`
10222    /// tries the primary file's `H5F_EXTPATH` — the directory it was opened
10223    /// from — before the bare relative name against the working directory
10224    /// (H5Fint.c:952-977). Measured against libhdf5 1.14.6 through h5py: a
10225    /// `VirtualSource("src.h5", ...)` beside its VDS reads its data with
10226    /// `HDF5_VDS_PREFIX` unset and the working directory elsewhere; before
10227    /// the reader took that step it read back all fill value.
10228    #[test]
10229    fn a_relative_source_resolves_next_to_the_virtual_dataset() {
10230        use crate::Selection;
10231        let dir = std::env::temp_dir().join(format!(
10232            "rust_hdf5_vds_beside_{}_{:?}",
10233            std::process::id(),
10234            std::thread::current().id()
10235        ));
10236        std::fs::create_dir_all(&dir).unwrap();
10237        {
10238            let file = H5File::create(dir.join("src.h5")).unwrap();
10239            file.new_dataset::<i32>()
10240                .shape([2usize, 4])
10241                .create("data")
10242                .unwrap()
10243                .write_raw(&(0..8i32).collect::<Vec<_>>())
10244                .unwrap();
10245            file.close().unwrap();
10246        }
10247        {
10248            let file = H5File::create(dir.join("v.h5")).unwrap();
10249            file.new_dataset::<i32>()
10250                .shape([2usize, 4])
10251                .fill_value(-9i32)
10252                // Named relatively, as h5py's `VirtualSource("src.h5", ...)`
10253                // stores it — nothing in the file says where it lives.
10254                .virtual_mapping(Selection::All, "src.h5", "data", Selection::All)
10255                .create("v")
10256                .unwrap();
10257            file.close().unwrap();
10258        }
10259        // The working directory is the crate root under `cargo test`, not
10260        // `dir`, so only the extpath step can find `src.h5`.
10261        assert_ne!(std::env::current_dir().unwrap(), dir);
10262        let file = H5File::open(dir.join("v.h5")).unwrap();
10263        let ds = file.dataset("v").unwrap();
10264        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8i32).collect::<Vec<_>>());
10265        drop(file);
10266        std::fs::remove_dir_all(&dir).ok();
10267    }
10268
10269    /// The first open of a virtual dataset fixes its access properties for
10270    /// every open that overlaps it: only the open that finds no shared info
10271    /// in `H5FO_opened` runs `H5D__open_oid(dataset, dapl_id)`, and a later
10272    /// one just points at that shared info without ever reading its own dapl
10273    /// (H5Dint.c:1496-1500, :1523-1528) — the view and the gap live in the
10274    /// shared layout storage `H5D__virtual_init` filled from that first dapl
10275    /// (H5Dvirtual.c:2178-2188). Measured against libhdf5 1.14.6 and 2.0.0
10276    /// through `h5d.open(..., dapl=...)` on the printf-gap VDS below: opening
10277    /// gap 0 then gap 1 gives both handles two rows; with every handle closed,
10278    /// opening gap 1 then gap 0 gives both four rows and the gap row reads as
10279    /// fill; with every handle closed again, gap 0 alone is back to two rows.
10280    #[test]
10281    fn the_first_open_of_a_virtual_dataset_fixes_the_properties_for_later_opens() {
10282        use crate::DatasetAccess;
10283        let path = printf_gap_file("vds_printf_first_open");
10284        let file = H5File::open(&path).unwrap();
10285        let gap = |g: u64| DatasetAccess::new().virtual_printf_gap(g);
10286        {
10287            let a = file.dataset_with("vds", gap(0)).unwrap();
10288            let b = file.dataset_with("vds", gap(1)).unwrap();
10289            assert_eq!(a.shape(), vec![2, 2]);
10290            assert_eq!(b.shape(), vec![2, 2]);
10291            assert_eq!(b.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10292        }
10293        {
10294            let c = file.dataset_with("vds", gap(1)).unwrap();
10295            let d = file.dataset_with("vds", gap(0)).unwrap();
10296            assert_eq!(c.shape(), vec![4, 2]);
10297            assert_eq!(d.shape(), vec![4, 2]);
10298            assert_eq!(
10299                d.read_raw::<i32>().unwrap(),
10300                vec![0, 1, 100, 101, -7, -7, 300, 301]
10301            );
10302        }
10303        let e = file.dataset_with("vds", gap(0)).unwrap();
10304        assert_eq!(e.shape(), vec![2, 2]);
10305        assert_eq!(e.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10306        drop(e);
10307        drop(file);
10308        std::fs::remove_file(&path).ok();
10309    }
10310
10311    /// `H5D_VDS_FIRST_MISSING` ignores the printf gap: `H5D__virtual_init`
10312    /// reads the gap property only under `H5D_VDS_LAST_AVAILABLE` and forces
10313    /// it to 0 otherwise (H5Dvirtual.c:2182-2188). Measured against libhdf5
10314    /// 1.14.6: gap 2 under this view still gives two rows.
10315    #[test]
10316    fn the_first_missing_view_ignores_the_printf_gap() {
10317        use crate::{DatasetAccess, VirtualView};
10318        let path = printf_gap_file("vds_printf_first_missing");
10319        let file = H5File::open(&path).unwrap();
10320        for gap in [0u64, 2] {
10321            let ds = file
10322                .dataset_with(
10323                    "vds",
10324                    DatasetAccess::new()
10325                        .virtual_view(VirtualView::FirstMissing)
10326                        .virtual_printf_gap(gap),
10327                )
10328                .unwrap();
10329            assert_eq!(ds.shape(), vec![2, 2], "gap {gap}");
10330            assert_eq!(
10331                ds.read_raw::<i32>().unwrap(),
10332                vec![0, 1, 100, 101],
10333                "gap {gap}"
10334            );
10335        }
10336        drop(file);
10337        std::fs::remove_file(&path).ok();
10338    }
10339
10340    /// On a mapping unlimited on both sides the view is
10341    /// `H5S_hyper_get_clip_extent_match`'s `incl_trail`
10342    /// (H5Dvirtual.c:1447-1451): with a stride wider than its block, the
10343    /// extent under `H5D_VDS_LAST_AVAILABLE` ends at the last mapped row,
10344    /// and under `H5D_VDS_FIRST_MISSING` it runs on to where the next block
10345    /// would start. Measured against libhdf5 1.14.6 over a three-row source
10346    /// with stride 3 and block 2: two rows and three rows.
10347    #[test]
10348    fn the_view_decides_whether_a_trailing_gap_is_inside_the_extent() {
10349        use crate::format::selection::UNLIMITED;
10350        use crate::{DatasetAccess, Hyperslab, RegularHyperslab, Selection, VirtualView};
10351        let strided = || Selection::Hyperslab {
10352            rank: 2,
10353            form: Hyperslab::Regular(RegularHyperslab {
10354                start: vec![0, 0],
10355                stride: vec![3, 1],
10356                count: vec![UNLIMITED, 1],
10357                block: vec![2, 2],
10358            }),
10359        };
10360        let path = temp_path("vds_view_trail");
10361        {
10362            let file = H5File::create(&path).unwrap();
10363            file.new_dataset::<i32>()
10364                .shape([3usize, 2])
10365                .max_shape(&[None, Some(2)])
10366                .chunk(&[1, 2])
10367                .create("src")
10368                .unwrap()
10369                .write_raw(&(0..6i32).collect::<Vec<_>>())
10370                .unwrap();
10371            file.new_dataset::<i32>()
10372                .shape([1usize, 2])
10373                .max_shape(&[None, Some(2)])
10374                .fill_value(-9i32)
10375                .virtual_mapping(strided(), ".", "src", strided())
10376                .create("vds")
10377                .unwrap();
10378            file.close().unwrap();
10379        }
10380        let file = H5File::open(&path).unwrap();
10381        {
10382            let last = file.dataset("vds").unwrap();
10383            assert_eq!(last.shape(), vec![2, 2]);
10384            assert_eq!(last.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3]);
10385        }
10386        // The handle above is gone, so this open is the one that resolves.
10387        let first = file
10388            .dataset_with(
10389                "vds",
10390                DatasetAccess::new().virtual_view(VirtualView::FirstMissing),
10391            )
10392            .unwrap();
10393        assert_eq!(first.shape(), vec![3, 2]);
10394        assert_eq!(first.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3, -9, -9]);
10395        drop(file);
10396        std::fs::remove_file(&path).ok();
10397    }
10398
10399    /// The property list reads back what was set (`H5Pget_virtual_view`,
10400    /// `H5Pget_virtual_printf_gap`), and the gap `HSIZE_UNDEF` that
10401    /// `H5Pset_virtual_printf_gap` refuses is refused by the open that would
10402    /// have used it.
10403    #[test]
10404    fn the_access_property_list_reads_back_and_refuses_hsize_undef() {
10405        use crate::{DatasetAccess, VirtualView};
10406        let plist = DatasetAccess::new();
10407        assert_eq!(plist.view(), VirtualView::LastAvailable);
10408        assert_eq!(plist.printf_gap(), 0);
10409        let set = plist
10410            .virtual_view(VirtualView::FirstMissing)
10411            .virtual_printf_gap(4);
10412        assert_eq!(set.view(), VirtualView::FirstMissing);
10413        // The *property* keeps what was set even though the resolution under
10414        // this view scans with 0.
10415        assert_eq!(set.printf_gap(), 4);
10416
10417        let path = printf_gap_file("vds_printf_gap_undef");
10418        let file = H5File::open(&path).unwrap();
10419        let err = match file.dataset_with("vds", DatasetAccess::new().virtual_printf_gap(u64::MAX))
10420        {
10421            Ok(_) => panic!("HSIZE_UNDEF is not a valid printf gap size"),
10422            Err(e) => e.to_string(),
10423        };
10424        assert!(err.contains("HSIZE_UNDEF"), "{err}");
10425        drop(file);
10426        std::fs::remove_file(&path).ok();
10427    }
10428
10429    /// The rank-2 `count = (H5S_UNLIMITED, 1)`, `block = (1, 2)` selection
10430    /// both sides of an unlimited row-wise mapping use.
10431    fn unlimited_rows() -> crate::Selection {
10432        use crate::format::selection::UNLIMITED;
10433        use crate::{Hyperslab, RegularHyperslab, Selection};
10434        Selection::Hyperslab {
10435            rank: 2,
10436            form: Hyperslab::Regular(RegularHyperslab {
10437                start: vec![0, 0],
10438                stride: vec![1, 1],
10439                count: vec![UNLIMITED, 1],
10440                block: vec![1, 2],
10441            }),
10442        }
10443    }
10444
10445    /// An unlimited virtual selection over a *limited* source selection is
10446    /// the printf shape, and without a `%b` in a source name there is no
10447    /// second dataset to fill the second block —
10448    /// `H5D_virtual_check_mapping_post` refuses it, and so does this.
10449    #[test]
10450    fn an_unlimited_virtual_selection_over_a_limited_source_is_refused() {
10451        use crate::Selection;
10452        let path = temp_path("vds_unlim_limited_src");
10453        let file = H5File::create(&path).unwrap();
10454        let err = match file
10455            .new_dataset::<i32>()
10456            .shape([1usize, 2])
10457            .max_shape(&[None, Some(2)])
10458            .virtual_mapping(unlimited_rows(), "src.h5", "src", Selection::All)
10459            .create("vds")
10460        {
10461            Ok(_) => panic!("no printf substitution names the mapping's later blocks"),
10462            Err(e) => e.to_string(),
10463        };
10464        assert!(err.contains("printf"), "{err}");
10465        file.close().unwrap();
10466        std::fs::remove_file(&path).ok();
10467    }
10468
10469    /// A virtual dataset stores nothing of its own, so it cannot also be one
10470    /// of the storage classes that do.
10471    #[test]
10472    fn a_virtual_dataset_cannot_also_be_chunked_or_external() {
10473        use crate::Selection;
10474        let path = temp_path("vds_exclusive");
10475        let file = H5File::create(&path).unwrap();
10476        let builder = || {
10477            file.new_dataset::<i32>().shape([16usize]).virtual_mapping(
10478                Selection::All,
10479                "src.h5",
10480                "src",
10481                Selection::All,
10482            )
10483        };
10484        for (which, res) in [
10485            ("chunked", builder().chunk(&[4]).create("a")),
10486            ("compact", builder().compact().create("b")),
10487            (
10488                "external",
10489                builder().external(&[("x.raw", 0, 64)]).create("c"),
10490            ),
10491            ("references", builder().object_references().create("d")),
10492        ] {
10493            match res {
10494                Ok(_) => panic!("a virtual dataset cannot also be {which}"),
10495                Err(e) => assert!(e.to_string().contains("virtual dataset"), "{which}: {e}"),
10496            }
10497        }
10498        file.close().unwrap();
10499        std::fs::remove_file(&path).ok();
10500    }
10501
10502    /// Deleting a virtual dataset frees the global heap object its mapping
10503    /// list lived in — `H5D__virtual_delete` reaches `H5HG_remove` — so the
10504    /// next dataset's own heap object reuses that space instead of the file
10505    /// growing by a whole collection per deleted virtual dataset.
10506    #[test]
10507    fn deleting_a_virtual_dataset_frees_its_mapping_list() {
10508        use crate::Selection;
10509        let path = temp_path("vds_delete");
10510        let baseline = {
10511            let file = H5File::create(&path).unwrap();
10512            file.new_dataset::<i32>()
10513                .shape([16usize])
10514                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10515                .create("vds")
10516                .unwrap();
10517            file.delete_dataset("vds").unwrap();
10518            file.close().unwrap();
10519            std::fs::metadata(&path).unwrap().len()
10520        };
10521        std::fs::remove_file(&path).ok();
10522
10523        // Ten more create-then-delete rounds must land on the same file size:
10524        // each round's heap object is removed, its collection becomes empty
10525        // and returns to the allocator, and the next round takes it back.
10526        let file = H5File::create(&path).unwrap();
10527        for i in 0..10 {
10528            let name = format!("vds{i}");
10529            file.new_dataset::<i32>()
10530                .shape([16usize])
10531                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10532                .create(&name)
10533                .unwrap();
10534            file.delete_dataset(&name).unwrap();
10535        }
10536        file.close().unwrap();
10537        assert_eq!(std::fs::metadata(&path).unwrap().len(), baseline);
10538        std::fs::remove_file(&path).ok();
10539    }
10540
10541    /// [`H5Dataset::storage_layout`] tells the four classes apart — the
10542    /// negative case for any one class is simply that it is not another.
10543    #[test]
10544    fn storage_layout_reports_each_class() {
10545        use crate::StorageLayout;
10546        let path = temp_path("storage_layout");
10547        {
10548            let file = H5File::create(&path).unwrap();
10549            file.new_dataset::<i32>()
10550                .shape([4usize])
10551                .create("contig")
10552                .unwrap();
10553            file.new_dataset::<i32>()
10554                .shape([4usize])
10555                .compact()
10556                .create("compact")
10557                .unwrap();
10558            file.new_dataset::<i32>()
10559                .shape([8usize])
10560                .chunk(&[4])
10561                .create("chunked")
10562                .unwrap();
10563            file.close().unwrap();
10564        }
10565        let file = H5File::open(&path).unwrap();
10566        assert_eq!(
10567            file.dataset("contig").unwrap().storage_layout().unwrap(),
10568            StorageLayout::Contiguous
10569        );
10570        assert_eq!(
10571            file.dataset("compact").unwrap().storage_layout().unwrap(),
10572            StorageLayout::Compact
10573        );
10574        assert_eq!(
10575            file.dataset("chunked").unwrap().storage_layout().unwrap(),
10576            StorageLayout::Chunked
10577        );
10578    }
10579
10580    /// Read-mode-only accessor, matching `datatype()`'s own contract.
10581    #[test]
10582    fn storage_layout_errors_in_write_mode() {
10583        let path = temp_path("storage_layout_write_mode");
10584        let file = H5File::create(&path).unwrap();
10585        let ds = file
10586            .new_dataset::<i32>()
10587            .shape([4usize])
10588            .create("data")
10589            .unwrap();
10590        assert!(ds.storage_layout().is_err());
10591        file.close().unwrap();
10592    }
10593
10594    /// [`H5Dataset::chunk_index`] reports the real on-disk index kind
10595    /// (extensible array for one unlimited dimension, version-1 B-tree
10596    /// under a legacy libver bound) and `None` for an unchunked dataset —
10597    /// the negative case.
10598    #[test]
10599    fn chunk_index_reports_the_stored_kind() {
10600        use crate::ChunkIndex;
10601        let path = temp_path("chunk_index");
10602        {
10603            let file = H5File::create(&path).unwrap();
10604            file.new_dataset::<i32>()
10605                .shape([4usize])
10606                .create("contig")
10607                .unwrap();
10608            file.new_dataset::<i32>()
10609                .shape([16usize])
10610                .chunk(&[4])
10611                .max_shape(&[None])
10612                .create("earray")
10613                .unwrap();
10614            file.set_libver_latest(false).unwrap();
10615            file.new_dataset::<i32>()
10616                .shape([8usize])
10617                .chunk(&[4])
10618                .max_shape(&[None])
10619                .create("btree1")
10620                .unwrap();
10621            file.close().unwrap();
10622        }
10623        let file = H5File::open(&path).unwrap();
10624        assert_eq!(file.dataset("contig").unwrap().chunk_index().unwrap(), None);
10625        assert_eq!(
10626            file.dataset("earray").unwrap().chunk_index().unwrap(),
10627            Some(ChunkIndex::ExtensibleArray)
10628        );
10629        assert_eq!(
10630            file.dataset("btree1").unwrap().chunk_index().unwrap(),
10631            Some(ChunkIndex::BtreeV1)
10632        );
10633    }
10634
10635    /// [`H5Dataset::filters`] reports the stored pipeline in order — and
10636    /// the negative case: an unfiltered dataset reports an empty pipeline,
10637    /// not an error.
10638    #[test]
10639    fn filters_reports_the_stored_pipeline() {
10640        use crate::format::messages::filter::{FILTER_DEFLATE, FILTER_SHUFFLE, FLAG_OPTIONAL};
10641        let path = temp_path("filters");
10642        {
10643            let file = H5File::create(&path).unwrap();
10644            file.new_dataset::<i32>()
10645                .shape([16usize])
10646                .create("unfiltered")
10647                .unwrap();
10648            file.new_dataset::<i32>()
10649                .shape([16usize])
10650                .chunk(&[4])
10651                .shuffle()
10652                .deflate(6)
10653                .create("filtered")
10654                .unwrap();
10655            file.close().unwrap();
10656        }
10657        let file = H5File::open(&path).unwrap();
10658        assert_eq!(
10659            file.dataset("unfiltered").unwrap().filters().unwrap(),
10660            Vec::new()
10661        );
10662        let filters = file.dataset("filtered").unwrap().filters().unwrap();
10663        assert_eq!(filters.len(), 2);
10664        assert_eq!(filters[0].id, FILTER_SHUFFLE);
10665        assert_eq!(filters[0].flags, FLAG_OPTIONAL);
10666        assert_eq!(filters[1].id, FILTER_DEFLATE);
10667        assert_eq!(filters[1].cd_values, vec![6]);
10668    }
10669
10670    /// [`H5Dataset::fill_value`] reports the explicit bytes for a dataset
10671    /// created with `.fill_value(...)`, and the negative case: a dataset
10672    /// with no fill value set reports [`FillValue::Default`], not an error.
10673    /// `FillValue::Undefined` has no constructor on either this crate's
10674    /// writer or h5py's public API, so it is not exercised here.
10675    #[test]
10676    fn fill_value_reports_the_stored_value() {
10677        use crate::FillValue;
10678        let path = temp_path("fill_value");
10679        {
10680            let file = H5File::create(&path).unwrap();
10681            file.new_dataset::<i32>()
10682                .shape([4usize])
10683                .create("unset")
10684                .unwrap();
10685            file.new_dataset::<i32>()
10686                .shape([4usize])
10687                .fill_value(-7i32)
10688                .create("set")
10689                .unwrap();
10690            file.close().unwrap();
10691        }
10692        let file = H5File::open(&path).unwrap();
10693        assert_eq!(
10694            file.dataset("unset").unwrap().fill_value().unwrap(),
10695            FillValue::Default
10696        );
10697        assert_eq!(
10698            file.dataset("set").unwrap().fill_value().unwrap(),
10699            FillValue::UserDefined((-7i32).to_le_bytes().to_vec())
10700        );
10701    }
10702
10703    /// The three `H5D__efl_construct` / `H5Pset_external` rules an
10704    /// `H5O_EFL_UNLIMITED` slot lives inside: it may only be the last slot,
10705    /// an unlimited dataspace must have one, and only the first dimension
10706    /// may be extendible.
10707    #[test]
10708    fn the_unlimited_external_slot_keeps_its_three_rules() {
10709        use crate::format::messages::external_file_list::UNLIMITED;
10710        let path = temp_path("efl_unlim_rules");
10711        let file = H5File::create(&path).unwrap();
10712
10713        // "previous file size is unlimited": nothing behind an unlimited slot
10714        // could ever be reached.
10715        let err = match file
10716            .new_dataset::<i32>()
10717            .shape([8usize])
10718            .external(&[("a.raw", 0, UNLIMITED), ("b.raw", 0, 32)])
10719            .create("mid")
10720        {
10721            Ok(_) => panic!("an unlimited slot absorbs everything behind it"),
10722            Err(e) => e.to_string(),
10723        };
10724        assert!(err.contains("only be the last"), "{err}");
10725
10726        // "unlimited dataspace but finite storage".
10727        let err = match file
10728            .new_dataset::<i32>()
10729            .shape([8usize])
10730            .max_shape(&[None])
10731            .external(&[("a.raw", 0, 32)])
10732            .create("finite")
10733        {
10734            Ok(_) => panic!("no finite reservation covers an unlimited extent"),
10735            Err(e) => e.to_string(),
10736        };
10737        assert!(err.contains("unlimited dataspace"), "{err}");
10738
10739        // "only the first dimension can be extendible".
10740        let err = match file
10741            .new_dataset::<i32>()
10742            .shape([2usize, 4])
10743            .max_shape(&[Some(2), None])
10744            .external(&[("a.raw", 0, UNLIMITED)])
10745            .create("dim1")
10746        {
10747            Ok(_) => panic!("only the slowest-varying dimension may be extendible"),
10748            Err(e) => e.to_string(),
10749        };
10750        assert!(err.contains("only the first dimension"), "{err}");
10751
10752        // And the legal shape: an unlimited last slot under an unlimited
10753        // first dimension.
10754        file.new_dataset::<i32>()
10755            .shape([8usize])
10756            .max_shape(&[None])
10757            .external(&[("a.raw", 0, 16), ("b.raw", 0, UNLIMITED)])
10758            .create("ok")
10759            .unwrap()
10760            .write_raw(&(0..8i32).collect::<Vec<_>>())
10761            .unwrap();
10762        file.close().unwrap();
10763        std::fs::remove_file(&path).ok();
10764        for raw in ["a.raw", "b.raw"] {
10765            std::fs::remove_file(raw).ok();
10766        }
10767    }
10768
10769    /// An unlimited slot reserves nothing, so a read of it is bounded by the
10770    /// dataset's extent and by what the file physically holds: the tail past
10771    /// the end of a short raw file reads back as zero, exactly as
10772    /// `H5D__efl_read` fills it.
10773    #[test]
10774    fn an_unlimited_external_slot_reads_zero_past_the_end_of_its_file() {
10775        use crate::format::messages::external_file_list::UNLIMITED;
10776        let dir = std::env::temp_dir().join(format!("rh5_efl_short_{}", std::process::id()));
10777        std::fs::create_dir_all(&dir).unwrap();
10778        let path = dir.join("f.h5");
10779        let raw = dir.join("short.raw");
10780        // Four elements' worth of bytes for an eight-element dataset.
10781        std::fs::write(&raw, [0u8; 16]).unwrap();
10782        {
10783            let file = H5File::create(&path).unwrap();
10784            file.new_dataset::<i32>()
10785                .shape([8usize])
10786                .max_shape(&[None])
10787                .external(&[(raw.to_str().unwrap(), 0, UNLIMITED)])
10788                .create("data")
10789                .unwrap();
10790            file.close().unwrap();
10791        }
10792        let file = H5File::open(&path).unwrap();
10793        assert_eq!(
10794            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
10795            vec![0i32; 8]
10796        );
10797        drop(file);
10798        std::fs::remove_dir_all(&dir).ok();
10799    }
10800
10801    /// [`H5Dataset::external_files`] reports the stored segment list in
10802    /// order, and the negative case: a dataset whose data lives in this
10803    /// file reports an empty list, not an error.
10804    #[test]
10805    fn external_files_reports_the_stored_segments() {
10806        let path = temp_path("external_files");
10807        {
10808            let file = H5File::create(&path).unwrap();
10809            file.new_dataset::<i32>()
10810                .shape([4usize])
10811                .create("contig")
10812                .unwrap();
10813            file.new_dataset::<i32>()
10814                .shape([16usize])
10815                .external(&[("a.raw", 0, 32), ("b.raw", 8, 32)])
10816                .create("external")
10817                .unwrap();
10818            file.close().unwrap();
10819        }
10820        let file = H5File::open(&path).unwrap();
10821        assert_eq!(
10822            file.dataset("contig").unwrap().external_files().unwrap(),
10823            Vec::new()
10824        );
10825        let segments = file.dataset("external").unwrap().external_files().unwrap();
10826        assert_eq!(segments.len(), 2);
10827        assert_eq!(segments[0].name, "a.raw");
10828        assert_eq!(segments[0].offset, 0);
10829        assert_eq!(segments[0].size, 32);
10830        assert_eq!(segments[1].name, "b.raw");
10831        assert_eq!(segments[1].offset, 8);
10832        assert_eq!(segments[1].size, 32);
10833    }
10834
10835    /// [`H5Dataset::max_shape`] reports an unlimited axis as `None` and a
10836    /// fixed one as its current size — and the negative case: a dataset
10837    /// with no maximum-dimensions message reports max == current, not an
10838    /// error.
10839    #[test]
10840    fn max_shape_reports_unlimited_and_fixed_axes() {
10841        let path = temp_path("max_shape");
10842        {
10843            let file = H5File::create(&path).unwrap();
10844            file.new_dataset::<i32>()
10845                .shape([4usize, 8])
10846                .create("fixed")
10847                .unwrap();
10848            file.new_dataset::<i32>()
10849                .shape([4usize, 8])
10850                .chunk(&[2, 8])
10851                .max_shape(&[None, Some(8)])
10852                .create("unlimited")
10853                .unwrap();
10854            file.close().unwrap();
10855        }
10856        let file = H5File::open(&path).unwrap();
10857        assert_eq!(
10858            file.dataset("fixed").unwrap().max_shape().unwrap(),
10859            vec![Some(4), Some(8)]
10860        );
10861        assert_eq!(
10862            file.dataset("unlimited").unwrap().max_shape().unwrap(),
10863            vec![None, Some(8)]
10864        );
10865    }
10866
10867    /// [`H5Dataset::virtual_mappings`] reports the stored source/virtual
10868    /// mapping list in order, and the negative case: a dataset with no
10869    /// virtual layout reports an empty list, not an error.
10870    #[test]
10871    fn virtual_mappings_reports_the_stored_mappings() {
10872        use crate::Selection;
10873        let path = temp_path("virtual_mappings");
10874        {
10875            let file = H5File::create(&path).unwrap();
10876            file.new_dataset::<i32>()
10877                .shape([4usize])
10878                .create("plain")
10879                .unwrap();
10880            file.new_dataset::<i32>()
10881                .shape([16usize])
10882                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10883                .create("vds")
10884                .unwrap();
10885            file.close().unwrap();
10886        }
10887        let file = H5File::open(&path).unwrap();
10888        assert_eq!(
10889            file.dataset("plain").unwrap().virtual_mappings().unwrap(),
10890            Vec::new()
10891        );
10892        let mappings = file.dataset("vds").unwrap().virtual_mappings().unwrap();
10893        assert_eq!(mappings.len(), 1);
10894        assert_eq!(mappings[0].source_file_name, "src.h5");
10895        assert_eq!(mappings[0].source_dset_name, "src");
10896        assert_eq!(mappings[0].source_selection, Selection::All);
10897        assert_eq!(mappings[0].virtual_selection, Selection::All);
10898    }
10899}