Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use std::borrow::Cow;
8
9use crate::attribute::AttrBuilder;
10use crate::error::{Hdf5Error, Result};
11use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
12use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
13use crate::format::messages::filter::Filter;
14use crate::format::messages::virtual_mapping::VirtualMapping;
15use crate::format::reference::{Reference, ReferenceTarget};
16use crate::format::selection::Selection;
17use crate::format::storage_kind::AttributeStorage;
18use crate::io::reader::{read_image_into_new, ExternalFileSegment};
19use crate::io::writer::ChunkIndexKind;
20use crate::types::H5Type;
21
22// ---------------------------------------------------------------------------
23// DatasetBuilder
24// ---------------------------------------------------------------------------
25
26/// A fluent builder for creating datasets.
27///
28/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
29///
30/// ```no_run
31/// # use rust_hdf5::H5File;
32/// let file = H5File::create("builder.h5").unwrap();
33/// let ds = file.new_dataset::<f32>()
34///     .shape(&[10, 20])
35///     .create("temperatures")
36///     .unwrap();
37/// ```
38pub struct DatasetBuilder<T: H5Type> {
39    file_inner: SharedInner,
40    shape: Option<Vec<usize>>,
41    is_null: bool,
42    chunk_dims: Option<Vec<usize>>,
43    max_shape: Option<Vec<Option<usize>>>,
44    is_compact: bool,
45    early_allocation: bool,
46    deflate_level: Option<u32>,
47    shuffle: bool,
48    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
49    group_path: Option<String>,
50    fill_value: Option<Vec<u8>>,
51    fill_time: Option<FillTime>,
52    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
53    committed_type: Option<String>,
54    references: Option<ReferenceElement>,
55    external: Option<Vec<(String, u64, u64)>>,
56    efile_prefix: Option<String>,
57    virtual_mappings: Vec<VirtualMapping>,
58    _marker: std::marker::PhantomData<T>,
59}
60
61/// Which reference a `*_references()` builder call asked the elements to be.
62///
63/// One field rather than a flag per kind: an element is a whole-object
64/// reference or a region reference, never both, and the width of each is only
65/// known once the file's address size is (see
66/// [`DatatypeMessage::object_reference`] and
67/// [`DatatypeMessage::region_reference`]).
68///
69/// [`DatatypeMessage::object_reference`]: crate::format::messages::datatype::DatatypeMessage::object_reference
70/// [`DatatypeMessage::region_reference`]: crate::format::messages::datatype::DatatypeMessage::region_reference
71#[derive(Debug, Clone, Copy, PartialEq, Eq)]
72enum ReferenceElement {
73    /// `H5T_STD_REF_OBJ`.
74    Object,
75    /// `H5T_STD_REF_DSETREG`.
76    Region,
77    /// `H5T_STD_REF`, the 1.12 form. One datatype for all three 1.12 kinds:
78    /// the element leads with the kind it holds, so `H5T__ref_disk_getsize`
79    /// sizes every element for the widest of them and a dataset of this type
80    /// may hold objects, regions and attributes alike.
81    Revised,
82}
83
84impl ReferenceElement {
85    /// The stored datatype for this kind in a file with `ctx`'s address size.
86    fn datatype(
87        self,
88        ctx: &crate::format::FormatContext,
89    ) -> crate::format::messages::datatype::DatatypeMessage {
90        use crate::format::messages::datatype::DatatypeMessage;
91        match self {
92            Self::Object => DatatypeMessage::object_reference(ctx),
93            Self::Region => DatatypeMessage::region_reference(ctx),
94            Self::Revised => DatatypeMessage::std_object_reference(ctx),
95        }
96    }
97}
98
99impl<T: H5Type> DatasetBuilder<T> {
100    pub(crate) fn new(file_inner: SharedInner) -> Self {
101        Self {
102            file_inner,
103            shape: None,
104            is_null: false,
105            chunk_dims: None,
106            max_shape: None,
107            is_compact: false,
108            early_allocation: false,
109            deflate_level: None,
110            shuffle: false,
111            custom_pipeline: None,
112            group_path: None,
113            fill_value: None,
114            fill_time: None,
115            datatype_override: None,
116            committed_type: None,
117            references: None,
118            external: None,
119            efile_prefix: None,
120            virtual_mappings: Vec::new(),
121            _marker: std::marker::PhantomData,
122        }
123    }
124
125    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
126        Self {
127            file_inner,
128            shape: None,
129            is_null: false,
130            chunk_dims: None,
131            max_shape: None,
132            is_compact: false,
133            early_allocation: false,
134            deflate_level: None,
135            shuffle: false,
136            custom_pipeline: None,
137            group_path: Some(group_path),
138            fill_value: None,
139            fill_time: None,
140            datatype_override: None,
141            committed_type: None,
142            references: None,
143            external: None,
144            efile_prefix: None,
145            virtual_mappings: Vec::new(),
146            _marker: std::marker::PhantomData,
147        }
148    }
149
150    /// Set the dataset dimensions.
151    ///
152    /// This is required before calling [`create`](Self::create), unless
153    /// [`null`](Self::null) was called instead.
154    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
155    #[must_use]
156    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
157        self.shape = Some(dims.as_ref().to_vec());
158        self
159    }
160
161    /// Create a scalar (0-dimensional) dataset holding a single value.
162    #[must_use]
163    pub fn scalar(mut self) -> Self {
164        self.shape = Some(vec![]);
165        self
166    }
167
168    /// Create a dataset with the NULL dataspace: no elements at all.
169    ///
170    /// Distinct from [`scalar`](Self::scalar), which holds exactly one
171    /// element. A NULL dataset holds zero bytes of data and cannot be
172    /// written to — [`write_raw`](H5Dataset::write_raw) and
173    /// [`write_raw_bytes`](H5Dataset::write_raw_bytes) return an error, and
174    /// it cannot be chunked or filtered, matching h5py's `h5py.Empty`.
175    #[must_use]
176    pub fn null(mut self) -> Self {
177        self.is_null = true;
178        self
179    }
180
181    /// Set chunk dimensions for chunked storage.
182    ///
183    /// When set, the dataset uses chunked storage with the extensible array
184    /// index. You should also call [`max_shape`](Self::max_shape) or
185    /// [`resizable`](Self::resizable) to allow extending.
186    #[must_use]
187    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
188        self.chunk_dims = Some(chunk_dims.to_vec());
189        self
190    }
191
192    /// Make all dimensions unlimited (resizable).
193    ///
194    /// This sets max_dims to u64::MAX for all dimensions.
195    #[must_use]
196    pub fn resizable(mut self) -> Self {
197        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
198        self
199    }
200
201    /// Set maximum dimensions. `None` means unlimited for that dimension.
202    #[must_use]
203    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
204        self.max_shape = Some(max.to_vec());
205        self
206    }
207
208    /// Store the raw data inside the dataset's object header —
209    /// `H5Pset_layout(dcpl, H5D_COMPACT)`.
210    ///
211    /// A compact dataset costs no data block and no second seek to read, which
212    /// suits the small per-run constants an analysis file is full of. It is
213    /// bounded by what one object header message can hold
214    /// ([`MAX_COMPACT_DATA`](crate::MAX_COMPACT_DATA) bytes) and it
215    /// is fixed in size: [`chunk`](Self::chunk), a filter, and an unlimited
216    /// [`max_shape`](Self::max_shape) are all rejected at
217    /// [`create`](Self::create), as libhdf5 rejects them.
218    ///
219    /// ```no_run
220    /// # use rust_hdf5::H5File;
221    /// let file = H5File::create("compact.h5").unwrap();
222    /// let ds = file.new_dataset::<i32>()
223    ///     .shape([16])
224    ///     .compact()
225    ///     .create("data")
226    ///     .unwrap();
227    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
228    /// ```
229    #[must_use]
230    pub fn compact(mut self) -> Self {
231        self.is_compact = true;
232        self
233    }
234
235    /// Allocate the whole of a chunked dataset's storage at create —
236    /// `H5Pset_alloc_time(dcpl, H5D_ALLOC_TIME_EARLY)`, h5py's
237    /// `alloc_time=h5d.ALLOC_TIME_EARLY`.
238    ///
239    /// Every chunk exists, holding the fill value, before anything is
240    /// written, so an unwritten chunk costs a read of fill bytes rather than
241    /// a miss. On a fixed-shape unfiltered dataset that is also what lets
242    /// libhdf5 pick its cheapest chunk index — the *implicit* index, which
243    /// is no index at all: the chunks are one contiguous run in grid order
244    /// and a chunk's address is arithmetic. This builder makes the same
245    /// choice under the same conditions, so such a dataset is written with
246    /// no index structure in the file.
247    ///
248    /// Ignored by storage that has no chunk grid to allocate: contiguous,
249    /// compact and NULL-dataspace datasets.
250    ///
251    /// ```no_run
252    /// # use rust_hdf5::H5File;
253    /// let file = H5File::create("implicit.h5").unwrap();
254    /// let ds = file.new_dataset::<i32>()
255    ///     .shape([16])
256    ///     .chunk(&[4])
257    ///     .early_allocation()
258    ///     .create("data")
259    ///     .unwrap();
260    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
261    /// ```
262    #[must_use]
263    pub fn early_allocation(mut self) -> Self {
264        self.early_allocation = true;
265        self
266    }
267
268    /// Enable deflate (gzip) compression with the given level (0-9).
269    ///
270    /// Requires chunked storage (call `.chunk()` before `.create()`).
271    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
272    #[must_use]
273    pub fn deflate(mut self, level: u32) -> Self {
274        self.deflate_level = Some(level);
275        self
276    }
277
278    /// Enable the shuffle filter — `H5Pset_shuffle(dcpl)`, h5py's
279    /// `shuffle=True`.
280    ///
281    /// Shuffle reorders a chunk's bytes by their position within an element,
282    /// which typically improves how well a compressor behind it does on
283    /// numeric data. It is a permutation, not a compressor: on its own it
284    /// leaves the chunk exactly as large as it was, which is what
285    /// `H5Pset_shuffle` without a compressor writes. Combine it with
286    /// [`deflate`](Self::deflate) to compress the shuffled stream. Requires
287    /// chunked storage.
288    ///
289    /// The element width the filter records is the dataset's, so a
290    /// [`datatype`](Self::datatype) override is what it follows when the
291    /// stored element is not `T` itself.
292    #[must_use]
293    pub fn shuffle(mut self) -> Self {
294        self.shuffle = true;
295        self
296    }
297
298    /// Enable shuffle + deflate compression — the same pipeline as
299    /// `.shuffle().deflate(level)`.
300    ///
301    /// Shuffle reorders bytes by position within elements before compression,
302    /// which typically improves compression ratios for numeric data.
303    /// Requires chunked storage.
304    #[must_use]
305    pub fn shuffle_deflate(mut self, level: u32) -> Self {
306        self.shuffle = true;
307        self.deflate_level = Some(level);
308        self
309    }
310
311    /// Enable Zstandard compression with the given level (1-22, default 3).
312    ///
313    /// Requires chunked storage (call `.chunk()` before `.create()`).
314    #[must_use]
315    pub fn zstd(mut self, level: u32) -> Self {
316        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
317        self
318    }
319
320    /// Set a custom filter pipeline for compression.
321    ///
322    /// This takes precedence over [`deflate`](Self::deflate) and
323    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
324    #[must_use]
325    pub fn filter_pipeline(
326        mut self,
327        pipeline: crate::format::messages::filter::FilterPipeline,
328    ) -> Self {
329        self.custom_pipeline = Some(pipeline);
330        self
331    }
332
333    /// Override the stored element datatype.
334    ///
335    /// By default the dataset is created with the datatype derived from the
336    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
337    /// different on-disk datatype than the in-memory element type — for
338    /// example a reduced-precision fixed-point type that matches an N-bit
339    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
340    /// override must equal `T::element_size()`; the N-bit filter packs the
341    /// significant bits within that fixed footprint.
342    ///
343    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
344    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
345    #[must_use]
346    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
347        self.datatype_override = Some(dt);
348        self
349    }
350
351    /// Build the dataset on the committed (named) datatype at `path` —
352    /// h5py's `dtype=f["name"]`, `H5Dcreate2` with a committed type id.
353    ///
354    /// The dataset does not describe its type: its header stores a pointer to
355    /// that object, so the type is defined once and every dataset sharing it
356    /// is guaranteed to agree. The type comes from the committed object, so
357    /// this supersedes both `T` and [`datatype`](Self::datatype).
358    ///
359    /// The path is resolved at [`create`](Self::create), which fails when no
360    /// committed datatype is there — commit it with
361    /// [`H5File::commit_datatype`](crate::file::H5File::commit_datatype)
362    /// first.
363    ///
364    /// ```no_run
365    /// # use rust_hdf5::H5File;
366    /// # use rust_hdf5::format::messages::datatype::DatatypeMessage;
367    /// let file = H5File::create("committed.h5").unwrap();
368    /// file.commit_datatype("temperature", DatatypeMessage::f64_type()).unwrap();
369    /// file.new_dataset::<f64>()
370    ///     .committed_type("temperature")
371    ///     .shape([4])
372    ///     .create("readings")
373    ///     .unwrap();
374    /// ```
375    #[must_use]
376    pub fn committed_type(mut self, path: &str) -> Self {
377        self.committed_type = Some(path.to_string());
378        self
379    }
380
381    /// Store object references — h5py's `h5py.ref_dtype`.
382    ///
383    /// The elements are written with
384    /// [`write_object_references`](H5Dataset::write_object_references) and
385    /// name objects by path. The element width is the file's address size, so
386    /// the datatype is resolved at [`create`](Self::create) rather than here;
387    /// it overrides both `T` and any [`datatype`](Self::datatype) call.
388    ///
389    /// ```no_run
390    /// # use rust_hdf5::H5File;
391    /// let file = H5File::create("refs.h5").unwrap();
392    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
393    /// let refs = file.new_dataset::<u64>()
394    ///     .object_references()
395    ///     .shape([1])
396    ///     .create("refs")
397    ///     .unwrap();
398    /// refs.write_object_references(&["/target"]).unwrap();
399    /// file.close().unwrap();
400    /// ```
401    #[must_use]
402    pub fn object_references(mut self) -> Self {
403        self.references = Some(ReferenceElement::Object);
404        self
405    }
406
407    /// Store revised object references — the 1.12 `H5T_STD_REF`.
408    ///
409    /// Same paths and same [`write_object_references`](H5Dataset::write_object_references)
410    /// call as [`object_references`](Self::object_references); only the stored
411    /// element differs, carrying the reference's kind alongside the address so
412    /// one datatype can hold every reference kind. h5py cannot read it, so
413    /// prefer the pre-1.12 form for files h5py will open.
414    ///
415    /// ```no_run
416    /// # use rust_hdf5::H5File;
417    /// let file = H5File::create("stdrefs.h5").unwrap();
418    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
419    /// let refs = file.new_dataset::<u64>()
420    ///     .std_object_references()
421    ///     .shape([1])
422    ///     .create("refs")
423    ///     .unwrap();
424    /// refs.write_object_references(&["/target"]).unwrap();
425    /// file.close().unwrap();
426    /// ```
427    #[must_use]
428    pub fn std_object_references(mut self) -> Self {
429        self.references = Some(ReferenceElement::Revised);
430        self
431    }
432
433    /// Store revised region references — `H5R_DATASET_REGION2`, written into
434    /// the same `H5T_STD_REF` datatype
435    /// [`std_object_references`](Self::std_object_references) makes.
436    ///
437    /// The elements are written with
438    /// [`write_std_region_references`](H5Dataset::write_std_region_references).
439    /// What distinguishes them from the pre-1.12
440    /// [`region_references`](Self::region_references) is the element, not the
441    /// datatype: a 1.12 element names its own kind, so one dataset of this type
442    /// may hold object, region and attribute references together. h5py 3.15
443    /// cannot read any of them, so prefer the pre-1.12 form for files h5py will
444    /// open.
445    ///
446    /// ```no_run
447    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, LibverBound, Selection};
448    /// let file = H5File::options().libver(LibverBound::V112).create("stdregions.h5").unwrap();
449    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
450    /// let refs = file.new_dataset::<u64>()
451    ///     .std_region_references()
452    ///     .shape([1])
453    ///     .create("refs")
454    ///     .unwrap();
455    /// let rows = Selection::Hyperslab {
456    ///     rank: 1,
457    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
458    /// };
459    /// refs.write_std_region_references(&[("/target", rows)]).unwrap();
460    /// file.close().unwrap();
461    /// ```
462    #[must_use]
463    pub fn std_region_references(self) -> Self {
464        self.std_object_references()
465    }
466
467    /// Store attribute references — `H5R_ATTR`, the one reference kind with no
468    /// pre-1.12 form, in the same `H5T_STD_REF` datatype
469    /// [`std_object_references`](Self::std_object_references) makes.
470    ///
471    /// The elements are written with
472    /// [`write_attribute_references`](H5Dataset::write_attribute_references) and
473    /// name an object and one of its attributes. h5py 3.15 cannot read them.
474    ///
475    /// ```no_run
476    /// # use rust_hdf5::{H5File, LibverBound};
477    /// let file = H5File::options().libver(LibverBound::V112).create("attrrefs.h5").unwrap();
478    /// let target = file.new_dataset::<i32>().shape([4]).create("target").unwrap();
479    /// target.new_attr::<i32>().shape([3]).create("note").unwrap()
480    ///     .write_array(&[7i32, 8, 9]).unwrap();
481    /// let refs = file.new_dataset::<u64>()
482    ///     .attribute_references()
483    ///     .shape([1])
484    ///     .create("refs")
485    ///     .unwrap();
486    /// refs.write_attribute_references(&[("/target", "note")]).unwrap();
487    /// file.close().unwrap();
488    /// ```
489    #[must_use]
490    pub fn attribute_references(self) -> Self {
491        self.std_object_references()
492    }
493
494    /// Store dataset region references — h5py's `h5py.regionref_dtype`.
495    ///
496    /// The elements are written with
497    /// [`write_region_references`](H5Dataset::write_region_references) and name
498    /// a dataset plus a selection over it. The element is a global-heap id, so
499    /// its width follows the file's address size and the datatype is resolved
500    /// at [`create`](Self::create) rather than here; it overrides both `T` and
501    /// any [`datatype`](Self::datatype) call.
502    ///
503    /// ```no_run
504    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, Selection};
505    /// let file = H5File::create("regions.h5").unwrap();
506    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
507    /// let refs = file.new_dataset::<u64>()
508    ///     .region_references()
509    ///     .shape([1])
510    ///     .create("refs")
511    ///     .unwrap();
512    /// let rows = Selection::Hyperslab {
513    ///     rank: 1,
514    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
515    /// };
516    /// refs.write_region_references(&[("/target", rows)]).unwrap();
517    /// file.close().unwrap();
518    /// ```
519    #[must_use]
520    pub fn region_references(mut self) -> Self {
521        self.references = Some(ReferenceElement::Region);
522        self
523    }
524
525    /// Keep the raw data in files outside this one — `H5Pset_external`,
526    /// h5py's `external=[(name, offset, size)]`.
527    ///
528    /// Each entry is a file name, the byte offset in it where that entry's
529    /// region starts, and how many bytes of the dataset it holds; the entries
530    /// concatenate, in order, into the dataset's bytes and together must cover
531    /// them. A relative name is resolved against `HDF5_EXTFILE_PREFIX` the way
532    /// libhdf5 resolves it, so the same name reads back through this crate and
533    /// through h5py. The storage is contiguous by definition, which rules out
534    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
535    /// [`null`](Self::null) and either reference kind.
536    ///
537    /// The named files are created on first write and never truncated, so
538    /// several datasets may own disjoint ranges of one file.
539    ///
540    /// The last entry may take the unlimited size
541    /// [`external_file_list::UNLIMITED`](crate::format::messages::external_file_list::UNLIMITED)
542    /// (`H5O_EFL_UNLIMITED`), which makes it absorb the whole rest of the
543    /// dataset however far it grows. A dataset whose
544    /// [`max_shape`](Self::max_shape) is unlimited must have one, since no
545    /// finite reservation could cover it, and only the first dimension may be
546    /// extendible — both `H5D__efl_construct`'s rules.
547    ///
548    /// ```no_run
549    /// # use rust_hdf5::H5File;
550    /// let file = H5File::create("ext.h5").unwrap();
551    /// let ds = file.new_dataset::<i32>()
552    ///     .shape([16])
553    ///     .external(&[("ext.raw", 0, 64)])
554    ///     .create("data")
555    ///     .unwrap();
556    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
557    /// ```
558    #[must_use]
559    pub fn external(mut self, files: &[(&str, u64, u64)]) -> Self {
560        self.external = Some(
561            files
562                .iter()
563                .map(|&(name, offset, size)| (name.to_string(), offset, size))
564                .collect(),
565        );
566        self
567    }
568
569    /// `H5Pset_efile_prefix` on the dapl `H5Dcreate2` takes — the directory
570    /// the raw data files named by [`external`](Self::external) are created
571    /// under, and looked for under on every later write through this handle.
572    ///
573    /// `H5D__create` builds `dset->shared->extfile_prefix` from the dapl
574    /// (H5Dint.c:1318) and `H5D__efl_write` joins each slot name against it
575    /// with the same single-path `H5_combine_path` the read side uses
576    /// (H5Defl.c:429-431) — so this decides where the bytes land, and the
577    /// prefix a later reader names must agree for it to find them.
578    ///
579    /// Measured under libhdf5 1.14.6 and 2.0.0: writing through a dapl that
580    /// names a directory creates the raw data file there and nowhere else,
581    /// and a directory that does not exist fails the write outright rather
582    /// than being created.
583    ///
584    /// It shares [`DatasetAccess::efile_prefix`]'s rules, both being
585    /// `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX` shadows this outright
586    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
587    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
588    /// (:1098-1102), which leaves a stored name to resolve against the
589    /// process's current directory.
590    ///
591    /// Ignored by a dataset that names no external files, which has no slot
592    /// name to join.
593    #[must_use]
594    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
595        self.efile_prefix = Some(prefix.into());
596        self
597    }
598
599    /// Map part of this dataset onto part of a dataset in another file, making
600    /// it virtual — `H5Pset_virtual`, one `VirtualLayout[...] =
601    /// VirtualSource(...)` assignment in h5py.
602    ///
603    /// The arguments are `H5Pset_virtual`'s, in its order: which elements of
604    /// *this* dataset the mapping fills, the file and dataset the data comes
605    /// from, and which elements of that source dataset it comes from. Call it
606    /// once per mapping; they apply in the order given, which is the order
607    /// libhdf5 resolves overlapping ones in.
608    ///
609    /// The source file is named exactly as stored — resolved against
610    /// `HDF5_VDS_PREFIX`, or the virtual dataset's own directory, when the
611    /// file is read — and `"."` means this file. Nothing is opened or checked
612    /// here: a source that does not exist yet is legal, and reads of the
613    /// unmapped or unresolvable parts return the [`fill_value`](Self::fill_value).
614    ///
615    /// A virtual dataset stores nothing of its own, which rules out
616    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
617    /// [`null`](Self::null), [`external`](Self::external) and either reference
618    /// kind — and makes writing to it an error, since its elements belong to
619    /// the source datasets.
620    ///
621    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: such a
622    /// mapping grows with its source, and the dataset's extent in that
623    /// dimension is whatever the sources reachable when it is opened supply
624    /// (`H5D__virtual_set_extent_unlim`). Give it a
625    /// [`max_shape`](Self::max_shape) unlimited in the same dimension, as
626    /// libhdf5 requires of the dataspace behind one.
627    ///
628    /// A source name may carry libhdf5's `printf`-style substitutions: `%b`
629    /// is the block index and `%%` an escaped literal `%`. One such mapping
630    /// stands for the family of source datasets that fill the successive
631    /// blocks of an unlimited virtual selection, so it is legal only with an
632    /// unlimited virtual selection over a limited source selection, and the
633    /// dataset's extent stops at the first block whose source is missing.
634    ///
635    /// ```no_run
636    /// # use rust_hdf5::{H5File, Selection};
637    /// let file = H5File::create("vds.h5").unwrap();
638    /// let ds = file.new_dataset::<i32>()
639    ///     .shape([16])
640    ///     .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
641    ///     .create("vds")
642    ///     .unwrap();
643    /// ```
644    #[must_use]
645    pub fn virtual_mapping(
646        mut self,
647        virtual_selection: Selection,
648        source_file: &str,
649        source_dataset: &str,
650        source_selection: Selection,
651    ) -> Self {
652        self.virtual_mappings.push(VirtualMapping {
653            source_file_name: source_file.to_string(),
654            source_dset_name: source_dataset.to_string(),
655            source_selection,
656            virtual_selection,
657        });
658        self
659    }
660
661    /// Set a user-defined fill value for unwritten elements.
662    ///
663    /// Without this, datasets use the HDF5 default zero-fill. When set,
664    /// the value is written into the dataset's fill-value message
665    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
666    /// unwritten regions as this value rather than zero.
667    ///
668    /// ```no_run
669    /// # use rust_hdf5::H5File;
670    /// let file = H5File::create("fv.h5").unwrap();
671    /// let ds = file.new_dataset::<f32>()
672    ///     .shape(&[100])
673    ///     .fill_value(f32::NAN)
674    ///     .create("data")
675    ///     .unwrap();
676    /// ```
677    #[must_use]
678    pub fn fill_value(mut self, value: T) -> Self {
679        let es = T::element_size();
680        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
681        // well-defined byte representation; `element_size()` matches
682        // `size_of::<T>()`. The slice borrows `value` only for this call.
683        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
684        self.fill_value = Some(raw.to_vec());
685        self
686    }
687
688    /// Set when the fill value is written into allocated storage —
689    /// `H5Pset_fill_time`.
690    ///
691    /// Without this, a dataset gets [`FillTime::IfSet`]
692    /// (`H5D_CRT_FILL_TIME_DEF`), the default every dataset creation
693    /// property list carries. [`FillTime::Never`] applies to a dataset with
694    /// no fill value too: it only stops this writer's own eager tiling of
695    /// the value into newly allocated storage, not the default zero-fill
696    /// that storage already has, so its only observable effect is on a
697    /// dataset that also calls [`fill_value`](Self::fill_value).
698    ///
699    /// ```no_run
700    /// # use rust_hdf5::{FillTime, H5File};
701    /// let file = H5File::create("fv.h5").unwrap();
702    /// let ds = file.new_dataset::<f32>()
703    ///     .shape(&[100])
704    ///     .fill_value(f32::NAN)
705    ///     .fill_time(FillTime::Never)
706    ///     .create("data")
707    ///     .unwrap();
708    /// ```
709    #[must_use]
710    pub fn fill_time(mut self, time: FillTime) -> Self {
711        self.fill_time = Some(time);
712        self
713    }
714
715    /// Finalize and create the dataset with the given `name`.
716    ///
717    /// The name is the link name within the root group (e.g. `"data"` or
718    /// `"group1/data"` once nested groups are supported).
719    pub fn create(self, name: &str) -> Result<H5Dataset> {
720        // A committed type is resolved before the dataset exists and recorded
721        // after, here rather than in each storage path: every path reaches
722        // this one return, so a dataset can never be built on a committed
723        // type and then fail to say so — which would silently write the type
724        // out in full instead of pointing at the object.
725        let committed = self.resolve_committed_type()?;
726        let file_inner = clone_inner(&self.file_inner);
727        let ds = self.create_object(name, committed.as_ref().map(|(_, dt)| dt.clone()))?;
728        if let (Some((share, _)), DatasetInfo::Writer { index, .. }) = (committed, &ds.info) {
729            let inner = borrow_inner(&file_inner);
730            if let H5FileInner::Writer(writer) = &*inner {
731                writer.share_committed_type(*index, share);
732            }
733        }
734        Ok(ds)
735    }
736
737    /// The committed datatype this dataset is built on, with the type it
738    /// holds; `None` when [`committed_type`](Self::committed_type) was not
739    /// called.
740    fn resolve_committed_type(&self) -> Result<Option<(usize, DatatypeMessage)>> {
741        let Some(path) = self.committed_type.as_deref() else {
742            return Ok(None);
743        };
744        if self.references.is_some() {
745            // Both name the stored type and they cannot both be it: the
746            // pointer would say the elements are the committed type while the
747            // reference writers write addresses. True of either reference
748            // kind — an object reference is an address, a region reference is
749            // a global-heap address plus a serialized selection.
750            return Err(Hdf5Error::InvalidState(
751                "a dataset cannot be built on a committed datatype and hold references".into(),
752            ));
753        }
754        let inner = borrow_inner(&self.file_inner);
755        match &*inner {
756            H5FileInner::Writer(writer) => Ok(Some(writer.committed_datatype_for_share(path)?)),
757            H5FileInner::Reader(_) => Err(Hdf5Error::InvalidState(
758                "cannot create a dataset in read mode".into(),
759            )),
760            H5FileInner::Closed => Err(Hdf5Error::InvalidState("file is closed".into())),
761        }
762    }
763
764    /// Everything [`create`](Self::create) does apart from recording the
765    /// committed-type share; `committed` is the type that object holds.
766    fn create_object(self, name: &str, committed: Option<DatatypeMessage>) -> Result<H5Dataset> {
767        // Build the full name: if created within a group, prefix with group path
768        let full_name = if let Some(ref gp) = self.group_path {
769            if gp == "/" {
770                name.to_string()
771            } else {
772                let trimmed = gp.trim_start_matches('/');
773                format!("{}/{}", trimmed, name)
774            }
775        } else {
776            name.to_string()
777        };
778
779        let datatype = if let Some(kind) = self.references {
780            // The element is measured in file addresses, and only the writer
781            // knows how wide one is for this file.
782            let inner = borrow_inner(&self.file_inner);
783            match &*inner {
784                H5FileInner::Writer(writer) => kind.datatype(writer.ctx()),
785                H5FileInner::Reader(_) => {
786                    return Err(Hdf5Error::InvalidState(
787                        "cannot create a dataset in read mode".into(),
788                    ))
789                }
790                H5FileInner::Closed => {
791                    return Err(Hdf5Error::InvalidState("file is closed".into()))
792                }
793            }
794        } else if let Some(dt) = committed {
795            // The object header holds the type; the dataset stores a pointer
796            // to it, but every size and payload check still needs the type
797            // itself.
798            dt
799        } else {
800            self.datatype_override.clone().unwrap_or_else(T::hdf5_type)
801        };
802        // Size one element from the on-disk datatype, not the carrier `T`. For
803        // the default path this equals `T::element_size()`; when a `datatype()`
804        // override is set (N-bit, or a runtime `CompoundType`), the stored type
805        // — not `T` — defines the element width, so the dataspace, the raw
806        // allocation, and the `write_raw` length check all agree with the bytes
807        // libhdf5/h5py will read.
808        let element_size = datatype.element_size() as usize;
809        // `fill_value` took the host image of a `T`; the fill-value message
810        // holds one element in the dataset's own datatype, so it is converted
811        // here — the order is only known once the override is resolved, and
812        // the builder's calls can arrive in either order.
813        let fill_value = match self.fill_value.as_deref() {
814            Some(bytes) => Some(to_stored_byte_order(bytes, &datatype, element_size)?.into_owned()),
815            None => None,
816        };
817
818        let wants_filter =
819            self.custom_pipeline.is_some() || self.shuffle || self.deflate_level.is_some();
820
821        // External storage *is* contiguous storage: the layout message says
822        // contiguous with an undefined address, and the External File List
823        // beside it says where the bytes really are. Every other storage class
824        // names bytes of its own, so none of them can also name these.
825        if self.external.is_some() {
826            if self.chunk_dims.is_some() || wants_filter || self.is_compact || self.is_null {
827                return Err(Hdf5Error::InvalidState(
828                    "a dataset whose raw data lives in external files is contiguous, so it \
829                     cannot also be chunked, filtered, compact or NULL"
830                        .into(),
831                ));
832            }
833            if self.references.is_some() {
834                return Err(Hdf5Error::InvalidState(
835                    "object and region references are stamped into the dataset's own \
836                     contiguous block, which a dataset stored in external files has none of"
837                        .into(),
838                ));
839            }
840        }
841
842        // A virtual dataset stores nothing of its own — its elements are read
843        // out of the datasets its mappings name — so it can be none of the
844        // storage classes that do, and there is no block for a reference
845        // writer to stamp into either.
846        if !self.virtual_mappings.is_empty() {
847            if self.chunk_dims.is_some()
848                || wants_filter
849                || self.is_compact
850                || self.is_null
851                || self.external.is_some()
852            {
853                return Err(Hdf5Error::InvalidState(
854                    "a virtual dataset's elements live in the datasets its mappings name, \
855                     so it cannot also be chunked, filtered, compact, NULL or stored in \
856                     external files"
857                        .into(),
858                ));
859            }
860            if self.references.is_some() {
861                return Err(Hdf5Error::InvalidState(
862                    "object and region references are stamped into the dataset's own \
863                     contiguous block, which a virtual dataset has none of"
864                        .into(),
865                ));
866            }
867        }
868
869        if self.is_null {
870            // A NULL dataspace holds no elements at all: no chunk grid to
871            // scatter into, no raw image to put in an object header, no fill
872            // value to apply to unwritten elements (there are none), matching
873            // upstream's rejection of these combinations (`H5Dchunk.c`'s
874            // chunked-layout dataspace check).
875            if self.chunk_dims.is_some() || wants_filter || self.is_compact {
876                return Err(Hdf5Error::InvalidState(
877                    "a NULL dataspace dataset cannot be chunked, filtered or compact".into(),
878                ));
879            }
880            if fill_value.is_some() {
881                return Err(Hdf5Error::InvalidState(
882                    "a NULL dataspace dataset cannot have a fill value".into(),
883                ));
884            }
885            if self.fill_time.is_some() {
886                return Err(Hdf5Error::InvalidState(
887                    "a NULL dataspace dataset cannot have a fill time".into(),
888                ));
889            }
890
891            let index = {
892                let inner = borrow_inner(&self.file_inner);
893                match &*inner {
894                    H5FileInner::Writer(writer) => {
895                        let idx = writer.create_null_dataset(&full_name, datatype)?;
896                        if let Some(ref gp) = self.group_path {
897                            if gp != "/" {
898                                writer.assign_dataset_to_group(gp, idx)?;
899                            }
900                        }
901                        idx
902                    }
903                    H5FileInner::Reader(_) => {
904                        return Err(Hdf5Error::InvalidState(
905                            "cannot create a dataset in read mode".into(),
906                        ));
907                    }
908                    H5FileInner::Closed => {
909                        return Err(Hdf5Error::InvalidState("file is closed".into()));
910                    }
911                }
912            };
913
914            return Ok(H5Dataset {
915                file_inner: clone_inner(&self.file_inner),
916                info: DatasetInfo::Writer {
917                    index,
918                    shape: Vec::new(),
919                    element_size,
920                    chunk_index: None,
921                    is_null: true,
922                },
923                _open: None,
924            });
925        }
926
927        let shape = self.shape.ok_or_else(|| {
928            Hdf5Error::InvalidState("shape must be set before calling create()".into())
929        })?;
930        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
931
932        if self.is_compact {
933            // The raw data is the layout message, so there is no chunk grid to
934            // filter and no room to grow into: `H5D__compact_construct` refuses
935            // a max dimension above the current one, and `H5Pset_layout` and
936            // `H5Pset_chunk` overwrite each other rather than combining.
937            if self.chunk_dims.is_some() || wants_filter {
938                return Err(Hdf5Error::InvalidState(
939                    "a compact dataset stores its data in the object header, so it \
940                     cannot be chunked or filtered"
941                        .into(),
942                ));
943            }
944            if self
945                .max_shape
946                .as_ref()
947                .is_some_and(|max| max.iter().zip(&shape).any(|(m, &d)| *m != Some(d)))
948            {
949                return Err(Hdf5Error::InvalidState(
950                    "a compact dataset cannot be extendible: its maximum shape must \
951                     equal its shape"
952                        .into(),
953                ));
954            }
955
956            let index = {
957                let inner = borrow_inner(&self.file_inner);
958                match &*inner {
959                    H5FileInner::Writer(writer) => {
960                        let idx = writer.create_compact_dataset(&full_name, datatype, &dims_u64)?;
961                        // Set before the fill value: NEVER must be in place
962                        // before that call decides whether to eager-tile it.
963                        if let Some(time) = self.fill_time {
964                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
965                        }
966                        if let Some(ref fv) = fill_value {
967                            writer.set_dataset_fill_value(idx, fv.clone())?;
968                        }
969                        idx
970                    }
971                    H5FileInner::Reader(_) => {
972                        return Err(Hdf5Error::InvalidState(
973                            "cannot create a dataset in read mode".into(),
974                        ));
975                    }
976                    H5FileInner::Closed => {
977                        return Err(Hdf5Error::InvalidState("file is closed".into()));
978                    }
979                }
980            };
981
982            return Ok(H5Dataset {
983                file_inner: clone_inner(&self.file_inner),
984                info: DatasetInfo::Writer {
985                    index,
986                    shape,
987                    element_size,
988                    chunk_index: None,
989                    is_null: false,
990                },
991                _open: None,
992            });
993        }
994
995        if !self.virtual_mappings.is_empty() {
996            let index = {
997                let inner = borrow_inner(&self.file_inner);
998                match &*inner {
999                    H5FileInner::Writer(writer) => {
1000                        let idx = writer.create_virtual_dataset(
1001                            &full_name,
1002                            datatype,
1003                            &dims_u64,
1004                            self.max_shape
1005                                .as_ref()
1006                                .map(|max| {
1007                                    max.iter()
1008                                        .map(|m| m.map_or(u64::MAX, |v| v as u64))
1009                                        .collect::<Vec<u64>>()
1010                                })
1011                                .as_deref(),
1012                            &self.virtual_mappings,
1013                        )?;
1014                        // The fill value is what a read of an unmapped — or
1015                        // unresolvable — element returns, so it is the one
1016                        // dataset property a virtual dataset carries about its
1017                        // own elements. Nothing is tiled into storage: it has
1018                        // none.
1019                        // Set before the fill value: NEVER must be in place
1020                        // before that call decides whether to eager-tile it.
1021                        if let Some(time) = self.fill_time {
1022                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1023                        }
1024                        if let Some(ref fv) = fill_value {
1025                            writer.set_dataset_fill_value(idx, fv.clone())?;
1026                        }
1027                        idx
1028                    }
1029                    H5FileInner::Reader(_) => {
1030                        return Err(Hdf5Error::InvalidState(
1031                            "cannot create a dataset in read mode".into(),
1032                        ));
1033                    }
1034                    H5FileInner::Closed => {
1035                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1036                    }
1037                }
1038            };
1039
1040            return Ok(H5Dataset {
1041                file_inner: clone_inner(&self.file_inner),
1042                info: DatasetInfo::Writer {
1043                    index,
1044                    shape,
1045                    element_size,
1046                    chunk_index: None,
1047                    is_null: false,
1048                },
1049                _open: None,
1050            });
1051        }
1052
1053        // A filter pipeline requires chunked storage. When a filter is
1054        // requested without explicit chunk dimensions, store the whole
1055        // dataset as a single chunk instead of silently dropping the filter
1056        // on the contiguous path. (This is one whole-dataset chunk, not
1057        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
1058        // for large datasets.)
1059        let auto_chunk: Option<Vec<usize>> =
1060            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
1061                Some(shape.iter().map(|&d| d.max(1)).collect())
1062            } else {
1063                None
1064            };
1065
1066        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
1067            // Chunked dataset
1068            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
1069            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
1070                max.iter()
1071                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
1072                    .collect()
1073            } else {
1074                // Default: max = current
1075                dims_u64.clone()
1076            };
1077
1078            // The file's format settles the question before the shape gets a
1079            // say. `H5D__chunk_set_info` reaches the index-selection block
1080            // only once the data layout message is at version 4 (H5Dchunk.c:936)
1081            // — which the file's library-version bound decides, not the
1082            // dataspace — and below it the version-3 message carries a
1083            // version-1 B-tree and nothing else. The writer owns that reading
1084            // of `H5O_layout_ver_bounds`; the chunk's byte count is the one
1085            // input from here, a chunk over 4 GiB being the one thing that
1086            // forces the newer message whatever the bound says.
1087            let chunk_bytes = chunk_u64.iter().product::<u64>() * element_size as u64;
1088            let v110_indexing = match &*borrow_inner(&self.file_inner) {
1089                H5FileInner::Writer(writer) => writer.uses_v110_chunk_indexing(chunk_bytes),
1090                // Neither can create a dataset at all; the creator below
1091                // reports which of the two it is.
1092                _ => true,
1093            };
1094
1095            // Inside the block libhdf5 selects the chunk index from the
1096            // dataspace and the creation properties, in this order
1097            // (`H5D__chunk_set_info`, H5Dchunk.c:955): a v2 B-tree for two or
1098            // more unlimited dimensions, an extensible array for exactly one;
1099            // for a fixed shape, the single-chunk index takes priority —
1100            // unconditional of filter or allocation time — whenever the shape
1101            // is exactly one whole chunk, ahead of the implicit index (no
1102            // filter, and early allocation, which is what puts every chunk at
1103            // a computable address) and the fixed array (everything else).
1104            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
1105            let one_chunk = chunk_u64 == dims_u64 && max_u64 == dims_u64;
1106            let kind = if !v110_indexing {
1107                ChunkIndexKind::BtreeV1
1108            } else if n_unlimited >= 2 {
1109                ChunkIndexKind::BtreeV2
1110            } else if n_unlimited == 1 {
1111                ChunkIndexKind::ExtensibleArray
1112            } else if one_chunk {
1113                ChunkIndexKind::SingleChunk
1114            } else if self.early_allocation && !wants_filter {
1115                ChunkIndexKind::Implicit
1116            } else {
1117                ChunkIndexKind::FixedArray
1118            };
1119
1120            let index = {
1121                let inner = borrow_inner(&self.file_inner);
1122                match &*inner {
1123                    H5FileInner::Writer(writer) => {
1124                        // The requested filter pipeline, if any. Every index
1125                        // builds it from the same options, so one owner
1126                        // resolves it: a second construction site is what let
1127                        // a request naming no compressor — shuffle on its own
1128                        // — fall through to unfiltered storage.
1129                        let explicit_pipeline = || {
1130                            use crate::format::messages::filter::FilterPipeline;
1131                            if let Some(p) = self.custom_pipeline.clone() {
1132                                return p;
1133                            }
1134                            // Shuffle records the width of the element it
1135                            // permutes, which is the stored one — a `datatype`
1136                            // override moves that away from `T`.
1137                            let es = element_size as u32;
1138                            match (self.shuffle, self.deflate_level) {
1139                                (true, Some(level)) => FilterPipeline::shuffle_deflate(es, level),
1140                                (true, None) => FilterPipeline::shuffle(es),
1141                                // deflate_level (checked by wants_filter).
1142                                (false, level) => FilterPipeline::deflate(level.unwrap()),
1143                            }
1144                        };
1145                        let idx = if kind == ChunkIndexKind::BtreeV1 {
1146                            // The classic index, which takes the pipeline the
1147                            // same way the others do — and is refused with it
1148                            // in a classic file, whose filter pipeline
1149                            // message is a version this crate does not write.
1150                            let pipeline = wants_filter.then(explicit_pipeline);
1151                            writer.create_btree_v1_dataset(
1152                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1153                            )?
1154                        } else if kind == ChunkIndexKind::BtreeV2 {
1155                            // Two or more unlimited dimensions: a v2 B-tree,
1156                            // whose records carry the stored size and filter
1157                            // mask when the dataset is compressed (libhdf5
1158                            // H5D_BT2_FILT).
1159                            if wants_filter {
1160                                writer.create_btree_v2_dataset_with_pipeline(
1161                                    &full_name,
1162                                    datatype,
1163                                    &dims_u64,
1164                                    &max_u64,
1165                                    &chunk_u64,
1166                                    explicit_pipeline(),
1167                                )?
1168                            } else {
1169                                writer.create_btree_v2_dataset(
1170                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1171                                )?
1172                            }
1173                        } else if kind == ChunkIndexKind::Implicit {
1174                            // No index structure at all: every chunk of the
1175                            // grid is allocated at create in one run, so
1176                            // there is no pipeline arm — a filter is what
1177                            // makes chunks different sizes, and this index
1178                            // has no room to say so.
1179                            writer.create_implicit_dataset(
1180                                &full_name, datatype, &dims_u64, &chunk_u64,
1181                            )?
1182                        } else if kind == ChunkIndexKind::SingleChunk {
1183                            // A fixed shape covered by exactly one chunk:
1184                            // libhdf5 picks this index ahead of Implicit and
1185                            // Fixed Array regardless of filter or allocation
1186                            // time. Filtered or not, it takes the same
1187                            // explicit pipeline the other indexes do; a
1188                            // filtered chunk's stored size isn't known ahead
1189                            // of its first write, so early allocation only
1190                            // ever applies to the unfiltered form.
1191                            if wants_filter {
1192                                writer.create_single_chunk_dataset_with_pipeline(
1193                                    &full_name,
1194                                    datatype,
1195                                    &dims_u64,
1196                                    &chunk_u64,
1197                                    explicit_pipeline(),
1198                                )?
1199                            } else {
1200                                writer.create_single_chunk_dataset(
1201                                    &full_name,
1202                                    datatype,
1203                                    &dims_u64,
1204                                    &chunk_u64,
1205                                    self.early_allocation,
1206                                )?
1207                            }
1208                        } else if kind == ChunkIndexKind::FixedArray {
1209                            // A chunked dataset with no unlimited dimension
1210                            // must use the fixed-array index — libhdf5
1211                            // rejects an extensible-array index here. A
1212                            // compressed fixed-shape dataset uses a *filtered*
1213                            // fixed array (FA client id 1). The maximum shape
1214                            // sizes the array, so a finite max above the
1215                            // current shape stays growable.
1216                            let pipeline = wants_filter.then(explicit_pipeline);
1217                            writer.create_fixed_array_dataset_with_max(
1218                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1219                            )?
1220                        } else if wants_filter {
1221                            // The extensible-array index takes the pipeline
1222                            // the same way, so it goes through the one owner
1223                            // too.
1224                            writer.create_chunked_dataset_with_pipeline(
1225                                &full_name,
1226                                datatype,
1227                                &dims_u64,
1228                                &max_u64,
1229                                &chunk_u64,
1230                                explicit_pipeline(),
1231                            )?
1232                        } else {
1233                            writer.create_chunked_dataset(
1234                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1235                            )?
1236                        };
1237                        // Set before the fill value: NEVER must be in place
1238                        // before that call decides whether to eager-tile it.
1239                        if let Some(time) = self.fill_time {
1240                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1241                        }
1242                        if let Some(ref fv) = fill_value {
1243                            writer.set_dataset_fill_value(idx, fv.clone())?;
1244                        }
1245                        idx
1246                    }
1247                    H5FileInner::Reader(_) => {
1248                        return Err(Hdf5Error::InvalidState(
1249                            "cannot create a dataset in read mode".into(),
1250                        ));
1251                    }
1252                    H5FileInner::Closed => {
1253                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1254                    }
1255                }
1256            };
1257
1258            Ok(H5Dataset {
1259                file_inner: clone_inner(&self.file_inner),
1260                info: DatasetInfo::Writer {
1261                    index,
1262                    shape,
1263                    element_size,
1264                    chunk_index: Some(kind),
1265                    is_null: false,
1266                },
1267                _open: None,
1268            })
1269        } else {
1270            // Contiguous dataset (original path)
1271            let efile_access = match self.efile_prefix.as_deref() {
1272                Some(p) => DatasetAccess::new().efile_prefix(p),
1273                None => DatasetAccess::new(),
1274            };
1275            let (index, open) = {
1276                let inner = borrow_inner(&self.file_inner);
1277                match &*inner {
1278                    H5FileInner::Writer(writer) => {
1279                        let idx = match self.external.as_deref() {
1280                            Some(files) => {
1281                                let slots: Vec<(&str, u64, u64)> = files
1282                                    .iter()
1283                                    .map(|(name, offset, size)| (name.as_str(), *offset, *size))
1284                                    .collect();
1285                                writer.create_external_dataset(
1286                                    &full_name,
1287                                    datatype,
1288                                    &dims_u64,
1289                                    self.max_shape
1290                                        .as_ref()
1291                                        .map(|max| {
1292                                            max.iter()
1293                                                .map(|m| m.map_or(u64::MAX, |v| v as u64))
1294                                                .collect::<Vec<u64>>()
1295                                        })
1296                                        .as_deref(),
1297                                    &slots,
1298                                )?
1299                            }
1300                            None => writer.create_dataset(&full_name, datatype, &dims_u64)?,
1301                        };
1302                        // Before anything that can write raw bytes: the
1303                        // prefix an external dataset's slot names are joined
1304                        // against is settled by the create, as
1305                        // `H5D__build_file_prefix` settles it for
1306                        // `H5D__create` (H5Dint.c:1318).
1307                        let open = writer.bind_efile_prefix(idx, &efile_access)?;
1308                        // Set before the fill value: NEVER must be in place
1309                        // before that call decides whether to eager-tile it.
1310                        if let Some(time) = self.fill_time {
1311                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1312                        }
1313                        if let Some(ref fv) = fill_value {
1314                            writer.set_dataset_fill_value(idx, fv.clone())?;
1315                        }
1316                        (idx, open)
1317                    }
1318                    H5FileInner::Reader(_) => {
1319                        return Err(Hdf5Error::InvalidState(
1320                            "cannot create a dataset in read mode".into(),
1321                        ));
1322                    }
1323                    H5FileInner::Closed => {
1324                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1325                    }
1326                }
1327            };
1328
1329            Ok(H5Dataset {
1330                file_inner: clone_inner(&self.file_inner),
1331                info: DatasetInfo::Writer {
1332                    index,
1333                    shape,
1334                    element_size,
1335                    chunk_index: None,
1336                    is_null: false,
1337                },
1338                _open: open,
1339            })
1340        }
1341    }
1342}
1343
1344// ---------------------------------------------------------------------------
1345// DatasetInfo
1346// ---------------------------------------------------------------------------
1347
1348/// Internal metadata about a dataset handle.
1349enum DatasetInfo {
1350    /// A dataset created via `new_dataset().create()` in write mode.
1351    Writer {
1352        /// Index into the writer's dataset list.
1353        index: usize,
1354        /// Shape (current dimensions).
1355        shape: Vec<usize>,
1356        /// Size of one element in bytes.
1357        element_size: usize,
1358        /// Which chunk index this dataset uses, `None` when its storage is
1359        /// not chunked. One field rather than a flag per index: a dataset
1360        /// has exactly one chunk index, and the flags could spell
1361        /// combinations ("not chunked, but indexed by a v2 B-tree") that no
1362        /// dataset has — which is what a write path reading only some of
1363        /// them turns into a write to the wrong index.
1364        chunk_index: Option<ChunkIndexKind>,
1365        /// Whether this is a NULL dataspace (no elements at all — distinct
1366        /// from a scalar, which holds exactly one). Always `false` when
1367        /// `chunk_index` is `Some`: a NULL dataspace can never be chunked.
1368        is_null: bool,
1369    },
1370    /// A dataset opened by name in read mode.
1371    Reader {
1372        /// The link name of the dataset.
1373        name: String,
1374        /// Shape (current dimensions).
1375        shape: Vec<usize>,
1376        /// Size of one element in bytes.
1377        element_size: usize,
1378    },
1379}
1380
1381// ---------------------------------------------------------------------------
1382// H5Dataset
1383// ---------------------------------------------------------------------------
1384
1385/// A handle to an HDF5 dataset, supporting typed read and write operations.
1386///
1387/// The dataset holds a shared reference to the file's I/O backend, so it
1388/// remains valid even if the originating [`H5File`](crate::file::H5File) is
1389/// moved or dropped (they share ownership via `Rc`).
1390pub struct H5Dataset {
1391    file_inner: SharedInner,
1392    info: DatasetInfo,
1393    /// Keeps this dataset's *open* alive for as long as the handle is, so
1394    /// the reader or the writer can tell whether a later open of the same
1395    /// name joins this one or starts fresh — libhdf5's `H5FO_opened`
1396    /// shared-info count (H5Dint.c:1496-1500). `None` for a dataset with no
1397    /// per-open answer to hold: in write mode, one whose raw data is in this
1398    /// file rather than in the files an external file list names.
1399    ///
1400    /// Held, never read: its whole job is to keep the reader's `Weak` on it
1401    /// upgradable until this handle goes away.
1402    _open: Option<crate::io::reader::DatasetOpenToken>,
1403}
1404
1405impl Drop for H5Dataset {
1406    /// Closing a virtual dataset's last handle closes the source files that
1407    /// open was holding, which is what `H5D__virtual_reset_layout` does at
1408    /// the last `H5Dclose` (H5Dvirtual.c:709-710, closing each
1409    /// `source_dset->dset` at :955 and with it the file that dataset kept
1410    /// open). Nothing else in this crate can end a virtual open, so this is
1411    /// where the reader is told.
1412    ///
1413    /// The token is dropped *before* the reader is asked, so the reader's
1414    /// `Weak` already reads dead for the handle going away here. A write-mode
1415    /// handle's token belongs to the writer's own external file prefix, which
1416    /// has no source files to close, so it takes no lock either.
1417    fn drop(&mut self) {
1418        let Some(open) = self._open.take() else {
1419            return;
1420        };
1421        drop(open);
1422        if matches!(self.info, DatasetInfo::Writer { .. }) {
1423            return;
1424        }
1425        let Some(mut inner) = crate::file::try_borrow_inner_mut(&self.file_inner) else {
1426            return;
1427        };
1428        if let crate::file::H5FileInner::Reader(reader) = &mut *inner {
1429            reader.release_closed_virtual_sources();
1430        }
1431    }
1432}
1433
1434/// One chunk's bytes on the way to the file, and who filtered them.
1435///
1436/// This is what separates a normal chunk write from a direct one; everything
1437/// else about placing a chunk is identical, so the two share a single dispatch.
1438#[derive(Clone, Copy)]
1439enum ChunkBytes<'a> {
1440    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
1441    /// are stored.
1442    Unfiltered(&'a [u8]),
1443    /// Bytes already in their stored form, with `filter_mask` naming the
1444    /// filters that were skipped.
1445    Prefiltered { data: &'a [u8], filter_mask: u32 },
1446}
1447
1448/// The byte order this build reads and writes natively.
1449pub(crate) const HOST_BYTE_ORDER: ByteOrder = if cfg!(target_endian = "big") {
1450    ByteOrder::BigEndian
1451} else {
1452    ByteOrder::LittleEndian
1453};
1454
1455/// The byte order this build does not read or write natively.
1456pub(crate) const FOREIGN_BYTE_ORDER: ByteOrder = match HOST_BYTE_ORDER {
1457    ByteOrder::LittleEndian => ByteOrder::BigEndian,
1458    ByteOrder::BigEndian => ByteOrder::LittleEndian,
1459};
1460
1461/// What a typed access has to do with an element image of a given datatype.
1462#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1463enum ByteOrderAction {
1464    /// Stored order is the host's: the image is already the typed value.
1465    Keep,
1466    /// The whole element is one scalar in the foreign order: reverse it.
1467    SwapElements,
1468    /// A composite storing something in the foreign order.
1469    Refuse,
1470}
1471
1472/// Classify a datatype for a typed access of element width `width`.
1473///
1474/// The single owner of the rule; both directions ask it, so a type a read
1475/// converts is exactly a type a write converts.
1476///
1477/// A composite element cannot be swapped as a unit — its members have their
1478/// own orders and offsets — so one that touches the foreign order is refused
1479/// rather than silently passed through in the wrong order.
1480fn byte_order_action(datatype: &DatatypeMessage, width: usize) -> ByteOrderAction {
1481    match datatype.scalar_byte_order() {
1482        Some(order) if order == FOREIGN_BYTE_ORDER && width > 1 => ByteOrderAction::SwapElements,
1483        Some(_) => ByteOrderAction::Keep,
1484        None if datatype.contains_byte_order(FOREIGN_BYTE_ORDER) => ByteOrderAction::Refuse,
1485        None => ByteOrderAction::Keep,
1486    }
1487}
1488
1489/// Why the stored image of an element is not already the host image of a
1490/// value of width `width` — `None` when it is, and a copying read would only
1491/// be memcpy-ing bytes it does not touch.
1492///
1493/// The two ways a stored element can need work before it is a value are the
1494/// two conversions a copying read performs in place: a byte-order swap
1495/// ([`to_host_byte_order`]) and the n-bit/scale-offset unpacking
1496/// (`Hdf5Reader::apply_post_filter_conversion`). Asking one question of both
1497/// is what lets a zero-copy view refuse exactly the datatypes a copying read
1498/// would have had to rewrite.
1499#[cfg(feature = "mmap")]
1500pub(crate) fn stored_image_mismatch(
1501    datatype: &DatatypeMessage,
1502    width: usize,
1503) -> Option<&'static str> {
1504    match byte_order_action(datatype, width) {
1505        ByteOrderAction::SwapElements => return Some("they are stored in the foreign byte order"),
1506        ByteOrderAction::Refuse => {
1507            return Some("it is a composite storing members in the foreign byte order")
1508        }
1509        ByteOrderAction::Keep => {}
1510    }
1511    if crate::format::nbit_scaleoffset::datatype_needs_bit_conversion(datatype) {
1512        return Some("the significant bits do not fill the stored element");
1513    }
1514    None
1515}
1516
1517/// Put a raw element image into host byte order, in place, for a typed read.
1518///
1519/// Every path that reinterprets the on-disk image as `T` — `read_raw`,
1520/// `read_slice`, `read_raw_into`, `read_slice_into` and their SWMR
1521/// counterparts — passes through here. Reinterpretation only yields the
1522/// stored value when the stored order is the host's.
1523///
1524/// A refused datatype is one no reinterpretation can decode;
1525/// [`H5Dataset::read_raw_bytes`] hands over the image for the caller to
1526/// decode member by member.
1527///
1528/// `width` is the element size, already checked equal to `T::element_size()`.
1529pub(crate) fn to_host_byte_order(
1530    bytes: &mut [u8],
1531    datatype: &DatatypeMessage,
1532    width: usize,
1533) -> Result<()> {
1534    match byte_order_action(datatype, width) {
1535        ByteOrderAction::Keep => {}
1536        ByteOrderAction::SwapElements => {
1537            for elem in bytes.chunks_exact_mut(width) {
1538                elem.reverse();
1539            }
1540        }
1541        ByteOrderAction::Refuse => {
1542            return Err(Hdf5Error::TypeMismatch(format!(
1543                "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a \
1544                 typed read cannot reinterpret element by element; read_raw_bytes() returns \
1545                 the image to decode member by member"
1546            )))
1547        }
1548    }
1549    Ok(())
1550}
1551
1552/// The stored byte image of each variable-length sequence in a batch.
1553///
1554/// The vlen writers take `&[&[T]]` and store one global-heap object per
1555/// sequence, so each sequence needs the same host-image-to-stored-image step
1556/// [`to_stored_byte_order`] performs for a fixed-shape write — a `T` is
1557/// written from its host bytes, and `T::hdf5_type()` declares little-endian.
1558/// Borrows on a little-endian host, which is every machine that does not have
1559/// to swap.
1560pub(crate) fn vlen_sequence_images<'a, T: H5Type>(
1561    items: &'a [&'a [T]],
1562) -> Result<Vec<std::borrow::Cow<'a, [u8]>>> {
1563    let base = T::hdf5_type();
1564    items
1565        .iter()
1566        .map(|item| {
1567            // Safety: the same contract `write_raw` relies on — `T: Copy +
1568            // 'static` is a numeric primitive whose byte image is its value —
1569            // and the extent comes from the slice itself, so it cannot name
1570            // memory past it. The result borrows `items` and outlives nothing.
1571            let host = unsafe {
1572                std::slice::from_raw_parts(item.as_ptr() as *const u8, std::mem::size_of_val(*item))
1573            };
1574            to_stored_byte_order(host, &base, T::element_size())
1575        })
1576        .collect()
1577}
1578
1579/// Put a typed value's host-order image into the order the datatype declares.
1580///
1581/// The write-side counterpart of [`to_host_byte_order`], and the one place
1582/// every path that hands a `&[T]` to the file — `write_raw`, `write_slice`,
1583/// `append`, and the builder's fill value — turns those bytes into stored
1584/// bytes. A `T` is written from its host image, so a dataset declaring the
1585/// foreign order would otherwise hold host bytes under that declaration: a
1586/// file that is wrong by its own header.
1587///
1588/// Borrows when the declared order is the host's, which is every write that
1589/// does not set a [`datatype`](DatasetBuilder::datatype) override.
1590///
1591/// `width` is the element size, already checked equal to `T::element_size()`.
1592pub(crate) fn to_stored_byte_order<'a>(
1593    bytes: &'a [u8],
1594    datatype: &DatatypeMessage,
1595    width: usize,
1596) -> Result<std::borrow::Cow<'a, [u8]>> {
1597    match byte_order_action(datatype, width) {
1598        ByteOrderAction::Keep => Ok(std::borrow::Cow::Borrowed(bytes)),
1599        ByteOrderAction::SwapElements => {
1600            let mut owned = bytes.to_vec();
1601            for elem in owned.chunks_exact_mut(width) {
1602                elem.reverse();
1603            }
1604            Ok(std::borrow::Cow::Owned(owned))
1605        }
1606        ByteOrderAction::Refuse => Err(Hdf5Error::TypeMismatch(format!(
1607            "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a typed \
1608             write cannot lay out element by element; write_raw_bytes() takes the image the \
1609             caller encodes member by member"
1610        ))),
1611    }
1612}
1613
1614/// Strip a fixed-string element's padding, leaving the bytes that carry the
1615/// value.
1616///
1617/// The rule itself lives with the datatype message
1618/// ([`fixed_string_content`]); this adds the element index a reserved padding
1619/// rule needs to be reported against.
1620fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
1621    crate::format::messages::datatype::fixed_string_content(elem, padding).ok_or_else(|| {
1622        Hdf5Error::InvalidState(format!(
1623            "string {index} uses padding rule {padding}, which the format reserves"
1624        ))
1625    })
1626}
1627
1628/// Decode one string element's bytes under the datatype's character set.
1629///
1630/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
1631/// `index` names the element in the error otherwise.
1632fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
1633    if lossy {
1634        return Ok(String::from_utf8_lossy(bytes).into_owned());
1635    }
1636    match charset {
1637        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
1638        0 => match bytes.iter().position(|&b| b >= 0x80) {
1639            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
1640            Some(at) => Err(Hdf5Error::InvalidState(format!(
1641                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
1642                bytes[at]
1643            ))),
1644        },
1645        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
1646            Hdf5Error::InvalidState(format!(
1647                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
1648            ))
1649        }),
1650        other => Err(Hdf5Error::InvalidState(format!(
1651            "string {index} uses character set {other}, which the format reserves"
1652        ))),
1653    }
1654}
1655
1656/// A dataset's storage layout class (read mode only) — `H5Pget_layout`'s
1657/// four values.
1658///
1659/// Distinct from [`ChunkIndex`], which names the structure a `Chunked`
1660/// layout's index uses; this only says which of the four storage classes
1661/// the dataset was created with.
1662#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1663pub enum StorageLayout {
1664    /// Raw data stored inline in the object header.
1665    Compact,
1666    /// Raw data in a single contiguous block — or, when the dataset also
1667    /// carries an external file list, in one or more blocks of an outside
1668    /// file instead ([`H5Dataset::external_files`]); the layout itself
1669    /// still reports `Contiguous` either way.
1670    Contiguous,
1671    /// Raw data split into fixed-size chunks, each independently
1672    /// allocated. [`H5Dataset::chunk_dims`] gives the chunk shape,
1673    /// [`H5Dataset::chunk_index`] the index structure.
1674    Chunked,
1675    /// No raw data of its own: every element comes from another dataset,
1676    /// possibly in another file ([`H5Dataset::virtual_mappings`]).
1677    Virtual,
1678}
1679
1680/// The chunk index structure a chunked dataset uses on disk (read mode
1681/// only) — which of libhdf5's chunk-lookup structures the layout message
1682/// names.
1683///
1684/// `BtreeV1` belongs to the version-3 chunked layout message (the only
1685/// index a file whose superblock predates version 2 can carry); the other
1686/// five are what a version-4 message's index-type byte selects, per
1687/// `H5D__layout_set_latest_indexing` (H5Dlayout.c).
1688#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1689pub enum ChunkIndex {
1690    /// Version-1 B-tree — the classic chunk index, and the only one a
1691    /// version-3 chunked layout message can carry.
1692    BtreeV1,
1693    /// Version-2 B-tree — two or more unlimited dimensions.
1694    BtreeV2,
1695    /// A single index entry for a dataset whose one chunk covers the whole
1696    /// dataspace (`dims == max_dims == chunk_dims`).
1697    SingleChunk,
1698    /// No index structure at all: chunk addresses are computed
1699    /// arithmetically over a contiguous run (no filter, early allocation).
1700    Implicit,
1701    /// Fixed-size array — a fixed shape needing per-chunk bookkeeping.
1702    FixedArray,
1703    /// Extensible array — exactly one unlimited dimension.
1704    ExtensibleArray,
1705}
1706
1707/// A dataset's fill-value state (read mode only) — `H5Pfill_value_defined`'s
1708/// tri-state (`H5D_fill_value_t`).
1709#[derive(Debug, Clone, PartialEq, Eq)]
1710pub enum FillValue {
1711    /// No fill value has ever been set: unwritten elements read back
1712    /// zero-filled, and no fill-value message named an explicit value.
1713    Default,
1714    /// The fill value was explicitly disabled: unallocated storage is never
1715    /// fill-initialized.
1716    Undefined,
1717    /// An explicit fill value, one element wide.
1718    UserDefined(Vec<u8>),
1719}
1720
1721/// When a dataset's fill value is written into allocated storage —
1722/// `H5Pset_fill_time`/`H5Pget_fill_time`'s `H5D_fill_time_t`.
1723///
1724/// Distinct from [`FillValue`], which says *what* the fill value is; this
1725/// says *when* it is written. The two agree everywhere except a dataset with
1726/// no fill value of its own: there, `Alloc` writes the default fill (zeros)
1727/// at allocation and `IfSet` writes nothing into space that already reads as
1728/// zeros — indistinguishable on disk in the value itself, only in this byte.
1729#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1730pub enum FillTime {
1731    /// Fill at allocation regardless of whether a fill value was ever set.
1732    Alloc,
1733    /// Never write the fill value into allocated storage.
1734    Never,
1735    /// Fill at allocation only when a fill value was set — the default
1736    /// every dataset gets unless [`fill_time`](DatasetBuilder::fill_time)
1737    /// says otherwise.
1738    IfSet,
1739}
1740
1741impl FillTime {
1742    /// The on-disk `H5D_fill_time_t` byte this variant is — what the writer
1743    /// stores and the fill-value message's write-time field carries.
1744    fn wire_byte(self) -> u8 {
1745        match self {
1746            Self::Alloc => 0,
1747            Self::Never => 1,
1748            Self::IfSet => 2,
1749        }
1750    }
1751}
1752
1753/// When a dataset's raw-data storage is allocated —
1754/// `H5Pset_alloc_time`/`H5Pget_alloc_time`'s `H5D_alloc_time_t`, read back
1755/// from the same fill-value message [`FillTime`] is.
1756///
1757/// Not user-settable: `H5P__set_layout` (H5Pdcpl.c) picks this from the
1758/// dataset's storage class alone (`H5D_ALLOC_TIME_DEFAULT` per layout —
1759/// compact is `Early`, chunked and virtual are `Incr`, contiguous is
1760/// `Late`), and this crate has no `DatasetBuilder` setter that overrides it.
1761/// [`H5Dataset::alloc_time`] exists to read back what the writer declared.
1762#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1763pub enum AllocTime {
1764    /// Space is allocated as soon as the dataset is created.
1765    Early,
1766    /// Space is allocated when data is first written.
1767    Late,
1768    /// Space is allocated incrementally, as chunks (or virtual source
1769    /// datasets) are written.
1770    Incr,
1771}
1772
1773/// Which mapped data an unlimited virtual dataset's extent covers —
1774/// libhdf5's `H5D_vds_view_t`, set with `H5Pset_virtual_view` and read back
1775/// with `H5Pget_virtual_view` (H5Pdapl.c:1067, :1102).
1776///
1777/// A *dataset access* property: it is never stored in the file, so it says
1778/// how *this* open reads a virtual dataset, not what its writer intended.
1779#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1780pub enum VirtualView {
1781    /// `H5D_VDS_LAST_AVAILABLE` — the extent reaches the end of the last
1782    /// mapped block that has a source, so a gap before it reads as the fill
1783    /// value. libhdf5's default (`H5D_ACS_VDS_VIEW_DEF`, H5Pdapl.c:62).
1784    #[default]
1785    LastAvailable,
1786    /// `H5D_VDS_FIRST_MISSING` — the extent stops where the first missing
1787    /// mapped block begins, so no unmapped block is inside it.
1788    ///
1789    /// Under this view libhdf5 ignores
1790    /// [`virtual_printf_gap`](DatasetAccess::virtual_printf_gap) entirely:
1791    /// `H5D__virtual_init` reads the gap property only for
1792    /// [`LastAvailable`](Self::LastAvailable) and forces it to 0 otherwise
1793    /// (H5Dvirtual.c:2182-2188).
1794    FirstMissing,
1795}
1796
1797/// The dataset *access* properties this crate models — libhdf5's
1798/// `H5P_DATASET_ACCESS` property list, as much of it as affects reading.
1799///
1800/// [`virtual_view`](Self::virtual_view) and
1801/// [`virtual_printf_gap`](Self::virtual_printf_gap) govern how a virtual
1802/// dataset's extent is resolved when it is opened
1803/// (`H5D__virtual_set_extent_unlim`, H5Dvirtual.c:1386);
1804/// [`virtual_prefix`](Self::virtual_prefix) and
1805/// [`efile_prefix`](Self::efile_prefix) say where the *other files* a
1806/// dataset's data lives in are looked for. None of them is stored in the
1807/// file: opening a dataset without naming them reads it exactly as
1808/// libhdf5's default dapl does.
1809///
1810/// Pass one to [`H5File::dataset_with`](crate::H5File::dataset_with).
1811///
1812/// ```no_run
1813/// use rust_hdf5::{DatasetAccess, H5File, VirtualView};
1814///
1815/// let file = H5File::open("vds.h5").unwrap();
1816/// let access = DatasetAccess::new()
1817///     .virtual_view(VirtualView::LastAvailable)
1818///     .virtual_printf_gap(2);
1819/// let ds = file.dataset_with("vds", access).unwrap();
1820/// ```
1821#[derive(Debug, Clone, PartialEq, Eq, Default)]
1822pub struct DatasetAccess {
1823    view: VirtualView,
1824    printf_gap: u64,
1825    virtual_prefix: Option<String>,
1826    efile_prefix: Option<String>,
1827}
1828
1829impl DatasetAccess {
1830    /// A property list holding libhdf5's defaults —
1831    /// [`VirtualView::LastAvailable`] and a printf gap of 0, the values
1832    /// `H5D_ACS_VDS_VIEW_DEF` and `H5D_ACS_VDS_PRINTF_GAP_DEF` register
1833    /// (H5Pdapl.c:62, :67).
1834    pub fn new() -> Self {
1835        Self::default()
1836    }
1837
1838    /// `H5Pset_virtual_view` (H5Pdapl.c:1067). The two legal values are the
1839    /// two [`VirtualView`] variants, so the "not a valid bounds option"
1840    /// argument check that call makes has nothing to reject here.
1841    pub fn virtual_view(mut self, view: VirtualView) -> Self {
1842        self.view = view;
1843        self
1844    }
1845
1846    /// `H5Pset_virtual_printf_gap` (H5Pdapl.c:1207): how many consecutive
1847    /// missing printf-named source datasets the extent resolution looks past
1848    /// before it stops. 0 — the default — stops at the first one missing.
1849    ///
1850    /// `u64::MAX` is libhdf5's `HSIZE_UNDEF`, which that call rejects as "not
1851    /// a valid printf gap size"; here the rejection surfaces from the open
1852    /// that uses the property, since a builder method has no way to report
1853    /// it.
1854    pub fn virtual_printf_gap(mut self, gap: u64) -> Self {
1855        self.printf_gap = gap;
1856        self
1857    }
1858
1859    /// `H5Pget_virtual_view` (H5Pdapl.c:1102).
1860    pub fn view(&self) -> VirtualView {
1861        self.view
1862    }
1863
1864    /// `H5Pset_virtual_prefix` (H5Pdapl.c:1478): a directory a virtual
1865    /// dataset's *source file names* are looked for under, before the
1866    /// virtual file's own directory and after `HDF5_VDS_PREFIX`.
1867    ///
1868    /// It is the third step of `H5F_prefix_open_file`'s search order
1869    /// (H5Fint.c:938-950), and it is reached only when `HDF5_VDS_PREFIX` is
1870    /// unset or empty: `H5D__build_file_prefix` reads the environment first
1871    /// and falls back to this property (H5Dint.c:1077-1082), so an
1872    /// environment prefix shadows this one outright rather than being tried
1873    /// alongside it.
1874    ///
1875    /// A leading `${ORIGIN}` stands for the directory holding the virtual
1876    /// dataset's own file (H5Dint.c:1105-1113), and `"."` or `""` means "no
1877    /// prefix" (:1096-1100), both exactly as for the environment variable.
1878    ///
1879    /// Like the other two, this is a *dataset access* property that is never
1880    /// stored in the file, and the first open of a virtual dataset fixes it
1881    /// for every open that overlaps it.
1882    pub fn virtual_prefix(mut self, prefix: impl Into<String>) -> Self {
1883        self.virtual_prefix = Some(prefix.into());
1884        self
1885    }
1886
1887    /// `H5Pset_efile_prefix` (H5Pdapl.c:1392): a directory the *raw data
1888    /// files* of a dataset stored through an external file list are looked
1889    /// for under.
1890    ///
1891    /// This one takes no search at all, unlike the other two prefixes:
1892    /// `H5D__efl_read` joins the prefix to the stored name with
1893    /// `H5_combine_path` and opens exactly that one path (H5Defl.c:315-317).
1894    /// With no prefix in force the stored name is used as written, so a
1895    /// relative one resolves against the *process's current directory* and
1896    /// not against the directory holding the HDF5 file — measured under
1897    /// libhdf5 1.14.6 and 2.0.0: a raw data file next to the HDF5 file is
1898    /// not found, while the same name under the current directory is.
1899    ///
1900    /// It shares [`virtual_prefix`](Self::virtual_prefix)'s expansion rules,
1901    /// because both are built by `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX`
1902    /// shadows this property outright rather than merely preceding it
1903    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
1904    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
1905    /// (:1098-1102).
1906    ///
1907    /// # A second open must name the same one
1908    ///
1909    /// Where a mismatched [`virtual_prefix`](Self::virtual_prefix) is
1910    /// silently ignored by the second open, a mismatched external file prefix
1911    /// is an *error*: `H5D_open` compares the expanded prefix against the one
1912    /// the already-open dataset resolved under and refuses the open when they
1913    /// differ (H5Dint.c:1533-1545). Expanded, so two opens that differ only
1914    /// in a property the environment shadows still agree. Closing every
1915    /// handle releases the answer, and the next open sets its own.
1916    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
1917        self.efile_prefix = Some(prefix.into());
1918        self
1919    }
1920
1921    /// `H5Pget_virtual_printf_gap` (H5Pdapl.c:1243) — the value set, not the
1922    /// one the extent resolution ends up using; see
1923    /// [`VirtualView::FirstMissing`].
1924    pub fn printf_gap(&self) -> u64 {
1925        self.printf_gap
1926    }
1927
1928    /// `H5Pget_virtual_prefix` (H5Pdapl.c:1510) — the property as set, before
1929    /// `HDF5_VDS_PREFIX` gets to shadow it and before `${ORIGIN}` is
1930    /// expanded. `None` is `H5D_ACS_VDS_PREFIX_DEF`, a null prefix
1931    /// (H5Pdapl.c:72).
1932    pub fn virtual_prefix_value(&self) -> Option<&str> {
1933        self.virtual_prefix.as_deref()
1934    }
1935
1936    /// `H5Pget_efile_prefix` (H5Pdapl.c:1422) — the property as set, before
1937    /// `HDF5_EXTFILE_PREFIX` gets to shadow it and before `${ORIGIN}` is
1938    /// expanded. `None` is `H5D_ACS_EFILE_PREFIX_DEF`, a null prefix
1939    /// (H5Pdapl.c:90).
1940    pub fn efile_prefix_value(&self) -> Option<&str> {
1941        self.efile_prefix.as_deref()
1942    }
1943
1944    /// The printf gap `H5D__virtual_set_extent_unlim` actually scans with:
1945    /// the property under [`VirtualView::LastAvailable`], and 0 under
1946    /// [`VirtualView::FirstMissing`], because `H5D__virtual_init` only reads
1947    /// the property in the first case (H5Dvirtual.c:2182-2188).
1948    ///
1949    /// The single owner of that rule — the resolution never reads
1950    /// [`printf_gap`](Self::printf_gap) directly.
1951    pub(crate) fn effective_printf_gap(&self) -> u64 {
1952        match self.view {
1953            VirtualView::LastAvailable => self.printf_gap,
1954            VirtualView::FirstMissing => 0,
1955        }
1956    }
1957
1958    /// Reject what `H5Pset_virtual_printf_gap` rejects, at the open that uses
1959    /// the property.
1960    pub(crate) fn validate(&self) -> Result<()> {
1961        if self.printf_gap == u64::MAX {
1962            return Err(Hdf5Error::InvalidState(
1963                "virtual_printf_gap(u64::MAX) is libhdf5's HSIZE_UNDEF, which \
1964                 H5Pset_virtual_printf_gap refuses as \"not a valid printf gap size\""
1965                    .into(),
1966            ));
1967        }
1968        Ok(())
1969    }
1970}
1971
1972/// Elements a selection of `counts` holds, refusing a product that overflows
1973/// `usize` rather than wrapping it into a small allocation.
1974fn element_count(counts: &[u64]) -> Result<usize> {
1975    counts
1976        .iter()
1977        .try_fold(1usize, |acc, &c| {
1978            usize::try_from(c).ok().and_then(|c| acc.checked_mul(c))
1979        })
1980        .ok_or_else(|| {
1981            Hdf5Error::InvalidState(format!("selection {counts:?} has more elements than usize"))
1982        })
1983}
1984
1985impl H5Dataset {
1986    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
1987    pub(crate) fn new_reader(
1988        file_inner: SharedInner,
1989        name: String,
1990        shape: Vec<usize>,
1991        element_size: usize,
1992        open: Option<crate::io::reader::DatasetOpenToken>,
1993    ) -> Self {
1994        Self {
1995            file_inner,
1996            info: DatasetInfo::Reader {
1997                name,
1998                shape,
1999                element_size,
2000            },
2001            _open: open,
2002        }
2003    }
2004
2005    /// Create a writer-mode dataset handle for an already-created dataset
2006    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
2007    ///
2008    /// Reconstructs the same handle `new_dataset().create()` returns, so the
2009    /// reopened dataset supports attribute writes and chunk appends.
2010    ///
2011    /// `is_null` is always `false` here: reopening an existing NULL-dataspace
2012    /// dataset for further writes is not a case this constructor's caller
2013    /// distinguishes (a NULL dataset has nothing to append or chunk-write in
2014    /// the first place).
2015    pub(crate) fn new_writer(
2016        file_inner: SharedInner,
2017        index: usize,
2018        parts: crate::io::writer::DatasetHandleParts,
2019    ) -> Self {
2020        Self {
2021            file_inner,
2022            info: DatasetInfo::Writer {
2023                index,
2024                shape: parts.shape,
2025                element_size: parts.element_size,
2026                chunk_index: parts.chunk_index,
2027                is_null: false,
2028            },
2029            _open: parts.open,
2030        }
2031    }
2032
2033    /// Return the dataset dimensions.
2034    pub fn shape(&self) -> Vec<usize> {
2035        match &self.info {
2036            DatasetInfo::Writer { shape, .. } => shape.clone(),
2037            DatasetInfo::Reader { shape, .. } => shape.clone(),
2038        }
2039    }
2040
2041    /// Return the number of dimensions (rank) of the dataset.
2042    pub fn ndims(&self) -> usize {
2043        match &self.info {
2044            DatasetInfo::Writer { shape, .. } => shape.len(),
2045            DatasetInfo::Reader { shape, .. } => shape.len(),
2046        }
2047    }
2048
2049    /// Return the total number of elements in the dataset.
2050    ///
2051    /// 0 for a NULL dataspace ([`is_null`](Self::is_null)) — unlike a scalar,
2052    /// whose `shape()` is the same empty `Vec` but which holds exactly one
2053    /// element, so `shape().iter().product()` cannot be used here.
2054    pub fn total_elements(&self) -> usize {
2055        if self.is_null() {
2056            return 0;
2057        }
2058        match &self.info {
2059            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
2060            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
2061        }
2062    }
2063
2064    /// Return the size of one element in bytes.
2065    pub fn element_size(&self) -> usize {
2066        match &self.info {
2067            DatasetInfo::Writer { element_size, .. } => *element_size,
2068            DatasetInfo::Reader { element_size, .. } => *element_size,
2069        }
2070    }
2071
2072    /// Return whether this dataset has the NULL dataspace: no elements at
2073    /// all, distinct from a scalar dataset (rank 0, exactly one element) —
2074    /// both report the same empty [`shape`](Self::shape). See
2075    /// [`DatasetBuilder::null`].
2076    pub fn is_null(&self) -> bool {
2077        match &self.info {
2078            DatasetInfo::Writer { is_null, .. } => *is_null,
2079            DatasetInfo::Reader { name, .. } => {
2080                let mut inner = borrow_inner_mut(&self.file_inner);
2081                match &mut *inner {
2082                    H5FileInner::Reader(reader) => reader
2083                        .dataset_info(name)
2084                        .map(|info| info.dataspace.is_null())
2085                        .unwrap_or(false),
2086                    _ => false,
2087                }
2088            }
2089        }
2090    }
2091
2092    /// Return the element datatype as parsed from the file (read mode only).
2093    ///
2094    /// Unlike [`element_size`](Self::element_size), which reports only the
2095    /// byte width, this exposes the full datatype: its class (integer vs
2096    /// floating-point vs string vs compound …), signedness, byte order and
2097    /// bit precision. Callers that must reconstruct the exact stored type —
2098    /// for example to map it to a NumPy / Arrow dtype — should use this
2099    /// instead of inferring a type from the byte width, which cannot
2100    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
2101    /// bytes).
2102    ///
2103    /// # Errors
2104    ///
2105    /// Returns an error if the file is in write mode, or if the dataset can
2106    /// no longer be found in the reader's metadata.
2107    ///
2108    /// ```no_run
2109    /// # use rust_hdf5::{H5File, DatatypeMessage};
2110    /// let file = H5File::open("data.h5").unwrap();
2111    /// let ds = file.dataset("image").unwrap();
2112    /// match ds.datatype().unwrap() {
2113    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
2114    ///         println!("integer: {} bytes, signed={}", size, signed);
2115    ///     }
2116    ///     DatatypeMessage::FloatingPoint { size, .. } => {
2117    ///         println!("float: {} bytes", size);
2118    ///     }
2119    ///     other => println!("other type: {other}"),
2120    /// }
2121    /// ```
2122    pub fn datatype(&self) -> Result<DatatypeMessage> {
2123        match &self.info {
2124            DatasetInfo::Reader { name, .. } => {
2125                let mut inner = borrow_inner_mut(&self.file_inner);
2126                match &mut *inner {
2127                    H5FileInner::Reader(reader) => reader
2128                        .dataset_info(name)
2129                        .map(|info| info.datatype.clone())
2130                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2131                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2132                }
2133            }
2134            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2135                "datatype() is only available in read mode".into(),
2136            )),
2137        }
2138    }
2139
2140    /// Return the chunk dimensions, if this is a chunked dataset.
2141    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
2142        match &self.info {
2143            DatasetInfo::Reader { name, .. } => {
2144                let mut inner = borrow_inner_mut(&self.file_inner);
2145                if let H5FileInner::Reader(reader) = &mut *inner {
2146                    if let Some(info) = reader.dataset_info(name) {
2147                        use crate::format::messages::data_layout::DataLayoutMessage;
2148                        let chunk_dims = match &info.layout {
2149                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
2150                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
2151                            _ => None,
2152                        };
2153                        if let Some(chunk_dims) = chunk_dims {
2154                            // Strip trailing element-size dimension
2155                            return Some(
2156                                chunk_dims[..chunk_dims.len() - 1]
2157                                    .iter()
2158                                    .map(|&d| d as usize)
2159                                    .collect(),
2160                            );
2161                        }
2162                    }
2163                }
2164                None
2165            }
2166            DatasetInfo::Writer { .. } => None,
2167        }
2168    }
2169
2170    /// Return whether this is a chunked dataset.
2171    pub fn is_chunked(&self) -> bool {
2172        match &self.info {
2173            DatasetInfo::Writer { chunk_index, .. } => chunk_index.is_some(),
2174            DatasetInfo::Reader { name, .. } => {
2175                let mut inner = borrow_inner_mut(&self.file_inner);
2176                match &mut *inner {
2177                    H5FileInner::Reader(reader) => {
2178                        if let Some(info) = reader.dataset_info(name) {
2179                            use crate::format::messages::data_layout::DataLayoutMessage;
2180                            matches!(
2181                                info.layout,
2182                                DataLayoutMessage::ChunkedV4 { .. }
2183                                    | DataLayoutMessage::ChunkedV3 { .. }
2184                            )
2185                        } else {
2186                            false
2187                        }
2188                    }
2189                    _ => false,
2190                }
2191            }
2192        }
2193    }
2194
2195    /// Return the dataset's storage layout class (read mode only).
2196    ///
2197    /// # Errors
2198    ///
2199    /// Returns an error if the file is in write mode, or if the dataset can
2200    /// no longer be found in the reader's metadata.
2201    pub fn storage_layout(&self) -> Result<StorageLayout> {
2202        match &self.info {
2203            DatasetInfo::Reader { name, .. } => {
2204                let mut inner = borrow_inner_mut(&self.file_inner);
2205                match &mut *inner {
2206                    H5FileInner::Reader(reader) => {
2207                        use crate::format::messages::data_layout::DataLayoutMessage;
2208                        reader
2209                            .dataset_info(name)
2210                            .map(|info| match &info.layout {
2211                                DataLayoutMessage::Compact { .. } => StorageLayout::Compact,
2212                                DataLayoutMessage::Contiguous { .. } => StorageLayout::Contiguous,
2213                                DataLayoutMessage::ChunkedV3 { .. }
2214                                | DataLayoutMessage::ChunkedV4 { .. } => StorageLayout::Chunked,
2215                                DataLayoutMessage::Virtual { .. } => StorageLayout::Virtual,
2216                            })
2217                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2218                    }
2219                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2220                }
2221            }
2222            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2223                "storage_layout() is only available in read mode".into(),
2224            )),
2225        }
2226    }
2227
2228    /// Return the chunk index structure this dataset's layout uses, or
2229    /// `None` for a dataset that is not chunked (read mode only).
2230    ///
2231    /// # Errors
2232    ///
2233    /// Returns an error if the file is in write mode, or if the dataset can
2234    /// no longer be found in the reader's metadata.
2235    pub fn chunk_index(&self) -> Result<Option<ChunkIndex>> {
2236        match &self.info {
2237            DatasetInfo::Reader { name, .. } => {
2238                let mut inner = borrow_inner_mut(&self.file_inner);
2239                match &mut *inner {
2240                    H5FileInner::Reader(reader) => {
2241                        use crate::format::messages::data_layout::{
2242                            ChunkIndexType, DataLayoutMessage,
2243                        };
2244                        reader
2245                            .dataset_info(name)
2246                            .map(|info| match &info.layout {
2247                                DataLayoutMessage::ChunkedV3 { .. } => Some(ChunkIndex::BtreeV1),
2248                                DataLayoutMessage::ChunkedV4 { index_type, .. } => {
2249                                    Some(match index_type {
2250                                        ChunkIndexType::SingleChunk => ChunkIndex::SingleChunk,
2251                                        ChunkIndexType::Implicit => ChunkIndex::Implicit,
2252                                        ChunkIndexType::FixedArray => ChunkIndex::FixedArray,
2253                                        ChunkIndexType::ExtensibleArray => {
2254                                            ChunkIndex::ExtensibleArray
2255                                        }
2256                                        ChunkIndexType::BTreeV2 => ChunkIndex::BtreeV2,
2257                                    })
2258                                }
2259                                _ => None,
2260                            })
2261                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2262                    }
2263                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2264                }
2265            }
2266            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2267                "chunk_index() is only available in read mode".into(),
2268            )),
2269        }
2270    }
2271
2272    /// Return this dataset's filter pipeline (read mode only), in
2273    /// application order. Empty when the dataset has no filter pipeline
2274    /// message at all — an unfiltered dataset, not an error.
2275    ///
2276    /// # Errors
2277    ///
2278    /// Returns an error if the file is in write mode, or if the dataset can
2279    /// no longer be found in the reader's metadata.
2280    pub fn filters(&self) -> Result<Vec<Filter>> {
2281        match &self.info {
2282            DatasetInfo::Reader { name, .. } => {
2283                let mut inner = borrow_inner_mut(&self.file_inner);
2284                match &mut *inner {
2285                    H5FileInner::Reader(reader) => reader
2286                        .dataset_info(name)
2287                        .map(|info| {
2288                            info.filter_pipeline
2289                                .as_ref()
2290                                .map(|fp| fp.filters.clone())
2291                                .unwrap_or_default()
2292                        })
2293                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2294                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2295                }
2296            }
2297            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2298                "filters() is only available in read mode".into(),
2299            )),
2300        }
2301    }
2302
2303    /// Return this dataset's fill-value state (read mode only).
2304    ///
2305    /// # Errors
2306    ///
2307    /// Returns an error if the file is in write mode, or if the dataset can
2308    /// no longer be found in the reader's metadata.
2309    pub fn fill_value(&self) -> Result<FillValue> {
2310        match &self.info {
2311            DatasetInfo::Reader { name, .. } => {
2312                let mut inner = borrow_inner_mut(&self.file_inner);
2313                match &mut *inner {
2314                    H5FileInner::Reader(reader) => reader
2315                        .dataset_info(name)
2316                        .map(|info| match info.fill_defined {
2317                            0 => FillValue::Undefined,
2318                            2 => {
2319                                FillValue::UserDefined(info.fill_value.clone().unwrap_or_default())
2320                            }
2321                            _ => FillValue::Default,
2322                        })
2323                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2324                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2325                }
2326            }
2327            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2328                "fill_value() is only available in read mode".into(),
2329            )),
2330        }
2331    }
2332
2333    /// Return when this dataset's fill value is written into allocated
2334    /// storage (read mode only) — `H5Pget_fill_time`.
2335    ///
2336    /// # Errors
2337    ///
2338    /// Returns an error if the file is in write mode, or if the dataset can
2339    /// no longer be found in the reader's metadata.
2340    pub fn fill_time(&self) -> Result<FillTime> {
2341        match &self.info {
2342            DatasetInfo::Reader { name, .. } => {
2343                let mut inner = borrow_inner_mut(&self.file_inner);
2344                match &mut *inner {
2345                    H5FileInner::Reader(reader) => reader
2346                        .dataset_info(name)
2347                        .map(|info| match info.fill_write_time {
2348                            0 => FillTime::Alloc,
2349                            1 => FillTime::Never,
2350                            _ => FillTime::IfSet,
2351                        })
2352                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2353                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2354                }
2355            }
2356            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2357                "fill_time() is only available in read mode".into(),
2358            )),
2359        }
2360    }
2361
2362    /// Return when this dataset's raw-data storage is allocated (read mode
2363    /// only) — `H5Pget_alloc_time`.
2364    ///
2365    /// # Errors
2366    ///
2367    /// Returns an error if the file is in write mode, or if the dataset can
2368    /// no longer be found in the reader's metadata.
2369    pub fn alloc_time(&self) -> Result<AllocTime> {
2370        match &self.info {
2371            DatasetInfo::Reader { name, .. } => {
2372                let mut inner = borrow_inner_mut(&self.file_inner);
2373                match &mut *inner {
2374                    H5FileInner::Reader(reader) => reader
2375                        .dataset_info(name)
2376                        .map(|info| match info.alloc_time {
2377                            1 => AllocTime::Early,
2378                            3 => AllocTime::Incr,
2379                            _ => AllocTime::Late,
2380                        })
2381                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2382                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2383                }
2384            }
2385            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2386                "alloc_time() is only available in read mode".into(),
2387            )),
2388        }
2389    }
2390
2391    /// Return this dataset's external raw-data file segments (read mode
2392    /// only), in the order the dataset's logical byte range concatenates
2393    /// them. Empty for a dataset whose data lives in this file.
2394    ///
2395    /// # Errors
2396    ///
2397    /// Returns an error if the file is in write mode, or if the dataset can
2398    /// no longer be found in the reader's metadata.
2399    pub fn external_files(&self) -> Result<Vec<ExternalFileSegment>> {
2400        match &self.info {
2401            DatasetInfo::Reader { name, .. } => {
2402                let mut inner = borrow_inner_mut(&self.file_inner);
2403                match &mut *inner {
2404                    H5FileInner::Reader(reader) => reader
2405                        .dataset_info(name)
2406                        .map(|info| info.external_files.clone())
2407                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2408                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2409                }
2410            }
2411            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2412                "external_files() is only available in read mode".into(),
2413            )),
2414        }
2415    }
2416
2417    /// Return this dataset's maximum dimension sizes (read mode only):
2418    /// `None` in a dimension marks that axis unlimited. A dataset with no
2419    /// maximum-dimensions message reports its current shape (max == current
2420    /// — the upstream convention for a fixed-extent dataset).
2421    ///
2422    /// # Errors
2423    ///
2424    /// Returns an error if the file is in write mode, or if the dataset can
2425    /// no longer be found in the reader's metadata.
2426    pub fn max_shape(&self) -> Result<Vec<Option<usize>>> {
2427        match &self.info {
2428            DatasetInfo::Reader { name, .. } => {
2429                let mut inner = borrow_inner_mut(&self.file_inner);
2430                match &mut *inner {
2431                    H5FileInner::Reader(reader) => reader
2432                        .dataset_info(name)
2433                        .map(|info| match &info.dataspace.max_dims {
2434                            Some(max_dims) => max_dims
2435                                .iter()
2436                                .map(|&d| (d != u64::MAX).then_some(d as usize))
2437                                .collect(),
2438                            None => info
2439                                .dataspace
2440                                .dims
2441                                .iter()
2442                                .map(|&d| Some(d as usize))
2443                                .collect(),
2444                        })
2445                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2446                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2447                }
2448            }
2449            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2450                "max_shape() is only available in read mode".into(),
2451            )),
2452        }
2453    }
2454
2455    /// Return this dataset's virtual-dataset source/virtual mappings (read
2456    /// mode only), in on-disk order. Empty for any dataset whose layout is
2457    /// not virtual, and for a virtual dataset that has no mappings yet.
2458    ///
2459    /// # Errors
2460    ///
2461    /// Returns an error if the file is in write mode, or if the dataset can
2462    /// no longer be found in the reader's metadata.
2463    pub fn virtual_mappings(&self) -> Result<Vec<VirtualMapping>> {
2464        match &self.info {
2465            DatasetInfo::Reader { name, .. } => {
2466                let mut inner = borrow_inner_mut(&self.file_inner);
2467                match &mut *inner {
2468                    H5FileInner::Reader(reader) => reader
2469                        .dataset_info(name)
2470                        .map(|info| {
2471                            info.virtual_mappings
2472                                .as_ref()
2473                                .map(|vml| vml.mappings.clone())
2474                                .unwrap_or_default()
2475                        })
2476                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2477                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2478                }
2479            }
2480            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2481                "virtual_mappings() is only available in read mode".into(),
2482            )),
2483        }
2484    }
2485
2486    /// Return the names of all attributes on this dataset (read mode only).
2487    pub fn attr_names(&self) -> Result<Vec<String>> {
2488        match &self.info {
2489            DatasetInfo::Reader { name, .. } => {
2490                let mut inner = borrow_inner_mut(&self.file_inner);
2491                match &mut *inner {
2492                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
2493                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2494                }
2495            }
2496            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2497                "attr_names not available in write mode".into(),
2498            )),
2499        }
2500    }
2501
2502    /// Why the attribute `attr_name` on this dataset cannot be read, or `None`
2503    /// when it can be.
2504    ///
2505    /// An attribute whose message this crate cannot decode is still listed by
2506    /// [`attr_names`](Self::attr_names) — the object header carries it — and
2507    /// this says what stands in the way. Opening it through
2508    /// [`attr`](Self::attr) fails with the same text.
2509    pub fn attr_unreadable_reason(&self, attr_name: &str) -> Result<Option<String>> {
2510        match &self.info {
2511            DatasetInfo::Reader { name, .. } => {
2512                let mut inner = borrow_inner_mut(&self.file_inner);
2513                match &mut *inner {
2514                    H5FileInner::Reader(reader) => Ok(reader
2515                        .dataset_attr_unreadable_reason(name, attr_name)
2516                        .map(str::to_string)),
2517                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2518                }
2519            }
2520            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2521                "attr_unreadable_reason not available in write mode".into(),
2522            )),
2523        }
2524    }
2525
2526    /// Why this dataset's attribute *set* cannot be listed, or `None` when it
2527    /// can be.
2528    ///
2529    /// The object-scope counterpart of
2530    /// [`attr_unreadable_reason`](Self::attr_unreadable_reason). A dense
2531    /// attribute set is indexed by name hash, so a heap or index that will not
2532    /// read yields no names to hang a per-attribute reason on;
2533    /// [`attr_names`](Self::attr_names) then returns the failure rather than a
2534    /// short list, and this reports it without an attribute name.
2535    pub fn attrs_unreadable_reason(&self) -> Result<Option<String>> {
2536        match &self.info {
2537            DatasetInfo::Reader { name, .. } => {
2538                let mut inner = borrow_inner_mut(&self.file_inner);
2539                match &mut *inner {
2540                    H5FileInner::Reader(reader) => Ok(reader
2541                        .dataset_attrs_unreadable_reason(name)
2542                        .map(str::to_string)),
2543                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2544                }
2545            }
2546            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2547                "attrs_unreadable_reason not available in write mode".into(),
2548            )),
2549        }
2550    }
2551
2552    /// This dataset's own compact-vs-dense attribute storage — the
2553    /// equivalent of `h5py.h5o.get_info(did.id).meta_size.attr.index_size`
2554    /// being nonzero (read mode only).
2555    pub fn attr_storage(&self) -> Result<AttributeStorage> {
2556        match &self.info {
2557            DatasetInfo::Reader { name, .. } => {
2558                let mut inner = borrow_inner_mut(&self.file_inner);
2559                match &mut *inner {
2560                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_storage(name)?),
2561                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2562                }
2563            }
2564            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2565                "attr_storage not available in write mode".into(),
2566            )),
2567        }
2568    }
2569
2570    /// This dataset's own object-header attribute count — the equivalent of
2571    /// `h5py.h5o.get_info(did.id).num_attrs` (read mode only).
2572    pub fn header_attr_count(&self) -> Result<u64> {
2573        match &self.info {
2574            DatasetInfo::Reader { name, .. } => {
2575                let mut inner = borrow_inner_mut(&self.file_inner);
2576                match &mut *inner {
2577                    H5FileInner::Reader(reader) => Ok(reader.dataset_header_attr_count(name)?),
2578                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2579                }
2580            }
2581            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2582                "header_attr_count not available in write mode".into(),
2583            )),
2584        }
2585    }
2586
2587    /// Open an attribute by name (read mode only).
2588    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
2589        match &self.info {
2590            DatasetInfo::Reader { name, .. } => {
2591                let mut inner = borrow_inner_mut(&self.file_inner);
2592                match &mut *inner {
2593                    H5FileInner::Reader(reader) => {
2594                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
2595                        Ok(crate::attribute::H5Attribute::new_reader(
2596                            clone_inner(&self.file_inner),
2597                            attr_msg,
2598                        ))
2599                    }
2600                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2601                }
2602            }
2603            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2604                "attr() not available in write mode".into(),
2605            )),
2606        }
2607    }
2608
2609    /// Start building a new attribute on this dataset.
2610    ///
2611    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
2612    /// and `.create("name")` to finalize.
2613    ///
2614    /// # Example
2615    ///
2616    /// ```no_run
2617    /// # use rust_hdf5::H5File;
2618    /// # use rust_hdf5::types::VarLenUnicode;
2619    /// let file = H5File::create("attr.h5").unwrap();
2620    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
2621    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
2622    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
2623    /// ```
2624    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
2625        let ds_index = match &self.info {
2626            DatasetInfo::Writer { index, .. } => *index,
2627            DatasetInfo::Reader { .. } => {
2628                // Reader mode: we'll return a builder that will error on create.
2629                // Using usize::MAX as sentinel.
2630                usize::MAX
2631            }
2632        };
2633        AttrBuilder::new(&self.file_inner, ds_index)
2634    }
2635
2636    /// Write a typed slice holding the dataset's whole image.
2637    ///
2638    /// The slice length must match the total number of elements declared by
2639    /// the dataset shape. The data is reinterpreted as raw bytes and written
2640    /// to the file: to the contiguous data block, or — for a chunked dataset —
2641    /// scattered across its chunk grid, through the filter pipeline if one is
2642    /// set. To write only part of a dataset, use
2643    /// [`write_slice`](Self::write_slice).
2644    ///
2645    /// # Errors
2646    ///
2647    /// Returns an error if:
2648    /// - The file is in read mode.
2649    /// - The data length does not match the declared shape.
2650    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
2651        match &self.info {
2652            DatasetInfo::Writer {
2653                index,
2654                shape,
2655                element_size,
2656                chunk_index,
2657                is_null,
2658            } => {
2659                if *is_null {
2660                    return Err(Hdf5Error::InvalidState(
2661                        "cannot write to a NULL dataspace dataset".into(),
2662                    ));
2663                }
2664                let total_elements: usize = shape.iter().product();
2665                if data.len() != total_elements {
2666                    return Err(Hdf5Error::InvalidState(format!(
2667                        "data length {} does not match dataset size {}",
2668                        data.len(),
2669                        total_elements,
2670                    )));
2671                }
2672
2673                // Verify element size matches
2674                if T::element_size() != *element_size {
2675                    return Err(Hdf5Error::TypeMismatch(format!(
2676                        "write type has element size {} but dataset expects {}",
2677                        T::element_size(),
2678                        element_size,
2679                    )));
2680                }
2681
2682                // Safety: T: Copy + 'static (numeric primitive) with well-defined
2683                // byte representation. The resulting slice borrows `data` and
2684                // lives only as long as this block.
2685                let byte_len = data.len() * T::element_size();
2686                let host =
2687                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
2688                let datatype = {
2689                    let inner = borrow_inner(&self.file_inner);
2690                    match &*inner {
2691                        H5FileInner::Writer(writer) => writer.dataset_datatype(*index),
2692                        _ => {
2693                            return Err(Hdf5Error::InvalidState(
2694                                "file is no longer in write mode".into(),
2695                            ))
2696                        }
2697                    }
2698                };
2699                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
2700
2701                if let Some(kind) = *chunk_index {
2702                    // A chunked dataset has no contiguous data block; scatter
2703                    // the full row-major image into its chunk grid and write
2704                    // each chunk through the dataset's filter pipeline.
2705                    return self.write_full_image_chunked(*index, kind, &stored, *element_size);
2706                }
2707
2708                let inner = borrow_inner(&self.file_inner);
2709                match &*inner {
2710                    H5FileInner::Writer(writer) => {
2711                        writer.write_dataset_raw(*index, &stored)?;
2712                        Ok(())
2713                    }
2714                    _ => Err(Hdf5Error::InvalidState(
2715                        "file is no longer in write mode".into(),
2716                    )),
2717                }
2718            }
2719            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2720                "cannot write to a dataset opened in read mode".into(),
2721            )),
2722        }
2723    }
2724
2725    /// Write the raw byte image of the whole dataset directly.
2726    ///
2727    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
2728    /// data block, or a chunk grid the image is scattered across.
2729    ///
2730    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
2731    /// `H5Type` carrier, so it works for element types that have no matching
2732    /// Rust primitive — in particular a runtime
2733    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
2734    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
2735    /// `product(shape) * element_size`, where `element_size` is taken from the
2736    /// dataset's on-disk datatype.
2737    ///
2738    /// ```no_run
2739    /// # use rust_hdf5::H5File;
2740    /// # use rust_hdf5::types::{CompoundType, H5Type};
2741    /// let file = H5File::create("c.h5").unwrap();
2742    /// let ct = CompoundType {
2743    ///     members: vec![
2744    ///         ("id".to_string(), i32::hdf5_type(), 0),
2745    ///         ("val".to_string(), f64::hdf5_type(), 4),
2746    ///     ],
2747    ///     total_size: 12,
2748    /// };
2749    /// let ds = file
2750    ///     .new_dataset::<u8>()
2751    ///     .datatype(ct.to_datatype())
2752    ///     .shape(&[2])
2753    ///     .create("records")
2754    ///     .unwrap();
2755    /// let mut bytes = Vec::new();
2756    /// bytes.extend_from_slice(&1i32.to_le_bytes());
2757    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
2758    /// bytes.extend_from_slice(&2i32.to_le_bytes());
2759    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
2760    /// ds.write_raw_bytes(&bytes).unwrap();
2761    /// ```
2762    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
2763        match &self.info {
2764            DatasetInfo::Writer {
2765                index,
2766                shape,
2767                element_size,
2768                chunk_index,
2769                is_null,
2770            } => {
2771                if *is_null {
2772                    return Err(Hdf5Error::InvalidState(
2773                        "cannot write to a NULL dataspace dataset".into(),
2774                    ));
2775                }
2776                let expected: usize = shape.iter().product::<usize>() * *element_size;
2777                if bytes.len() != expected {
2778                    return Err(Hdf5Error::InvalidState(format!(
2779                        "raw byte length {} does not match dataset size {} \
2780                         (product(shape) * element_size {})",
2781                        bytes.len(),
2782                        expected,
2783                        element_size,
2784                    )));
2785                }
2786                if let Some(kind) = *chunk_index {
2787                    // Scatter the full row-major image into the chunk grid
2788                    // (same path as write_raw, carrier-agnostic bytes).
2789                    return self.write_full_image_chunked(*index, kind, bytes, *element_size);
2790                }
2791                let inner = borrow_inner(&self.file_inner);
2792                match &*inner {
2793                    H5FileInner::Writer(writer) => {
2794                        writer.write_dataset_raw(*index, bytes)?;
2795                        Ok(())
2796                    }
2797                    _ => Err(Hdf5Error::InvalidState(
2798                        "file is no longer in write mode".into(),
2799                    )),
2800                }
2801            }
2802            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2803                "cannot write to a dataset opened in read mode".into(),
2804            )),
2805        }
2806    }
2807
2808    /// Scatter a full row-major dataset image into its chunk grid, writing
2809    /// every chunk through the dataset's filter pipeline.
2810    ///
2811    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
2812    /// — it is how [`write_raw`](Self::write_raw) and
2813    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
2814    /// (including the single auto-chunk created when a filter is set without
2815    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
2816    /// chunk footprint, exactly as libhdf5 stores them.
2817    fn write_full_image_chunked(
2818        &self,
2819        index: usize,
2820        kind: ChunkIndexKind,
2821        bytes: &[u8],
2822        element_size: usize,
2823    ) -> Result<()> {
2824        let inner = borrow_inner(&self.file_inner);
2825        let writer = match &*inner {
2826            H5FileInner::Writer(w) => w,
2827            _ => {
2828                return Err(Hdf5Error::InvalidState(
2829                    "file is no longer in write mode".into(),
2830                ))
2831            }
2832        };
2833        // Whole-operation guard: the flush, the grid snapshot and the chunk
2834        // writes below must not interleave with a concurrent same-dataset
2835        // operation.
2836        let cell = writer.ds(index);
2837        let _op = cell.op.lock();
2838        // A buffered append tail would flush over the image at close; hand
2839        // it to the chunks first, the image below overwrites everything.
2840        writer.flush_append_buffer(index)?;
2841        let chunk_dims = writer
2842            .dataset_chunk_dims(index)
2843            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
2844            .to_vec();
2845        let dims = writer.dataset_dims(index).to_vec();
2846        let rank = dims.len();
2847
2848        // Chunk grid: number of chunks along each dimension (row-major).
2849        let mut grid = vec![0u64; rank];
2850        for d in 0..rank {
2851            grid[d] = if chunk_dims[d] > 0 {
2852                dims[d].div_ceil(chunk_dims[d])
2853            } else {
2854                0
2855            };
2856        }
2857        let total_chunks: u64 = grid.iter().product();
2858
2859        // Decode the iteration counter into row-major coordinates over the
2860        // *current* image's chunk grid. This is only an odometer over the
2861        // chunks the image spans — the slot a chunk is recorded under comes
2862        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
2863        // extent decides.
2864        let coords_of = |linear: u64| -> Vec<u64> {
2865            let mut rem = linear;
2866            let mut coords = vec![0u64; rank];
2867            for d in (0..rank).rev() {
2868                coords[d] = rem % grid[d];
2869                rem /= grid[d];
2870            }
2871            coords
2872        };
2873
2874        // The batch entry points exist for one reason: to run the filter
2875        // pipeline over a window of chunks in parallel. An unfiltered dataset
2876        // has no pipeline to run, so it takes the plain per-chunk owner
2877        // whatever its index is, and only a filtered extensible or fixed array
2878        // — the two indexes with a batch entry point — takes the window below.
2879        let batched = writer.dataset_is_filtered(index)
2880            && matches!(
2881                kind,
2882                ChunkIndexKind::ExtensibleArray | ChunkIndexKind::FixedArray
2883            );
2884        if !batched {
2885            // One staging buffer for the whole image, reused chunk after
2886            // chunk: a chunk that already sits as one complete run of `bytes`
2887            // needs no staging at all and goes to the file straight out of the
2888            // caller's slice, so only an n-D interleave or a short edge pays
2889            // for a gather.
2890            let mut staging = Vec::new();
2891            for linear in 0..total_chunks {
2892                let coords = coords_of(linear);
2893                let chunk =
2894                    match Self::contiguous_chunk_span(&dims, &chunk_dims, &coords, element_size) {
2895                        Some(span) => &bytes[span],
2896                        None => {
2897                            Self::gather_chunk_into(
2898                                &mut staging,
2899                                bytes,
2900                                &dims,
2901                                &chunk_dims,
2902                                &coords,
2903                                element_size,
2904                            );
2905                            &staging[..]
2906                        }
2907                    };
2908                writer.write_chunk_at_coords(index, &coords, chunk)?;
2909            }
2910        } else {
2911            // Hand the pipeline a window of chunks so it compresses them in
2912            // parallel (with the `parallel` feature). A fixed-size window
2913            // bounds peak memory instead of materializing every chunk at once;
2914            // 256 keeps every rayon worker fed while capping the transient
2915            // buffers to window * chunk bytes. The compressors read the window
2916            // concurrently, so a gathered chunk here cannot share one reused
2917            // buffer the way the sequential path above does — but a chunk that
2918            // is already a complete run of `bytes` is borrowed, not copied.
2919            // The two indexes differ only in how a chunk is addressed: EA by
2920            // its linear grid index, FA by grid coordinates.
2921            const BATCH_WINDOW: u64 = 256;
2922            let mut start = 0u64;
2923            while start < total_chunks {
2924                let end = (start + BATCH_WINDOW).min(total_chunks);
2925                let items: Vec<(Vec<u64>, Cow<'_, [u8]>)> = (start..end)
2926                    .map(|counter| {
2927                        let coords = coords_of(counter);
2928                        let data = match Self::contiguous_chunk_span(
2929                            &dims,
2930                            &chunk_dims,
2931                            &coords,
2932                            element_size,
2933                        ) {
2934                            Some(span) => Cow::Borrowed(&bytes[span]),
2935                            None => {
2936                                let mut buf = Vec::new();
2937                                Self::gather_chunk_into(
2938                                    &mut buf,
2939                                    bytes,
2940                                    &dims,
2941                                    &chunk_dims,
2942                                    &coords,
2943                                    element_size,
2944                                );
2945                                Cow::Owned(buf)
2946                            }
2947                        };
2948                        (coords, data)
2949                    })
2950                    .collect();
2951                if kind == ChunkIndexKind::FixedArray {
2952                    let pairs: Vec<(&[u64], &[u8])> = items
2953                        .iter()
2954                        .map(|(c, d)| (c.as_slice(), d.as_ref()))
2955                        .collect();
2956                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
2957                } else {
2958                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
2959                    for (c, d) in &items {
2960                        pairs.push((writer.chunk_slot(index, c)?, d.as_ref()));
2961                    }
2962                    writer.write_chunks_batch_inner(index, &pairs)?;
2963                }
2964                start = end;
2965            }
2966        }
2967        Ok(())
2968    }
2969
2970    /// The byte range one chunk occupies in a row-major full-dataset image,
2971    /// for a chunk that needs no gather at all: its elements are one
2972    /// contiguous run of `source` *and* they fill the chunk shape exactly, so
2973    /// the bytes that go to the file are already sitting in the caller's
2974    /// buffer.
2975    ///
2976    /// Both halves hold when every dimension after the first spans the whole
2977    /// dataset (`chunk_dims[d] == dims[d]`, leaving nothing interleaved and no
2978    /// padding along those axes) and the chunk does not hang off the far edge
2979    /// of the first — which is every full chunk of a 1-D dataset. `None` means
2980    /// the chunk has to be gathered.
2981    fn contiguous_chunk_span(
2982        dims: &[u64],
2983        chunk_dims: &[u64],
2984        coords: &[u64],
2985        element_size: usize,
2986    ) -> Option<std::ops::Range<usize>> {
2987        let rank = dims.len();
2988        if rank == 0 || chunk_dims[1..] != dims[1..] {
2989            return None;
2990        }
2991        if (coords[0] + 1) * chunk_dims[0] > dims[0] {
2992            return None;
2993        }
2994        let plane: u64 = dims[1..].iter().product::<u64>() * element_size as u64;
2995        let start = usize::try_from(coords[0] * chunk_dims[0] * plane).ok()?;
2996        let len = usize::try_from(chunk_dims[0] * plane).ok()?;
2997        Some(start..start.checked_add(len)?)
2998    }
2999
3000    /// Gather one chunk's bytes from a row-major full-dataset image into
3001    /// `out`, replacing whatever it held.
3002    ///
3003    /// `coords` are the chunk's grid coordinates. `out` is left exactly
3004    /// `product(chunk_dims) * element_size` bytes long, holding the chunk's
3005    /// elements and zero where the chunk extends past the dataset edge — so a
3006    /// caller may hand the same buffer to one chunk after another.
3007    fn gather_chunk_into(
3008        out: &mut Vec<u8>,
3009        source: &[u8],
3010        dims: &[u64],
3011        chunk_dims: &[u64],
3012        coords: &[u64],
3013        element_size: usize,
3014    ) {
3015        let rank = dims.len();
3016        let chunk_elems: u64 = chunk_dims.iter().product();
3017        let chunk_bytes = chunk_elems as usize * element_size;
3018        if rank == 0 {
3019            // Scalar dataset: a single element, no chunking dimension.
3020            out.clear();
3021            out.resize(chunk_bytes, 0);
3022            if source.len() >= element_size {
3023                out[..element_size].copy_from_slice(&source[..element_size]);
3024            }
3025            return;
3026        }
3027
3028        // Actual extent of this chunk along each dimension (edge chunks are
3029        // smaller than the nominal chunk shape).
3030        let mut extent = vec![0u64; rank];
3031        for d in 0..rank {
3032            let start = coords[d] * chunk_dims[d];
3033            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
3034            extent[d] = end.saturating_sub(start);
3035        }
3036        // Size the buffer, then zero it only when this chunk leaves part of
3037        // its shape uncovered: a full chunk has every byte overwritten below,
3038        // while an edge chunk's padding must read as zero even though a
3039        // reused buffer still holds the previous chunk's bytes.
3040        if out.len() != chunk_bytes {
3041            out.clear();
3042            out.resize(chunk_bytes, 0);
3043        } else if extent != chunk_dims {
3044            out.fill(0);
3045        }
3046        if extent.contains(&0) {
3047            return; // nothing of the dataset falls in this chunk
3048        }
3049
3050        // Row-major strides (in elements) for the source (over `dims`) and the
3051        // destination chunk buffer (over `chunk_dims`).
3052        let mut src_stride = vec![1u64; rank];
3053        let mut dst_stride = vec![1u64; rank];
3054        for d in (0..rank - 1).rev() {
3055            src_stride[d] = src_stride[d + 1] * dims[d + 1];
3056            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
3057        }
3058
3059        // Copy one contiguous run along the last axis per outer multi-index.
3060        let last = rank - 1;
3061        let run = extent[last] as usize * element_size;
3062        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
3063        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
3064        for _ in 0..outer {
3065            let mut src_off = 0u64;
3066            let mut dst_off = 0u64;
3067            for d in 0..rank {
3068                let global = coords[d] * chunk_dims[d] + idx[d];
3069                src_off += global * src_stride[d];
3070                dst_off += idx[d] * dst_stride[d];
3071            }
3072            let s = src_off as usize * element_size;
3073            let dpos = dst_off as usize * element_size;
3074            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
3075
3076            // Advance the multi-index over axes [0..last); the last axis is the
3077            // contiguous run handled above.
3078            let mut d = last;
3079            while d > 0 {
3080                d -= 1;
3081                idx[d] += 1;
3082                if idx[d] < extent[d] {
3083                    break;
3084                }
3085                idx[d] = 0;
3086            }
3087        }
3088    }
3089
3090    /// Write a single chunk to a chunked dataset.
3091    ///
3092    /// `chunk_idx` is the linear chunk index (typically the frame number for
3093    /// streaming datasets). `data` is the raw byte data for one chunk.
3094    ///
3095    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
3096    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
3097    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
3098        match &self.info {
3099            DatasetInfo::Writer {
3100                index, chunk_index, ..
3101            } => {
3102                let Some(kind) = *chunk_index else {
3103                    return Err(Hdf5Error::InvalidState(
3104                        "write_chunk is only for chunked datasets".into(),
3105                    ));
3106                };
3107                if kind == ChunkIndexKind::BtreeV2 {
3108                    return Err(Hdf5Error::InvalidState(
3109                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
3110                         with the chunk's grid coordinates"
3111                            .into(),
3112                    ));
3113                }
3114
3115                let inner = borrow_inner(&self.file_inner);
3116                match &*inner {
3117                    H5FileInner::Writer(writer) => {
3118                        // One op: the slot decode and the write see the same
3119                        // extents.
3120                        let cell = writer.ds(*index);
3121                        let _op = cell.op.lock();
3122                        match kind {
3123                            // All four address a chunk by its grid
3124                            // coordinates, so the linear slot is decoded back
3125                            // into them.
3126                            ChunkIndexKind::FixedArray
3127                            | ChunkIndexKind::Implicit
3128                            | ChunkIndexKind::SingleChunk
3129                            | ChunkIndexKind::BtreeV1 => {
3130                                let coords =
3131                                    writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3132                                writer.write_chunk_at_coords(*index, &coords, data)?;
3133                            }
3134                            _ => writer.write_chunk_inner(*index, chunk_idx as u64, data)?,
3135                        }
3136                        Ok(())
3137                    }
3138                    _ => Err(Hdf5Error::InvalidState(
3139                        "file is no longer in write mode".into(),
3140                    )),
3141                }
3142            }
3143            DatasetInfo::Reader { .. } => {
3144                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3145            }
3146        }
3147    }
3148
3149    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
3150    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
3151    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
3152    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
3153    ///
3154    /// `chunk_idx` is the linear chunk index (the frame number for streaming
3155    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
3156    /// the already-filtered bytes of one chunk — its length is the *stored*
3157    /// (compressed) size, not the uncompressed chunk size.
3158    ///
3159    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
3160    /// dataset's pipeline was **not** applied to this chunk and must be skipped
3161    /// on read. Pass 0 when the full pipeline was already applied upstream (the
3162    /// common case: a codec plugin handed you compressed frames).
3163    ///
3164    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3165    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
3166    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
3167    /// against, so address its chunks with
3168    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
3169    ///
3170    /// # Reading back
3171    ///
3172    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
3173    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
3174    /// the reader skipping exactly the filters the mask marks as not applied.
3175    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
3176        match &self.info {
3177            DatasetInfo::Writer {
3178                index, chunk_index, ..
3179            } => {
3180                let Some(kind) = *chunk_index else {
3181                    return Err(Hdf5Error::InvalidState(
3182                        "write_chunk_raw is only for chunked datasets".into(),
3183                    ));
3184                };
3185                if kind == ChunkIndexKind::BtreeV2 {
3186                    return Err(Hdf5Error::InvalidState(
3187                        "this dataset uses a v2 B-tree chunk index; use \
3188                         write_chunk_raw_at with the chunk's grid coordinates"
3189                            .into(),
3190                    ));
3191                }
3192                if kind == ChunkIndexKind::Implicit {
3193                    return Err(Hdf5Error::InvalidState(
3194                        "this dataset uses the implicit chunk index, which stores \
3195                         every chunk at its full unfiltered size and has nowhere to \
3196                         record a stored size or a filter mask"
3197                            .into(),
3198                    ));
3199                }
3200
3201                let inner = borrow_inner(&self.file_inner);
3202                match &*inner {
3203                    H5FileInner::Writer(writer) => {
3204                        // One op: the slot decode and the write see the same
3205                        // extents.
3206                        let cell = writer.ds(*index);
3207                        let _op = cell.op.lock();
3208                        if kind == ChunkIndexKind::FixedArray {
3209                            // Fixed-array dataset: decode the index-grid slot
3210                            // into row-major grid coordinates.
3211                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3212                            writer.write_compressed_chunk_fixed_array_inner(
3213                                *index,
3214                                &coords,
3215                                data,
3216                                filter_mask,
3217                            )?;
3218                        } else if kind == ChunkIndexKind::BtreeV1 {
3219                            // Same for the classic index, whose key carries a
3220                            // stored size and a filter mask of its own.
3221                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3222                            writer.write_compressed_chunk_btree_v1_inner(
3223                                *index,
3224                                &coords,
3225                                data,
3226                                filter_mask,
3227                            )?;
3228                        } else if kind == ChunkIndexKind::SingleChunk {
3229                            // Same again for the single-chunk index, whose
3230                            // layout message carries the stored size and mask
3231                            // inline.
3232                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3233                            writer.write_compressed_chunk_single_chunk_inner(
3234                                *index,
3235                                &coords,
3236                                data,
3237                                filter_mask,
3238                            )?;
3239                        } else {
3240                            writer.write_compressed_chunk_inner(
3241                                *index,
3242                                chunk_idx as u64,
3243                                data,
3244                                filter_mask,
3245                            )?;
3246                        }
3247                        Ok(())
3248                    }
3249                    _ => Err(Hdf5Error::InvalidState(
3250                        "file is no longer in write mode".into(),
3251                    )),
3252                }
3253            }
3254            DatasetInfo::Reader { .. } => {
3255                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3256            }
3257        }
3258    }
3259
3260    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
3261    /// chunk-grid coordinates (one per dimension).
3262    ///
3263    /// This is the entry point for datasets with two or more unlimited
3264    /// dimensions. The dataset's logical dimensions are extended to cover
3265    /// the written chunk. `data` is the raw bytes of one full chunk.
3266    ///
3267    /// ```no_run
3268    /// # use rust_hdf5::H5File;
3269    /// let file = H5File::create("bt2.h5").unwrap();
3270    /// let ds = file.new_dataset::<i32>()
3271    ///     .shape(&[0, 0])
3272    ///     .chunk(&[2, 2])
3273    ///     .max_shape(&[None, None])
3274    ///     .create("grid")
3275    ///     .unwrap();
3276    /// let chunk = [0i32, 1, 2, 3];
3277    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
3278    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3279    /// ```
3280    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
3281        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
3282    }
3283
3284    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
3285    /// addressed by its chunk-grid coordinates.
3286    ///
3287    /// The coordinate-addressed twin of
3288    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
3289    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
3290    /// there is no fixed chunk grid for a linear index to mean anything against.
3291    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
3292    /// to cover the written chunk.
3293    ///
3294    /// `data` is the already-filtered bytes of one chunk — its length is the
3295    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
3296    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
3297    /// full pipeline already ran upstream.
3298    ///
3299    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3300    /// has no slot to record a stored size or mask.
3301    pub fn write_chunk_raw_at(
3302        &self,
3303        chunk_coords: &[usize],
3304        data: &[u8],
3305        filter_mask: u32,
3306    ) -> Result<()> {
3307        self.write_chunk_at_inner(
3308            chunk_coords,
3309            ChunkBytes::Prefiltered { data, filter_mask },
3310            "write_chunk_raw_at",
3311        )
3312    }
3313
3314    /// The single owner of coordinate-addressed chunk writes: validates the
3315    /// coordinates, grows the dataspace to cover them, and routes the bytes to
3316    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
3317    /// here or already ran upstream is carried by `bytes`, not by a second copy
3318    /// of this dispatch.
3319    fn write_chunk_at_inner(
3320        &self,
3321        chunk_coords: &[usize],
3322        bytes: ChunkBytes<'_>,
3323        what: &str,
3324    ) -> Result<()> {
3325        match &self.info {
3326            DatasetInfo::Writer {
3327                index, chunk_index, ..
3328            } => {
3329                let Some(kind) = *chunk_index else {
3330                    return Err(Hdf5Error::InvalidState(format!(
3331                        "{what} is only for chunked datasets"
3332                    )));
3333                };
3334                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3335                let inner = borrow_inner(&self.file_inner);
3336                let writer = match &*inner {
3337                    H5FileInner::Writer(w) => w,
3338                    _ => {
3339                        return Err(Hdf5Error::InvalidState(
3340                            "file is no longer in write mode".into(),
3341                        ))
3342                    }
3343                };
3344                // Whole-operation guard: the dims snapshot, the chunk write
3345                // and the extend below must not interleave with a concurrent
3346                // same-dataset operation.
3347                let cell = writer.ds(*index);
3348                let _op = cell.op.lock();
3349                let chunk_dims = writer
3350                    .dataset_chunk_dims(*index)
3351                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3352                    .to_vec();
3353                let dims = writer.dataset_dims(*index).to_vec();
3354                if coords.len() != dims.len() {
3355                    return Err(Hdf5Error::InvalidState(format!(
3356                        "chunk_coords has {} entries but the dataset has {} dimensions",
3357                        coords.len(),
3358                        dims.len()
3359                    )));
3360                }
3361                if chunk_dims.len() != dims.len() {
3362                    return Err(Hdf5Error::InvalidState(format!(
3363                        "dataset chunk shape has {} dimensions but the dataspace has {}",
3364                        chunk_dims.len(),
3365                        dims.len()
3366                    )));
3367                }
3368
3369                if kind == ChunkIndexKind::FixedArray {
3370                    // Fixed-array (fixed-shape) dataset: no dimension growth.
3371                    match bytes {
3372                        ChunkBytes::Unfiltered(data) => {
3373                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
3374                        }
3375                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3376                            .write_compressed_chunk_fixed_array_inner(
3377                                *index,
3378                                &coords,
3379                                data,
3380                                filter_mask,
3381                            )?,
3382                    }
3383                    return Ok(());
3384                }
3385
3386                if kind == ChunkIndexKind::Implicit {
3387                    // Implicit index: fixed shape, so no dimension growth
3388                    // either, and no slot to record a stored size in.
3389                    match bytes {
3390                        ChunkBytes::Unfiltered(data) => {
3391                            writer.write_chunk_implicit_inner(*index, &coords, data)?
3392                        }
3393                        ChunkBytes::Prefiltered { .. } => {
3394                            return Err(Hdf5Error::InvalidState(
3395                                "this dataset uses the implicit chunk index, which stores \
3396                                 every chunk at its full unfiltered size and has nowhere \
3397                                 to record a stored size or a filter mask"
3398                                    .into(),
3399                            ))
3400                        }
3401                    }
3402                    return Ok(());
3403                }
3404
3405                if kind == ChunkIndexKind::SingleChunk {
3406                    // Single-chunk index: fixed shape covered by exactly one
3407                    // chunk, so no dimension growth either.
3408                    match bytes {
3409                        ChunkBytes::Unfiltered(data) => {
3410                            writer.write_chunk_single_chunk_inner(*index, &coords, data)?
3411                        }
3412                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3413                            .write_compressed_chunk_single_chunk_inner(
3414                                *index,
3415                                &coords,
3416                                data,
3417                                filter_mask,
3418                            )?,
3419                    }
3420                    return Ok(());
3421                }
3422
3423                // The remaining indexes (v2 B-tree, v1 B-tree, extensible
3424                // array) can all grow: validate the coordinates and compute
3425                // the grown dimensions up-front, before any chunk is
3426                // written, so an overflowing coordinate cannot leave an
3427                // orphaned chunk in the file.
3428                //
3429                // The last chunk of a dimension usually hangs past the extent
3430                // — a length of 10 in chunks of 4 ends at 12 — so the growth
3431                // is capped at the declared maximum, which is what the chunk
3432                // still covers. Without the cap a legal edge chunk would be
3433                // written and then rejected by the extend below.
3434                let max_dims = writer.dataset_max_dims(*index);
3435                let mut new_dims = dims.clone();
3436                for d in 0..dims.len() {
3437                    let needed = coords[d]
3438                        .checked_add(1)
3439                        .and_then(|c| c.checked_mul(chunk_dims[d]))
3440                        .ok_or_else(|| {
3441                            Hdf5Error::InvalidState(format!(
3442                                "chunk coordinate {} in dimension {} is too large",
3443                                coords[d], d
3444                            ))
3445                        })?;
3446                    let needed = needed.min(max_dims[d]);
3447                    if needed > new_dims[d] {
3448                        new_dims[d] = needed;
3449                    }
3450                }
3451
3452                if kind == ChunkIndexKind::BtreeV2 {
3453                    match bytes {
3454                        ChunkBytes::Unfiltered(data) => {
3455                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
3456                        }
3457                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3458                            .write_compressed_chunk_btree_v2_inner(
3459                                *index,
3460                                &coords,
3461                                data,
3462                                filter_mask,
3463                            )?,
3464                    }
3465                } else if kind == ChunkIndexKind::BtreeV1 {
3466                    // The classic index takes any shape, fixed or unlimited,
3467                    // so it grows the dataspace with the chunk the way the v2
3468                    // B-tree does — bounded below by the maximum extent.
3469                    match bytes {
3470                        ChunkBytes::Unfiltered(data) => {
3471                            writer.write_chunk_btree_v1_inner(*index, &coords, data)?
3472                        }
3473                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3474                            .write_compressed_chunk_btree_v1_inner(
3475                                *index,
3476                                &coords,
3477                                data,
3478                                filter_mask,
3479                            )?,
3480                    }
3481                } else {
3482                    // Extensible array: the chunk's index-grid slot (row-major
3483                    // against the maximum extent).
3484                    let linear = writer.chunk_slot(*index, &coords)?;
3485                    match bytes {
3486                        ChunkBytes::Unfiltered(data) => {
3487                            writer.write_chunk_inner(*index, linear, data)?
3488                        }
3489                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3490                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
3491                    }
3492                }
3493
3494                if new_dims != dims {
3495                    writer.extend_dataset_inner(*index, &new_dims)?;
3496                }
3497                Ok(())
3498            }
3499            DatasetInfo::Reader { .. } => {
3500                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3501            }
3502        }
3503    }
3504
3505    /// Write multiple chunks in a batch, optionally compressing in parallel.
3506    ///
3507    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
3508    /// pipeline is configured and the `parallel` feature is enabled, all
3509    /// chunks are compressed concurrently via rayon.
3510    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
3511        match &self.info {
3512            DatasetInfo::Writer {
3513                index, chunk_index, ..
3514            } => {
3515                if chunk_index.is_none() {
3516                    return Err(Hdf5Error::InvalidState(
3517                        "write_chunks_batch is only for chunked datasets".into(),
3518                    ));
3519                }
3520                let pairs: Vec<(u64, &[u8])> = chunks
3521                    .iter()
3522                    .map(|(idx, data)| (*idx as u64, *data))
3523                    .collect();
3524                let inner = borrow_inner(&self.file_inner);
3525                match &*inner {
3526                    H5FileInner::Writer(writer) => {
3527                        writer.write_chunks_batch(*index, &pairs)?;
3528                        Ok(())
3529                    }
3530                    _ => Err(Hdf5Error::InvalidState(
3531                        "file is no longer in write mode".into(),
3532                    )),
3533                }
3534            }
3535            DatasetInfo::Reader { .. } => {
3536                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3537            }
3538        }
3539    }
3540
3541    /// Append data along the first dimension of a chunked dataset.
3542    ///
3543    /// `data` must contain a whole number of "frames" — slices along
3544    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
3545    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
3546    /// multiple of `H * W`.
3547    ///
3548    /// This method writes the necessary chunks and extends the dataset
3549    /// shape automatically.
3550    ///
3551    /// ```no_run
3552    /// # use rust_hdf5::H5File;
3553    /// let file = H5File::create("append.h5").unwrap();
3554    /// let ds = file.new_dataset::<f64>()
3555    ///     .shape(&[0, 3])
3556    ///     .chunk(&[1, 3])
3557    ///     .max_shape(&[None, Some(3)])
3558    ///     .create("data")
3559    ///     .unwrap();
3560    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
3561    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
3562    /// ```
3563    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
3564        match &self.info {
3565            DatasetInfo::Writer {
3566                index,
3567                element_size,
3568                chunk_index,
3569                ..
3570            } => {
3571                if chunk_index.is_none() {
3572                    return Err(Hdf5Error::InvalidState(
3573                        "append is only for chunked datasets".into(),
3574                    ));
3575                }
3576                if T::element_size() != *element_size {
3577                    return Err(Hdf5Error::TypeMismatch(format!(
3578                        "append type has element size {} but dataset expects {}",
3579                        T::element_size(),
3580                        element_size,
3581                    )));
3582                }
3583
3584                let ds_index = *index;
3585                let es = *element_size;
3586
3587                let inner = borrow_inner(&self.file_inner);
3588                let writer = match &*inner {
3589                    H5FileInner::Writer(w) => w,
3590                    _ => {
3591                        return Err(Hdf5Error::InvalidState(
3592                            "file is no longer in write mode".into(),
3593                        ))
3594                    }
3595                };
3596
3597                // Whole-operation guard: the buffer take, the frame writes,
3598                // the re-buffer and the extend below are separate slot
3599                // acquisitions that a concurrent same-dataset append must not
3600                // interleave with.
3601                let cell = writer.ds(ds_index);
3602                let _op = cell.op.lock();
3603
3604                let chunk_dims = writer
3605                    .dataset_chunk_dims(ds_index)
3606                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3607                    .to_vec();
3608                let dims = writer.dataset_dims(ds_index).to_vec();
3609
3610                // Frame size = product of dims[1..]
3611                let frame_elems: usize = if dims.len() > 1 {
3612                    dims[1..].iter().map(|&d| d as usize).product()
3613                } else {
3614                    1
3615                };
3616
3617                if frame_elems == 0 {
3618                    return Err(Hdf5Error::InvalidState(
3619                        "cannot append to dataset with zero-size trailing dimensions".into(),
3620                    ));
3621                }
3622
3623                if !data.len().is_multiple_of(frame_elems) {
3624                    return Err(Hdf5Error::InvalidState(format!(
3625                        "data length {} is not a multiple of frame size {}",
3626                        data.len(),
3627                        frame_elems,
3628                    )));
3629                }
3630
3631                let n_new_frames = data.len() / frame_elems;
3632                let current_dim0 = dims[0] as usize;
3633
3634                // Chunk size along first dimension
3635                let chunk_dim0 = chunk_dims[0] as usize;
3636                let frame_bytes = frame_elems * es;
3637
3638                let host = unsafe {
3639                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
3640                };
3641                let datatype = writer.dataset_datatype(ds_index);
3642                let raw = to_stored_byte_order(host, &datatype, es)?;
3643
3644                // Merge the buffer with the new frames when it is the
3645                // dataset's tail; a buffer left mid-extent (the extent moved
3646                // past it) keeps its recorded place — flush it and start
3647                // fresh at the current end.
3648                let taken = { writer.ds(ds_index).lock().append.take() };
3649                let (base_dim0, buffered_frames, mut combined) = match taken {
3650                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
3651                        (b.base as usize, b.frames as usize, b.bytes)
3652                    }
3653                    Some(b) => {
3654                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
3655                        (current_dim0, 0, Vec::new())
3656                    }
3657                    None => (current_dim0, 0, Vec::new()),
3658                };
3659                combined.extend_from_slice(&raw);
3660
3661                let total_frames = buffered_frames + n_new_frames;
3662
3663                // Rows up to the last chunk boundary are written now; the
3664                // tail that does not complete a chunk goes back in the
3665                // buffer for the next append (or the flush at close). The
3666                // boundary can precede `base_dim0` — a reopened file's
3667                // flushed partial chunk leaves the base mid-chunk — in
3668                // which case everything is tail.
3669                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
3670                let write_frames = last_boundary.saturating_sub(base_dim0);
3671                let tail_frames = total_frames - write_frames;
3672                if write_frames > 0 {
3673                    writer.write_append_frames(
3674                        ds_index,
3675                        base_dim0 as u64,
3676                        write_frames as u64,
3677                        &combined[..write_frames * frame_bytes],
3678                    )?;
3679                }
3680                if tail_frames > 0 {
3681                    let ds = writer.ds(ds_index);
3682                    let mut m = ds.lock();
3683                    m.append = Some(crate::io::writer::AppendBuffer {
3684                        base: (base_dim0 + write_frames) as u64,
3685                        frames: tail_frames as u64,
3686                        bytes: combined[write_frames * frame_bytes..].to_vec(),
3687                    });
3688                }
3689
3690                // Extend dims to include all frames (buffered + new)
3691                let logical_dim0 = base_dim0 + total_frames;
3692                let mut new_dims: Vec<u64> = dims;
3693                new_dims[0] = logical_dim0 as u64;
3694                writer.extend_dataset_inner(ds_index, &new_dims)?;
3695
3696                Ok(())
3697            }
3698            DatasetInfo::Reader { .. } => {
3699                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
3700            }
3701        }
3702    }
3703
3704    /// Extend the dimensions of a chunked dataset.
3705    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
3706        match &self.info {
3707            DatasetInfo::Writer {
3708                index, chunk_index, ..
3709            } => {
3710                if chunk_index.is_none() {
3711                    return Err(Hdf5Error::InvalidState(
3712                        "extend is only for chunked datasets".into(),
3713                    ));
3714                }
3715
3716                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3717                let inner = borrow_inner(&self.file_inner);
3718                match &*inner {
3719                    H5FileInner::Writer(writer) => {
3720                        writer.extend_dataset(*index, &dims_u64)?;
3721                        Ok(())
3722                    }
3723                    _ => Err(Hdf5Error::InvalidState(
3724                        "file is no longer in write mode".into(),
3725                    )),
3726                }
3727            }
3728            DatasetInfo::Reader { .. } => {
3729                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
3730            }
3731        }
3732    }
3733
3734    /// Set the logical extent of a chunked dataset, growing **or
3735    /// shrinking** any dimension.
3736    ///
3737    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
3738    /// dimension — for example to correct an over-extended frame count
3739    /// after writing a partial multi-frame chunk. Shrinking prunes the
3740    /// stored chunks the way libhdf5's `H5Dset_extent` does: a chunk
3741    /// entirely beyond the new extent is removed from the chunk index and
3742    /// its storage freed for reuse, and a chunk the new extent cuts
3743    /// through has its out-of-extent region overwritten with the fill
3744    /// value — so growing the extent back exposes fill values, not the
3745    /// old data. The new extent must not exceed the dataset's maximum
3746    /// dimensions.
3747    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
3748        match &self.info {
3749            DatasetInfo::Writer { index, .. } => {
3750                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3751                let inner = borrow_inner(&self.file_inner);
3752                match &*inner {
3753                    H5FileInner::Writer(writer) => {
3754                        writer.set_dataset_extent(*index, &dims_u64)?;
3755                        Ok(())
3756                    }
3757                    _ => Err(Hdf5Error::InvalidState(
3758                        "file is no longer in write mode".into(),
3759                    )),
3760                }
3761            }
3762            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
3763                "cannot set extent in read mode".into(),
3764            )),
3765        }
3766    }
3767
3768    /// Flush a chunked dataset's index structures to disk.
3769    pub fn flush(&self) -> Result<()> {
3770        match &self.info {
3771            DatasetInfo::Writer { index, .. } => {
3772                let inner = borrow_inner(&self.file_inner);
3773                match &*inner {
3774                    H5FileInner::Writer(writer) => {
3775                        writer.flush_dataset(*index)?;
3776                        Ok(())
3777                    }
3778                    _ => Ok(()),
3779                }
3780            }
3781            DatasetInfo::Reader { .. } => Ok(()),
3782        }
3783    }
3784
3785    /// Read a slice (hyperslab) of the dataset as a typed vector.
3786    ///
3787    /// `starts` and `counts` define the N-dimensional selection:
3788    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
3789    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
3790        match &self.info {
3791            DatasetInfo::Reader {
3792                name, element_size, ..
3793            } => {
3794                if T::element_size() != *element_size {
3795                    return Err(Hdf5Error::TypeMismatch(format!(
3796                        "read type has element size {} but dataset has element size {}",
3797                        T::element_size(),
3798                        element_size,
3799                    )));
3800                }
3801                let datatype = self.datatype()?;
3802                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
3803                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
3804
3805                let count = element_count(&counts_u64)?;
3806                let mut inner = borrow_inner_mut(&self.file_inner);
3807                let H5FileInner::Reader(reader) = &mut *inner else {
3808                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3809                };
3810                // The selection lands in the vector this returns, so its bytes
3811                // are touched once instead of being read into a byte buffer and
3812                // copied into a second one of the same size.
3813                read_image_into_new(count, |image| {
3814                    reader.read_slice_into(name, &starts_u64, &counts_u64, image)?;
3815                    to_host_byte_order(image, &datatype, T::element_size())
3816                })
3817            }
3818            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3819                "cannot read_slice from a dataset in write mode".into(),
3820            )),
3821        }
3822    }
3823
3824    /// Read a strided hyperslab as a typed vector — h5py's stepped slicing
3825    /// (`ds[a:b:s]`) or the general `start`/`stride`/`count`/`block` form of
3826    /// `H5Sselect_hyperslab`.
3827    ///
3828    /// One entry per dimension: `start[d]` is the first index, `stride[d]`
3829    /// the spacing between selected blocks (all-`1` is the same selection
3830    /// [`read_slice`](Self::read_slice) reads), `count[d]` how many blocks,
3831    /// and `block[d]` how many contiguous elements each block covers. The
3832    /// returned vector is row-major over `count[d] * block[d]` per
3833    /// dimension — exactly the shape h5py's stepped slicing produces.
3834    ///
3835    /// ```no_run
3836    /// # use rust_hdf5::H5File;
3837    /// let file = H5File::open("data.h5").unwrap();
3838    /// let ds = file.dataset("series").unwrap(); // shape [100]
3839    /// // Python: ds[0:100:2] — every other element.
3840    /// let evens: Vec<f64> = ds.read_hyperslab(&[0], &[2], &[50], &[1]).unwrap();
3841    /// ```
3842    pub fn read_hyperslab<T: H5Type>(
3843        &self,
3844        start: &[usize],
3845        stride: &[usize],
3846        count: &[usize],
3847        block: &[usize],
3848    ) -> Result<Vec<T>> {
3849        match &self.info {
3850            DatasetInfo::Reader {
3851                name, element_size, ..
3852            } => {
3853                if T::element_size() != *element_size {
3854                    return Err(Hdf5Error::TypeMismatch(format!(
3855                        "read type has element size {} but dataset has element size {}",
3856                        T::element_size(),
3857                        element_size,
3858                    )));
3859                }
3860                let datatype = self.datatype()?;
3861                let start_u64: Vec<u64> = start.iter().map(|&s| s as u64).collect();
3862                let stride_u64: Vec<u64> = stride.iter().map(|&s| s as u64).collect();
3863                let count_u64: Vec<u64> = count.iter().map(|&c| c as u64).collect();
3864                let block_u64: Vec<u64> = block.iter().map(|&b| b as u64).collect();
3865
3866                let selected: Vec<u64> = count_u64
3867                    .iter()
3868                    .zip(&block_u64)
3869                    .map(|(&c, &b)| c.saturating_mul(b))
3870                    .collect();
3871                let n = element_count(&selected)?;
3872                let mut inner = borrow_inner_mut(&self.file_inner);
3873                let H5FileInner::Reader(reader) = &mut *inner else {
3874                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3875                };
3876                read_image_into_new(n, |image| {
3877                    reader.read_hyperslab_into(
3878                        name,
3879                        &start_u64,
3880                        &stride_u64,
3881                        &count_u64,
3882                        &block_u64,
3883                        image,
3884                    )?;
3885                    to_host_byte_order(image, &datatype, T::element_size())
3886                })
3887            }
3888            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3889                "cannot read_hyperslab from a dataset in write mode".into(),
3890            )),
3891        }
3892    }
3893
3894    /// Read a list of coordinates in one call, as a typed vector — h5py
3895    /// fancy indexing with a coordinate list.
3896    ///
3897    /// `points[i]` is a coordinate with one entry per dimension. The
3898    /// returned vector holds one element per point, in the same order as
3899    /// `points`, regardless of the dataset's rank.
3900    ///
3901    /// ```no_run
3902    /// # use rust_hdf5::H5File;
3903    /// let file = H5File::open("data.h5").unwrap();
3904    /// let ds = file.dataset("grid").unwrap(); // shape [10, 10]
3905    /// // Python: ds[np.array([[0, 0], [3, 4], [9, 9]])]
3906    /// let picked: Vec<f64> = ds.read_points(&[vec![0, 0], vec![3, 4], vec![9, 9]]).unwrap();
3907    /// ```
3908    pub fn read_points<T: H5Type>(&self, points: &[Vec<usize>]) -> Result<Vec<T>> {
3909        match &self.info {
3910            DatasetInfo::Reader {
3911                name, element_size, ..
3912            } => {
3913                if T::element_size() != *element_size {
3914                    return Err(Hdf5Error::TypeMismatch(format!(
3915                        "read type has element size {} but dataset has element size {}",
3916                        T::element_size(),
3917                        element_size,
3918                    )));
3919                }
3920                let datatype = self.datatype()?;
3921                let points_u64: Vec<Vec<u64>> = points
3922                    .iter()
3923                    .map(|p| p.iter().map(|&c| c as u64).collect())
3924                    .collect();
3925
3926                let mut inner = borrow_inner_mut(&self.file_inner);
3927                let H5FileInner::Reader(reader) = &mut *inner else {
3928                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3929                };
3930                read_image_into_new(points_u64.len(), |image| {
3931                    reader.read_points_into(name, &points_u64, image)?;
3932                    to_host_byte_order(image, &datatype, T::element_size())
3933                })
3934            }
3935            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3936                "cannot read_points from a dataset in write mode".into(),
3937            )),
3938        }
3939    }
3940
3941    /// Read one chunk's raw (still-filtered) bytes and its filter mask,
3942    /// addressed by chunk-grid coordinates — the read half of
3943    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) and the HDF5 "direct
3944    /// chunk read" (`H5Dread_chunk`, formerly `H5DOread_chunk`; h5py's
3945    /// `Dataset.id.read_direct_chunk`).
3946    ///
3947    /// The bytes are exactly what is stored on disk: filtered/compressed if
3948    /// the dataset has a filter pipeline, with no decompression applied. The
3949    /// returned `u32` is the chunk's filter mask: bit *i* set means filter
3950    /// *i* of the pipeline was **not** applied to this particular chunk and
3951    /// must be skipped when reversing it.
3952    ///
3953    /// `Err` if the dataset is not chunked, `chunk_coords` has the wrong
3954    /// rank, or the chunk at those coordinates has never been written.
3955    ///
3956    /// ```no_run
3957    /// # use rust_hdf5::H5File;
3958    /// let file = H5File::open("data.h5").unwrap();
3959    /// let ds = file.dataset("frames").unwrap();
3960    /// let (raw, filter_mask) = ds.read_chunk_raw_at(&[0, 0]).unwrap();
3961    /// ```
3962    pub fn read_chunk_raw_at(&self, chunk_coords: &[usize]) -> Result<(Vec<u8>, u32)> {
3963        match &self.info {
3964            DatasetInfo::Reader { name, .. } => {
3965                let coords_u64: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3966                let mut inner = borrow_inner_mut(&self.file_inner);
3967                match &mut *inner {
3968                    H5FileInner::Reader(reader) => Ok(reader.read_chunk_raw_at(name, &coords_u64)?),
3969                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
3970                }
3971            }
3972            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3973                "cannot read_chunk_raw_at from a dataset in write mode".into(),
3974            )),
3975        }
3976    }
3977
3978    /// Write a typed slice to a sub-region of the dataset.
3979    ///
3980    /// `starts` and `counts` define the N-dimensional selection, which must lie
3981    /// inside the dataset's current extent.
3982    ///
3983    /// Works for both contiguous and chunked datasets. For a chunked dataset
3984    /// only the chunks the selection touches are rewritten — a partially
3985    /// covered chunk is read back, patched, and written again, so updating one
3986    /// row of an appendable dataset costs the chunks that row crosses rather
3987    /// than the whole dataset. Elements of a touched chunk that the selection
3988    /// does not cover keep their stored value, or the dataset's fill value if
3989    /// the chunk did not exist yet.
3990    pub fn write_slice<T: H5Type>(
3991        &self,
3992        starts: &[usize],
3993        counts: &[usize],
3994        data: &[T],
3995    ) -> Result<()> {
3996        match &self.info {
3997            DatasetInfo::Writer {
3998                index,
3999                element_size,
4000                ..
4001            } => {
4002                if T::element_size() != *element_size {
4003                    return Err(Hdf5Error::TypeMismatch(format!(
4004                        "write type has element size {} but dataset expects {}",
4005                        T::element_size(),
4006                        element_size,
4007                    )));
4008                }
4009
4010                let expected: usize = counts.iter().product();
4011                if data.len() != expected {
4012                    return Err(Hdf5Error::InvalidState(format!(
4013                        "data length {} does not match slice size {}",
4014                        data.len(),
4015                        expected,
4016                    )));
4017                }
4018
4019                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4020                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4021
4022                let byte_len = data.len() * T::element_size();
4023                let host =
4024                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
4025
4026                let inner = borrow_inner(&self.file_inner);
4027                match &*inner {
4028                    H5FileInner::Writer(writer) => {
4029                        let datatype = writer.dataset_datatype(*index);
4030                        let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
4031                        writer.write_slice(*index, &starts_u64, &counts_u64, &stored)?;
4032                        Ok(())
4033                    }
4034                    _ => Err(Hdf5Error::InvalidState(
4035                        "file is no longer in write mode".into(),
4036                    )),
4037                }
4038            }
4039            DatasetInfo::Reader { .. } => {
4040                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4041            }
4042        }
4043    }
4044
4045    /// Replace elements `start .. start + strings.len()` of a 1-D
4046    /// variable-length string dataset.
4047    ///
4048    /// The extent and every element outside the range are left alone, and the
4049    /// cost is the new strings plus the chunks holding their references — not
4050    /// the column. The dataset's character set is enforced: a non-ASCII
4051    /// replacement in a dataset that declares ASCII is rejected rather than
4052    /// stored under a datatype that misdescribes it.
4053    ///
4054    /// The global heap objects the replaced references pointed at are freed —
4055    /// the same reclaim libhdf5 performs on an overwrite — so updating one
4056    /// element repeatedly reuses space rather than growing the file. A
4057    /// collection emptied by the update returns its block to the allocator.
4058    /// Under SWMR nothing is freed, because a reader may still be following
4059    /// those references.
4060    ///
4061    /// ```no_run
4062    /// # use rust_hdf5::H5File;
4063    /// let file = H5File::open_rw("meta.h5").unwrap();
4064    /// let ds = file.dataset_writer("notes").unwrap();
4065    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
4066    /// file.close().unwrap();
4067    /// ```
4068    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
4069        match &self.info {
4070            DatasetInfo::Writer { index, .. } => {
4071                let inner = borrow_inner(&self.file_inner);
4072                match &*inner {
4073                    H5FileInner::Writer(writer) => {
4074                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
4075                        Ok(())
4076                    }
4077                    _ => Err(Hdf5Error::InvalidState(
4078                        "file is no longer in write mode".into(),
4079                    )),
4080                }
4081            }
4082            DatasetInfo::Reader { .. } => {
4083                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4084            }
4085        }
4086    }
4087
4088    /// Read variable-length strings from a dataset.
4089    ///
4090    /// This handles h5py-style vlen string datasets that store strings
4091    /// as global heap references. Returns one String per element.
4092    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
4093        match &self.info {
4094            DatasetInfo::Reader { name, .. } => {
4095                let mut inner = borrow_inner_mut(&self.file_inner);
4096                match &mut *inner {
4097                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
4098                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4099                }
4100            }
4101            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4102                "cannot read vlen strings from a dataset in write mode".into(),
4103            )),
4104        }
4105    }
4106
4107    /// Read variable-length byte arrays from a dataset.
4108    ///
4109    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
4110    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
4111    /// that store each element as a global heap reference. Returns one
4112    /// `Vec<u8>` per element.
4113    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
4114        match &self.info {
4115            DatasetInfo::Reader { name, .. } => {
4116                let mut inner = borrow_inner_mut(&self.file_inner);
4117                match &mut *inner {
4118                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
4119                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4120                }
4121            }
4122            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4123                "cannot read vlen bytes from a dataset in write mode".into(),
4124            )),
4125        }
4126    }
4127
4128    /// Write object references naming `paths` into elements `0..paths.len()`
4129    /// — h5py's `refs[i] = f['/target'].ref`.
4130    ///
4131    /// The dataset must have been created with
4132    /// [`object_references`](DatasetBuilder::object_references). A path names
4133    /// a dataset or a group (`/` is the root group) and must already exist;
4134    /// what reaches the file is the target's object header address, which is
4135    /// assigned when the file is finalized. Elements left unwritten read back
4136    /// as null references.
4137    pub fn write_object_references(&self, paths: &[&str]) -> Result<()> {
4138        match &self.info {
4139            DatasetInfo::Writer { index, .. } => {
4140                let inner = borrow_inner(&self.file_inner);
4141                match &*inner {
4142                    H5FileInner::Writer(writer) => {
4143                        writer.write_object_references(*index, 0, paths)?;
4144                        Ok(())
4145                    }
4146                    _ => Err(Hdf5Error::InvalidState(
4147                        "file is no longer in write mode".into(),
4148                    )),
4149                }
4150            }
4151            DatasetInfo::Reader { .. } => {
4152                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4153            }
4154        }
4155    }
4156
4157    /// Write region references over `targets` into elements
4158    /// `0..targets.len()` — h5py's `refs[i] = f['/target'].regionref[0:3]`.
4159    ///
4160    /// The dataset must have been created with
4161    /// [`region_references`](DatasetBuilder::region_references). Each target is
4162    /// the path of an existing *dataset* and a [`Selection`] over it, which
4163    /// must fit that dataset's extent — the rule `H5Rcreate` applies. What
4164    /// reaches the file is a global-heap object holding the target's object
4165    /// header address (assigned when the file is finalized) and the serialized
4166    /// selection. Elements left unwritten read back as null references.
4167    ///
4168    /// ```no_run
4169    /// # use rust_hdf5::{H5File, PointSelection, Selection};
4170    /// let file = H5File::create("regions.h5").unwrap();
4171    /// file.new_dataset::<i32>().shape([4, 6]).create("m").unwrap();
4172    /// let refs = file.new_dataset::<u64>()
4173    ///     .region_references()
4174    ///     .shape([1])
4175    ///     .create("refs")
4176    ///     .unwrap();
4177    /// let points = Selection::Points(PointSelection {
4178    ///     rank: 2,
4179    ///     points: vec![vec![0, 1], vec![3, 5]],
4180    /// });
4181    /// refs.write_region_references(&[("/m", points)]).unwrap();
4182    /// file.close().unwrap();
4183    /// ```
4184    pub fn write_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4185        match &self.info {
4186            DatasetInfo::Writer { index, .. } => {
4187                let inner = borrow_inner(&self.file_inner);
4188                match &*inner {
4189                    H5FileInner::Writer(writer) => {
4190                        writer.write_region_references(*index, 0, targets)?;
4191                        Ok(())
4192                    }
4193                    _ => Err(Hdf5Error::InvalidState(
4194                        "file is no longer in write mode".into(),
4195                    )),
4196                }
4197            }
4198            DatasetInfo::Reader { .. } => {
4199                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4200            }
4201        }
4202    }
4203
4204    /// Write revised region references over `targets` into elements
4205    /// `0..targets.len()` — `H5Rcreate_region` plus `H5Dwrite` of an
4206    /// `H5T_STD_REF` dataset.
4207    ///
4208    /// The dataset must have been created with
4209    /// [`std_region_references`](DatasetBuilder::std_region_references) or one
4210    /// of its two siblings, which make the same datatype. Each target is the
4211    /// path of an existing *dataset* and a [`Selection`] over it, which must
4212    /// fit that dataset's extent. What reaches the file is a global-heap blob
4213    /// holding the target's object header address (assigned when the file is
4214    /// finalized) and the serialized selection, and an element carrying the
4215    /// blob's id and its byte count. Elements left unwritten read back as null
4216    /// references.
4217    pub fn write_std_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4218        let targets: Vec<(&str, ReferenceTarget)> = targets
4219            .iter()
4220            .map(|(path, selection)| (*path, ReferenceTarget::Region(selection.clone())))
4221            .collect();
4222        self.write_revised_references(&targets)
4223    }
4224
4225    /// Write attribute references naming `targets` into elements
4226    /// `0..targets.len()` — `H5Rcreate_attr` plus `H5Dwrite` of an
4227    /// `H5T_STD_REF` dataset.
4228    ///
4229    /// Each target is the path of an existing object — a dataset, a group, or
4230    /// `/` for the root group — and the name of an attribute it already
4231    /// carries. There is no pre-1.12 form of this reference kind, so the
4232    /// dataset must have been created with
4233    /// [`attribute_references`](DatasetBuilder::attribute_references) or one of
4234    /// its two siblings. Elements left unwritten read back as null references.
4235    pub fn write_attribute_references(&self, targets: &[(&str, &str)]) -> Result<()> {
4236        let targets: Vec<(&str, ReferenceTarget)> = targets
4237            .iter()
4238            .map(|(path, name)| (*path, ReferenceTarget::Attribute((*name).to_string())))
4239            .collect();
4240        self.write_revised_references(&targets)
4241    }
4242
4243    /// Store `targets` as 1.12 reference elements, whatever mix of kinds they
4244    /// are: the one path both revised-reference writers take.
4245    fn write_revised_references(&self, targets: &[(&str, ReferenceTarget)]) -> Result<()> {
4246        match &self.info {
4247            DatasetInfo::Writer { index, .. } => {
4248                let inner = borrow_inner(&self.file_inner);
4249                match &*inner {
4250                    H5FileInner::Writer(writer) => {
4251                        writer.write_revised_references(*index, 0, targets)?;
4252                        Ok(())
4253                    }
4254                    _ => Err(Hdf5Error::InvalidState(
4255                        "file is no longer in write mode".into(),
4256                    )),
4257                }
4258            }
4259            DatasetInfo::Reader { .. } => {
4260                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4261            }
4262        }
4263    }
4264
4265    /// Read a reference dataset's elements, each resolved to the object it
4266    /// names.
4267    ///
4268    /// Every reference kind is read: the pre-1.12 pair h5py writes —
4269    /// `Reference` (an object header address) and `RegionReference` (a heap id
4270    /// whose heap object holds the target plus a serialized selection) — and
4271    /// the 1.12 `H5T_STD_REF` trio, `H5R_OBJECT2`, `H5R_DATASET_REGION2` and
4272    /// `H5R_ATTR`. An object reference comes back as [`Reference::Object`]
4273    /// carrying the target's path, a region reference as
4274    /// [`Reference::Region`], whose [`bounds`](Reference::bounds) is the
4275    /// selection's bounding box — libhdf5's `H5Sget_select_bounds` — and an
4276    /// attribute reference as [`Reference::Attr`], which adds the attribute's
4277    /// name.
4278    ///
4279    /// A 1.12 reference written into a file other than its target's carries
4280    /// that file's name, and [`Reference::file`] reports it; the path is then
4281    /// a path inside that file, resolved by opening it under the name the
4282    /// reference carries, and `None` when nothing is there.
4283    ///
4284    /// ```no_run
4285    /// # use rust_hdf5::H5File;
4286    /// let file = H5File::open("refs.h5").unwrap();
4287    /// for r in file.dataset("refs").unwrap().read_references().unwrap() {
4288    ///     println!("{:?} {:?}", r.path(), r.bounds());
4289    /// }
4290    /// ```
4291    pub fn read_references(&self) -> Result<Vec<Reference>> {
4292        match &self.info {
4293            DatasetInfo::Reader { name, .. } => {
4294                let mut inner = borrow_inner_mut(&self.file_inner);
4295                match &mut *inner {
4296                    H5FileInner::Reader(reader) => Ok(reader.read_references(name)?),
4297                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4298                }
4299            }
4300            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4301                "cannot read references from a dataset in write mode".into(),
4302            )),
4303        }
4304    }
4305
4306    /// Read a string dataset, fixed-width or variable-length, as one `String`
4307    /// per element.
4308    ///
4309    /// The width of a `FixedString` dataset is whatever the file says, so a
4310    /// 24-byte label column and a 100-byte one are read by the same call. The
4311    /// padding rule the datatype declares decides where each element ends —
4312    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
4313    /// character set decides how the remaining bytes are decoded: ASCII (0)
4314    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
4315    /// violates either is an error naming the element, not a silent
4316    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
4317    /// call that accepts such a file, replacing what it cannot decode.
4318    ///
4319    /// ```no_run
4320    /// # use rust_hdf5::H5File;
4321    /// let file = H5File::open("labels.h5").unwrap();
4322    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
4323    /// ```
4324    pub fn read_strings(&self) -> Result<Vec<String>> {
4325        self.read_strings_inner(false)
4326    }
4327
4328    /// [`read_strings`](Self::read_strings), but bytes that do not decode
4329    /// under the dataset's character set become U+FFFD instead of an error.
4330    ///
4331    /// Producers do mislabel the character set — a file that declares ASCII
4332    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
4333    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
4334        self.read_strings_inner(true)
4335    }
4336
4337    /// The single owner of string decoding for both string datatypes: the
4338    /// element bytes are found differently, the padding and character-set
4339    /// rules that turn them into a `String` are the same.
4340    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
4341        if matches!(self.info, DatasetInfo::Writer { .. }) {
4342            return Err(Hdf5Error::InvalidState(
4343                "cannot read strings from a dataset in write mode".into(),
4344            ));
4345        }
4346        match self.datatype()? {
4347            DatatypeMessage::VarLenString { charset, .. } => self
4348                .read_vlen_bytes()?
4349                .iter()
4350                .enumerate()
4351                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
4352                .collect(),
4353            DatatypeMessage::FixedString {
4354                size,
4355                padding,
4356                charset,
4357            } => {
4358                let width = size as usize;
4359                if width == 0 {
4360                    // A corrupt file can declare it; `chunks_exact(0)` panics.
4361                    return Err(Hdf5Error::InvalidState(
4362                        "fixed-string datatype has zero width".into(),
4363                    ));
4364                }
4365                // `read_raw_bytes` returns `product(dims) * width` bytes, so
4366                // `chunks_exact` leaves no remainder.
4367                let raw = self.read_raw_bytes()?;
4368                raw.chunks_exact(width)
4369                    .enumerate()
4370                    .map(|(i, elem)| {
4371                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
4372                    })
4373                    .collect()
4374            }
4375            other => Err(Hdf5Error::InvalidState(format!(
4376                "read_strings is only for string datasets, this one is {other:?}"
4377            ))),
4378        }
4379    }
4380
4381    /// Read the entire dataset as a typed vector.
4382    ///
4383    /// The raw bytes are read from the file and reinterpreted as `T`. The
4384    /// caller must ensure that `T` matches the datatype used when the dataset
4385    /// was written.
4386    ///
4387    /// # Errors
4388    ///
4389    /// Returns an error if:
4390    /// - The file is in write mode.
4391    /// - The raw data size is not a multiple of `T::element_size()`.
4392    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
4393        match &self.info {
4394            DatasetInfo::Reader {
4395                name, element_size, ..
4396            } => {
4397                if T::element_size() != *element_size {
4398                    return Err(Hdf5Error::TypeMismatch(format!(
4399                        "read type has element size {} but dataset has element size {}",
4400                        T::element_size(),
4401                        element_size,
4402                    )));
4403                }
4404
4405                let datatype = self.datatype()?;
4406                let mut inner = borrow_inner_mut(&self.file_inner);
4407                let H5FileInner::Reader(reader) = &mut *inner else {
4408                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4409                };
4410                let total = reader.dataset_raw_size(name)? as usize;
4411                if !total.is_multiple_of(T::element_size()) {
4412                    return Err(Hdf5Error::TypeMismatch(format!(
4413                        "raw data size {total} is not a multiple of element size {}",
4414                        T::element_size(),
4415                    )));
4416                }
4417
4418                // The image is read into the vector this returns, so the
4419                // bytes are touched once rather than being zeroed, read, and
4420                // then copied into a second buffer of the same size.
4421                read_image_into_new(total / T::element_size(), |image| {
4422                    reader.read_dataset_raw_into(name, image)?;
4423                    to_host_byte_order(image, &datatype, T::element_size())
4424                })
4425            }
4426            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4427                "cannot read from a dataset in write mode".into(),
4428            )),
4429        }
4430    }
4431
4432    /// View the entire dataset as `&[T]` pointing straight into the file's
4433    /// memory map — no read, no copy, no allocation.
4434    ///
4435    /// The returned [`MappedView<T>`](crate::MappedView) dereferences to
4436    /// `&[T]` holding exactly what [`read_raw`](Self::read_raw) would have
4437    /// returned, bit for bit.
4438    ///
4439    /// # When it works
4440    ///
4441    /// The file must be open read-only and mapped (which a read-only open
4442    /// does whenever the OS allows), the dataset's raw data must be one
4443    /// contiguous stretch of that file, and its stored elements must already
4444    /// be the host image of a `T` — same width, host byte order, significant
4445    /// bits filling the element. That is the ordinary case for an
4446    /// uncompressed, non-chunked numeric dataset written by either this crate
4447    /// or libhdf5.
4448    ///
4449    /// # When it refuses
4450    ///
4451    /// Zero-copy is a contract, not an optimization: when the bytes cannot be
4452    /// handed over as they lie, this returns
4453    /// [`Hdf5Error::NotViewable`](crate::Hdf5Error::NotViewable) naming the
4454    /// reason and never quietly falls back to copying. Every
4455    /// [`ViewRefusal`](crate::ViewRefusal) is a case where
4456    /// [`read_raw`](Self::read_raw) still works: the file is not mapped, the
4457    /// layout is chunked, compact, virtual, or external, no storage is
4458    /// allocated (the dataset reads as its fill value), `T` is the wrong
4459    /// width, the stored elements need a byte-order swap or bit unpacking,
4460    /// the data lands at an offset `T`'s alignment does not permit, or the
4461    /// image runs past the end of the map.
4462    ///
4463    /// # Snapshot semantics
4464    ///
4465    /// The view owns a share of the map rather than borrowing the file
4466    /// handle, so it stays readable after the dataset and the file are
4467    /// dropped, and after a SWMR refresh has retaken the map — a live view
4468    /// keeps showing the file as it was when *its* map was taken, while the
4469    /// refreshed handle reads the new one. Nothing about a view is
4470    /// invalidated by anything this process does.
4471    ///
4472    /// # Truncation
4473    ///
4474    /// The pages are the file's own. Another process writing the file in
4475    /// place is seen through the view, and one *truncating* it under the map
4476    /// faults with `SIGBUS` on the pages that went away. That is the standing
4477    /// risk of mapping the file at all; a view does not add to it, and no
4478    /// guard inside this process can close it.
4479    ///
4480    /// ```no_run
4481    /// # use rust_hdf5::H5File;
4482    /// let file = H5File::open("data.h5")?;
4483    /// let ds = file.dataset("matrix")?;
4484    /// let view = ds.read_mapped::<f64>()?;
4485    /// let total: f64 = view.iter().sum();
4486    /// # Ok::<(), rust_hdf5::Hdf5Error>(())
4487    /// ```
4488    #[cfg(feature = "mmap")]
4489    pub fn read_mapped<T: H5Type>(&self) -> Result<crate::mapped::MappedView<T>> {
4490        self.mapped_view(crate::mapped::ViewRange::Whole)
4491    }
4492
4493    /// View a contiguous sub-range of the dataset as `&[T]` pointing straight
4494    /// into the file's memory map.
4495    ///
4496    /// `starts` and `counts` name the same N-dimensional selection
4497    /// [`read_slice`](Self::read_slice) takes, and the view holds exactly what
4498    /// that call would have returned — but only when the selection is one
4499    /// contiguous run of the stored image: a trailing group of dimensions
4500    /// taken whole, the dimension before it taken as one span, and a single
4501    /// index along every dimension before that. Anything else steps over
4502    /// elements a single slice cannot skip, and is refused with
4503    /// [`ViewRefusal::Range`](crate::ViewRefusal::Range) rather than gathered
4504    /// into a copy.
4505    ///
4506    /// Everything [`read_mapped`](Self::read_mapped) documents about when a
4507    /// dataset can be viewed, snapshot semantics, and truncation applies here
4508    /// unchanged.
4509    #[cfg(feature = "mmap")]
4510    pub fn read_mapped_slice<T: H5Type>(
4511        &self,
4512        starts: &[usize],
4513        counts: &[usize],
4514    ) -> Result<crate::mapped::MappedView<T>> {
4515        self.mapped_view(crate::mapped::ViewRange::Slab { starts, counts })
4516    }
4517
4518    /// The one route from a dataset handle to the file's map: ask the reader
4519    /// that owns the dataset for the facts, and hand them to
4520    /// [`crate::mapped::view`], which is the only thing that can turn them
4521    /// into a view.
4522    #[cfg(feature = "mmap")]
4523    fn mapped_view<T: H5Type>(
4524        &self,
4525        range: crate::mapped::ViewRange<'_>,
4526    ) -> Result<crate::mapped::MappedView<T>> {
4527        let DatasetInfo::Reader { name, .. } = &self.info else {
4528            return Err(Hdf5Error::InvalidState(
4529                "cannot read from a dataset in write mode".into(),
4530            ));
4531        };
4532        let mut inner = borrow_inner_mut(&self.file_inner);
4533        let H5FileInner::Reader(reader) = &mut *inner else {
4534            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4535        };
4536        let src = reader.dataset_view_source(name)?;
4537        crate::mapped::view::<T>(&src, range).map_err(Hdf5Error::NotViewable)
4538    }
4539
4540    /// Read the raw byte image of a dataset without an `H5Type` carrier.
4541    ///
4542    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
4543    /// the element bytes verbatim regardless of the on-disk element type, so a
4544    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
4545    /// no matching Rust primitive can be read back and decoded by the caller.
4546    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
4547        match &self.info {
4548            DatasetInfo::Reader { name, .. } => {
4549                let mut inner = borrow_inner_mut(&self.file_inner);
4550                match &mut *inner {
4551                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
4552                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4553                }
4554            }
4555            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4556                "cannot read from a dataset in write mode".into(),
4557            )),
4558        }
4559    }
4560
4561    /// Read a numeric dataset as `T`, converting each element from the
4562    /// on-disk datatype.
4563    ///
4564    /// Unlike [`read_raw`](Self::read_raw), which requires `T`'s size to match
4565    /// the stored element size exactly, this inspects the dataset's datatype
4566    /// message — class, signedness, byte order, width — and converts per
4567    /// element:
4568    ///
4569    /// - integer → integer: checked; a stored value that does not fit in `T`
4570    ///   is an error naming the element index and value, never a silent wrap.
4571    /// - `f32` source → `f64`: exact widening.
4572    /// - `f64` source → `f32`, float → integer, and integer → float are
4573    ///   rejected as [`TypeMismatch`](Hdf5Error::TypeMismatch).
4574    ///
4575    /// Big-endian sources are decoded according to the datatype's byte order,
4576    /// which [`read_raw`](Self::read_raw)'s size-only check would misread.
4577    ///
4578    /// ```no_run
4579    /// # use rust_hdf5::H5File;
4580    /// let file = H5File::open("data.h5").unwrap();
4581    /// let ds = file.dataset("counts").unwrap(); // stored as e.g. i16
4582    /// let counts = ds.read_numeric_as::<i64>().unwrap();
4583    /// ```
4584    pub fn read_numeric_as<T: ReadNumeric>(&self) -> Result<Vec<T>> {
4585        match &self.info {
4586            DatasetInfo::Reader { name, .. } => {
4587                let (kind, raw) = {
4588                    let mut inner = borrow_inner_mut(&self.file_inner);
4589                    match &mut *inner {
4590                        H5FileInner::Reader(reader) => {
4591                            let info = reader
4592                                .dataset_info(name)
4593                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4594                            let kind = numeric::classify(&info.datatype)?;
4595                            (kind, reader.read_dataset_raw(name)?)
4596                        }
4597                        _ => {
4598                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4599                        }
4600                    }
4601                };
4602                numeric::convert(kind, &raw)
4603            }
4604            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4605                "cannot read from a dataset in write mode".into(),
4606            )),
4607        }
4608    }
4609
4610    /// Read a slice (hyperslab) of a numeric dataset as `T`, with the same
4611    /// per-element datatype conversion as
4612    /// [`read_numeric_as`](Self::read_numeric_as).
4613    ///
4614    /// `starts` and `counts` define the N-dimensional selection exactly as in
4615    /// [`read_slice`](Self::read_slice).
4616    pub fn read_numeric_slice_as<T: ReadNumeric>(
4617        &self,
4618        starts: &[usize],
4619        counts: &[usize],
4620    ) -> Result<Vec<T>> {
4621        match &self.info {
4622            DatasetInfo::Reader { name, .. } => {
4623                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4624                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4625                let (kind, raw) = {
4626                    let mut inner = borrow_inner_mut(&self.file_inner);
4627                    match &mut *inner {
4628                        H5FileInner::Reader(reader) => {
4629                            let info = reader
4630                                .dataset_info(name)
4631                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4632                            let kind = numeric::classify(&info.datatype)?;
4633                            (kind, reader.read_slice(name, &starts_u64, &counts_u64)?)
4634                        }
4635                        _ => {
4636                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4637                        }
4638                    }
4639                };
4640                numeric::convert(kind, &raw)
4641            }
4642            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4643                "cannot read_slice from a dataset in write mode".into(),
4644            )),
4645        }
4646    }
4647
4648    /// Read the whole dataset into a caller-provided buffer, with no allocation.
4649    ///
4650    /// `out` must have exactly `product(dims)` elements (the dataset's element
4651    /// count) and `T::element_size()` must match the dataset's on-disk element
4652    /// size, otherwise an error is returned and `out` is left unspecified. The
4653    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
4654    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
4655    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
4656    /// without the extra staging copy a `read_raw` + copy-into-pinned would
4657    /// incur. Works for every layout (contiguous, compact, and chunked under
4658    /// any index); for chunked data each decoded chunk is scattered directly
4659    /// into `out`.
4660    ///
4661    /// ```no_run
4662    /// # use rust_hdf5::H5File;
4663    /// let file = H5File::open("data.h5").unwrap();
4664    /// let ds = file.dataset("frames").unwrap();
4665    /// let n: usize = ds.shape().iter().product();
4666    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
4667    /// ds.read_raw_into(&mut buf).unwrap();
4668    /// ```
4669    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
4670        match &self.info {
4671            DatasetInfo::Reader {
4672                name, element_size, ..
4673            } => {
4674                if T::element_size() != *element_size {
4675                    return Err(Hdf5Error::TypeMismatch(format!(
4676                        "read type has element size {} but dataset has element size {}",
4677                        T::element_size(),
4678                        element_size,
4679                    )));
4680                }
4681                let datatype = self.datatype()?;
4682                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
4683                // byte representation; every bit pattern the read writes is a
4684                // valid `T`. The byte view borrows `out` exclusively for this
4685                // call, and `out.len() * element_size` cannot overflow because
4686                // it is the byte length of an existing slice (<= isize::MAX).
4687                let byte_len = out.len() * T::element_size();
4688                let bytes = unsafe {
4689                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4690                };
4691                {
4692                    let mut inner = borrow_inner_mut(&self.file_inner);
4693                    match &mut *inner {
4694                        H5FileInner::Reader(reader) => reader.read_dataset_raw_into(name, bytes)?,
4695                        _ => {
4696                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4697                        }
4698                    }
4699                }
4700                to_host_byte_order(bytes, &datatype, T::element_size())
4701            }
4702            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4703                "cannot read from a dataset in write mode".into(),
4704            )),
4705        }
4706    }
4707
4708    /// Read a hyperslab into a caller-provided buffer, with no allocation.
4709    ///
4710    /// `out` must have exactly `product(counts)` elements and
4711    /// `T::element_size()` must match the dataset's element size. The zero-copy
4712    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
4713    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
4714    /// selection are read, and the selected bytes land directly in `out` — the
4715    /// entry point for reading one frame / block straight into a pinned host
4716    /// buffer for an H2D transfer.
4717    ///
4718    /// ```no_run
4719    /// # use rust_hdf5::H5File;
4720    /// let file = H5File::open("vol.h5").unwrap();
4721    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
4722    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
4723    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
4724    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
4725    /// ```
4726    pub fn read_slice_into<T: H5Type>(
4727        &self,
4728        out: &mut [T],
4729        starts: &[usize],
4730        counts: &[usize],
4731    ) -> Result<()> {
4732        match &self.info {
4733            DatasetInfo::Reader {
4734                name, element_size, ..
4735            } => {
4736                if T::element_size() != *element_size {
4737                    return Err(Hdf5Error::TypeMismatch(format!(
4738                        "read type has element size {} but dataset has element size {}",
4739                        T::element_size(),
4740                        element_size,
4741                    )));
4742                }
4743                let datatype = self.datatype()?;
4744                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4745                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4746                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
4747                // borrow of `out`, byte length within bounds.
4748                let byte_len = out.len() * T::element_size();
4749                let bytes = unsafe {
4750                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4751                };
4752                {
4753                    let mut inner = borrow_inner_mut(&self.file_inner);
4754                    match &mut *inner {
4755                        H5FileInner::Reader(reader) => {
4756                            reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?
4757                        }
4758                        _ => {
4759                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4760                        }
4761                    }
4762                }
4763                to_host_byte_order(bytes, &datatype, T::element_size())
4764            }
4765            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4766                "cannot read from a dataset in write mode".into(),
4767            )),
4768        }
4769    }
4770}
4771
4772// ---------------------------------------------------------------------------
4773// Datatype-aware numeric conversion (read_numeric_as)
4774// ---------------------------------------------------------------------------
4775
4776/// Marker trait for the Rust types [`H5Dataset::read_numeric_as`] can convert
4777/// into: the integer primitives (checked, never wrapping) plus `f32`/`f64`
4778/// (widening only).
4779///
4780/// Sealed — the conversion policy is part of the library contract, so the
4781/// trait cannot be implemented outside this crate.
4782pub trait ReadNumeric: numeric::Sealed {}
4783impl<T: numeric::Sealed> ReadNumeric for T {}
4784
4785pub(crate) mod numeric {
4786    //! Per-element decode + checked conversion for `read_numeric_as` (and the
4787    //! attribute counterpart `H5Attribute::read_numeric_as`).
4788
4789    use crate::error::{Hdf5Error, Result};
4790    use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, IeeeFormat};
4791
4792    /// A source element, normalized: every standard integer width — u64::MAX
4793    /// included — fits in `i128` without loss.
4794    pub enum NumericSource {
4795        Int(i128),
4796        F32(f32),
4797        F64(f64),
4798    }
4799
4800    /// The on-disk element shape `classify` accepted.
4801    #[derive(Clone, Copy)]
4802    pub enum SourceKind {
4803        Int {
4804            size: usize,
4805            signed: bool,
4806            byte_order: ByteOrder,
4807        },
4808        F16(ByteOrder),
4809        F32(ByteOrder),
4810        F64(ByteOrder),
4811    }
4812
4813    impl SourceKind {
4814        fn element_size(self) -> usize {
4815            match self {
4816                SourceKind::Int { size, .. } => size,
4817                SourceKind::F16(_) => 2,
4818                SourceKind::F32(_) => 4,
4819                SourceKind::F64(_) => 8,
4820            }
4821        }
4822    }
4823
4824    /// Widen an IEEE 754 binary16 bit pattern to `f32`, which represents every
4825    /// half — including subnormals, infinities and NaN payloads — exactly.
4826    ///
4827    /// Rust has no stable `f16` to convert through.
4828    fn f16_bits_to_f32(bits: u16) -> f32 {
4829        let sign = u32::from(bits >> 15);
4830        let exponent = u32::from((bits >> 10) & 0x1f);
4831        let mantissa = u32::from(bits & 0x03ff);
4832        if exponent == 0 {
4833            // Zero and subnormals: the value is mantissa * 2^-24, which is a
4834            // normal f32 for every mantissa, so the multiply is exact. Going
4835            // through the sign separately keeps -0.0.
4836            let magnitude = mantissa as f32 * (1.0 / 16_777_216.0);
4837            return if sign == 1 { -magnitude } else { magnitude };
4838        }
4839        let out = if exponent == 0x1f {
4840            // Infinity and NaN; shifting the mantissa maps the quiet bit onto
4841            // f32's quiet bit and preserves the rest of the payload.
4842            (sign << 31) | 0x7f80_0000 | (mantissa << 13)
4843        } else {
4844            // Normal: rebias the exponent (127 - 15) and left-align the
4845            // mantissa.
4846            (sign << 31) | ((exponent + 112) << 23) | (mantissa << 13)
4847        };
4848        f32::from_bits(out)
4849    }
4850
4851    /// Map a datatype message to a supported numeric source shape.
4852    ///
4853    /// Accepts standard-width integers (1/2/4/8 bytes, full precision, zero
4854    /// bit offset) and IEEE binary32/binary64 floats; everything else is a
4855    /// `TypeMismatch` naming what was found.
4856    pub fn classify(dt: &DatatypeMessage) -> Result<SourceKind> {
4857        match *dt {
4858            DatatypeMessage::FixedPoint {
4859                size,
4860                byte_order,
4861                signed,
4862                bit_offset,
4863                bit_precision,
4864            } => {
4865                if !matches!(size, 1 | 2 | 4 | 8)
4866                    || bit_offset != 0
4867                    || u32::from(bit_precision) != size * 8
4868                {
4869                    return Err(Hdf5Error::TypeMismatch(format!(
4870                        "fixed-point datatype (size {size}, bit offset {bit_offset}, \
4871                         precision {bit_precision}) is not a standard-width integer",
4872                    )));
4873                }
4874                Ok(SourceKind::Int {
4875                    size: size as usize,
4876                    signed,
4877                    byte_order,
4878                })
4879            }
4880            DatatypeMessage::BitField {
4881                size,
4882                byte_order,
4883                bit_offset,
4884                bit_precision,
4885            } => {
4886                // A bit field has no signed form; a full-width one is the
4887                // unsigned integer of the stored width. A narrower one would
4888                // need a shift-and-mask conversion this path does not model.
4889                if !matches!(size, 1 | 2 | 4 | 8)
4890                    || bit_offset != 0
4891                    || u32::from(bit_precision) != size * 8
4892                {
4893                    return Err(Hdf5Error::TypeMismatch(format!(
4894                        "bit-field datatype (size {size}, bit offset {bit_offset}, \
4895                         precision {bit_precision}) is not a whole-width bit field",
4896                    )));
4897                }
4898                Ok(SourceKind::Int {
4899                    size: size as usize,
4900                    signed: false,
4901                    byte_order,
4902                })
4903            }
4904            DatatypeMessage::FloatingPoint {
4905                size,
4906                byte_order,
4907                exponent_size,
4908                mantissa_size,
4909                ..
4910            } => match dt.ieee_format() {
4911                Some(IeeeFormat::Binary16) => Ok(SourceKind::F16(byte_order)),
4912                Some(IeeeFormat::Binary32) => Ok(SourceKind::F32(byte_order)),
4913                Some(IeeeFormat::Binary64) => Ok(SourceKind::F64(byte_order)),
4914                None => Err(Hdf5Error::TypeMismatch(format!(
4915                    "floating-point datatype (size {size}, exponent {exponent_size} bits, \
4916                     mantissa {mantissa_size} bits) is not an IEEE 754 interchange format",
4917                ))),
4918            },
4919            ref other => Err(Hdf5Error::TypeMismatch(format!(
4920                "dataset datatype '{other}' is not numeric",
4921            ))),
4922        }
4923    }
4924
4925    fn decode_element(kind: SourceKind, bytes: &[u8]) -> NumericSource {
4926        match kind {
4927            SourceKind::Int {
4928                size,
4929                signed,
4930                byte_order,
4931            } => {
4932                let mut le = [0u8; 8];
4933                match byte_order {
4934                    ByteOrder::LittleEndian => le[..size].copy_from_slice(bytes),
4935                    ByteOrder::BigEndian => {
4936                        for (dst, src) in le[..size].iter_mut().zip(bytes.iter().rev()) {
4937                            *dst = *src;
4938                        }
4939                    }
4940                }
4941                let zero_extended = u64::from_le_bytes(le);
4942                let value = if signed {
4943                    // Arithmetic right shift sign-extends the low `size` bytes.
4944                    let shift = 64 - 8 * size as u32;
4945                    i128::from(((zero_extended as i64) << shift) >> shift)
4946                } else {
4947                    i128::from(zero_extended)
4948                };
4949                NumericSource::Int(value)
4950            }
4951            SourceKind::F16(byte_order) => {
4952                let arr: [u8; 2] = bytes.try_into().unwrap();
4953                let bits = match byte_order {
4954                    ByteOrder::LittleEndian => u16::from_le_bytes(arr),
4955                    ByteOrder::BigEndian => u16::from_be_bytes(arr),
4956                };
4957                NumericSource::F32(f16_bits_to_f32(bits))
4958            }
4959            SourceKind::F32(byte_order) => {
4960                let arr: [u8; 4] = bytes.try_into().unwrap();
4961                NumericSource::F32(match byte_order {
4962                    ByteOrder::LittleEndian => f32::from_le_bytes(arr),
4963                    ByteOrder::BigEndian => f32::from_be_bytes(arr),
4964                })
4965            }
4966            SourceKind::F64(byte_order) => {
4967                let arr: [u8; 8] = bytes.try_into().unwrap();
4968                NumericSource::F64(match byte_order {
4969                    ByteOrder::LittleEndian => f64::from_le_bytes(arr),
4970                    ByteOrder::BigEndian => f64::from_be_bytes(arr),
4971                })
4972            }
4973        }
4974    }
4975
4976    /// Decode and convert every element of `raw` into `T`.
4977    pub fn convert<T: Sealed>(kind: SourceKind, raw: &[u8]) -> Result<Vec<T>> {
4978        let size = kind.element_size();
4979        if !raw.len().is_multiple_of(size) {
4980            return Err(Hdf5Error::TypeMismatch(format!(
4981                "raw data size {} is not a multiple of element size {size}",
4982                raw.len(),
4983            )));
4984        }
4985        raw.chunks_exact(size)
4986            .enumerate()
4987            .map(|(index, bytes)| T::from_source(decode_element(kind, bytes), index))
4988            .collect()
4989    }
4990
4991    /// The sealed half of `ReadNumeric`: how one normalized source element
4992    /// becomes a `Self`, or a `TypeMismatch` explaining why it cannot.
4993    pub trait Sealed: Sized {
4994        fn from_source(src: NumericSource, index: usize) -> Result<Self>;
4995    }
4996
4997    macro_rules! int_targets {
4998        ($($t:ty),* $(,)?) => {$(
4999            impl Sealed for $t {
5000                fn from_source(src: NumericSource, index: usize) -> Result<Self> {
5001                    match src {
5002                        NumericSource::Int(v) => <$t>::try_from(v).map_err(|_| {
5003                            Hdf5Error::TypeMismatch(format!(
5004                                concat!(
5005                                    "value {} at element {} does not fit in ",
5006                                    stringify!($t),
5007                                ),
5008                                v, index,
5009                            ))
5010                        }),
5011                        NumericSource::F32(_) | NumericSource::F64(_) => {
5012                            Err(Hdf5Error::TypeMismatch(
5013                                concat!(
5014                                    "cannot read a floating-point dataset as ",
5015                                    stringify!($t),
5016                                    "; read as f64 and convert explicitly",
5017                                )
5018                                .into(),
5019                            ))
5020                        }
5021                    }
5022                }
5023            }
5024        )*};
5025    }
5026    int_targets!(i8, i16, i32, i64, u8, u16, u32, u64, u128);
5027
5028    // Not in the macro: `i128::try_from(i128)` is infallible, which trips
5029    // clippy::unnecessary_fallible_conversions.
5030    impl Sealed for i128 {
5031        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5032            match src {
5033                NumericSource::Int(v) => Ok(v),
5034                NumericSource::F32(_) | NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
5035                    "cannot read a floating-point dataset as i128; read as f64 and \
5036                     convert explicitly"
5037                        .into(),
5038                )),
5039            }
5040        }
5041    }
5042
5043    impl Sealed for f32 {
5044        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5045            match src {
5046                NumericSource::F32(v) => Ok(v),
5047                NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
5048                    "narrowing an f64 dataset to f32 loses precision; read as f64".into(),
5049                )),
5050                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
5051                    "cannot read an integer dataset as f32; read as an integer type and \
5052                     convert explicitly"
5053                        .into(),
5054                )),
5055            }
5056        }
5057    }
5058
5059    impl Sealed for f64 {
5060        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
5061            match src {
5062                NumericSource::F64(v) => Ok(v),
5063                // Every f32 is exactly representable as f64.
5064                NumericSource::F32(v) => Ok(f64::from(v)),
5065                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
5066                    "cannot read an integer dataset as f64; integers above 2^53 lose \
5067                     precision — read as an integer type and convert explicitly"
5068                        .into(),
5069                )),
5070            }
5071        }
5072    }
5073
5074    #[cfg(test)]
5075    mod tests {
5076        use super::*;
5077
5078        /// Every binary16 bit pattern class widens to the f32 with the same
5079        /// value: zeros keep their sign, subnormals stay exact, infinities and
5080        /// NaN payloads survive.
5081        #[test]
5082        fn f16_widening_is_exact() {
5083            let cases: [(u16, f32); 10] = [
5084                (0x0000, 0.0),
5085                (0x3c00, 1.0),
5086                (0xc000, -2.0),
5087                (0x3555, 0.333_251_95),   // nearest half to 1/3
5088                (0x0001, 5.960_464_5e-8), // smallest subnormal, 2^-24
5089                (0x03ff, 6.097_555e-5),   // largest subnormal
5090                (0x0400, 6.103_515_6e-5), // smallest normal
5091                (0x7bff, 65504.0),        // largest finite
5092                (0x7c00, f32::INFINITY),
5093                (0xfc00, f32::NEG_INFINITY),
5094            ];
5095            for (bits, expected) in cases {
5096                let got = f16_bits_to_f32(bits);
5097                assert_eq!(got, expected, "0x{bits:04x} widened to {got}");
5098            }
5099
5100            let neg_zero = f16_bits_to_f32(0x8000);
5101            assert_eq!(neg_zero, 0.0);
5102            assert!(neg_zero.is_sign_negative(), "-0.0 lost its sign");
5103
5104            let nan = f16_bits_to_f32(0x7e01);
5105            assert!(nan.is_nan());
5106            // The quiet bit and the payload land in f32's mantissa.
5107            assert_eq!(nan.to_bits(), 0x7fc0_2000);
5108        }
5109
5110        #[test]
5111        fn f16_source_converts_and_honors_byte_order() {
5112            let kind = classify(&DatatypeMessage::f16_type()).unwrap();
5113            // 1.0, -2.0, 0.333..., 65504
5114            let raw = [0x00, 0x3c, 0x00, 0xc0, 0x55, 0x35, 0xff, 0x7b];
5115            assert_eq!(
5116                convert::<f32>(kind, &raw).unwrap(),
5117                vec![1.0, -2.0, 0.333_251_95, 65504.0]
5118            );
5119            assert_eq!(
5120                convert::<f64>(kind, &raw).unwrap(),
5121                vec![1.0, -2.0, 0.333_251_953_125, 65504.0]
5122            );
5123
5124            let DatatypeMessage::FloatingPoint { .. } = DatatypeMessage::f16_type() else {
5125                unreachable!()
5126            };
5127            let mut be = DatatypeMessage::f16_type();
5128            if let DatatypeMessage::FloatingPoint { byte_order, .. } = &mut be {
5129                *byte_order = ByteOrder::BigEndian;
5130            }
5131            let be_kind = classify(&be).unwrap();
5132            assert_eq!(convert::<f32>(be_kind, &[0x3c, 0x00]).unwrap(), vec![1.0]);
5133        }
5134
5135        /// A float whose layout is not an interchange format is refused, not
5136        /// reinterpreted.
5137        #[test]
5138        fn non_ieee_float_is_refused() {
5139            let mut odd = DatatypeMessage::f32_type();
5140            if let DatatypeMessage::FloatingPoint { exponent_bias, .. } = &mut odd {
5141                *exponent_bias = 63;
5142            }
5143            assert!(odd.ieee_format().is_none());
5144            let err = classify(&odd).err().expect("non-IEEE float was accepted");
5145            assert!(
5146                err.to_string().contains("IEEE 754 interchange format"),
5147                "unexpected error: {err}"
5148            );
5149        }
5150    }
5151}
5152
5153#[cfg(test)]
5154mod tests {
5155    use crate::H5File;
5156    use std::path::PathBuf;
5157
5158    fn temp_path(name: &str) -> PathBuf {
5159        // Include PID + a per-call atomic counter so that concurrent
5160        // cargo invocations and any kernel-level "lock not yet
5161        // released" races between sequential opens cannot collide.
5162        use std::sync::atomic::{AtomicU64, Ordering};
5163        static COUNTER: AtomicU64 = AtomicU64::new(0);
5164        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
5165        std::env::temp_dir().join(format!(
5166            "hdf5_dataset_test_{}_{}_{}.h5",
5167            name,
5168            std::process::id(),
5169            n
5170        ))
5171    }
5172
5173    #[test]
5174    fn runtime_compound_via_datatype_override_and_raw_bytes() {
5175        use crate::format::messages::datatype::DatatypeMessage;
5176        use crate::types::{CompoundType, H5Type};
5177
5178        let path = temp_path("compound_raw");
5179        // A 12-byte packed compound with NO matching Rust primitive carrier,
5180        // so it can only be written through the datatype() override +
5181        // write_raw_bytes path (the runtime-CompoundType use case).
5182        let ct = CompoundType {
5183            members: vec![
5184                ("id".to_string(), i32::hdf5_type(), 0),
5185                ("val".to_string(), f64::hdf5_type(), 4),
5186            ],
5187            total_size: 12,
5188        };
5189        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
5190        let mut bytes = Vec::new();
5191        for (id, val) in recs {
5192            bytes.extend_from_slice(&id.to_le_bytes());
5193            bytes.extend_from_slice(&val.to_le_bytes());
5194        }
5195
5196        {
5197            let file = H5File::create(&path).unwrap();
5198            let ds = file
5199                .new_dataset::<u8>()
5200                .datatype(ct.to_datatype())
5201                .shape([recs.len()])
5202                .create("records")
5203                .unwrap();
5204            ds.write_raw_bytes(&bytes).unwrap();
5205            file.close().unwrap();
5206        }
5207        {
5208            let file = H5File::open(&path).unwrap();
5209            let ds = file.dataset("records").unwrap();
5210            // The on-disk element type is the compound we specified (size 12),
5211            // not the u8 carrier.
5212            match ds.datatype().unwrap() {
5213                DatatypeMessage::Compound { size, members } => {
5214                    assert_eq!(size, 12);
5215                    assert_eq!(members.len(), 2);
5216                    assert_eq!(members[0].name, "id");
5217                    assert_eq!(members[0].offset, 0);
5218                    assert_eq!(members[1].name, "val");
5219                    assert_eq!(members[1].offset, 4);
5220                }
5221                other => panic!("expected compound datatype, got {other:?}"),
5222            }
5223            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
5224        }
5225        std::fs::remove_file(&path).ok();
5226    }
5227
5228    #[test]
5229    fn builder_requires_shape() {
5230        let path = temp_path("no_shape");
5231        let file = H5File::create(&path).unwrap();
5232        let result = file.new_dataset::<u8>().create("data");
5233        assert!(result.is_err());
5234        std::fs::remove_file(&path).ok();
5235    }
5236
5237    // The last chunk along a *fixed* dimension covers more elements than the
5238    // extent has, so growing the dataspace to the chunk's far edge asks for
5239    // more than the declared maximum. Before the clamp the chunk was written
5240    // and then the call failed on that extend, leaving the bytes in the file
5241    // and the caller an error.
5242    #[test]
5243    fn a_partial_edge_chunk_does_not_grow_past_the_declared_maximum() {
5244        let path = temp_path("edge_chunk_extent");
5245        // Extensible array: dimension 1 is unlimited, dimension 0 is fixed at
5246        // 10 and not a multiple of the chunk's 4.
5247        let file = H5File::create(&path).unwrap();
5248        let ds = file
5249            .new_dataset::<i32>()
5250            .shape([10usize, 4])
5251            .max_shape(&[Some(10), None])
5252            .chunk(&[4, 4])
5253            .create("grid")
5254            .unwrap();
5255        let chunk: Vec<u8> = (0i32..16).flat_map(|v| v.to_le_bytes()).collect();
5256        // Chunk row 2 spans elements 8..12 of a dimension that stops at 10.
5257        ds.write_chunk_at(&[2, 0], &chunk).unwrap();
5258        assert_eq!(ds.shape(), vec![10, 4]);
5259        file.close().unwrap();
5260
5261        let file = H5File::open(&path).unwrap();
5262        let back = file.dataset("grid").unwrap().read_raw::<i32>().unwrap();
5263        assert_eq!(back.len(), 40);
5264        assert_eq!(&back[32..40], &[0, 1, 2, 3, 4, 5, 6, 7]);
5265        drop(file);
5266        std::fs::remove_file(&path).ok();
5267    }
5268
5269    // The cap is in `write_chunk_at_inner`, so it belongs to every chunk index
5270    // whose write reaches the extend below it — the v2 B-tree as much as the
5271    // extensible array. A rank-3 dataset with two unlimited dimensions gets
5272    // that index, and its third, fixed dimension is where the last chunk
5273    // overhangs. (The fixed array and the implicit index return before the
5274    // extend: their shape cannot grow at all. The version-1 B-tree does reach
5275    // it — `tests/legacy_append.rs` carries that case, which needs a classic
5276    // file.)
5277    #[test]
5278    fn the_edge_write_cap_holds_for_the_v2_btree_index() {
5279        let path = temp_path("edge_chunk_bt2");
5280        let file = H5File::create(&path).unwrap();
5281        let ds = file
5282            .new_dataset::<i32>()
5283            .shape([10usize, 4, 4])
5284            .max_shape(&[Some(10), None, None])
5285            .chunk(&[4, 4, 4])
5286            .create("cube")
5287            .unwrap();
5288        let chunk: Vec<u8> = (0i32..64).flat_map(|v| v.to_le_bytes()).collect();
5289        // Chunk plane 2 spans elements 8..12 of a dimension that stops at 10.
5290        ds.write_chunk_at(&[2, 0, 0], &chunk).unwrap();
5291        assert_eq!(ds.shape(), vec![10, 4, 4]);
5292        file.close().unwrap();
5293
5294        let file = H5File::open(&path).unwrap();
5295        let back = file.dataset("cube").unwrap().read_raw::<i32>().unwrap();
5296        assert_eq!(back.len(), 160);
5297        // Rows 8 and 9 of the written plane, 16 elements each.
5298        assert_eq!(&back[128..160], &(0i32..32).collect::<Vec<_>>()[..]);
5299        drop(file);
5300        std::fs::remove_file(&path).ok();
5301    }
5302
5303    // The cap guards a chunk-coordinate write, and neither an externally
5304    // stored nor a virtual dataset has chunk coordinates to guard: both are
5305    // contiguous storage classes, refused at build together with chunked
5306    // storage, and `write_chunk_at` refuses what is not chunked. So the path
5307    // the case above exercises cannot be entered for either — asserted here
5308    // rather than left to inspection, since both classes route their raw bytes
5309    // through the same writer as the chunk grid does.
5310    #[test]
5311    fn an_external_or_virtual_dataset_never_reaches_the_edge_write_cap() {
5312        use crate::Selection;
5313        let dir = std::env::temp_dir().join(format!(
5314            "rust_hdf5_edge_cap_{}_{}",
5315            std::process::id(),
5316            temp_path("x").file_name().unwrap().to_string_lossy()
5317        ));
5318        std::fs::create_dir_all(&dir).unwrap();
5319        let path = dir.join("edge_cap.h5");
5320        let file = H5File::create(&path).unwrap();
5321        let payload = dir.join("payload.raw");
5322
5323        // Chunked storage and these two are mutually exclusive at build.
5324        for (which, res) in [
5325            (
5326                "external",
5327                file.new_dataset::<i32>()
5328                    .shape([10usize])
5329                    .external(&[(payload.to_str().unwrap(), 0, 40)])
5330                    .chunk(&[4])
5331                    .create("a"),
5332            ),
5333            (
5334                "virtual",
5335                file.new_dataset::<i32>()
5336                    .shape([10usize])
5337                    .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5338                    .chunk(&[4])
5339                    .create("b"),
5340            ),
5341        ] {
5342            match res {
5343                Ok(_) => panic!("a {which} dataset cannot also be chunked"),
5344                Err(e) => assert!(e.to_string().contains("chunked"), "{which}: {e}"),
5345            }
5346        }
5347
5348        // And the coordinate write itself is refused on both, with the extent
5349        // left exactly where it was.
5350        let ext = file
5351            .new_dataset::<i32>()
5352            .shape([10usize])
5353            .external(&[(payload.to_str().unwrap(), 0, 40)])
5354            .create("outside")
5355            .unwrap();
5356        let vds = file
5357            .new_dataset::<i32>()
5358            .shape([10usize])
5359            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5360            .create("elsewhere")
5361            .unwrap();
5362        let chunk: Vec<u8> = (0i32..4).flat_map(|v| v.to_le_bytes()).collect();
5363        for (which, ds) in [("external", &ext), ("virtual", &vds)] {
5364            let err = ds.write_chunk_at(&[2], &chunk).unwrap_err().to_string();
5365            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5366            let err = ds
5367                .write_chunk_raw_at(&[2], &chunk, 0)
5368                .unwrap_err()
5369                .to_string();
5370            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5371            assert_eq!(ds.shape(), vec![10], "{which}");
5372        }
5373        file.close().unwrap();
5374        std::fs::remove_dir_all(&dir).ok();
5375    }
5376
5377    #[test]
5378    fn write_raw_size_mismatch() {
5379        let path = temp_path("size_mismatch");
5380        let file = H5File::create(&path).unwrap();
5381        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
5382        // Provide 3 elements instead of 4
5383        let result = ds.write_raw(&[1u8, 2, 3]);
5384        assert!(result.is_err());
5385        std::fs::remove_file(&path).ok();
5386    }
5387
5388    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
5389    // dataset = one chunk) rather than silently drop the filter on the
5390    // contiguous path. write_raw then populates that single chunk.
5391    #[cfg(feature = "deflate")]
5392    #[test]
5393    fn filter_without_chunk_autochunks_and_roundtrips() {
5394        let path = temp_path("autochunk_filter");
5395        let data: Vec<i32> = (0..8).collect();
5396        {
5397            let file = H5File::create(&path).unwrap();
5398            let ds = file
5399                .new_dataset::<i32>()
5400                .deflate(6)
5401                .shape([8])
5402                .create("seq")
5403                .unwrap();
5404            ds.write_raw(&data).unwrap();
5405            file.close().unwrap();
5406        }
5407        {
5408            let file = H5File::open(&path).unwrap();
5409            let ds = file.dataset("seq").unwrap();
5410            // The filter forced chunked storage: a single whole-dataset chunk.
5411            assert!(
5412                ds.is_chunked(),
5413                "auto-chunk did not produce chunked storage"
5414            );
5415            assert_eq!(ds.chunk_dims(), Some(vec![8]));
5416            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5417        }
5418        std::fs::remove_file(&path).ok();
5419    }
5420
5421    // B: write_raw on an explicitly chunked + compressed dataset scatters the
5422    // full row-major image across a multi-chunk grid, including edge chunks
5423    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
5424    #[cfg(feature = "deflate")]
5425    #[test]
5426    fn an_edge_chunk_pads_with_zero_not_the_chunk_written_before_it() {
5427        // 4x5 over a 2x3 chunk: the second chunk of each row covers only two
5428        // of its three columns, so a third of it is padding. The image write
5429        // stages every chunk of a shape like this through one reused buffer,
5430        // and the padding has to reach the file as zero rather than as
5431        // whatever the chunk before it left in that buffer.
5432        let path = temp_path("edge_chunk_padding");
5433        let data: Vec<i32> = (1..=20).collect(); // no zeros of its own
5434        {
5435            let file = H5File::create(&path).unwrap();
5436            let ds = file
5437                .new_dataset::<i32>()
5438                .shape([4, 5])
5439                .chunk(&[2, 3])
5440                .create("grid")
5441                .unwrap();
5442            ds.write_raw(&data).unwrap();
5443            file.close().unwrap();
5444        }
5445        {
5446            let file = H5File::open(&path).unwrap();
5447            let ds = file.dataset("grid").unwrap();
5448            let as_i32 = |bytes: Vec<u8>| -> Vec<i32> {
5449                bytes
5450                    .chunks_exact(4)
5451                    .map(|b| i32::from_le_bytes(b.try_into().unwrap()))
5452                    .collect()
5453            };
5454            // The chunk that precedes each edge chunk is full, so a leak would
5455            // show as its 3rd and 6th elements (3 and 8, then 13 and 18).
5456            assert_eq!(
5457                as_i32(ds.read_chunk_raw_at(&[0, 0]).unwrap().0),
5458                [1, 2, 3, 6, 7, 8]
5459            );
5460            assert_eq!(
5461                as_i32(ds.read_chunk_raw_at(&[0, 1]).unwrap().0),
5462                [4, 5, 0, 9, 10, 0]
5463            );
5464            assert_eq!(
5465                as_i32(ds.read_chunk_raw_at(&[1, 1]).unwrap().0),
5466                [14, 15, 0, 19, 20, 0]
5467            );
5468            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5469        }
5470        std::fs::remove_file(&path).ok();
5471    }
5472
5473    #[test]
5474    #[cfg(feature = "deflate")]
5475    fn write_raw_multichunk_edge_roundtrips() {
5476        let path = temp_path("multichunk_edge");
5477        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
5478        {
5479            let file = H5File::create(&path).unwrap();
5480            let ds = file
5481                .new_dataset::<i32>()
5482                .shape([7, 5])
5483                .chunk(&[3, 2])
5484                .deflate(4)
5485                .create("grid")
5486                .unwrap();
5487            ds.write_raw(&data).unwrap();
5488            file.close().unwrap();
5489        }
5490        {
5491            let file = H5File::open(&path).unwrap();
5492            let ds = file.dataset("grid").unwrap();
5493            assert_eq!(ds.shape(), vec![7, 5]);
5494            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5495            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5496        }
5497        std::fs::remove_file(&path).ok();
5498    }
5499
5500    // B: write_raw on an unfiltered chunked dataset (previously rejected with
5501    // "use write_chunk for chunked datasets") now gathers and round-trips.
5502    #[test]
5503    fn write_raw_unfiltered_chunked_roundtrips() {
5504        let path = temp_path("chunked_unfiltered");
5505        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
5506        {
5507            let file = H5File::create(&path).unwrap();
5508            let ds = file
5509                .new_dataset::<f64>()
5510                .shape([4, 3])
5511                .chunk(&[2, 2])
5512                .create("m")
5513                .unwrap();
5514            ds.write_raw(&data).unwrap();
5515            file.close().unwrap();
5516        }
5517        {
5518            let file = H5File::open(&path).unwrap();
5519            let ds = file.dataset("m").unwrap();
5520            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
5521            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
5522        }
5523        std::fs::remove_file(&path).ok();
5524    }
5525
5526    // write_raw on an extensible-array (unlimited first dim) compressed dataset
5527    // drives write_full_image_chunked's EA branch, which gathers chunks and
5528    // compresses them through the windowed batch path. Round-trips the full
5529    // image, including a partial edge chunk along the unlimited dimension.
5530    #[cfg(feature = "deflate")]
5531    #[test]
5532    fn write_raw_ea_compressed_roundtrips() {
5533        let path = temp_path("write_raw_ea_deflate");
5534        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
5535        {
5536            let file = H5File::create(&path).unwrap();
5537            let ds = file
5538                .new_dataset::<i32>()
5539                .shape([5, 4])
5540                .chunk(&[2, 4])
5541                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
5542                .deflate(5)
5543                .create("stream")
5544                .unwrap();
5545            ds.write_raw(&data).unwrap();
5546            file.close().unwrap();
5547        }
5548        {
5549            let file = H5File::open(&path).unwrap();
5550            let ds = file.dataset("stream").unwrap();
5551            assert_eq!(ds.shape(), vec![5, 4]);
5552            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5553        }
5554        std::fs::remove_file(&path).ok();
5555    }
5556
5557    // 3D chunked Full read with partial-edge chunks in every dimension. This
5558    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
5559    // dimensions, exercising the nested outer-coordinate carry and the
5560    // last-axis edge clamp (chunks hang off the high edge in all three axes).
5561    #[cfg(feature = "deflate")]
5562    #[test]
5563    fn read_full_3d_chunked_edge_roundtrips() {
5564        let path = temp_path("full_3d_chunked_edge");
5565        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
5566        // along each axis is partial (1, 1, and 1 element respectively).
5567        let total: usize = 5 * 4 * 3;
5568        let data: Vec<i32> = (0..total as i32).collect();
5569        {
5570            let file = H5File::create(&path).unwrap();
5571            let ds = file
5572                .new_dataset::<i32>()
5573                .shape([5, 4, 3])
5574                .chunk(&[2, 3, 2])
5575                .deflate(4)
5576                .create("vol")
5577                .unwrap();
5578            ds.write_raw(&data).unwrap();
5579            file.close().unwrap();
5580        }
5581        {
5582            let file = H5File::open(&path).unwrap();
5583            let ds = file.dataset("vol").unwrap();
5584            assert_eq!(ds.shape(), vec![5, 4, 3]);
5585            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
5586            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5587        }
5588        std::fs::remove_file(&path).ok();
5589    }
5590
5591    #[test]
5592    fn roundtrip_u8_1d() {
5593        let path = temp_path("rt_u8_1d");
5594        let data: Vec<u8> = (0..10).collect();
5595
5596        {
5597            let file = H5File::create(&path).unwrap();
5598            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
5599            ds.write_raw(&data).unwrap();
5600            file.close().unwrap();
5601        }
5602
5603        {
5604            let file = H5File::open(&path).unwrap();
5605            let ds = file.dataset("seq").unwrap();
5606            assert_eq!(ds.shape(), vec![10]);
5607            let readback = ds.read_raw::<u8>().unwrap();
5608            assert_eq!(readback, data);
5609        }
5610
5611        std::fs::remove_file(&path).ok();
5612    }
5613
5614    #[test]
5615    fn roundtrip_i32_2d() {
5616        let path = temp_path("rt_i32_2d");
5617        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
5618
5619        {
5620            let file = H5File::create(&path).unwrap();
5621            let ds = file
5622                .new_dataset::<i32>()
5623                .shape([2, 3])
5624                .create("matrix")
5625                .unwrap();
5626            ds.write_raw(&data).unwrap();
5627            file.close().unwrap();
5628        }
5629
5630        {
5631            let file = H5File::open(&path).unwrap();
5632            let ds = file.dataset("matrix").unwrap();
5633            assert_eq!(ds.shape(), vec![2, 3]);
5634            let readback = ds.read_raw::<i32>().unwrap();
5635            assert_eq!(readback, data);
5636        }
5637
5638        std::fs::remove_file(&path).ok();
5639    }
5640
5641    #[test]
5642    fn roundtrip_f64_3d() {
5643        let path = temp_path("rt_f64_3d");
5644        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
5645
5646        {
5647            let file = H5File::create(&path).unwrap();
5648            let ds = file
5649                .new_dataset::<f64>()
5650                .shape([2, 3, 4])
5651                .create("cube")
5652                .unwrap();
5653            ds.write_raw(&data).unwrap();
5654            file.close().unwrap();
5655        }
5656
5657        {
5658            let file = H5File::open(&path).unwrap();
5659            let ds = file.dataset("cube").unwrap();
5660            assert_eq!(ds.shape(), vec![2, 3, 4]);
5661            let readback = ds.read_raw::<f64>().unwrap();
5662            assert_eq!(readback, data);
5663        }
5664
5665        std::fs::remove_file(&path).ok();
5666    }
5667
5668    #[test]
5669    fn cannot_read_in_write_mode() {
5670        let path = temp_path("no_read_write");
5671        let file = H5File::create(&path).unwrap();
5672        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5673        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5674        let result = ds.read_raw::<u8>();
5675        assert!(result.is_err());
5676        std::fs::remove_file(&path).ok();
5677    }
5678
5679    #[test]
5680    fn cannot_write_in_read_mode() {
5681        let path = temp_path("no_write_read");
5682
5683        {
5684            let file = H5File::create(&path).unwrap();
5685            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5686            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5687            file.close().unwrap();
5688        }
5689
5690        {
5691            let file = H5File::open(&path).unwrap();
5692            let ds = file.dataset("x").unwrap();
5693            let result = ds.write_raw(&[5u8, 6, 7, 8]);
5694            assert!(result.is_err());
5695        }
5696
5697        std::fs::remove_file(&path).ok();
5698    }
5699
5700    #[test]
5701    fn numeric_attr_roundtrip() {
5702        let path = temp_path("num_attr");
5703        {
5704            let file = H5File::create(&path).unwrap();
5705            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5706            ds.write_raw(&[1.0f32; 4]).unwrap();
5707
5708            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5709            a1.write_numeric(&1.2345f64).unwrap();
5710
5711            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
5712            a2.write_numeric(&42i32).unwrap();
5713
5714            file.close().unwrap();
5715        }
5716        {
5717            let file = H5File::open(&path).unwrap();
5718            let ds = file.dataset("data").unwrap();
5719
5720            let scale = ds.attr("scale").unwrap();
5721            let val: f64 = scale.read_numeric().unwrap();
5722            assert!((val - 1.2345).abs() < 1e-10);
5723
5724            let count = ds.attr("count").unwrap();
5725            let val: i32 = count.read_numeric().unwrap();
5726            assert_eq!(val, 42);
5727        }
5728        std::fs::remove_file(&path).ok();
5729    }
5730
5731    #[test]
5732    fn array_attr_roundtrip() {
5733        let path = temp_path("array_attr");
5734        let offsets = [10i32, -20, 30];
5735        {
5736            let file = H5File::create(&path).unwrap();
5737            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5738            ds.write_raw(&[1.0f32; 4]).unwrap();
5739
5740            // 1-D int32 array attribute (NDArrayDimOffset-style).
5741            let a = ds
5742                .new_attr::<i32>()
5743                .shape([3])
5744                .create("dim_offset")
5745                .unwrap();
5746            a.write_array(&offsets).unwrap();
5747
5748            // Wrong element count is rejected.
5749            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
5750            assert!(bad.write_array(&[1i32, 2]).is_err());
5751
5752            file.close().unwrap();
5753        }
5754        {
5755            let file = H5File::open(&path).unwrap();
5756            let ds = file.dataset("data").unwrap();
5757            let a = ds.attr("dim_offset").unwrap();
5758            let raw = a.read_raw().unwrap();
5759            assert_eq!(raw.len(), 3 * 4);
5760            let got: Vec<i32> = raw
5761                .chunks_exact(4)
5762                .map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
5763                .collect();
5764            assert_eq!(got, offsets);
5765        }
5766        std::fs::remove_file(&path).ok();
5767    }
5768
5769    #[test]
5770    fn attr_datatype_exposes_class_and_sign() {
5771        // H5Attribute::datatype() must report the stored datatype class and
5772        // signedness so a generic attr->metadata mapper need not infer it from
5773        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
5774        use crate::format::messages::datatype::DatatypeMessage;
5775
5776        let path = temp_path("attr_datatype");
5777        {
5778            let file = H5File::create(&path).unwrap();
5779            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5780            ds.new_attr::<f64>()
5781                .shape(())
5782                .create("scale")
5783                .unwrap()
5784                .write_numeric(&1.5f64)
5785                .unwrap();
5786            ds.new_attr::<i32>()
5787                .shape(())
5788                .create("count")
5789                .unwrap()
5790                .write_numeric(&7i32)
5791                .unwrap();
5792            file.close().unwrap();
5793        }
5794        {
5795            let file = H5File::open(&path).unwrap();
5796            let ds = file.dataset("data").unwrap();
5797
5798            match ds.attr("scale").unwrap().datatype().unwrap() {
5799                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
5800                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
5801            }
5802
5803            match ds.attr("count").unwrap().datatype().unwrap() {
5804                DatatypeMessage::FixedPoint { size, signed, .. } => {
5805                    assert_eq!(size, 4);
5806                    assert!(signed, "i32 attr must be signed");
5807                }
5808                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
5809            }
5810        }
5811        std::fs::remove_file(&path).ok();
5812    }
5813
5814    #[test]
5815    fn attr_datatype_in_write_mode_errors() {
5816        let path = temp_path("attr_datatype_write_mode");
5817        let file = H5File::create(&path).unwrap();
5818        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5819        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5820        assert!(attr.datatype().is_err());
5821        std::fs::remove_file(&path).ok();
5822    }
5823
5824    #[test]
5825    fn cannot_create_dataset_in_read_mode() {
5826        let path = temp_path("no_create_read");
5827
5828        {
5829            let _file = H5File::create(&path).unwrap();
5830        }
5831
5832        {
5833            let file = H5File::open(&path).unwrap();
5834            let result = file.new_dataset::<u8>().shape([4]).create("x");
5835            assert!(result.is_err());
5836        }
5837
5838        std::fs::remove_file(&path).ok();
5839    }
5840
5841    #[test]
5842    fn shape_accessor() {
5843        let path = temp_path("shape_acc");
5844
5845        let file = H5File::create(&path).unwrap();
5846        let ds = file
5847            .new_dataset::<f32>()
5848            .shape([5, 10, 3])
5849            .create("tensor")
5850            .unwrap();
5851        assert_eq!(ds.shape(), vec![5, 10, 3]);
5852
5853        std::fs::remove_file(&path).ok();
5854    }
5855
5856    #[test]
5857    fn slice_roundtrip_2d() {
5858        let path = temp_path("slice_2d");
5859
5860        // Create a 4x5 dataset, write full, then read a slice
5861        let data: Vec<i32> = (0..20).collect();
5862        {
5863            let file = H5File::create(&path).unwrap();
5864            let ds = file
5865                .new_dataset::<i32>()
5866                .shape([4, 5])
5867                .create("mat")
5868                .unwrap();
5869            ds.write_raw(&data).unwrap();
5870            file.close().unwrap();
5871        }
5872        {
5873            let file = H5File::open(&path).unwrap();
5874            let ds = file.dataset("mat").unwrap();
5875            // Read rows 1..3, cols 2..4 (2x2 slice)
5876            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
5877            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
5878            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
5879            assert_eq!(slice, vec![7, 8, 12, 13]);
5880        }
5881
5882        std::fs::remove_file(&path).ok();
5883    }
5884
5885    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
5886    // caller-provided buffer and MUST produce byte-for-byte the same data as
5887    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
5888    // creatable layout, since both now share one buffer-filling core.
5889    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
5890    where
5891        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
5892    {
5893        let n: usize = ds.shape().iter().product();
5894        let want_full = ds.read_raw::<T>().unwrap();
5895        let mut got_full = vec![T::default(); n];
5896        ds.read_raw_into::<T>(&mut got_full).unwrap();
5897        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
5898
5899        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
5900        let sn: usize = counts.iter().product();
5901        let mut got_slice = vec![T::default(); sn];
5902        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
5903            .unwrap();
5904        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
5905    }
5906
5907    #[test]
5908    fn read_into_matches_vec_contiguous() {
5909        let path = temp_path("into_contig");
5910        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
5911        {
5912            let file = H5File::create(&path).unwrap();
5913            let ds = file
5914                .new_dataset::<i32>()
5915                .shape([4, 5])
5916                .create("mat")
5917                .unwrap();
5918            ds.write_raw(&data).unwrap();
5919            file.close().unwrap();
5920        }
5921        {
5922            let file = H5File::open(&path).unwrap();
5923            let ds = file.dataset("mat").unwrap();
5924            assert_eq!(ds.chunk_dims(), None);
5925            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
5926        }
5927        std::fs::remove_file(&path).ok();
5928    }
5929
5930    #[test]
5931    fn read_into_matches_vec_chunked_unfiltered() {
5932        let path = temp_path("into_chunk");
5933        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
5934        {
5935            let file = H5File::create(&path).unwrap();
5936            let ds = file
5937                .new_dataset::<f64>()
5938                .shape([7, 5])
5939                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
5940                .create("grid")
5941                .unwrap();
5942            ds.write_raw(&data).unwrap();
5943            file.close().unwrap();
5944        }
5945        {
5946            let file = H5File::open(&path).unwrap();
5947            let ds = file.dataset("grid").unwrap();
5948            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5949            // Slice spans multiple chunks (rows 2..5, cols 1..4).
5950            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
5951        }
5952        std::fs::remove_file(&path).ok();
5953    }
5954
5955    #[test]
5956    fn read_into_matches_vec_single_chunk() {
5957        let path = temp_path("into_single_chunk");
5958        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
5959        {
5960            let file = H5File::create(&path).unwrap();
5961            let ds = file
5962                .new_dataset::<i32>()
5963                .shape([3, 4])
5964                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
5965                .create("g")
5966                .unwrap();
5967            ds.write_raw(&data).unwrap();
5968            file.close().unwrap();
5969        }
5970        {
5971            let file = H5File::open(&path).unwrap();
5972            let ds = file.dataset("g").unwrap();
5973            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
5974            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
5975        }
5976        std::fs::remove_file(&path).ok();
5977    }
5978
5979    #[cfg(feature = "deflate")]
5980    #[test]
5981    fn read_into_matches_vec_chunked_deflate() {
5982        let path = temp_path("into_chunk_deflate");
5983        let data: Vec<i32> = (0..35).collect(); // 7 x 5
5984        {
5985            let file = H5File::create(&path).unwrap();
5986            let ds = file
5987                .new_dataset::<i32>()
5988                .shape([7, 5])
5989                .chunk(&[3, 2])
5990                .deflate(4)
5991                .create("grid")
5992                .unwrap();
5993            ds.write_raw(&data).unwrap();
5994            file.close().unwrap();
5995        }
5996        {
5997            let file = H5File::open(&path).unwrap();
5998            let ds = file.dataset("grid").unwrap();
5999            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6000            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
6001        }
6002        std::fs::remove_file(&path).ok();
6003    }
6004
6005    #[test]
6006    fn read_into_wrong_buffer_size_rejected() {
6007        let path = temp_path("into_badlen");
6008        let data: Vec<i32> = (0..20).collect(); // 4 x 5
6009        {
6010            let file = H5File::create(&path).unwrap();
6011            let ds = file
6012                .new_dataset::<i32>()
6013                .shape([4, 5])
6014                .create("mat")
6015                .unwrap();
6016            ds.write_raw(&data).unwrap();
6017            file.close().unwrap();
6018        }
6019        {
6020            let file = H5File::open(&path).unwrap();
6021            let ds = file.dataset("mat").unwrap();
6022
6023            // Too small / too large full-read buffers are both rejected.
6024            let mut small = vec![0i32; 19];
6025            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
6026            let mut large = vec![0i32; 21];
6027            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
6028
6029            // Slice buffer must be exactly product(counts) = 4.
6030            let mut bad_slice = vec![0i32; 3];
6031            assert!(ds
6032                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
6033                .is_err());
6034            // The correctly sized slice buffer succeeds.
6035            let mut ok_slice = vec![0i32; 4];
6036            assert!(ds
6037                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
6038                .is_ok());
6039        }
6040        std::fs::remove_file(&path).ok();
6041    }
6042
6043    #[test]
6044    fn read_into_wrong_element_size_rejected() {
6045        let path = temp_path("into_badtype");
6046        let data: Vec<i32> = (0..20).collect(); // element size 4
6047        {
6048            let file = H5File::create(&path).unwrap();
6049            let ds = file
6050                .new_dataset::<i32>()
6051                .shape([4, 5])
6052                .create("mat")
6053                .unwrap();
6054            ds.write_raw(&data).unwrap();
6055            file.close().unwrap();
6056        }
6057        {
6058            let file = H5File::open(&path).unwrap();
6059            let ds = file.dataset("mat").unwrap();
6060            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
6061            // element size -> TypeMismatch, even with a "correctly sized" Vec.
6062            let mut as_u8 = vec![0u8; 20];
6063            assert!(matches!(
6064                ds.read_raw_into::<u8>(&mut as_u8),
6065                Err(crate::Hdf5Error::TypeMismatch(_))
6066            ));
6067            let mut as_i64 = vec![0i64; 20];
6068            assert!(matches!(
6069                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
6070                Err(crate::Hdf5Error::TypeMismatch(_))
6071            ));
6072        }
6073        std::fs::remove_file(&path).ok();
6074    }
6075
6076    #[test]
6077    fn write_slice_2d() {
6078        let path = temp_path("write_slice_2d");
6079
6080        {
6081            let file = H5File::create(&path).unwrap();
6082            let ds = file
6083                .new_dataset::<f32>()
6084                .shape([3, 4])
6085                .create("data")
6086                .unwrap();
6087            ds.write_raw(&[0.0f32; 12]).unwrap();
6088            // Overwrite a 2x2 sub-region
6089            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
6090                .unwrap();
6091            file.close().unwrap();
6092        }
6093        {
6094            let file = H5File::open(&path).unwrap();
6095            let ds = file.dataset("data").unwrap();
6096            let full = ds.read_raw::<f32>().unwrap();
6097            // Row 0: [0,0,0,0]
6098            // Row 1: [0,10,20,0]
6099            // Row 2: [0,30,40,0]
6100            assert_eq!(
6101                full,
6102                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
6103            );
6104        }
6105
6106        std::fs::remove_file(&path).ok();
6107    }
6108
6109    /// One 2x4 i32 chunk whose every element is `v`.
6110    fn chunk_of(v: i32) -> Vec<u8> {
6111        (0..8).flat_map(|_| v.to_le_bytes()).collect()
6112    }
6113
6114    /// Write chunk (0,0) `rewrites` times — each time with a different value,
6115    /// so no write can be skipped — and return the closed file's size along
6116    /// with what the chunk reads back as.
6117    fn rewrite_chunk(
6118        tag: &str,
6119        rewrites: i32,
6120        build: impl Fn(&H5File) -> crate::H5Dataset,
6121    ) -> (u64, i32) {
6122        let path = temp_path(tag);
6123        {
6124            let file = H5File::create(&path).unwrap();
6125            let ds = build(&file);
6126            for v in 1..=rewrites {
6127                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
6128            }
6129            file.close().unwrap();
6130        }
6131        let size = std::fs::metadata(&path).unwrap().len();
6132        let first = {
6133            let file = H5File::open(&path).unwrap();
6134            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
6135        };
6136        std::fs::remove_file(&path).ok();
6137        (size, first)
6138    }
6139
6140    // An unfiltered chunk's stored size is fixed by the chunk shape, so
6141    // rewriting it must overwrite the block it already occupies rather than
6142    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
6143    // leaves must_alloc false for exactly this case). The file must therefore
6144    // be byte-identical in size no matter how many times the chunk is written.
6145    #[test]
6146    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
6147        let build = |f: &H5File| {
6148            f.new_dataset::<i32>()
6149                .shape([2, 4])
6150                .chunk(&[2, 4])
6151                .max_shape(&[None, Some(4)])
6152                .create("d")
6153                .unwrap()
6154        };
6155        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
6156        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
6157        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6158        assert_eq!(last, 8, "the last write must be the one that survives");
6159    }
6160
6161    #[test]
6162    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
6163        let build = |f: &H5File| {
6164            f.new_dataset::<i32>()
6165                .shape([2, 4])
6166                .chunk(&[2, 4])
6167                .create("d")
6168                .unwrap()
6169        };
6170        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
6171        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
6172        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6173        assert_eq!(last, 8);
6174    }
6175
6176    #[test]
6177    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
6178        let build = |f: &H5File| {
6179            f.new_dataset::<i32>()
6180                .shape([2, 4])
6181                .chunk(&[2, 4])
6182                .max_shape(&[None, None])
6183                .create("d")
6184                .unwrap()
6185        };
6186        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
6187        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
6188        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6189        assert_eq!(last, 8);
6190    }
6191
6192    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
6193    // pool. Every node is the same size, so the blocks already on disk are
6194    // reused and repeated flushes cost nothing; sizing the root to its record
6195    // count instead would relocate it each time and orphan the block it left.
6196    #[test]
6197    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
6198        let flush_n = |label: &str, flushes: usize| -> u64 {
6199            let path = temp_path(label);
6200            {
6201                let file = H5File::create(&path).unwrap();
6202                let ds = file
6203                    .new_dataset::<i32>()
6204                    .shape([2, 4])
6205                    .chunk(&[2, 4])
6206                    .max_shape(&[None, None])
6207                    .create("d")
6208                    .unwrap();
6209                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
6210                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
6211                for _ in 0..flushes {
6212                    ds.flush().unwrap();
6213                }
6214                file.close().unwrap();
6215            }
6216            let size = std::fs::metadata(&path).unwrap().len();
6217            // The data must survive every rewrite of the index.
6218            {
6219                let file = H5File::open(&path).unwrap();
6220                let ds = file.dataset("d").unwrap();
6221                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
6222            }
6223            std::fs::remove_file(&path).ok();
6224            size
6225        };
6226        assert_eq!(
6227            flush_n("bt2_flush_8", 8),
6228            flush_n("bt2_flush_1", 1),
6229            "8 index flushes grew the file past a single one"
6230        );
6231    }
6232
6233    // A filtered chunk whose compressed size changes cannot stay put, so it
6234    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
6235    // H5MF_xfree). Alternating between two payloads of different compressed
6236    // size must therefore keep reusing the same two blocks instead of
6237    // appending a fresh one each time.
6238    #[cfg(feature = "deflate")]
6239    #[test]
6240    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
6241        // All-equal elements deflate to far fewer bytes than a varied payload,
6242        // so the two writes below land at different stored sizes.
6243        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
6244        let varied: Vec<u8> = (0..8i32)
6245            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
6246            .collect();
6247
6248        let sizes: Vec<u64> = [1usize, 8]
6249            .iter()
6250            .map(|&rounds| {
6251                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
6252                {
6253                    let file = H5File::create(&path).unwrap();
6254                    let ds = file
6255                        .new_dataset::<i32>()
6256                        .shape([2, 4])
6257                        .chunk(&[2, 4])
6258                        .max_shape(&[None, Some(4)])
6259                        .deflate(6)
6260                        .create("d")
6261                        .unwrap();
6262                    for _ in 0..rounds {
6263                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
6264                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
6265                    }
6266                    file.close().unwrap();
6267                }
6268                let size = std::fs::metadata(&path).unwrap().len();
6269                {
6270                    let file = H5File::open(&path).unwrap();
6271                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
6272                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
6273                    assert_eq!(got, want, "the last write must survive the round trip");
6274                }
6275                std::fs::remove_file(&path).ok();
6276                size
6277            })
6278            .collect();
6279
6280        assert_eq!(
6281            sizes[1], sizes[0],
6282            "8 alternating rewrites grew the file past a single pair"
6283        );
6284    }
6285
6286    #[test]
6287    fn write_slice_out_of_bounds_rejected() {
6288        let path = temp_path("write_slice_oob");
6289        let file = H5File::create(&path).unwrap();
6290        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
6291        ds.write_raw(&[0i32; 4]).unwrap();
6292        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
6293        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
6294        // An in-bounds slice still works.
6295        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
6296        std::fs::remove_file(&path).ok();
6297    }
6298
6299    #[test]
6300    fn duplicate_dataset_name_rejected() {
6301        let path = temp_path("dup_name");
6302        let file = H5File::create(&path).unwrap();
6303        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
6304        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
6305        std::fs::remove_file(&path).ok();
6306    }
6307
6308    #[test]
6309    fn extend_cannot_shrink() {
6310        let path = temp_path("extend_shrink");
6311        let file = H5File::create(&path).unwrap();
6312        let ds = file
6313            .new_dataset::<i32>()
6314            .shape([0])
6315            .chunk(&[2])
6316            .max_shape(&[None])
6317            .create("d")
6318            .unwrap();
6319        ds.append(&[1i32, 2, 3, 4]).unwrap();
6320        // Shrinking below the written extent must be rejected.
6321        assert!(ds.extend(&[2]).is_err());
6322        // Growing is fine.
6323        assert!(ds.extend(&[6]).is_ok());
6324        std::fs::remove_file(&path).ok();
6325    }
6326
6327    #[test]
6328    fn attr_read_roundtrip() {
6329        use crate::types::VarLenUnicode;
6330        let path = temp_path("attr_read");
6331
6332        {
6333            let file = H5File::create(&path).unwrap();
6334            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6335            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6336            let a1 = ds
6337                .new_attr::<VarLenUnicode>()
6338                .shape(())
6339                .create("units")
6340                .unwrap();
6341            a1.write_string("meters").unwrap();
6342            let a2 = ds
6343                .new_attr::<VarLenUnicode>()
6344                .shape(())
6345                .create("desc")
6346                .unwrap();
6347            a2.write_string("test data").unwrap();
6348            file.close().unwrap();
6349        }
6350        {
6351            let file = H5File::open(&path).unwrap();
6352            let ds = file.dataset("data").unwrap();
6353
6354            let names = ds.attr_names().unwrap();
6355            assert!(names.contains(&"units".to_string()));
6356            assert!(names.contains(&"desc".to_string()));
6357
6358            let units = ds.attr("units").unwrap();
6359            assert_eq!(units.read_string().unwrap(), "meters");
6360
6361            let desc = ds.attr("desc").unwrap();
6362            assert_eq!(desc.read_string().unwrap(), "test data");
6363        }
6364
6365        std::fs::remove_file(&path).ok();
6366    }
6367
6368    #[test]
6369    fn type_mismatch_element_size() {
6370        let path = temp_path("type_mismatch");
6371
6372        {
6373            let file = H5File::create(&path).unwrap();
6374            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
6375            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
6376            file.close().unwrap();
6377        }
6378
6379        {
6380            let file = H5File::open(&path).unwrap();
6381            let ds = file.dataset("data").unwrap();
6382            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
6383            let result = ds.read_raw::<u8>();
6384            assert!(result.is_err());
6385        }
6386
6387        std::fs::remove_file(&path).ok();
6388    }
6389
6390    #[test]
6391    fn dataset_survives_file_move() {
6392        let path = temp_path("ds_survives");
6393
6394        let ds = {
6395            let file = H5File::create(&path).unwrap();
6396            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
6397        };
6398        // file is dropped here, but ds still holds Rc to the inner state
6399        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6400        // The writer will finalize on drop of the last Rc
6401
6402        std::fs::remove_file(&path).ok();
6403    }
6404
6405    #[test]
6406    fn new_attr_scalar_string() {
6407        use crate::types::VarLenUnicode;
6408
6409        let path = temp_path("attr_scalar_string");
6410        {
6411            let file = H5File::create(&path).unwrap();
6412            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6413            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6414
6415            let attr = ds
6416                .new_attr::<VarLenUnicode>()
6417                .shape(())
6418                .create("name")
6419                .unwrap();
6420            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
6421                .unwrap();
6422
6423            file.close().unwrap();
6424        }
6425
6426        // Verify the file is still valid and readable
6427        {
6428            use crate::format::messages::datatype::DatatypeMessage;
6429            let file = H5File::open(&path).unwrap();
6430            let ds = file.dataset("data").unwrap();
6431            assert_eq!(ds.shape(), vec![4]);
6432            let readback = ds.read_raw::<u8>().unwrap();
6433            assert_eq!(readback, vec![1u8, 2, 3, 4]);
6434
6435            // The string attribute is stored as a true variable-length string
6436            // (not fixed-length) and round-trips its value.
6437            let attr = ds.attr("name").unwrap();
6438            assert!(
6439                matches!(
6440                    attr.datatype().unwrap(),
6441                    DatatypeMessage::VarLenString { .. }
6442                ),
6443                "string attribute should have a variable-length string datatype"
6444            );
6445            assert_eq!(attr.read_string().unwrap(), "test_value");
6446        }
6447
6448        std::fs::remove_file(&path).ok();
6449    }
6450
6451    #[test]
6452    fn all_numeric_types_roundtrip() {
6453        let path = temp_path("all_types");
6454
6455        {
6456            let file = H5File::create(&path).unwrap();
6457
6458            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
6459            ds.write_raw(&[1u8, 2]).unwrap();
6460
6461            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
6462            ds.write_raw(&[-1i8, 1]).unwrap();
6463
6464            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
6465            ds.write_raw(&[100u16, 200]).unwrap();
6466
6467            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
6468            ds.write_raw(&[-100i16, 100]).unwrap();
6469
6470            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
6471            ds.write_raw(&[1000u32, 2000]).unwrap();
6472
6473            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
6474            ds.write_raw(&[-1000i32, 1000]).unwrap();
6475
6476            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
6477            ds.write_raw(&[10000u64, 20000]).unwrap();
6478
6479            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
6480            ds.write_raw(&[-10000i64, 10000]).unwrap();
6481
6482            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
6483            ds.write_raw(&[1.5f32, 2.5]).unwrap();
6484
6485            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
6486            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
6487
6488            file.close().unwrap();
6489        }
6490
6491        {
6492            let file = H5File::open(&path).unwrap();
6493
6494            assert_eq!(
6495                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
6496                vec![1u8, 2]
6497            );
6498            assert_eq!(
6499                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
6500                vec![-1i8, 1]
6501            );
6502            assert_eq!(
6503                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
6504                vec![100u16, 200]
6505            );
6506            assert_eq!(
6507                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
6508                vec![-100i16, 100]
6509            );
6510            assert_eq!(
6511                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
6512                vec![1000u32, 2000]
6513            );
6514            assert_eq!(
6515                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
6516                vec![-1000i32, 1000]
6517            );
6518            assert_eq!(
6519                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
6520                vec![10000u64, 20000]
6521            );
6522            assert_eq!(
6523                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
6524                vec![-10000i64, 10000]
6525            );
6526            assert_eq!(
6527                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
6528                vec![1.5f32, 2.5]
6529            );
6530            assert_eq!(
6531                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
6532                vec![1.23456f64, 7.89012]
6533            );
6534        }
6535
6536        std::fs::remove_file(&path).ok();
6537    }
6538
6539    #[test]
6540    fn append_chunked_roundtrip() {
6541        let path = temp_path("append_chunked");
6542
6543        {
6544            let file = H5File::create(&path).unwrap();
6545            let ds = file
6546                .new_dataset::<f64>()
6547                .shape([0, 3])
6548                .chunk(&[1, 3])
6549                .max_shape(&[None, Some(3)])
6550                .create("data")
6551                .unwrap();
6552
6553            // Append one frame
6554            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
6555            // Append two frames at once
6556            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
6557
6558            file.close().unwrap();
6559        }
6560
6561        {
6562            let file = H5File::open(&path).unwrap();
6563            let ds = file.dataset("data").unwrap();
6564            assert_eq!(ds.shape(), vec![3, 3]);
6565            let all = ds.read_raw::<f64>().unwrap();
6566            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
6567        }
6568
6569        std::fs::remove_file(&path).ok();
6570    }
6571
6572    #[test]
6573    fn append_1d_chunked() {
6574        let path = temp_path("append_1d");
6575
6576        {
6577            let file = H5File::create(&path).unwrap();
6578            let ds = file
6579                .new_dataset::<i32>()
6580                .shape([0])
6581                .chunk(&[4])
6582                .max_shape(&[None])
6583                .create("values")
6584                .unwrap();
6585
6586            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
6587            ds.append(&[40i32]).unwrap(); // fills chunk boundary
6588            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
6589
6590            file.close().unwrap();
6591        }
6592
6593        {
6594            let file = H5File::open(&path).unwrap();
6595            let ds = file.dataset("values").unwrap();
6596            assert_eq!(ds.shape(), vec![8]);
6597            let all = ds.read_raw::<i32>().unwrap();
6598            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
6599        }
6600
6601        std::fs::remove_file(&path).ok();
6602    }
6603
6604    #[test]
6605    fn append_partial_chunk_flushed_on_close() {
6606        let path = temp_path("append_partial_close");
6607
6608        {
6609            let file = H5File::create(&path).unwrap();
6610            let ds = file
6611                .new_dataset::<f64>()
6612                .shape([0])
6613                .chunk(&[4])
6614                .max_shape(&[None])
6615                .create("vals")
6616                .unwrap();
6617
6618            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
6619            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
6620            file.close().unwrap();
6621        }
6622
6623        {
6624            let file = H5File::open(&path).unwrap();
6625            let ds = file.dataset("vals").unwrap();
6626            assert_eq!(ds.shape(), vec![5]);
6627            let all = ds.read_raw::<f64>().unwrap();
6628            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
6629            // read_raw reads total shape elements
6630            assert_eq!(all.len(), 5);
6631            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
6632        }
6633
6634        std::fs::remove_file(&path).ok();
6635    }
6636
6637    /// An append that leaves its chunk partial is buffered until close, and
6638    /// the flush has to keep the frames that chunk already holds. It built a
6639    /// fresh fill-value chunk around the buffered frame instead, so reopening
6640    /// a file and appending one row erased every earlier row of that chunk
6641    /// (issue #3). Four sessions: the second lands beside an existing row, the
6642    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
6643    /// the third left in chunk 1.
6644    #[test]
6645    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
6646        let path = temp_path("append_reopen_partial");
6647
6648        {
6649            let file = H5File::create(&path).unwrap();
6650            let ds = file
6651                .new_dataset::<i32>()
6652                .shape([0, 3])
6653                .chunk(&[4, 3])
6654                .max_shape(&[None, Some(3)])
6655                .create("values")
6656                .unwrap();
6657            ds.append(&[1, 2, 3]).unwrap();
6658            file.close().unwrap();
6659        }
6660        for rows in [
6661            vec![4, 5, 6],
6662            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
6663            vec![16, 17, 18],
6664        ] {
6665            let file = H5File::open_rw(&path).unwrap();
6666            file.dataset_writer("values")
6667                .unwrap()
6668                .append(&rows)
6669                .unwrap();
6670            file.close().unwrap();
6671        }
6672
6673        let file = H5File::open(&path).unwrap();
6674        let ds = file.dataset("values").unwrap();
6675        assert_eq!(ds.shape(), vec![6, 3]);
6676        assert_eq!(
6677            ds.read_raw::<i32>().unwrap(),
6678            (1..=18).collect::<Vec<i32>>()
6679        );
6680        std::fs::remove_file(&path).ok();
6681    }
6682
6683    #[cfg(feature = "deflate")]
6684    #[test]
6685    fn vlen_append_after_reopen_filtered() {
6686        // Reopen + append into a partially-written *compressed* vlen chunk
6687        // (index-block chunk). Exercises filtered-index-block reconstruction
6688        // in open_append plus filtered read-modify-write.
6689        let path = temp_path("vlen_reopen_filtered");
6690        {
6691            let file = H5File::create(&path).unwrap();
6692            file.create_appendable_vlen_dataset(
6693                "strs",
6694                4,
6695                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6696            )
6697            .unwrap();
6698            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
6699                .unwrap();
6700            file.close().unwrap();
6701        }
6702        {
6703            let file = H5File::open_rw(&path).unwrap();
6704            file.append_vlen_strings("strs", &["delta"]).unwrap();
6705            file.close().unwrap();
6706        }
6707        {
6708            let file = H5File::open(&path).unwrap();
6709            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6710            assert_eq!(
6711                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
6712                vec!["alpha", "beta", "gamma", "delta"]
6713            );
6714        }
6715        std::fs::remove_file(&path).ok();
6716    }
6717
6718    #[test]
6719    fn vlen_append_after_reopen_data_block() {
6720        // Reopen + append into a partial chunk that lives in an extensible-
6721        // array *data block* (chunk index >= idx_blk_elmts). Exercises
6722        // data-block resolution in read_chunk_if_present and write_chunk.
6723        let path = temp_path("vlen_reopen_datablk");
6724        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
6725        {
6726            let file = H5File::create(&path).unwrap();
6727            file.create_appendable_vlen_dataset("strs", 2, None)
6728                .unwrap();
6729            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6730            file.append_vlen_strings("strs", &refs).unwrap();
6731            file.close().unwrap();
6732        }
6733        {
6734            let file = H5File::open_rw(&path).unwrap();
6735            file.append_vlen_strings("strs", &["s9"]).unwrap();
6736            file.close().unwrap();
6737        }
6738        {
6739            let file = H5File::open(&path).unwrap();
6740            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6741            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
6742            assert_eq!(got, want);
6743        }
6744        std::fs::remove_file(&path).ok();
6745    }
6746
6747    #[test]
6748    fn vlen_append_after_reopen_super_block() {
6749        // Reopen + append into a partial chunk whose index falls in an
6750        // extensible-array *super block* (chunk index 244 with the default
6751        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
6752        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
6753        // index block or its direct data blocks, so chunk 244 is reached
6754        // via a super block read from disk). Exercises the ViaSblk branch
6755        // of read_chunk_if_present.
6756        let path = temp_path("vlen_reopen_super");
6757        // 489 strings, chunk size 2 -> chunk 244 holds one string only
6758        // (partially filled) and is flushed to disk on close.
6759        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
6760        {
6761            let file = H5File::create(&path).unwrap();
6762            file.create_appendable_vlen_dataset("strs", 2, None)
6763                .unwrap();
6764            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6765            file.append_vlen_strings("strs", &refs).unwrap();
6766            file.close().unwrap();
6767        }
6768        {
6769            let file = H5File::open_rw(&path).unwrap();
6770            file.append_vlen_strings("strs", &["v489"]).unwrap();
6771            file.close().unwrap();
6772        }
6773        {
6774            let file = H5File::open(&path).unwrap();
6775            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6776            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
6777            assert_eq!(got, want);
6778        }
6779        std::fs::remove_file(&path).ok();
6780    }
6781
6782    #[cfg(feature = "deflate")]
6783    #[test]
6784    fn vlen_append_after_reopen_filtered_data_block() {
6785        // The hardest path: compressed + chunk in a data block + partial
6786        // read-modify-write across a reopen.
6787        let path = temp_path("vlen_reopen_filt_datablk");
6788        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
6789        {
6790            let file = H5File::create(&path).unwrap();
6791            file.create_appendable_vlen_dataset(
6792                "strs",
6793                2,
6794                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6795            )
6796            .unwrap();
6797            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6798            file.append_vlen_strings("strs", &refs).unwrap();
6799            file.close().unwrap();
6800        }
6801        {
6802            let file = H5File::open_rw(&path).unwrap();
6803            file.append_vlen_strings("strs", &["item09"]).unwrap();
6804            file.close().unwrap();
6805        }
6806        {
6807            let file = H5File::open(&path).unwrap();
6808            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6809            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
6810            assert_eq!(got, want);
6811        }
6812        std::fs::remove_file(&path).ok();
6813    }
6814
6815    #[test]
6816    fn group_nx_class_attribute_roundtrip() {
6817        // Non-root groups carry attributes (NeXus `NX_class`) in their
6818        // own object header, and the reader reads them back by path.
6819        let path = temp_path("group_nx_class");
6820        {
6821            let file = H5File::create(&path).unwrap();
6822            let entry = file.create_group("entry").unwrap();
6823            entry.set_attr_string("NX_class", "NXentry").unwrap();
6824            let det = entry.create_group("detector").unwrap();
6825            det.set_attr_string("NX_class", "NXdetector").unwrap();
6826            det.set_attr_numeric("frame_count", &7i32).unwrap();
6827            det.new_dataset::<f32>()
6828                .shape([4])
6829                .create("data")
6830                .unwrap()
6831                .write_raw(&[1.0f32; 4])
6832                .unwrap();
6833            file.close().unwrap();
6834        }
6835        {
6836            let file = H5File::open(&path).unwrap();
6837            let entry = file.root_group().group("entry").unwrap();
6838            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
6839            let det = entry.group("detector").unwrap();
6840            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
6841            let names = det.attr_names().unwrap();
6842            assert!(names.contains(&"NX_class".to_string()));
6843            assert!(names.contains(&"frame_count".to_string()));
6844        }
6845        std::fs::remove_file(&path).ok();
6846    }
6847
6848    #[test]
6849    fn ea_super_block_roundtrip() {
6850        // 2000 chunks span several extensible-array super blocks. Before
6851        // super-block support the writer errored at chunk index 228.
6852        let path = temp_path("ea_super_rt");
6853        {
6854            let file = H5File::create(&path).unwrap();
6855            let ds = file
6856                .new_dataset::<i32>()
6857                .shape([0])
6858                .chunk(&[1])
6859                .max_shape(&[None])
6860                .create("v")
6861                .unwrap();
6862            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
6863            file.close().unwrap();
6864        }
6865        {
6866            let file = H5File::open(&path).unwrap();
6867            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6868            assert_eq!(v.len(), 2000);
6869            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
6870        }
6871        std::fs::remove_file(&path).ok();
6872    }
6873
6874    #[cfg(feature = "deflate")]
6875    #[test]
6876    fn ea_filtered_super_block_roundtrip() {
6877        // Compressed chunks across super blocks.
6878        let path = temp_path("ea_filt_super");
6879        {
6880            let file = H5File::create(&path).unwrap();
6881            let ds = file
6882                .new_dataset::<i32>()
6883                .shape([0])
6884                .chunk(&[1])
6885                .max_shape(&[None])
6886                .deflate(4)
6887                .create("v")
6888                .unwrap();
6889            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
6890            file.close().unwrap();
6891        }
6892        {
6893            let file = H5File::open(&path).unwrap();
6894            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6895            assert_eq!(v, (0..600).collect::<Vec<i32>>());
6896        }
6897        std::fs::remove_file(&path).ok();
6898    }
6899
6900    #[test]
6901    fn ea_super_block_open_append() {
6902        // Reopen a dataset and append chunks that fall in super blocks.
6903        let path = temp_path("ea_super_append");
6904        {
6905            let file = H5File::create(&path).unwrap();
6906            let ds = file
6907                .new_dataset::<i32>()
6908                .shape([0])
6909                .chunk(&[1])
6910                .max_shape(&[None])
6911                .create("v")
6912                .unwrap();
6913            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
6914            file.close().unwrap();
6915        }
6916        {
6917            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
6918            let idx = w.dataset_index("v").unwrap();
6919            for c in 300..900u64 {
6920                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
6921            }
6922            w.extend_dataset(idx, &[900]).unwrap();
6923            w.close().unwrap();
6924        }
6925        {
6926            let file = H5File::open(&path).unwrap();
6927            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
6928            assert_eq!(v.len(), 900);
6929            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
6930        }
6931        std::fs::remove_file(&path).ok();
6932    }
6933
6934    // Two or more unlimited dimensions select the v2 B-tree index; with a
6935    // filter its records become type 11, carrying each chunk's stored size and
6936    // mask. The payload is highly compressible, so the chunks really are
6937    // stored smaller than the extent — the file would be at least
6938    // 6*8*4 = 192 bytes of raw chunk data otherwise.
6939    #[cfg(feature = "deflate")]
6940    #[test]
6941    fn compressed_multi_unlimited_dataset_roundtrips() {
6942        let path = temp_path("bt2_filtered");
6943        {
6944            let file = H5File::create(&path).unwrap();
6945            let ds = file
6946                .new_dataset::<i32>()
6947                .shape([6, 8])
6948                .chunk(&[2, 4])
6949                .max_shape(&[None, None])
6950                .deflate(6)
6951                .create("d")
6952                .unwrap();
6953            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
6954            // A partial write forces a decompress-patch-recompress of one
6955            // chunk, whose new compressed size may not fit its old block.
6956            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
6957            file.close().unwrap();
6958        }
6959        {
6960            let file = H5File::open(&path).unwrap();
6961            let ds = file.dataset("d").unwrap();
6962            assert_eq!(ds.shape(), vec![6, 8]);
6963            let mut want = vec![7i32; 48];
6964            want[9] = 1;
6965            want[10] = 2;
6966            want[17] = 3;
6967            want[18] = 4;
6968            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
6969        }
6970        std::fs::remove_file(&path).ok();
6971    }
6972
6973    #[test]
6974    fn btree_v2_multi_unlimited_roundtrip() {
6975        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
6976        // index; chunks are written by grid coordinates with write_chunk_at.
6977        let path = temp_path("bt2_multi");
6978        {
6979            let file = H5File::create(&path).unwrap();
6980            let ds = file
6981                .new_dataset::<i32>()
6982                .shape([0, 0])
6983                .chunk(&[2, 2])
6984                .max_shape(&[None, None])
6985                .create("grid")
6986                .unwrap();
6987            assert!(ds.is_chunked());
6988            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
6989            for cr in 0..2usize {
6990                for cc in 0..2usize {
6991                    let mut bytes = Vec::new();
6992                    for i in 0..2usize {
6993                        for j in 0..2usize {
6994                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
6995                            bytes.extend_from_slice(&v.to_le_bytes());
6996                        }
6997                    }
6998                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
6999                }
7000            }
7001            file.close().unwrap();
7002        }
7003        {
7004            let file = H5File::open(&path).unwrap();
7005            let ds = file.dataset("grid").unwrap();
7006            assert_eq!(ds.shape(), vec![4, 4]);
7007            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
7008        }
7009        std::fs::remove_file(&path).ok();
7010    }
7011
7012    #[test]
7013    fn subframe_chunking_roundtrip() {
7014        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
7015        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
7016        // the chunk-grid coordinates.
7017        let path = temp_path("subframe");
7018        {
7019            let file = H5File::create(&path).unwrap();
7020            let ds = file
7021                .new_dataset::<i32>()
7022                .shape([0, 8, 8])
7023                .chunk(&[1, 4, 4])
7024                .max_shape(&[None, Some(8), Some(8)])
7025                .create("v")
7026                .unwrap();
7027            for f in 0..3usize {
7028                for cr in 0..2usize {
7029                    for cc in 0..2usize {
7030                        let mut bytes = Vec::new();
7031                        for i in 0..4usize {
7032                            for j in 0..4usize {
7033                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
7034                                bytes.extend_from_slice(&v.to_le_bytes());
7035                            }
7036                        }
7037                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
7038                    }
7039                }
7040            }
7041            file.close().unwrap();
7042        }
7043        {
7044            let file = H5File::open(&path).unwrap();
7045            let ds = file.dataset("v").unwrap();
7046            assert_eq!(ds.shape(), vec![3, 8, 8]);
7047            assert_eq!(
7048                ds.read_raw::<i32>().unwrap(),
7049                (0..192).collect::<Vec<i32>>()
7050            );
7051        }
7052        std::fs::remove_file(&path).ok();
7053    }
7054
7055    #[test]
7056    fn fill_value_contiguous_roundtrip() {
7057        let path = temp_path("fill_value_contig");
7058        {
7059            let file = H5File::create(&path).unwrap();
7060            let ds = file
7061                .new_dataset::<f32>()
7062                .shape([4])
7063                .fill_value(2.5f32)
7064                .create("data")
7065                .unwrap();
7066            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
7067            file.close().unwrap();
7068        }
7069        // open_append decodes the fill-value message back from the header.
7070        {
7071            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7072            let idx = writer.dataset_index("data").unwrap();
7073            assert_eq!(
7074                writer.ds(idx).lock().fill_value,
7075                Some(2.5f32.to_le_bytes().to_vec())
7076            );
7077        }
7078        // Data still reads back correctly.
7079        {
7080            let file = H5File::open(&path).unwrap();
7081            let ds = file.dataset("data").unwrap();
7082            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
7083        }
7084        std::fs::remove_file(&path).ok();
7085    }
7086
7087    /// Early allocation on a fixed unfiltered shape selects the implicit
7088    /// index, and "implicit" is literal: the file holds no index structure
7089    /// at all, only a version-4 layout message of index type 2 pointing at
7090    /// the run of chunk space the create allocated.
7091    #[test]
7092    fn early_allocation_writes_the_implicit_index() {
7093        let path = temp_path("implicit_index");
7094        {
7095            let file = H5File::create(&path).unwrap();
7096            let ds = file
7097                .new_dataset::<i32>()
7098                .shape([16])
7099                .chunk(&[4])
7100                .early_allocation()
7101                .create("data")
7102                .unwrap();
7103            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7104            file.close().unwrap();
7105        }
7106        let bytes = std::fs::read(&path).unwrap();
7107        for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7108            assert!(
7109                !bytes.windows(4).any(|w| w == magic),
7110                "{} appears in a file whose chunk index is supposed to be no \
7111                 structure at all",
7112                String::from_utf8_lossy(magic)
7113            );
7114        }
7115        {
7116            let file = H5File::open(&path).unwrap();
7117            let ds = file.dataset("data").unwrap();
7118            assert_eq!(
7119                ds.read_raw::<i32>().unwrap(),
7120                (0..16i32).collect::<Vec<_>>()
7121            );
7122        }
7123        std::fs::remove_file(&path).ok();
7124    }
7125
7126    /// Two of the conditions are conditions: an unlimited dimension or a
7127    /// filter each send the dataset to the index libhdf5 would pick
7128    /// instead, early allocation or not. (The third — one whole-dataset
7129    /// chunk — sends it to the single-chunk index instead of Fixed Array;
7130    /// see `one_whole_dataset_chunk_writes_the_single_chunk_index`.)
7131    #[test]
7132    #[cfg(feature = "deflate")]
7133    fn early_allocation_only_picks_implicit_where_libhdf5_does() {
7134        // Every case writes and reads back its data, so a mis-selected index
7135        // shows up as wrong bytes and not just as a different structure.
7136        for (which, magic) in [("unlimited", b"EAHD"), ("filtered", b"FAHD")] {
7137            let path = temp_path("implicit_not");
7138            {
7139                let file = H5File::create(&path).unwrap();
7140                let builder = file.new_dataset::<i32>().shape([16]);
7141                let builder = match which {
7142                    "unlimited" => builder.chunk(&[4]).max_shape(&[None]),
7143                    _ => builder.chunk(&[4]).deflate(6),
7144                };
7145                let ds = builder.early_allocation().create("data").unwrap();
7146                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7147                file.close().unwrap();
7148            }
7149            let bytes = std::fs::read(&path).unwrap();
7150            assert!(
7151                bytes.windows(4).any(|w| w == magic),
7152                "{which}: expected a {} index",
7153                String::from_utf8_lossy(magic)
7154            );
7155            let file = H5File::open(&path).unwrap();
7156            assert_eq!(
7157                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7158                (0..16i32).collect::<Vec<_>>(),
7159                "{which}"
7160            );
7161            std::fs::remove_file(&path).ok();
7162        }
7163    }
7164
7165    /// One whole-dataset chunk always selects the single-chunk index —
7166    /// ahead of Fixed Array, and ahead of Implicit too, whether or not early
7167    /// allocation was requested (`H5D__layout_set_latest_indexing` checks it
7168    /// unconditionally). Like Implicit, "single chunk" is literal: no index
7169    /// structure at all, just the one chunk's address — and, unfiltered and
7170    /// early-allocated, that address exists before anything is written — in
7171    /// the layout message directly.
7172    #[test]
7173    fn one_whole_dataset_chunk_writes_the_single_chunk_index() {
7174        for early in [false, true] {
7175            let path = temp_path("single_chunk_index");
7176            {
7177                let file = H5File::create(&path).unwrap();
7178                let builder = file.new_dataset::<i32>().shape([16]).chunk(&[16]);
7179                let builder = if early {
7180                    builder.early_allocation()
7181                } else {
7182                    builder
7183                };
7184                let ds = builder.create("data").unwrap();
7185                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7186                file.close().unwrap();
7187            }
7188            let bytes = std::fs::read(&path).unwrap();
7189            for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7190                assert!(
7191                    !bytes.windows(4).any(|w| w == magic),
7192                    "early={early}: {} appears in a file whose chunk index is \
7193                     supposed to be no structure at all",
7194                    String::from_utf8_lossy(magic)
7195                );
7196            }
7197            let file = H5File::open(&path).unwrap();
7198            assert_eq!(
7199                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7200                (0..16i32).collect::<Vec<_>>(),
7201                "early={early}"
7202            );
7203            std::fs::remove_file(&path).ok();
7204        }
7205    }
7206
7207    /// An implicitly indexed dataset's chunks all exist from create, so an
7208    /// unwritten one reads back as the fill value — and the fill value is
7209    /// tiled over the whole run at create, not per chunk on demand.
7210    #[test]
7211    fn implicit_index_fills_every_chunk_at_create() {
7212        let path = temp_path("implicit_fill");
7213        {
7214            let file = H5File::create(&path).unwrap();
7215            let ds = file
7216                .new_dataset::<i32>()
7217                .shape([8])
7218                .chunk(&[4])
7219                .early_allocation()
7220                .fill_value(-3i32)
7221                .create("data")
7222                .unwrap();
7223            // Only chunk 0.
7224            let chunk: Vec<u8> = [1i32, 2, 3, 4]
7225                .iter()
7226                .flat_map(|v| v.to_le_bytes())
7227                .collect();
7228            ds.write_chunk(0, &chunk).unwrap();
7229            file.close().unwrap();
7230        }
7231        let file = H5File::open(&path).unwrap();
7232        let ds = file.dataset("data").unwrap();
7233        assert_eq!(
7234            ds.read_raw::<i32>().unwrap(),
7235            vec![1, 2, 3, 4, -3, -3, -3, -3]
7236        );
7237        std::fs::remove_file(&path).ok();
7238    }
7239
7240    /// A reopen has to reconstruct the run's address *and* its length from
7241    /// the layout message alone — there is no index structure to read it
7242    /// back from — or the close would rewrite the dataset as unallocated
7243    /// contiguous storage and drop every byte.
7244    #[test]
7245    fn implicit_index_survives_a_reopen() {
7246        let path = temp_path("implicit_reopen");
7247        {
7248            let file = H5File::create(&path).unwrap();
7249            file.new_dataset::<i32>()
7250                .shape([8])
7251                .chunk(&[4])
7252                .early_allocation()
7253                .create("data")
7254                .unwrap()
7255                .write_raw(&[0i32, 1, 2, 3, 4, 5, 6, 7])
7256                .unwrap();
7257            file.close().unwrap();
7258        }
7259        {
7260            let file = H5File::open_rw(&path).unwrap();
7261            let ds = file.dataset_writer("data").unwrap();
7262            let chunk: Vec<u8> = [10i32, 11, 12, 13]
7263                .iter()
7264                .flat_map(|v| v.to_le_bytes())
7265                .collect();
7266            ds.write_chunk(1, &chunk).unwrap();
7267            file.close().unwrap();
7268        }
7269        let file = H5File::open(&path).unwrap();
7270        assert_eq!(
7271            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7272            vec![0, 1, 2, 3, 10, 11, 12, 13]
7273        );
7274        std::fs::remove_file(&path).ok();
7275    }
7276
7277    #[test]
7278    fn fill_value_chunked_roundtrip() {
7279        let path = temp_path("fill_value_chunked");
7280        {
7281            let file = H5File::create(&path).unwrap();
7282            let ds = file
7283                .new_dataset::<i32>()
7284                .shape([0])
7285                .chunk(&[4])
7286                .max_shape(&[None])
7287                .fill_value(-7i32)
7288                .create("vals")
7289                .unwrap();
7290            ds.append(&[1i32, 2, 3, 4]).unwrap();
7291            file.close().unwrap();
7292        }
7293        {
7294            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7295            let idx = writer.dataset_index("vals").unwrap();
7296            assert_eq!(
7297                writer.ds(idx).lock().fill_value,
7298                Some((-7i32).to_le_bytes().to_vec())
7299            );
7300        }
7301        std::fs::remove_file(&path).ok();
7302    }
7303
7304    #[test]
7305    fn fill_value_read_missing_chunks() {
7306        // A chunked dataset with chunk 1 left unwritten must read that
7307        // gap back as the user-defined fill value, not zero.
7308        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
7309            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
7310        }
7311        let path = temp_path("fill_value_read_missing");
7312        {
7313            let file = H5File::create(&path).unwrap();
7314            let ds = file
7315                .new_dataset::<i32>()
7316                .shape([0])
7317                .chunk(&[2])
7318                .max_shape(&[None])
7319                .fill_value(-1i32)
7320                .create("vals")
7321                .unwrap();
7322            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
7323            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
7324            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
7325            ds.extend(&[6]).unwrap();
7326            file.close().unwrap();
7327        }
7328        {
7329            let file = H5File::open(&path).unwrap();
7330            let ds = file.dataset("vals").unwrap();
7331            let all = ds.read_raw::<i32>().unwrap();
7332            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
7333        }
7334        std::fs::remove_file(&path).ok();
7335    }
7336
7337    #[test]
7338    fn fill_value_partial_chunk_padded_with_fill() {
7339        // A partial trailing chunk flushed at close must pad its unwritten
7340        // tail with the fill value. That pad sits beyond the logical shape,
7341        // so it is verified by scanning the on-disk chunk bytes directly.
7342        let path = temp_path("fill_value_partial_pad");
7343        {
7344            let file = H5File::create(&path).unwrap();
7345            let ds = file
7346                .new_dataset::<i32>()
7347                .shape([0])
7348                .chunk(&[4])
7349                .max_shape(&[None])
7350                .fill_value(-9i32)
7351                .create("vals")
7352                .unwrap();
7353            // 3 of 4 frames -> flushed as a partial chunk on close.
7354            ds.append(&[1i32, 2, 3]).unwrap();
7355            file.close().unwrap();
7356        }
7357        let bytes = std::fs::read(&path).unwrap();
7358        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
7359        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
7360        let pos = bytes
7361            .windows(needle.len())
7362            .position(|w| w == needle)
7363            .expect("chunk data [1,2,3] not found in file");
7364        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
7365        assert_eq!(
7366            pad,
7367            &(-9i32).to_le_bytes(),
7368            "partial chunk tail must be padded with fill value -9, got {:?}",
7369            pad
7370        );
7371        std::fs::remove_file(&path).ok();
7372    }
7373
7374    #[test]
7375    fn vlen_append_after_reopen_preserves_existing() {
7376        // Reopening and appending into a partially-written vlen chunk must
7377        // read-modify-write: the strings already on disk must survive.
7378        let path = temp_path("vlen_append_reopen");
7379        {
7380            let file = H5File::create(&path).unwrap();
7381            file.create_appendable_vlen_dataset("strs", 4, None)
7382                .unwrap();
7383            // 3 of 4 frames -> flushed as a partial chunk on close.
7384            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
7385            file.close().unwrap();
7386        }
7387        {
7388            // Append a 4th string -> partial-chunk write into chunk 0.
7389            let file = H5File::open_rw(&path).unwrap();
7390            file.append_vlen_strings("strs", &["d"]).unwrap();
7391            file.close().unwrap();
7392        }
7393        {
7394            let file = H5File::open(&path).unwrap();
7395            let ds = file.dataset("strs").unwrap();
7396            let got = ds.read_vlen_strings().unwrap();
7397            assert_eq!(
7398                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
7399                vec!["a", "b", "c", "d"]
7400            );
7401        }
7402        std::fs::remove_file(&path).ok();
7403    }
7404
7405    #[test]
7406    fn fill_value_size_mismatch_errors() {
7407        let path = temp_path("fill_value_mismatch");
7408        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
7409        let dt = <f64 as crate::types::H5Type>::hdf5_type();
7410        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
7411        // f64 element size is 8; a 4-byte fill value must be rejected.
7412        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
7413        // The correct width succeeds.
7414        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
7415        writer.close().unwrap();
7416        std::fs::remove_file(&path).ok();
7417    }
7418
7419    #[test]
7420    fn datatype_exposes_class_sign_and_byteorder() {
7421        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
7422        // from f32 (both 4 bytes). datatype() must report the real class and
7423        // signedness so a reader does not have to guess from element_size.
7424        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
7425
7426        let path = temp_path("datatype_accessor");
7427        {
7428            let file = H5File::create(&path).unwrap();
7429            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
7430            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
7431            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
7432            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
7433            file.close().unwrap();
7434        }
7435
7436        let file = H5File::open(&path).unwrap();
7437
7438        match file.dataset("u8d").unwrap().datatype().unwrap() {
7439            DatatypeMessage::FixedPoint {
7440                size,
7441                signed,
7442                byte_order,
7443                ..
7444            } => {
7445                assert_eq!(size, 1);
7446                assert!(!signed, "u8 must be unsigned");
7447                assert_eq!(byte_order, ByteOrder::LittleEndian);
7448            }
7449            other => panic!("expected FixedPoint for u8, got {other:?}"),
7450        }
7451
7452        match file.dataset("i8d").unwrap().datatype().unwrap() {
7453            DatatypeMessage::FixedPoint { size, signed, .. } => {
7454                assert_eq!(size, 1);
7455                assert!(signed, "i8 must be signed");
7456            }
7457            other => panic!("expected FixedPoint for i8, got {other:?}"),
7458        }
7459
7460        match file.dataset("i32d").unwrap().datatype().unwrap() {
7461            DatatypeMessage::FixedPoint { size, signed, .. } => {
7462                assert_eq!(size, 4);
7463                assert!(signed, "i32 must be signed");
7464            }
7465            other => panic!("expected FixedPoint for i32, got {other:?}"),
7466        }
7467
7468        match file.dataset("f32d").unwrap().datatype().unwrap() {
7469            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
7470            other => panic!("expected FloatingPoint for f32, got {other:?}"),
7471        }
7472
7473        std::fs::remove_file(&path).ok();
7474    }
7475
7476    #[test]
7477    fn datatype_in_write_mode_errors() {
7478        let path = temp_path("datatype_write_mode");
7479        let file = H5File::create(&path).unwrap();
7480        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
7481        assert!(ds.datatype().is_err());
7482        std::fs::remove_file(&path).ok();
7483    }
7484
7485    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
7486
7487    /// Extensible-array path: pre-compress with the dataset's pipeline, write
7488    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
7489    /// the data round-trips through the reader unchanged.
7490    #[cfg(feature = "deflate")]
7491    #[test]
7492    fn write_chunk_raw_ea_roundtrip_mask0() {
7493        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7494        let path = temp_path("wcr_ea_mask0");
7495        let original: Vec<i32> = (0..12).collect();
7496        {
7497            let file = H5File::create(&path).unwrap();
7498            let ds = file
7499                .new_dataset::<i32>()
7500                .shape([0])
7501                .chunk(&[4])
7502                .max_shape(&[None])
7503                .deflate(4)
7504                .create("v")
7505                .unwrap();
7506            assert!(ds.is_chunked());
7507            let pipeline = FilterPipeline::deflate(4);
7508            for c in 0..3usize {
7509                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7510                    .iter()
7511                    .flat_map(|v| v.to_le_bytes())
7512                    .collect();
7513                let compressed = apply_filters(&pipeline, &raw).unwrap();
7514                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7515            }
7516            ds.set_extent(&[12]).unwrap();
7517            file.close().unwrap();
7518        }
7519        {
7520            let file = H5File::open(&path).unwrap();
7521            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7522            assert_eq!(v, original);
7523        }
7524        std::fs::remove_file(&path).ok();
7525    }
7526
7527    /// Fixed-array path (all dimensions bounded): same verbatim write through
7528    /// the linear-index dispatch, round-tripped through the reader.
7529    #[cfg(feature = "deflate")]
7530    #[test]
7531    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
7532        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7533        let path = temp_path("wcr_fa_mask0");
7534        let original: Vec<i32> = (0..12).collect();
7535        {
7536            let file = H5File::create(&path).unwrap();
7537            let ds = file
7538                .new_dataset::<i32>()
7539                .shape([12])
7540                .chunk(&[4])
7541                .deflate(4)
7542                .create("v")
7543                .unwrap();
7544            assert!(ds.is_chunked());
7545            let pipeline = FilterPipeline::deflate(4);
7546            for c in 0..3usize {
7547                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7548                    .iter()
7549                    .flat_map(|v| v.to_le_bytes())
7550                    .collect();
7551                let compressed = apply_filters(&pipeline, &raw).unwrap();
7552                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7553            }
7554            file.close().unwrap();
7555        }
7556        {
7557            let file = H5File::open(&path).unwrap();
7558            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7559            assert_eq!(v, original);
7560        }
7561        std::fs::remove_file(&path).ok();
7562    }
7563
7564    /// The caller-supplied filter_mask must reach the on-disk filtered index
7565    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
7566    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
7567    /// the extensible-array filtered entry to read the mask back at the format
7568    /// level (independent of the data reader's mask handling).
7569    #[cfg(feature = "deflate")]
7570    #[test]
7571    fn write_chunk_raw_records_filter_mask() {
7572        let path = temp_path("wcr_records_mask");
7573        let raw: Vec<u8> = [10i32, 20, 30, 40]
7574            .iter()
7575            .flat_map(|v| v.to_le_bytes())
7576            .collect();
7577        assert_eq!(raw.len(), 16);
7578        {
7579            let file = H5File::create(&path).unwrap();
7580            let ds = file
7581                .new_dataset::<i32>()
7582                .shape([0])
7583                .chunk(&[4])
7584                .max_shape(&[None])
7585                .deflate(4)
7586                .create("v")
7587                .unwrap();
7588            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
7589            // chunk is stored uncompressed (its raw bytes).
7590            ds.write_chunk_raw(0, &raw, 1).unwrap();
7591            ds.set_extent(&[4]).unwrap();
7592            file.close().unwrap();
7593        }
7594        // Reopen the writer; open_append decodes the filtered index block from
7595        // disk, so the entry reflects exactly what was committed.
7596        {
7597            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7598            let idx = w.dataset_index("v").unwrap();
7599            let ds = w.ds(idx);
7600            let m = ds.lock();
7601            let entry = &m
7602                .chunked
7603                .as_ref()
7604                .unwrap()
7605                .filt_iblk
7606                .as_ref()
7607                .unwrap()
7608                .elements[0];
7609            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
7610            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
7611        }
7612        std::fs::remove_file(&path).ok();
7613    }
7614
7615    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
7616    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
7617    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
7618    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
7619    /// A chunk whose stored stream decodes to less than its image places no
7620    /// run at all: the whole chunk reads as the fill value, whether the read
7621    /// laid the fill down first (a plan that leaves output uncovered — here the
7622    /// unallocated middle chunk) or fills only what nothing wrote. The decode
7623    /// writes into the output image itself, so the bytes a short image leaves
7624    /// behind are the ones this covers.
7625    #[cfg(feature = "deflate")]
7626    #[test]
7627    fn a_chunk_that_decodes_short_of_its_image_reads_as_fill() {
7628        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7629        let path = temp_path("short_chunk_image_is_fill");
7630        let pipeline = FilterPipeline::deflate(4);
7631        {
7632            let file = H5File::create(&path).unwrap();
7633            let ds = file
7634                .new_dataset::<i32>()
7635                .shape([0])
7636                .chunk(&[4])
7637                .max_shape(&[None])
7638                .fill_value(-1i32)
7639                .deflate(4)
7640                .create("v")
7641                .unwrap();
7642            // Chunk 0 carries two elements where its image wants four.
7643            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7644            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7645                .unwrap();
7646            // Chunk 1 is never written; chunk 2 is whole.
7647            let whole: Vec<u8> = [9i32, 10, 11, 12]
7648                .iter()
7649                .flat_map(|v| v.to_le_bytes())
7650                .collect();
7651            ds.write_chunk_raw(2, &apply_filters(&pipeline, &whole).unwrap(), 0)
7652                .unwrap();
7653            ds.set_extent(&[12]).unwrap();
7654            file.close().unwrap();
7655        }
7656        {
7657            let file = H5File::open(&path).unwrap();
7658            let ds = file.dataset("v").unwrap();
7659            assert_eq!(
7660                ds.read_raw::<i32>().unwrap(),
7661                vec![-1, -1, -1, -1, -1, -1, -1, -1, 9, 10, 11, 12]
7662            );
7663            // The same verdict when the plan covers every output byte, so no
7664            // fill goes down first: chunks 0 and 2 alone.
7665            assert_eq!(ds.read_slice::<i32>(&[0], &[4]).unwrap(), vec![-1; 4]);
7666            assert_eq!(
7667                ds.read_slice::<i32>(&[8], &[4]).unwrap(),
7668                vec![9, 10, 11, 12]
7669            );
7670        }
7671        std::fs::remove_file(&path).ok();
7672    }
7673
7674    /// The staged spelling of the case above: a selection that takes only part
7675    /// of the short chunk decodes it into a buffer sized from the layout, and
7676    /// what that buffer holds past the stream is cut off rather than kept — a
7677    /// run inside the decoded bytes is real data, a run reaching past them is
7678    /// fill.
7679    #[cfg(feature = "deflate")]
7680    #[test]
7681    fn a_staged_chunk_carries_only_what_its_stream_decoded() {
7682        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7683        let path = temp_path("short_chunk_image_staged");
7684        let pipeline = FilterPipeline::deflate(4);
7685        {
7686            let file = H5File::create(&path).unwrap();
7687            let ds = file
7688                .new_dataset::<i32>()
7689                .shape([0])
7690                .chunk(&[4])
7691                .max_shape(&[None])
7692                .fill_value(-1i32)
7693                .deflate(4)
7694                .create("v")
7695                .unwrap();
7696            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7697            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7698                .unwrap();
7699            ds.set_extent(&[4]).unwrap();
7700            file.close().unwrap();
7701        }
7702        {
7703            let file = H5File::open(&path).unwrap();
7704            let ds = file.dataset("v").unwrap();
7705            // Inside the decoded bytes.
7706            assert_eq!(ds.read_slice::<i32>(&[0], &[2]).unwrap(), vec![7, 8]);
7707            // Straddling their end: the run is not placed at all.
7708            assert_eq!(ds.read_slice::<i32>(&[1], &[2]).unwrap(), vec![-1, -1]);
7709        }
7710        std::fs::remove_file(&path).ok();
7711    }
7712
7713    #[cfg(feature = "deflate")]
7714    #[test]
7715    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
7716        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7717        let path = temp_path("wcr_ea_per_chunk_mask");
7718        let original: Vec<i32> = (0..8).collect();
7719        let pipeline = FilterPipeline::deflate(4);
7720        {
7721            let file = H5File::create(&path).unwrap();
7722            let ds = file
7723                .new_dataset::<i32>()
7724                .shape([0])
7725                .chunk(&[4])
7726                .max_shape(&[None])
7727                .deflate(4)
7728                .create("v")
7729                .unwrap();
7730            let raw0: Vec<u8> = original[0..4]
7731                .iter()
7732                .flat_map(|v| v.to_le_bytes())
7733                .collect();
7734            // chunk 0: compressed through the pipeline, mask 0.
7735            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7736                .unwrap();
7737            let raw1: Vec<u8> = original[4..8]
7738                .iter()
7739                .flat_map(|v| v.to_le_bytes())
7740                .collect();
7741            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
7742            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7743            ds.set_extent(&[8]).unwrap();
7744            file.close().unwrap();
7745        }
7746        {
7747            let file = H5File::open(&path).unwrap();
7748            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7749            assert_eq!(v, original);
7750        }
7751        std::fs::remove_file(&path).ok();
7752    }
7753
7754    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
7755    /// compressed/raw chunks as the EA case, through the fixed-array index.
7756    #[cfg(feature = "deflate")]
7757    #[test]
7758    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
7759        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7760        let path = temp_path("wcr_fa_per_chunk_mask");
7761        let original: Vec<i32> = (0..8).collect();
7762        let pipeline = FilterPipeline::deflate(4);
7763        {
7764            let file = H5File::create(&path).unwrap();
7765            let ds = file
7766                .new_dataset::<i32>()
7767                .shape([8])
7768                .chunk(&[4])
7769                .deflate(4)
7770                .create("v")
7771                .unwrap();
7772            let raw0: Vec<u8> = original[0..4]
7773                .iter()
7774                .flat_map(|v| v.to_le_bytes())
7775                .collect();
7776            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7777                .unwrap();
7778            let raw1: Vec<u8> = original[4..8]
7779                .iter()
7780                .flat_map(|v| v.to_le_bytes())
7781                .collect();
7782            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7783            file.close().unwrap();
7784        }
7785        {
7786            let file = H5File::open(&path).unwrap();
7787            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7788            assert_eq!(v, original);
7789        }
7790        std::fs::remove_file(&path).ok();
7791    }
7792
7793    /// An unfiltered chunk index has no slot for a stored size or mask, so a
7794    /// direct chunk write must be rejected rather than silently dropping them.
7795    #[test]
7796    fn write_chunk_raw_rejects_unfiltered() {
7797        let path = temp_path("wcr_unfiltered");
7798        let file = H5File::create(&path).unwrap();
7799        let ds = file
7800            .new_dataset::<i32>()
7801            .shape([0])
7802            .chunk(&[4])
7803            .max_shape(&[None])
7804            .create("v")
7805            .unwrap();
7806        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7807        assert!(
7808            err.to_string().contains("filtered dataset"),
7809            "expected a filtered-dataset error, got: {err}"
7810        );
7811        std::fs::remove_file(&path).ok();
7812    }
7813
7814    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
7815    /// index to mean anything against, so the linear entry point points the
7816    /// caller at the coordinate-addressed one rather than guessing a grid.
7817    #[test]
7818    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
7819        let path = temp_path("wcr_btree2");
7820        let file = H5File::create(&path).unwrap();
7821        let ds = file
7822            .new_dataset::<i32>()
7823            .shape([0, 0])
7824            .chunk(&[2, 2])
7825            .max_shape(&[None, None])
7826            .create("grid")
7827            .unwrap();
7828        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7829        assert!(
7830            err.to_string().contains("write_chunk_raw_at"),
7831            "expected a pointer to the coordinate form, got: {err}"
7832        );
7833        std::fs::remove_file(&path).ok();
7834    }
7835
7836    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
7837    /// and the type-11 record carries their size and the caller's mask, so a
7838    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
7839    /// bytes while one written compressed (mask 0) is decompressed.
7840    #[cfg(feature = "deflate")]
7841    #[test]
7842    fn write_chunk_raw_at_round_trips_on_btree_v2() {
7843        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7844
7845        let path = temp_path("wcr_at_btree2");
7846        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
7847        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
7848        {
7849            let file = H5File::create(&path).unwrap();
7850            let ds = file
7851                .new_dataset::<i32>()
7852                .shape([0, 0])
7853                .chunk(&[2, 2])
7854                .max_shape(&[None, None])
7855                .deflate(6)
7856                .create("grid")
7857                .unwrap();
7858            let pipeline = FilterPipeline::deflate(6);
7859            // Chunk (0,0): pipeline already applied upstream, mask 0.
7860            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
7861                .unwrap();
7862            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
7863            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
7864            file.close().unwrap();
7865        }
7866        let file = H5File::open(&path).unwrap();
7867        let ds = file.dataset("grid").unwrap();
7868        assert_eq!(ds.shape(), vec![4, 4]);
7869        let all = ds.read_raw::<i32>().unwrap();
7870        // Chunk (0,0) occupies rows 0..2, columns 0..2.
7871        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
7872        // Chunk (1,1) occupies rows 2..4, columns 2..4.
7873        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
7874        drop(file);
7875        std::fs::remove_file(&path).ok();
7876    }
7877
7878    /// The coordinate form is not BT2-only: it addresses an extensible- or
7879    /// fixed-array dataset's grid just as well, and records the same mask.
7880    #[cfg(feature = "deflate")]
7881    #[test]
7882    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
7883        for (label, max_shape) in [
7884            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
7885            ("wcr_at_fa", None),
7886        ] {
7887            let path = temp_path(label);
7888            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
7889            {
7890                let file = H5File::create(&path).unwrap();
7891                let mut b = file
7892                    .new_dataset::<i32>()
7893                    .shape([4usize, 4])
7894                    .chunk(&[2, 4])
7895                    .deflate(6);
7896                if let Some(ref ms) = max_shape {
7897                    b = b.max_shape(ms);
7898                }
7899                let ds = b.create("grid").unwrap();
7900                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
7901                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
7902                file.close().unwrap();
7903            }
7904            let file = H5File::open(&path).unwrap();
7905            let ds = file.dataset("grid").unwrap();
7906            let all = ds.read_raw::<i32>().unwrap();
7907            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
7908            drop(file);
7909            std::fs::remove_file(&path).ok();
7910        }
7911    }
7912
7913    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
7914    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
7915    /// chunk gives chunk_size_len = 2 (max 65535).
7916    #[cfg(feature = "deflate")]
7917    #[test]
7918    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
7919        let path = temp_path("wcr_at_oversized");
7920        let file = H5File::create(&path).unwrap();
7921        let ds = file
7922            .new_dataset::<i32>()
7923            .shape([0, 0])
7924            .chunk(&[1, 1])
7925            .max_shape(&[None, None])
7926            .deflate(4)
7927            .create("grid")
7928            .unwrap();
7929        let err = ds
7930            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
7931            .unwrap_err();
7932        assert!(
7933            err.to_string().contains("does not fit"),
7934            "expected a chunk-size-field overflow error, got: {err}"
7935        );
7936        std::fs::remove_file(&path).ok();
7937    }
7938
7939    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
7940    /// the same reason the array indexes reject a direct write.
7941    #[test]
7942    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
7943        let path = temp_path("wcr_at_unfiltered");
7944        let file = H5File::create(&path).unwrap();
7945        let ds = file
7946            .new_dataset::<i32>()
7947            .shape([0, 0])
7948            .chunk(&[2, 2])
7949            .max_shape(&[None, None])
7950            .create("grid")
7951            .unwrap();
7952        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
7953        assert!(
7954            err.to_string().contains("filtered dataset"),
7955            "expected a filtered-dataset error, got: {err}"
7956        );
7957        std::fs::remove_file(&path).ok();
7958    }
7959
7960    /// A stored size that does not fit the index's chunk-size field must error
7961    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
7962    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
7963    /// a 70000-byte stored chunk overflows it.
7964    #[cfg(feature = "deflate")]
7965    #[test]
7966    fn write_chunk_raw_rejects_oversized_chunk() {
7967        let path = temp_path("wcr_oversized");
7968        let file = H5File::create(&path).unwrap();
7969        let ds = file
7970            .new_dataset::<i32>()
7971            .shape([0])
7972            .chunk(&[1])
7973            .max_shape(&[None])
7974            .deflate(4)
7975            .create("v")
7976            .unwrap();
7977        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
7978        assert!(
7979            err.to_string().contains("does not fit"),
7980            "expected a chunk-size-field overflow error, got: {err}"
7981        );
7982        std::fs::remove_file(&path).ok();
7983    }
7984
7985    // ---- issue #5: runtime-width fixed-string reading ----------------------
7986
7987    use crate::format::messages::datatype::{CompoundMember, DatatypeMessage};
7988
7989    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
7990    /// element images, optionally chunked and deflated.
7991    fn write_fixed_string_dataset(
7992        path: &std::path::Path,
7993        dt: DatatypeMessage,
7994        width: usize,
7995        elems: &[&[u8]],
7996        compressed: bool,
7997    ) {
7998        let mut raw = Vec::with_capacity(elems.len() * width);
7999        for e in elems {
8000            assert!(e.len() <= width);
8001            raw.extend_from_slice(e);
8002            raw.resize(raw.len() + (width - e.len()), 0);
8003        }
8004        let file = H5File::create(path).unwrap();
8005        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
8006        if compressed {
8007            b = b.chunk(&[2]).deflate(6);
8008        }
8009        let ds = b.create("labels").unwrap();
8010        ds.write_raw_bytes(&raw).unwrap();
8011        file.close().unwrap();
8012    }
8013
8014    /// The width is whatever the file says, so one call reads a 24-byte label
8015    /// column and a 100-byte one. Producers like VASP pick it per dataset.
8016    #[test]
8017    fn read_strings_handles_any_fixed_width() {
8018        for width in [4usize, 24, 100] {
8019            let path = temp_path(&format!("fixed_str_{width}"));
8020            write_fixed_string_dataset(
8021                &path,
8022                DatatypeMessage::fixed_string(width as u32),
8023                width,
8024                &[b"ab", b"cde", b""],
8025                false,
8026            );
8027            let file = H5File::open(&path).unwrap();
8028            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8029            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
8030            std::fs::remove_file(&path).ok();
8031        }
8032    }
8033
8034    /// Each padding rule decides where the value ends. Null-terminated and
8035    /// null-padded both stop at the first NUL and ignore the bytes after it;
8036    /// space-padded strips only a tail of spaces, so an embedded NUL is
8037    /// content there.
8038    ///
8039    /// Checked against libhdf5 1.14.6: reading this same `"ab\0X\0\0"`
8040    /// null-padded element into a wider null-terminated destination gives
8041    /// `"ab"`, and reading a space-padded `"a\0b     "` gives `"a\0b"` —
8042    /// `H5T__conv_s_s` runs the same `!s[nchars]` loop for both null rules.
8043    #[test]
8044    fn read_strings_honors_every_padding_rule() {
8045        // "ab" then a NUL then trailing junk that both null rules must drop.
8046        let elem: &[u8] = b"ab\0X\0\0";
8047        for (padding, want) in [(0u8, "ab"), (1, "ab")] {
8048            let path = temp_path(&format!("fixed_pad_{padding}"));
8049            write_fixed_string_dataset(
8050                &path,
8051                DatatypeMessage::FixedString {
8052                    size: 6,
8053                    padding,
8054                    charset: 0,
8055                },
8056                6,
8057                &[elem],
8058                false,
8059            );
8060            let file = H5File::open(&path).unwrap();
8061            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8062            assert_eq!(got, vec![want.to_string()], "padding {padding}");
8063            std::fs::remove_file(&path).ok();
8064        }
8065        // Space-padded keeps interior spaces and strips only the tail.
8066        let path = temp_path("fixed_pad_2");
8067        write_fixed_string_dataset(
8068            &path,
8069            DatatypeMessage::FixedString {
8070                size: 8,
8071                padding: 2,
8072                charset: 0,
8073            },
8074            8,
8075            &[b"a b     "],
8076            false,
8077        );
8078        let file = H5File::open(&path).unwrap();
8079        assert_eq!(
8080            file.dataset("labels").unwrap().read_strings().unwrap(),
8081            vec!["a b".to_string()]
8082        );
8083        std::fs::remove_file(&path).ok();
8084
8085        // ... and an embedded NUL, which no space rule marks as an end.
8086        let path = temp_path("fixed_pad_2_nul");
8087        write_fixed_string_dataset(
8088            &path,
8089            DatatypeMessage::FixedString {
8090                size: 8,
8091                padding: 2,
8092                charset: 0,
8093            },
8094            8,
8095            &[b"a\0b     "],
8096            false,
8097        );
8098        let file = H5File::open(&path).unwrap();
8099        assert_eq!(
8100            file.dataset("labels").unwrap().read_strings().unwrap(),
8101            vec!["a\0b".to_string()]
8102        );
8103        std::fs::remove_file(&path).ok();
8104    }
8105
8106    /// A reserved padding or character-set code is an error naming the element,
8107    /// not a guess.
8108    #[test]
8109    fn read_strings_rejects_reserved_datatype_codes() {
8110        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
8111            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
8112            write_fixed_string_dataset(
8113                &path,
8114                DatatypeMessage::FixedString {
8115                    size: 4,
8116                    padding,
8117                    charset,
8118                },
8119                4,
8120                &[b"ab"],
8121                false,
8122            );
8123            let file = H5File::open(&path).unwrap();
8124            let err = file
8125                .dataset("labels")
8126                .unwrap()
8127                .read_strings()
8128                .unwrap_err()
8129                .to_string();
8130            assert!(err.contains(want), "got: {err}");
8131            std::fs::remove_file(&path).ok();
8132        }
8133    }
8134
8135    /// The typed read paths reinterpret the element image, so the stored order
8136    /// has to be the host's first. A scalar is swapped; a composite cannot be
8137    /// (its members have their own orders and offsets) and is refused.
8138    #[test]
8139    fn to_host_byte_order_converts_scalars_and_refuses_composites() {
8140        use crate::dataset::{to_host_byte_order, HOST_BYTE_ORDER};
8141        use crate::format::messages::datatype::ByteOrder;
8142
8143        let foreign = match HOST_BYTE_ORDER {
8144            ByteOrder::LittleEndian => ByteOrder::BigEndian,
8145            ByteOrder::BigEndian => ByteOrder::LittleEndian,
8146        };
8147        let int = |order, size| DatatypeMessage::FixedPoint {
8148            size,
8149            byte_order: order,
8150            signed: false,
8151            bit_offset: 0,
8152            bit_precision: (size * 8) as u16,
8153        };
8154
8155        // Foreign order: each element is reversed, elementwise.
8156        let mut buf = [1u8, 2, 3, 4, 5, 6, 7, 8];
8157        to_host_byte_order(&mut buf, &int(foreign, 4), 4).unwrap();
8158        assert_eq!(buf, [4, 3, 2, 1, 8, 7, 6, 5]);
8159
8160        // Host order: untouched.
8161        let mut buf = [1u8, 2, 3, 4];
8162        to_host_byte_order(&mut buf, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8163        assert_eq!(buf, [1, 2, 3, 4]);
8164
8165        // One byte wide: no order to convert.
8166        let mut buf = [1u8, 2, 3, 4];
8167        to_host_byte_order(&mut buf, &int(foreign, 1), 1).unwrap();
8168        assert_eq!(buf, [1, 2, 3, 4]);
8169
8170        // An enum stores its values in its base type's order.
8171        let mut buf = [1u8, 2];
8172        let enumeration = DatatypeMessage::Enum {
8173            base: Box::new(int(foreign, 2)),
8174            members: Vec::new(),
8175        };
8176        to_host_byte_order(&mut buf, &enumeration, 2).unwrap();
8177        assert_eq!(buf, [2, 1]);
8178
8179        // A string has no byte order at all.
8180        let mut buf = *b"abcd";
8181        to_host_byte_order(&mut buf, &DatatypeMessage::fixed_string(4), 4).unwrap();
8182        assert_eq!(&buf, b"abcd");
8183
8184        // A compound whose members are all host-order is reinterpretable.
8185        let compound = |order| DatatypeMessage::Compound {
8186            size: 4,
8187            members: vec![CompoundMember {
8188                name: "x".into(),
8189                offset: 0,
8190                datatype: int(order, 4),
8191            }],
8192        };
8193        let mut buf = [1u8, 2, 3, 4];
8194        to_host_byte_order(&mut buf, &compound(HOST_BYTE_ORDER), 4).unwrap();
8195        assert_eq!(buf, [1, 2, 3, 4]);
8196
8197        // One that is not says so, rather than handing back the raw bytes.
8198        let mut buf = [1u8, 2, 3, 4];
8199        let err = to_host_byte_order(&mut buf, &compound(foreign), 4)
8200            .expect_err("a foreign-order compound was reinterpreted")
8201            .to_string();
8202        assert!(err.contains("read_raw_bytes"), "got: {err}");
8203        assert_eq!(buf, [1, 2, 3, 4], "the refused image is left alone");
8204    }
8205
8206    /// The write direction answers for exactly the types the read direction
8207    /// does — same classifier — and borrows the caller's bytes whenever the
8208    /// declared order is already the host's.
8209    #[test]
8210    fn to_stored_byte_order_converts_scalars_and_refuses_composites() {
8211        use crate::dataset::{to_stored_byte_order, FOREIGN_BYTE_ORDER, HOST_BYTE_ORDER};
8212        use std::borrow::Cow;
8213
8214        let int = |order, size| DatatypeMessage::FixedPoint {
8215            size,
8216            byte_order: order,
8217            signed: false,
8218            bit_offset: 0,
8219            bit_precision: (size * 8) as u16,
8220        };
8221
8222        // Declared foreign: each element is reversed on the way out.
8223        let host = [1u8, 2, 3, 4, 5, 6, 7, 8];
8224        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 4), 4).unwrap();
8225        assert_eq!(&*stored, &[4, 3, 2, 1, 8, 7, 6, 5]);
8226        assert!(matches!(stored, Cow::Owned(_)), "a swap needs its own copy");
8227
8228        // Declared host order: handed through without a copy.
8229        let stored = to_stored_byte_order(&host, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8230        assert!(matches!(stored, Cow::Borrowed(_)), "no copy without a swap");
8231        assert_eq!(&*stored, &host);
8232
8233        // One byte wide: no order to lay out.
8234        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 1), 1).unwrap();
8235        assert_eq!(&*stored, &host);
8236
8237        // An enum stores its values in its base type's order.
8238        let enumeration = DatatypeMessage::Enum {
8239            base: Box::new(int(FOREIGN_BYTE_ORDER, 2)),
8240            members: Vec::new(),
8241        };
8242        let stored = to_stored_byte_order(&[1u8, 2], &enumeration, 2).unwrap();
8243        assert_eq!(&*stored, &[2, 1]);
8244
8245        // A compound cannot be laid out as a unit; one that declares the
8246        // foreign order for a member is refused, not written host-order.
8247        let compound = |order| DatatypeMessage::Compound {
8248            size: 4,
8249            members: vec![CompoundMember {
8250                name: "x".into(),
8251                offset: 0,
8252                datatype: int(order, 4),
8253            }],
8254        };
8255        let stored = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(HOST_BYTE_ORDER), 4).unwrap();
8256        assert_eq!(&*stored, &[1, 2, 3, 4]);
8257        let err = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(FOREIGN_BYTE_ORDER), 4)
8258            .expect_err("a foreign-order compound was written from host bytes")
8259            .to_string();
8260        assert!(err.contains("write_raw_bytes"), "got: {err}");
8261    }
8262
8263    /// The declared character set is enforced: a byte that cannot be decoded is
8264    /// an error naming the element, and the lossy call is what accepts the file
8265    /// instead of a silent substitution here.
8266    #[test]
8267    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
8268        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
8269        // in one that declares UTF-8.
8270        for (charset, bytes, want) in [
8271            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
8272            (1, b"a\xff".as_slice(), "not valid UTF-8"),
8273        ] {
8274            let path = temp_path(&format!("fixed_charset_{charset}"));
8275            write_fixed_string_dataset(
8276                &path,
8277                DatatypeMessage::FixedString {
8278                    size: 6,
8279                    padding: 1,
8280                    charset,
8281                },
8282                6,
8283                &[b"ok", bytes],
8284                false,
8285            );
8286            let file = H5File::open(&path).unwrap();
8287            let ds = file.dataset("labels").unwrap();
8288            let err = ds.read_strings().unwrap_err().to_string();
8289            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
8290            let lossy = ds.read_strings_lossy().unwrap();
8291            assert_eq!(lossy[0], "ok");
8292            assert_eq!(
8293                lossy[1].chars().next().unwrap(),
8294                if charset == 0 { 'c' } else { 'a' }
8295            );
8296            std::fs::remove_file(&path).ok();
8297        }
8298    }
8299
8300    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
8301    /// split a character.
8302    #[test]
8303    fn read_strings_reads_utf8_fixed_strings() {
8304        let path = temp_path("fixed_utf8");
8305        write_fixed_string_dataset(
8306            &path,
8307            DatatypeMessage::fixed_string_utf8(12),
8308            12,
8309            &["héllo".as_bytes(), "안녕".as_bytes()],
8310            false,
8311        );
8312        let file = H5File::open(&path).unwrap();
8313        assert_eq!(
8314            file.dataset("labels").unwrap().read_strings().unwrap(),
8315            vec!["héllo".to_string(), "안녕".to_string()]
8316        );
8317        std::fs::remove_file(&path).ok();
8318    }
8319
8320    /// The decode sits on the decoded raw-data path, so a chunked and deflated
8321    /// dataset reads the same as a contiguous one.
8322    #[cfg(feature = "deflate")]
8323    #[test]
8324    fn read_strings_reads_a_compressed_fixed_string_dataset() {
8325        let path = temp_path("fixed_str_deflate");
8326        write_fixed_string_dataset(
8327            &path,
8328            DatatypeMessage::fixed_string(16),
8329            16,
8330            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
8331            true,
8332        );
8333        let file = H5File::open(&path).unwrap();
8334        assert_eq!(
8335            file.dataset("labels").unwrap().read_strings().unwrap(),
8336            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
8337        );
8338        std::fs::remove_file(&path).ok();
8339    }
8340
8341    /// One call covers both string datatypes, so a caller need not branch on
8342    /// which one the file used.
8343    #[test]
8344    fn read_strings_also_reads_variable_length_strings() {
8345        let path = temp_path("read_strings_vlen");
8346        {
8347            let file = H5File::create(&path).unwrap();
8348            file.write_vlen_strings("names", &["alpha", "", "안녕"])
8349                .unwrap();
8350            file.close().unwrap();
8351        }
8352        let file = H5File::open(&path).unwrap();
8353        assert_eq!(
8354            file.dataset("names").unwrap().read_strings().unwrap(),
8355            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
8356        );
8357        std::fs::remove_file(&path).ok();
8358    }
8359
8360    /// A file declaring a zero-width fixed string is an error, not the panic
8361    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
8362    /// test patches the width in the encoded datatype message down to zero and
8363    /// re-stamps the object header's checksum over the result.
8364    #[test]
8365    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
8366        use crate::format::checksum::checksum_metadata;
8367        use crate::format::object_header::OHDR_SIGNATURE;
8368
8369        let path = temp_path("fixed_str_zero_width");
8370        write_fixed_string_dataset(
8371            &path,
8372            DatatypeMessage::fixed_string(37),
8373            37,
8374            &[b"ab", b"cd"],
8375            false,
8376        );
8377
8378        // Version 1 string datatype: class|version, padding|charset, two
8379        // reserved bytes, then the width as a little-endian u32. The width is
8380        // 37 so the eight bytes occur once in the file.
8381        let mut bytes = std::fs::read(&path).unwrap();
8382        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
8383        let at = bytes
8384            .windows(needle.len())
8385            .position(|w| w == needle)
8386            .expect("encoded fixed-string datatype message");
8387        assert!(
8388            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
8389            "the datatype message pattern is not unique in the file"
8390        );
8391
8392        // The enclosing v2 object header ends in a checksum over everything
8393        // from its signature onwards; find the offset where the stored value
8394        // still agrees, so the patched header can be re-stamped there.
8395        let ohdr = bytes[..at]
8396            .windows(4)
8397            .rposition(|w| w == OHDR_SIGNATURE)
8398            .expect("enclosing object header");
8399        let cksum_at = (at + needle.len()..bytes.len() - 4)
8400            .find(|&e| {
8401                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
8402                    == checksum_metadata(&bytes[ohdr..e])
8403            })
8404            .expect("object header checksum");
8405
8406        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
8407        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
8408        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
8409        std::fs::write(&path, &bytes).unwrap();
8410
8411        let file = H5File::open(&path).unwrap();
8412        let err = file
8413            .dataset("labels")
8414            .unwrap()
8415            .read_strings()
8416            .unwrap_err()
8417            .to_string();
8418        assert!(err.contains("zero width"), "got: {err}");
8419        std::fs::remove_file(&path).ok();
8420    }
8421
8422    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
8423    #[test]
8424    fn read_strings_rejects_a_non_string_dataset() {
8425        let path = temp_path("read_strings_numeric");
8426        {
8427            let file = H5File::create(&path).unwrap();
8428            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8429            ds.write_raw(&[1i32, 2, 3]).unwrap();
8430            file.close().unwrap();
8431        }
8432        let file = H5File::open(&path).unwrap();
8433        let err = file
8434            .dataset("nums")
8435            .unwrap()
8436            .read_strings()
8437            .unwrap_err()
8438            .to_string();
8439        assert!(err.contains("only for string datasets"), "got: {err}");
8440        std::fs::remove_file(&path).ok();
8441    }
8442
8443    // ---- issue #6: random updates to vlen string datasets ------------------
8444
8445    /// One element changes; the extent and every other element stay as they
8446    /// were, on a contiguous vlen dataset.
8447    #[test]
8448    fn write_vlen_strings_slice_replaces_one_element() {
8449        let path = temp_path("vlen_slice_contig");
8450        {
8451            let file = H5File::create(&path).unwrap();
8452            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8453                .unwrap();
8454            file.close().unwrap();
8455        }
8456        {
8457            let file = H5File::open_rw(&path).unwrap();
8458            file.dataset_writer("notes")
8459                .unwrap()
8460                .write_vlen_strings_slice(1, &["replacement"])
8461                .unwrap();
8462            file.close().unwrap();
8463        }
8464        let file = H5File::open(&path).unwrap();
8465        let ds = file.dataset("notes").unwrap();
8466        assert_eq!(ds.shape(), vec![4]);
8467        assert_eq!(
8468            ds.read_vlen_strings().unwrap(),
8469            vec!["a", "replacement", "c", "d"]
8470        );
8471        std::fs::remove_file(&path).ok();
8472    }
8473
8474    /// The same on an appendable chunked dataset, across a reopen, over a range
8475    /// that spans a chunk boundary.
8476    #[test]
8477    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
8478        let path = temp_path("vlen_slice_chunked");
8479        {
8480            let file = H5File::create(&path).unwrap();
8481            file.create_appendable_vlen_dataset("notes", 2, None)
8482                .unwrap();
8483            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
8484            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
8485            file.append_vlen_strings("notes", &refs).unwrap();
8486            file.close().unwrap();
8487        }
8488        {
8489            // Elements 1..4 cross the 2-element chunk boundary twice.
8490            let file = H5File::open_rw(&path).unwrap();
8491            file.dataset_writer("notes")
8492                .unwrap()
8493                .write_vlen_strings_slice(1, &["x", "y", "z"])
8494                .unwrap();
8495            file.close().unwrap();
8496        }
8497        let file = H5File::open(&path).unwrap();
8498        let ds = file.dataset("notes").unwrap();
8499        assert_eq!(ds.shape(), vec![6]);
8500        assert_eq!(
8501            ds.read_vlen_strings().unwrap(),
8502            vec!["v0", "x", "y", "z", "v4", "v5"]
8503        );
8504        std::fs::remove_file(&path).ok();
8505    }
8506
8507    /// Elements the append buffer still holds are not on disk yet; the
8508    /// update flushes them to their chunks first, so the flush at close has
8509    /// nothing left to write the pre-update reference over.
8510    #[test]
8511    fn write_vlen_strings_slice_updates_buffered_elements() {
8512        let path = temp_path("vlen_slice_buffered");
8513        {
8514            let file = H5File::create(&path).unwrap();
8515            file.create_appendable_vlen_dataset("notes", 4, None)
8516                .unwrap();
8517            // 3 of a 4-element chunk: all three stay in the append buffer.
8518            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8519            file.dataset_writer("notes")
8520                .unwrap()
8521                .write_vlen_strings_slice(1, &["patched"])
8522                .unwrap();
8523            file.close().unwrap();
8524        }
8525        let file = H5File::open(&path).unwrap();
8526        assert_eq!(
8527            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8528            vec!["a", "patched", "c"]
8529        );
8530        std::fs::remove_file(&path).ok();
8531    }
8532
8533    /// A range past the end is rejected before anything is written, and an
8534    /// empty batch costs the file nothing — without the early return it would
8535    /// still allocate and write an empty global-heap collection.
8536    #[test]
8537    fn write_vlen_strings_slice_checks_its_range() {
8538        let build = |name: &str, empty_call: bool| {
8539            let path = temp_path(name);
8540            let file = H5File::create(&path).unwrap();
8541            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8542            let ds = file.dataset_writer("notes").unwrap();
8543            let err = ds
8544                .write_vlen_strings_slice(1, &["x", "y"])
8545                .unwrap_err()
8546                .to_string();
8547            assert!(
8548                err.contains("outside the dataset's 2 elements"),
8549                "got: {err}"
8550            );
8551            if empty_call {
8552                ds.write_vlen_strings_slice(0, &[]).unwrap();
8553            }
8554            file.close().unwrap();
8555            path
8556        };
8557
8558        let with_empty = build("vlen_slice_range", true);
8559        let control = build("vlen_slice_range_control", false);
8560        assert_eq!(
8561            std::fs::metadata(&with_empty).unwrap().len(),
8562            std::fs::metadata(&control).unwrap().len(),
8563            "the rejected and empty calls must leave the file untouched"
8564        );
8565
8566        let file = H5File::open(&with_empty).unwrap();
8567        assert_eq!(
8568            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8569            vec!["a", "b"]
8570        );
8571        std::fs::remove_file(&with_empty).ok();
8572        std::fs::remove_file(&control).ok();
8573    }
8574
8575    /// The element offset is one-dimensional, so a multi-dimensional dataset is
8576    /// rejected rather than silently indexed along the first axis.
8577    #[test]
8578    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
8579        let path = temp_path("vlen_slice_2d");
8580        let file = H5File::create(&path).unwrap();
8581        let ds = file
8582            .new_dataset::<u8>()
8583            .datatype(DatatypeMessage::vlen_string_utf8())
8584            .shape([2, 3])
8585            .create("grid")
8586            .unwrap();
8587        let err = ds
8588            .write_vlen_strings_slice(0, &["x"])
8589            .unwrap_err()
8590            .to_string();
8591        assert!(err.contains("1-dimension datasets"), "got: {err}");
8592        file.close().unwrap();
8593        std::fs::remove_file(&path).ok();
8594    }
8595
8596    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
8597    /// declares the ASCII character set would mislabel the bytes.
8598    #[test]
8599    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
8600        let path = temp_path("vlen_slice_ascii");
8601        let file = H5File::create(&path).unwrap();
8602        let ds = file
8603            .new_dataset::<u8>()
8604            .datatype(DatatypeMessage::vlen_string_ascii())
8605            .shape([3])
8606            .create("notes")
8607            .unwrap();
8608        let err = ds
8609            .write_vlen_strings_slice(0, &["ok", "안녕"])
8610            .unwrap_err()
8611            .to_string();
8612        assert!(
8613            err.contains("string 1") && err.contains("is not ASCII"),
8614            "got: {err}"
8615        );
8616        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
8617        file.close().unwrap();
8618        std::fs::remove_file(&path).ok();
8619    }
8620
8621    /// A numeric dataset is rejected: its elements are not vlen references and
8622    /// writing one would corrupt the column.
8623    #[test]
8624    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
8625        let path = temp_path("vlen_slice_numeric");
8626        let file = H5File::create(&path).unwrap();
8627        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8628        ds.write_raw(&[1i32, 2, 3]).unwrap();
8629        let err = ds
8630            .write_vlen_strings_slice(0, &["x"])
8631            .unwrap_err()
8632            .to_string();
8633        assert!(
8634            err.contains("only for variable-length string datasets"),
8635            "got: {err}"
8636        );
8637        file.close().unwrap();
8638        std::fs::remove_file(&path).ok();
8639    }
8640
8641    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
8642
8643    /// Repeatedly replacing the same element must not grow the file per
8644    /// update: the collection each update supersedes is freed and the next
8645    /// update's collection lands in that block. Without the release every
8646    /// update costs another `H5HG_MINALLOC` (4096) bytes.
8647    #[test]
8648    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
8649        let size_after = |updates: usize| {
8650            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
8651            let file = H5File::create(&path).unwrap();
8652            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8653            let ds = file.dataset_writer("notes").unwrap();
8654            for i in 0..updates {
8655                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
8656                    .unwrap();
8657            }
8658            file.close().unwrap();
8659            let n = std::fs::metadata(&path).unwrap().len();
8660            let read = H5File::open(&path).unwrap();
8661            assert_eq!(
8662                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8663                vec![format!("update {}", updates - 1), "b".to_string()]
8664            );
8665            drop(read);
8666            std::fs::remove_file(&path).ok();
8667            n
8668        };
8669
8670        // The allocator settles once a freed block is available to reuse, so
8671        // every count past that produces the same file.
8672        let settled = size_after(3);
8673        assert_eq!(size_after(20), settled, "20 updates against 3");
8674        assert_eq!(size_after(50), settled, "50 updates against 3");
8675    }
8676
8677    /// An empty string is stored as a real heap object under a reference whose
8678    /// sequence length is zero, so the release must go by the address, not the
8679    /// length — a length test strands the object and its collection forever.
8680    #[test]
8681    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
8682        let size_after = |updates: usize| {
8683            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
8684            let file = H5File::create(&path).unwrap();
8685            file.write_vlen_strings("notes", &["", "b"]).unwrap();
8686            let ds = file.dataset_writer("notes").unwrap();
8687            for _ in 0..updates {
8688                ds.write_vlen_strings_slice(0, &[""]).unwrap();
8689            }
8690            file.close().unwrap();
8691            let n = std::fs::metadata(&path).unwrap().len();
8692            let read = H5File::open(&path).unwrap();
8693            assert_eq!(
8694                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8695                vec!["".to_string(), "b".to_string()]
8696            );
8697            drop(read);
8698            std::fs::remove_file(&path).ok();
8699            n
8700        };
8701
8702        let settled = size_after(3);
8703        assert_eq!(size_after(20), settled, "20 empty updates against 3");
8704        assert_eq!(size_after(50), settled, "50 empty updates against 3");
8705    }
8706
8707    /// The elements the update does not name keep their strings, so freeing
8708    /// the superseded objects must not disturb the collection's survivors.
8709    #[test]
8710    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
8711        let path = temp_path("vlen_slice_heap_survivors");
8712        {
8713            let file = H5File::create(&path).unwrap();
8714            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8715                .unwrap();
8716            let ds = file.dataset_writer("notes").unwrap();
8717            // Two updates inside the one collection the create wrote, so the
8718            // second reads a collection the first already rewrote.
8719            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
8720            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
8721            file.close().unwrap();
8722        }
8723        let file = H5File::open(&path).unwrap();
8724        assert_eq!(
8725            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8726            vec!["a", "B", "c", "D"]
8727        );
8728        std::fs::remove_file(&path).ok();
8729    }
8730
8731    /// Replacing every element of a chunked dataset empties the collection the
8732    /// append wrote, and the file must still read back correctly after its
8733    /// block goes to the allocator.
8734    #[test]
8735    fn write_vlen_strings_slice_frees_an_emptied_collection() {
8736        let path = temp_path("vlen_slice_heap_emptied");
8737        {
8738            let file = H5File::create(&path).unwrap();
8739            file.create_appendable_vlen_dataset("notes", 2, None)
8740                .unwrap();
8741            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
8742                .unwrap();
8743            file.close().unwrap();
8744        }
8745        {
8746            let file = H5File::open_rw(&path).unwrap();
8747            file.dataset_writer("notes")
8748                .unwrap()
8749                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
8750                .unwrap();
8751            file.close().unwrap();
8752        }
8753        let file = H5File::open(&path).unwrap();
8754        assert_eq!(
8755            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8756            vec!["w", "x", "y", "z"]
8757        );
8758        std::fs::remove_file(&path).ok();
8759    }
8760
8761    /// A collection larger than the 4096-byte minimum must keep its size when
8762    /// an object leaves it. Re-encoding at the natural size instead shrinks
8763    /// what the header declares, so the block's tail stops being part of the
8764    /// collection and the eventual free returns less than was allocated —
8765    /// stranding the difference on every cycle.
8766    #[test]
8767    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
8768        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
8769        let size_after = |cycles: usize| {
8770            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
8771            let file = H5File::create(&path).unwrap();
8772            let seed: Vec<String> = "abcd".chars().map(big).collect();
8773            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
8774            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
8775            // is one collection well above it.
8776            file.write_vlen_strings("notes", &refs).unwrap();
8777            let ds = file.dataset_writer("notes").unwrap();
8778            for _ in 0..cycles {
8779                // Partially empty the collection, then finish it off: the
8780                // block is freed only after it has been rewritten once.
8781                let head = big('x');
8782                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
8783                let tail: Vec<String> = "yzw".chars().map(big).collect();
8784                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
8785                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
8786            }
8787            file.close().unwrap();
8788            let n = std::fs::metadata(&path).unwrap().len();
8789            let read = H5File::open(&path).unwrap();
8790            assert_eq!(
8791                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8792                vec![big('x'), big('y'), big('z'), big('w')]
8793            );
8794            drop(read);
8795            std::fs::remove_file(&path).ok();
8796            n
8797        };
8798
8799        let settled = size_after(4);
8800        assert_eq!(size_after(30), settled, "30 cycles against 4");
8801    }
8802
8803    /// An element still in the append buffer has never been on disk, so its
8804    /// superseded object has to be found in the buffer or it is stranded.
8805    #[test]
8806    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
8807        let path = temp_path("vlen_slice_heap_buffered");
8808        let file = H5File::create(&path).unwrap();
8809        file.create_appendable_vlen_dataset("notes", 4, None)
8810            .unwrap();
8811        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8812        let ds = file.dataset_writer("notes").unwrap();
8813        for i in 0..20 {
8814            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
8815                .unwrap();
8816        }
8817        file.close().unwrap();
8818
8819        let file = H5File::open(&path).unwrap();
8820        assert_eq!(
8821            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8822            vec!["a", "patch 19", "c"]
8823        );
8824        let size = std::fs::metadata(&path).unwrap().len();
8825        std::fs::remove_file(&path).ok();
8826        assert!(
8827            size < 20 * 4096,
8828            "20 buffered updates left {size} bytes, one collection per update"
8829        );
8830    }
8831
8832    /// Regression: a typed `write_slice` into rows the append buffer still
8833    /// held wrote the chunks, and the flush at close wrote the stale buffered
8834    /// rows back over it — write 99, read 50. The slice now flushes the
8835    /// buffer first, making the chunks the single authority for those rows.
8836    #[test]
8837    fn write_slice_into_the_buffered_tail_survives_close() {
8838        let path = temp_path("slice_into_buffered_tail");
8839        {
8840            let file = H5File::create(&path).unwrap();
8841            let ds = file
8842                .new_dataset::<i32>()
8843                .shape([0])
8844                .chunk(&[4])
8845                .max_shape(&[None])
8846                .create("d")
8847                .unwrap();
8848            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
8849            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
8850            ds.write_slice(&[4], &[1], &[99]).unwrap();
8851            file.close().unwrap();
8852        }
8853        {
8854            let file = H5File::open(&path).unwrap();
8855            let ds = file.dataset("d").unwrap();
8856            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
8857        }
8858        std::fs::remove_file(&path).ok();
8859    }
8860
8861    /// Extending a dataset while appends sit in the buffer must not move
8862    /// them: the buffer records the absolute row its frames belong to, so
8863    /// the flush at close lands them there, and the grown region reads as
8864    /// fill.
8865    #[test]
8866    fn extend_does_not_move_buffered_appends() {
8867        let path = temp_path("extend_keeps_buffered_rows");
8868        {
8869            let file = H5File::create(&path).unwrap();
8870            let ds = file
8871                .new_dataset::<i32>()
8872                .shape([0])
8873                .chunk(&[4])
8874                .max_shape(&[None])
8875                .create("d")
8876                .unwrap();
8877            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
8878            ds.extend(&[10]).unwrap();
8879            file.close().unwrap();
8880        }
8881        {
8882            let file = H5File::open(&path).unwrap();
8883            let ds = file.dataset("d").unwrap();
8884            assert_eq!(
8885                ds.read_raw::<i32>().unwrap(),
8886                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
8887            );
8888        }
8889        std::fs::remove_file(&path).ok();
8890    }
8891
8892    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
8893    /// dimensions) buffered fine but close() failed "not a chunked dataset"
8894    /// and lost the buffered rows — the append's chunk writes required the
8895    /// extensible-array index. They now go through the index-generic
8896    /// hyperslab engine.
8897    #[test]
8898    fn append_to_a_btree_v2_dataset_survives_close() {
8899        let path = temp_path("append_bt2_close");
8900        {
8901            let file = H5File::create(&path).unwrap();
8902            let ds = file
8903                .new_dataset::<i32>()
8904                .shape([0, 3])
8905                .chunk(&[4, 3])
8906                .max_shape(&[None, None])
8907                .create("d")
8908                .unwrap();
8909            // One buffered row, then a batch that crosses the chunk
8910            // boundary: 4 rows fill chunk band 0, one row stays buffered
8911            // for the flush at close.
8912            ds.append(&[1, 2, 3]).unwrap();
8913            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
8914            file.close().unwrap();
8915        }
8916        {
8917            let file = H5File::open(&path).unwrap();
8918            let ds = file.dataset("d").unwrap();
8919            assert_eq!(ds.shape(), vec![5, 3]);
8920            assert_eq!(
8921                ds.read_raw::<i32>().unwrap(),
8922                (1..=15).collect::<Vec<i32>>()
8923            );
8924        }
8925        std::fs::remove_file(&path).ok();
8926    }
8927
8928    /// A chunk row narrower than the frame row is legal geometry (libhdf5
8929    /// creates it); appended frames must be scattered across the row's
8930    /// tiles at the chunk stride, not packed at the frame stride.
8931    #[test]
8932    fn append_scatters_frames_across_narrow_chunk_tiles() {
8933        let path = temp_path("append_narrow_chunks");
8934        {
8935            let file = H5File::create(&path).unwrap();
8936            let ds = file
8937                .new_dataset::<i32>()
8938                .shape([0, 8])
8939                .chunk(&[2, 4])
8940                .max_shape(&[None, Some(8)])
8941                .create("d")
8942                .unwrap();
8943            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
8944            // row 2 is flushed partial at close.
8945            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
8946            file.close().unwrap();
8947        }
8948        {
8949            let file = H5File::open(&path).unwrap();
8950            let ds = file.dataset("d").unwrap();
8951            assert_eq!(ds.shape(), vec![3, 8]);
8952            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
8953        }
8954        std::fs::remove_file(&path).ok();
8955    }
8956
8957    /// A fixed-array dataset has no room to grow: appending must surface an
8958    /// error naming the chunk grid, not lose rows silently. (Before the
8959    /// index-generic append it failed as "not a chunked dataset".)
8960    #[test]
8961    fn append_to_a_full_fixed_array_dataset_errors() {
8962        let path = temp_path("append_fa_errors");
8963        let file = H5File::create(&path).unwrap();
8964        let ds = file
8965            .new_dataset::<i32>()
8966            .shape([4, 3])
8967            .chunk(&[2, 3])
8968            .create("d")
8969            .unwrap();
8970        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
8971        assert!(
8972            err.to_string().contains("chunk grid"),
8973            "unexpected error: {err}"
8974        );
8975        file.close().unwrap();
8976        std::fs::remove_file(&path).ok();
8977    }
8978
8979    /// A finite max_shape above the current shape used to be dropped on the
8980    /// fixed-array path: the array was sized from the current dims and the
8981    /// stored dataspace had no maximum, so growth failed. The array is now
8982    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
8983    /// fixed-max dataset appends up to its maximum and roundtrips.
8984    #[test]
8985    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
8986        let path = temp_path("fa_growable_dim0");
8987        {
8988            let file = H5File::create(&path).unwrap();
8989            let ds = file
8990                .new_dataset::<i32>()
8991                .shape([4, 3])
8992                .chunk(&[2, 3])
8993                .max_shape(&[Some(10), Some(3)])
8994                .create("d")
8995                .unwrap();
8996            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
8997            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
8998            file.close().unwrap();
8999        }
9000        {
9001            let file = H5File::open(&path).unwrap();
9002            let ds = file.dataset("d").unwrap();
9003            assert_eq!(ds.shape(), vec![6, 3]);
9004            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
9005        }
9006        std::fs::remove_file(&path).ok();
9007    }
9008
9009    /// The multiplier-dimension boundary: growing a dimension other than 0
9010    /// changes the current chunk grid but not the index grid. Chunk slots
9011    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
9012    /// chunks written before the extend are looked up under different
9013    /// indices after it.
9014    #[test]
9015    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
9016        let path = temp_path("fa_growable_dim1");
9017        {
9018            let file = H5File::create(&path).unwrap();
9019            let ds = file
9020                .new_dataset::<i32>()
9021                .shape([4, 3])
9022                .chunk(&[2, 3])
9023                .max_shape(&[Some(4), Some(9)])
9024                .create("d")
9025                .unwrap();
9026            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9027            ds.extend(&[4, 6]).unwrap();
9028            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
9029                .unwrap();
9030            file.close().unwrap();
9031        }
9032        {
9033            let file = H5File::open(&path).unwrap();
9034            let ds = file.dataset("d").unwrap();
9035            assert_eq!(ds.shape(), vec![4, 6]);
9036            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
9037            // then [12 + r*3 ..) from the second.
9038            let mut expect = Vec::new();
9039            for r in 0i32..4 {
9040                expect.extend((r * 3)..(r * 3 + 3));
9041                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
9042            }
9043            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
9044        }
9045        std::fs::remove_file(&path).ok();
9046    }
9047
9048    /// Growth boundaries: past the stored maximum is rejected, and a dataset
9049    /// without a stored maximum is fixed at its extent (libhdf5 defaults
9050    /// maxdims to dims at creation).
9051    #[test]
9052    fn extend_beyond_the_maximum_is_rejected() {
9053        let path = temp_path("extend_beyond_max");
9054        let file = H5File::create(&path).unwrap();
9055        let ds = file
9056            .new_dataset::<i32>()
9057            .shape([4, 3])
9058            .chunk(&[2, 3])
9059            .max_shape(&[Some(6), Some(3)])
9060            .create("d")
9061            .unwrap();
9062        ds.extend(&[6, 3]).unwrap();
9063        let err = ds.extend(&[8, 3]).unwrap_err();
9064        assert!(
9065            err.to_string().contains("exceeds the maximum"),
9066            "unexpected error: {err}"
9067        );
9068        file.close().unwrap();
9069        std::fs::remove_file(&path).ok();
9070    }
9071
9072    /// An unlimited dimension other than 0 has no fixed linear slot without
9073    /// libhdf5's extensible-array swizzling; `chunk_grid::linear_index` now
9074    /// implements that swizzle for any dimension, so this creates cleanly
9075    /// and every extend keeps writing new chunks to new slots, never
9076    /// re-addressing one already on disk.
9077    #[test]
9078    fn builder_accepts_an_unlimited_inner_dimension() {
9079        let path = temp_path("unlimited_inner_dim");
9080        let file = H5File::create(&path).unwrap();
9081        let ds = file
9082            .new_dataset::<i32>()
9083            .shape([4, 0])
9084            .chunk(&[2, 2])
9085            .max_shape(&[Some(4), None])
9086            .create("d")
9087            .unwrap();
9088        assert_eq!(ds.shape(), vec![4, 0]);
9089
9090        // Write, then extend and write again: if the linear index were
9091        // recomputed from the *current* extent instead of the maximum one,
9092        // the second extend would shift every slot number and the first
9093        // write's chunks would decode under the wrong coordinates below.
9094        ds.extend(&[4, 2]).unwrap();
9095        ds.write_slice(&[0, 0], &[4, 2], &[1, 2, 3, 4, 5, 6, 7, 8])
9096            .unwrap();
9097        ds.extend(&[4, 4]).unwrap();
9098        ds.write_slice(&[0, 2], &[4, 2], &[9, 10, 11, 12, 13, 14, 15, 16])
9099            .unwrap();
9100
9101        file.close().unwrap();
9102        let file = H5File::open(&path).unwrap();
9103        let ds = file.dataset("d").unwrap();
9104        assert_eq!(
9105            ds.read_slice::<i32>(&[0, 0], &[4, 4]).unwrap(),
9106            vec![1, 2, 9, 10, 3, 4, 11, 12, 5, 6, 13, 14, 7, 8, 15, 16]
9107        );
9108        std::fs::remove_file(&path).ok();
9109    }
9110
9111    /// Regression: a chunk wider than a fixed max dimension used to be
9112    /// accepted, and appends then packed rows at the chunk stride — writing
9113    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
9114    /// geometry at create (`H5D__chunk_construct`); so do we now.
9115    #[test]
9116    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
9117        let path = temp_path("builder_chunk_wider_than_max");
9118        let file = H5File::create(&path).unwrap();
9119        let err = match file
9120            .new_dataset::<i32>()
9121            .shape([0, 2])
9122            .chunk(&[2, 4])
9123            .max_shape(&[None, Some(2)])
9124            .create("v5")
9125        {
9126            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
9127            Err(e) => e,
9128        };
9129        assert!(
9130            err.to_string().contains("maximum dimension size"),
9131            "unexpected error: {err}"
9132        );
9133        file.close().unwrap();
9134        std::fs::remove_file(&path).ok();
9135    }
9136
9137    /// Boundary: `fits in T` vs `does not fit in T`, for both the too-large
9138    /// (u64::MAX → i64) and the negative-to-unsigned (−1 → u32) directions.
9139    #[test]
9140    fn numeric_int_checked_conversion_boundaries() {
9141        let path = temp_path("numeric_int_bounds");
9142        {
9143            let file = H5File::create(&path).unwrap();
9144            let ds = file.new_dataset::<u64>().shape([2]).create("u").unwrap();
9145            ds.write_raw(&[1u64, u64::MAX]).unwrap();
9146            let ds = file.new_dataset::<i32>().shape([2]).create("i").unwrap();
9147            ds.write_raw(&[-1i32, 5]).unwrap();
9148            file.close().unwrap();
9149        }
9150        let file = H5File::open(&path).unwrap();
9151
9152        let u = file.dataset("u").unwrap();
9153        assert_eq!(u.read_numeric_as::<u64>().unwrap(), vec![1, u64::MAX]);
9154        assert_eq!(
9155            u.read_numeric_as::<i128>().unwrap(),
9156            vec![1, i128::from(u64::MAX)]
9157        );
9158        let err = u.read_numeric_as::<i64>().unwrap_err();
9159        assert!(
9160            err.to_string()
9161                .contains("value 18446744073709551615 at element 1 does not fit in i64"),
9162            "unexpected error: {err}"
9163        );
9164
9165        let i = file.dataset("i").unwrap();
9166        assert_eq!(i.read_numeric_as::<i64>().unwrap(), vec![-1, 5]);
9167        let err = i.read_numeric_as::<u32>().unwrap_err();
9168        assert!(
9169            err.to_string()
9170                .contains("value -1 at element 0 does not fit in u32"),
9171            "unexpected error: {err}"
9172        );
9173        std::fs::remove_file(&path).ok();
9174    }
9175
9176    /// Boundary: f32 → f64 is exact widening; f64 → f32 is rejected.
9177    #[test]
9178    fn numeric_float_widening_only() {
9179        let path = temp_path("numeric_float");
9180        {
9181            let file = H5File::create(&path).unwrap();
9182            let ds = file.new_dataset::<f32>().shape([2]).create("f4").unwrap();
9183            ds.write_raw(&[1.5f32, -2.25]).unwrap();
9184            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9185            ds.write_raw(&[3.75f64]).unwrap();
9186            file.close().unwrap();
9187        }
9188        let file = H5File::open(&path).unwrap();
9189
9190        let f4 = file.dataset("f4").unwrap();
9191        assert_eq!(f4.read_numeric_as::<f32>().unwrap(), vec![1.5, -2.25]);
9192        assert_eq!(f4.read_numeric_as::<f64>().unwrap(), vec![1.5, -2.25]);
9193
9194        let f8 = file.dataset("f8").unwrap();
9195        assert_eq!(f8.read_numeric_as::<f64>().unwrap(), vec![3.75]);
9196        let err = f8.read_numeric_as::<f32>().unwrap_err();
9197        assert!(
9198            err.to_string().contains("narrowing"),
9199            "unexpected error: {err}"
9200        );
9201        std::fs::remove_file(&path).ok();
9202    }
9203
9204    /// Boundary: cross-class conversions (float ↔ integer) are rejected in
9205    /// both directions, and a non-numeric datatype is rejected at classify.
9206    #[test]
9207    fn numeric_cross_class_and_non_numeric_rejected() {
9208        let path = temp_path("numeric_cross_class");
9209        {
9210            let file = H5File::create(&path).unwrap();
9211            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9212            ds.write_raw(&[1.0f64]).unwrap();
9213            let ds = file.new_dataset::<i32>().shape([1]).create("i4").unwrap();
9214            ds.write_raw(&[7i32]).unwrap();
9215            file.write_vlen_strings("s", &["a", "b"]).unwrap();
9216            file.close().unwrap();
9217        }
9218        let file = H5File::open(&path).unwrap();
9219
9220        let err = file
9221            .dataset("f8")
9222            .unwrap()
9223            .read_numeric_as::<i64>()
9224            .unwrap_err();
9225        assert!(
9226            err.to_string().contains("floating-point dataset as i64"),
9227            "unexpected error: {err}"
9228        );
9229        let err = file
9230            .dataset("i4")
9231            .unwrap()
9232            .read_numeric_as::<f64>()
9233            .unwrap_err();
9234        assert!(
9235            err.to_string().contains("integer dataset as f64"),
9236            "unexpected error: {err}"
9237        );
9238        let err = file
9239            .dataset("s")
9240            .unwrap()
9241            .read_numeric_as::<i64>()
9242            .unwrap_err();
9243        assert!(
9244            err.to_string().contains("is not numeric"),
9245            "unexpected error: {err}"
9246        );
9247        std::fs::remove_file(&path).ok();
9248    }
9249
9250    /// Boundary: big-endian sources decode per the datatype's byte order.
9251    /// Unit-level (the writer only emits little-endian): feed `convert` a
9252    /// big-endian datatype plus big-endian bytes directly.
9253    #[test]
9254    fn numeric_big_endian_decode() {
9255        use super::numeric;
9256        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
9257
9258        let dt = DatatypeMessage::FixedPoint {
9259            size: 4,
9260            byte_order: ByteOrder::BigEndian,
9261            signed: true,
9262            bit_offset: 0,
9263            bit_precision: 32,
9264        };
9265        let mut raw = Vec::new();
9266        raw.extend_from_slice(&(-2i32).to_be_bytes());
9267        raw.extend_from_slice(&(100_000i32).to_be_bytes());
9268        let kind = numeric::classify(&dt).unwrap();
9269        assert_eq!(
9270            numeric::convert::<i64>(kind, &raw).unwrap(),
9271            vec![-2, 100_000]
9272        );
9273
9274        let dt = DatatypeMessage::FloatingPoint {
9275            size: 8,
9276            byte_order: ByteOrder::BigEndian,
9277            sign_location: 63,
9278            bit_offset: 0,
9279            bit_precision: 64,
9280            exponent_location: 52,
9281            exponent_size: 11,
9282            mantissa_location: 0,
9283            mantissa_size: 52,
9284            exponent_bias: 1023,
9285        };
9286        let raw = (-2.25f64).to_be_bytes();
9287        let kind = numeric::classify(&dt).unwrap();
9288        assert_eq!(numeric::convert::<f64>(kind, &raw).unwrap(), vec![-2.25]);
9289    }
9290
9291    /// `H5Attribute::read_numeric` validates the stored datatype before
9292    /// reinterpreting bytes: cross-width, cross-class, and non-numeric
9293    /// attributes error instead of returning bit-garbage, while the exact
9294    /// type and the HBool / complex-compound paths keep working.
9295    #[test]
9296    fn attr_read_numeric_validates_datatype() {
9297        use crate::types::{Complex64, HBool, VarLenUnicode};
9298        let path = temp_path("attr_read_numeric_validate");
9299        {
9300            let file = H5File::create(&path).unwrap();
9301            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9302            ds.write_raw(&[1.0f32; 2]).unwrap();
9303            let a = ds.new_attr::<f64>().shape(()).create("f8").unwrap();
9304            a.write_numeric(&1.5f64).unwrap();
9305            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9306            a.write_numeric(&-7i32).unwrap();
9307            let a = ds.new_attr::<HBool>().shape(()).create("b").unwrap();
9308            a.write_numeric(&HBool::from(true)).unwrap();
9309            let a = ds.new_attr::<Complex64>().shape(()).create("z").unwrap();
9310            a.write_numeric(&Complex64 { re: 1.0, im: -2.0 }).unwrap();
9311            let a = ds
9312                .new_attr::<VarLenUnicode>()
9313                .shape(())
9314                .create("s")
9315                .unwrap();
9316            a.write_scalar(&VarLenUnicode("text".into())).unwrap();
9317            file.close().unwrap();
9318        }
9319        let file = H5File::open(&path).unwrap();
9320        let ds = file.dataset("d").unwrap();
9321
9322        let f8 = ds.attr("f8").unwrap();
9323        assert_eq!(f8.read_numeric::<f64>().unwrap(), 1.5);
9324        // Previously returned the low half of the f64 image as an f32.
9325        let err = f8.read_numeric::<f32>().unwrap_err();
9326        assert!(
9327            err.to_string().contains("read_numeric_as"),
9328            "unexpected error: {err}"
9329        );
9330        assert!(f8.read_numeric::<i64>().is_err());
9331
9332        let i4 = ds.attr("i4").unwrap();
9333        assert_eq!(i4.read_numeric::<i32>().unwrap(), -7);
9334        assert!(i4.read_numeric::<u32>().is_err());
9335
9336        assert!(bool::from(
9337            ds.attr("b").unwrap().read_numeric::<HBool>().unwrap()
9338        ));
9339        let z = ds.attr("z").unwrap().read_numeric::<Complex64>().unwrap();
9340        assert_eq!((z.re, z.im), (1.0, -2.0));
9341
9342        // A vlen string attribute: read_numeric used to transmute the heap
9343        // reference bytes into the requested type.
9344        let s = ds.attr("s").unwrap();
9345        assert!(s.read_numeric::<f64>().is_err());
9346        assert!(s.read_numeric_as::<f64>().is_err());
9347        std::fs::remove_file(&path).ok();
9348    }
9349
9350    /// The attribute conversion read applies the dataset rules: checked
9351    /// int → int naming index and value on overflow, widening-only floats,
9352    /// cross-class rejected; an array attribute converts every element.
9353    #[test]
9354    fn attr_read_numeric_as_converts() {
9355        let path = temp_path("attr_read_numeric_as");
9356        {
9357            let file = H5File::create(&path).unwrap();
9358            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9359            ds.write_raw(&[1.0f32; 2]).unwrap();
9360            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9361            a.write_numeric(&-7i32).unwrap();
9362            let a = ds.new_attr::<u64>().shape(()).create("u8max").unwrap();
9363            a.write_numeric(&u64::MAX).unwrap();
9364            let a = ds.new_attr::<i16>().shape([3]).create("arr").unwrap();
9365            a.write_array(&[1i16, -2, 3]).unwrap();
9366            file.close().unwrap();
9367        }
9368        let file = H5File::open(&path).unwrap();
9369        let ds = file.dataset("d").unwrap();
9370        assert_eq!(
9371            ds.attr("i4").unwrap().read_numeric_as::<i64>().unwrap(),
9372            vec![-7]
9373        );
9374        let err = ds
9375            .attr("u8max")
9376            .unwrap()
9377            .read_numeric_as::<i64>()
9378            .unwrap_err();
9379        assert!(
9380            err.to_string().contains("does not fit in i64"),
9381            "unexpected error: {err}"
9382        );
9383        assert_eq!(
9384            ds.attr("arr").unwrap().read_numeric_as::<i32>().unwrap(),
9385            vec![1, -2, 3]
9386        );
9387        assert!(ds.attr("i4").unwrap().read_numeric_as::<f64>().is_err());
9388        std::fs::remove_file(&path).ok();
9389    }
9390
9391    /// The hyperslab variant applies the same conversion to a sub-selection.
9392    #[test]
9393    fn numeric_slice_conversion() {
9394        let path = temp_path("numeric_slice");
9395        {
9396            let file = H5File::create(&path).unwrap();
9397            let ds = file.new_dataset::<i16>().shape([2, 3]).create("m").unwrap();
9398            ds.write_raw(&[1i16, 2, 3, 4, 5, 6]).unwrap();
9399            file.close().unwrap();
9400        }
9401        let file = H5File::open(&path).unwrap();
9402        let m = file.dataset("m").unwrap();
9403        assert_eq!(
9404            m.read_numeric_slice_as::<i32>(&[0, 1], &[2, 2]).unwrap(),
9405            vec![2, 3, 5, 6]
9406        );
9407        std::fs::remove_file(&path).ok();
9408    }
9409
9410    /// A NULL dataspace round-trips through the public API as `is_null() ==
9411    /// true`, `shape() == []`, and `read_raw_bytes()` empty — and stays
9412    /// distinguishable from a scalar dataset, which shares the same empty
9413    /// `shape()` but holds exactly one element.
9414    #[test]
9415    fn null_dataspace_distinct_from_scalar() {
9416        let path = temp_path("null_vs_scalar");
9417        {
9418            let file = H5File::create(&path).unwrap();
9419            file.new_dataset::<i32>().null().create("empty").unwrap();
9420            let scalar = file.new_dataset::<i32>().scalar().create("scalar").unwrap();
9421            scalar.write_raw(&[42i32]).unwrap();
9422            file.close().unwrap();
9423        }
9424        let file = H5File::open(&path).unwrap();
9425
9426        let empty = file.dataset("empty").unwrap();
9427        assert!(empty.is_null());
9428        assert_eq!(empty.shape(), Vec::<usize>::new());
9429        assert_eq!(empty.total_elements(), 0);
9430        assert_eq!(empty.read_raw_bytes().unwrap(), Vec::<u8>::new());
9431
9432        let scalar = file.dataset("scalar").unwrap();
9433        assert!(!scalar.is_null());
9434        assert_eq!(scalar.shape(), Vec::<usize>::new());
9435        assert_eq!(scalar.total_elements(), 1);
9436        assert_eq!(scalar.read_raw::<i32>().unwrap(), vec![42]);
9437
9438        std::fs::remove_file(&path).ok();
9439    }
9440
9441    /// A NULL dataspace dataset rejects writes outright — there is nothing
9442    /// to write into — rather than silently accepting a scalar-shaped
9443    /// write against unallocated storage.
9444    #[test]
9445    fn null_dataspace_rejects_writes() {
9446        let path = temp_path("null_write_rejected");
9447        let file = H5File::create(&path).unwrap();
9448        let ds = file.new_dataset::<i32>().null().create("empty").unwrap();
9449        assert!(ds.write_raw(&[1i32]).is_err());
9450        assert!(ds.write_raw_bytes(&[0u8; 4]).is_err());
9451        file.close().unwrap();
9452        std::fs::remove_file(&path).ok();
9453    }
9454
9455    /// `.null()` combined with `.chunk()` or a fill value is rejected at
9456    /// `create()` rather than silently dropping the conflicting option —
9457    /// a NULL dataspace can never be chunked or filtered upstream.
9458    #[test]
9459    fn null_dataspace_rejects_chunking_and_fill_value() {
9460        let path = temp_path("null_chunk_rejected");
9461        let file = H5File::create(&path).unwrap();
9462        assert!(file
9463            .new_dataset::<i32>()
9464            .null()
9465            .chunk(&[4])
9466            .create("a")
9467            .is_err());
9468        assert!(file
9469            .new_dataset::<i32>()
9470            .null()
9471            .fill_value(7i32)
9472            .create("b")
9473            .is_err());
9474        file.close().unwrap();
9475        std::fs::remove_file(&path).ok();
9476    }
9477
9478    /// A committed type is resolved before the dataset is created, so a name
9479    /// that is not one — or one paired with object references, which would
9480    /// make the stored type disagree with the payload — leaves no dataset
9481    /// behind.
9482    #[test]
9483    fn a_committed_type_that_cannot_be_shared_creates_no_dataset() {
9484        use crate::format::messages::datatype::DatatypeMessage;
9485
9486        let path = temp_path("committed_refused");
9487        let file = H5File::create(&path).unwrap();
9488        file.commit_datatype("t", DatatypeMessage::i32_type())
9489            .unwrap();
9490
9491        assert!(file
9492            .new_dataset::<i32>()
9493            .committed_type("absent")
9494            .shape([2usize])
9495            .create("a")
9496            .is_err());
9497        assert!(file
9498            .new_dataset::<u64>()
9499            .committed_type("t")
9500            .object_references()
9501            .shape([2usize])
9502            .create("b")
9503            .is_err());
9504        // A dataset already exists under that name, so the type cannot take
9505        // it either.
9506        file.new_dataset::<i32>()
9507            .shape([2usize])
9508            .create("taken")
9509            .unwrap();
9510        assert!(file
9511            .commit_datatype("taken", DatatypeMessage::i32_type())
9512            .is_err());
9513        assert!(file
9514            .commit_datatype("t", DatatypeMessage::f64_type())
9515            .is_err());
9516
9517        assert_eq!(file.dataset_names(), vec!["taken".to_string()]);
9518        assert_eq!(file.named_datatype_names(), vec!["t".to_string()]);
9519        file.close().unwrap();
9520        std::fs::remove_file(&path).ok();
9521    }
9522
9523    /// Deleting the group that named a committed datatype takes the name with
9524    /// it: nothing in the file reaches the type, so it is not written, the
9525    /// name is free again, and it can no longer be shared by that name.
9526    #[test]
9527    fn deleting_a_group_takes_the_committed_datatypes_it_named() {
9528        use crate::format::messages::datatype::DatatypeMessage;
9529
9530        let path = temp_path("committed_group_deleted");
9531        {
9532            let file = H5File::create(&path).unwrap();
9533            let types = file.create_group("types").unwrap();
9534            types
9535                .commit_datatype("t", DatatypeMessage::i32_type())
9536                .unwrap();
9537            assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9538
9539            file.delete_group("types").unwrap();
9540            assert!(file.named_datatype_names().is_empty());
9541            assert!(file
9542                .new_dataset::<i32>()
9543                .committed_type("types/t")
9544                .shape([2usize])
9545                .create("d")
9546                .is_err());
9547
9548            // The name is free, so a new group may take it back.
9549            let types = file.create_group("types").unwrap();
9550            types
9551                .commit_datatype("t", DatatypeMessage::f64_type())
9552                .unwrap();
9553            file.close().unwrap();
9554        }
9555        let file = H5File::open(&path).unwrap();
9556        assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9557        assert_eq!(
9558            file.named_datatype("types/t").unwrap().datatype().unwrap(),
9559            crate::format::messages::datatype::DatatypeMessage::f64_type()
9560        );
9561        drop(file);
9562        std::fs::remove_file(&path).ok();
9563    }
9564
9565    /// The layout class the reader sees for `name`, plus the image a compact
9566    /// layout carries — what distinguishes compact storage from contiguous
9567    /// storage that happens to hold the same bytes.
9568    fn compact_image(path: &std::path::Path, name: &str) -> Option<Vec<u8>> {
9569        use crate::format::messages::data_layout::DataLayoutMessage;
9570        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9571        match &reader.dataset_info(name).unwrap().layout {
9572            DataLayoutMessage::Compact { data } => Some(data.clone()),
9573            other => panic!("{name}: expected a compact layout, got {other:?}"),
9574        }
9575    }
9576
9577    /// `.compact()` puts the raw data inside the data layout message: the
9578    /// dataset has no data block of its own, and the image the layout carries
9579    /// is what a read returns.
9580    #[test]
9581    fn a_compact_dataset_stores_its_data_in_the_layout_message() {
9582        let path = temp_path("compact_roundtrip");
9583        let values: Vec<i32> = (0..16).collect();
9584        {
9585            let file = H5File::create(&path).unwrap();
9586            file.new_dataset::<i32>()
9587                .shape([16usize])
9588                .compact()
9589                .create("d")
9590                .unwrap()
9591                .write_raw(&values)
9592                .unwrap();
9593            file.close().unwrap();
9594        }
9595
9596        let image = compact_image(&path, "d").unwrap();
9597        assert_eq!(
9598            image,
9599            values
9600                .iter()
9601                .flat_map(|v| v.to_le_bytes())
9602                .collect::<Vec<_>>()
9603        );
9604
9605        let file = H5File::open(&path).unwrap();
9606        let ds = file.dataset("d").unwrap();
9607        assert_eq!(ds.shape(), vec![16]);
9608        assert_eq!(ds.read_raw::<i32>().unwrap(), values);
9609        std::fs::remove_file(&path).ok();
9610    }
9611
9612    /// A compact dataset created inside a group is linked from that group,
9613    /// not from the root: its create path goes through the same parent
9614    /// resolution every other layout uses.
9615    #[test]
9616    fn a_compact_dataset_lands_in_its_group() {
9617        let path = temp_path("compact_in_group");
9618        {
9619            let file = H5File::create(&path).unwrap();
9620            let g = file.root_group().create_group("g").unwrap();
9621            g.new_dataset::<u8>()
9622                .shape([4usize])
9623                .compact()
9624                .create("d")
9625                .unwrap()
9626                .write_raw(&[1u8, 2, 3, 4])
9627                .unwrap();
9628            file.close().unwrap();
9629        }
9630        assert_eq!(compact_image(&path, "/g/d").unwrap(), vec![1u8, 2, 3, 4]);
9631
9632        let file = H5File::open(&path).unwrap();
9633        assert_eq!(
9634            file.dataset("/g/d").unwrap().read_raw::<u8>().unwrap(),
9635            vec![1u8, 2, 3, 4]
9636        );
9637        std::fs::remove_file(&path).ok();
9638    }
9639
9640    /// A compact dataset's storage is the image itself, so the fill value has
9641    /// to be tiled into it at create — `H5D__compact_fill`'s job. An unwritten
9642    /// element must read back as the fill value, not as zero.
9643    #[test]
9644    fn an_unwritten_compact_dataset_reads_back_as_its_fill_value() {
9645        let path = temp_path("compact_fill");
9646        {
9647            let file = H5File::create(&path).unwrap();
9648            file.new_dataset::<i32>()
9649                .shape([4usize])
9650                .compact()
9651                .fill_value(-7i32)
9652                .create("d")
9653                .unwrap();
9654            file.close().unwrap();
9655        }
9656        let file = H5File::open(&path).unwrap();
9657        assert_eq!(
9658            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9659            vec![-7i32; 4]
9660        );
9661        std::fs::remove_file(&path).ok();
9662    }
9663
9664    /// The image is the layout message, so anything that rewrites the header
9665    /// rewrites the data with it. Reopening and attaching an attribute makes
9666    /// the header stale; the rebuilt one must still carry the image rather
9667    /// than fall back to an unallocated contiguous layout.
9668    #[test]
9669    fn a_reopened_compact_dataset_keeps_its_image() {
9670        let path = temp_path("compact_reopen");
9671        let values: Vec<i32> = (100..108).collect();
9672        {
9673            let file = H5File::create(&path).unwrap();
9674            file.new_dataset::<i32>()
9675                .shape([8usize])
9676                .compact()
9677                .create("d")
9678                .unwrap()
9679                .write_raw(&values)
9680                .unwrap();
9681            file.close().unwrap();
9682        }
9683        {
9684            let file = H5File::open_rw(&path).unwrap();
9685            file.dataset_writer("d")
9686                .unwrap()
9687                .new_attr::<i32>()
9688                .shape(())
9689                .create("note")
9690                .unwrap()
9691                .write_numeric(&1i32)
9692                .unwrap();
9693            file.close().unwrap();
9694        }
9695        assert_eq!(
9696            compact_image(&path, "d").unwrap(),
9697            values
9698                .iter()
9699                .flat_map(|v| v.to_le_bytes())
9700                .collect::<Vec<_>>()
9701        );
9702
9703        let file = H5File::open(&path).unwrap();
9704        assert_eq!(
9705            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9706            values
9707        );
9708        std::fs::remove_file(&path).ok();
9709    }
9710
9711    /// The ceiling is what a data layout message can hold, so it is checked
9712    /// in bytes and names them: the largest image that fits is accepted and
9713    /// one element more is refused.
9714    #[test]
9715    fn the_compact_ceiling_is_checked_in_bytes() {
9716        let path = temp_path("compact_ceiling");
9717        let file = H5File::create(&path).unwrap();
9718
9719        let fits = crate::MAX_COMPACT_DATA / 4;
9720        file.new_dataset::<i32>()
9721            .shape([fits])
9722            .compact()
9723            .create("fits")
9724            .unwrap();
9725
9726        let over = fits + 1;
9727        let err = match file
9728            .new_dataset::<i32>()
9729            .shape([over])
9730            .compact()
9731            .create("over")
9732        {
9733            Err(e) => e.to_string(),
9734            Ok(_) => panic!("an image {} bytes wide must be refused", over * 4),
9735        };
9736        assert!(
9737            err.contains(&(over * 4).to_string())
9738                && err.contains(&crate::MAX_COMPACT_DATA.to_string()),
9739            "the error must name both sizes: {err}"
9740        );
9741
9742        file.close().unwrap();
9743        std::fs::remove_file(&path).ok();
9744    }
9745
9746    /// Compact storage has no chunk grid to filter and no room to grow, so
9747    /// each conflicting option is refused at `create()` rather than silently
9748    /// overriding the layout the way `H5Pset_chunk` does.
9749    #[test]
9750    fn compact_rejects_chunking_filters_growth_and_a_null_dataspace() {
9751        let path = temp_path("compact_rejects");
9752        let file = H5File::create(&path).unwrap();
9753        assert!(file
9754            .new_dataset::<i32>()
9755            .shape([4usize])
9756            .compact()
9757            .chunk(&[4])
9758            .create("a")
9759            .is_err());
9760        assert!(file
9761            .new_dataset::<i32>()
9762            .shape([4usize])
9763            .compact()
9764            .deflate(4)
9765            .create("b")
9766            .is_err());
9767        assert!(file
9768            .new_dataset::<i32>()
9769            .shape([4usize])
9770            .compact()
9771            .max_shape(&[None])
9772            .create("c")
9773            .is_err());
9774        assert!(file
9775            .new_dataset::<i32>()
9776            .shape([4usize])
9777            .compact()
9778            .max_shape(&[Some(8)])
9779            .create("d")
9780            .is_err());
9781        assert!(file
9782            .new_dataset::<i32>()
9783            .null()
9784            .compact()
9785            .create("e")
9786            .is_err());
9787        file.close().unwrap();
9788        std::fs::remove_file(&path).ok();
9789    }
9790
9791    /// The filter pipeline the reader decodes from `name`'s header.
9792    fn stored_pipeline(
9793        path: &std::path::Path,
9794        name: &str,
9795    ) -> crate::format::messages::filter::FilterPipeline {
9796        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9797        reader
9798            .dataset_info(name)
9799            .unwrap()
9800            .filter_pipeline
9801            .clone()
9802            .unwrap_or_else(|| panic!("{name}: no filter pipeline"))
9803    }
9804
9805    /// Shuffle is a permutation, not a compressor, so it is a pipeline on its
9806    /// own — `H5Pset_shuffle` with nothing behind it. Its stage must reach the
9807    /// header, and the reader must unpermute what it wrote.
9808    #[test]
9809    fn shuffle_alone_is_a_filter_pipeline() {
9810        use crate::format::messages::filter::{FilterPipeline, FILTER_SHUFFLE};
9811        let path = temp_path("shuffle_alone");
9812        let values: Vec<i32> = (0..64).collect();
9813        {
9814            let file = H5File::create(&path).unwrap();
9815            file.new_dataset::<i32>()
9816                .shape([64usize])
9817                .chunk(&[16])
9818                .shuffle()
9819                .create("d")
9820                .unwrap()
9821                .write_raw(&values)
9822                .unwrap();
9823            file.close().unwrap();
9824        }
9825        let pipeline = stored_pipeline(&path, "d");
9826        assert_eq!(pipeline, FilterPipeline::shuffle(4));
9827        assert_eq!(pipeline.filters[0].id, FILTER_SHUFFLE);
9828
9829        let file = H5File::open(&path).unwrap();
9830        assert_eq!(
9831            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9832            values
9833        );
9834        std::fs::remove_file(&path).ok();
9835    }
9836
9837    /// `.shuffle()` and `.deflate()` are separate stages that compose, and
9838    /// `.shuffle_deflate()` is the shorthand for both — one pipeline, built
9839    /// once, whichever way it was asked for.
9840    #[cfg(feature = "deflate")]
9841    #[test]
9842    fn shuffle_composes_with_deflate() {
9843        let path = temp_path("shuffle_then_deflate");
9844        let values: Vec<i32> = (0..64).collect();
9845        {
9846            let file = H5File::create(&path).unwrap();
9847            for (name, ds) in [
9848                ("split", file.new_dataset::<i32>().shuffle().deflate(6)),
9849                ("combined", file.new_dataset::<i32>().shuffle_deflate(6)),
9850            ] {
9851                ds.shape([64usize])
9852                    .chunk(&[16])
9853                    .create(name)
9854                    .unwrap()
9855                    .write_raw(&values)
9856                    .unwrap();
9857            }
9858            file.close().unwrap();
9859        }
9860        assert_eq!(
9861            stored_pipeline(&path, "split"),
9862            crate::format::messages::filter::FilterPipeline::shuffle_deflate(4, 6)
9863        );
9864        assert_eq!(
9865            stored_pipeline(&path, "split"),
9866            stored_pipeline(&path, "combined")
9867        );
9868
9869        let file = H5File::open(&path).unwrap();
9870        for name in ["split", "combined"] {
9871            assert_eq!(
9872                file.dataset(name).unwrap().read_raw::<i32>().unwrap(),
9873                values,
9874                "{name}"
9875            );
9876        }
9877        std::fs::remove_file(&path).ok();
9878    }
9879
9880    /// The width shuffle permutes by is the stored element's, which a
9881    /// `datatype` override moves away from the carrier type `T`: recording
9882    /// `T`'s width would permute a 4-byte element as four 1-byte ones and
9883    /// hand libhdf5 a chunk it unshuffles into different bytes.
9884    #[test]
9885    fn shuffle_records_the_stored_element_width() {
9886        use crate::format::messages::datatype::DatatypeMessage;
9887        use crate::format::messages::filter::FilterPipeline;
9888        let path = temp_path("shuffle_override_width");
9889        let bytes: Vec<u8> = (0..16u8).collect();
9890        {
9891            let file = H5File::create(&path).unwrap();
9892            file.new_dataset::<u8>()
9893                .datatype(DatatypeMessage::i32_type())
9894                .shape([4usize])
9895                .chunk(&[4])
9896                .shuffle()
9897                .create("d")
9898                .unwrap()
9899                .write_raw_bytes(&bytes)
9900                .unwrap();
9901            file.close().unwrap();
9902        }
9903        assert_eq!(stored_pipeline(&path, "d"), FilterPipeline::shuffle(4));
9904
9905        let file = H5File::open(&path).unwrap();
9906        assert_eq!(
9907            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9908            bytes
9909                .chunks(4)
9910                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
9911                .collect::<Vec<_>>()
9912        );
9913        std::fs::remove_file(&path).ok();
9914    }
9915
9916    /// A write into a virtual dataset is refused by name rather than landing
9917    /// somewhere no reader would look: libhdf5 pushes such a write through
9918    /// the mapping into the source dataset (`H5D__virtual_write`), which this
9919    /// writer does not do.
9920    #[test]
9921    fn a_virtual_dataset_refuses_every_write() {
9922        use crate::Selection;
9923        let path = temp_path("vds_write_refused");
9924        let file = H5File::create(&path).unwrap();
9925        let ds = file
9926            .new_dataset::<i32>()
9927            .shape([16usize])
9928            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
9929            .create("vds")
9930            .unwrap();
9931        for err in [
9932            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap_err(),
9933            ds.write_slice(&[0], &[2], &[1i32, 2]).unwrap_err(),
9934        ] {
9935            let msg = err.to_string();
9936            assert!(msg.contains("virtual dataset"), "{msg}");
9937        }
9938        file.close().unwrap();
9939        std::fs::remove_file(&path).ok();
9940    }
9941
9942    /// A `%b` substitution only means something when the virtual selection is
9943    /// unlimited and the source selection is not: that is the shape where
9944    /// each block draws from a different source dataset. On any other mapping
9945    /// there is only one block, so `H5D_virtual_check_mapping_post` refuses
9946    /// the specifier — and an illegal conversion is refused wherever it
9947    /// appears.
9948    #[test]
9949    fn a_printf_source_name_needs_the_mapping_shape_that_uses_it() {
9950        use crate::Selection;
9951        let path = temp_path("vds_printf");
9952        let file = H5File::create(&path).unwrap();
9953        for (f, d) in [("src_%b.h5", "src"), ("src.h5", "block_%b")] {
9954            let err = match file
9955                .new_dataset::<i32>()
9956                .shape([16usize])
9957                .virtual_mapping(Selection::All, f, d, Selection::All)
9958                .create("vds")
9959            {
9960                Ok(_) => panic!("a bounded mapping has one block, so %b names nothing"),
9961                Err(e) => e.to_string(),
9962            };
9963            assert!(err.contains("printf specifier"), "{err}");
9964        }
9965        // `%z` is not a conversion libhdf5 has, in any mapping shape.
9966        let err = match file
9967            .new_dataset::<i32>()
9968            .shape([1usize, 2])
9969            .max_shape(&[None, Some(2)])
9970            .virtual_mapping(unlimited_rows(), "src_%z.h5", "src", Selection::All)
9971            .create("vds_bad")
9972        {
9973            Ok(_) => panic!("%z is not a legal conversion"),
9974            Err(e) => e.to_string(),
9975        };
9976        assert!(err.contains("invalid format specifier"), "{err}");
9977        file.close().unwrap();
9978        std::fs::remove_file(&path).ok();
9979    }
9980
9981    /// A printf mapping stitches one source dataset per block of its
9982    /// unlimited virtual selection, and the extent stops at the first block
9983    /// with no source (`H5D__virtual_set_extent_unlim`'s printf arm, at the
9984    /// default `H5D_VDS_LAST_AVAILABLE` view and `printf_gap` 0).
9985    #[test]
9986    fn a_printf_mapping_stitches_one_source_per_block() {
9987        use crate::Selection;
9988        let path = temp_path("vds_printf_blocks");
9989        {
9990            let file = H5File::create(&path).unwrap();
9991            for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
9992                file.new_dataset::<i32>()
9993                    .shape([2usize])
9994                    .create(&format!("block{b}"))
9995                    .unwrap()
9996                    .write_raw(&[base, base + 1])
9997                    .unwrap();
9998            }
9999            file.new_dataset::<i32>()
10000                .shape([1usize, 2])
10001                .max_shape(&[None, Some(2)])
10002                .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10003                .create("vds")
10004                .unwrap();
10005            file.close().unwrap();
10006        }
10007        let file = H5File::open(&path).unwrap();
10008        let ds = file.dataset("vds").unwrap();
10009        // `block2` is missing, so `block3` is never reached: two rows.
10010        assert_eq!(ds.shape(), vec![2, 2]);
10011        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10012        drop(file);
10013        std::fs::remove_file(&path).ok();
10014    }
10015
10016    /// A mapping whose source cannot be opened reads back as the fill value,
10017    /// not as an error: `H5D__virtual_open_source_dset` accepts a null source
10018    /// file and clears the error stack for a missing source dataset
10019    /// (H5Dvirtual.c:877-909), so `H5D__virtual_read_one` finds no projected
10020    /// memory space and reads nothing for it (H5Dvirtual.c:2661-2665).
10021    #[test]
10022    fn a_source_that_cannot_be_opened_reads_as_the_fill_value() {
10023        use crate::{Hyperslab, HyperslabBlock, Selection};
10024        let block = |start: u64, end: u64| Selection::Hyperslab {
10025            rank: 1,
10026            form: Hyperslab::Blocks(vec![HyperslabBlock {
10027                start: vec![start],
10028                end: vec![end],
10029            }]),
10030        };
10031        let path = temp_path("vds_absent_source");
10032        {
10033            let file = H5File::create(&path).unwrap();
10034            file.new_dataset::<i32>()
10035                .shape([4usize])
10036                .create("here")
10037                .unwrap()
10038                .write_raw(&[1i32, 2, 3, 4])
10039                .unwrap();
10040            file.new_dataset::<i32>()
10041                .shape([12usize])
10042                .fill_value(-3i32)
10043                .virtual_mapping(block(0, 3), ".", "here", block(0, 3))
10044                // A dataset that is not in this file.
10045                .virtual_mapping(block(4, 7), ".", "absent", block(0, 3))
10046                // A file that does not exist beside this one.
10047                .virtual_mapping(block(8, 11), "no_such_vds_source.h5", "src", block(0, 3))
10048                .create("vds")
10049                .unwrap();
10050            file.close().unwrap();
10051        }
10052        let file = H5File::open(&path).unwrap();
10053        let ds = file.dataset("vds").unwrap();
10054        assert_eq!(
10055            ds.read_raw::<i32>().unwrap(),
10056            vec![1, 2, 3, 4, -3, -3, -3, -3, -3, -3, -3, -3]
10057        );
10058        // The same rule on the slice path, which stitches the whole image
10059        // before extracting the region.
10060        assert_eq!(
10061            ds.read_slice::<i32>(&[2], &[6]).unwrap(),
10062            vec![3, 4, -3, -3, -3, -3]
10063        );
10064        drop(file);
10065        std::fs::remove_file(&path).ok();
10066    }
10067
10068    /// `%%` is an escaped literal `%`, not a substitution: the mapping is an
10069    /// ordinary bounded one, and the source it resolves against is the name
10070    /// with a single `%` in it.
10071    #[test]
10072    fn an_escaped_percent_is_a_literal_in_a_source_name() {
10073        use crate::Selection;
10074        let path = temp_path("vds_escaped_pct");
10075        {
10076            let file = H5File::create(&path).unwrap();
10077            file.new_dataset::<i32>()
10078                .shape([4usize])
10079                .create("od%d")
10080                .unwrap()
10081                .write_raw(&[5i32, 6, 7, 8])
10082                .unwrap();
10083            file.new_dataset::<i32>()
10084                .shape([4usize])
10085                .virtual_mapping(Selection::All, ".", "od%%d", Selection::All)
10086                .create("vds")
10087                .unwrap();
10088            file.close().unwrap();
10089        }
10090        let file = H5File::open(&path).unwrap();
10091        let ds = file.dataset("vds").unwrap();
10092        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![5, 6, 7, 8]);
10093        // The stored name keeps its escape; only the resolution unescapes.
10094        assert_eq!(ds.virtual_mappings().unwrap()[0].source_dset_name, "od%%d");
10095        drop(file);
10096        std::fs::remove_file(&path).ok();
10097    }
10098
10099    /// An unlimited mapping takes its extent from the source it names, so a
10100    /// virtual dataset written with one reports the source's rows, not the
10101    /// seed extent its dataspace message stores
10102    /// (`H5D__virtual_set_extent_unlim`). Same file, so the resolution runs
10103    /// without opening another one.
10104    #[test]
10105    fn an_unlimited_mapping_takes_its_extent_from_its_source() {
10106        let path = temp_path("vds_unlimited");
10107        {
10108            let file = H5File::create(&path).unwrap();
10109            file.new_dataset::<i32>()
10110                .shape([5usize, 2])
10111                .chunk(&[5, 2])
10112                .max_shape(&[None, Some(2)])
10113                .create("src")
10114                .unwrap()
10115                .write_raw(&(0..10i32).collect::<Vec<_>>())
10116                .unwrap();
10117            file.new_dataset::<i32>()
10118                .shape([1usize, 2])
10119                .max_shape(&[None, Some(2)])
10120                .virtual_mapping(unlimited_rows(), ".", "src", unlimited_rows())
10121                .create("vds")
10122                .unwrap();
10123            file.close().unwrap();
10124        }
10125        let file = H5File::open(&path).unwrap();
10126        let ds = file.dataset("vds").unwrap();
10127        assert_eq!(ds.shape(), vec![5, 2]);
10128        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..10).collect::<Vec<i32>>());
10129        drop(file);
10130        std::fs::remove_file(&path).ok();
10131    }
10132
10133    /// The blocks-0/1/3 printf file every dataset-access test below reads,
10134    /// laid out exactly like `a_printf_mapping_stitches_one_source_per_block`
10135    /// so the gap is the only thing that changes.
10136    fn printf_gap_file(tag: &str) -> std::path::PathBuf {
10137        use crate::Selection;
10138        let path = temp_path(tag);
10139        let file = H5File::create(&path).unwrap();
10140        for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10141            file.new_dataset::<i32>()
10142                .shape([2usize])
10143                .create(&format!("block{b}"))
10144                .unwrap()
10145                .write_raw(&[base, base + 1])
10146                .unwrap();
10147        }
10148        file.new_dataset::<i32>()
10149            .shape([1usize, 2])
10150            .max_shape(&[None, Some(2)])
10151            .fill_value(-7i32)
10152            .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10153            .create("vds")
10154            .unwrap();
10155        file.close().unwrap();
10156        path
10157    }
10158
10159    /// `H5Pset_virtual_printf_gap` lets the block scan look past a missing
10160    /// source, and the blocks it looked past stay inside the extent reading
10161    /// as the fill value (H5Dvirtual.c:1519-1614, :2661-2665). Measured
10162    /// against libhdf5 1.14.6 through h5py's `h5p.PropDAID`: gap 0 gives
10163    /// two rows, gap 1 and gap 2 both give four with row 2 filled.
10164    #[test]
10165    fn a_printf_gap_looks_past_the_missing_block() {
10166        use crate::DatasetAccess;
10167        let path = printf_gap_file("vds_printf_gap");
10168        let file = H5File::open(&path).unwrap();
10169        for (gap, shape, data) in [
10170            (0u64, vec![2usize, 2], vec![0i32, 1, 100, 101]),
10171            (1, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10172            (2, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10173        ] {
10174            let ds = file
10175                .dataset_with("vds", DatasetAccess::new().virtual_printf_gap(gap))
10176                .unwrap();
10177            assert_eq!(ds.shape(), shape, "gap {gap}");
10178            assert_eq!(ds.read_raw::<i32>().unwrap(), data, "gap {gap}");
10179        }
10180        // Back to the default: the extent follows the properties the open
10181        // names, in both directions.
10182        let ds = file.dataset("vds").unwrap();
10183        assert_eq!(ds.shape(), vec![2, 2]);
10184        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10185        drop(file);
10186        std::fs::remove_file(&path).ok();
10187    }
10188
10189    /// A relatively-named source is found next to the virtual dataset even
10190    /// when the process is somewhere else entirely: `H5F_prefix_open_file`
10191    /// tries the primary file's `H5F_EXTPATH` — the directory it was opened
10192    /// from — before the bare relative name against the working directory
10193    /// (H5Fint.c:952-977). Measured against libhdf5 1.14.6 through h5py: a
10194    /// `VirtualSource("src.h5", ...)` beside its VDS reads its data with
10195    /// `HDF5_VDS_PREFIX` unset and the working directory elsewhere; before
10196    /// the reader took that step it read back all fill value.
10197    #[test]
10198    fn a_relative_source_resolves_next_to_the_virtual_dataset() {
10199        use crate::Selection;
10200        let dir = std::env::temp_dir().join(format!(
10201            "rust_hdf5_vds_beside_{}_{:?}",
10202            std::process::id(),
10203            std::thread::current().id()
10204        ));
10205        std::fs::create_dir_all(&dir).unwrap();
10206        {
10207            let file = H5File::create(dir.join("src.h5")).unwrap();
10208            file.new_dataset::<i32>()
10209                .shape([2usize, 4])
10210                .create("data")
10211                .unwrap()
10212                .write_raw(&(0..8i32).collect::<Vec<_>>())
10213                .unwrap();
10214            file.close().unwrap();
10215        }
10216        {
10217            let file = H5File::create(dir.join("v.h5")).unwrap();
10218            file.new_dataset::<i32>()
10219                .shape([2usize, 4])
10220                .fill_value(-9i32)
10221                // Named relatively, as h5py's `VirtualSource("src.h5", ...)`
10222                // stores it — nothing in the file says where it lives.
10223                .virtual_mapping(Selection::All, "src.h5", "data", Selection::All)
10224                .create("v")
10225                .unwrap();
10226            file.close().unwrap();
10227        }
10228        // The working directory is the crate root under `cargo test`, not
10229        // `dir`, so only the extpath step can find `src.h5`.
10230        assert_ne!(std::env::current_dir().unwrap(), dir);
10231        let file = H5File::open(dir.join("v.h5")).unwrap();
10232        let ds = file.dataset("v").unwrap();
10233        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8i32).collect::<Vec<_>>());
10234        drop(file);
10235        std::fs::remove_dir_all(&dir).ok();
10236    }
10237
10238    /// The first open of a virtual dataset fixes its access properties for
10239    /// every open that overlaps it: only the open that finds no shared info
10240    /// in `H5FO_opened` runs `H5D__open_oid(dataset, dapl_id)`, and a later
10241    /// one just points at that shared info without ever reading its own dapl
10242    /// (H5Dint.c:1496-1500, :1523-1528) — the view and the gap live in the
10243    /// shared layout storage `H5D__virtual_init` filled from that first dapl
10244    /// (H5Dvirtual.c:2178-2188). Measured against libhdf5 1.14.6 and 2.0.0
10245    /// through `h5d.open(..., dapl=...)` on the printf-gap VDS below: opening
10246    /// gap 0 then gap 1 gives both handles two rows; with every handle closed,
10247    /// opening gap 1 then gap 0 gives both four rows and the gap row reads as
10248    /// fill; with every handle closed again, gap 0 alone is back to two rows.
10249    #[test]
10250    fn the_first_open_of_a_virtual_dataset_fixes_the_properties_for_later_opens() {
10251        use crate::DatasetAccess;
10252        let path = printf_gap_file("vds_printf_first_open");
10253        let file = H5File::open(&path).unwrap();
10254        let gap = |g: u64| DatasetAccess::new().virtual_printf_gap(g);
10255        {
10256            let a = file.dataset_with("vds", gap(0)).unwrap();
10257            let b = file.dataset_with("vds", gap(1)).unwrap();
10258            assert_eq!(a.shape(), vec![2, 2]);
10259            assert_eq!(b.shape(), vec![2, 2]);
10260            assert_eq!(b.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10261        }
10262        {
10263            let c = file.dataset_with("vds", gap(1)).unwrap();
10264            let d = file.dataset_with("vds", gap(0)).unwrap();
10265            assert_eq!(c.shape(), vec![4, 2]);
10266            assert_eq!(d.shape(), vec![4, 2]);
10267            assert_eq!(
10268                d.read_raw::<i32>().unwrap(),
10269                vec![0, 1, 100, 101, -7, -7, 300, 301]
10270            );
10271        }
10272        let e = file.dataset_with("vds", gap(0)).unwrap();
10273        assert_eq!(e.shape(), vec![2, 2]);
10274        assert_eq!(e.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10275        drop(e);
10276        drop(file);
10277        std::fs::remove_file(&path).ok();
10278    }
10279
10280    /// `H5D_VDS_FIRST_MISSING` ignores the printf gap: `H5D__virtual_init`
10281    /// reads the gap property only under `H5D_VDS_LAST_AVAILABLE` and forces
10282    /// it to 0 otherwise (H5Dvirtual.c:2182-2188). Measured against libhdf5
10283    /// 1.14.6: gap 2 under this view still gives two rows.
10284    #[test]
10285    fn the_first_missing_view_ignores_the_printf_gap() {
10286        use crate::{DatasetAccess, VirtualView};
10287        let path = printf_gap_file("vds_printf_first_missing");
10288        let file = H5File::open(&path).unwrap();
10289        for gap in [0u64, 2] {
10290            let ds = file
10291                .dataset_with(
10292                    "vds",
10293                    DatasetAccess::new()
10294                        .virtual_view(VirtualView::FirstMissing)
10295                        .virtual_printf_gap(gap),
10296                )
10297                .unwrap();
10298            assert_eq!(ds.shape(), vec![2, 2], "gap {gap}");
10299            assert_eq!(
10300                ds.read_raw::<i32>().unwrap(),
10301                vec![0, 1, 100, 101],
10302                "gap {gap}"
10303            );
10304        }
10305        drop(file);
10306        std::fs::remove_file(&path).ok();
10307    }
10308
10309    /// On a mapping unlimited on both sides the view is
10310    /// `H5S_hyper_get_clip_extent_match`'s `incl_trail`
10311    /// (H5Dvirtual.c:1447-1451): with a stride wider than its block, the
10312    /// extent under `H5D_VDS_LAST_AVAILABLE` ends at the last mapped row,
10313    /// and under `H5D_VDS_FIRST_MISSING` it runs on to where the next block
10314    /// would start. Measured against libhdf5 1.14.6 over a three-row source
10315    /// with stride 3 and block 2: two rows and three rows.
10316    #[test]
10317    fn the_view_decides_whether_a_trailing_gap_is_inside_the_extent() {
10318        use crate::format::selection::UNLIMITED;
10319        use crate::{DatasetAccess, Hyperslab, RegularHyperslab, Selection, VirtualView};
10320        let strided = || Selection::Hyperslab {
10321            rank: 2,
10322            form: Hyperslab::Regular(RegularHyperslab {
10323                start: vec![0, 0],
10324                stride: vec![3, 1],
10325                count: vec![UNLIMITED, 1],
10326                block: vec![2, 2],
10327            }),
10328        };
10329        let path = temp_path("vds_view_trail");
10330        {
10331            let file = H5File::create(&path).unwrap();
10332            file.new_dataset::<i32>()
10333                .shape([3usize, 2])
10334                .max_shape(&[None, Some(2)])
10335                .chunk(&[1, 2])
10336                .create("src")
10337                .unwrap()
10338                .write_raw(&(0..6i32).collect::<Vec<_>>())
10339                .unwrap();
10340            file.new_dataset::<i32>()
10341                .shape([1usize, 2])
10342                .max_shape(&[None, Some(2)])
10343                .fill_value(-9i32)
10344                .virtual_mapping(strided(), ".", "src", strided())
10345                .create("vds")
10346                .unwrap();
10347            file.close().unwrap();
10348        }
10349        let file = H5File::open(&path).unwrap();
10350        {
10351            let last = file.dataset("vds").unwrap();
10352            assert_eq!(last.shape(), vec![2, 2]);
10353            assert_eq!(last.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3]);
10354        }
10355        // The handle above is gone, so this open is the one that resolves.
10356        let first = file
10357            .dataset_with(
10358                "vds",
10359                DatasetAccess::new().virtual_view(VirtualView::FirstMissing),
10360            )
10361            .unwrap();
10362        assert_eq!(first.shape(), vec![3, 2]);
10363        assert_eq!(first.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3, -9, -9]);
10364        drop(file);
10365        std::fs::remove_file(&path).ok();
10366    }
10367
10368    /// The property list reads back what was set (`H5Pget_virtual_view`,
10369    /// `H5Pget_virtual_printf_gap`), and the gap `HSIZE_UNDEF` that
10370    /// `H5Pset_virtual_printf_gap` refuses is refused by the open that would
10371    /// have used it.
10372    #[test]
10373    fn the_access_property_list_reads_back_and_refuses_hsize_undef() {
10374        use crate::{DatasetAccess, VirtualView};
10375        let plist = DatasetAccess::new();
10376        assert_eq!(plist.view(), VirtualView::LastAvailable);
10377        assert_eq!(plist.printf_gap(), 0);
10378        let set = plist
10379            .virtual_view(VirtualView::FirstMissing)
10380            .virtual_printf_gap(4);
10381        assert_eq!(set.view(), VirtualView::FirstMissing);
10382        // The *property* keeps what was set even though the resolution under
10383        // this view scans with 0.
10384        assert_eq!(set.printf_gap(), 4);
10385
10386        let path = printf_gap_file("vds_printf_gap_undef");
10387        let file = H5File::open(&path).unwrap();
10388        let err = match file.dataset_with("vds", DatasetAccess::new().virtual_printf_gap(u64::MAX))
10389        {
10390            Ok(_) => panic!("HSIZE_UNDEF is not a valid printf gap size"),
10391            Err(e) => e.to_string(),
10392        };
10393        assert!(err.contains("HSIZE_UNDEF"), "{err}");
10394        drop(file);
10395        std::fs::remove_file(&path).ok();
10396    }
10397
10398    /// The rank-2 `count = (H5S_UNLIMITED, 1)`, `block = (1, 2)` selection
10399    /// both sides of an unlimited row-wise mapping use.
10400    fn unlimited_rows() -> crate::Selection {
10401        use crate::format::selection::UNLIMITED;
10402        use crate::{Hyperslab, RegularHyperslab, Selection};
10403        Selection::Hyperslab {
10404            rank: 2,
10405            form: Hyperslab::Regular(RegularHyperslab {
10406                start: vec![0, 0],
10407                stride: vec![1, 1],
10408                count: vec![UNLIMITED, 1],
10409                block: vec![1, 2],
10410            }),
10411        }
10412    }
10413
10414    /// An unlimited virtual selection over a *limited* source selection is
10415    /// the printf shape, and without a `%b` in a source name there is no
10416    /// second dataset to fill the second block —
10417    /// `H5D_virtual_check_mapping_post` refuses it, and so does this.
10418    #[test]
10419    fn an_unlimited_virtual_selection_over_a_limited_source_is_refused() {
10420        use crate::Selection;
10421        let path = temp_path("vds_unlim_limited_src");
10422        let file = H5File::create(&path).unwrap();
10423        let err = match file
10424            .new_dataset::<i32>()
10425            .shape([1usize, 2])
10426            .max_shape(&[None, Some(2)])
10427            .virtual_mapping(unlimited_rows(), "src.h5", "src", Selection::All)
10428            .create("vds")
10429        {
10430            Ok(_) => panic!("no printf substitution names the mapping's later blocks"),
10431            Err(e) => e.to_string(),
10432        };
10433        assert!(err.contains("printf"), "{err}");
10434        file.close().unwrap();
10435        std::fs::remove_file(&path).ok();
10436    }
10437
10438    /// A virtual dataset stores nothing of its own, so it cannot also be one
10439    /// of the storage classes that do.
10440    #[test]
10441    fn a_virtual_dataset_cannot_also_be_chunked_or_external() {
10442        use crate::Selection;
10443        let path = temp_path("vds_exclusive");
10444        let file = H5File::create(&path).unwrap();
10445        let builder = || {
10446            file.new_dataset::<i32>().shape([16usize]).virtual_mapping(
10447                Selection::All,
10448                "src.h5",
10449                "src",
10450                Selection::All,
10451            )
10452        };
10453        for (which, res) in [
10454            ("chunked", builder().chunk(&[4]).create("a")),
10455            ("compact", builder().compact().create("b")),
10456            (
10457                "external",
10458                builder().external(&[("x.raw", 0, 64)]).create("c"),
10459            ),
10460            ("references", builder().object_references().create("d")),
10461        ] {
10462            match res {
10463                Ok(_) => panic!("a virtual dataset cannot also be {which}"),
10464                Err(e) => assert!(e.to_string().contains("virtual dataset"), "{which}: {e}"),
10465            }
10466        }
10467        file.close().unwrap();
10468        std::fs::remove_file(&path).ok();
10469    }
10470
10471    /// Deleting a virtual dataset frees the global heap object its mapping
10472    /// list lived in — `H5D__virtual_delete` reaches `H5HG_remove` — so the
10473    /// next dataset's own heap object reuses that space instead of the file
10474    /// growing by a whole collection per deleted virtual dataset.
10475    #[test]
10476    fn deleting_a_virtual_dataset_frees_its_mapping_list() {
10477        use crate::Selection;
10478        let path = temp_path("vds_delete");
10479        let baseline = {
10480            let file = H5File::create(&path).unwrap();
10481            file.new_dataset::<i32>()
10482                .shape([16usize])
10483                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10484                .create("vds")
10485                .unwrap();
10486            file.delete_dataset("vds").unwrap();
10487            file.close().unwrap();
10488            std::fs::metadata(&path).unwrap().len()
10489        };
10490        std::fs::remove_file(&path).ok();
10491
10492        // Ten more create-then-delete rounds must land on the same file size:
10493        // each round's heap object is removed, its collection becomes empty
10494        // and returns to the allocator, and the next round takes it back.
10495        let file = H5File::create(&path).unwrap();
10496        for i in 0..10 {
10497            let name = format!("vds{i}");
10498            file.new_dataset::<i32>()
10499                .shape([16usize])
10500                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10501                .create(&name)
10502                .unwrap();
10503            file.delete_dataset(&name).unwrap();
10504        }
10505        file.close().unwrap();
10506        assert_eq!(std::fs::metadata(&path).unwrap().len(), baseline);
10507        std::fs::remove_file(&path).ok();
10508    }
10509
10510    /// [`H5Dataset::storage_layout`] tells the four classes apart — the
10511    /// negative case for any one class is simply that it is not another.
10512    #[test]
10513    fn storage_layout_reports_each_class() {
10514        use crate::StorageLayout;
10515        let path = temp_path("storage_layout");
10516        {
10517            let file = H5File::create(&path).unwrap();
10518            file.new_dataset::<i32>()
10519                .shape([4usize])
10520                .create("contig")
10521                .unwrap();
10522            file.new_dataset::<i32>()
10523                .shape([4usize])
10524                .compact()
10525                .create("compact")
10526                .unwrap();
10527            file.new_dataset::<i32>()
10528                .shape([8usize])
10529                .chunk(&[4])
10530                .create("chunked")
10531                .unwrap();
10532            file.close().unwrap();
10533        }
10534        let file = H5File::open(&path).unwrap();
10535        assert_eq!(
10536            file.dataset("contig").unwrap().storage_layout().unwrap(),
10537            StorageLayout::Contiguous
10538        );
10539        assert_eq!(
10540            file.dataset("compact").unwrap().storage_layout().unwrap(),
10541            StorageLayout::Compact
10542        );
10543        assert_eq!(
10544            file.dataset("chunked").unwrap().storage_layout().unwrap(),
10545            StorageLayout::Chunked
10546        );
10547    }
10548
10549    /// Read-mode-only accessor, matching `datatype()`'s own contract.
10550    #[test]
10551    fn storage_layout_errors_in_write_mode() {
10552        let path = temp_path("storage_layout_write_mode");
10553        let file = H5File::create(&path).unwrap();
10554        let ds = file
10555            .new_dataset::<i32>()
10556            .shape([4usize])
10557            .create("data")
10558            .unwrap();
10559        assert!(ds.storage_layout().is_err());
10560        file.close().unwrap();
10561    }
10562
10563    /// [`H5Dataset::chunk_index`] reports the real on-disk index kind
10564    /// (extensible array for one unlimited dimension, version-1 B-tree
10565    /// under a legacy libver bound) and `None` for an unchunked dataset —
10566    /// the negative case.
10567    #[test]
10568    fn chunk_index_reports_the_stored_kind() {
10569        use crate::ChunkIndex;
10570        let path = temp_path("chunk_index");
10571        {
10572            let file = H5File::create(&path).unwrap();
10573            file.new_dataset::<i32>()
10574                .shape([4usize])
10575                .create("contig")
10576                .unwrap();
10577            file.new_dataset::<i32>()
10578                .shape([16usize])
10579                .chunk(&[4])
10580                .max_shape(&[None])
10581                .create("earray")
10582                .unwrap();
10583            file.set_libver_latest(false).unwrap();
10584            file.new_dataset::<i32>()
10585                .shape([8usize])
10586                .chunk(&[4])
10587                .max_shape(&[None])
10588                .create("btree1")
10589                .unwrap();
10590            file.close().unwrap();
10591        }
10592        let file = H5File::open(&path).unwrap();
10593        assert_eq!(file.dataset("contig").unwrap().chunk_index().unwrap(), None);
10594        assert_eq!(
10595            file.dataset("earray").unwrap().chunk_index().unwrap(),
10596            Some(ChunkIndex::ExtensibleArray)
10597        );
10598        assert_eq!(
10599            file.dataset("btree1").unwrap().chunk_index().unwrap(),
10600            Some(ChunkIndex::BtreeV1)
10601        );
10602    }
10603
10604    /// [`H5Dataset::filters`] reports the stored pipeline in order — and
10605    /// the negative case: an unfiltered dataset reports an empty pipeline,
10606    /// not an error.
10607    #[test]
10608    fn filters_reports_the_stored_pipeline() {
10609        use crate::format::messages::filter::{FILTER_DEFLATE, FILTER_SHUFFLE, FLAG_OPTIONAL};
10610        let path = temp_path("filters");
10611        {
10612            let file = H5File::create(&path).unwrap();
10613            file.new_dataset::<i32>()
10614                .shape([16usize])
10615                .create("unfiltered")
10616                .unwrap();
10617            file.new_dataset::<i32>()
10618                .shape([16usize])
10619                .chunk(&[4])
10620                .shuffle()
10621                .deflate(6)
10622                .create("filtered")
10623                .unwrap();
10624            file.close().unwrap();
10625        }
10626        let file = H5File::open(&path).unwrap();
10627        assert_eq!(
10628            file.dataset("unfiltered").unwrap().filters().unwrap(),
10629            Vec::new()
10630        );
10631        let filters = file.dataset("filtered").unwrap().filters().unwrap();
10632        assert_eq!(filters.len(), 2);
10633        assert_eq!(filters[0].id, FILTER_SHUFFLE);
10634        assert_eq!(filters[0].flags, FLAG_OPTIONAL);
10635        assert_eq!(filters[1].id, FILTER_DEFLATE);
10636        assert_eq!(filters[1].cd_values, vec![6]);
10637    }
10638
10639    /// [`H5Dataset::fill_value`] reports the explicit bytes for a dataset
10640    /// created with `.fill_value(...)`, and the negative case: a dataset
10641    /// with no fill value set reports [`FillValue::Default`], not an error.
10642    /// `FillValue::Undefined` has no constructor on either this crate's
10643    /// writer or h5py's public API, so it is not exercised here.
10644    #[test]
10645    fn fill_value_reports_the_stored_value() {
10646        use crate::FillValue;
10647        let path = temp_path("fill_value");
10648        {
10649            let file = H5File::create(&path).unwrap();
10650            file.new_dataset::<i32>()
10651                .shape([4usize])
10652                .create("unset")
10653                .unwrap();
10654            file.new_dataset::<i32>()
10655                .shape([4usize])
10656                .fill_value(-7i32)
10657                .create("set")
10658                .unwrap();
10659            file.close().unwrap();
10660        }
10661        let file = H5File::open(&path).unwrap();
10662        assert_eq!(
10663            file.dataset("unset").unwrap().fill_value().unwrap(),
10664            FillValue::Default
10665        );
10666        assert_eq!(
10667            file.dataset("set").unwrap().fill_value().unwrap(),
10668            FillValue::UserDefined((-7i32).to_le_bytes().to_vec())
10669        );
10670    }
10671
10672    /// The three `H5D__efl_construct` / `H5Pset_external` rules an
10673    /// `H5O_EFL_UNLIMITED` slot lives inside: it may only be the last slot,
10674    /// an unlimited dataspace must have one, and only the first dimension
10675    /// may be extendible.
10676    #[test]
10677    fn the_unlimited_external_slot_keeps_its_three_rules() {
10678        use crate::format::messages::external_file_list::UNLIMITED;
10679        let path = temp_path("efl_unlim_rules");
10680        let file = H5File::create(&path).unwrap();
10681
10682        // "previous file size is unlimited": nothing behind an unlimited slot
10683        // could ever be reached.
10684        let err = match file
10685            .new_dataset::<i32>()
10686            .shape([8usize])
10687            .external(&[("a.raw", 0, UNLIMITED), ("b.raw", 0, 32)])
10688            .create("mid")
10689        {
10690            Ok(_) => panic!("an unlimited slot absorbs everything behind it"),
10691            Err(e) => e.to_string(),
10692        };
10693        assert!(err.contains("only be the last"), "{err}");
10694
10695        // "unlimited dataspace but finite storage".
10696        let err = match file
10697            .new_dataset::<i32>()
10698            .shape([8usize])
10699            .max_shape(&[None])
10700            .external(&[("a.raw", 0, 32)])
10701            .create("finite")
10702        {
10703            Ok(_) => panic!("no finite reservation covers an unlimited extent"),
10704            Err(e) => e.to_string(),
10705        };
10706        assert!(err.contains("unlimited dataspace"), "{err}");
10707
10708        // "only the first dimension can be extendible".
10709        let err = match file
10710            .new_dataset::<i32>()
10711            .shape([2usize, 4])
10712            .max_shape(&[Some(2), None])
10713            .external(&[("a.raw", 0, UNLIMITED)])
10714            .create("dim1")
10715        {
10716            Ok(_) => panic!("only the slowest-varying dimension may be extendible"),
10717            Err(e) => e.to_string(),
10718        };
10719        assert!(err.contains("only the first dimension"), "{err}");
10720
10721        // And the legal shape: an unlimited last slot under an unlimited
10722        // first dimension.
10723        file.new_dataset::<i32>()
10724            .shape([8usize])
10725            .max_shape(&[None])
10726            .external(&[("a.raw", 0, 16), ("b.raw", 0, UNLIMITED)])
10727            .create("ok")
10728            .unwrap()
10729            .write_raw(&(0..8i32).collect::<Vec<_>>())
10730            .unwrap();
10731        file.close().unwrap();
10732        std::fs::remove_file(&path).ok();
10733        for raw in ["a.raw", "b.raw"] {
10734            std::fs::remove_file(raw).ok();
10735        }
10736    }
10737
10738    /// An unlimited slot reserves nothing, so a read of it is bounded by the
10739    /// dataset's extent and by what the file physically holds: the tail past
10740    /// the end of a short raw file reads back as zero, exactly as
10741    /// `H5D__efl_read` fills it.
10742    #[test]
10743    fn an_unlimited_external_slot_reads_zero_past_the_end_of_its_file() {
10744        use crate::format::messages::external_file_list::UNLIMITED;
10745        let dir = std::env::temp_dir().join(format!("rh5_efl_short_{}", std::process::id()));
10746        std::fs::create_dir_all(&dir).unwrap();
10747        let path = dir.join("f.h5");
10748        let raw = dir.join("short.raw");
10749        // Four elements' worth of bytes for an eight-element dataset.
10750        std::fs::write(&raw, [0u8; 16]).unwrap();
10751        {
10752            let file = H5File::create(&path).unwrap();
10753            file.new_dataset::<i32>()
10754                .shape([8usize])
10755                .max_shape(&[None])
10756                .external(&[(raw.to_str().unwrap(), 0, UNLIMITED)])
10757                .create("data")
10758                .unwrap();
10759            file.close().unwrap();
10760        }
10761        let file = H5File::open(&path).unwrap();
10762        assert_eq!(
10763            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
10764            vec![0i32; 8]
10765        );
10766        drop(file);
10767        std::fs::remove_dir_all(&dir).ok();
10768    }
10769
10770    /// [`H5Dataset::external_files`] reports the stored segment list in
10771    /// order, and the negative case: a dataset whose data lives in this
10772    /// file reports an empty list, not an error.
10773    #[test]
10774    fn external_files_reports_the_stored_segments() {
10775        let path = temp_path("external_files");
10776        {
10777            let file = H5File::create(&path).unwrap();
10778            file.new_dataset::<i32>()
10779                .shape([4usize])
10780                .create("contig")
10781                .unwrap();
10782            file.new_dataset::<i32>()
10783                .shape([16usize])
10784                .external(&[("a.raw", 0, 32), ("b.raw", 8, 32)])
10785                .create("external")
10786                .unwrap();
10787            file.close().unwrap();
10788        }
10789        let file = H5File::open(&path).unwrap();
10790        assert_eq!(
10791            file.dataset("contig").unwrap().external_files().unwrap(),
10792            Vec::new()
10793        );
10794        let segments = file.dataset("external").unwrap().external_files().unwrap();
10795        assert_eq!(segments.len(), 2);
10796        assert_eq!(segments[0].name, "a.raw");
10797        assert_eq!(segments[0].offset, 0);
10798        assert_eq!(segments[0].size, 32);
10799        assert_eq!(segments[1].name, "b.raw");
10800        assert_eq!(segments[1].offset, 8);
10801        assert_eq!(segments[1].size, 32);
10802    }
10803
10804    /// [`H5Dataset::max_shape`] reports an unlimited axis as `None` and a
10805    /// fixed one as its current size — and the negative case: a dataset
10806    /// with no maximum-dimensions message reports max == current, not an
10807    /// error.
10808    #[test]
10809    fn max_shape_reports_unlimited_and_fixed_axes() {
10810        let path = temp_path("max_shape");
10811        {
10812            let file = H5File::create(&path).unwrap();
10813            file.new_dataset::<i32>()
10814                .shape([4usize, 8])
10815                .create("fixed")
10816                .unwrap();
10817            file.new_dataset::<i32>()
10818                .shape([4usize, 8])
10819                .chunk(&[2, 8])
10820                .max_shape(&[None, Some(8)])
10821                .create("unlimited")
10822                .unwrap();
10823            file.close().unwrap();
10824        }
10825        let file = H5File::open(&path).unwrap();
10826        assert_eq!(
10827            file.dataset("fixed").unwrap().max_shape().unwrap(),
10828            vec![Some(4), Some(8)]
10829        );
10830        assert_eq!(
10831            file.dataset("unlimited").unwrap().max_shape().unwrap(),
10832            vec![None, Some(8)]
10833        );
10834    }
10835
10836    /// [`H5Dataset::virtual_mappings`] reports the stored source/virtual
10837    /// mapping list in order, and the negative case: a dataset with no
10838    /// virtual layout reports an empty list, not an error.
10839    #[test]
10840    fn virtual_mappings_reports_the_stored_mappings() {
10841        use crate::Selection;
10842        let path = temp_path("virtual_mappings");
10843        {
10844            let file = H5File::create(&path).unwrap();
10845            file.new_dataset::<i32>()
10846                .shape([4usize])
10847                .create("plain")
10848                .unwrap();
10849            file.new_dataset::<i32>()
10850                .shape([16usize])
10851                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10852                .create("vds")
10853                .unwrap();
10854            file.close().unwrap();
10855        }
10856        let file = H5File::open(&path).unwrap();
10857        assert_eq!(
10858            file.dataset("plain").unwrap().virtual_mappings().unwrap(),
10859            Vec::new()
10860        );
10861        let mappings = file.dataset("vds").unwrap().virtual_mappings().unwrap();
10862        assert_eq!(mappings.len(), 1);
10863        assert_eq!(mappings[0].source_file_name, "src.h5");
10864        assert_eq!(mappings[0].source_dset_name, "src");
10865        assert_eq!(mappings[0].source_selection, Selection::All);
10866        assert_eq!(mappings[0].virtual_selection, Selection::All);
10867    }
10868}