Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use std::borrow::Cow;
8
9use crate::attribute::AttrBuilder;
10use crate::error::{Hdf5Error, Result};
11use crate::file::{
12    borrow_inner, borrow_inner_mut, clone_inner, same_inner, H5FileInner, SharedInner,
13};
14use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
15use crate::format::messages::filter::Filter;
16use crate::format::messages::virtual_mapping::VirtualMapping;
17use crate::format::reference::{Reference, ReferenceTarget};
18use crate::format::selection::check_hyperslab;
19use crate::format::selection::Selection;
20use crate::format::storage_kind::AttributeStorage;
21use crate::io::file_handle::ReadDst;
22use crate::io::reader::{read_image_into_new, ExternalFileSegment};
23use crate::io::writer::ChunkIndexKind;
24use crate::types::H5Type;
25
26// ---------------------------------------------------------------------------
27// DatasetBuilder
28// ---------------------------------------------------------------------------
29
30/// A fluent builder for creating datasets.
31///
32/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
33///
34/// ```no_run
35/// # use rust_hdf5::H5File;
36/// let file = H5File::create("builder.h5").unwrap();
37/// let ds = file.new_dataset::<f32>()
38///     .shape(&[10, 20])
39///     .create("temperatures")
40///     .unwrap();
41/// ```
42pub struct DatasetBuilder<T: H5Type> {
43    file_inner: SharedInner,
44    shape: Option<Vec<usize>>,
45    is_null: bool,
46    chunk_dims: Option<Vec<usize>>,
47    max_shape: Option<Vec<Option<usize>>>,
48    is_compact: bool,
49    early_allocation: bool,
50    deflate_level: Option<u32>,
51    shuffle: bool,
52    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
53    group_path: Option<String>,
54    fill_value: Option<Vec<u8>>,
55    fill_time: Option<FillTime>,
56    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
57    committed_type: Option<String>,
58    references: Option<ReferenceElement>,
59    external: Option<Vec<(String, u64, u64)>>,
60    efile_prefix: Option<String>,
61    virtual_mappings: Vec<VirtualMapping>,
62    _marker: std::marker::PhantomData<T>,
63}
64
65/// Which reference a `*_references()` builder call asked the elements to be.
66///
67/// One field rather than a flag per kind: an element is a whole-object
68/// reference or a region reference, never both, and the width of each is only
69/// known once the file's address size is (see
70/// [`DatatypeMessage::object_reference`] and
71/// [`DatatypeMessage::region_reference`]).
72///
73/// [`DatatypeMessage::object_reference`]: crate::format::messages::datatype::DatatypeMessage::object_reference
74/// [`DatatypeMessage::region_reference`]: crate::format::messages::datatype::DatatypeMessage::region_reference
75#[derive(Debug, Clone, Copy, PartialEq, Eq)]
76enum ReferenceElement {
77    /// `H5T_STD_REF_OBJ`.
78    Object,
79    /// `H5T_STD_REF_DSETREG`.
80    Region,
81    /// `H5T_STD_REF`, the 1.12 form. One datatype for all three 1.12 kinds:
82    /// the element leads with the kind it holds, so `H5T__ref_disk_getsize`
83    /// sizes every element for the widest of them and a dataset of this type
84    /// may hold objects, regions and attributes alike.
85    Revised,
86}
87
88impl ReferenceElement {
89    /// The stored datatype for this kind in a file with `ctx`'s address size.
90    fn datatype(
91        self,
92        ctx: &crate::format::FormatContext,
93    ) -> crate::format::messages::datatype::DatatypeMessage {
94        use crate::format::messages::datatype::DatatypeMessage;
95        match self {
96            Self::Object => DatatypeMessage::object_reference(ctx),
97            Self::Region => DatatypeMessage::region_reference(ctx),
98            Self::Revised => DatatypeMessage::std_object_reference(ctx),
99        }
100    }
101}
102
103impl<T: H5Type> DatasetBuilder<T> {
104    pub(crate) fn new(file_inner: SharedInner) -> Self {
105        Self {
106            file_inner,
107            shape: None,
108            is_null: false,
109            chunk_dims: None,
110            max_shape: None,
111            is_compact: false,
112            early_allocation: false,
113            deflate_level: None,
114            shuffle: false,
115            custom_pipeline: None,
116            group_path: None,
117            fill_value: None,
118            fill_time: None,
119            datatype_override: None,
120            committed_type: None,
121            references: None,
122            external: None,
123            efile_prefix: None,
124            virtual_mappings: Vec::new(),
125            _marker: std::marker::PhantomData,
126        }
127    }
128
129    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
130        Self {
131            file_inner,
132            shape: None,
133            is_null: false,
134            chunk_dims: None,
135            max_shape: None,
136            is_compact: false,
137            early_allocation: false,
138            deflate_level: None,
139            shuffle: false,
140            custom_pipeline: None,
141            group_path: Some(group_path),
142            fill_value: None,
143            fill_time: None,
144            datatype_override: None,
145            committed_type: None,
146            references: None,
147            external: None,
148            efile_prefix: None,
149            virtual_mappings: Vec::new(),
150            _marker: std::marker::PhantomData,
151        }
152    }
153
154    /// Set the dataset dimensions.
155    ///
156    /// This is required before calling [`create`](Self::create), unless
157    /// [`null`](Self::null) was called instead.
158    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
159    #[must_use]
160    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
161        self.shape = Some(dims.as_ref().to_vec());
162        self
163    }
164
165    /// Create a scalar (0-dimensional) dataset holding a single value.
166    #[must_use]
167    pub fn scalar(mut self) -> Self {
168        self.shape = Some(vec![]);
169        self
170    }
171
172    /// Create a dataset with the NULL dataspace: no elements at all.
173    ///
174    /// Distinct from [`scalar`](Self::scalar), which holds exactly one
175    /// element. A NULL dataset holds zero bytes of data and cannot be
176    /// written to — [`write_raw`](H5Dataset::write_raw) and
177    /// [`write_raw_bytes`](H5Dataset::write_raw_bytes) return an error, and
178    /// it cannot be chunked or filtered, matching h5py's `h5py.Empty`.
179    #[must_use]
180    pub fn null(mut self) -> Self {
181        self.is_null = true;
182        self
183    }
184
185    /// Set chunk dimensions for chunked storage.
186    ///
187    /// When set, the dataset uses chunked storage with the extensible array
188    /// index. You should also call [`max_shape`](Self::max_shape) or
189    /// [`resizable`](Self::resizable) to allow extending.
190    #[must_use]
191    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
192        self.chunk_dims = Some(chunk_dims.to_vec());
193        self
194    }
195
196    /// Make all dimensions unlimited (resizable).
197    ///
198    /// This sets max_dims to u64::MAX for all dimensions.
199    #[must_use]
200    pub fn resizable(mut self) -> Self {
201        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
202        self
203    }
204
205    /// Set maximum dimensions. `None` means unlimited for that dimension.
206    #[must_use]
207    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
208        self.max_shape = Some(max.to_vec());
209        self
210    }
211
212    /// Store the raw data inside the dataset's object header —
213    /// `H5Pset_layout(dcpl, H5D_COMPACT)`.
214    ///
215    /// A compact dataset costs no data block and no second seek to read, which
216    /// suits the small per-run constants an analysis file is full of. It is
217    /// bounded by what one object header message can hold
218    /// ([`MAX_COMPACT_DATA`](crate::MAX_COMPACT_DATA) bytes) and it
219    /// is fixed in size: [`chunk`](Self::chunk), a filter, and an unlimited
220    /// [`max_shape`](Self::max_shape) are all rejected at
221    /// [`create`](Self::create), as libhdf5 rejects them.
222    ///
223    /// ```no_run
224    /// # use rust_hdf5::H5File;
225    /// let file = H5File::create("compact.h5").unwrap();
226    /// let ds = file.new_dataset::<i32>()
227    ///     .shape([16])
228    ///     .compact()
229    ///     .create("data")
230    ///     .unwrap();
231    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
232    /// ```
233    #[must_use]
234    pub fn compact(mut self) -> Self {
235        self.is_compact = true;
236        self
237    }
238
239    /// Allocate the whole of a chunked dataset's storage at create —
240    /// `H5Pset_alloc_time(dcpl, H5D_ALLOC_TIME_EARLY)`, h5py's
241    /// `alloc_time=h5d.ALLOC_TIME_EARLY`.
242    ///
243    /// Every chunk exists, holding the fill value, before anything is
244    /// written, so an unwritten chunk costs a read of fill bytes rather than
245    /// a miss. On a fixed-shape unfiltered dataset that is also what lets
246    /// libhdf5 pick its cheapest chunk index — the *implicit* index, which
247    /// is no index at all: the chunks are one contiguous run in grid order
248    /// and a chunk's address is arithmetic. This builder makes the same
249    /// choice under the same conditions, so such a dataset is written with
250    /// no index structure in the file.
251    ///
252    /// Ignored by storage that has no chunk grid to allocate: contiguous,
253    /// compact and NULL-dataspace datasets.
254    ///
255    /// ```no_run
256    /// # use rust_hdf5::H5File;
257    /// let file = H5File::create("implicit.h5").unwrap();
258    /// let ds = file.new_dataset::<i32>()
259    ///     .shape([16])
260    ///     .chunk(&[4])
261    ///     .early_allocation()
262    ///     .create("data")
263    ///     .unwrap();
264    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
265    /// ```
266    #[must_use]
267    pub fn early_allocation(mut self) -> Self {
268        self.early_allocation = true;
269        self
270    }
271
272    /// Enable deflate (gzip) compression with the given level (0-9).
273    ///
274    /// Requires chunked storage (call `.chunk()` before `.create()`).
275    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
276    #[must_use]
277    pub fn deflate(mut self, level: u32) -> Self {
278        self.deflate_level = Some(level);
279        self
280    }
281
282    /// Enable the shuffle filter — `H5Pset_shuffle(dcpl)`, h5py's
283    /// `shuffle=True`.
284    ///
285    /// Shuffle reorders a chunk's bytes by their position within an element,
286    /// which typically improves how well a compressor behind it does on
287    /// numeric data. It is a permutation, not a compressor: on its own it
288    /// leaves the chunk exactly as large as it was, which is what
289    /// `H5Pset_shuffle` without a compressor writes. Combine it with
290    /// [`deflate`](Self::deflate) to compress the shuffled stream. Requires
291    /// chunked storage.
292    ///
293    /// The element width the filter records is the dataset's, so a
294    /// [`datatype`](Self::datatype) override is what it follows when the
295    /// stored element is not `T` itself.
296    #[must_use]
297    pub fn shuffle(mut self) -> Self {
298        self.shuffle = true;
299        self
300    }
301
302    /// Enable shuffle + deflate compression — the same pipeline as
303    /// `.shuffle().deflate(level)`.
304    ///
305    /// Shuffle reorders bytes by position within elements before compression,
306    /// which typically improves compression ratios for numeric data.
307    /// Requires chunked storage.
308    #[must_use]
309    pub fn shuffle_deflate(mut self, level: u32) -> Self {
310        self.shuffle = true;
311        self.deflate_level = Some(level);
312        self
313    }
314
315    /// Enable Zstandard compression with the given level (1-22, default 3).
316    ///
317    /// Requires chunked storage (call `.chunk()` before `.create()`).
318    #[must_use]
319    pub fn zstd(mut self, level: u32) -> Self {
320        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
321        self
322    }
323
324    /// Set a custom filter pipeline for compression.
325    ///
326    /// This takes precedence over [`deflate`](Self::deflate) and
327    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
328    #[must_use]
329    pub fn filter_pipeline(
330        mut self,
331        pipeline: crate::format::messages::filter::FilterPipeline,
332    ) -> Self {
333        self.custom_pipeline = Some(pipeline);
334        self
335    }
336
337    /// Override the stored element datatype.
338    ///
339    /// By default the dataset is created with the datatype derived from the
340    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
341    /// different on-disk datatype than the in-memory element type — for
342    /// example a reduced-precision fixed-point type that matches an N-bit
343    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
344    /// override must equal `T::element_size()`; the N-bit filter packs the
345    /// significant bits within that fixed footprint.
346    ///
347    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
348    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
349    #[must_use]
350    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
351        self.datatype_override = Some(dt);
352        self
353    }
354
355    /// Build the dataset on the committed (named) datatype at `path` —
356    /// h5py's `dtype=f["name"]`, `H5Dcreate2` with a committed type id.
357    ///
358    /// The dataset does not describe its type: its header stores a pointer to
359    /// that object, so the type is defined once and every dataset sharing it
360    /// is guaranteed to agree. The type comes from the committed object, so
361    /// this supersedes both `T` and [`datatype`](Self::datatype).
362    ///
363    /// The path is resolved at [`create`](Self::create), which fails when no
364    /// committed datatype is there — commit it with
365    /// [`H5File::commit_datatype`](crate::file::H5File::commit_datatype)
366    /// first.
367    ///
368    /// ```no_run
369    /// # use rust_hdf5::H5File;
370    /// # use rust_hdf5::format::messages::datatype::DatatypeMessage;
371    /// let file = H5File::create("committed.h5").unwrap();
372    /// file.commit_datatype("temperature", DatatypeMessage::f64_type()).unwrap();
373    /// file.new_dataset::<f64>()
374    ///     .committed_type("temperature")
375    ///     .shape([4])
376    ///     .create("readings")
377    ///     .unwrap();
378    /// ```
379    #[must_use]
380    pub fn committed_type(mut self, path: &str) -> Self {
381        self.committed_type = Some(path.to_string());
382        self
383    }
384
385    /// Store object references — h5py's `h5py.ref_dtype`.
386    ///
387    /// The elements are written with
388    /// [`write_object_references`](H5Dataset::write_object_references) and
389    /// name objects by path. The element width is the file's address size, so
390    /// the datatype is resolved at [`create`](Self::create) rather than here;
391    /// it overrides both `T` and any [`datatype`](Self::datatype) call.
392    ///
393    /// ```no_run
394    /// # use rust_hdf5::H5File;
395    /// let file = H5File::create("refs.h5").unwrap();
396    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
397    /// let refs = file.new_dataset::<u64>()
398    ///     .object_references()
399    ///     .shape([1])
400    ///     .create("refs")
401    ///     .unwrap();
402    /// refs.write_object_references(&["/target"]).unwrap();
403    /// file.close().unwrap();
404    /// ```
405    #[must_use]
406    pub fn object_references(mut self) -> Self {
407        self.references = Some(ReferenceElement::Object);
408        self
409    }
410
411    /// Store revised object references — the 1.12 `H5T_STD_REF`.
412    ///
413    /// Same paths and same [`write_object_references`](H5Dataset::write_object_references)
414    /// call as [`object_references`](Self::object_references); only the stored
415    /// element differs, carrying the reference's kind alongside the address so
416    /// one datatype can hold every reference kind. h5py cannot read it, so
417    /// prefer the pre-1.12 form for files h5py will open.
418    ///
419    /// ```no_run
420    /// # use rust_hdf5::H5File;
421    /// let file = H5File::create("stdrefs.h5").unwrap();
422    /// file.new_dataset::<i32>().shape([4]).create("target").unwrap();
423    /// let refs = file.new_dataset::<u64>()
424    ///     .std_object_references()
425    ///     .shape([1])
426    ///     .create("refs")
427    ///     .unwrap();
428    /// refs.write_object_references(&["/target"]).unwrap();
429    /// file.close().unwrap();
430    /// ```
431    #[must_use]
432    pub fn std_object_references(mut self) -> Self {
433        self.references = Some(ReferenceElement::Revised);
434        self
435    }
436
437    /// Store revised region references — `H5R_DATASET_REGION2`, written into
438    /// the same `H5T_STD_REF` datatype
439    /// [`std_object_references`](Self::std_object_references) makes.
440    ///
441    /// The elements are written with
442    /// [`write_std_region_references`](H5Dataset::write_std_region_references).
443    /// What distinguishes them from the pre-1.12
444    /// [`region_references`](Self::region_references) is the element, not the
445    /// datatype: a 1.12 element names its own kind, so one dataset of this type
446    /// may hold object, region and attribute references together. h5py 3.15
447    /// cannot read any of them, so prefer the pre-1.12 form for files h5py will
448    /// open.
449    ///
450    /// ```no_run
451    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, LibverBound, Selection};
452    /// let file = H5File::options().libver(LibverBound::V112).create("stdregions.h5").unwrap();
453    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
454    /// let refs = file.new_dataset::<u64>()
455    ///     .std_region_references()
456    ///     .shape([1])
457    ///     .create("refs")
458    ///     .unwrap();
459    /// let rows = Selection::Hyperslab {
460    ///     rank: 1,
461    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
462    /// };
463    /// refs.write_std_region_references(&[("/target", rows)]).unwrap();
464    /// file.close().unwrap();
465    /// ```
466    #[must_use]
467    pub fn std_region_references(self) -> Self {
468        self.std_object_references()
469    }
470
471    /// Store attribute references — `H5R_ATTR`, the one reference kind with no
472    /// pre-1.12 form, in the same `H5T_STD_REF` datatype
473    /// [`std_object_references`](Self::std_object_references) makes.
474    ///
475    /// The elements are written with
476    /// [`write_attribute_references`](H5Dataset::write_attribute_references) and
477    /// name an object and one of its attributes. h5py 3.15 cannot read them.
478    ///
479    /// ```no_run
480    /// # use rust_hdf5::{H5File, LibverBound};
481    /// let file = H5File::options().libver(LibverBound::V112).create("attrrefs.h5").unwrap();
482    /// let target = file.new_dataset::<i32>().shape([4]).create("target").unwrap();
483    /// target.new_attr::<i32>().shape([3]).create("note").unwrap()
484    ///     .write_array(&[7i32, 8, 9]).unwrap();
485    /// let refs = file.new_dataset::<u64>()
486    ///     .attribute_references()
487    ///     .shape([1])
488    ///     .create("refs")
489    ///     .unwrap();
490    /// refs.write_attribute_references(&[("/target", "note")]).unwrap();
491    /// file.close().unwrap();
492    /// ```
493    #[must_use]
494    pub fn attribute_references(self) -> Self {
495        self.std_object_references()
496    }
497
498    /// Store dataset region references — h5py's `h5py.regionref_dtype`.
499    ///
500    /// The elements are written with
501    /// [`write_region_references`](H5Dataset::write_region_references) and name
502    /// a dataset plus a selection over it. The element is a global-heap id, so
503    /// its width follows the file's address size and the datatype is resolved
504    /// at [`create`](Self::create) rather than here; it overrides both `T` and
505    /// any [`datatype`](Self::datatype) call.
506    ///
507    /// ```no_run
508    /// # use rust_hdf5::{H5File, Hyperslab, HyperslabBlock, Selection};
509    /// let file = H5File::create("regions.h5").unwrap();
510    /// file.new_dataset::<i32>().shape([8]).create("target").unwrap();
511    /// let refs = file.new_dataset::<u64>()
512    ///     .region_references()
513    ///     .shape([1])
514    ///     .create("refs")
515    ///     .unwrap();
516    /// let rows = Selection::Hyperslab {
517    ///     rank: 1,
518    ///     form: Hyperslab::Blocks(vec![HyperslabBlock { start: vec![0], end: vec![2] }]),
519    /// };
520    /// refs.write_region_references(&[("/target", rows)]).unwrap();
521    /// file.close().unwrap();
522    /// ```
523    #[must_use]
524    pub fn region_references(mut self) -> Self {
525        self.references = Some(ReferenceElement::Region);
526        self
527    }
528
529    /// Keep the raw data in files outside this one — `H5Pset_external`,
530    /// h5py's `external=[(name, offset, size)]`.
531    ///
532    /// Each entry is a file name, the byte offset in it where that entry's
533    /// region starts, and how many bytes of the dataset it holds; the entries
534    /// concatenate, in order, into the dataset's bytes and together must cover
535    /// them. A relative name is resolved against `HDF5_EXTFILE_PREFIX` the way
536    /// libhdf5 resolves it, so the same name reads back through this crate and
537    /// through h5py. The storage is contiguous by definition, which rules out
538    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
539    /// [`null`](Self::null) and either reference kind.
540    ///
541    /// The named files are created on first write and never truncated, so
542    /// several datasets may own disjoint ranges of one file.
543    ///
544    /// The last entry may take the unlimited size
545    /// [`external_file_list::UNLIMITED`](crate::format::messages::external_file_list::UNLIMITED)
546    /// (`H5O_EFL_UNLIMITED`), which makes it absorb the whole rest of the
547    /// dataset however far it grows. A dataset whose
548    /// [`max_shape`](Self::max_shape) is unlimited must have one, since no
549    /// finite reservation could cover it, and only the first dimension may be
550    /// extendible — both `H5D__efl_construct`'s rules.
551    ///
552    /// ```no_run
553    /// # use rust_hdf5::H5File;
554    /// let file = H5File::create("ext.h5").unwrap();
555    /// let ds = file.new_dataset::<i32>()
556    ///     .shape([16])
557    ///     .external(&[("ext.raw", 0, 64)])
558    ///     .create("data")
559    ///     .unwrap();
560    /// ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
561    /// ```
562    #[must_use]
563    pub fn external(mut self, files: &[(&str, u64, u64)]) -> Self {
564        self.external = Some(
565            files
566                .iter()
567                .map(|&(name, offset, size)| (name.to_string(), offset, size))
568                .collect(),
569        );
570        self
571    }
572
573    /// `H5Pset_efile_prefix` on the dapl `H5Dcreate2` takes — the directory
574    /// the raw data files named by [`external`](Self::external) are created
575    /// under, and looked for under on every later write through this handle.
576    ///
577    /// `H5D__create` builds `dset->shared->extfile_prefix` from the dapl
578    /// (H5Dint.c:1318) and `H5D__efl_write` joins each slot name against it
579    /// with the same single-path `H5_combine_path` the read side uses
580    /// (H5Defl.c:429-431) — so this decides where the bytes land, and the
581    /// prefix a later reader names must agree for it to find them.
582    ///
583    /// Measured under libhdf5 1.14.6 and 2.0.0: writing through a dapl that
584    /// names a directory creates the raw data file there and nowhere else,
585    /// and a directory that does not exist fails the write outright rather
586    /// than being created.
587    ///
588    /// It shares [`DatasetAccess::efile_prefix`]'s rules, both being
589    /// `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX` shadows this outright
590    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
591    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
592    /// (:1098-1102), which leaves a stored name to resolve against the
593    /// process's current directory.
594    ///
595    /// Ignored by a dataset that names no external files, which has no slot
596    /// name to join.
597    #[must_use]
598    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
599        self.efile_prefix = Some(prefix.into());
600        self
601    }
602
603    /// Map part of this dataset onto part of a dataset in another file, making
604    /// it virtual — `H5Pset_virtual`, one `VirtualLayout[...] =
605    /// VirtualSource(...)` assignment in h5py.
606    ///
607    /// The arguments are `H5Pset_virtual`'s, in its order: which elements of
608    /// *this* dataset the mapping fills, the file and dataset the data comes
609    /// from, and which elements of that source dataset it comes from. Call it
610    /// once per mapping; they apply in the order given, which is the order
611    /// libhdf5 resolves overlapping ones in.
612    ///
613    /// The source file is named exactly as stored — resolved against
614    /// `HDF5_VDS_PREFIX`, or the virtual dataset's own directory, when the
615    /// file is read — and `"."` means this file. Nothing is opened or checked
616    /// here: a source that does not exist yet is legal, and reads of the
617    /// unmapped or unresolvable parts return the [`fill_value`](Self::fill_value).
618    ///
619    /// A virtual dataset stores nothing of its own, which rules out
620    /// [`chunk`](Self::chunk), a filter, [`compact`](Self::compact),
621    /// [`null`](Self::null), [`external`](Self::external) and either reference
622    /// kind — and makes writing to it an error, since its elements belong to
623    /// the source datasets.
624    ///
625    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: such a
626    /// mapping grows with its source, and the dataset's extent in that
627    /// dimension is whatever the sources reachable when it is opened supply
628    /// (`H5D__virtual_set_extent_unlim`). Give it a
629    /// [`max_shape`](Self::max_shape) unlimited in the same dimension, as
630    /// libhdf5 requires of the dataspace behind one.
631    ///
632    /// A source name may carry libhdf5's `printf`-style substitutions: `%b`
633    /// is the block index and `%%` an escaped literal `%`. One such mapping
634    /// stands for the family of source datasets that fill the successive
635    /// blocks of an unlimited virtual selection, so it is legal only with an
636    /// unlimited virtual selection over a limited source selection, and the
637    /// dataset's extent stops at the first block whose source is missing.
638    ///
639    /// ```no_run
640    /// # use rust_hdf5::{H5File, Selection};
641    /// let file = H5File::create("vds.h5").unwrap();
642    /// let ds = file.new_dataset::<i32>()
643    ///     .shape([16])
644    ///     .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
645    ///     .create("vds")
646    ///     .unwrap();
647    /// ```
648    #[must_use]
649    pub fn virtual_mapping(
650        mut self,
651        virtual_selection: Selection,
652        source_file: &str,
653        source_dataset: &str,
654        source_selection: Selection,
655    ) -> Self {
656        self.virtual_mappings.push(VirtualMapping {
657            source_file_name: source_file.to_string(),
658            source_dset_name: source_dataset.to_string(),
659            source_selection,
660            virtual_selection,
661        });
662        self
663    }
664
665    /// Set a user-defined fill value for unwritten elements.
666    ///
667    /// Without this, datasets use the HDF5 default zero-fill. When set,
668    /// the value is written into the dataset's fill-value message
669    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
670    /// unwritten regions as this value rather than zero.
671    ///
672    /// ```no_run
673    /// # use rust_hdf5::H5File;
674    /// let file = H5File::create("fv.h5").unwrap();
675    /// let ds = file.new_dataset::<f32>()
676    ///     .shape(&[100])
677    ///     .fill_value(f32::NAN)
678    ///     .create("data")
679    ///     .unwrap();
680    /// ```
681    #[must_use]
682    pub fn fill_value(mut self, value: T) -> Self {
683        let es = T::element_size();
684        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
685        // well-defined byte representation; `element_size()` matches
686        // `size_of::<T>()`. The slice borrows `value` only for this call.
687        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
688        self.fill_value = Some(raw.to_vec());
689        self
690    }
691
692    /// Set when the fill value is written into allocated storage —
693    /// `H5Pset_fill_time`.
694    ///
695    /// Without this, a dataset gets [`FillTime::IfSet`]
696    /// (`H5D_CRT_FILL_TIME_DEF`), the default every dataset creation
697    /// property list carries. [`FillTime::Never`] applies to a dataset with
698    /// no fill value too: it only stops this writer's own eager tiling of
699    /// the value into newly allocated storage, not the default zero-fill
700    /// that storage already has, so its only observable effect is on a
701    /// dataset that also calls [`fill_value`](Self::fill_value).
702    ///
703    /// ```no_run
704    /// # use rust_hdf5::{FillTime, H5File};
705    /// let file = H5File::create("fv.h5").unwrap();
706    /// let ds = file.new_dataset::<f32>()
707    ///     .shape(&[100])
708    ///     .fill_value(f32::NAN)
709    ///     .fill_time(FillTime::Never)
710    ///     .create("data")
711    ///     .unwrap();
712    /// ```
713    #[must_use]
714    pub fn fill_time(mut self, time: FillTime) -> Self {
715        self.fill_time = Some(time);
716        self
717    }
718
719    /// Finalize and create the dataset with the given `name`.
720    ///
721    /// The name is the link name within the root group (e.g. `"data"` or
722    /// `"group1/data"` once nested groups are supported).
723    pub fn create(self, name: &str) -> Result<H5Dataset> {
724        // A committed type is resolved before the dataset exists and recorded
725        // after, here rather than in each storage path: every path reaches
726        // this one return, so a dataset can never be built on a committed
727        // type and then fail to say so — which would silently write the type
728        // out in full instead of pointing at the object.
729        let committed = self.resolve_committed_type()?;
730        let file_inner = clone_inner(&self.file_inner);
731        let ds = self.create_object(name, committed.as_ref().map(|(_, dt)| dt.clone()))?;
732        if let (Some((share, _)), DatasetInfo::Writer { index, .. }) = (committed, &ds.info) {
733            let inner = borrow_inner(&file_inner);
734            if let H5FileInner::Writer(writer) = &*inner {
735                writer.share_committed_type(*index, share);
736            }
737        }
738        Ok(ds)
739    }
740
741    /// The committed datatype this dataset is built on, with the type it
742    /// holds; `None` when [`committed_type`](Self::committed_type) was not
743    /// called.
744    fn resolve_committed_type(&self) -> Result<Option<(usize, DatatypeMessage)>> {
745        let Some(path) = self.committed_type.as_deref() else {
746            return Ok(None);
747        };
748        if self.references.is_some() {
749            // Both name the stored type and they cannot both be it: the
750            // pointer would say the elements are the committed type while the
751            // reference writers write addresses. True of either reference
752            // kind — an object reference is an address, a region reference is
753            // a global-heap address plus a serialized selection.
754            return Err(Hdf5Error::InvalidState(
755                "a dataset cannot be built on a committed datatype and hold references".into(),
756            ));
757        }
758        let inner = borrow_inner(&self.file_inner);
759        match &*inner {
760            H5FileInner::Writer(writer) => Ok(Some(writer.committed_datatype_for_share(path)?)),
761            H5FileInner::Reader(_) => Err(Hdf5Error::InvalidState(
762                "cannot create a dataset in read mode".into(),
763            )),
764            H5FileInner::Closed => Err(Hdf5Error::InvalidState("file is closed".into())),
765        }
766    }
767
768    /// Everything [`create`](Self::create) does apart from recording the
769    /// committed-type share; `committed` is the type that object holds.
770    fn create_object(self, name: &str, committed: Option<DatatypeMessage>) -> Result<H5Dataset> {
771        // Build the full name: if created within a group, prefix with group path
772        let full_name = if let Some(ref gp) = self.group_path {
773            if gp == "/" {
774                name.to_string()
775            } else {
776                let trimmed = gp.trim_start_matches('/');
777                format!("{}/{}", trimmed, name)
778            }
779        } else {
780            name.to_string()
781        };
782
783        let datatype = if let Some(kind) = self.references {
784            // The element is measured in file addresses, and only the writer
785            // knows how wide one is for this file.
786            let inner = borrow_inner(&self.file_inner);
787            match &*inner {
788                H5FileInner::Writer(writer) => kind.datatype(writer.ctx()),
789                H5FileInner::Reader(_) => {
790                    return Err(Hdf5Error::InvalidState(
791                        "cannot create a dataset in read mode".into(),
792                    ))
793                }
794                H5FileInner::Closed => {
795                    return Err(Hdf5Error::InvalidState("file is closed".into()))
796                }
797            }
798        } else if let Some(dt) = committed {
799            // The object header holds the type; the dataset stores a pointer
800            // to it, but every size and payload check still needs the type
801            // itself.
802            dt
803        } else {
804            self.datatype_override.clone().unwrap_or_else(T::hdf5_type)
805        };
806        // Size one element from the on-disk datatype, not the carrier `T`. For
807        // the default path this equals `T::element_size()`; when a `datatype()`
808        // override is set (N-bit, or a runtime `CompoundType`), the stored type
809        // — not `T` — defines the element width, so the dataspace, the raw
810        // allocation, and the `write_raw` length check all agree with the bytes
811        // libhdf5/h5py will read.
812        let element_size = datatype.element_size() as usize;
813        // `fill_value` took the host image of a `T`; the fill-value message
814        // holds one element in the dataset's own datatype, so it is converted
815        // here — the order is only known once the override is resolved, and
816        // the builder's calls can arrive in either order.
817        let fill_value = match self.fill_value.as_deref() {
818            Some(bytes) => Some(to_stored_byte_order(bytes, &datatype, element_size)?.into_owned()),
819            None => None,
820        };
821
822        let wants_filter =
823            self.custom_pipeline.is_some() || self.shuffle || self.deflate_level.is_some();
824
825        // External storage *is* contiguous storage: the layout message says
826        // contiguous with an undefined address, and the External File List
827        // beside it says where the bytes really are. Every other storage class
828        // names bytes of its own, so none of them can also name these.
829        if self.external.is_some() {
830            if self.chunk_dims.is_some() || wants_filter || self.is_compact || self.is_null {
831                return Err(Hdf5Error::InvalidState(
832                    "a dataset whose raw data lives in external files is contiguous, so it \
833                     cannot also be chunked, filtered, compact or NULL"
834                        .into(),
835                ));
836            }
837            if self.references.is_some() {
838                return Err(Hdf5Error::InvalidState(
839                    "object and region references are stamped into the dataset's own \
840                     contiguous block, which a dataset stored in external files has none of"
841                        .into(),
842                ));
843            }
844        }
845
846        // A virtual dataset stores nothing of its own — its elements are read
847        // out of the datasets its mappings name — so it can be none of the
848        // storage classes that do, and there is no block for a reference
849        // writer to stamp into either.
850        if !self.virtual_mappings.is_empty() {
851            if self.chunk_dims.is_some()
852                || wants_filter
853                || self.is_compact
854                || self.is_null
855                || self.external.is_some()
856            {
857                return Err(Hdf5Error::InvalidState(
858                    "a virtual dataset's elements live in the datasets its mappings name, \
859                     so it cannot also be chunked, filtered, compact, NULL or stored in \
860                     external files"
861                        .into(),
862                ));
863            }
864            if self.references.is_some() {
865                return Err(Hdf5Error::InvalidState(
866                    "object and region references are stamped into the dataset's own \
867                     contiguous block, which a virtual dataset has none of"
868                        .into(),
869                ));
870            }
871        }
872
873        if self.is_null {
874            // A NULL dataspace holds no elements at all: no chunk grid to
875            // scatter into, no raw image to put in an object header, no fill
876            // value to apply to unwritten elements (there are none), matching
877            // upstream's rejection of these combinations (`H5Dchunk.c`'s
878            // chunked-layout dataspace check).
879            if self.chunk_dims.is_some() || wants_filter || self.is_compact {
880                return Err(Hdf5Error::InvalidState(
881                    "a NULL dataspace dataset cannot be chunked, filtered or compact".into(),
882                ));
883            }
884            if fill_value.is_some() {
885                return Err(Hdf5Error::InvalidState(
886                    "a NULL dataspace dataset cannot have a fill value".into(),
887                ));
888            }
889            if self.fill_time.is_some() {
890                return Err(Hdf5Error::InvalidState(
891                    "a NULL dataspace dataset cannot have a fill time".into(),
892                ));
893            }
894
895            let index = {
896                let inner = borrow_inner(&self.file_inner);
897                match &*inner {
898                    H5FileInner::Writer(writer) => {
899                        let idx = writer.create_null_dataset(&full_name, datatype)?;
900                        if let Some(ref gp) = self.group_path {
901                            if gp != "/" {
902                                writer.assign_dataset_to_group(gp, idx)?;
903                            }
904                        }
905                        idx
906                    }
907                    H5FileInner::Reader(_) => {
908                        return Err(Hdf5Error::InvalidState(
909                            "cannot create a dataset in read mode".into(),
910                        ));
911                    }
912                    H5FileInner::Closed => {
913                        return Err(Hdf5Error::InvalidState("file is closed".into()));
914                    }
915                }
916            };
917
918            return Ok(H5Dataset {
919                file_inner: clone_inner(&self.file_inner),
920                info: DatasetInfo::Writer {
921                    index,
922                    shape: Vec::new(),
923                    element_size,
924                    chunk_index: None,
925                    is_null: true,
926                },
927                _open: None,
928            });
929        }
930
931        let shape = self.shape.ok_or_else(|| {
932            Hdf5Error::InvalidState("shape must be set before calling create()".into())
933        })?;
934        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
935
936        if self.is_compact {
937            // The raw data is the layout message, so there is no chunk grid to
938            // filter and no room to grow into: `H5D__compact_construct` refuses
939            // a max dimension above the current one, and `H5Pset_layout` and
940            // `H5Pset_chunk` overwrite each other rather than combining.
941            if self.chunk_dims.is_some() || wants_filter {
942                return Err(Hdf5Error::InvalidState(
943                    "a compact dataset stores its data in the object header, so it \
944                     cannot be chunked or filtered"
945                        .into(),
946                ));
947            }
948            if self
949                .max_shape
950                .as_ref()
951                .is_some_and(|max| max.iter().zip(&shape).any(|(m, &d)| *m != Some(d)))
952            {
953                return Err(Hdf5Error::InvalidState(
954                    "a compact dataset cannot be extendible: its maximum shape must \
955                     equal its shape"
956                        .into(),
957                ));
958            }
959
960            let index = {
961                let inner = borrow_inner(&self.file_inner);
962                match &*inner {
963                    H5FileInner::Writer(writer) => {
964                        let idx = writer.create_compact_dataset(&full_name, datatype, &dims_u64)?;
965                        // Set before the fill value: NEVER must be in place
966                        // before that call decides whether to eager-tile it.
967                        if let Some(time) = self.fill_time {
968                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
969                        }
970                        if let Some(ref fv) = fill_value {
971                            writer.set_dataset_fill_value(idx, fv.clone())?;
972                        }
973                        idx
974                    }
975                    H5FileInner::Reader(_) => {
976                        return Err(Hdf5Error::InvalidState(
977                            "cannot create a dataset in read mode".into(),
978                        ));
979                    }
980                    H5FileInner::Closed => {
981                        return Err(Hdf5Error::InvalidState("file is closed".into()));
982                    }
983                }
984            };
985
986            return Ok(H5Dataset {
987                file_inner: clone_inner(&self.file_inner),
988                info: DatasetInfo::Writer {
989                    index,
990                    shape,
991                    element_size,
992                    chunk_index: None,
993                    is_null: false,
994                },
995                _open: None,
996            });
997        }
998
999        if !self.virtual_mappings.is_empty() {
1000            let index = {
1001                let inner = borrow_inner(&self.file_inner);
1002                match &*inner {
1003                    H5FileInner::Writer(writer) => {
1004                        let idx = writer.create_virtual_dataset(
1005                            &full_name,
1006                            datatype,
1007                            &dims_u64,
1008                            self.max_shape
1009                                .as_ref()
1010                                .map(|max| {
1011                                    max.iter()
1012                                        .map(|m| m.map_or(u64::MAX, |v| v as u64))
1013                                        .collect::<Vec<u64>>()
1014                                })
1015                                .as_deref(),
1016                            &self.virtual_mappings,
1017                        )?;
1018                        // The fill value is what a read of an unmapped — or
1019                        // unresolvable — element returns, so it is the one
1020                        // dataset property a virtual dataset carries about its
1021                        // own elements. Nothing is tiled into storage: it has
1022                        // none.
1023                        // Set before the fill value: NEVER must be in place
1024                        // before that call decides whether to eager-tile it.
1025                        if let Some(time) = self.fill_time {
1026                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1027                        }
1028                        if let Some(ref fv) = fill_value {
1029                            writer.set_dataset_fill_value(idx, fv.clone())?;
1030                        }
1031                        idx
1032                    }
1033                    H5FileInner::Reader(_) => {
1034                        return Err(Hdf5Error::InvalidState(
1035                            "cannot create a dataset in read mode".into(),
1036                        ));
1037                    }
1038                    H5FileInner::Closed => {
1039                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1040                    }
1041                }
1042            };
1043
1044            return Ok(H5Dataset {
1045                file_inner: clone_inner(&self.file_inner),
1046                info: DatasetInfo::Writer {
1047                    index,
1048                    shape,
1049                    element_size,
1050                    chunk_index: None,
1051                    is_null: false,
1052                },
1053                _open: None,
1054            });
1055        }
1056
1057        // A filter pipeline requires chunked storage. When a filter is
1058        // requested without explicit chunk dimensions, store the whole
1059        // dataset as a single chunk instead of silently dropping the filter
1060        // on the contiguous path. (This is one whole-dataset chunk, not
1061        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
1062        // for large datasets.)
1063        let auto_chunk: Option<Vec<usize>> =
1064            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
1065                Some(shape.iter().map(|&d| d.max(1)).collect())
1066            } else {
1067                None
1068            };
1069
1070        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
1071            // Chunked dataset
1072            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
1073            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
1074                max.iter()
1075                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
1076                    .collect()
1077            } else {
1078                // Default: max = current
1079                dims_u64.clone()
1080            };
1081
1082            // The file's format settles the question before the shape gets a
1083            // say. `H5D__chunk_set_info` reaches the index-selection block
1084            // only once the data layout message is at version 4 (H5Dchunk.c:936)
1085            // — which the file's library-version bound decides, not the
1086            // dataspace — and below it the version-3 message carries a
1087            // version-1 B-tree and nothing else. The writer owns that reading
1088            // of `H5O_layout_ver_bounds`; the chunk's byte count is the one
1089            // input from here, a chunk over 4 GiB being the one thing that
1090            // forces the newer message whatever the bound says.
1091            let chunk_bytes = chunk_u64.iter().product::<u64>() * element_size as u64;
1092            let v110_indexing = match &*borrow_inner(&self.file_inner) {
1093                H5FileInner::Writer(writer) => writer.uses_v110_chunk_indexing(chunk_bytes),
1094                // Neither can create a dataset at all; the creator below
1095                // reports which of the two it is.
1096                _ => true,
1097            };
1098
1099            // Inside the block libhdf5 selects the chunk index from the
1100            // dataspace and the creation properties, in this order
1101            // (`H5D__chunk_set_info`, H5Dchunk.c:955): a v2 B-tree for two or
1102            // more unlimited dimensions, an extensible array for exactly one;
1103            // for a fixed shape, the single-chunk index takes priority —
1104            // unconditional of filter or allocation time — whenever the shape
1105            // is exactly one whole chunk, ahead of the implicit index (no
1106            // filter, and early allocation, which is what puts every chunk at
1107            // a computable address) and the fixed array (everything else).
1108            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
1109            let one_chunk = chunk_u64 == dims_u64 && max_u64 == dims_u64;
1110            let kind = if !v110_indexing {
1111                ChunkIndexKind::BtreeV1
1112            } else if n_unlimited >= 2 {
1113                ChunkIndexKind::BtreeV2
1114            } else if n_unlimited == 1 {
1115                ChunkIndexKind::ExtensibleArray
1116            } else if one_chunk {
1117                ChunkIndexKind::SingleChunk
1118            } else if self.early_allocation && !wants_filter {
1119                ChunkIndexKind::Implicit
1120            } else {
1121                ChunkIndexKind::FixedArray
1122            };
1123
1124            let index = {
1125                let inner = borrow_inner(&self.file_inner);
1126                match &*inner {
1127                    H5FileInner::Writer(writer) => {
1128                        // The requested filter pipeline, if any. Every index
1129                        // builds it from the same options, so one owner
1130                        // resolves it: a second construction site is what let
1131                        // a request naming no compressor — shuffle on its own
1132                        // — fall through to unfiltered storage.
1133                        let explicit_pipeline = || {
1134                            use crate::format::messages::filter::FilterPipeline;
1135                            if let Some(p) = self.custom_pipeline.clone() {
1136                                return p;
1137                            }
1138                            // Shuffle records the width of the element it
1139                            // permutes, which is the stored one — a `datatype`
1140                            // override moves that away from `T`.
1141                            let es = element_size as u32;
1142                            match (self.shuffle, self.deflate_level) {
1143                                (true, Some(level)) => FilterPipeline::shuffle_deflate(es, level),
1144                                (true, None) => FilterPipeline::shuffle(es),
1145                                // deflate_level (checked by wants_filter).
1146                                (false, level) => FilterPipeline::deflate(level.unwrap()),
1147                            }
1148                        };
1149                        let idx = if kind == ChunkIndexKind::BtreeV1 {
1150                            // The classic index, which takes the pipeline the
1151                            // same way the others do — and is refused with it
1152                            // in a classic file, whose filter pipeline
1153                            // message is a version this crate does not write.
1154                            let pipeline = wants_filter.then(explicit_pipeline);
1155                            writer.create_btree_v1_dataset(
1156                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1157                            )?
1158                        } else if kind == ChunkIndexKind::BtreeV2 {
1159                            // Two or more unlimited dimensions: a v2 B-tree,
1160                            // whose records carry the stored size and filter
1161                            // mask when the dataset is compressed (libhdf5
1162                            // H5D_BT2_FILT).
1163                            if wants_filter {
1164                                writer.create_btree_v2_dataset_with_pipeline(
1165                                    &full_name,
1166                                    datatype,
1167                                    &dims_u64,
1168                                    &max_u64,
1169                                    &chunk_u64,
1170                                    explicit_pipeline(),
1171                                )?
1172                            } else {
1173                                writer.create_btree_v2_dataset(
1174                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1175                                )?
1176                            }
1177                        } else if kind == ChunkIndexKind::Implicit {
1178                            // No index structure at all: every chunk of the
1179                            // grid is allocated at create in one run, so
1180                            // there is no pipeline arm — a filter is what
1181                            // makes chunks different sizes, and this index
1182                            // has no room to say so.
1183                            writer.create_implicit_dataset(
1184                                &full_name, datatype, &dims_u64, &chunk_u64,
1185                            )?
1186                        } else if kind == ChunkIndexKind::SingleChunk {
1187                            // A fixed shape covered by exactly one chunk:
1188                            // libhdf5 picks this index ahead of Implicit and
1189                            // Fixed Array regardless of filter or allocation
1190                            // time. Filtered or not, it takes the same
1191                            // explicit pipeline the other indexes do; a
1192                            // filtered chunk's stored size isn't known ahead
1193                            // of its first write, so early allocation only
1194                            // ever applies to the unfiltered form.
1195                            if wants_filter {
1196                                writer.create_single_chunk_dataset_with_pipeline(
1197                                    &full_name,
1198                                    datatype,
1199                                    &dims_u64,
1200                                    &chunk_u64,
1201                                    explicit_pipeline(),
1202                                )?
1203                            } else {
1204                                writer.create_single_chunk_dataset(
1205                                    &full_name,
1206                                    datatype,
1207                                    &dims_u64,
1208                                    &chunk_u64,
1209                                    self.early_allocation,
1210                                )?
1211                            }
1212                        } else if kind == ChunkIndexKind::FixedArray {
1213                            // A chunked dataset with no unlimited dimension
1214                            // must use the fixed-array index — libhdf5
1215                            // rejects an extensible-array index here. A
1216                            // compressed fixed-shape dataset uses a *filtered*
1217                            // fixed array (FA client id 1). The maximum shape
1218                            // sizes the array, so a finite max above the
1219                            // current shape stays growable.
1220                            let pipeline = wants_filter.then(explicit_pipeline);
1221                            writer.create_fixed_array_dataset_with_max(
1222                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
1223                            )?
1224                        } else if wants_filter {
1225                            // The extensible-array index takes the pipeline
1226                            // the same way, so it goes through the one owner
1227                            // too.
1228                            writer.create_chunked_dataset_with_pipeline(
1229                                &full_name,
1230                                datatype,
1231                                &dims_u64,
1232                                &max_u64,
1233                                &chunk_u64,
1234                                explicit_pipeline(),
1235                            )?
1236                        } else {
1237                            writer.create_chunked_dataset(
1238                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
1239                            )?
1240                        };
1241                        // Set before the fill value: NEVER must be in place
1242                        // before that call decides whether to eager-tile it.
1243                        if let Some(time) = self.fill_time {
1244                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1245                        }
1246                        if let Some(ref fv) = fill_value {
1247                            writer.set_dataset_fill_value(idx, fv.clone())?;
1248                        }
1249                        idx
1250                    }
1251                    H5FileInner::Reader(_) => {
1252                        return Err(Hdf5Error::InvalidState(
1253                            "cannot create a dataset in read mode".into(),
1254                        ));
1255                    }
1256                    H5FileInner::Closed => {
1257                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1258                    }
1259                }
1260            };
1261
1262            Ok(H5Dataset {
1263                file_inner: clone_inner(&self.file_inner),
1264                info: DatasetInfo::Writer {
1265                    index,
1266                    shape,
1267                    element_size,
1268                    chunk_index: Some(kind),
1269                    is_null: false,
1270                },
1271                _open: None,
1272            })
1273        } else {
1274            // Contiguous dataset (original path)
1275            let efile_access = match self.efile_prefix.as_deref() {
1276                Some(p) => DatasetAccess::new().efile_prefix(p),
1277                None => DatasetAccess::new(),
1278            };
1279            let (index, open) = {
1280                let inner = borrow_inner(&self.file_inner);
1281                match &*inner {
1282                    H5FileInner::Writer(writer) => {
1283                        let idx = match self.external.as_deref() {
1284                            Some(files) => {
1285                                let slots: Vec<(&str, u64, u64)> = files
1286                                    .iter()
1287                                    .map(|(name, offset, size)| (name.as_str(), *offset, *size))
1288                                    .collect();
1289                                writer.create_external_dataset(
1290                                    &full_name,
1291                                    datatype,
1292                                    &dims_u64,
1293                                    self.max_shape
1294                                        .as_ref()
1295                                        .map(|max| {
1296                                            max.iter()
1297                                                .map(|m| m.map_or(u64::MAX, |v| v as u64))
1298                                                .collect::<Vec<u64>>()
1299                                        })
1300                                        .as_deref(),
1301                                    &slots,
1302                                )?
1303                            }
1304                            None => writer.create_dataset(&full_name, datatype, &dims_u64)?,
1305                        };
1306                        // Before anything that can write raw bytes: the
1307                        // prefix an external dataset's slot names are joined
1308                        // against is settled by the create, as
1309                        // `H5D__build_file_prefix` settles it for
1310                        // `H5D__create` (H5Dint.c:1318).
1311                        let open = writer.bind_efile_prefix(idx, &efile_access)?;
1312                        // Set before the fill value: NEVER must be in place
1313                        // before that call decides whether to eager-tile it.
1314                        if let Some(time) = self.fill_time {
1315                            writer.set_dataset_fill_time(idx, time.wire_byte())?;
1316                        }
1317                        if let Some(ref fv) = fill_value {
1318                            writer.set_dataset_fill_value(idx, fv.clone())?;
1319                        }
1320                        (idx, open)
1321                    }
1322                    H5FileInner::Reader(_) => {
1323                        return Err(Hdf5Error::InvalidState(
1324                            "cannot create a dataset in read mode".into(),
1325                        ));
1326                    }
1327                    H5FileInner::Closed => {
1328                        return Err(Hdf5Error::InvalidState("file is closed".into()));
1329                    }
1330                }
1331            };
1332
1333            Ok(H5Dataset {
1334                file_inner: clone_inner(&self.file_inner),
1335                info: DatasetInfo::Writer {
1336                    index,
1337                    shape,
1338                    element_size,
1339                    chunk_index: None,
1340                    is_null: false,
1341                },
1342                _open: open,
1343            })
1344        }
1345    }
1346}
1347
1348// ---------------------------------------------------------------------------
1349// DatasetInfo
1350// ---------------------------------------------------------------------------
1351
1352/// Internal metadata about a dataset handle.
1353enum DatasetInfo {
1354    /// A dataset created via `new_dataset().create()` in write mode.
1355    Writer {
1356        /// Index into the writer's dataset list.
1357        index: usize,
1358        /// Shape (current dimensions).
1359        shape: Vec<usize>,
1360        /// Size of one element in bytes.
1361        element_size: usize,
1362        /// Which chunk index this dataset uses, `None` when its storage is
1363        /// not chunked. One field rather than a flag per index: a dataset
1364        /// has exactly one chunk index, and the flags could spell
1365        /// combinations ("not chunked, but indexed by a v2 B-tree") that no
1366        /// dataset has — which is what a write path reading only some of
1367        /// them turns into a write to the wrong index.
1368        chunk_index: Option<ChunkIndexKind>,
1369        /// Whether this is a NULL dataspace (no elements at all — distinct
1370        /// from a scalar, which holds exactly one). Always `false` when
1371        /// `chunk_index` is `Some`: a NULL dataspace can never be chunked.
1372        is_null: bool,
1373    },
1374    /// A dataset opened by name in read mode.
1375    Reader {
1376        /// The link name of the dataset.
1377        name: String,
1378        /// Shape (current dimensions).
1379        shape: Vec<usize>,
1380        /// Size of one element in bytes.
1381        element_size: usize,
1382    },
1383}
1384
1385// ---------------------------------------------------------------------------
1386// H5Dataset
1387// ---------------------------------------------------------------------------
1388
1389/// A handle to an HDF5 dataset, supporting typed read and write operations.
1390///
1391/// The dataset holds a shared reference to the file's I/O backend, so it
1392/// remains valid even if the originating [`H5File`](crate::file::H5File) is
1393/// moved or dropped (they share ownership via `Rc`).
1394pub struct H5Dataset {
1395    file_inner: SharedInner,
1396    info: DatasetInfo,
1397    /// Keeps this dataset's *open* alive for as long as the handle is, so
1398    /// the reader or the writer can tell whether a later open of the same
1399    /// name joins this one or starts fresh — libhdf5's `H5FO_opened`
1400    /// shared-info count (H5Dint.c:1496-1500). `None` for a dataset with no
1401    /// per-open answer to hold: in write mode, one whose raw data is in this
1402    /// file rather than in the files an external file list names.
1403    ///
1404    /// Held, never read: its whole job is to keep the reader's `Weak` on it
1405    /// upgradable until this handle goes away.
1406    _open: Option<crate::io::reader::DatasetOpenToken>,
1407}
1408
1409impl Drop for H5Dataset {
1410    /// Closing a virtual dataset's last handle closes the source files that
1411    /// open was holding, which is what `H5D__virtual_reset_layout` does at
1412    /// the last `H5Dclose` (H5Dvirtual.c:709-710, closing each
1413    /// `source_dset->dset` at :955 and with it the file that dataset kept
1414    /// open). Nothing else in this crate can end a virtual open, so this is
1415    /// where the reader is told.
1416    ///
1417    /// The token is dropped *before* the reader is asked, so the reader's
1418    /// `Weak` already reads dead for the handle going away here. A write-mode
1419    /// handle's token belongs to the writer's own external file prefix, which
1420    /// has no source files to close, so it takes no lock either.
1421    fn drop(&mut self) {
1422        let Some(open) = self._open.take() else {
1423            return;
1424        };
1425        drop(open);
1426        if matches!(self.info, DatasetInfo::Writer { .. }) {
1427            return;
1428        }
1429        let Some(mut inner) = crate::file::try_borrow_inner_mut(&self.file_inner) else {
1430            return;
1431        };
1432        if let crate::file::H5FileInner::Reader(reader) = &mut *inner {
1433            reader.release_closed_virtual_sources();
1434        }
1435    }
1436}
1437
1438/// One chunk's bytes on the way to the file, and who filtered them.
1439///
1440/// This is what separates a normal chunk write from a direct one; everything
1441/// else about placing a chunk is identical, so the two share a single dispatch.
1442#[derive(Clone, Copy)]
1443enum ChunkBytes<'a> {
1444    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
1445    /// are stored.
1446    Unfiltered(&'a [u8]),
1447    /// Bytes already in their stored form, with `filter_mask` naming the
1448    /// filters that were skipped.
1449    Prefiltered { data: &'a [u8], filter_mask: u32 },
1450}
1451
1452/// The byte order this build reads and writes natively.
1453pub(crate) const HOST_BYTE_ORDER: ByteOrder = if cfg!(target_endian = "big") {
1454    ByteOrder::BigEndian
1455} else {
1456    ByteOrder::LittleEndian
1457};
1458
1459/// The byte order this build does not read or write natively.
1460pub(crate) const FOREIGN_BYTE_ORDER: ByteOrder = match HOST_BYTE_ORDER {
1461    ByteOrder::LittleEndian => ByteOrder::BigEndian,
1462    ByteOrder::BigEndian => ByteOrder::LittleEndian,
1463};
1464
1465/// What a typed access has to do with an element image of a given datatype.
1466#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1467enum ByteOrderAction {
1468    /// Stored order is the host's: the image is already the typed value.
1469    Keep,
1470    /// The whole element is one scalar in the foreign order: reverse it.
1471    SwapElements,
1472    /// A composite storing something in the foreign order.
1473    Refuse,
1474}
1475
1476/// Classify a datatype for a typed access of element width `width`.
1477///
1478/// The single owner of the rule; both directions ask it, so a type a read
1479/// converts is exactly a type a write converts.
1480///
1481/// A composite element cannot be swapped as a unit — its members have their
1482/// own orders and offsets — so one that touches the foreign order is refused
1483/// rather than silently passed through in the wrong order.
1484fn byte_order_action(datatype: &DatatypeMessage, width: usize) -> ByteOrderAction {
1485    match datatype.scalar_byte_order() {
1486        Some(order) if order == FOREIGN_BYTE_ORDER && width > 1 => ByteOrderAction::SwapElements,
1487        Some(_) => ByteOrderAction::Keep,
1488        None if datatype.contains_byte_order(FOREIGN_BYTE_ORDER) => ByteOrderAction::Refuse,
1489        None => ByteOrderAction::Keep,
1490    }
1491}
1492
1493/// Why the stored image of an element is not already the host image of a
1494/// value of width `width` — `None` when it is, and a copying read would only
1495/// be memcpy-ing bytes it does not touch.
1496///
1497/// The two ways a stored element can need work before it is a value are the
1498/// two conversions a copying read performs in place: a byte-order swap
1499/// ([`to_host_byte_order`]) and the n-bit/scale-offset unpacking
1500/// (`Hdf5Reader::apply_post_filter_conversion`). Asking one question of both
1501/// is what lets a zero-copy view refuse exactly the datatypes a copying read
1502/// would have had to rewrite.
1503#[cfg(feature = "mmap")]
1504pub(crate) fn stored_image_mismatch(
1505    datatype: &DatatypeMessage,
1506    width: usize,
1507) -> Option<&'static str> {
1508    match byte_order_action(datatype, width) {
1509        ByteOrderAction::SwapElements => return Some("they are stored in the foreign byte order"),
1510        ByteOrderAction::Refuse => {
1511            return Some("it is a composite storing members in the foreign byte order")
1512        }
1513        ByteOrderAction::Keep => {}
1514    }
1515    if crate::format::nbit_scaleoffset::datatype_needs_bit_conversion(datatype) {
1516        return Some("the significant bits do not fill the stored element");
1517    }
1518    None
1519}
1520
1521/// Put a raw element image into host byte order, in place, for a typed read.
1522///
1523/// Every path that reinterprets the on-disk image as `T` — `read_raw`,
1524/// `read_slice`, `read_raw_into`, `read_slice_into` and their SWMR
1525/// counterparts — passes through here. Reinterpretation only yields the
1526/// stored value when the stored order is the host's.
1527///
1528/// A refused datatype is one no reinterpretation can decode;
1529/// [`H5Dataset::read_raw_bytes`] hands over the image for the caller to
1530/// decode member by member.
1531///
1532/// `width` is the element size, already checked equal to `T::element_size()`.
1533pub(crate) fn to_host_byte_order(
1534    bytes: &mut [u8],
1535    datatype: &DatatypeMessage,
1536    width: usize,
1537) -> Result<()> {
1538    match byte_order_action(datatype, width) {
1539        ByteOrderAction::Keep => {}
1540        ByteOrderAction::SwapElements => {
1541            for elem in bytes.chunks_exact_mut(width) {
1542                elem.reverse();
1543            }
1544        }
1545        ByteOrderAction::Refuse => {
1546            return Err(Hdf5Error::TypeMismatch(format!(
1547                "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a \
1548                 typed read cannot reinterpret element by element; read_raw_bytes() returns \
1549                 the image to decode member by member"
1550            )))
1551        }
1552    }
1553    Ok(())
1554}
1555
1556/// The stored byte image of each variable-length sequence in a batch.
1557///
1558/// The vlen writers take `&[&[T]]` and store one global-heap object per
1559/// sequence, so each sequence needs the same host-image-to-stored-image step
1560/// [`to_stored_byte_order`] performs for a fixed-shape write — a `T` is
1561/// written from its host bytes, and `T::hdf5_type()` declares little-endian.
1562/// Borrows on a little-endian host, which is every machine that does not have
1563/// to swap.
1564pub(crate) fn vlen_sequence_images<'a, T: H5Type>(
1565    items: &'a [&'a [T]],
1566) -> Result<Vec<std::borrow::Cow<'a, [u8]>>> {
1567    let base = T::hdf5_type();
1568    items
1569        .iter()
1570        .map(|item| {
1571            // Safety: the same contract `write_raw` relies on — `T: Copy +
1572            // 'static` is a numeric primitive whose byte image is its value —
1573            // and the extent comes from the slice itself, so it cannot name
1574            // memory past it. The result borrows `items` and outlives nothing.
1575            let host = unsafe {
1576                std::slice::from_raw_parts(item.as_ptr() as *const u8, std::mem::size_of_val(*item))
1577            };
1578            to_stored_byte_order(host, &base, T::element_size())
1579        })
1580        .collect()
1581}
1582
1583/// Put a typed value's host-order image into the order the datatype declares.
1584///
1585/// The write-side counterpart of [`to_host_byte_order`], and the one place
1586/// every path that hands a `&[T]` to the file — `write_raw`, `write_slice`,
1587/// `append`, and the builder's fill value — turns those bytes into stored
1588/// bytes. A `T` is written from its host image, so a dataset declaring the
1589/// foreign order would otherwise hold host bytes under that declaration: a
1590/// file that is wrong by its own header.
1591///
1592/// Borrows when the declared order is the host's, which is every write that
1593/// does not set a [`datatype`](DatasetBuilder::datatype) override.
1594///
1595/// `width` is the element size, already checked equal to `T::element_size()`.
1596pub(crate) fn to_stored_byte_order<'a>(
1597    bytes: &'a [u8],
1598    datatype: &DatatypeMessage,
1599    width: usize,
1600) -> Result<std::borrow::Cow<'a, [u8]>> {
1601    match byte_order_action(datatype, width) {
1602        ByteOrderAction::Keep => Ok(std::borrow::Cow::Borrowed(bytes)),
1603        ByteOrderAction::SwapElements => {
1604            let mut owned = bytes.to_vec();
1605            for elem in owned.chunks_exact_mut(width) {
1606                elem.reverse();
1607            }
1608            Ok(std::borrow::Cow::Owned(owned))
1609        }
1610        ByteOrderAction::Refuse => Err(Hdf5Error::TypeMismatch(format!(
1611            "dataset datatype {datatype} stores {FOREIGN_BYTE_ORDER:?} values, which a typed \
1612             write cannot lay out element by element; write_raw_bytes() takes the image the \
1613             caller encodes member by member"
1614        ))),
1615    }
1616}
1617
1618/// Strip a fixed-string element's padding, leaving the bytes that carry the
1619/// value.
1620///
1621/// The rule itself lives with the datatype message
1622/// ([`fixed_string_content`]); this adds the element index a reserved padding
1623/// rule needs to be reported against.
1624fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
1625    crate::format::messages::datatype::fixed_string_content(elem, padding).ok_or_else(|| {
1626        Hdf5Error::InvalidState(format!(
1627            "string {index} uses padding rule {padding}, which the format reserves"
1628        ))
1629    })
1630}
1631
1632/// Decode one string element's bytes under the datatype's character set.
1633///
1634/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
1635/// `index` names the element in the error otherwise.
1636fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
1637    if lossy {
1638        return Ok(String::from_utf8_lossy(bytes).into_owned());
1639    }
1640    match charset {
1641        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
1642        0 => match bytes.iter().position(|&b| b >= 0x80) {
1643            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
1644            Some(at) => Err(Hdf5Error::InvalidState(format!(
1645                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
1646                bytes[at]
1647            ))),
1648        },
1649        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
1650            Hdf5Error::InvalidState(format!(
1651                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
1652            ))
1653        }),
1654        other => Err(Hdf5Error::InvalidState(format!(
1655            "string {index} uses character set {other}, which the format reserves"
1656        ))),
1657    }
1658}
1659
1660/// A dataset's storage layout class (read mode only) — `H5Pget_layout`'s
1661/// four values.
1662///
1663/// Distinct from [`ChunkIndex`], which names the structure a `Chunked`
1664/// layout's index uses; this only says which of the four storage classes
1665/// the dataset was created with.
1666#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1667pub enum StorageLayout {
1668    /// Raw data stored inline in the object header.
1669    Compact,
1670    /// Raw data in a single contiguous block — or, when the dataset also
1671    /// carries an external file list, in one or more blocks of an outside
1672    /// file instead ([`H5Dataset::external_files`]); the layout itself
1673    /// still reports `Contiguous` either way.
1674    Contiguous,
1675    /// Raw data split into fixed-size chunks, each independently
1676    /// allocated. [`H5Dataset::chunk_dims`] gives the chunk shape,
1677    /// [`H5Dataset::chunk_index`] the index structure.
1678    Chunked,
1679    /// No raw data of its own: every element comes from another dataset,
1680    /// possibly in another file ([`H5Dataset::virtual_mappings`]).
1681    Virtual,
1682}
1683
1684/// The chunk index structure a chunked dataset uses on disk (read mode
1685/// only) — which of libhdf5's chunk-lookup structures the layout message
1686/// names.
1687///
1688/// `BtreeV1` belongs to the version-3 chunked layout message (the only
1689/// index a file whose superblock predates version 2 can carry); the other
1690/// five are what a version-4 message's index-type byte selects, per
1691/// `H5D__layout_set_latest_indexing` (H5Dlayout.c).
1692#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1693pub enum ChunkIndex {
1694    /// Version-1 B-tree — the classic chunk index, and the only one a
1695    /// version-3 chunked layout message can carry.
1696    BtreeV1,
1697    /// Version-2 B-tree — two or more unlimited dimensions.
1698    BtreeV2,
1699    /// A single index entry for a dataset whose one chunk covers the whole
1700    /// dataspace (`dims == max_dims == chunk_dims`).
1701    SingleChunk,
1702    /// No index structure at all: chunk addresses are computed
1703    /// arithmetically over a contiguous run (no filter, early allocation).
1704    Implicit,
1705    /// Fixed-size array — a fixed shape needing per-chunk bookkeeping.
1706    FixedArray,
1707    /// Extensible array — exactly one unlimited dimension.
1708    ExtensibleArray,
1709}
1710
1711/// A dataset's fill-value state (read mode only) — `H5Pfill_value_defined`'s
1712/// tri-state (`H5D_fill_value_t`).
1713#[derive(Debug, Clone, PartialEq, Eq)]
1714pub enum FillValue {
1715    /// No fill value has ever been set: unwritten elements read back
1716    /// zero-filled, and no fill-value message named an explicit value.
1717    Default,
1718    /// The fill value was explicitly disabled: unallocated storage is never
1719    /// fill-initialized.
1720    Undefined,
1721    /// An explicit fill value, one element wide.
1722    UserDefined(Vec<u8>),
1723}
1724
1725/// When a dataset's fill value is written into allocated storage —
1726/// `H5Pset_fill_time`/`H5Pget_fill_time`'s `H5D_fill_time_t`.
1727///
1728/// Distinct from [`FillValue`], which says *what* the fill value is; this
1729/// says *when* it is written. The two agree everywhere except a dataset with
1730/// no fill value of its own: there, `Alloc` writes the default fill (zeros)
1731/// at allocation and `IfSet` writes nothing into space that already reads as
1732/// zeros — indistinguishable on disk in the value itself, only in this byte.
1733#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1734pub enum FillTime {
1735    /// Fill at allocation regardless of whether a fill value was ever set.
1736    Alloc,
1737    /// Never write the fill value into allocated storage.
1738    Never,
1739    /// Fill at allocation only when a fill value was set — the default
1740    /// every dataset gets unless [`fill_time`](DatasetBuilder::fill_time)
1741    /// says otherwise.
1742    IfSet,
1743}
1744
1745impl FillTime {
1746    /// The on-disk `H5D_fill_time_t` byte this variant is — what the writer
1747    /// stores and the fill-value message's write-time field carries.
1748    fn wire_byte(self) -> u8 {
1749        match self {
1750            Self::Alloc => 0,
1751            Self::Never => 1,
1752            Self::IfSet => 2,
1753        }
1754    }
1755}
1756
1757/// When a dataset's raw-data storage is allocated —
1758/// `H5Pset_alloc_time`/`H5Pget_alloc_time`'s `H5D_alloc_time_t`, read back
1759/// from the same fill-value message [`FillTime`] is.
1760///
1761/// `H5P__set_layout` (H5Pdcpl.c) picks this from the dataset's storage
1762/// class (`H5D_ALLOC_TIME_DEFAULT` per layout — compact is `Early`,
1763/// chunked and virtual are `Incr`, contiguous is `Late`). The one override
1764/// this crate's builder offers is [`DatasetBuilder::early_allocation`],
1765/// which a chunked dataset reads back as `Early` where the writer took it
1766/// up: an implicit index, or a single unfiltered chunk.
1767/// [`H5Dataset::alloc_time`] reads back what the writer declared.
1768#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1769pub enum AllocTime {
1770    /// Space is allocated as soon as the dataset is created.
1771    Early,
1772    /// Space is allocated when data is first written.
1773    Late,
1774    /// Space is allocated incrementally, as chunks (or virtual source
1775    /// datasets) are written.
1776    Incr,
1777}
1778
1779/// Which mapped data an unlimited virtual dataset's extent covers —
1780/// libhdf5's `H5D_vds_view_t`, set with `H5Pset_virtual_view` and read back
1781/// with `H5Pget_virtual_view` (H5Pdapl.c:1067, :1102).
1782///
1783/// A *dataset access* property: it is never stored in the file, so it says
1784/// how *this* open reads a virtual dataset, not what its writer intended.
1785#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1786pub enum VirtualView {
1787    /// `H5D_VDS_LAST_AVAILABLE` — the extent reaches the end of the last
1788    /// mapped block that has a source, so a gap before it reads as the fill
1789    /// value. libhdf5's default (`H5D_ACS_VDS_VIEW_DEF`, H5Pdapl.c:62).
1790    #[default]
1791    LastAvailable,
1792    /// `H5D_VDS_FIRST_MISSING` — the extent stops where the first missing
1793    /// mapped block begins, so no unmapped block is inside it.
1794    ///
1795    /// Under this view libhdf5 ignores
1796    /// [`virtual_printf_gap`](DatasetAccess::virtual_printf_gap) entirely:
1797    /// `H5D__virtual_init` reads the gap property only for
1798    /// [`LastAvailable`](Self::LastAvailable) and forces it to 0 otherwise
1799    /// (H5Dvirtual.c:2182-2188).
1800    FirstMissing,
1801}
1802
1803/// The dataset *access* properties this crate models — libhdf5's
1804/// `H5P_DATASET_ACCESS` property list, as much of it as affects reading.
1805///
1806/// [`virtual_view`](Self::virtual_view) and
1807/// [`virtual_printf_gap`](Self::virtual_printf_gap) govern how a virtual
1808/// dataset's extent is resolved when it is opened
1809/// (`H5D__virtual_set_extent_unlim`, H5Dvirtual.c:1386);
1810/// [`virtual_prefix`](Self::virtual_prefix) and
1811/// [`efile_prefix`](Self::efile_prefix) say where the *other files* a
1812/// dataset's data lives in are looked for. None of them is stored in the
1813/// file: opening a dataset without naming them reads it exactly as
1814/// libhdf5's default dapl does.
1815///
1816/// Pass one to [`H5File::dataset_with`](crate::H5File::dataset_with).
1817///
1818/// ```no_run
1819/// use rust_hdf5::{DatasetAccess, H5File, VirtualView};
1820///
1821/// let file = H5File::open("vds.h5").unwrap();
1822/// let access = DatasetAccess::new()
1823///     .virtual_view(VirtualView::LastAvailable)
1824///     .virtual_printf_gap(2);
1825/// let ds = file.dataset_with("vds", access).unwrap();
1826/// ```
1827#[derive(Debug, Clone, PartialEq, Eq, Default)]
1828pub struct DatasetAccess {
1829    view: VirtualView,
1830    printf_gap: u64,
1831    virtual_prefix: Option<String>,
1832    efile_prefix: Option<String>,
1833}
1834
1835impl DatasetAccess {
1836    /// A property list holding libhdf5's defaults —
1837    /// [`VirtualView::LastAvailable`] and a printf gap of 0, the values
1838    /// `H5D_ACS_VDS_VIEW_DEF` and `H5D_ACS_VDS_PRINTF_GAP_DEF` register
1839    /// (H5Pdapl.c:62, :67).
1840    pub fn new() -> Self {
1841        Self::default()
1842    }
1843
1844    /// `H5Pset_virtual_view` (H5Pdapl.c:1067). The two legal values are the
1845    /// two [`VirtualView`] variants, so the "not a valid bounds option"
1846    /// argument check that call makes has nothing to reject here.
1847    pub fn virtual_view(mut self, view: VirtualView) -> Self {
1848        self.view = view;
1849        self
1850    }
1851
1852    /// `H5Pset_virtual_printf_gap` (H5Pdapl.c:1207): how many consecutive
1853    /// missing printf-named source datasets the extent resolution looks past
1854    /// before it stops. 0 — the default — stops at the first one missing.
1855    ///
1856    /// `u64::MAX` is libhdf5's `HSIZE_UNDEF`, which that call rejects as "not
1857    /// a valid printf gap size"; here the rejection surfaces from the open
1858    /// that uses the property, since a builder method has no way to report
1859    /// it.
1860    pub fn virtual_printf_gap(mut self, gap: u64) -> Self {
1861        self.printf_gap = gap;
1862        self
1863    }
1864
1865    /// `H5Pget_virtual_view` (H5Pdapl.c:1102).
1866    pub fn view(&self) -> VirtualView {
1867        self.view
1868    }
1869
1870    /// `H5Pset_virtual_prefix` (H5Pdapl.c:1478): a directory a virtual
1871    /// dataset's *source file names* are looked for under, before the
1872    /// virtual file's own directory and after `HDF5_VDS_PREFIX`.
1873    ///
1874    /// It is the third step of `H5F_prefix_open_file`'s search order
1875    /// (H5Fint.c:938-950), and it is reached only when `HDF5_VDS_PREFIX` is
1876    /// unset or empty: `H5D__build_file_prefix` reads the environment first
1877    /// and falls back to this property (H5Dint.c:1077-1082), so an
1878    /// environment prefix shadows this one outright rather than being tried
1879    /// alongside it.
1880    ///
1881    /// A leading `${ORIGIN}` stands for the directory holding the virtual
1882    /// dataset's own file (H5Dint.c:1105-1113), and `"."` or `""` means "no
1883    /// prefix" (:1096-1100), both exactly as for the environment variable.
1884    ///
1885    /// Like the other two, this is a *dataset access* property that is never
1886    /// stored in the file, and the first open of a virtual dataset fixes it
1887    /// for every open that overlaps it.
1888    pub fn virtual_prefix(mut self, prefix: impl Into<String>) -> Self {
1889        self.virtual_prefix = Some(prefix.into());
1890        self
1891    }
1892
1893    /// `H5Pset_efile_prefix` (H5Pdapl.c:1392): a directory the *raw data
1894    /// files* of a dataset stored through an external file list are looked
1895    /// for under.
1896    ///
1897    /// This one takes no search at all, unlike the other two prefixes:
1898    /// `H5D__efl_read` joins the prefix to the stored name with
1899    /// `H5_combine_path` and opens exactly that one path (H5Defl.c:315-317).
1900    /// With no prefix in force the stored name is used as written, so a
1901    /// relative one resolves against the *process's current directory* and
1902    /// not against the directory holding the HDF5 file — measured under
1903    /// libhdf5 1.14.6 and 2.0.0: a raw data file next to the HDF5 file is
1904    /// not found, while the same name under the current directory is.
1905    ///
1906    /// It shares [`virtual_prefix`](Self::virtual_prefix)'s expansion rules,
1907    /// because both are built by `H5D__build_file_prefix`: `HDF5_EXTFILE_PREFIX`
1908    /// shadows this property outright rather than merely preceding it
1909    /// (H5Dint.c:1084-1090), a leading `${ORIGIN}` stands for the directory
1910    /// holding the HDF5 file (:1105-1113), and `"."` or `""` means no prefix
1911    /// (:1098-1102).
1912    ///
1913    /// # A second open must name the same one
1914    ///
1915    /// Where a mismatched [`virtual_prefix`](Self::virtual_prefix) is
1916    /// silently ignored by the second open, a mismatched external file prefix
1917    /// is an *error*: `H5D_open` compares the expanded prefix against the one
1918    /// the already-open dataset resolved under and refuses the open when they
1919    /// differ (H5Dint.c:1533-1545). Expanded, so two opens that differ only
1920    /// in a property the environment shadows still agree. Closing every
1921    /// handle releases the answer, and the next open sets its own.
1922    pub fn efile_prefix(mut self, prefix: impl Into<String>) -> Self {
1923        self.efile_prefix = Some(prefix.into());
1924        self
1925    }
1926
1927    /// `H5Pget_virtual_printf_gap` (H5Pdapl.c:1243) — the value set, not the
1928    /// one the extent resolution ends up using; see
1929    /// [`VirtualView::FirstMissing`].
1930    pub fn printf_gap(&self) -> u64 {
1931        self.printf_gap
1932    }
1933
1934    /// `H5Pget_virtual_prefix` (H5Pdapl.c:1510) — the property as set, before
1935    /// `HDF5_VDS_PREFIX` gets to shadow it and before `${ORIGIN}` is
1936    /// expanded. `None` is `H5D_ACS_VDS_PREFIX_DEF`, a null prefix
1937    /// (H5Pdapl.c:72).
1938    pub fn virtual_prefix_value(&self) -> Option<&str> {
1939        self.virtual_prefix.as_deref()
1940    }
1941
1942    /// `H5Pget_efile_prefix` (H5Pdapl.c:1422) — the property as set, before
1943    /// `HDF5_EXTFILE_PREFIX` gets to shadow it and before `${ORIGIN}` is
1944    /// expanded. `None` is `H5D_ACS_EFILE_PREFIX_DEF`, a null prefix
1945    /// (H5Pdapl.c:90).
1946    pub fn efile_prefix_value(&self) -> Option<&str> {
1947        self.efile_prefix.as_deref()
1948    }
1949
1950    /// The printf gap `H5D__virtual_set_extent_unlim` actually scans with:
1951    /// the property under [`VirtualView::LastAvailable`], and 0 under
1952    /// [`VirtualView::FirstMissing`], because `H5D__virtual_init` only reads
1953    /// the property in the first case (H5Dvirtual.c:2182-2188).
1954    ///
1955    /// The single owner of that rule — the resolution never reads
1956    /// [`printf_gap`](Self::printf_gap) directly.
1957    pub(crate) fn effective_printf_gap(&self) -> u64 {
1958        match self.view {
1959            VirtualView::LastAvailable => self.printf_gap,
1960            VirtualView::FirstMissing => 0,
1961        }
1962    }
1963
1964    /// Reject what `H5Pset_virtual_printf_gap` rejects, at the open that uses
1965    /// the property.
1966    pub(crate) fn validate(&self) -> Result<()> {
1967        if self.printf_gap == u64::MAX {
1968            return Err(Hdf5Error::InvalidState(
1969                "virtual_printf_gap(u64::MAX) is libhdf5's HSIZE_UNDEF, which \
1970                 H5Pset_virtual_printf_gap refuses as \"not a valid printf gap size\""
1971                    .into(),
1972            ));
1973        }
1974        Ok(())
1975    }
1976}
1977
1978/// Elements a selection of `counts` holds, refusing a product that overflows
1979/// `usize` rather than wrapping it into a small allocation.
1980fn element_count(counts: &[u64]) -> Result<usize> {
1981    counts
1982        .iter()
1983        .try_fold(1usize, |acc, &c| {
1984            usize::try_from(c).ok().and_then(|c| acc.checked_mul(c))
1985        })
1986        .ok_or_else(|| {
1987            Hdf5Error::InvalidState(format!("selection {counts:?} has more elements than usize"))
1988        })
1989}
1990
1991impl H5Dataset {
1992    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
1993    pub(crate) fn new_reader(
1994        file_inner: SharedInner,
1995        name: String,
1996        shape: Vec<usize>,
1997        element_size: usize,
1998        open: Option<crate::io::reader::DatasetOpenToken>,
1999    ) -> Self {
2000        Self {
2001            file_inner,
2002            info: DatasetInfo::Reader {
2003                name,
2004                shape,
2005                element_size,
2006            },
2007            _open: open,
2008        }
2009    }
2010
2011    /// Create a writer-mode dataset handle for an already-created dataset
2012    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
2013    ///
2014    /// Reconstructs the same handle `new_dataset().create()` returns, so the
2015    /// reopened dataset supports attribute writes and chunk appends.
2016    ///
2017    /// `is_null` is always `false` here: reopening an existing NULL-dataspace
2018    /// dataset for further writes is not a case this constructor's caller
2019    /// distinguishes (a NULL dataset has nothing to append or chunk-write in
2020    /// the first place).
2021    pub(crate) fn new_writer(
2022        file_inner: SharedInner,
2023        index: usize,
2024        parts: crate::io::writer::DatasetHandleParts,
2025    ) -> Self {
2026        Self {
2027            file_inner,
2028            info: DatasetInfo::Writer {
2029                index,
2030                shape: parts.shape,
2031                element_size: parts.element_size,
2032                chunk_index: parts.chunk_index,
2033                is_null: false,
2034            },
2035            _open: parts.open,
2036        }
2037    }
2038
2039    /// Return the dataset dimensions.
2040    pub fn shape(&self) -> Vec<usize> {
2041        match &self.info {
2042            DatasetInfo::Writer { shape, .. } => shape.clone(),
2043            DatasetInfo::Reader { shape, .. } => shape.clone(),
2044        }
2045    }
2046
2047    /// Return the number of dimensions (rank) of the dataset.
2048    pub fn ndims(&self) -> usize {
2049        match &self.info {
2050            DatasetInfo::Writer { shape, .. } => shape.len(),
2051            DatasetInfo::Reader { shape, .. } => shape.len(),
2052        }
2053    }
2054
2055    /// Return the total number of elements in the dataset.
2056    ///
2057    /// 0 for a NULL dataspace ([`is_null`](Self::is_null)) — unlike a scalar,
2058    /// whose `shape()` is the same empty `Vec` but which holds exactly one
2059    /// element, so `shape().iter().product()` cannot be used here.
2060    pub fn total_elements(&self) -> usize {
2061        if self.is_null() {
2062            return 0;
2063        }
2064        match &self.info {
2065            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
2066            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
2067        }
2068    }
2069
2070    /// Return the size of one element in bytes.
2071    pub fn element_size(&self) -> usize {
2072        match &self.info {
2073            DatasetInfo::Writer { element_size, .. } => *element_size,
2074            DatasetInfo::Reader { element_size, .. } => *element_size,
2075        }
2076    }
2077
2078    /// Return whether this dataset has the NULL dataspace: no elements at
2079    /// all, distinct from a scalar dataset (rank 0, exactly one element) —
2080    /// both report the same empty [`shape`](Self::shape). See
2081    /// [`DatasetBuilder::null`].
2082    pub fn is_null(&self) -> bool {
2083        match &self.info {
2084            DatasetInfo::Writer { is_null, .. } => *is_null,
2085            DatasetInfo::Reader { name, .. } => {
2086                let mut inner = borrow_inner_mut(&self.file_inner);
2087                match &mut *inner {
2088                    H5FileInner::Reader(reader) => reader
2089                        .dataset_info(name)
2090                        .map(|info| info.dataspace.is_null())
2091                        .unwrap_or(false),
2092                    _ => false,
2093                }
2094            }
2095        }
2096    }
2097
2098    /// Return the element datatype as parsed from the file (read mode only).
2099    ///
2100    /// Unlike [`element_size`](Self::element_size), which reports only the
2101    /// byte width, this exposes the full datatype: its class (integer vs
2102    /// floating-point vs string vs compound …), signedness, byte order and
2103    /// bit precision. Callers that must reconstruct the exact stored type —
2104    /// for example to map it to a NumPy / Arrow dtype — should use this
2105    /// instead of inferring a type from the byte width, which cannot
2106    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
2107    /// bytes).
2108    ///
2109    /// # Errors
2110    ///
2111    /// Returns an error if the file is in write mode, or if the dataset can
2112    /// no longer be found in the reader's metadata.
2113    ///
2114    /// ```no_run
2115    /// # use rust_hdf5::{H5File, DatatypeMessage};
2116    /// let file = H5File::open("data.h5").unwrap();
2117    /// let ds = file.dataset("image").unwrap();
2118    /// match ds.datatype().unwrap() {
2119    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
2120    ///         println!("integer: {} bytes, signed={}", size, signed);
2121    ///     }
2122    ///     DatatypeMessage::FloatingPoint { size, .. } => {
2123    ///         println!("float: {} bytes", size);
2124    ///     }
2125    ///     other => println!("other type: {other}"),
2126    /// }
2127    /// ```
2128    pub fn datatype(&self) -> Result<DatatypeMessage> {
2129        match &self.info {
2130            DatasetInfo::Reader { name, .. } => {
2131                let mut inner = borrow_inner_mut(&self.file_inner);
2132                match &mut *inner {
2133                    H5FileInner::Reader(reader) => reader
2134                        .dataset_info(name)
2135                        .map(|info| info.datatype.clone())
2136                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2137                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2138                }
2139            }
2140            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2141                "datatype() is only available in read mode".into(),
2142            )),
2143        }
2144    }
2145
2146    /// Return the chunk dimensions, if this is a chunked dataset.
2147    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
2148        match &self.info {
2149            DatasetInfo::Reader { name, .. } => {
2150                let mut inner = borrow_inner_mut(&self.file_inner);
2151                if let H5FileInner::Reader(reader) = &mut *inner {
2152                    if let Some(info) = reader.dataset_info(name) {
2153                        use crate::format::messages::data_layout::DataLayoutMessage;
2154                        let chunk_dims = match &info.layout {
2155                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
2156                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
2157                            _ => None,
2158                        };
2159                        if let Some(chunk_dims) = chunk_dims {
2160                            // Strip trailing element-size dimension
2161                            return Some(
2162                                chunk_dims[..chunk_dims.len() - 1]
2163                                    .iter()
2164                                    .map(|&d| d as usize)
2165                                    .collect(),
2166                            );
2167                        }
2168                    }
2169                }
2170                None
2171            }
2172            DatasetInfo::Writer { .. } => None,
2173        }
2174    }
2175
2176    /// Return whether this is a chunked dataset.
2177    pub fn is_chunked(&self) -> bool {
2178        match &self.info {
2179            DatasetInfo::Writer { chunk_index, .. } => chunk_index.is_some(),
2180            DatasetInfo::Reader { name, .. } => {
2181                let mut inner = borrow_inner_mut(&self.file_inner);
2182                match &mut *inner {
2183                    H5FileInner::Reader(reader) => {
2184                        if let Some(info) = reader.dataset_info(name) {
2185                            use crate::format::messages::data_layout::DataLayoutMessage;
2186                            matches!(
2187                                info.layout,
2188                                DataLayoutMessage::ChunkedV4 { .. }
2189                                    | DataLayoutMessage::ChunkedV3 { .. }
2190                            )
2191                        } else {
2192                            false
2193                        }
2194                    }
2195                    _ => false,
2196                }
2197            }
2198        }
2199    }
2200
2201    /// Return the dataset's storage layout class (read mode only).
2202    ///
2203    /// # Errors
2204    ///
2205    /// Returns an error if the file is in write mode, or if the dataset can
2206    /// no longer be found in the reader's metadata.
2207    pub fn storage_layout(&self) -> Result<StorageLayout> {
2208        match &self.info {
2209            DatasetInfo::Reader { name, .. } => {
2210                let mut inner = borrow_inner_mut(&self.file_inner);
2211                match &mut *inner {
2212                    H5FileInner::Reader(reader) => {
2213                        use crate::format::messages::data_layout::DataLayoutMessage;
2214                        reader
2215                            .dataset_info(name)
2216                            .map(|info| match &info.layout {
2217                                DataLayoutMessage::Compact { .. } => StorageLayout::Compact,
2218                                DataLayoutMessage::Contiguous { .. } => StorageLayout::Contiguous,
2219                                DataLayoutMessage::ChunkedV3 { .. }
2220                                | DataLayoutMessage::ChunkedV4 { .. } => StorageLayout::Chunked,
2221                                DataLayoutMessage::Virtual { .. } => StorageLayout::Virtual,
2222                            })
2223                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2224                    }
2225                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2226                }
2227            }
2228            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2229                "storage_layout() is only available in read mode".into(),
2230            )),
2231        }
2232    }
2233
2234    /// Return the chunk index structure this dataset's layout uses, or
2235    /// `None` for a dataset that is not chunked (read mode only).
2236    ///
2237    /// # Errors
2238    ///
2239    /// Returns an error if the file is in write mode, or if the dataset can
2240    /// no longer be found in the reader's metadata.
2241    pub fn chunk_index(&self) -> Result<Option<ChunkIndex>> {
2242        match &self.info {
2243            DatasetInfo::Reader { name, .. } => {
2244                let mut inner = borrow_inner_mut(&self.file_inner);
2245                match &mut *inner {
2246                    H5FileInner::Reader(reader) => {
2247                        use crate::format::messages::data_layout::{
2248                            ChunkIndexType, DataLayoutMessage,
2249                        };
2250                        reader
2251                            .dataset_info(name)
2252                            .map(|info| match &info.layout {
2253                                DataLayoutMessage::ChunkedV3 { .. } => Some(ChunkIndex::BtreeV1),
2254                                DataLayoutMessage::ChunkedV4 { index_type, .. } => {
2255                                    Some(match index_type {
2256                                        ChunkIndexType::SingleChunk => ChunkIndex::SingleChunk,
2257                                        ChunkIndexType::Implicit => ChunkIndex::Implicit,
2258                                        ChunkIndexType::FixedArray => ChunkIndex::FixedArray,
2259                                        ChunkIndexType::ExtensibleArray => {
2260                                            ChunkIndex::ExtensibleArray
2261                                        }
2262                                        ChunkIndexType::BTreeV2 => ChunkIndex::BtreeV2,
2263                                    })
2264                                }
2265                                _ => None,
2266                            })
2267                            .ok_or_else(|| Hdf5Error::NotFound(name.clone()))
2268                    }
2269                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2270                }
2271            }
2272            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2273                "chunk_index() is only available in read mode".into(),
2274            )),
2275        }
2276    }
2277
2278    /// Return this dataset's filter pipeline (read mode only), in
2279    /// application order. Empty when the dataset has no filter pipeline
2280    /// message at all — an unfiltered dataset, not an error.
2281    ///
2282    /// # Errors
2283    ///
2284    /// Returns an error if the file is in write mode, or if the dataset can
2285    /// no longer be found in the reader's metadata.
2286    pub fn filters(&self) -> Result<Vec<Filter>> {
2287        match &self.info {
2288            DatasetInfo::Reader { name, .. } => {
2289                let mut inner = borrow_inner_mut(&self.file_inner);
2290                match &mut *inner {
2291                    H5FileInner::Reader(reader) => reader
2292                        .dataset_info(name)
2293                        .map(|info| {
2294                            info.filter_pipeline
2295                                .as_ref()
2296                                .map(|fp| fp.filters.clone())
2297                                .unwrap_or_default()
2298                        })
2299                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2300                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2301                }
2302            }
2303            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2304                "filters() is only available in read mode".into(),
2305            )),
2306        }
2307    }
2308
2309    /// Return this dataset's fill-value state (read mode only).
2310    ///
2311    /// # Errors
2312    ///
2313    /// Returns an error if the file is in write mode, or if the dataset can
2314    /// no longer be found in the reader's metadata.
2315    pub fn fill_value(&self) -> Result<FillValue> {
2316        match &self.info {
2317            DatasetInfo::Reader { name, .. } => {
2318                let mut inner = borrow_inner_mut(&self.file_inner);
2319                match &mut *inner {
2320                    H5FileInner::Reader(reader) => reader
2321                        .dataset_info(name)
2322                        .map(|info| match info.fill_defined {
2323                            0 => FillValue::Undefined,
2324                            2 => {
2325                                FillValue::UserDefined(info.fill_value.clone().unwrap_or_default())
2326                            }
2327                            _ => FillValue::Default,
2328                        })
2329                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2330                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2331                }
2332            }
2333            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2334                "fill_value() is only available in read mode".into(),
2335            )),
2336        }
2337    }
2338
2339    /// Return when this dataset's fill value is written into allocated
2340    /// storage (read mode only) — `H5Pget_fill_time`.
2341    ///
2342    /// # Errors
2343    ///
2344    /// Returns an error if the file is in write mode, or if the dataset can
2345    /// no longer be found in the reader's metadata.
2346    pub fn fill_time(&self) -> Result<FillTime> {
2347        match &self.info {
2348            DatasetInfo::Reader { name, .. } => {
2349                let mut inner = borrow_inner_mut(&self.file_inner);
2350                match &mut *inner {
2351                    H5FileInner::Reader(reader) => reader
2352                        .dataset_info(name)
2353                        .map(|info| match info.fill_write_time {
2354                            0 => FillTime::Alloc,
2355                            1 => FillTime::Never,
2356                            _ => FillTime::IfSet,
2357                        })
2358                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2359                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2360                }
2361            }
2362            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2363                "fill_time() is only available in read mode".into(),
2364            )),
2365        }
2366    }
2367
2368    /// Return when this dataset's raw-data storage is allocated (read mode
2369    /// only) — `H5Pget_alloc_time`.
2370    ///
2371    /// # Errors
2372    ///
2373    /// Returns an error if the file is in write mode, or if the dataset can
2374    /// no longer be found in the reader's metadata.
2375    pub fn alloc_time(&self) -> Result<AllocTime> {
2376        match &self.info {
2377            DatasetInfo::Reader { name, .. } => {
2378                let mut inner = borrow_inner_mut(&self.file_inner);
2379                match &mut *inner {
2380                    H5FileInner::Reader(reader) => reader
2381                        .dataset_info(name)
2382                        .map(|info| match info.alloc_time {
2383                            1 => AllocTime::Early,
2384                            3 => AllocTime::Incr,
2385                            _ => AllocTime::Late,
2386                        })
2387                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2388                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2389                }
2390            }
2391            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2392                "alloc_time() is only available in read mode".into(),
2393            )),
2394        }
2395    }
2396
2397    /// Return this dataset's external raw-data file segments (read mode
2398    /// only), in the order the dataset's logical byte range concatenates
2399    /// them. Empty for a dataset whose data lives in this file.
2400    ///
2401    /// # Errors
2402    ///
2403    /// Returns an error if the file is in write mode, or if the dataset can
2404    /// no longer be found in the reader's metadata.
2405    pub fn external_files(&self) -> Result<Vec<ExternalFileSegment>> {
2406        match &self.info {
2407            DatasetInfo::Reader { name, .. } => {
2408                let mut inner = borrow_inner_mut(&self.file_inner);
2409                match &mut *inner {
2410                    H5FileInner::Reader(reader) => reader
2411                        .dataset_info(name)
2412                        .map(|info| info.external_files.clone())
2413                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2414                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2415                }
2416            }
2417            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2418                "external_files() is only available in read mode".into(),
2419            )),
2420        }
2421    }
2422
2423    /// Return this dataset's maximum dimension sizes (read mode only):
2424    /// `None` in a dimension marks that axis unlimited. A dataset with no
2425    /// maximum-dimensions message reports its current shape (max == current
2426    /// — the upstream convention for a fixed-extent dataset).
2427    ///
2428    /// # Errors
2429    ///
2430    /// Returns an error if the file is in write mode, or if the dataset can
2431    /// no longer be found in the reader's metadata.
2432    pub fn max_shape(&self) -> Result<Vec<Option<usize>>> {
2433        match &self.info {
2434            DatasetInfo::Reader { name, .. } => {
2435                let mut inner = borrow_inner_mut(&self.file_inner);
2436                match &mut *inner {
2437                    H5FileInner::Reader(reader) => reader
2438                        .dataset_info(name)
2439                        .map(|info| match &info.dataspace.max_dims {
2440                            Some(max_dims) => max_dims
2441                                .iter()
2442                                .map(|&d| (d != u64::MAX).then_some(d as usize))
2443                                .collect(),
2444                            None => info
2445                                .dataspace
2446                                .dims
2447                                .iter()
2448                                .map(|&d| Some(d as usize))
2449                                .collect(),
2450                        })
2451                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2452                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2453                }
2454            }
2455            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2456                "max_shape() is only available in read mode".into(),
2457            )),
2458        }
2459    }
2460
2461    /// Return this dataset's virtual-dataset source/virtual mappings (read
2462    /// mode only), in on-disk order. Empty for any dataset whose layout is
2463    /// not virtual, and for a virtual dataset that has no mappings yet.
2464    ///
2465    /// # Errors
2466    ///
2467    /// Returns an error if the file is in write mode, or if the dataset can
2468    /// no longer be found in the reader's metadata.
2469    pub fn virtual_mappings(&self) -> Result<Vec<VirtualMapping>> {
2470        match &self.info {
2471            DatasetInfo::Reader { name, .. } => {
2472                let mut inner = borrow_inner_mut(&self.file_inner);
2473                match &mut *inner {
2474                    H5FileInner::Reader(reader) => reader
2475                        .dataset_info(name)
2476                        .map(|info| {
2477                            info.virtual_mappings
2478                                .as_ref()
2479                                .map(|vml| vml.mappings.clone())
2480                                .unwrap_or_default()
2481                        })
2482                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
2483                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2484                }
2485            }
2486            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2487                "virtual_mappings() is only available in read mode".into(),
2488            )),
2489        }
2490    }
2491
2492    /// Return the names of all attributes on this dataset (read mode only).
2493    pub fn attr_names(&self) -> Result<Vec<String>> {
2494        match &self.info {
2495            DatasetInfo::Reader { name, .. } => {
2496                let mut inner = borrow_inner_mut(&self.file_inner);
2497                match &mut *inner {
2498                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
2499                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2500                }
2501            }
2502            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2503                "attr_names not available in write mode".into(),
2504            )),
2505        }
2506    }
2507
2508    /// Why the attribute `attr_name` on this dataset cannot be read, or `None`
2509    /// when it can be.
2510    ///
2511    /// An attribute whose message this crate cannot decode is still listed by
2512    /// [`attr_names`](Self::attr_names) — the object header carries it — and
2513    /// this says what stands in the way. Opening it through
2514    /// [`attr`](Self::attr) fails with the same text.
2515    pub fn attr_unreadable_reason(&self, attr_name: &str) -> Result<Option<String>> {
2516        match &self.info {
2517            DatasetInfo::Reader { name, .. } => {
2518                let mut inner = borrow_inner_mut(&self.file_inner);
2519                match &mut *inner {
2520                    H5FileInner::Reader(reader) => Ok(reader
2521                        .dataset_attr_unreadable_reason(name, attr_name)
2522                        .map(str::to_string)),
2523                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2524                }
2525            }
2526            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2527                "attr_unreadable_reason not available in write mode".into(),
2528            )),
2529        }
2530    }
2531
2532    /// Why this dataset's attribute *set* cannot be listed, or `None` when it
2533    /// can be.
2534    ///
2535    /// The object-scope counterpart of
2536    /// [`attr_unreadable_reason`](Self::attr_unreadable_reason). A dense
2537    /// attribute set is indexed by name hash, so a heap or index that will not
2538    /// read yields no names to hang a per-attribute reason on;
2539    /// [`attr_names`](Self::attr_names) then returns the failure rather than a
2540    /// short list, and this reports it without an attribute name.
2541    pub fn attrs_unreadable_reason(&self) -> Result<Option<String>> {
2542        match &self.info {
2543            DatasetInfo::Reader { name, .. } => {
2544                let mut inner = borrow_inner_mut(&self.file_inner);
2545                match &mut *inner {
2546                    H5FileInner::Reader(reader) => Ok(reader
2547                        .dataset_attrs_unreadable_reason(name)
2548                        .map(str::to_string)),
2549                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2550                }
2551            }
2552            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2553                "attrs_unreadable_reason not available in write mode".into(),
2554            )),
2555        }
2556    }
2557
2558    /// This dataset's own compact-vs-dense attribute storage — the
2559    /// equivalent of `h5py.h5o.get_info(did.id).meta_size.attr.index_size`
2560    /// being nonzero (read mode only).
2561    pub fn attr_storage(&self) -> Result<AttributeStorage> {
2562        match &self.info {
2563            DatasetInfo::Reader { name, .. } => {
2564                let mut inner = borrow_inner_mut(&self.file_inner);
2565                match &mut *inner {
2566                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_storage(name)?),
2567                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2568                }
2569            }
2570            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2571                "attr_storage not available in write mode".into(),
2572            )),
2573        }
2574    }
2575
2576    /// This dataset's own object-header attribute count — the equivalent of
2577    /// `h5py.h5o.get_info(did.id).num_attrs` (read mode only).
2578    pub fn header_attr_count(&self) -> Result<u64> {
2579        match &self.info {
2580            DatasetInfo::Reader { name, .. } => {
2581                let mut inner = borrow_inner_mut(&self.file_inner);
2582                match &mut *inner {
2583                    H5FileInner::Reader(reader) => Ok(reader.dataset_header_attr_count(name)?),
2584                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2585                }
2586            }
2587            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2588                "header_attr_count not available in write mode".into(),
2589            )),
2590        }
2591    }
2592
2593    /// Open an attribute by name (read mode only).
2594    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
2595        match &self.info {
2596            DatasetInfo::Reader { name, .. } => {
2597                let mut inner = borrow_inner_mut(&self.file_inner);
2598                match &mut *inner {
2599                    H5FileInner::Reader(reader) => {
2600                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
2601                        Ok(crate::attribute::H5Attribute::new_reader(
2602                            clone_inner(&self.file_inner),
2603                            attr_msg,
2604                        ))
2605                    }
2606                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2607                }
2608            }
2609            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2610                "attr() not available in write mode".into(),
2611            )),
2612        }
2613    }
2614
2615    /// Start building a new attribute on this dataset.
2616    ///
2617    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
2618    /// and `.create("name")` to finalize.
2619    ///
2620    /// # Example
2621    ///
2622    /// ```no_run
2623    /// # use rust_hdf5::H5File;
2624    /// # use rust_hdf5::types::VarLenUnicode;
2625    /// let file = H5File::create("attr.h5").unwrap();
2626    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
2627    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
2628    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
2629    /// ```
2630    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
2631        let ds_index = match &self.info {
2632            DatasetInfo::Writer { index, .. } => *index,
2633            DatasetInfo::Reader { .. } => {
2634                // Reader mode: we'll return a builder that will error on create.
2635                // Using usize::MAX as sentinel.
2636                usize::MAX
2637            }
2638        };
2639        AttrBuilder::new(&self.file_inner, ds_index)
2640    }
2641
2642    /// Write a typed slice holding the dataset's whole image.
2643    ///
2644    /// The slice length must match the total number of elements declared by
2645    /// the dataset shape. The data is reinterpreted as raw bytes and written
2646    /// to the file: to the contiguous data block, or — for a chunked dataset —
2647    /// scattered across its chunk grid, through the filter pipeline if one is
2648    /// set. To write only part of a dataset, use
2649    /// [`write_slice`](Self::write_slice).
2650    ///
2651    /// # Errors
2652    ///
2653    /// Returns an error if:
2654    /// - The file is in read mode.
2655    /// - The data length does not match the declared shape.
2656    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
2657        match &self.info {
2658            DatasetInfo::Writer {
2659                index,
2660                shape,
2661                element_size,
2662                chunk_index,
2663                is_null,
2664            } => {
2665                if *is_null {
2666                    return Err(Hdf5Error::InvalidState(
2667                        "cannot write to a NULL dataspace dataset".into(),
2668                    ));
2669                }
2670                let total_elements: usize = shape.iter().product();
2671                if data.len() != total_elements {
2672                    return Err(Hdf5Error::InvalidState(format!(
2673                        "data length {} does not match dataset size {}",
2674                        data.len(),
2675                        total_elements,
2676                    )));
2677                }
2678
2679                // Verify element size matches
2680                if T::element_size() != *element_size {
2681                    return Err(Hdf5Error::TypeMismatch(format!(
2682                        "write type has element size {} but dataset expects {}",
2683                        T::element_size(),
2684                        element_size,
2685                    )));
2686                }
2687
2688                // Safety: T: Copy + 'static (numeric primitive) with well-defined
2689                // byte representation. The resulting slice borrows `data` and
2690                // lives only as long as this block.
2691                let byte_len = data.len() * T::element_size();
2692                let host =
2693                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
2694                let datatype = {
2695                    let inner = borrow_inner(&self.file_inner);
2696                    match &*inner {
2697                        H5FileInner::Writer(writer) => writer.dataset_datatype(*index),
2698                        _ => {
2699                            return Err(Hdf5Error::InvalidState(
2700                                "file is no longer in write mode".into(),
2701                            ))
2702                        }
2703                    }
2704                };
2705                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
2706
2707                if let Some(kind) = *chunk_index {
2708                    // A chunked dataset has no contiguous data block; scatter
2709                    // the full row-major image into its chunk grid and write
2710                    // each chunk through the dataset's filter pipeline.
2711                    return self.write_full_image_chunked(*index, kind, &stored, *element_size);
2712                }
2713
2714                let inner = borrow_inner(&self.file_inner);
2715                match &*inner {
2716                    H5FileInner::Writer(writer) => {
2717                        writer.write_dataset_raw(*index, &stored)?;
2718                        Ok(())
2719                    }
2720                    _ => Err(Hdf5Error::InvalidState(
2721                        "file is no longer in write mode".into(),
2722                    )),
2723                }
2724            }
2725            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2726                "cannot write to a dataset opened in read mode".into(),
2727            )),
2728        }
2729    }
2730
2731    /// Write the raw byte image of the whole dataset directly.
2732    ///
2733    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
2734    /// data block, or a chunk grid the image is scattered across.
2735    ///
2736    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
2737    /// `H5Type` carrier, so it works for element types that have no matching
2738    /// Rust primitive — in particular a runtime
2739    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
2740    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
2741    /// `product(shape) * element_size`, where `element_size` is taken from the
2742    /// dataset's on-disk datatype.
2743    ///
2744    /// ```no_run
2745    /// # use rust_hdf5::H5File;
2746    /// # use rust_hdf5::types::{CompoundType, H5Type};
2747    /// let file = H5File::create("c.h5").unwrap();
2748    /// let ct = CompoundType {
2749    ///     members: vec![
2750    ///         ("id".to_string(), i32::hdf5_type(), 0),
2751    ///         ("val".to_string(), f64::hdf5_type(), 4),
2752    ///     ],
2753    ///     total_size: 12,
2754    /// };
2755    /// let ds = file
2756    ///     .new_dataset::<u8>()
2757    ///     .datatype(ct.to_datatype())
2758    ///     .shape(&[2])
2759    ///     .create("records")
2760    ///     .unwrap();
2761    /// let mut bytes = Vec::new();
2762    /// bytes.extend_from_slice(&1i32.to_le_bytes());
2763    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
2764    /// bytes.extend_from_slice(&2i32.to_le_bytes());
2765    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
2766    /// ds.write_raw_bytes(&bytes).unwrap();
2767    /// ```
2768    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
2769        match &self.info {
2770            DatasetInfo::Writer {
2771                index,
2772                shape,
2773                element_size,
2774                chunk_index,
2775                is_null,
2776            } => {
2777                if *is_null {
2778                    return Err(Hdf5Error::InvalidState(
2779                        "cannot write to a NULL dataspace dataset".into(),
2780                    ));
2781                }
2782                let expected: usize = shape.iter().product::<usize>() * *element_size;
2783                if bytes.len() != expected {
2784                    return Err(Hdf5Error::InvalidState(format!(
2785                        "raw byte length {} does not match dataset size {} \
2786                         (product(shape) * element_size {})",
2787                        bytes.len(),
2788                        expected,
2789                        element_size,
2790                    )));
2791                }
2792                if let Some(kind) = *chunk_index {
2793                    // Scatter the full row-major image into the chunk grid
2794                    // (same path as write_raw, carrier-agnostic bytes).
2795                    return self.write_full_image_chunked(*index, kind, bytes, *element_size);
2796                }
2797                let inner = borrow_inner(&self.file_inner);
2798                match &*inner {
2799                    H5FileInner::Writer(writer) => {
2800                        writer.write_dataset_raw(*index, bytes)?;
2801                        Ok(())
2802                    }
2803                    _ => Err(Hdf5Error::InvalidState(
2804                        "file is no longer in write mode".into(),
2805                    )),
2806                }
2807            }
2808            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
2809                "cannot write to a dataset opened in read mode".into(),
2810            )),
2811        }
2812    }
2813
2814    /// Scatter a full row-major dataset image into its chunk grid, writing
2815    /// every chunk through the dataset's filter pipeline.
2816    ///
2817    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
2818    /// — it is how [`write_raw`](Self::write_raw) and
2819    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
2820    /// (including the single auto-chunk created when a filter is set without
2821    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
2822    /// chunk footprint, exactly as libhdf5 stores them.
2823    fn write_full_image_chunked(
2824        &self,
2825        index: usize,
2826        kind: ChunkIndexKind,
2827        bytes: &[u8],
2828        element_size: usize,
2829    ) -> Result<()> {
2830        let inner = borrow_inner(&self.file_inner);
2831        let writer = match &*inner {
2832            H5FileInner::Writer(w) => w,
2833            _ => {
2834                return Err(Hdf5Error::InvalidState(
2835                    "file is no longer in write mode".into(),
2836                ))
2837            }
2838        };
2839        // Whole-operation guard: the flush, the grid snapshot and the chunk
2840        // writes below must not interleave with a concurrent same-dataset
2841        // operation.
2842        let cell = writer.ds(index);
2843        let _op = cell.op.lock();
2844        // A buffered append tail would flush over the image at close; hand
2845        // it to the chunks first, the image below overwrites everything.
2846        writer.flush_append_buffer(index)?;
2847        let chunk_dims = writer
2848            .dataset_chunk_dims(index)
2849            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
2850            .to_vec();
2851        let dims = writer.dataset_dims(index).to_vec();
2852        let rank = dims.len();
2853
2854        // Chunk grid: number of chunks along each dimension (row-major).
2855        let mut grid = vec![0u64; rank];
2856        for d in 0..rank {
2857            grid[d] = if chunk_dims[d] > 0 {
2858                dims[d].div_ceil(chunk_dims[d])
2859            } else {
2860                0
2861            };
2862        }
2863        let total_chunks: u64 = grid.iter().product();
2864
2865        // Decode the iteration counter into row-major coordinates over the
2866        // *current* image's chunk grid. This is only an odometer over the
2867        // chunks the image spans — the slot a chunk is recorded under comes
2868        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
2869        // extent decides.
2870        let coords_of = |linear: u64| -> Vec<u64> {
2871            let mut rem = linear;
2872            let mut coords = vec![0u64; rank];
2873            for d in (0..rank).rev() {
2874                coords[d] = rem % grid[d];
2875                rem /= grid[d];
2876            }
2877            coords
2878        };
2879
2880        // The batch entry points exist for one reason: to run the filter
2881        // pipeline over a window of chunks in parallel. An unfiltered dataset
2882        // has no pipeline to run, so it takes the plain per-chunk owner
2883        // whatever its index is, and only a filtered extensible or fixed array
2884        // — the two indexes with a batch entry point — takes the window below.
2885        let batched = writer.dataset_is_filtered(index)
2886            && matches!(
2887                kind,
2888                ChunkIndexKind::ExtensibleArray | ChunkIndexKind::FixedArray
2889            );
2890        if !batched {
2891            // One staging buffer for the whole image, reused chunk after
2892            // chunk: a chunk that already sits as one complete run of `bytes`
2893            // needs no staging at all and goes to the file straight out of the
2894            // caller's slice, so only an n-D interleave or a short edge pays
2895            // for a gather.
2896            let mut staging = Vec::new();
2897            for linear in 0..total_chunks {
2898                let coords = coords_of(linear);
2899                let chunk =
2900                    match Self::contiguous_chunk_span(&dims, &chunk_dims, &coords, element_size) {
2901                        Some(span) => &bytes[span],
2902                        None => {
2903                            Self::gather_chunk_into(
2904                                &mut staging,
2905                                bytes,
2906                                &dims,
2907                                &chunk_dims,
2908                                &coords,
2909                                element_size,
2910                            );
2911                            &staging[..]
2912                        }
2913                    };
2914                writer.write_chunk_at_coords(index, &coords, chunk)?;
2915            }
2916        } else {
2917            // Hand the pipeline a window of chunks so it compresses them in
2918            // parallel (with the `parallel` feature). A fixed-size window
2919            // bounds peak memory instead of materializing every chunk at once;
2920            // 256 keeps every rayon worker fed while capping the transient
2921            // buffers to window * chunk bytes. The compressors read the window
2922            // concurrently, so a gathered chunk here cannot share one reused
2923            // buffer the way the sequential path above does — but a chunk that
2924            // is already a complete run of `bytes` is borrowed, not copied.
2925            // The two indexes differ only in how a chunk is addressed: EA by
2926            // its linear grid index, FA by grid coordinates.
2927            const BATCH_WINDOW: u64 = 256;
2928            let mut start = 0u64;
2929            while start < total_chunks {
2930                let end = (start + BATCH_WINDOW).min(total_chunks);
2931                let items: Vec<(Vec<u64>, Cow<'_, [u8]>)> = (start..end)
2932                    .map(|counter| {
2933                        let coords = coords_of(counter);
2934                        let data = match Self::contiguous_chunk_span(
2935                            &dims,
2936                            &chunk_dims,
2937                            &coords,
2938                            element_size,
2939                        ) {
2940                            Some(span) => Cow::Borrowed(&bytes[span]),
2941                            None => {
2942                                let mut buf = Vec::new();
2943                                Self::gather_chunk_into(
2944                                    &mut buf,
2945                                    bytes,
2946                                    &dims,
2947                                    &chunk_dims,
2948                                    &coords,
2949                                    element_size,
2950                                );
2951                                Cow::Owned(buf)
2952                            }
2953                        };
2954                        (coords, data)
2955                    })
2956                    .collect();
2957                if kind == ChunkIndexKind::FixedArray {
2958                    let pairs: Vec<(&[u64], &[u8])> = items
2959                        .iter()
2960                        .map(|(c, d)| (c.as_slice(), d.as_ref()))
2961                        .collect();
2962                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
2963                } else {
2964                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
2965                    for (c, d) in &items {
2966                        pairs.push((writer.chunk_slot(index, c)?, d.as_ref()));
2967                    }
2968                    writer.write_chunks_batch_inner(index, &pairs)?;
2969                }
2970                start = end;
2971            }
2972        }
2973        Ok(())
2974    }
2975
2976    /// The byte range one chunk occupies in a row-major full-dataset image,
2977    /// for a chunk that needs no gather at all: its elements are one
2978    /// contiguous run of `source` *and* they fill the chunk shape exactly, so
2979    /// the bytes that go to the file are already sitting in the caller's
2980    /// buffer.
2981    ///
2982    /// Both halves hold when every dimension after the first spans the whole
2983    /// dataset (`chunk_dims[d] == dims[d]`, leaving nothing interleaved and no
2984    /// padding along those axes) and the chunk does not hang off the far edge
2985    /// of the first — which is every full chunk of a 1-D dataset. `None` means
2986    /// the chunk has to be gathered.
2987    fn contiguous_chunk_span(
2988        dims: &[u64],
2989        chunk_dims: &[u64],
2990        coords: &[u64],
2991        element_size: usize,
2992    ) -> Option<std::ops::Range<usize>> {
2993        let rank = dims.len();
2994        if rank == 0 || chunk_dims[1..] != dims[1..] {
2995            return None;
2996        }
2997        if (coords[0] + 1) * chunk_dims[0] > dims[0] {
2998            return None;
2999        }
3000        let plane: u64 = dims[1..].iter().product::<u64>() * element_size as u64;
3001        let start = usize::try_from(coords[0] * chunk_dims[0] * plane).ok()?;
3002        let len = usize::try_from(chunk_dims[0] * plane).ok()?;
3003        Some(start..start.checked_add(len)?)
3004    }
3005
3006    /// Gather one chunk's bytes from a row-major full-dataset image into
3007    /// `out`, replacing whatever it held.
3008    ///
3009    /// `coords` are the chunk's grid coordinates. `out` is left exactly
3010    /// `product(chunk_dims) * element_size` bytes long, holding the chunk's
3011    /// elements and zero where the chunk extends past the dataset edge — so a
3012    /// caller may hand the same buffer to one chunk after another.
3013    fn gather_chunk_into(
3014        out: &mut Vec<u8>,
3015        source: &[u8],
3016        dims: &[u64],
3017        chunk_dims: &[u64],
3018        coords: &[u64],
3019        element_size: usize,
3020    ) {
3021        let rank = dims.len();
3022        let chunk_elems: u64 = chunk_dims.iter().product();
3023        let chunk_bytes = chunk_elems as usize * element_size;
3024        if rank == 0 {
3025            // Scalar dataset: a single element, no chunking dimension.
3026            out.clear();
3027            out.resize(chunk_bytes, 0);
3028            if source.len() >= element_size {
3029                out[..element_size].copy_from_slice(&source[..element_size]);
3030            }
3031            return;
3032        }
3033
3034        // Actual extent of this chunk along each dimension (edge chunks are
3035        // smaller than the nominal chunk shape).
3036        let mut extent = vec![0u64; rank];
3037        for d in 0..rank {
3038            let start = coords[d] * chunk_dims[d];
3039            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
3040            extent[d] = end.saturating_sub(start);
3041        }
3042        // Size the buffer, then zero it only when this chunk leaves part of
3043        // its shape uncovered: a full chunk has every byte overwritten below,
3044        // while an edge chunk's padding must read as zero even though a
3045        // reused buffer still holds the previous chunk's bytes.
3046        if out.len() != chunk_bytes {
3047            out.clear();
3048            out.resize(chunk_bytes, 0);
3049        } else if extent != chunk_dims {
3050            out.fill(0);
3051        }
3052        if extent.contains(&0) {
3053            return; // nothing of the dataset falls in this chunk
3054        }
3055
3056        // Row-major strides (in elements) for the source (over `dims`) and the
3057        // destination chunk buffer (over `chunk_dims`).
3058        let mut src_stride = vec![1u64; rank];
3059        let mut dst_stride = vec![1u64; rank];
3060        for d in (0..rank - 1).rev() {
3061            src_stride[d] = src_stride[d + 1] * dims[d + 1];
3062            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
3063        }
3064
3065        // Copy one contiguous run along the last axis per outer multi-index.
3066        let last = rank - 1;
3067        let run = extent[last] as usize * element_size;
3068        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
3069        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
3070        for _ in 0..outer {
3071            let mut src_off = 0u64;
3072            let mut dst_off = 0u64;
3073            for d in 0..rank {
3074                let global = coords[d] * chunk_dims[d] + idx[d];
3075                src_off += global * src_stride[d];
3076                dst_off += idx[d] * dst_stride[d];
3077            }
3078            let s = src_off as usize * element_size;
3079            let dpos = dst_off as usize * element_size;
3080            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
3081
3082            // Advance the multi-index over axes [0..last); the last axis is the
3083            // contiguous run handled above.
3084            let mut d = last;
3085            while d > 0 {
3086                d -= 1;
3087                idx[d] += 1;
3088                if idx[d] < extent[d] {
3089                    break;
3090                }
3091                idx[d] = 0;
3092            }
3093        }
3094    }
3095
3096    /// Write a single chunk to a chunked dataset.
3097    ///
3098    /// `chunk_idx` is the linear chunk index (typically the frame number for
3099    /// streaming datasets). `data` is the raw byte data for one chunk.
3100    ///
3101    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
3102    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
3103    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
3104        match &self.info {
3105            DatasetInfo::Writer {
3106                index, chunk_index, ..
3107            } => {
3108                let Some(kind) = *chunk_index else {
3109                    return Err(Hdf5Error::InvalidState(
3110                        "write_chunk is only for chunked datasets".into(),
3111                    ));
3112                };
3113                if kind == ChunkIndexKind::BtreeV2 {
3114                    return Err(Hdf5Error::InvalidState(
3115                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
3116                         with the chunk's grid coordinates"
3117                            .into(),
3118                    ));
3119                }
3120
3121                let inner = borrow_inner(&self.file_inner);
3122                match &*inner {
3123                    H5FileInner::Writer(writer) => {
3124                        // One op: the slot decode and the write see the same
3125                        // extents.
3126                        let cell = writer.ds(*index);
3127                        let _op = cell.op.lock();
3128                        match kind {
3129                            // All four address a chunk by its grid
3130                            // coordinates, so the linear slot is decoded back
3131                            // into them.
3132                            ChunkIndexKind::FixedArray
3133                            | ChunkIndexKind::Implicit
3134                            | ChunkIndexKind::SingleChunk
3135                            | ChunkIndexKind::BtreeV1 => {
3136                                let coords =
3137                                    writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3138                                writer.write_chunk_at_coords(*index, &coords, data)?;
3139                            }
3140                            _ => writer.write_chunk_inner(*index, chunk_idx as u64, data)?,
3141                        }
3142                        Ok(())
3143                    }
3144                    _ => Err(Hdf5Error::InvalidState(
3145                        "file is no longer in write mode".into(),
3146                    )),
3147                }
3148            }
3149            DatasetInfo::Reader { .. } => {
3150                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3151            }
3152        }
3153    }
3154
3155    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
3156    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
3157    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
3158    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
3159    ///
3160    /// `chunk_idx` is the linear chunk index (the frame number for streaming
3161    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
3162    /// the already-filtered bytes of one chunk — its length is the *stored*
3163    /// (compressed) size, not the uncompressed chunk size.
3164    ///
3165    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
3166    /// dataset's pipeline was **not** applied to this chunk and must be skipped
3167    /// on read. Pass 0 when the full pipeline was already applied upstream (the
3168    /// common case: a codec plugin handed you compressed frames).
3169    ///
3170    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3171    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
3172    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
3173    /// against, so address its chunks with
3174    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
3175    ///
3176    /// # Reading back
3177    ///
3178    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
3179    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
3180    /// the reader skipping exactly the filters the mask marks as not applied.
3181    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
3182        match &self.info {
3183            DatasetInfo::Writer {
3184                index, chunk_index, ..
3185            } => {
3186                let Some(kind) = *chunk_index else {
3187                    return Err(Hdf5Error::InvalidState(
3188                        "write_chunk_raw is only for chunked datasets".into(),
3189                    ));
3190                };
3191                if kind == ChunkIndexKind::BtreeV2 {
3192                    return Err(Hdf5Error::InvalidState(
3193                        "this dataset uses a v2 B-tree chunk index; use \
3194                         write_chunk_raw_at with the chunk's grid coordinates"
3195                            .into(),
3196                    ));
3197                }
3198                if kind == ChunkIndexKind::Implicit {
3199                    return Err(Hdf5Error::InvalidState(
3200                        "this dataset uses the implicit chunk index, which stores \
3201                         every chunk at its full unfiltered size and has nowhere to \
3202                         record a stored size or a filter mask"
3203                            .into(),
3204                    ));
3205                }
3206
3207                let inner = borrow_inner(&self.file_inner);
3208                match &*inner {
3209                    H5FileInner::Writer(writer) => {
3210                        // One op: the slot decode and the write see the same
3211                        // extents.
3212                        let cell = writer.ds(*index);
3213                        let _op = cell.op.lock();
3214                        if kind == ChunkIndexKind::FixedArray {
3215                            // Fixed-array dataset: decode the index-grid slot
3216                            // into row-major grid coordinates.
3217                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3218                            writer.write_compressed_chunk_fixed_array_inner(
3219                                *index,
3220                                &coords,
3221                                data,
3222                                filter_mask,
3223                            )?;
3224                        } else if kind == ChunkIndexKind::BtreeV1 {
3225                            // Same for the classic index, whose key carries a
3226                            // stored size and a filter mask of its own.
3227                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3228                            writer.write_compressed_chunk_btree_v1_inner(
3229                                *index,
3230                                &coords,
3231                                data,
3232                                filter_mask,
3233                            )?;
3234                        } else if kind == ChunkIndexKind::SingleChunk {
3235                            // Same again for the single-chunk index, whose
3236                            // layout message carries the stored size and mask
3237                            // inline.
3238                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
3239                            writer.write_compressed_chunk_single_chunk_inner(
3240                                *index,
3241                                &coords,
3242                                data,
3243                                filter_mask,
3244                            )?;
3245                        } else {
3246                            writer.write_compressed_chunk_inner(
3247                                *index,
3248                                chunk_idx as u64,
3249                                data,
3250                                filter_mask,
3251                            )?;
3252                        }
3253                        Ok(())
3254                    }
3255                    _ => Err(Hdf5Error::InvalidState(
3256                        "file is no longer in write mode".into(),
3257                    )),
3258                }
3259            }
3260            DatasetInfo::Reader { .. } => {
3261                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3262            }
3263        }
3264    }
3265
3266    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
3267    /// chunk-grid coordinates (one per dimension).
3268    ///
3269    /// This is the entry point for datasets with two or more unlimited
3270    /// dimensions. The dataset's logical dimensions are extended to cover
3271    /// the written chunk. `data` is the raw bytes of one full chunk.
3272    ///
3273    /// ```no_run
3274    /// # use rust_hdf5::H5File;
3275    /// let file = H5File::create("bt2.h5").unwrap();
3276    /// let ds = file.new_dataset::<i32>()
3277    ///     .shape(&[0, 0])
3278    ///     .chunk(&[2, 2])
3279    ///     .max_shape(&[None, None])
3280    ///     .create("grid")
3281    ///     .unwrap();
3282    /// let chunk = [0i32, 1, 2, 3];
3283    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
3284    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3285    /// ```
3286    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
3287        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
3288    }
3289
3290    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
3291    /// addressed by its chunk-grid coordinates.
3292    ///
3293    /// The coordinate-addressed twin of
3294    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
3295    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
3296    /// there is no fixed chunk grid for a linear index to mean anything against.
3297    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
3298    /// to cover the written chunk.
3299    ///
3300    /// `data` is the already-filtered bytes of one chunk — its length is the
3301    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
3302    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
3303    /// full pipeline already ran upstream.
3304    ///
3305    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
3306    /// has no slot to record a stored size or mask.
3307    pub fn write_chunk_raw_at(
3308        &self,
3309        chunk_coords: &[usize],
3310        data: &[u8],
3311        filter_mask: u32,
3312    ) -> Result<()> {
3313        self.write_chunk_at_inner(
3314            chunk_coords,
3315            ChunkBytes::Prefiltered { data, filter_mask },
3316            "write_chunk_raw_at",
3317        )
3318    }
3319
3320    /// The single owner of coordinate-addressed chunk writes: validates the
3321    /// coordinates, grows the dataspace to cover them, and routes the bytes to
3322    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
3323    /// here or already ran upstream is carried by `bytes`, not by a second copy
3324    /// of this dispatch.
3325    fn write_chunk_at_inner(
3326        &self,
3327        chunk_coords: &[usize],
3328        bytes: ChunkBytes<'_>,
3329        what: &str,
3330    ) -> Result<()> {
3331        match &self.info {
3332            DatasetInfo::Writer {
3333                index, chunk_index, ..
3334            } => {
3335                let Some(kind) = *chunk_index else {
3336                    return Err(Hdf5Error::InvalidState(format!(
3337                        "{what} is only for chunked datasets"
3338                    )));
3339                };
3340                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3341                let inner = borrow_inner(&self.file_inner);
3342                let writer = match &*inner {
3343                    H5FileInner::Writer(w) => w,
3344                    _ => {
3345                        return Err(Hdf5Error::InvalidState(
3346                            "file is no longer in write mode".into(),
3347                        ))
3348                    }
3349                };
3350                // Whole-operation guard: the dims snapshot, the chunk write
3351                // and the extend below must not interleave with a concurrent
3352                // same-dataset operation.
3353                let cell = writer.ds(*index);
3354                let _op = cell.op.lock();
3355                let chunk_dims = writer
3356                    .dataset_chunk_dims(*index)
3357                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3358                    .to_vec();
3359                let dims = writer.dataset_dims(*index).to_vec();
3360                if coords.len() != dims.len() {
3361                    return Err(Hdf5Error::InvalidState(format!(
3362                        "chunk_coords has {} entries but the dataset has {} dimensions",
3363                        coords.len(),
3364                        dims.len()
3365                    )));
3366                }
3367                if chunk_dims.len() != dims.len() {
3368                    return Err(Hdf5Error::InvalidState(format!(
3369                        "dataset chunk shape has {} dimensions but the dataspace has {}",
3370                        chunk_dims.len(),
3371                        dims.len()
3372                    )));
3373                }
3374
3375                if kind == ChunkIndexKind::FixedArray {
3376                    // Fixed-array (fixed-shape) dataset: no dimension growth.
3377                    match bytes {
3378                        ChunkBytes::Unfiltered(data) => {
3379                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
3380                        }
3381                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3382                            .write_compressed_chunk_fixed_array_inner(
3383                                *index,
3384                                &coords,
3385                                data,
3386                                filter_mask,
3387                            )?,
3388                    }
3389                    return Ok(());
3390                }
3391
3392                if kind == ChunkIndexKind::Implicit {
3393                    // Implicit index: fixed shape, so no dimension growth
3394                    // either, and no slot to record a stored size in.
3395                    match bytes {
3396                        ChunkBytes::Unfiltered(data) => {
3397                            writer.write_chunk_implicit_inner(*index, &coords, data)?
3398                        }
3399                        ChunkBytes::Prefiltered { .. } => {
3400                            return Err(Hdf5Error::InvalidState(
3401                                "this dataset uses the implicit chunk index, which stores \
3402                                 every chunk at its full unfiltered size and has nowhere \
3403                                 to record a stored size or a filter mask"
3404                                    .into(),
3405                            ))
3406                        }
3407                    }
3408                    return Ok(());
3409                }
3410
3411                if kind == ChunkIndexKind::SingleChunk {
3412                    // Single-chunk index: fixed shape covered by exactly one
3413                    // chunk, so no dimension growth either.
3414                    match bytes {
3415                        ChunkBytes::Unfiltered(data) => {
3416                            writer.write_chunk_single_chunk_inner(*index, &coords, data)?
3417                        }
3418                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3419                            .write_compressed_chunk_single_chunk_inner(
3420                                *index,
3421                                &coords,
3422                                data,
3423                                filter_mask,
3424                            )?,
3425                    }
3426                    return Ok(());
3427                }
3428
3429                // The remaining indexes (v2 B-tree, v1 B-tree, extensible
3430                // array) can all grow: validate the coordinates and compute
3431                // the grown dimensions up-front, before any chunk is
3432                // written, so an overflowing coordinate cannot leave an
3433                // orphaned chunk in the file.
3434                //
3435                // The last chunk of a dimension usually hangs past the extent
3436                // — a length of 10 in chunks of 4 ends at 12 — so the growth
3437                // is capped at the declared maximum, which is what the chunk
3438                // still covers. Without the cap a legal edge chunk would be
3439                // written and then rejected by the extend below.
3440                let max_dims = writer.dataset_max_dims(*index);
3441                let mut new_dims = dims.clone();
3442                for d in 0..dims.len() {
3443                    let needed = coords[d]
3444                        .checked_add(1)
3445                        .and_then(|c| c.checked_mul(chunk_dims[d]))
3446                        .ok_or_else(|| {
3447                            Hdf5Error::InvalidState(format!(
3448                                "chunk coordinate {} in dimension {} is too large",
3449                                coords[d], d
3450                            ))
3451                        })?;
3452                    let needed = needed.min(max_dims[d]);
3453                    if needed > new_dims[d] {
3454                        new_dims[d] = needed;
3455                    }
3456                }
3457
3458                if kind == ChunkIndexKind::BtreeV2 {
3459                    match bytes {
3460                        ChunkBytes::Unfiltered(data) => {
3461                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
3462                        }
3463                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3464                            .write_compressed_chunk_btree_v2_inner(
3465                                *index,
3466                                &coords,
3467                                data,
3468                                filter_mask,
3469                            )?,
3470                    }
3471                } else if kind == ChunkIndexKind::BtreeV1 {
3472                    // The classic index takes any shape, fixed or unlimited,
3473                    // so it grows the dataspace with the chunk the way the v2
3474                    // B-tree does — bounded below by the maximum extent.
3475                    match bytes {
3476                        ChunkBytes::Unfiltered(data) => {
3477                            writer.write_chunk_btree_v1_inner(*index, &coords, data)?
3478                        }
3479                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3480                            .write_compressed_chunk_btree_v1_inner(
3481                                *index,
3482                                &coords,
3483                                data,
3484                                filter_mask,
3485                            )?,
3486                    }
3487                } else {
3488                    // Extensible array: the chunk's index-grid slot (row-major
3489                    // against the maximum extent).
3490                    let linear = writer.chunk_slot(*index, &coords)?;
3491                    match bytes {
3492                        ChunkBytes::Unfiltered(data) => {
3493                            writer.write_chunk_inner(*index, linear, data)?
3494                        }
3495                        ChunkBytes::Prefiltered { data, filter_mask } => writer
3496                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
3497                    }
3498                }
3499
3500                if new_dims != dims {
3501                    writer.extend_dataset_inner(*index, &new_dims)?;
3502                }
3503                Ok(())
3504            }
3505            DatasetInfo::Reader { .. } => {
3506                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3507            }
3508        }
3509    }
3510
3511    /// Write multiple chunks in a batch, optionally compressing in parallel.
3512    ///
3513    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
3514    /// pipeline is configured and the `parallel` feature is enabled, all
3515    /// chunks are compressed concurrently via rayon.
3516    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
3517        match &self.info {
3518            DatasetInfo::Writer {
3519                index, chunk_index, ..
3520            } => {
3521                if chunk_index.is_none() {
3522                    return Err(Hdf5Error::InvalidState(
3523                        "write_chunks_batch is only for chunked datasets".into(),
3524                    ));
3525                }
3526                let pairs: Vec<(u64, &[u8])> = chunks
3527                    .iter()
3528                    .map(|(idx, data)| (*idx as u64, *data))
3529                    .collect();
3530                let inner = borrow_inner(&self.file_inner);
3531                match &*inner {
3532                    H5FileInner::Writer(writer) => {
3533                        writer.write_chunks_batch(*index, &pairs)?;
3534                        Ok(())
3535                    }
3536                    _ => Err(Hdf5Error::InvalidState(
3537                        "file is no longer in write mode".into(),
3538                    )),
3539                }
3540            }
3541            DatasetInfo::Reader { .. } => {
3542                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
3543            }
3544        }
3545    }
3546
3547    /// Append data along the first dimension of a chunked dataset.
3548    ///
3549    /// `data` must contain a whole number of "frames" — slices along
3550    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
3551    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
3552    /// multiple of `H * W`.
3553    ///
3554    /// This method writes the necessary chunks and extends the dataset
3555    /// shape automatically.
3556    ///
3557    /// ```no_run
3558    /// # use rust_hdf5::H5File;
3559    /// let file = H5File::create("append.h5").unwrap();
3560    /// let ds = file.new_dataset::<f64>()
3561    ///     .shape(&[0, 3])
3562    ///     .chunk(&[1, 3])
3563    ///     .max_shape(&[None, Some(3)])
3564    ///     .create("data")
3565    ///     .unwrap();
3566    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
3567    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
3568    /// ```
3569    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
3570        match &self.info {
3571            DatasetInfo::Writer {
3572                index,
3573                element_size,
3574                chunk_index,
3575                ..
3576            } => {
3577                if chunk_index.is_none() {
3578                    return Err(Hdf5Error::InvalidState(
3579                        "append is only for chunked datasets".into(),
3580                    ));
3581                }
3582                if T::element_size() != *element_size {
3583                    return Err(Hdf5Error::TypeMismatch(format!(
3584                        "append type has element size {} but dataset expects {}",
3585                        T::element_size(),
3586                        element_size,
3587                    )));
3588                }
3589
3590                let ds_index = *index;
3591                let es = *element_size;
3592
3593                let inner = borrow_inner(&self.file_inner);
3594                let writer = match &*inner {
3595                    H5FileInner::Writer(w) => w,
3596                    _ => {
3597                        return Err(Hdf5Error::InvalidState(
3598                            "file is no longer in write mode".into(),
3599                        ))
3600                    }
3601                };
3602
3603                // Whole-operation guard: the buffer take, the frame writes,
3604                // the re-buffer and the extend below are separate slot
3605                // acquisitions that a concurrent same-dataset append must not
3606                // interleave with.
3607                let cell = writer.ds(ds_index);
3608                let _op = cell.op.lock();
3609
3610                let chunk_dims = writer
3611                    .dataset_chunk_dims(ds_index)
3612                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
3613                    .to_vec();
3614                let dims = writer.dataset_dims(ds_index).to_vec();
3615
3616                // Frame size = product of dims[1..]
3617                let frame_elems: usize = if dims.len() > 1 {
3618                    dims[1..].iter().map(|&d| d as usize).product()
3619                } else {
3620                    1
3621                };
3622
3623                if frame_elems == 0 {
3624                    return Err(Hdf5Error::InvalidState(
3625                        "cannot append to dataset with zero-size trailing dimensions".into(),
3626                    ));
3627                }
3628
3629                if !data.len().is_multiple_of(frame_elems) {
3630                    return Err(Hdf5Error::InvalidState(format!(
3631                        "data length {} is not a multiple of frame size {}",
3632                        data.len(),
3633                        frame_elems,
3634                    )));
3635                }
3636
3637                let n_new_frames = data.len() / frame_elems;
3638                let current_dim0 = dims[0] as usize;
3639
3640                // Chunk size along first dimension
3641                let chunk_dim0 = chunk_dims[0] as usize;
3642                let frame_bytes = frame_elems * es;
3643
3644                let host = unsafe {
3645                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
3646                };
3647                let datatype = writer.dataset_datatype(ds_index);
3648                let raw = to_stored_byte_order(host, &datatype, es)?;
3649
3650                // Merge the buffer with the new frames when it is the
3651                // dataset's tail; a buffer left mid-extent (the extent moved
3652                // past it) keeps its recorded place — flush it and start
3653                // fresh at the current end.
3654                let taken = { writer.ds(ds_index).lock().append.take() };
3655                let (base_dim0, buffered_frames, mut combined) = match taken {
3656                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
3657                        (b.base as usize, b.frames as usize, b.bytes)
3658                    }
3659                    Some(b) => {
3660                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
3661                        (current_dim0, 0, Vec::new())
3662                    }
3663                    None => (current_dim0, 0, Vec::new()),
3664                };
3665                combined.extend_from_slice(&raw);
3666
3667                let total_frames = buffered_frames + n_new_frames;
3668
3669                // Rows up to the last chunk boundary are written now; the
3670                // tail that does not complete a chunk goes back in the
3671                // buffer for the next append (or the flush at close). The
3672                // boundary can precede `base_dim0` — a reopened file's
3673                // flushed partial chunk leaves the base mid-chunk — in
3674                // which case everything is tail.
3675                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
3676                let write_frames = last_boundary.saturating_sub(base_dim0);
3677                let tail_frames = total_frames - write_frames;
3678                if write_frames > 0 {
3679                    writer.write_append_frames(
3680                        ds_index,
3681                        base_dim0 as u64,
3682                        write_frames as u64,
3683                        &combined[..write_frames * frame_bytes],
3684                    )?;
3685                }
3686                if tail_frames > 0 {
3687                    let ds = writer.ds(ds_index);
3688                    let mut m = ds.lock();
3689                    m.append = Some(crate::io::writer::AppendBuffer {
3690                        base: (base_dim0 + write_frames) as u64,
3691                        frames: tail_frames as u64,
3692                        bytes: combined[write_frames * frame_bytes..].to_vec(),
3693                    });
3694                }
3695
3696                // Extend dims to include all frames (buffered + new)
3697                let logical_dim0 = base_dim0 + total_frames;
3698                let mut new_dims: Vec<u64> = dims;
3699                new_dims[0] = logical_dim0 as u64;
3700                writer.extend_dataset_inner(ds_index, &new_dims)?;
3701
3702                Ok(())
3703            }
3704            DatasetInfo::Reader { .. } => {
3705                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
3706            }
3707        }
3708    }
3709
3710    /// Extend the dimensions of a chunked dataset.
3711    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
3712        match &self.info {
3713            DatasetInfo::Writer {
3714                index, chunk_index, ..
3715            } => {
3716                if chunk_index.is_none() {
3717                    return Err(Hdf5Error::InvalidState(
3718                        "extend is only for chunked datasets".into(),
3719                    ));
3720                }
3721
3722                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3723                let inner = borrow_inner(&self.file_inner);
3724                match &*inner {
3725                    H5FileInner::Writer(writer) => {
3726                        writer.extend_dataset(*index, &dims_u64)?;
3727                        Ok(())
3728                    }
3729                    _ => Err(Hdf5Error::InvalidState(
3730                        "file is no longer in write mode".into(),
3731                    )),
3732                }
3733            }
3734            DatasetInfo::Reader { .. } => {
3735                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
3736            }
3737        }
3738    }
3739
3740    /// Set the logical extent of a chunked dataset, growing **or
3741    /// shrinking** any dimension.
3742    ///
3743    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
3744    /// dimension — for example to correct an over-extended frame count
3745    /// after writing a partial multi-frame chunk. Shrinking prunes the
3746    /// stored chunks the way libhdf5's `H5Dset_extent` does: a chunk
3747    /// entirely beyond the new extent is removed from the chunk index and
3748    /// its storage freed for reuse, and a chunk the new extent cuts
3749    /// through has its out-of-extent region overwritten with the fill
3750    /// value — so growing the extent back exposes fill values, not the
3751    /// old data. The new extent must not exceed the dataset's maximum
3752    /// dimensions.
3753    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
3754        match &self.info {
3755            DatasetInfo::Writer { index, .. } => {
3756                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
3757                let inner = borrow_inner(&self.file_inner);
3758                match &*inner {
3759                    H5FileInner::Writer(writer) => {
3760                        writer.set_dataset_extent(*index, &dims_u64)?;
3761                        Ok(())
3762                    }
3763                    _ => Err(Hdf5Error::InvalidState(
3764                        "file is no longer in write mode".into(),
3765                    )),
3766                }
3767            }
3768            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
3769                "cannot set extent in read mode".into(),
3770            )),
3771        }
3772    }
3773
3774    /// Flush a chunked dataset's index structures to disk.
3775    pub fn flush(&self) -> Result<()> {
3776        match &self.info {
3777            DatasetInfo::Writer { index, .. } => {
3778                let inner = borrow_inner(&self.file_inner);
3779                match &*inner {
3780                    H5FileInner::Writer(writer) => {
3781                        writer.flush_dataset(*index)?;
3782                        Ok(())
3783                    }
3784                    _ => Ok(()),
3785                }
3786            }
3787            DatasetInfo::Reader { .. } => Ok(()),
3788        }
3789    }
3790
3791    /// Read a slice (hyperslab) of the dataset as a typed vector.
3792    ///
3793    /// `starts` and `counts` define the N-dimensional selection:
3794    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
3795    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
3796        match &self.info {
3797            DatasetInfo::Reader {
3798                name,
3799                shape,
3800                element_size,
3801            } => {
3802                if T::element_size() != *element_size {
3803                    return Err(Hdf5Error::TypeMismatch(format!(
3804                        "read type has element size {} but dataset has element size {}",
3805                        T::element_size(),
3806                        element_size,
3807                    )));
3808                }
3809                let datatype = self.datatype()?;
3810                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
3811                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
3812
3813                // Bounds before sizing: the destination is allocated here,
3814                // ahead of the reader's own check, so a selection the extent
3815                // does not admit must be refused before its size is computed.
3816                let dims: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
3817                check_hyperslab(&dims, &starts_u64, &counts_u64)?;
3818                let count = element_count(&counts_u64)?;
3819                let mut inner = borrow_inner_mut(&self.file_inner);
3820                let H5FileInner::Reader(reader) = &mut *inner else {
3821                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3822                };
3823                // The selection lands in the vector this returns, so its bytes
3824                // are touched once instead of being read into a byte buffer and
3825                // copied into a second one of the same size.
3826                read_image_into_new(count, |image| {
3827                    reader.read_slice_into_dst(
3828                        name,
3829                        &starts_u64,
3830                        &counts_u64,
3831                        image,
3832                        ReadDst::Fresh,
3833                    )?;
3834                    to_host_byte_order(image, &datatype, T::element_size())
3835                })
3836            }
3837            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3838                "cannot read_slice from a dataset in write mode".into(),
3839            )),
3840        }
3841    }
3842
3843    /// Read a strided hyperslab as a typed vector — h5py's stepped slicing
3844    /// (`ds[a:b:s]`) or the general `start`/`stride`/`count`/`block` form of
3845    /// `H5Sselect_hyperslab`.
3846    ///
3847    /// One entry per dimension: `start[d]` is the first index, `stride[d]`
3848    /// the spacing between selected blocks (all-`1` is the same selection
3849    /// [`read_slice`](Self::read_slice) reads), `count[d]` how many blocks,
3850    /// and `block[d]` how many contiguous elements each block covers. The
3851    /// returned vector is row-major over `count[d] * block[d]` per
3852    /// dimension — exactly the shape h5py's stepped slicing produces.
3853    ///
3854    /// ```no_run
3855    /// # use rust_hdf5::H5File;
3856    /// let file = H5File::open("data.h5").unwrap();
3857    /// let ds = file.dataset("series").unwrap(); // shape [100]
3858    /// // Python: ds[0:100:2] — every other element.
3859    /// let evens: Vec<f64> = ds.read_hyperslab(&[0], &[2], &[50], &[1]).unwrap();
3860    /// ```
3861    pub fn read_hyperslab<T: H5Type>(
3862        &self,
3863        start: &[usize],
3864        stride: &[usize],
3865        count: &[usize],
3866        block: &[usize],
3867    ) -> Result<Vec<T>> {
3868        match &self.info {
3869            DatasetInfo::Reader {
3870                name, element_size, ..
3871            } => {
3872                if T::element_size() != *element_size {
3873                    return Err(Hdf5Error::TypeMismatch(format!(
3874                        "read type has element size {} but dataset has element size {}",
3875                        T::element_size(),
3876                        element_size,
3877                    )));
3878                }
3879                let datatype = self.datatype()?;
3880                let start_u64: Vec<u64> = start.iter().map(|&s| s as u64).collect();
3881                let stride_u64: Vec<u64> = stride.iter().map(|&s| s as u64).collect();
3882                let count_u64: Vec<u64> = count.iter().map(|&c| c as u64).collect();
3883                let block_u64: Vec<u64> = block.iter().map(|&b| b as u64).collect();
3884
3885                let selected: Vec<u64> = count_u64
3886                    .iter()
3887                    .zip(&block_u64)
3888                    .map(|(&c, &b)| c.saturating_mul(b))
3889                    .collect();
3890                let n = element_count(&selected)?;
3891                let mut inner = borrow_inner_mut(&self.file_inner);
3892                let H5FileInner::Reader(reader) = &mut *inner else {
3893                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3894                };
3895                read_image_into_new(n, |image| {
3896                    reader.read_hyperslab_into(
3897                        name,
3898                        &start_u64,
3899                        &stride_u64,
3900                        &count_u64,
3901                        &block_u64,
3902                        image,
3903                    )?;
3904                    to_host_byte_order(image, &datatype, T::element_size())
3905                })
3906            }
3907            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3908                "cannot read_hyperslab from a dataset in write mode".into(),
3909            )),
3910        }
3911    }
3912
3913    /// Read a list of coordinates in one call, as a typed vector — h5py
3914    /// fancy indexing with a coordinate list.
3915    ///
3916    /// `points[i]` is a coordinate with one entry per dimension. The
3917    /// returned vector holds one element per point, in the same order as
3918    /// `points`, regardless of the dataset's rank.
3919    ///
3920    /// ```no_run
3921    /// # use rust_hdf5::H5File;
3922    /// let file = H5File::open("data.h5").unwrap();
3923    /// let ds = file.dataset("grid").unwrap(); // shape [10, 10]
3924    /// // Python: ds[np.array([[0, 0], [3, 4], [9, 9]])]
3925    /// let picked: Vec<f64> = ds.read_points(&[vec![0, 0], vec![3, 4], vec![9, 9]]).unwrap();
3926    /// ```
3927    pub fn read_points<T: H5Type>(&self, points: &[Vec<usize>]) -> Result<Vec<T>> {
3928        match &self.info {
3929            DatasetInfo::Reader {
3930                name, element_size, ..
3931            } => {
3932                if T::element_size() != *element_size {
3933                    return Err(Hdf5Error::TypeMismatch(format!(
3934                        "read type has element size {} but dataset has element size {}",
3935                        T::element_size(),
3936                        element_size,
3937                    )));
3938                }
3939                let datatype = self.datatype()?;
3940                let points_u64: Vec<Vec<u64>> = points
3941                    .iter()
3942                    .map(|p| p.iter().map(|&c| c as u64).collect())
3943                    .collect();
3944
3945                let mut inner = borrow_inner_mut(&self.file_inner);
3946                let H5FileInner::Reader(reader) = &mut *inner else {
3947                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
3948                };
3949                read_image_into_new(points_u64.len(), |image| {
3950                    reader.read_points_into(name, &points_u64, image, ReadDst::Fresh)?;
3951                    to_host_byte_order(image, &datatype, T::element_size())
3952                })
3953            }
3954            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3955                "cannot read_points from a dataset in write mode".into(),
3956            )),
3957        }
3958    }
3959
3960    /// Read one chunk's raw (still-filtered) bytes and its filter mask,
3961    /// addressed by chunk-grid coordinates — the read half of
3962    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) and the HDF5 "direct
3963    /// chunk read" (`H5Dread_chunk`, formerly `H5DOread_chunk`; h5py's
3964    /// `Dataset.id.read_direct_chunk`).
3965    ///
3966    /// The bytes are exactly what is stored on disk: filtered/compressed if
3967    /// the dataset has a filter pipeline, with no decompression applied. The
3968    /// returned `u32` is the chunk's filter mask: bit *i* set means filter
3969    /// *i* of the pipeline was **not** applied to this particular chunk and
3970    /// must be skipped when reversing it.
3971    ///
3972    /// `Err` if the dataset is not chunked, `chunk_coords` has the wrong
3973    /// rank, or the chunk at those coordinates has never been written.
3974    ///
3975    /// ```no_run
3976    /// # use rust_hdf5::H5File;
3977    /// let file = H5File::open("data.h5").unwrap();
3978    /// let ds = file.dataset("frames").unwrap();
3979    /// let (raw, filter_mask) = ds.read_chunk_raw_at(&[0, 0]).unwrap();
3980    /// ```
3981    pub fn read_chunk_raw_at(&self, chunk_coords: &[usize]) -> Result<(Vec<u8>, u32)> {
3982        match &self.info {
3983            DatasetInfo::Reader { name, .. } => {
3984                let coords_u64: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
3985                let mut inner = borrow_inner_mut(&self.file_inner);
3986                match &mut *inner {
3987                    H5FileInner::Reader(reader) => Ok(reader.read_chunk_raw_at(name, &coords_u64)?),
3988                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
3989                }
3990            }
3991            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
3992                "cannot read_chunk_raw_at from a dataset in write mode".into(),
3993            )),
3994        }
3995    }
3996
3997    /// Write a typed slice to a sub-region of the dataset.
3998    ///
3999    /// `starts` and `counts` define the N-dimensional selection, which must lie
4000    /// inside the dataset's current extent.
4001    ///
4002    /// Works for both contiguous and chunked datasets. For a chunked dataset
4003    /// only the chunks the selection touches are rewritten — a partially
4004    /// covered chunk is read back, patched, and written again, so updating one
4005    /// row of an appendable dataset costs the chunks that row crosses rather
4006    /// than the whole dataset. Elements of a touched chunk that the selection
4007    /// does not cover keep their stored value, or the dataset's fill value if
4008    /// the chunk did not exist yet.
4009    pub fn write_slice<T: H5Type>(
4010        &self,
4011        starts: &[usize],
4012        counts: &[usize],
4013        data: &[T],
4014    ) -> Result<()> {
4015        match &self.info {
4016            DatasetInfo::Writer {
4017                index,
4018                element_size,
4019                ..
4020            } => {
4021                if T::element_size() != *element_size {
4022                    return Err(Hdf5Error::TypeMismatch(format!(
4023                        "write type has element size {} but dataset expects {}",
4024                        T::element_size(),
4025                        element_size,
4026                    )));
4027                }
4028
4029                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4030                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4031
4032                let inner = borrow_inner(&self.file_inner);
4033                let H5FileInner::Writer(writer) = &*inner else {
4034                    return Err(Hdf5Error::InvalidState(
4035                        "file is no longer in write mode".into(),
4036                    ));
4037                };
4038
4039                // Bounds before sizing, against the extent the writer holds
4040                // now (an extend since this handle was taken counts): a
4041                // selection the extent does not admit is refused for that
4042                // reason, not for the size it would have had.
4043                check_hyperslab(&writer.dataset_dims(*index), &starts_u64, &counts_u64)?;
4044                let expected = element_count(&counts_u64)?;
4045                if data.len() != expected {
4046                    return Err(Hdf5Error::InvalidState(format!(
4047                        "data length {} does not match slice size {}",
4048                        data.len(),
4049                        expected,
4050                    )));
4051                }
4052
4053                let byte_len = data.len() * T::element_size();
4054                let host =
4055                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
4056
4057                let datatype = writer.dataset_datatype(*index);
4058                let stored = to_stored_byte_order(host, &datatype, T::element_size())?;
4059                writer.write_slice(*index, &starts_u64, &counts_u64, &stored)?;
4060                Ok(())
4061            }
4062            DatasetInfo::Reader { .. } => {
4063                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4064            }
4065        }
4066    }
4067
4068    /// Replace elements `start .. start + strings.len()` of a 1-D
4069    /// variable-length string dataset.
4070    ///
4071    /// The extent and every element outside the range are left alone, and the
4072    /// cost is the new strings plus the chunks holding their references — not
4073    /// the column. The dataset's character set is enforced: a non-ASCII
4074    /// replacement in a dataset that declares ASCII is rejected rather than
4075    /// stored under a datatype that misdescribes it.
4076    ///
4077    /// The global heap objects the replaced references pointed at are freed —
4078    /// the same reclaim libhdf5 performs on an overwrite — so updating one
4079    /// element repeatedly reuses space rather than growing the file. A
4080    /// collection emptied by the update returns its block to the allocator.
4081    /// Under SWMR nothing is freed, because a reader may still be following
4082    /// those references.
4083    ///
4084    /// ```no_run
4085    /// # use rust_hdf5::H5File;
4086    /// let file = H5File::open_rw("meta.h5").unwrap();
4087    /// let ds = file.dataset_writer("notes").unwrap();
4088    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
4089    /// file.close().unwrap();
4090    /// ```
4091    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
4092        match &self.info {
4093            DatasetInfo::Writer { index, .. } => {
4094                let inner = borrow_inner(&self.file_inner);
4095                match &*inner {
4096                    H5FileInner::Writer(writer) => {
4097                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
4098                        Ok(())
4099                    }
4100                    _ => Err(Hdf5Error::InvalidState(
4101                        "file is no longer in write mode".into(),
4102                    )),
4103                }
4104            }
4105            DatasetInfo::Reader { .. } => {
4106                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4107            }
4108        }
4109    }
4110
4111    /// Read variable-length strings from a dataset.
4112    ///
4113    /// This handles h5py-style vlen string datasets that store strings
4114    /// as global heap references. Returns one String per element.
4115    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
4116        match &self.info {
4117            DatasetInfo::Reader { name, .. } => {
4118                let mut inner = borrow_inner_mut(&self.file_inner);
4119                match &mut *inner {
4120                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
4121                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4122                }
4123            }
4124            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4125                "cannot read vlen strings from a dataset in write mode".into(),
4126            )),
4127        }
4128    }
4129
4130    /// Read variable-length byte arrays from a dataset.
4131    ///
4132    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
4133    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
4134    /// that store each element as a global heap reference. Returns one
4135    /// `Vec<u8>` per element.
4136    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
4137        match &self.info {
4138            DatasetInfo::Reader { name, .. } => {
4139                let mut inner = borrow_inner_mut(&self.file_inner);
4140                match &mut *inner {
4141                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
4142                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4143                }
4144            }
4145            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4146                "cannot read vlen bytes from a dataset in write mode".into(),
4147            )),
4148        }
4149    }
4150
4151    /// Write object references naming `paths` into elements `0..paths.len()`
4152    /// — h5py's `refs[i] = f['/target'].ref`.
4153    ///
4154    /// The dataset must have been created with
4155    /// [`object_references`](DatasetBuilder::object_references). A path names
4156    /// a dataset or a group (`/` is the root group) and must already exist;
4157    /// what reaches the file is the target's object header address, which is
4158    /// assigned when the file is finalized. Elements left unwritten read back
4159    /// as null references.
4160    pub fn write_object_references(&self, paths: &[&str]) -> Result<()> {
4161        match &self.info {
4162            DatasetInfo::Writer { index, .. } => {
4163                let inner = borrow_inner(&self.file_inner);
4164                match &*inner {
4165                    H5FileInner::Writer(writer) => {
4166                        writer.write_object_references(*index, 0, paths)?;
4167                        Ok(())
4168                    }
4169                    _ => Err(Hdf5Error::InvalidState(
4170                        "file is no longer in write mode".into(),
4171                    )),
4172                }
4173            }
4174            DatasetInfo::Reader { .. } => {
4175                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4176            }
4177        }
4178    }
4179
4180    /// Mark this dataset as a dimension scale — `H5DSset_scale`, h5py's
4181    /// `ds.make_scale(name)`.
4182    ///
4183    /// Writes the `CLASS` attribute as the fixed-length null-terminated
4184    /// string `DIMENSION_SCALE` (16 bytes, the width `H5DSis_scale` checks
4185    /// for) and, when `name` is given, `NAME` the same way. A dataset that
4186    /// has scales attached to it cannot become one. Write mode only.
4187    ///
4188    /// ```no_run
4189    /// # use rust_hdf5::H5File;
4190    /// let file = H5File::create("scales.h5").unwrap();
4191    /// let x = file.new_dataset::<f64>().shape([4]).create("x").unwrap();
4192    /// x.write_raw(&[0.0, 0.5, 1.0, 1.5]).unwrap();
4193    /// x.set_scale(Some("x")).unwrap();
4194    /// ```
4195    pub fn set_scale(&self, name: Option<&str>) -> Result<()> {
4196        let index = self.writer_index("set_scale")?;
4197        let inner = borrow_inner(&self.file_inner);
4198        match &*inner {
4199            H5FileInner::Writer(writer) => Ok(writer.set_dimension_scale(index, name)?),
4200            _ => Err(Hdf5Error::InvalidState(
4201                "file is no longer in write mode".into(),
4202            )),
4203        }
4204    }
4205
4206    /// Attach `scale` as a dimension scale of this dataset's `axis` —
4207    /// `H5DSattach_scale`, h5py's `ds.dims[axis].attach_scale(scale)`.
4208    ///
4209    /// Records the attachment in this dataset's `DIMENSION_LIST` and the
4210    /// scale's `REFERENCE_LIST`, and marks `scale` as a dimension scale if it
4211    /// is not one yet. An axis may carry several scales: each attach appends.
4212    /// Attaching a scale already on that axis changes nothing. Both datasets
4213    /// must belong to this file, in write mode; a scale cannot have scales of
4214    /// its own, a dataset that is a scale cannot have scales attached, and
4215    /// `axis` must be below the rank (a scalar dataset counts as rank 1).
4216    ///
4217    /// ```no_run
4218    /// # use rust_hdf5::H5File;
4219    /// let file = H5File::create("scales.h5").unwrap();
4220    /// let data = file.new_dataset::<i32>().shape([2, 3]).create("data").unwrap();
4221    /// let x = file.new_dataset::<f64>().shape([3]).create("x").unwrap();
4222    /// x.set_scale(Some("x")).unwrap();
4223    /// data.attach_scale(1, &x).unwrap();
4224    /// ```
4225    pub fn attach_scale(&self, axis: usize, scale: &H5Dataset) -> Result<()> {
4226        let did = self.writer_index("attach_scale")?;
4227        let dsid = scale.writer_index("attach_scale")?;
4228        if !same_inner(&self.file_inner, &scale.file_inner) {
4229            return Err(Hdf5Error::InvalidState(
4230                "the dimension scale belongs to another file".into(),
4231            ));
4232        }
4233        let inner = borrow_inner(&self.file_inner);
4234        match &*inner {
4235            H5FileInner::Writer(writer) => Ok(writer.attach_dimension_scale(did, dsid, axis)?),
4236            _ => Err(Hdf5Error::InvalidState(
4237                "file is no longer in write mode".into(),
4238            )),
4239        }
4240    }
4241
4242    /// This dataset's index in the writer's registry, or the error a
4243    /// write-only operation `what` reports on a read-mode handle.
4244    fn writer_index(&self, what: &str) -> Result<usize> {
4245        match &self.info {
4246            DatasetInfo::Writer { index, .. } => Ok(*index),
4247            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(format!(
4248                "{what} is only available in write mode"
4249            ))),
4250        }
4251    }
4252
4253    /// Write region references over `targets` into elements
4254    /// `0..targets.len()` — h5py's `refs[i] = f['/target'].regionref[0:3]`.
4255    ///
4256    /// The dataset must have been created with
4257    /// [`region_references`](DatasetBuilder::region_references). Each target is
4258    /// the path of an existing *dataset* and a [`Selection`] over it, which
4259    /// must fit that dataset's extent — the rule `H5Rcreate` applies. What
4260    /// reaches the file is a global-heap object holding the target's object
4261    /// header address (assigned when the file is finalized) and the serialized
4262    /// selection. Elements left unwritten read back as null references.
4263    ///
4264    /// ```no_run
4265    /// # use rust_hdf5::{H5File, PointSelection, Selection};
4266    /// let file = H5File::create("regions.h5").unwrap();
4267    /// file.new_dataset::<i32>().shape([4, 6]).create("m").unwrap();
4268    /// let refs = file.new_dataset::<u64>()
4269    ///     .region_references()
4270    ///     .shape([1])
4271    ///     .create("refs")
4272    ///     .unwrap();
4273    /// let points = Selection::Points(PointSelection {
4274    ///     rank: 2,
4275    ///     points: vec![vec![0, 1], vec![3, 5]],
4276    /// });
4277    /// refs.write_region_references(&[("/m", points)]).unwrap();
4278    /// file.close().unwrap();
4279    /// ```
4280    pub fn write_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4281        match &self.info {
4282            DatasetInfo::Writer { index, .. } => {
4283                let inner = borrow_inner(&self.file_inner);
4284                match &*inner {
4285                    H5FileInner::Writer(writer) => {
4286                        writer.write_region_references(*index, 0, targets)?;
4287                        Ok(())
4288                    }
4289                    _ => Err(Hdf5Error::InvalidState(
4290                        "file is no longer in write mode".into(),
4291                    )),
4292                }
4293            }
4294            DatasetInfo::Reader { .. } => {
4295                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4296            }
4297        }
4298    }
4299
4300    /// Write revised region references over `targets` into elements
4301    /// `0..targets.len()` — `H5Rcreate_region` plus `H5Dwrite` of an
4302    /// `H5T_STD_REF` dataset.
4303    ///
4304    /// The dataset must have been created with
4305    /// [`std_region_references`](DatasetBuilder::std_region_references) or one
4306    /// of its two siblings, which make the same datatype. Each target is the
4307    /// path of an existing *dataset* and a [`Selection`] over it, which must
4308    /// fit that dataset's extent. What reaches the file is a global-heap blob
4309    /// holding the target's object header address (assigned when the file is
4310    /// finalized) and the serialized selection, and an element carrying the
4311    /// blob's id and its byte count. Elements left unwritten read back as null
4312    /// references.
4313    pub fn write_std_region_references(&self, targets: &[(&str, Selection)]) -> Result<()> {
4314        let targets: Vec<(&str, ReferenceTarget)> = targets
4315            .iter()
4316            .map(|(path, selection)| (*path, ReferenceTarget::Region(selection.clone())))
4317            .collect();
4318        self.write_revised_references(&targets)
4319    }
4320
4321    /// Write attribute references naming `targets` into elements
4322    /// `0..targets.len()` — `H5Rcreate_attr` plus `H5Dwrite` of an
4323    /// `H5T_STD_REF` dataset.
4324    ///
4325    /// Each target is the path of an existing object — a dataset, a group, or
4326    /// `/` for the root group — and the name of an attribute it already
4327    /// carries. There is no pre-1.12 form of this reference kind, so the
4328    /// dataset must have been created with
4329    /// [`attribute_references`](DatasetBuilder::attribute_references) or one of
4330    /// its two siblings. Elements left unwritten read back as null references.
4331    pub fn write_attribute_references(&self, targets: &[(&str, &str)]) -> Result<()> {
4332        let targets: Vec<(&str, ReferenceTarget)> = targets
4333            .iter()
4334            .map(|(path, name)| (*path, ReferenceTarget::Attribute((*name).to_string())))
4335            .collect();
4336        self.write_revised_references(&targets)
4337    }
4338
4339    /// Store `targets` as 1.12 reference elements, whatever mix of kinds they
4340    /// are: the one path both revised-reference writers take.
4341    fn write_revised_references(&self, targets: &[(&str, ReferenceTarget)]) -> Result<()> {
4342        match &self.info {
4343            DatasetInfo::Writer { index, .. } => {
4344                let inner = borrow_inner(&self.file_inner);
4345                match &*inner {
4346                    H5FileInner::Writer(writer) => {
4347                        writer.write_revised_references(*index, 0, targets)?;
4348                        Ok(())
4349                    }
4350                    _ => Err(Hdf5Error::InvalidState(
4351                        "file is no longer in write mode".into(),
4352                    )),
4353                }
4354            }
4355            DatasetInfo::Reader { .. } => {
4356                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
4357            }
4358        }
4359    }
4360
4361    /// Read a reference dataset's elements, each resolved to the object it
4362    /// names.
4363    ///
4364    /// Every reference kind is read: the pre-1.12 pair h5py writes —
4365    /// `Reference` (an object header address) and `RegionReference` (a heap id
4366    /// whose heap object holds the target plus a serialized selection) — and
4367    /// the 1.12 `H5T_STD_REF` trio, `H5R_OBJECT2`, `H5R_DATASET_REGION2` and
4368    /// `H5R_ATTR`. An object reference comes back as [`Reference::Object`]
4369    /// carrying the target's path, a region reference as
4370    /// [`Reference::Region`], whose [`bounds`](Reference::bounds) is the
4371    /// selection's bounding box — libhdf5's `H5Sget_select_bounds` — and an
4372    /// attribute reference as [`Reference::Attr`], which adds the attribute's
4373    /// name.
4374    ///
4375    /// A 1.12 reference written into a file other than its target's carries
4376    /// that file's name, and [`Reference::file`] reports it; the path is then
4377    /// a path inside that file, resolved by opening it under the name the
4378    /// reference carries, and `None` when nothing is there.
4379    ///
4380    /// ```no_run
4381    /// # use rust_hdf5::H5File;
4382    /// let file = H5File::open("refs.h5").unwrap();
4383    /// for r in file.dataset("refs").unwrap().read_references().unwrap() {
4384    ///     println!("{:?} {:?}", r.path(), r.bounds());
4385    /// }
4386    /// ```
4387    pub fn read_references(&self) -> Result<Vec<Reference>> {
4388        match &self.info {
4389            DatasetInfo::Reader { name, .. } => {
4390                let mut inner = borrow_inner_mut(&self.file_inner);
4391                match &mut *inner {
4392                    H5FileInner::Reader(reader) => Ok(reader.read_references(name)?),
4393                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4394                }
4395            }
4396            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4397                "cannot read references from a dataset in write mode".into(),
4398            )),
4399        }
4400    }
4401
4402    /// Read a string dataset, fixed-width or variable-length, as one `String`
4403    /// per element.
4404    ///
4405    /// The width of a `FixedString` dataset is whatever the file says, so a
4406    /// 24-byte label column and a 100-byte one are read by the same call. The
4407    /// padding rule the datatype declares decides where each element ends —
4408    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
4409    /// character set decides how the remaining bytes are decoded: ASCII (0)
4410    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
4411    /// violates either is an error naming the element, not a silent
4412    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
4413    /// call that accepts such a file, replacing what it cannot decode.
4414    ///
4415    /// ```no_run
4416    /// # use rust_hdf5::H5File;
4417    /// let file = H5File::open("labels.h5").unwrap();
4418    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
4419    /// ```
4420    pub fn read_strings(&self) -> Result<Vec<String>> {
4421        self.read_strings_inner(false)
4422    }
4423
4424    /// [`read_strings`](Self::read_strings), but bytes that do not decode
4425    /// under the dataset's character set become U+FFFD instead of an error.
4426    ///
4427    /// Producers do mislabel the character set — a file that declares ASCII
4428    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
4429    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
4430        self.read_strings_inner(true)
4431    }
4432
4433    /// The single owner of string decoding for both string datatypes: the
4434    /// element bytes are found differently, the padding and character-set
4435    /// rules that turn them into a `String` are the same.
4436    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
4437        if matches!(self.info, DatasetInfo::Writer { .. }) {
4438            return Err(Hdf5Error::InvalidState(
4439                "cannot read strings from a dataset in write mode".into(),
4440            ));
4441        }
4442        match self.datatype()? {
4443            DatatypeMessage::VarLenString { charset, .. } => self
4444                .read_vlen_bytes()?
4445                .iter()
4446                .enumerate()
4447                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
4448                .collect(),
4449            DatatypeMessage::FixedString {
4450                size,
4451                padding,
4452                charset,
4453            } => {
4454                let width = size as usize;
4455                if width == 0 {
4456                    // A corrupt file can declare it; `chunks_exact(0)` panics.
4457                    return Err(Hdf5Error::InvalidState(
4458                        "fixed-string datatype has zero width".into(),
4459                    ));
4460                }
4461                // `read_raw_bytes` returns `product(dims) * width` bytes, so
4462                // `chunks_exact` leaves no remainder.
4463                let raw = self.read_raw_bytes()?;
4464                raw.chunks_exact(width)
4465                    .enumerate()
4466                    .map(|(i, elem)| {
4467                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
4468                    })
4469                    .collect()
4470            }
4471            other => Err(Hdf5Error::InvalidState(format!(
4472                "read_strings is only for string datasets, this one is {other:?}"
4473            ))),
4474        }
4475    }
4476
4477    /// Read the entire dataset as a typed vector.
4478    ///
4479    /// The raw bytes are read from the file and reinterpreted as `T`. The
4480    /// caller must ensure that `T` matches the datatype used when the dataset
4481    /// was written.
4482    ///
4483    /// # Errors
4484    ///
4485    /// Returns an error if:
4486    /// - The file is in write mode.
4487    /// - The raw data size is not a multiple of `T::element_size()`.
4488    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
4489        match &self.info {
4490            DatasetInfo::Reader {
4491                name, element_size, ..
4492            } => {
4493                if T::element_size() != *element_size {
4494                    return Err(Hdf5Error::TypeMismatch(format!(
4495                        "read type has element size {} but dataset has element size {}",
4496                        T::element_size(),
4497                        element_size,
4498                    )));
4499                }
4500
4501                let datatype = self.datatype()?;
4502                let mut inner = borrow_inner_mut(&self.file_inner);
4503                let H5FileInner::Reader(reader) = &mut *inner else {
4504                    return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4505                };
4506                let total = reader.dataset_raw_size(name)? as usize;
4507                if !total.is_multiple_of(T::element_size()) {
4508                    return Err(Hdf5Error::TypeMismatch(format!(
4509                        "raw data size {total} is not a multiple of element size {}",
4510                        T::element_size(),
4511                    )));
4512                }
4513
4514                // The image is read into the vector this returns, so the
4515                // bytes are touched once rather than being zeroed, read, and
4516                // then copied into a second buffer of the same size.
4517                read_image_into_new(total / T::element_size(), |image| {
4518                    reader.read_dataset_raw_into_dst(name, image, ReadDst::Fresh)?;
4519                    to_host_byte_order(image, &datatype, T::element_size())
4520                })
4521            }
4522            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4523                "cannot read from a dataset in write mode".into(),
4524            )),
4525        }
4526    }
4527
4528    /// View the entire dataset as `&[T]` pointing straight into the file's
4529    /// memory map — no read, no copy, no allocation.
4530    ///
4531    /// The returned [`MappedView<T>`](crate::MappedView) dereferences to
4532    /// `&[T]` holding exactly what [`read_raw`](Self::read_raw) would have
4533    /// returned, bit for bit.
4534    ///
4535    /// # When it works
4536    ///
4537    /// The file must be open read-only and mapped (which a read-only open
4538    /// does whenever the OS allows), the dataset's raw data must be one
4539    /// contiguous stretch of that file, and its stored elements must already
4540    /// be the host image of a `T` — same width, host byte order, significant
4541    /// bits filling the element. That is the ordinary case for an
4542    /// uncompressed, non-chunked numeric dataset written by either this crate
4543    /// or libhdf5.
4544    ///
4545    /// # When it refuses
4546    ///
4547    /// Zero-copy is a contract, not an optimization: when the bytes cannot be
4548    /// handed over as they lie, this returns
4549    /// [`Hdf5Error::NotViewable`] naming the
4550    /// reason and never quietly falls back to copying. Every
4551    /// [`ViewRefusal`](crate::ViewRefusal) is a case where
4552    /// [`read_raw`](Self::read_raw) still works: the file is not mapped (an
4553    /// open that holds no shared lock never maps), the layout is chunked,
4554    /// compact, virtual, or external, no storage is
4555    /// allocated (the dataset reads as its fill value), `T` is the wrong
4556    /// width, the stored elements need a byte-order swap or bit unpacking,
4557    /// the data lands at an offset `T`'s alignment does not permit, or the
4558    /// image runs past the end of the map.
4559    ///
4560    /// # Snapshot semantics
4561    ///
4562    /// The view owns a share of the map rather than borrowing the file
4563    /// handle, so it stays readable after the dataset and the file are
4564    /// dropped, and after a SWMR refresh has retaken the map — a live view
4565    /// keeps showing the file as it was when *its* map was taken, while the
4566    /// refreshed handle reads the new one. Nothing about a view is
4567    /// invalidated by anything this process does. The share carries the
4568    /// shared file lock the map was taken under, so for as long as any view
4569    /// is alive a writer that honours locks cannot open the file, whether or
4570    /// not the reader that took the map is still open.
4571    ///
4572    /// # Truncation
4573    ///
4574    /// The pages are the file's own. Another process writing the file in
4575    /// place is seen through the view, and one *truncating* it under the map
4576    /// faults with `SIGBUS` on the pages that went away. The shared lock the
4577    /// view keeps is what stands between the map and such a writer; one that
4578    /// waives locks ([`FileLocking::Disabled`](crate::FileLocking::Disabled),
4579    /// or a filesystem without them) is outside what any guard inside this
4580    /// process can see.
4581    ///
4582    /// ```no_run
4583    /// # use rust_hdf5::H5File;
4584    /// let file = H5File::open("data.h5")?;
4585    /// let ds = file.dataset("matrix")?;
4586    /// let view = ds.read_mapped::<f64>()?;
4587    /// let total: f64 = view.iter().sum();
4588    /// # Ok::<(), rust_hdf5::Hdf5Error>(())
4589    /// ```
4590    #[cfg(feature = "mmap")]
4591    pub fn read_mapped<T: H5Type>(&self) -> Result<crate::mapped::MappedView<T>> {
4592        self.mapped_view(crate::mapped::ViewRange::Whole)
4593    }
4594
4595    /// View a contiguous sub-range of the dataset as `&[T]` pointing straight
4596    /// into the file's memory map.
4597    ///
4598    /// `starts` and `counts` name the same N-dimensional selection
4599    /// [`read_slice`](Self::read_slice) takes, and the view holds exactly what
4600    /// that call would have returned — but only when the selection is one
4601    /// contiguous run of the stored image: a trailing group of dimensions
4602    /// taken whole, the dimension before it taken as one span, and a single
4603    /// index along every dimension before that. Anything else steps over
4604    /// elements a single slice cannot skip, and is refused with
4605    /// [`ViewRefusal::Range`](crate::ViewRefusal::Range) rather than gathered
4606    /// into a copy.
4607    ///
4608    /// Everything [`read_mapped`](Self::read_mapped) documents about when a
4609    /// dataset can be viewed, snapshot semantics, and truncation applies here
4610    /// unchanged.
4611    #[cfg(feature = "mmap")]
4612    pub fn read_mapped_slice<T: H5Type>(
4613        &self,
4614        starts: &[usize],
4615        counts: &[usize],
4616    ) -> Result<crate::mapped::MappedView<T>> {
4617        self.mapped_view(crate::mapped::ViewRange::Slab { starts, counts })
4618    }
4619
4620    /// The one route from a dataset handle to the file's map: ask the reader
4621    /// that owns the dataset for the facts, and hand them to
4622    /// [`crate::mapped::view`], which is the only thing that can turn them
4623    /// into a view.
4624    #[cfg(feature = "mmap")]
4625    fn mapped_view<T: H5Type>(
4626        &self,
4627        range: crate::mapped::ViewRange<'_>,
4628    ) -> Result<crate::mapped::MappedView<T>> {
4629        let DatasetInfo::Reader { name, .. } = &self.info else {
4630            return Err(Hdf5Error::InvalidState(
4631                "cannot read from a dataset in write mode".into(),
4632            ));
4633        };
4634        let mut inner = borrow_inner_mut(&self.file_inner);
4635        let H5FileInner::Reader(reader) = &mut *inner else {
4636            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
4637        };
4638        let src = reader.dataset_view_source(name)?;
4639        crate::mapped::view::<T>(&src, range).map_err(Hdf5Error::NotViewable)
4640    }
4641
4642    /// Read the raw byte image of a dataset without an `H5Type` carrier.
4643    ///
4644    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
4645    /// the element bytes verbatim regardless of the on-disk element type, so a
4646    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
4647    /// no matching Rust primitive can be read back and decoded by the caller.
4648    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
4649        match &self.info {
4650            DatasetInfo::Reader { name, .. } => {
4651                let mut inner = borrow_inner_mut(&self.file_inner);
4652                match &mut *inner {
4653                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
4654                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
4655                }
4656            }
4657            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4658                "cannot read from a dataset in write mode".into(),
4659            )),
4660        }
4661    }
4662
4663    /// Read a numeric dataset as `T`, converting each element from the
4664    /// on-disk datatype.
4665    ///
4666    /// Unlike [`read_raw`](Self::read_raw), which requires `T`'s size to match
4667    /// the stored element size exactly, this inspects the dataset's datatype
4668    /// message — class, signedness, byte order, width — and converts per
4669    /// element:
4670    ///
4671    /// - integer → integer: checked; a stored value that does not fit in `T`
4672    ///   is an error naming the element index and value, never a silent wrap.
4673    /// - `f32` source → `f64`: exact widening.
4674    /// - `f64` source → `f32`, float → integer, and integer → float are
4675    ///   rejected as [`TypeMismatch`](Hdf5Error::TypeMismatch).
4676    ///
4677    /// Big-endian sources are decoded according to the datatype's byte order,
4678    /// which [`read_raw`](Self::read_raw)'s size-only check would misread.
4679    ///
4680    /// ```no_run
4681    /// # use rust_hdf5::H5File;
4682    /// let file = H5File::open("data.h5").unwrap();
4683    /// let ds = file.dataset("counts").unwrap(); // stored as e.g. i16
4684    /// let counts = ds.read_numeric_as::<i64>().unwrap();
4685    /// ```
4686    pub fn read_numeric_as<T: ReadNumeric>(&self) -> Result<Vec<T>> {
4687        match &self.info {
4688            DatasetInfo::Reader { name, .. } => {
4689                let (kind, raw) = {
4690                    let mut inner = borrow_inner_mut(&self.file_inner);
4691                    match &mut *inner {
4692                        H5FileInner::Reader(reader) => {
4693                            let info = reader
4694                                .dataset_info(name)
4695                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4696                            let kind = numeric::classify(&info.datatype)?;
4697                            (kind, reader.read_dataset_raw(name)?)
4698                        }
4699                        _ => {
4700                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4701                        }
4702                    }
4703                };
4704                numeric::convert(kind, &raw)
4705            }
4706            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4707                "cannot read from a dataset in write mode".into(),
4708            )),
4709        }
4710    }
4711
4712    /// Read a slice (hyperslab) of a numeric dataset as `T`, with the same
4713    /// per-element datatype conversion as
4714    /// [`read_numeric_as`](Self::read_numeric_as).
4715    ///
4716    /// `starts` and `counts` define the N-dimensional selection exactly as in
4717    /// [`read_slice`](Self::read_slice).
4718    pub fn read_numeric_slice_as<T: ReadNumeric>(
4719        &self,
4720        starts: &[usize],
4721        counts: &[usize],
4722    ) -> Result<Vec<T>> {
4723        match &self.info {
4724            DatasetInfo::Reader { name, .. } => {
4725                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4726                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4727                let (kind, raw) = {
4728                    let mut inner = borrow_inner_mut(&self.file_inner);
4729                    match &mut *inner {
4730                        H5FileInner::Reader(reader) => {
4731                            let info = reader
4732                                .dataset_info(name)
4733                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
4734                            let kind = numeric::classify(&info.datatype)?;
4735                            (kind, reader.read_slice(name, &starts_u64, &counts_u64)?)
4736                        }
4737                        _ => {
4738                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4739                        }
4740                    }
4741                };
4742                numeric::convert(kind, &raw)
4743            }
4744            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4745                "cannot read_slice from a dataset in write mode".into(),
4746            )),
4747        }
4748    }
4749
4750    /// Read the whole dataset into a caller-provided buffer, with no allocation.
4751    ///
4752    /// `out` must have exactly `product(dims)` elements (the dataset's element
4753    /// count) and `T::element_size()` must match the dataset's on-disk element
4754    /// size, otherwise an error is returned and `out` is left unspecified. The
4755    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
4756    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
4757    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
4758    /// without the extra staging copy a `read_raw` + copy-into-pinned would
4759    /// incur. Works for every layout (contiguous, compact, and chunked under
4760    /// any index); for chunked data each decoded chunk is scattered directly
4761    /// into `out`.
4762    ///
4763    /// ```no_run
4764    /// # use rust_hdf5::H5File;
4765    /// let file = H5File::open("data.h5").unwrap();
4766    /// let ds = file.dataset("frames").unwrap();
4767    /// let n: usize = ds.shape().iter().product();
4768    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
4769    /// ds.read_raw_into(&mut buf).unwrap();
4770    /// ```
4771    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
4772        match &self.info {
4773            DatasetInfo::Reader {
4774                name, element_size, ..
4775            } => {
4776                if T::element_size() != *element_size {
4777                    return Err(Hdf5Error::TypeMismatch(format!(
4778                        "read type has element size {} but dataset has element size {}",
4779                        T::element_size(),
4780                        element_size,
4781                    )));
4782                }
4783                let datatype = self.datatype()?;
4784                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
4785                // byte representation; every bit pattern the read writes is a
4786                // valid `T`. The byte view borrows `out` exclusively for this
4787                // call, and `out.len() * element_size` cannot overflow because
4788                // it is the byte length of an existing slice (<= isize::MAX).
4789                let byte_len = out.len() * T::element_size();
4790                let bytes = unsafe {
4791                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4792                };
4793                {
4794                    let mut inner = borrow_inner_mut(&self.file_inner);
4795                    match &mut *inner {
4796                        H5FileInner::Reader(reader) => reader.read_dataset_raw_into(name, bytes)?,
4797                        _ => {
4798                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4799                        }
4800                    }
4801                }
4802                to_host_byte_order(bytes, &datatype, T::element_size())
4803            }
4804            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4805                "cannot read from a dataset in write mode".into(),
4806            )),
4807        }
4808    }
4809
4810    /// Read a hyperslab into a caller-provided buffer, with no allocation.
4811    ///
4812    /// `out` must have exactly `product(counts)` elements and
4813    /// `T::element_size()` must match the dataset's element size. The zero-copy
4814    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
4815    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
4816    /// selection are read, and the selected bytes land directly in `out` — the
4817    /// entry point for reading one frame / block straight into a pinned host
4818    /// buffer for an H2D transfer.
4819    ///
4820    /// ```no_run
4821    /// # use rust_hdf5::H5File;
4822    /// let file = H5File::open("vol.h5").unwrap();
4823    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
4824    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
4825    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
4826    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
4827    /// ```
4828    pub fn read_slice_into<T: H5Type>(
4829        &self,
4830        out: &mut [T],
4831        starts: &[usize],
4832        counts: &[usize],
4833    ) -> Result<()> {
4834        match &self.info {
4835            DatasetInfo::Reader {
4836                name, element_size, ..
4837            } => {
4838                if T::element_size() != *element_size {
4839                    return Err(Hdf5Error::TypeMismatch(format!(
4840                        "read type has element size {} but dataset has element size {}",
4841                        T::element_size(),
4842                        element_size,
4843                    )));
4844                }
4845                let datatype = self.datatype()?;
4846                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
4847                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
4848                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
4849                // borrow of `out`, byte length within bounds.
4850                let byte_len = out.len() * T::element_size();
4851                let bytes = unsafe {
4852                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
4853                };
4854                {
4855                    let mut inner = borrow_inner_mut(&self.file_inner);
4856                    match &mut *inner {
4857                        H5FileInner::Reader(reader) => {
4858                            reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?
4859                        }
4860                        _ => {
4861                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
4862                        }
4863                    }
4864                }
4865                to_host_byte_order(bytes, &datatype, T::element_size())
4866            }
4867            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
4868                "cannot read from a dataset in write mode".into(),
4869            )),
4870        }
4871    }
4872}
4873
4874// ---------------------------------------------------------------------------
4875// Datatype-aware numeric conversion (read_numeric_as)
4876// ---------------------------------------------------------------------------
4877
4878/// Marker trait for the Rust types [`H5Dataset::read_numeric_as`] can convert
4879/// into: the integer primitives (checked, never wrapping) plus `f32`/`f64`
4880/// (widening only).
4881///
4882/// Sealed — the conversion policy is part of the library contract, so the
4883/// trait cannot be implemented outside this crate.
4884pub trait ReadNumeric: numeric::Sealed {}
4885impl<T: numeric::Sealed> ReadNumeric for T {}
4886
4887pub(crate) mod numeric {
4888    //! Per-element decode + checked conversion for `read_numeric_as` (and the
4889    //! attribute counterpart `H5Attribute::read_numeric_as`).
4890
4891    use crate::error::{Hdf5Error, Result};
4892    use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, IeeeFormat};
4893
4894    /// The on-disk element shape `classify` accepted.
4895    #[derive(Clone, Copy)]
4896    pub enum SourceKind {
4897        Int {
4898            size: usize,
4899            signed: bool,
4900            byte_order: ByteOrder,
4901        },
4902        F16(ByteOrder),
4903        F32(ByteOrder),
4904        F64(ByteOrder),
4905    }
4906
4907    impl SourceKind {
4908        fn element_size(self) -> usize {
4909            match self {
4910                SourceKind::Int { size, .. } => size,
4911                SourceKind::F16(_) => 2,
4912                SourceKind::F32(_) => 4,
4913                SourceKind::F64(_) => 8,
4914            }
4915        }
4916    }
4917
4918    /// Widen an IEEE 754 binary16 bit pattern to `f32`, which represents every
4919    /// half — including subnormals, infinities and NaN payloads — exactly.
4920    ///
4921    /// Rust has no stable `f16` to convert through.
4922    fn f16_bits_to_f32(bits: u16) -> f32 {
4923        let sign = u32::from(bits >> 15);
4924        let exponent = u32::from((bits >> 10) & 0x1f);
4925        let mantissa = u32::from(bits & 0x03ff);
4926        if exponent == 0 {
4927            // Zero and subnormals: the value is mantissa * 2^-24, which is a
4928            // normal f32 for every mantissa, so the multiply is exact. Going
4929            // through the sign separately keeps -0.0.
4930            let magnitude = mantissa as f32 * (1.0 / 16_777_216.0);
4931            return if sign == 1 { -magnitude } else { magnitude };
4932        }
4933        let out = if exponent == 0x1f {
4934            // Infinity and NaN; shifting the mantissa maps the quiet bit onto
4935            // f32's quiet bit and preserves the rest of the payload.
4936            (sign << 31) | 0x7f80_0000 | (mantissa << 13)
4937        } else {
4938            // Normal: rebias the exponent (127 - 15) and left-align the
4939            // mantissa.
4940            (sign << 31) | ((exponent + 112) << 23) | (mantissa << 13)
4941        };
4942        f32::from_bits(out)
4943    }
4944
4945    /// Map a datatype message to a supported numeric source shape.
4946    ///
4947    /// Accepts standard-width integers (1/2/4/8 bytes, full precision, zero
4948    /// bit offset) and IEEE binary32/binary64 floats; everything else is a
4949    /// `TypeMismatch` naming what was found.
4950    pub fn classify(dt: &DatatypeMessage) -> Result<SourceKind> {
4951        match *dt {
4952            DatatypeMessage::FixedPoint {
4953                size,
4954                byte_order,
4955                signed,
4956                bit_offset,
4957                bit_precision,
4958            } => {
4959                if !matches!(size, 1 | 2 | 4 | 8)
4960                    || bit_offset != 0
4961                    || u32::from(bit_precision) != size * 8
4962                {
4963                    return Err(Hdf5Error::TypeMismatch(format!(
4964                        "fixed-point datatype (size {size}, bit offset {bit_offset}, \
4965                         precision {bit_precision}) is not a standard-width integer",
4966                    )));
4967                }
4968                Ok(SourceKind::Int {
4969                    size: size as usize,
4970                    signed,
4971                    byte_order,
4972                })
4973            }
4974            DatatypeMessage::BitField {
4975                size,
4976                byte_order,
4977                bit_offset,
4978                bit_precision,
4979            } => {
4980                // A bit field has no signed form; a full-width one is the
4981                // unsigned integer of the stored width. A narrower one would
4982                // need a shift-and-mask conversion this path does not model.
4983                if !matches!(size, 1 | 2 | 4 | 8)
4984                    || bit_offset != 0
4985                    || u32::from(bit_precision) != size * 8
4986                {
4987                    return Err(Hdf5Error::TypeMismatch(format!(
4988                        "bit-field datatype (size {size}, bit offset {bit_offset}, \
4989                         precision {bit_precision}) is not a whole-width bit field",
4990                    )));
4991                }
4992                Ok(SourceKind::Int {
4993                    size: size as usize,
4994                    signed: false,
4995                    byte_order,
4996                })
4997            }
4998            DatatypeMessage::FloatingPoint {
4999                size,
5000                byte_order,
5001                exponent_size,
5002                mantissa_size,
5003                ..
5004            } => match dt.ieee_format() {
5005                Some(IeeeFormat::Binary16) => Ok(SourceKind::F16(byte_order)),
5006                Some(IeeeFormat::Binary32) => Ok(SourceKind::F32(byte_order)),
5007                Some(IeeeFormat::Binary64) => Ok(SourceKind::F64(byte_order)),
5008                None => Err(Hdf5Error::TypeMismatch(format!(
5009                    "floating-point datatype (size {size}, exponent {exponent_size} bits, \
5010                     mantissa {mantissa_size} bits) is not an IEEE 754 interchange format",
5011                ))),
5012            },
5013            ref other => Err(Hdf5Error::TypeMismatch(format!(
5014                "dataset datatype '{other}' is not numeric",
5015            ))),
5016        }
5017    }
5018
5019    /// Decode and convert every element of `raw` into `T`.
5020    ///
5021    /// One loop per source width and lane type, so an element is a
5022    /// fixed-size load, a sign extension and a range check, and the
5023    /// `TypeMismatch` for an element that does not fit is built only when
5024    /// one is found. An empty input converts to an empty vector whatever
5025    /// the classes.
5026    pub fn convert<T: Sealed>(kind: SourceKind, raw: &[u8]) -> Result<Vec<T>> {
5027        let size = kind.element_size();
5028        if !raw.len().is_multiple_of(size) {
5029            return Err(Hdf5Error::TypeMismatch(format!(
5030                "raw data size {} is not a multiple of element size {size}",
5031                raw.len(),
5032            )));
5033        }
5034        let mut out = vec![T::default(); raw.len() / size];
5035        match kind {
5036            SourceKind::Int {
5037                size,
5038                signed,
5039                byte_order,
5040            } => {
5041                let le = byte_order == ByteOrder::LittleEndian;
5042                match (size, signed) {
5043                    (1, true) => ints::<1, i64, T>(raw, le, &mut out),
5044                    (1, false) => ints::<1, u64, T>(raw, le, &mut out),
5045                    (2, true) => ints::<2, i64, T>(raw, le, &mut out),
5046                    (2, false) => ints::<2, u64, T>(raw, le, &mut out),
5047                    (4, true) => ints::<4, i64, T>(raw, le, &mut out),
5048                    (4, false) => ints::<4, u64, T>(raw, le, &mut out),
5049                    (8, true) => ints::<8, i64, T>(raw, le, &mut out),
5050                    (8, false) => ints::<8, u64, T>(raw, le, &mut out),
5051                    _ => unreachable!("classify admits 1-, 2-, 4- and 8-byte integers only"),
5052                }
5053            }
5054            SourceKind::F16(order) => floats::<2, T>(raw, order, &mut out, |b| {
5055                T::from_f32(f16_bits_to_f32(u16::from_le_bytes(b)))
5056            }),
5057            SourceKind::F32(order) => {
5058                floats::<4, T>(raw, order, &mut out, |b| T::from_f32(f32::from_le_bytes(b)))
5059            }
5060            SourceKind::F64(order) => {
5061                floats::<8, T>(raw, order, &mut out, |b| T::from_f64(f64::from_le_bytes(b)))
5062            }
5063        }?;
5064        Ok(out)
5065    }
5066
5067    /// The integer elements of `raw`, `N` bytes each, through lane `L`.
5068    fn ints<const N: usize, L: IntLane, T: Sealed>(
5069        raw: &[u8],
5070        le: bool,
5071        out: &mut [T],
5072    ) -> Result<()> {
5073        let (chunks, _) = raw.as_chunks::<N>();
5074        for (index, (o, &chunk)) in out.iter_mut().zip(chunks).enumerate() {
5075            let mut bytes = chunk;
5076            if !le {
5077                bytes.reverse();
5078            }
5079            *o = L::load(bytes).into_target(index)?;
5080        }
5081        Ok(())
5082    }
5083
5084    /// The float elements of `raw`, `N` bytes each, `decode` taking each
5085    /// one's little-endian bytes.
5086    fn floats<const N: usize, T: Sealed>(
5087        raw: &[u8],
5088        order: ByteOrder,
5089        out: &mut [T],
5090        decode: impl Fn([u8; N]) -> Result<T>,
5091    ) -> Result<()> {
5092        let (chunks, _) = raw.as_chunks::<N>();
5093        for (o, &chunk) in out.iter_mut().zip(chunks) {
5094            let mut bytes = chunk;
5095            if order == ByteOrder::BigEndian {
5096                bytes.reverse();
5097            }
5098            *o = decode(bytes)?;
5099        }
5100        Ok(())
5101    }
5102
5103    /// The lane an integer source decodes to: `i64` for a signed source,
5104    /// `u64` for an unsigned one, so every standard width — u64::MAX
5105    /// included — is held without loss.
5106    pub trait IntLane: Copy {
5107        /// The value of an element's little-endian `bytes`, `N` in 1..=8.
5108        fn load<const N: usize>(bytes: [u8; N]) -> Self;
5109        fn into_target<T: Sealed>(self, index: usize) -> Result<T>;
5110    }
5111
5112    impl IntLane for u64 {
5113        fn load<const N: usize>(bytes: [u8; N]) -> Self {
5114            let mut padded = [0u8; 8];
5115            padded[..N].copy_from_slice(&bytes);
5116            u64::from_le_bytes(padded)
5117        }
5118        fn into_target<T: Sealed>(self, index: usize) -> Result<T> {
5119            T::from_u64(self, index)
5120        }
5121    }
5122
5123    impl IntLane for i64 {
5124        fn load<const N: usize>(bytes: [u8; N]) -> Self {
5125            // Arithmetic right shift sign-extends the low `N` bytes.
5126            let shift = 64 - 8 * N as u32;
5127            ((u64::load(bytes) << shift) as i64) >> shift
5128        }
5129        fn into_target<T: Sealed>(self, index: usize) -> Result<T> {
5130            T::from_i64(self, index)
5131        }
5132    }
5133
5134    /// The sealed half of `ReadNumeric`: how one source element becomes a
5135    /// `Self`, or a `TypeMismatch` explaining why it cannot. `index` is the
5136    /// element's position, for the message.
5137    pub trait Sealed: Copy + Default {
5138        fn from_i64(v: i64, index: usize) -> Result<Self>;
5139        fn from_u64(v: u64, index: usize) -> Result<Self>;
5140        fn from_f32(v: f32) -> Result<Self>;
5141        fn from_f64(v: f64) -> Result<Self>;
5142    }
5143
5144    macro_rules! int_targets {
5145        ($($t:ty),* $(,)?) => {$(
5146            impl Sealed for $t {
5147                fn from_i64(v: i64, index: usize) -> Result<Self> {
5148                    <$t>::try_from(v).map_err(|_| int_overflow::<$t>(v, index))
5149                }
5150                fn from_u64(v: u64, index: usize) -> Result<Self> {
5151                    <$t>::try_from(v).map_err(|_| int_overflow::<$t>(v, index))
5152                }
5153                fn from_f32(_: f32) -> Result<Self> {
5154                    Err(float_as_int::<$t>())
5155                }
5156                fn from_f64(_: f64) -> Result<Self> {
5157                    Err(float_as_int::<$t>())
5158                }
5159            }
5160        )*};
5161    }
5162    int_targets!(i8, i16, i32, i64, u8, u16, u32, u64);
5163
5164    // Not in the macro: `i128::try_from` and `u128::try_from(u64)` are
5165    // infallible, which trips clippy::unnecessary_fallible_conversions.
5166    impl Sealed for i128 {
5167        fn from_i64(v: i64, _index: usize) -> Result<Self> {
5168            Ok(Self::from(v))
5169        }
5170        fn from_u64(v: u64, _index: usize) -> Result<Self> {
5171            Ok(Self::from(v))
5172        }
5173        fn from_f32(_: f32) -> Result<Self> {
5174            Err(float_as_int::<i128>())
5175        }
5176        fn from_f64(_: f64) -> Result<Self> {
5177            Err(float_as_int::<i128>())
5178        }
5179    }
5180
5181    impl Sealed for u128 {
5182        fn from_i64(v: i64, index: usize) -> Result<Self> {
5183            Self::try_from(v).map_err(|_| int_overflow::<u128>(v, index))
5184        }
5185        fn from_u64(v: u64, _index: usize) -> Result<Self> {
5186            Ok(Self::from(v))
5187        }
5188        fn from_f32(_: f32) -> Result<Self> {
5189            Err(float_as_int::<u128>())
5190        }
5191        fn from_f64(_: f64) -> Result<Self> {
5192            Err(float_as_int::<u128>())
5193        }
5194    }
5195
5196    fn int_overflow<T>(v: impl std::fmt::Display, index: usize) -> Hdf5Error {
5197        Hdf5Error::TypeMismatch(format!(
5198            "value {v} at element {index} does not fit in {}",
5199            std::any::type_name::<T>()
5200        ))
5201    }
5202
5203    fn float_as_int<T>() -> Hdf5Error {
5204        Hdf5Error::TypeMismatch(format!(
5205            "cannot read a floating-point dataset as {}; read as f64 and convert explicitly",
5206            std::any::type_name::<T>()
5207        ))
5208    }
5209
5210    impl Sealed for f32 {
5211        fn from_i64(_: i64, _index: usize) -> Result<Self> {
5212            Err(int_as_f32())
5213        }
5214        fn from_u64(_: u64, _index: usize) -> Result<Self> {
5215            Err(int_as_f32())
5216        }
5217        fn from_f32(v: f32) -> Result<Self> {
5218            Ok(v)
5219        }
5220        fn from_f64(_: f64) -> Result<Self> {
5221            Err(Hdf5Error::TypeMismatch(
5222                "narrowing an f64 dataset to f32 loses precision; read as f64".into(),
5223            ))
5224        }
5225    }
5226
5227    fn int_as_f32() -> Hdf5Error {
5228        Hdf5Error::TypeMismatch(
5229            "cannot read an integer dataset as f32; read as an integer type and convert \
5230             explicitly"
5231                .into(),
5232        )
5233    }
5234
5235    impl Sealed for f64 {
5236        fn from_i64(_: i64, _index: usize) -> Result<Self> {
5237            Err(int_as_f64())
5238        }
5239        fn from_u64(_: u64, _index: usize) -> Result<Self> {
5240            Err(int_as_f64())
5241        }
5242        // Every f32 is exactly representable as f64.
5243        fn from_f32(v: f32) -> Result<Self> {
5244            Ok(f64::from(v))
5245        }
5246        fn from_f64(v: f64) -> Result<Self> {
5247            Ok(v)
5248        }
5249    }
5250
5251    fn int_as_f64() -> Hdf5Error {
5252        Hdf5Error::TypeMismatch(
5253            "cannot read an integer dataset as f64; integers above 2^53 lose precision — \
5254             read as an integer type and convert explicitly"
5255                .into(),
5256        )
5257    }
5258
5259    #[cfg(test)]
5260    mod tests {
5261        use super::*;
5262
5263        /// Every binary16 bit pattern class widens to the f32 with the same
5264        /// value: zeros keep their sign, subnormals stay exact, infinities and
5265        /// NaN payloads survive.
5266        #[test]
5267        fn f16_widening_is_exact() {
5268            let cases: [(u16, f32); 10] = [
5269                (0x0000, 0.0),
5270                (0x3c00, 1.0),
5271                (0xc000, -2.0),
5272                (0x3555, 0.333_251_95),   // nearest half to 1/3
5273                (0x0001, 5.960_464_5e-8), // smallest subnormal, 2^-24
5274                (0x03ff, 6.097_555e-5),   // largest subnormal
5275                (0x0400, 6.103_515_6e-5), // smallest normal
5276                (0x7bff, 65504.0),        // largest finite
5277                (0x7c00, f32::INFINITY),
5278                (0xfc00, f32::NEG_INFINITY),
5279            ];
5280            for (bits, expected) in cases {
5281                let got = f16_bits_to_f32(bits);
5282                assert_eq!(got, expected, "0x{bits:04x} widened to {got}");
5283            }
5284
5285            let neg_zero = f16_bits_to_f32(0x8000);
5286            assert_eq!(neg_zero, 0.0);
5287            assert!(neg_zero.is_sign_negative(), "-0.0 lost its sign");
5288
5289            let nan = f16_bits_to_f32(0x7e01);
5290            assert!(nan.is_nan());
5291            // The quiet bit and the payload land in f32's mantissa.
5292            assert_eq!(nan.to_bits(), 0x7fc0_2000);
5293        }
5294
5295        #[test]
5296        fn f16_source_converts_and_honors_byte_order() {
5297            let kind = classify(&DatatypeMessage::f16_type()).unwrap();
5298            // 1.0, -2.0, 0.333..., 65504
5299            let raw = [0x00, 0x3c, 0x00, 0xc0, 0x55, 0x35, 0xff, 0x7b];
5300            assert_eq!(
5301                convert::<f32>(kind, &raw).unwrap(),
5302                vec![1.0, -2.0, 0.333_251_95, 65504.0]
5303            );
5304            assert_eq!(
5305                convert::<f64>(kind, &raw).unwrap(),
5306                vec![1.0, -2.0, 0.333_251_953_125, 65504.0]
5307            );
5308
5309            let DatatypeMessage::FloatingPoint { .. } = DatatypeMessage::f16_type() else {
5310                unreachable!()
5311            };
5312            let mut be = DatatypeMessage::f16_type();
5313            if let DatatypeMessage::FloatingPoint { byte_order, .. } = &mut be {
5314                *byte_order = ByteOrder::BigEndian;
5315            }
5316            let be_kind = classify(&be).unwrap();
5317            assert_eq!(convert::<f32>(be_kind, &[0x3c, 0x00]).unwrap(), vec![1.0]);
5318        }
5319
5320        /// A float whose layout is not an interchange format is refused, not
5321        /// reinterpreted.
5322        #[test]
5323        fn non_ieee_float_is_refused() {
5324            let mut odd = DatatypeMessage::f32_type();
5325            if let DatatypeMessage::FloatingPoint { exponent_bias, .. } = &mut odd {
5326                *exponent_bias = 63;
5327            }
5328            assert!(odd.ieee_format().is_none());
5329            let err = classify(&odd).err().expect("non-IEEE float was accepted");
5330            assert!(
5331                err.to_string().contains("IEEE 754 interchange format"),
5332                "unexpected error: {err}"
5333            );
5334        }
5335    }
5336}
5337
5338#[cfg(test)]
5339mod tests {
5340    use crate::H5File;
5341    use std::path::PathBuf;
5342
5343    fn temp_path(name: &str) -> PathBuf {
5344        // Include PID + a per-call atomic counter so that concurrent
5345        // cargo invocations and any kernel-level "lock not yet
5346        // released" races between sequential opens cannot collide.
5347        use std::sync::atomic::{AtomicU64, Ordering};
5348        static COUNTER: AtomicU64 = AtomicU64::new(0);
5349        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
5350        std::env::temp_dir().join(format!(
5351            "hdf5_dataset_test_{}_{}_{}.h5",
5352            name,
5353            std::process::id(),
5354            n
5355        ))
5356    }
5357
5358    #[test]
5359    fn runtime_compound_via_datatype_override_and_raw_bytes() {
5360        use crate::format::messages::datatype::DatatypeMessage;
5361        use crate::types::{CompoundType, H5Type};
5362
5363        let path = temp_path("compound_raw");
5364        // A 12-byte packed compound with NO matching Rust primitive carrier,
5365        // so it can only be written through the datatype() override +
5366        // write_raw_bytes path (the runtime-CompoundType use case).
5367        let ct = CompoundType {
5368            members: vec![
5369                ("id".to_string(), i32::hdf5_type(), 0),
5370                ("val".to_string(), f64::hdf5_type(), 4),
5371            ],
5372            total_size: 12,
5373        };
5374        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
5375        let mut bytes = Vec::new();
5376        for (id, val) in recs {
5377            bytes.extend_from_slice(&id.to_le_bytes());
5378            bytes.extend_from_slice(&val.to_le_bytes());
5379        }
5380
5381        {
5382            let file = H5File::create(&path).unwrap();
5383            let ds = file
5384                .new_dataset::<u8>()
5385                .datatype(ct.to_datatype())
5386                .shape([recs.len()])
5387                .create("records")
5388                .unwrap();
5389            ds.write_raw_bytes(&bytes).unwrap();
5390            file.close().unwrap();
5391        }
5392        {
5393            let file = H5File::open(&path).unwrap();
5394            let ds = file.dataset("records").unwrap();
5395            // The on-disk element type is the compound we specified (size 12),
5396            // not the u8 carrier.
5397            match ds.datatype().unwrap() {
5398                DatatypeMessage::Compound { size, members } => {
5399                    assert_eq!(size, 12);
5400                    assert_eq!(members.len(), 2);
5401                    assert_eq!(members[0].name, "id");
5402                    assert_eq!(members[0].offset, 0);
5403                    assert_eq!(members[1].name, "val");
5404                    assert_eq!(members[1].offset, 4);
5405                }
5406                other => panic!("expected compound datatype, got {other:?}"),
5407            }
5408            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
5409        }
5410        std::fs::remove_file(&path).ok();
5411    }
5412
5413    #[test]
5414    fn builder_requires_shape() {
5415        let path = temp_path("no_shape");
5416        let file = H5File::create(&path).unwrap();
5417        let result = file.new_dataset::<u8>().create("data");
5418        assert!(result.is_err());
5419        std::fs::remove_file(&path).ok();
5420    }
5421
5422    // The last chunk along a *fixed* dimension covers more elements than the
5423    // extent has, so growing the dataspace to the chunk's far edge asks for
5424    // more than the declared maximum. Before the clamp the chunk was written
5425    // and then the call failed on that extend, leaving the bytes in the file
5426    // and the caller an error.
5427    #[test]
5428    fn a_partial_edge_chunk_does_not_grow_past_the_declared_maximum() {
5429        let path = temp_path("edge_chunk_extent");
5430        // Extensible array: dimension 1 is unlimited, dimension 0 is fixed at
5431        // 10 and not a multiple of the chunk's 4.
5432        let file = H5File::create(&path).unwrap();
5433        let ds = file
5434            .new_dataset::<i32>()
5435            .shape([10usize, 4])
5436            .max_shape(&[Some(10), None])
5437            .chunk(&[4, 4])
5438            .create("grid")
5439            .unwrap();
5440        let chunk: Vec<u8> = (0i32..16).flat_map(|v| v.to_le_bytes()).collect();
5441        // Chunk row 2 spans elements 8..12 of a dimension that stops at 10.
5442        ds.write_chunk_at(&[2, 0], &chunk).unwrap();
5443        assert_eq!(ds.shape(), vec![10, 4]);
5444        file.close().unwrap();
5445
5446        let file = H5File::open(&path).unwrap();
5447        let back = file.dataset("grid").unwrap().read_raw::<i32>().unwrap();
5448        assert_eq!(back.len(), 40);
5449        assert_eq!(&back[32..40], &[0, 1, 2, 3, 4, 5, 6, 7]);
5450        drop(file);
5451        std::fs::remove_file(&path).ok();
5452    }
5453
5454    // The cap is in `write_chunk_at_inner`, so it belongs to every chunk index
5455    // whose write reaches the extend below it — the v2 B-tree as much as the
5456    // extensible array. A rank-3 dataset with two unlimited dimensions gets
5457    // that index, and its third, fixed dimension is where the last chunk
5458    // overhangs. (The fixed array and the implicit index return before the
5459    // extend: their shape cannot grow at all. The version-1 B-tree does reach
5460    // it — `tests/legacy_append.rs` carries that case, which needs a classic
5461    // file.)
5462    #[test]
5463    fn the_edge_write_cap_holds_for_the_v2_btree_index() {
5464        let path = temp_path("edge_chunk_bt2");
5465        let file = H5File::create(&path).unwrap();
5466        let ds = file
5467            .new_dataset::<i32>()
5468            .shape([10usize, 4, 4])
5469            .max_shape(&[Some(10), None, None])
5470            .chunk(&[4, 4, 4])
5471            .create("cube")
5472            .unwrap();
5473        let chunk: Vec<u8> = (0i32..64).flat_map(|v| v.to_le_bytes()).collect();
5474        // Chunk plane 2 spans elements 8..12 of a dimension that stops at 10.
5475        ds.write_chunk_at(&[2, 0, 0], &chunk).unwrap();
5476        assert_eq!(ds.shape(), vec![10, 4, 4]);
5477        file.close().unwrap();
5478
5479        let file = H5File::open(&path).unwrap();
5480        let back = file.dataset("cube").unwrap().read_raw::<i32>().unwrap();
5481        assert_eq!(back.len(), 160);
5482        // Rows 8 and 9 of the written plane, 16 elements each.
5483        assert_eq!(&back[128..160], &(0i32..32).collect::<Vec<_>>()[..]);
5484        drop(file);
5485        std::fs::remove_file(&path).ok();
5486    }
5487
5488    // The cap guards a chunk-coordinate write, and neither an externally
5489    // stored nor a virtual dataset has chunk coordinates to guard: both are
5490    // contiguous storage classes, refused at build together with chunked
5491    // storage, and `write_chunk_at` refuses what is not chunked. So the path
5492    // the case above exercises cannot be entered for either — asserted here
5493    // rather than left to inspection, since both classes route their raw bytes
5494    // through the same writer as the chunk grid does.
5495    #[test]
5496    fn an_external_or_virtual_dataset_never_reaches_the_edge_write_cap() {
5497        use crate::Selection;
5498        let dir = std::env::temp_dir().join(format!(
5499            "rust_hdf5_edge_cap_{}_{}",
5500            std::process::id(),
5501            temp_path("x").file_name().unwrap().to_string_lossy()
5502        ));
5503        std::fs::create_dir_all(&dir).unwrap();
5504        let path = dir.join("edge_cap.h5");
5505        let file = H5File::create(&path).unwrap();
5506        let payload = dir.join("payload.raw");
5507
5508        // Chunked storage and these two are mutually exclusive at build.
5509        for (which, res) in [
5510            (
5511                "external",
5512                file.new_dataset::<i32>()
5513                    .shape([10usize])
5514                    .external(&[(payload.to_str().unwrap(), 0, 40)])
5515                    .chunk(&[4])
5516                    .create("a"),
5517            ),
5518            (
5519                "virtual",
5520                file.new_dataset::<i32>()
5521                    .shape([10usize])
5522                    .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5523                    .chunk(&[4])
5524                    .create("b"),
5525            ),
5526        ] {
5527            match res {
5528                Ok(_) => panic!("a {which} dataset cannot also be chunked"),
5529                Err(e) => assert!(e.to_string().contains("chunked"), "{which}: {e}"),
5530            }
5531        }
5532
5533        // And the coordinate write itself is refused on both, with the extent
5534        // left exactly where it was.
5535        let ext = file
5536            .new_dataset::<i32>()
5537            .shape([10usize])
5538            .external(&[(payload.to_str().unwrap(), 0, 40)])
5539            .create("outside")
5540            .unwrap();
5541        let vds = file
5542            .new_dataset::<i32>()
5543            .shape([10usize])
5544            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
5545            .create("elsewhere")
5546            .unwrap();
5547        let chunk: Vec<u8> = (0i32..4).flat_map(|v| v.to_le_bytes()).collect();
5548        for (which, ds) in [("external", &ext), ("virtual", &vds)] {
5549            let err = ds.write_chunk_at(&[2], &chunk).unwrap_err().to_string();
5550            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5551            let err = ds
5552                .write_chunk_raw_at(&[2], &chunk, 0)
5553                .unwrap_err()
5554                .to_string();
5555            assert!(err.contains("only for chunked datasets"), "{which}: {err}");
5556            assert_eq!(ds.shape(), vec![10], "{which}");
5557        }
5558        file.close().unwrap();
5559        std::fs::remove_dir_all(&dir).ok();
5560    }
5561
5562    #[test]
5563    fn write_raw_size_mismatch() {
5564        let path = temp_path("size_mismatch");
5565        let file = H5File::create(&path).unwrap();
5566        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
5567        // Provide 3 elements instead of 4
5568        let result = ds.write_raw(&[1u8, 2, 3]);
5569        assert!(result.is_err());
5570        std::fs::remove_file(&path).ok();
5571    }
5572
5573    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
5574    // dataset = one chunk) rather than silently drop the filter on the
5575    // contiguous path. write_raw then populates that single chunk.
5576    #[cfg(feature = "deflate")]
5577    #[test]
5578    fn filter_without_chunk_autochunks_and_roundtrips() {
5579        let path = temp_path("autochunk_filter");
5580        let data: Vec<i32> = (0..8).collect();
5581        {
5582            let file = H5File::create(&path).unwrap();
5583            let ds = file
5584                .new_dataset::<i32>()
5585                .deflate(6)
5586                .shape([8])
5587                .create("seq")
5588                .unwrap();
5589            ds.write_raw(&data).unwrap();
5590            file.close().unwrap();
5591        }
5592        {
5593            let file = H5File::open(&path).unwrap();
5594            let ds = file.dataset("seq").unwrap();
5595            // The filter forced chunked storage: a single whole-dataset chunk.
5596            assert!(
5597                ds.is_chunked(),
5598                "auto-chunk did not produce chunked storage"
5599            );
5600            assert_eq!(ds.chunk_dims(), Some(vec![8]));
5601            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5602        }
5603        std::fs::remove_file(&path).ok();
5604    }
5605
5606    // B: write_raw on an explicitly chunked + compressed dataset scatters the
5607    // full row-major image across a multi-chunk grid, including edge chunks
5608    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
5609    #[cfg(feature = "deflate")]
5610    #[test]
5611    fn an_edge_chunk_pads_with_zero_not_the_chunk_written_before_it() {
5612        // 4x5 over a 2x3 chunk: the second chunk of each row covers only two
5613        // of its three columns, so a third of it is padding. The image write
5614        // stages every chunk of a shape like this through one reused buffer,
5615        // and the padding has to reach the file as zero rather than as
5616        // whatever the chunk before it left in that buffer.
5617        let path = temp_path("edge_chunk_padding");
5618        let data: Vec<i32> = (1..=20).collect(); // no zeros of its own
5619        {
5620            let file = H5File::create(&path).unwrap();
5621            let ds = file
5622                .new_dataset::<i32>()
5623                .shape([4, 5])
5624                .chunk(&[2, 3])
5625                .create("grid")
5626                .unwrap();
5627            ds.write_raw(&data).unwrap();
5628            file.close().unwrap();
5629        }
5630        {
5631            let file = H5File::open(&path).unwrap();
5632            let ds = file.dataset("grid").unwrap();
5633            let as_i32 = |bytes: Vec<u8>| -> Vec<i32> {
5634                bytes
5635                    .as_chunks::<4>()
5636                    .0
5637                    .iter()
5638                    .map(|b| i32::from_le_bytes(*b))
5639                    .collect()
5640            };
5641            // The chunk that precedes each edge chunk is full, so a leak would
5642            // show as its 3rd and 6th elements (3 and 8, then 13 and 18).
5643            assert_eq!(
5644                as_i32(ds.read_chunk_raw_at(&[0, 0]).unwrap().0),
5645                [1, 2, 3, 6, 7, 8]
5646            );
5647            assert_eq!(
5648                as_i32(ds.read_chunk_raw_at(&[0, 1]).unwrap().0),
5649                [4, 5, 0, 9, 10, 0]
5650            );
5651            assert_eq!(
5652                as_i32(ds.read_chunk_raw_at(&[1, 1]).unwrap().0),
5653                [14, 15, 0, 19, 20, 0]
5654            );
5655            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5656        }
5657        std::fs::remove_file(&path).ok();
5658    }
5659
5660    #[test]
5661    #[cfg(feature = "deflate")]
5662    fn write_raw_multichunk_edge_roundtrips() {
5663        let path = temp_path("multichunk_edge");
5664        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
5665        {
5666            let file = H5File::create(&path).unwrap();
5667            let ds = file
5668                .new_dataset::<i32>()
5669                .shape([7, 5])
5670                .chunk(&[3, 2])
5671                .deflate(4)
5672                .create("grid")
5673                .unwrap();
5674            ds.write_raw(&data).unwrap();
5675            file.close().unwrap();
5676        }
5677        {
5678            let file = H5File::open(&path).unwrap();
5679            let ds = file.dataset("grid").unwrap();
5680            assert_eq!(ds.shape(), vec![7, 5]);
5681            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
5682            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5683        }
5684        std::fs::remove_file(&path).ok();
5685    }
5686
5687    // B: write_raw on an unfiltered chunked dataset (previously rejected with
5688    // "use write_chunk for chunked datasets") now gathers and round-trips.
5689    #[test]
5690    fn write_raw_unfiltered_chunked_roundtrips() {
5691        let path = temp_path("chunked_unfiltered");
5692        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
5693        {
5694            let file = H5File::create(&path).unwrap();
5695            let ds = file
5696                .new_dataset::<f64>()
5697                .shape([4, 3])
5698                .chunk(&[2, 2])
5699                .create("m")
5700                .unwrap();
5701            ds.write_raw(&data).unwrap();
5702            file.close().unwrap();
5703        }
5704        {
5705            let file = H5File::open(&path).unwrap();
5706            let ds = file.dataset("m").unwrap();
5707            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
5708            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
5709        }
5710        std::fs::remove_file(&path).ok();
5711    }
5712
5713    // write_raw on an extensible-array (unlimited first dim) compressed dataset
5714    // drives write_full_image_chunked's EA branch, which gathers chunks and
5715    // compresses them through the windowed batch path. Round-trips the full
5716    // image, including a partial edge chunk along the unlimited dimension.
5717    #[cfg(feature = "deflate")]
5718    #[test]
5719    fn write_raw_ea_compressed_roundtrips() {
5720        let path = temp_path("write_raw_ea_deflate");
5721        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
5722        {
5723            let file = H5File::create(&path).unwrap();
5724            let ds = file
5725                .new_dataset::<i32>()
5726                .shape([5, 4])
5727                .chunk(&[2, 4])
5728                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
5729                .deflate(5)
5730                .create("stream")
5731                .unwrap();
5732            ds.write_raw(&data).unwrap();
5733            file.close().unwrap();
5734        }
5735        {
5736            let file = H5File::open(&path).unwrap();
5737            let ds = file.dataset("stream").unwrap();
5738            assert_eq!(ds.shape(), vec![5, 4]);
5739            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5740        }
5741        std::fs::remove_file(&path).ok();
5742    }
5743
5744    // 3D chunked Full read with partial-edge chunks in every dimension. This
5745    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
5746    // dimensions, exercising the nested outer-coordinate carry and the
5747    // last-axis edge clamp (chunks hang off the high edge in all three axes).
5748    #[cfg(feature = "deflate")]
5749    #[test]
5750    fn read_full_3d_chunked_edge_roundtrips() {
5751        let path = temp_path("full_3d_chunked_edge");
5752        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
5753        // along each axis is partial (1, 1, and 1 element respectively).
5754        let total: usize = 5 * 4 * 3;
5755        let data: Vec<i32> = (0..total as i32).collect();
5756        {
5757            let file = H5File::create(&path).unwrap();
5758            let ds = file
5759                .new_dataset::<i32>()
5760                .shape([5, 4, 3])
5761                .chunk(&[2, 3, 2])
5762                .deflate(4)
5763                .create("vol")
5764                .unwrap();
5765            ds.write_raw(&data).unwrap();
5766            file.close().unwrap();
5767        }
5768        {
5769            let file = H5File::open(&path).unwrap();
5770            let ds = file.dataset("vol").unwrap();
5771            assert_eq!(ds.shape(), vec![5, 4, 3]);
5772            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
5773            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
5774        }
5775        std::fs::remove_file(&path).ok();
5776    }
5777
5778    #[test]
5779    fn roundtrip_u8_1d() {
5780        let path = temp_path("rt_u8_1d");
5781        let data: Vec<u8> = (0..10).collect();
5782
5783        {
5784            let file = H5File::create(&path).unwrap();
5785            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
5786            ds.write_raw(&data).unwrap();
5787            file.close().unwrap();
5788        }
5789
5790        {
5791            let file = H5File::open(&path).unwrap();
5792            let ds = file.dataset("seq").unwrap();
5793            assert_eq!(ds.shape(), vec![10]);
5794            let readback = ds.read_raw::<u8>().unwrap();
5795            assert_eq!(readback, data);
5796        }
5797
5798        std::fs::remove_file(&path).ok();
5799    }
5800
5801    #[test]
5802    fn roundtrip_i32_2d() {
5803        let path = temp_path("rt_i32_2d");
5804        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
5805
5806        {
5807            let file = H5File::create(&path).unwrap();
5808            let ds = file
5809                .new_dataset::<i32>()
5810                .shape([2, 3])
5811                .create("matrix")
5812                .unwrap();
5813            ds.write_raw(&data).unwrap();
5814            file.close().unwrap();
5815        }
5816
5817        {
5818            let file = H5File::open(&path).unwrap();
5819            let ds = file.dataset("matrix").unwrap();
5820            assert_eq!(ds.shape(), vec![2, 3]);
5821            let readback = ds.read_raw::<i32>().unwrap();
5822            assert_eq!(readback, data);
5823        }
5824
5825        std::fs::remove_file(&path).ok();
5826    }
5827
5828    #[test]
5829    fn roundtrip_f64_3d() {
5830        let path = temp_path("rt_f64_3d");
5831        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
5832
5833        {
5834            let file = H5File::create(&path).unwrap();
5835            let ds = file
5836                .new_dataset::<f64>()
5837                .shape([2, 3, 4])
5838                .create("cube")
5839                .unwrap();
5840            ds.write_raw(&data).unwrap();
5841            file.close().unwrap();
5842        }
5843
5844        {
5845            let file = H5File::open(&path).unwrap();
5846            let ds = file.dataset("cube").unwrap();
5847            assert_eq!(ds.shape(), vec![2, 3, 4]);
5848            let readback = ds.read_raw::<f64>().unwrap();
5849            assert_eq!(readback, data);
5850        }
5851
5852        std::fs::remove_file(&path).ok();
5853    }
5854
5855    #[test]
5856    fn cannot_read_in_write_mode() {
5857        let path = temp_path("no_read_write");
5858        let file = H5File::create(&path).unwrap();
5859        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5860        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5861        let result = ds.read_raw::<u8>();
5862        assert!(result.is_err());
5863        std::fs::remove_file(&path).ok();
5864    }
5865
5866    #[test]
5867    fn cannot_write_in_read_mode() {
5868        let path = temp_path("no_write_read");
5869
5870        {
5871            let file = H5File::create(&path).unwrap();
5872            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
5873            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
5874            file.close().unwrap();
5875        }
5876
5877        {
5878            let file = H5File::open(&path).unwrap();
5879            let ds = file.dataset("x").unwrap();
5880            let result = ds.write_raw(&[5u8, 6, 7, 8]);
5881            assert!(result.is_err());
5882        }
5883
5884        std::fs::remove_file(&path).ok();
5885    }
5886
5887    #[test]
5888    fn numeric_attr_roundtrip() {
5889        let path = temp_path("num_attr");
5890        {
5891            let file = H5File::create(&path).unwrap();
5892            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5893            ds.write_raw(&[1.0f32; 4]).unwrap();
5894
5895            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
5896            a1.write_numeric(&1.2345f64).unwrap();
5897
5898            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
5899            a2.write_numeric(&42i32).unwrap();
5900
5901            file.close().unwrap();
5902        }
5903        {
5904            let file = H5File::open(&path).unwrap();
5905            let ds = file.dataset("data").unwrap();
5906
5907            let scale = ds.attr("scale").unwrap();
5908            let val: f64 = scale.read_numeric().unwrap();
5909            assert!((val - 1.2345).abs() < 1e-10);
5910
5911            let count = ds.attr("count").unwrap();
5912            let val: i32 = count.read_numeric().unwrap();
5913            assert_eq!(val, 42);
5914        }
5915        std::fs::remove_file(&path).ok();
5916    }
5917
5918    #[test]
5919    fn array_attr_roundtrip() {
5920        let path = temp_path("array_attr");
5921        let offsets = [10i32, -20, 30];
5922        {
5923            let file = H5File::create(&path).unwrap();
5924            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5925            ds.write_raw(&[1.0f32; 4]).unwrap();
5926
5927            // 1-D int32 array attribute (NDArrayDimOffset-style).
5928            let a = ds
5929                .new_attr::<i32>()
5930                .shape([3])
5931                .create("dim_offset")
5932                .unwrap();
5933            a.write_array(&offsets).unwrap();
5934
5935            // Wrong element count is rejected.
5936            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
5937            assert!(bad.write_array(&[1i32, 2]).is_err());
5938
5939            file.close().unwrap();
5940        }
5941        {
5942            let file = H5File::open(&path).unwrap();
5943            let ds = file.dataset("data").unwrap();
5944            let a = ds.attr("dim_offset").unwrap();
5945            let raw = a.read_raw().unwrap();
5946            assert_eq!(raw.len(), 3 * 4);
5947            let got: Vec<i32> = raw
5948                .as_chunks::<4>()
5949                .0
5950                .iter()
5951                .map(|b| i32::from_le_bytes(*b))
5952                .collect();
5953            assert_eq!(got, offsets);
5954        }
5955        std::fs::remove_file(&path).ok();
5956    }
5957
5958    #[test]
5959    fn attr_datatype_exposes_class_and_sign() {
5960        // H5Attribute::datatype() must report the stored datatype class and
5961        // signedness so a generic attr->metadata mapper need not infer it from
5962        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
5963        use crate::format::messages::datatype::DatatypeMessage;
5964
5965        let path = temp_path("attr_datatype");
5966        {
5967            let file = H5File::create(&path).unwrap();
5968            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
5969            ds.new_attr::<f64>()
5970                .shape(())
5971                .create("scale")
5972                .unwrap()
5973                .write_numeric(&1.5f64)
5974                .unwrap();
5975            ds.new_attr::<i32>()
5976                .shape(())
5977                .create("count")
5978                .unwrap()
5979                .write_numeric(&7i32)
5980                .unwrap();
5981            file.close().unwrap();
5982        }
5983        {
5984            let file = H5File::open(&path).unwrap();
5985            let ds = file.dataset("data").unwrap();
5986
5987            match ds.attr("scale").unwrap().datatype().unwrap() {
5988                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
5989                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
5990            }
5991
5992            match ds.attr("count").unwrap().datatype().unwrap() {
5993                DatatypeMessage::FixedPoint { size, signed, .. } => {
5994                    assert_eq!(size, 4);
5995                    assert!(signed, "i32 attr must be signed");
5996                }
5997                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
5998            }
5999        }
6000        std::fs::remove_file(&path).ok();
6001    }
6002
6003    #[test]
6004    fn attr_datatype_in_write_mode_errors() {
6005        let path = temp_path("attr_datatype_write_mode");
6006        let file = H5File::create(&path).unwrap();
6007        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
6008        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
6009        assert!(attr.datatype().is_err());
6010        std::fs::remove_file(&path).ok();
6011    }
6012
6013    #[test]
6014    fn cannot_create_dataset_in_read_mode() {
6015        let path = temp_path("no_create_read");
6016
6017        {
6018            let _file = H5File::create(&path).unwrap();
6019        }
6020
6021        {
6022            let file = H5File::open(&path).unwrap();
6023            let result = file.new_dataset::<u8>().shape([4]).create("x");
6024            assert!(result.is_err());
6025        }
6026
6027        std::fs::remove_file(&path).ok();
6028    }
6029
6030    #[test]
6031    fn shape_accessor() {
6032        let path = temp_path("shape_acc");
6033
6034        let file = H5File::create(&path).unwrap();
6035        let ds = file
6036            .new_dataset::<f32>()
6037            .shape([5, 10, 3])
6038            .create("tensor")
6039            .unwrap();
6040        assert_eq!(ds.shape(), vec![5, 10, 3]);
6041
6042        std::fs::remove_file(&path).ok();
6043    }
6044
6045    #[test]
6046    fn slice_roundtrip_2d() {
6047        let path = temp_path("slice_2d");
6048
6049        // Create a 4x5 dataset, write full, then read a slice
6050        let data: Vec<i32> = (0..20).collect();
6051        {
6052            let file = H5File::create(&path).unwrap();
6053            let ds = file
6054                .new_dataset::<i32>()
6055                .shape([4, 5])
6056                .create("mat")
6057                .unwrap();
6058            ds.write_raw(&data).unwrap();
6059            file.close().unwrap();
6060        }
6061        {
6062            let file = H5File::open(&path).unwrap();
6063            let ds = file.dataset("mat").unwrap();
6064            // Read rows 1..3, cols 2..4 (2x2 slice)
6065            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
6066            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
6067            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
6068            assert_eq!(slice, vec![7, 8, 12, 13]);
6069        }
6070
6071        std::fs::remove_file(&path).ok();
6072    }
6073
6074    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
6075    // caller-provided buffer and MUST produce byte-for-byte the same data as
6076    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
6077    // creatable layout, since both now share one buffer-filling core.
6078    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
6079    where
6080        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
6081    {
6082        let n: usize = ds.shape().iter().product();
6083        let want_full = ds.read_raw::<T>().unwrap();
6084        let mut got_full = vec![T::default(); n];
6085        ds.read_raw_into::<T>(&mut got_full).unwrap();
6086        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
6087
6088        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
6089        let sn: usize = counts.iter().product();
6090        let mut got_slice = vec![T::default(); sn];
6091        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
6092            .unwrap();
6093        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
6094    }
6095
6096    #[test]
6097    fn read_into_matches_vec_contiguous() {
6098        let path = temp_path("into_contig");
6099        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
6100        {
6101            let file = H5File::create(&path).unwrap();
6102            let ds = file
6103                .new_dataset::<i32>()
6104                .shape([4, 5])
6105                .create("mat")
6106                .unwrap();
6107            ds.write_raw(&data).unwrap();
6108            file.close().unwrap();
6109        }
6110        {
6111            let file = H5File::open(&path).unwrap();
6112            let ds = file.dataset("mat").unwrap();
6113            assert_eq!(ds.chunk_dims(), None);
6114            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
6115        }
6116        std::fs::remove_file(&path).ok();
6117    }
6118
6119    #[test]
6120    fn read_into_matches_vec_chunked_unfiltered() {
6121        let path = temp_path("into_chunk");
6122        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
6123        {
6124            let file = H5File::create(&path).unwrap();
6125            let ds = file
6126                .new_dataset::<f64>()
6127                .shape([7, 5])
6128                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
6129                .create("grid")
6130                .unwrap();
6131            ds.write_raw(&data).unwrap();
6132            file.close().unwrap();
6133        }
6134        {
6135            let file = H5File::open(&path).unwrap();
6136            let ds = file.dataset("grid").unwrap();
6137            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6138            // Slice spans multiple chunks (rows 2..5, cols 1..4).
6139            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
6140        }
6141        std::fs::remove_file(&path).ok();
6142    }
6143
6144    #[test]
6145    fn read_into_matches_vec_single_chunk() {
6146        let path = temp_path("into_single_chunk");
6147        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
6148        {
6149            let file = H5File::create(&path).unwrap();
6150            let ds = file
6151                .new_dataset::<i32>()
6152                .shape([3, 4])
6153                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
6154                .create("g")
6155                .unwrap();
6156            ds.write_raw(&data).unwrap();
6157            file.close().unwrap();
6158        }
6159        {
6160            let file = H5File::open(&path).unwrap();
6161            let ds = file.dataset("g").unwrap();
6162            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
6163            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
6164        }
6165        std::fs::remove_file(&path).ok();
6166    }
6167
6168    #[cfg(feature = "deflate")]
6169    #[test]
6170    fn read_into_matches_vec_chunked_deflate() {
6171        let path = temp_path("into_chunk_deflate");
6172        let data: Vec<i32> = (0..35).collect(); // 7 x 5
6173        {
6174            let file = H5File::create(&path).unwrap();
6175            let ds = file
6176                .new_dataset::<i32>()
6177                .shape([7, 5])
6178                .chunk(&[3, 2])
6179                .deflate(4)
6180                .create("grid")
6181                .unwrap();
6182            ds.write_raw(&data).unwrap();
6183            file.close().unwrap();
6184        }
6185        {
6186            let file = H5File::open(&path).unwrap();
6187            let ds = file.dataset("grid").unwrap();
6188            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
6189            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
6190        }
6191        std::fs::remove_file(&path).ok();
6192    }
6193
6194    #[test]
6195    fn read_into_wrong_buffer_size_rejected() {
6196        let path = temp_path("into_badlen");
6197        let data: Vec<i32> = (0..20).collect(); // 4 x 5
6198        {
6199            let file = H5File::create(&path).unwrap();
6200            let ds = file
6201                .new_dataset::<i32>()
6202                .shape([4, 5])
6203                .create("mat")
6204                .unwrap();
6205            ds.write_raw(&data).unwrap();
6206            file.close().unwrap();
6207        }
6208        {
6209            let file = H5File::open(&path).unwrap();
6210            let ds = file.dataset("mat").unwrap();
6211
6212            // Too small / too large full-read buffers are both rejected.
6213            let mut small = vec![0i32; 19];
6214            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
6215            let mut large = vec![0i32; 21];
6216            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
6217
6218            // Slice buffer must be exactly product(counts) = 4.
6219            let mut bad_slice = vec![0i32; 3];
6220            assert!(ds
6221                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
6222                .is_err());
6223            // The correctly sized slice buffer succeeds.
6224            let mut ok_slice = vec![0i32; 4];
6225            assert!(ds
6226                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
6227                .is_ok());
6228        }
6229        std::fs::remove_file(&path).ok();
6230    }
6231
6232    #[test]
6233    fn read_into_wrong_element_size_rejected() {
6234        let path = temp_path("into_badtype");
6235        let data: Vec<i32> = (0..20).collect(); // element size 4
6236        {
6237            let file = H5File::create(&path).unwrap();
6238            let ds = file
6239                .new_dataset::<i32>()
6240                .shape([4, 5])
6241                .create("mat")
6242                .unwrap();
6243            ds.write_raw(&data).unwrap();
6244            file.close().unwrap();
6245        }
6246        {
6247            let file = H5File::open(&path).unwrap();
6248            let ds = file.dataset("mat").unwrap();
6249            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
6250            // element size -> TypeMismatch, even with a "correctly sized" Vec.
6251            let mut as_u8 = vec![0u8; 20];
6252            assert!(matches!(
6253                ds.read_raw_into::<u8>(&mut as_u8),
6254                Err(crate::Hdf5Error::TypeMismatch(_))
6255            ));
6256            let mut as_i64 = vec![0i64; 20];
6257            assert!(matches!(
6258                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
6259                Err(crate::Hdf5Error::TypeMismatch(_))
6260            ));
6261        }
6262        std::fs::remove_file(&path).ok();
6263    }
6264
6265    #[test]
6266    fn write_slice_2d() {
6267        let path = temp_path("write_slice_2d");
6268
6269        {
6270            let file = H5File::create(&path).unwrap();
6271            let ds = file
6272                .new_dataset::<f32>()
6273                .shape([3, 4])
6274                .create("data")
6275                .unwrap();
6276            ds.write_raw(&[0.0f32; 12]).unwrap();
6277            // Overwrite a 2x2 sub-region
6278            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
6279                .unwrap();
6280            file.close().unwrap();
6281        }
6282        {
6283            let file = H5File::open(&path).unwrap();
6284            let ds = file.dataset("data").unwrap();
6285            let full = ds.read_raw::<f32>().unwrap();
6286            // Row 0: [0,0,0,0]
6287            // Row 1: [0,10,20,0]
6288            // Row 2: [0,30,40,0]
6289            assert_eq!(
6290                full,
6291                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
6292            );
6293        }
6294
6295        std::fs::remove_file(&path).ok();
6296    }
6297
6298    /// One 2x4 i32 chunk whose every element is `v`.
6299    fn chunk_of(v: i32) -> Vec<u8> {
6300        (0..8).flat_map(|_| v.to_le_bytes()).collect()
6301    }
6302
6303    /// Write chunk (0,0) `rewrites` times — each time with a different value,
6304    /// so no write can be skipped — and return the closed file's size along
6305    /// with what the chunk reads back as.
6306    fn rewrite_chunk(
6307        tag: &str,
6308        rewrites: i32,
6309        build: impl Fn(&H5File) -> crate::H5Dataset,
6310    ) -> (u64, i32) {
6311        let path = temp_path(tag);
6312        {
6313            let file = H5File::create(&path).unwrap();
6314            let ds = build(&file);
6315            for v in 1..=rewrites {
6316                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
6317            }
6318            file.close().unwrap();
6319        }
6320        let size = std::fs::metadata(&path).unwrap().len();
6321        let first = {
6322            let file = H5File::open(&path).unwrap();
6323            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
6324        };
6325        std::fs::remove_file(&path).ok();
6326        (size, first)
6327    }
6328
6329    // An unfiltered chunk's stored size is fixed by the chunk shape, so
6330    // rewriting it must overwrite the block it already occupies rather than
6331    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
6332    // leaves must_alloc false for exactly this case). The file must therefore
6333    // be byte-identical in size no matter how many times the chunk is written.
6334    #[test]
6335    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
6336        let build = |f: &H5File| {
6337            f.new_dataset::<i32>()
6338                .shape([2, 4])
6339                .chunk(&[2, 4])
6340                .max_shape(&[None, Some(4)])
6341                .create("d")
6342                .unwrap()
6343        };
6344        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
6345        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
6346        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6347        assert_eq!(last, 8, "the last write must be the one that survives");
6348    }
6349
6350    #[test]
6351    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
6352        let build = |f: &H5File| {
6353            f.new_dataset::<i32>()
6354                .shape([2, 4])
6355                .chunk(&[2, 4])
6356                .create("d")
6357                .unwrap()
6358        };
6359        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
6360        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
6361        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6362        assert_eq!(last, 8);
6363    }
6364
6365    #[test]
6366    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
6367        let build = |f: &H5File| {
6368            f.new_dataset::<i32>()
6369                .shape([2, 4])
6370                .chunk(&[2, 4])
6371                .max_shape(&[None, None])
6372                .create("d")
6373                .unwrap()
6374        };
6375        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
6376        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
6377        assert_eq!(many, once, "8 rewrites grew the file past a single write");
6378        assert_eq!(last, 8);
6379    }
6380
6381    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
6382    // pool. Every node is the same size, so the blocks already on disk are
6383    // reused and repeated flushes cost nothing; sizing the root to its record
6384    // count instead would relocate it each time and orphan the block it left.
6385    #[test]
6386    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
6387        let flush_n = |label: &str, flushes: usize| -> u64 {
6388            let path = temp_path(label);
6389            {
6390                let file = H5File::create(&path).unwrap();
6391                let ds = file
6392                    .new_dataset::<i32>()
6393                    .shape([2, 4])
6394                    .chunk(&[2, 4])
6395                    .max_shape(&[None, None])
6396                    .create("d")
6397                    .unwrap();
6398                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
6399                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
6400                for _ in 0..flushes {
6401                    ds.flush().unwrap();
6402                }
6403                file.close().unwrap();
6404            }
6405            let size = std::fs::metadata(&path).unwrap().len();
6406            // The data must survive every rewrite of the index.
6407            {
6408                let file = H5File::open(&path).unwrap();
6409                let ds = file.dataset("d").unwrap();
6410                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
6411            }
6412            std::fs::remove_file(&path).ok();
6413            size
6414        };
6415        assert_eq!(
6416            flush_n("bt2_flush_8", 8),
6417            flush_n("bt2_flush_1", 1),
6418            "8 index flushes grew the file past a single one"
6419        );
6420    }
6421
6422    // A filtered chunk whose compressed size changes cannot stay put, so it
6423    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
6424    // H5MF_xfree). Alternating between two payloads of different compressed
6425    // size must therefore keep reusing the same two blocks instead of
6426    // appending a fresh one each time.
6427    #[cfg(feature = "deflate")]
6428    #[test]
6429    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
6430        // All-equal elements deflate to far fewer bytes than a varied payload,
6431        // so the two writes below land at different stored sizes.
6432        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
6433        let varied: Vec<u8> = (0..8i32)
6434            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
6435            .collect();
6436
6437        let sizes: Vec<u64> = [1usize, 8]
6438            .iter()
6439            .map(|&rounds| {
6440                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
6441                {
6442                    let file = H5File::create(&path).unwrap();
6443                    let ds = file
6444                        .new_dataset::<i32>()
6445                        .shape([2, 4])
6446                        .chunk(&[2, 4])
6447                        .max_shape(&[None, Some(4)])
6448                        .deflate(6)
6449                        .create("d")
6450                        .unwrap();
6451                    for _ in 0..rounds {
6452                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
6453                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
6454                    }
6455                    file.close().unwrap();
6456                }
6457                let size = std::fs::metadata(&path).unwrap().len();
6458                {
6459                    let file = H5File::open(&path).unwrap();
6460                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
6461                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
6462                    assert_eq!(got, want, "the last write must survive the round trip");
6463                }
6464                std::fs::remove_file(&path).ok();
6465                size
6466            })
6467            .collect();
6468
6469        assert_eq!(
6470            sizes[1], sizes[0],
6471            "8 alternating rewrites grew the file past a single pair"
6472        );
6473    }
6474
6475    #[test]
6476    fn write_slice_out_of_bounds_rejected() {
6477        let path = temp_path("write_slice_oob");
6478        let file = H5File::create(&path).unwrap();
6479        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
6480        ds.write_raw(&[0i32; 4]).unwrap();
6481        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
6482        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
6483        // An in-bounds slice still works.
6484        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
6485        std::fs::remove_file(&path).ok();
6486    }
6487
6488    #[test]
6489    fn duplicate_dataset_name_rejected() {
6490        let path = temp_path("dup_name");
6491        let file = H5File::create(&path).unwrap();
6492        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
6493        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
6494        std::fs::remove_file(&path).ok();
6495    }
6496
6497    #[test]
6498    fn extend_cannot_shrink() {
6499        let path = temp_path("extend_shrink");
6500        let file = H5File::create(&path).unwrap();
6501        let ds = file
6502            .new_dataset::<i32>()
6503            .shape([0])
6504            .chunk(&[2])
6505            .max_shape(&[None])
6506            .create("d")
6507            .unwrap();
6508        ds.append(&[1i32, 2, 3, 4]).unwrap();
6509        // Shrinking below the written extent must be rejected.
6510        assert!(ds.extend(&[2]).is_err());
6511        // Growing is fine.
6512        assert!(ds.extend(&[6]).is_ok());
6513        std::fs::remove_file(&path).ok();
6514    }
6515
6516    #[test]
6517    fn attr_read_roundtrip() {
6518        use crate::types::VarLenUnicode;
6519        let path = temp_path("attr_read");
6520
6521        {
6522            let file = H5File::create(&path).unwrap();
6523            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6524            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6525            let a1 = ds
6526                .new_attr::<VarLenUnicode>()
6527                .shape(())
6528                .create("units")
6529                .unwrap();
6530            a1.write_string("meters").unwrap();
6531            let a2 = ds
6532                .new_attr::<VarLenUnicode>()
6533                .shape(())
6534                .create("desc")
6535                .unwrap();
6536            a2.write_string("test data").unwrap();
6537            file.close().unwrap();
6538        }
6539        {
6540            let file = H5File::open(&path).unwrap();
6541            let ds = file.dataset("data").unwrap();
6542
6543            let names = ds.attr_names().unwrap();
6544            assert!(names.contains(&"units".to_string()));
6545            assert!(names.contains(&"desc".to_string()));
6546
6547            let units = ds.attr("units").unwrap();
6548            assert_eq!(units.read_string().unwrap(), "meters");
6549
6550            let desc = ds.attr("desc").unwrap();
6551            assert_eq!(desc.read_string().unwrap(), "test data");
6552        }
6553
6554        std::fs::remove_file(&path).ok();
6555    }
6556
6557    #[test]
6558    fn type_mismatch_element_size() {
6559        let path = temp_path("type_mismatch");
6560
6561        {
6562            let file = H5File::create(&path).unwrap();
6563            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
6564            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
6565            file.close().unwrap();
6566        }
6567
6568        {
6569            let file = H5File::open(&path).unwrap();
6570            let ds = file.dataset("data").unwrap();
6571            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
6572            let result = ds.read_raw::<u8>();
6573            assert!(result.is_err());
6574        }
6575
6576        std::fs::remove_file(&path).ok();
6577    }
6578
6579    #[test]
6580    fn dataset_survives_file_move() {
6581        let path = temp_path("ds_survives");
6582
6583        let ds = {
6584            let file = H5File::create(&path).unwrap();
6585            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
6586        };
6587        // file is dropped here, but ds still holds Rc to the inner state
6588        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6589        // The writer will finalize on drop of the last Rc
6590
6591        std::fs::remove_file(&path).ok();
6592    }
6593
6594    #[test]
6595    fn new_attr_scalar_string() {
6596        use crate::types::VarLenUnicode;
6597
6598        let path = temp_path("attr_scalar_string");
6599        {
6600            let file = H5File::create(&path).unwrap();
6601            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
6602            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
6603
6604            let attr = ds
6605                .new_attr::<VarLenUnicode>()
6606                .shape(())
6607                .create("name")
6608                .unwrap();
6609            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
6610                .unwrap();
6611
6612            file.close().unwrap();
6613        }
6614
6615        // Verify the file is still valid and readable
6616        {
6617            use crate::format::messages::datatype::DatatypeMessage;
6618            let file = H5File::open(&path).unwrap();
6619            let ds = file.dataset("data").unwrap();
6620            assert_eq!(ds.shape(), vec![4]);
6621            let readback = ds.read_raw::<u8>().unwrap();
6622            assert_eq!(readback, vec![1u8, 2, 3, 4]);
6623
6624            // The string attribute is stored as a true variable-length string
6625            // (not fixed-length) and round-trips its value.
6626            let attr = ds.attr("name").unwrap();
6627            assert!(
6628                matches!(
6629                    attr.datatype().unwrap(),
6630                    DatatypeMessage::VarLenString { .. }
6631                ),
6632                "string attribute should have a variable-length string datatype"
6633            );
6634            assert_eq!(attr.read_string().unwrap(), "test_value");
6635        }
6636
6637        std::fs::remove_file(&path).ok();
6638    }
6639
6640    #[test]
6641    fn all_numeric_types_roundtrip() {
6642        let path = temp_path("all_types");
6643
6644        {
6645            let file = H5File::create(&path).unwrap();
6646
6647            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
6648            ds.write_raw(&[1u8, 2]).unwrap();
6649
6650            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
6651            ds.write_raw(&[-1i8, 1]).unwrap();
6652
6653            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
6654            ds.write_raw(&[100u16, 200]).unwrap();
6655
6656            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
6657            ds.write_raw(&[-100i16, 100]).unwrap();
6658
6659            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
6660            ds.write_raw(&[1000u32, 2000]).unwrap();
6661
6662            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
6663            ds.write_raw(&[-1000i32, 1000]).unwrap();
6664
6665            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
6666            ds.write_raw(&[10000u64, 20000]).unwrap();
6667
6668            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
6669            ds.write_raw(&[-10000i64, 10000]).unwrap();
6670
6671            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
6672            ds.write_raw(&[1.5f32, 2.5]).unwrap();
6673
6674            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
6675            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
6676
6677            file.close().unwrap();
6678        }
6679
6680        {
6681            let file = H5File::open(&path).unwrap();
6682
6683            assert_eq!(
6684                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
6685                vec![1u8, 2]
6686            );
6687            assert_eq!(
6688                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
6689                vec![-1i8, 1]
6690            );
6691            assert_eq!(
6692                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
6693                vec![100u16, 200]
6694            );
6695            assert_eq!(
6696                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
6697                vec![-100i16, 100]
6698            );
6699            assert_eq!(
6700                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
6701                vec![1000u32, 2000]
6702            );
6703            assert_eq!(
6704                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
6705                vec![-1000i32, 1000]
6706            );
6707            assert_eq!(
6708                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
6709                vec![10000u64, 20000]
6710            );
6711            assert_eq!(
6712                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
6713                vec![-10000i64, 10000]
6714            );
6715            assert_eq!(
6716                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
6717                vec![1.5f32, 2.5]
6718            );
6719            assert_eq!(
6720                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
6721                vec![1.23456f64, 7.89012]
6722            );
6723        }
6724
6725        std::fs::remove_file(&path).ok();
6726    }
6727
6728    #[test]
6729    fn append_chunked_roundtrip() {
6730        let path = temp_path("append_chunked");
6731
6732        {
6733            let file = H5File::create(&path).unwrap();
6734            let ds = file
6735                .new_dataset::<f64>()
6736                .shape([0, 3])
6737                .chunk(&[1, 3])
6738                .max_shape(&[None, Some(3)])
6739                .create("data")
6740                .unwrap();
6741
6742            // Append one frame
6743            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
6744            // Append two frames at once
6745            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
6746
6747            file.close().unwrap();
6748        }
6749
6750        {
6751            let file = H5File::open(&path).unwrap();
6752            let ds = file.dataset("data").unwrap();
6753            assert_eq!(ds.shape(), vec![3, 3]);
6754            let all = ds.read_raw::<f64>().unwrap();
6755            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
6756        }
6757
6758        std::fs::remove_file(&path).ok();
6759    }
6760
6761    #[test]
6762    fn append_1d_chunked() {
6763        let path = temp_path("append_1d");
6764
6765        {
6766            let file = H5File::create(&path).unwrap();
6767            let ds = file
6768                .new_dataset::<i32>()
6769                .shape([0])
6770                .chunk(&[4])
6771                .max_shape(&[None])
6772                .create("values")
6773                .unwrap();
6774
6775            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
6776            ds.append(&[40i32]).unwrap(); // fills chunk boundary
6777            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
6778
6779            file.close().unwrap();
6780        }
6781
6782        {
6783            let file = H5File::open(&path).unwrap();
6784            let ds = file.dataset("values").unwrap();
6785            assert_eq!(ds.shape(), vec![8]);
6786            let all = ds.read_raw::<i32>().unwrap();
6787            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
6788        }
6789
6790        std::fs::remove_file(&path).ok();
6791    }
6792
6793    #[test]
6794    fn append_partial_chunk_flushed_on_close() {
6795        let path = temp_path("append_partial_close");
6796
6797        {
6798            let file = H5File::create(&path).unwrap();
6799            let ds = file
6800                .new_dataset::<f64>()
6801                .shape([0])
6802                .chunk(&[4])
6803                .max_shape(&[None])
6804                .create("vals")
6805                .unwrap();
6806
6807            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
6808            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
6809            file.close().unwrap();
6810        }
6811
6812        {
6813            let file = H5File::open(&path).unwrap();
6814            let ds = file.dataset("vals").unwrap();
6815            assert_eq!(ds.shape(), vec![5]);
6816            let all = ds.read_raw::<f64>().unwrap();
6817            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
6818            // read_raw reads total shape elements
6819            assert_eq!(all.len(), 5);
6820            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
6821        }
6822
6823        std::fs::remove_file(&path).ok();
6824    }
6825
6826    /// An append that leaves its chunk partial is buffered until close, and
6827    /// the flush has to keep the frames that chunk already holds. It built a
6828    /// fresh fill-value chunk around the buffered frame instead, so reopening
6829    /// a file and appending one row erased every earlier row of that chunk
6830    /// (issue #3). Four sessions: the second lands beside an existing row, the
6831    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
6832    /// the third left in chunk 1.
6833    #[test]
6834    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
6835        let path = temp_path("append_reopen_partial");
6836
6837        {
6838            let file = H5File::create(&path).unwrap();
6839            let ds = file
6840                .new_dataset::<i32>()
6841                .shape([0, 3])
6842                .chunk(&[4, 3])
6843                .max_shape(&[None, Some(3)])
6844                .create("values")
6845                .unwrap();
6846            ds.append(&[1, 2, 3]).unwrap();
6847            file.close().unwrap();
6848        }
6849        for rows in [
6850            vec![4, 5, 6],
6851            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
6852            vec![16, 17, 18],
6853        ] {
6854            let file = H5File::open_rw(&path).unwrap();
6855            file.dataset_writer("values")
6856                .unwrap()
6857                .append(&rows)
6858                .unwrap();
6859            file.close().unwrap();
6860        }
6861
6862        let file = H5File::open(&path).unwrap();
6863        let ds = file.dataset("values").unwrap();
6864        assert_eq!(ds.shape(), vec![6, 3]);
6865        assert_eq!(
6866            ds.read_raw::<i32>().unwrap(),
6867            (1..=18).collect::<Vec<i32>>()
6868        );
6869        std::fs::remove_file(&path).ok();
6870    }
6871
6872    #[cfg(feature = "deflate")]
6873    #[test]
6874    fn vlen_append_after_reopen_filtered() {
6875        // Reopen + append into a partially-written *compressed* vlen chunk
6876        // (index-block chunk). Exercises filtered-index-block reconstruction
6877        // in open_append plus filtered read-modify-write.
6878        let path = temp_path("vlen_reopen_filtered");
6879        {
6880            let file = H5File::create(&path).unwrap();
6881            file.create_appendable_vlen_dataset(
6882                "strs",
6883                4,
6884                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6885            )
6886            .unwrap();
6887            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
6888                .unwrap();
6889            file.close().unwrap();
6890        }
6891        {
6892            let file = H5File::open_rw(&path).unwrap();
6893            file.append_vlen_strings("strs", &["delta"]).unwrap();
6894            file.close().unwrap();
6895        }
6896        {
6897            let file = H5File::open(&path).unwrap();
6898            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6899            assert_eq!(
6900                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
6901                vec!["alpha", "beta", "gamma", "delta"]
6902            );
6903        }
6904        std::fs::remove_file(&path).ok();
6905    }
6906
6907    #[test]
6908    fn vlen_append_after_reopen_data_block() {
6909        // Reopen + append into a partial chunk that lives in an extensible-
6910        // array *data block* (chunk index >= idx_blk_elmts). Exercises
6911        // data-block resolution in read_chunk_if_present and write_chunk.
6912        let path = temp_path("vlen_reopen_datablk");
6913        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
6914        {
6915            let file = H5File::create(&path).unwrap();
6916            file.create_appendable_vlen_dataset("strs", 2, None)
6917                .unwrap();
6918            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6919            file.append_vlen_strings("strs", &refs).unwrap();
6920            file.close().unwrap();
6921        }
6922        {
6923            let file = H5File::open_rw(&path).unwrap();
6924            file.append_vlen_strings("strs", &["s9"]).unwrap();
6925            file.close().unwrap();
6926        }
6927        {
6928            let file = H5File::open(&path).unwrap();
6929            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6930            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
6931            assert_eq!(got, want);
6932        }
6933        std::fs::remove_file(&path).ok();
6934    }
6935
6936    #[test]
6937    fn vlen_append_after_reopen_super_block() {
6938        // Reopen + append into a partial chunk whose index falls in an
6939        // extensible-array *super block* (chunk index 244 with the default
6940        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
6941        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
6942        // index block or its direct data blocks, so chunk 244 is reached
6943        // via a super block read from disk). Exercises the ViaSblk branch
6944        // of read_chunk_if_present.
6945        let path = temp_path("vlen_reopen_super");
6946        // 489 strings, chunk size 2 -> chunk 244 holds one string only
6947        // (partially filled) and is flushed to disk on close.
6948        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
6949        {
6950            let file = H5File::create(&path).unwrap();
6951            file.create_appendable_vlen_dataset("strs", 2, None)
6952                .unwrap();
6953            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6954            file.append_vlen_strings("strs", &refs).unwrap();
6955            file.close().unwrap();
6956        }
6957        {
6958            let file = H5File::open_rw(&path).unwrap();
6959            file.append_vlen_strings("strs", &["v489"]).unwrap();
6960            file.close().unwrap();
6961        }
6962        {
6963            let file = H5File::open(&path).unwrap();
6964            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6965            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
6966            assert_eq!(got, want);
6967        }
6968        std::fs::remove_file(&path).ok();
6969    }
6970
6971    #[cfg(feature = "deflate")]
6972    #[test]
6973    fn vlen_append_after_reopen_filtered_data_block() {
6974        // The hardest path: compressed + chunk in a data block + partial
6975        // read-modify-write across a reopen.
6976        let path = temp_path("vlen_reopen_filt_datablk");
6977        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
6978        {
6979            let file = H5File::create(&path).unwrap();
6980            file.create_appendable_vlen_dataset(
6981                "strs",
6982                2,
6983                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
6984            )
6985            .unwrap();
6986            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
6987            file.append_vlen_strings("strs", &refs).unwrap();
6988            file.close().unwrap();
6989        }
6990        {
6991            let file = H5File::open_rw(&path).unwrap();
6992            file.append_vlen_strings("strs", &["item09"]).unwrap();
6993            file.close().unwrap();
6994        }
6995        {
6996            let file = H5File::open(&path).unwrap();
6997            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
6998            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
6999            assert_eq!(got, want);
7000        }
7001        std::fs::remove_file(&path).ok();
7002    }
7003
7004    #[test]
7005    fn group_nx_class_attribute_roundtrip() {
7006        // Non-root groups carry attributes (NeXus `NX_class`) in their
7007        // own object header, and the reader reads them back by path.
7008        let path = temp_path("group_nx_class");
7009        {
7010            let file = H5File::create(&path).unwrap();
7011            let entry = file.create_group("entry").unwrap();
7012            entry.set_attr_string("NX_class", "NXentry").unwrap();
7013            let det = entry.create_group("detector").unwrap();
7014            det.set_attr_string("NX_class", "NXdetector").unwrap();
7015            det.set_attr_numeric("frame_count", &7i32).unwrap();
7016            det.new_dataset::<f32>()
7017                .shape([4])
7018                .create("data")
7019                .unwrap()
7020                .write_raw(&[1.0f32; 4])
7021                .unwrap();
7022            file.close().unwrap();
7023        }
7024        {
7025            let file = H5File::open(&path).unwrap();
7026            let entry = file.root_group().group("entry").unwrap();
7027            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
7028            let det = entry.group("detector").unwrap();
7029            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
7030            let names = det.attr_names().unwrap();
7031            assert!(names.contains(&"NX_class".to_string()));
7032            assert!(names.contains(&"frame_count".to_string()));
7033        }
7034        std::fs::remove_file(&path).ok();
7035    }
7036
7037    #[test]
7038    fn ea_super_block_roundtrip() {
7039        // 2000 chunks span several extensible-array super blocks. Before
7040        // super-block support the writer errored at chunk index 228.
7041        let path = temp_path("ea_super_rt");
7042        {
7043            let file = H5File::create(&path).unwrap();
7044            let ds = file
7045                .new_dataset::<i32>()
7046                .shape([0])
7047                .chunk(&[1])
7048                .max_shape(&[None])
7049                .create("v")
7050                .unwrap();
7051            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
7052            file.close().unwrap();
7053        }
7054        {
7055            let file = H5File::open(&path).unwrap();
7056            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7057            assert_eq!(v.len(), 2000);
7058            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
7059        }
7060        std::fs::remove_file(&path).ok();
7061    }
7062
7063    #[cfg(feature = "deflate")]
7064    #[test]
7065    fn ea_filtered_super_block_roundtrip() {
7066        // Compressed chunks across super blocks.
7067        let path = temp_path("ea_filt_super");
7068        {
7069            let file = H5File::create(&path).unwrap();
7070            let ds = file
7071                .new_dataset::<i32>()
7072                .shape([0])
7073                .chunk(&[1])
7074                .max_shape(&[None])
7075                .deflate(4)
7076                .create("v")
7077                .unwrap();
7078            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
7079            file.close().unwrap();
7080        }
7081        {
7082            let file = H5File::open(&path).unwrap();
7083            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7084            assert_eq!(v, (0..600).collect::<Vec<i32>>());
7085        }
7086        std::fs::remove_file(&path).ok();
7087    }
7088
7089    #[test]
7090    fn ea_super_block_open_append() {
7091        // Reopen a dataset and append chunks that fall in super blocks.
7092        let path = temp_path("ea_super_append");
7093        {
7094            let file = H5File::create(&path).unwrap();
7095            let ds = file
7096                .new_dataset::<i32>()
7097                .shape([0])
7098                .chunk(&[1])
7099                .max_shape(&[None])
7100                .create("v")
7101                .unwrap();
7102            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
7103            file.close().unwrap();
7104        }
7105        {
7106            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7107            let idx = w.dataset_index("v").unwrap();
7108            for c in 300..900u64 {
7109                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
7110            }
7111            w.extend_dataset(idx, &[900]).unwrap();
7112            w.close().unwrap();
7113        }
7114        {
7115            let file = H5File::open(&path).unwrap();
7116            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7117            assert_eq!(v.len(), 900);
7118            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
7119        }
7120        std::fs::remove_file(&path).ok();
7121    }
7122
7123    // Two or more unlimited dimensions select the v2 B-tree index; with a
7124    // filter its records become type 11, carrying each chunk's stored size and
7125    // mask. The payload is highly compressible, so the chunks really are
7126    // stored smaller than the extent — the file would be at least
7127    // 6*8*4 = 192 bytes of raw chunk data otherwise.
7128    #[cfg(feature = "deflate")]
7129    #[test]
7130    fn compressed_multi_unlimited_dataset_roundtrips() {
7131        let path = temp_path("bt2_filtered");
7132        {
7133            let file = H5File::create(&path).unwrap();
7134            let ds = file
7135                .new_dataset::<i32>()
7136                .shape([6, 8])
7137                .chunk(&[2, 4])
7138                .max_shape(&[None, None])
7139                .deflate(6)
7140                .create("d")
7141                .unwrap();
7142            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
7143            // A partial write forces a decompress-patch-recompress of one
7144            // chunk, whose new compressed size may not fit its old block.
7145            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
7146            file.close().unwrap();
7147        }
7148        {
7149            let file = H5File::open(&path).unwrap();
7150            let ds = file.dataset("d").unwrap();
7151            assert_eq!(ds.shape(), vec![6, 8]);
7152            let mut want = vec![7i32; 48];
7153            want[9] = 1;
7154            want[10] = 2;
7155            want[17] = 3;
7156            want[18] = 4;
7157            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
7158        }
7159        std::fs::remove_file(&path).ok();
7160    }
7161
7162    #[test]
7163    fn btree_v2_multi_unlimited_roundtrip() {
7164        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
7165        // index; chunks are written by grid coordinates with write_chunk_at.
7166        let path = temp_path("bt2_multi");
7167        {
7168            let file = H5File::create(&path).unwrap();
7169            let ds = file
7170                .new_dataset::<i32>()
7171                .shape([0, 0])
7172                .chunk(&[2, 2])
7173                .max_shape(&[None, None])
7174                .create("grid")
7175                .unwrap();
7176            assert!(ds.is_chunked());
7177            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
7178            for cr in 0..2usize {
7179                for cc in 0..2usize {
7180                    let mut bytes = Vec::new();
7181                    for i in 0..2usize {
7182                        for j in 0..2usize {
7183                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
7184                            bytes.extend_from_slice(&v.to_le_bytes());
7185                        }
7186                    }
7187                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
7188                }
7189            }
7190            file.close().unwrap();
7191        }
7192        {
7193            let file = H5File::open(&path).unwrap();
7194            let ds = file.dataset("grid").unwrap();
7195            assert_eq!(ds.shape(), vec![4, 4]);
7196            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
7197        }
7198        std::fs::remove_file(&path).ok();
7199    }
7200
7201    #[test]
7202    fn subframe_chunking_roundtrip() {
7203        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
7204        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
7205        // the chunk-grid coordinates.
7206        let path = temp_path("subframe");
7207        {
7208            let file = H5File::create(&path).unwrap();
7209            let ds = file
7210                .new_dataset::<i32>()
7211                .shape([0, 8, 8])
7212                .chunk(&[1, 4, 4])
7213                .max_shape(&[None, Some(8), Some(8)])
7214                .create("v")
7215                .unwrap();
7216            for f in 0..3usize {
7217                for cr in 0..2usize {
7218                    for cc in 0..2usize {
7219                        let mut bytes = Vec::new();
7220                        for i in 0..4usize {
7221                            for j in 0..4usize {
7222                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
7223                                bytes.extend_from_slice(&v.to_le_bytes());
7224                            }
7225                        }
7226                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
7227                    }
7228                }
7229            }
7230            file.close().unwrap();
7231        }
7232        {
7233            let file = H5File::open(&path).unwrap();
7234            let ds = file.dataset("v").unwrap();
7235            assert_eq!(ds.shape(), vec![3, 8, 8]);
7236            assert_eq!(
7237                ds.read_raw::<i32>().unwrap(),
7238                (0..192).collect::<Vec<i32>>()
7239            );
7240        }
7241        std::fs::remove_file(&path).ok();
7242    }
7243
7244    #[test]
7245    fn fill_value_contiguous_roundtrip() {
7246        let path = temp_path("fill_value_contig");
7247        {
7248            let file = H5File::create(&path).unwrap();
7249            let ds = file
7250                .new_dataset::<f32>()
7251                .shape([4])
7252                .fill_value(2.5f32)
7253                .create("data")
7254                .unwrap();
7255            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
7256            file.close().unwrap();
7257        }
7258        // open_append decodes the fill-value message back from the header.
7259        {
7260            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7261            let idx = writer.dataset_index("data").unwrap();
7262            assert_eq!(
7263                writer.ds(idx).lock().fill_value,
7264                Some(2.5f32.to_le_bytes().to_vec())
7265            );
7266        }
7267        // Data still reads back correctly.
7268        {
7269            let file = H5File::open(&path).unwrap();
7270            let ds = file.dataset("data").unwrap();
7271            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
7272        }
7273        std::fs::remove_file(&path).ok();
7274    }
7275
7276    /// Early allocation on a fixed unfiltered shape selects the implicit
7277    /// index, and "implicit" is literal: the file holds no index structure
7278    /// at all, only a version-4 layout message of index type 2 pointing at
7279    /// the run of chunk space the create allocated.
7280    #[test]
7281    fn early_allocation_writes_the_implicit_index() {
7282        let path = temp_path("implicit_index");
7283        {
7284            let file = H5File::create(&path).unwrap();
7285            let ds = file
7286                .new_dataset::<i32>()
7287                .shape([16])
7288                .chunk(&[4])
7289                .early_allocation()
7290                .create("data")
7291                .unwrap();
7292            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7293            file.close().unwrap();
7294        }
7295        let bytes = std::fs::read(&path).unwrap();
7296        for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7297            assert!(
7298                !bytes.windows(4).any(|w| w == magic),
7299                "{} appears in a file whose chunk index is supposed to be no \
7300                 structure at all",
7301                String::from_utf8_lossy(magic)
7302            );
7303        }
7304        {
7305            let file = H5File::open(&path).unwrap();
7306            let ds = file.dataset("data").unwrap();
7307            assert_eq!(
7308                ds.read_raw::<i32>().unwrap(),
7309                (0..16i32).collect::<Vec<_>>()
7310            );
7311        }
7312        std::fs::remove_file(&path).ok();
7313    }
7314
7315    /// Two of the conditions are conditions: an unlimited dimension or a
7316    /// filter each send the dataset to the index libhdf5 would pick
7317    /// instead, early allocation or not. (The third — one whole-dataset
7318    /// chunk — sends it to the single-chunk index instead of Fixed Array;
7319    /// see `one_whole_dataset_chunk_writes_the_single_chunk_index`.)
7320    #[test]
7321    #[cfg(feature = "deflate")]
7322    fn early_allocation_only_picks_implicit_where_libhdf5_does() {
7323        // Every case writes and reads back its data, so a mis-selected index
7324        // shows up as wrong bytes and not just as a different structure.
7325        for (which, magic) in [("unlimited", b"EAHD"), ("filtered", b"FAHD")] {
7326            let path = temp_path("implicit_not");
7327            {
7328                let file = H5File::create(&path).unwrap();
7329                let builder = file.new_dataset::<i32>().shape([16]);
7330                let builder = match which {
7331                    "unlimited" => builder.chunk(&[4]).max_shape(&[None]),
7332                    _ => builder.chunk(&[4]).deflate(6),
7333                };
7334                let ds = builder.early_allocation().create("data").unwrap();
7335                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7336                file.close().unwrap();
7337            }
7338            let bytes = std::fs::read(&path).unwrap();
7339            assert!(
7340                bytes.windows(4).any(|w| w == magic),
7341                "{which}: expected a {} index",
7342                String::from_utf8_lossy(magic)
7343            );
7344            let file = H5File::open(&path).unwrap();
7345            assert_eq!(
7346                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7347                (0..16i32).collect::<Vec<_>>(),
7348                "{which}"
7349            );
7350            std::fs::remove_file(&path).ok();
7351        }
7352    }
7353
7354    /// One whole-dataset chunk always selects the single-chunk index —
7355    /// ahead of Fixed Array, and ahead of Implicit too, whether or not early
7356    /// allocation was requested (`H5D__layout_set_latest_indexing` checks it
7357    /// unconditionally). Like Implicit, "single chunk" is literal: no index
7358    /// structure at all, just the one chunk's address — and, unfiltered and
7359    /// early-allocated, that address exists before anything is written — in
7360    /// the layout message directly.
7361    #[test]
7362    fn one_whole_dataset_chunk_writes_the_single_chunk_index() {
7363        for early in [false, true] {
7364            let path = temp_path("single_chunk_index");
7365            {
7366                let file = H5File::create(&path).unwrap();
7367                let builder = file.new_dataset::<i32>().shape([16]).chunk(&[16]);
7368                let builder = if early {
7369                    builder.early_allocation()
7370                } else {
7371                    builder
7372                };
7373                let ds = builder.create("data").unwrap();
7374                ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap();
7375                file.close().unwrap();
7376            }
7377            let bytes = std::fs::read(&path).unwrap();
7378            for magic in [b"EAHD", b"EAIB", b"FAHD", b"FADB", b"BTHD", b"TREE"] {
7379                assert!(
7380                    !bytes.windows(4).any(|w| w == magic),
7381                    "early={early}: {} appears in a file whose chunk index is \
7382                     supposed to be no structure at all",
7383                    String::from_utf8_lossy(magic)
7384                );
7385            }
7386            let file = H5File::open(&path).unwrap();
7387            assert_eq!(
7388                file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7389                (0..16i32).collect::<Vec<_>>(),
7390                "early={early}"
7391            );
7392            std::fs::remove_file(&path).ok();
7393        }
7394    }
7395
7396    /// An implicitly indexed dataset's chunks all exist from create, so an
7397    /// unwritten one reads back as the fill value — and the fill value is
7398    /// tiled over the whole run at create, not per chunk on demand.
7399    #[test]
7400    fn implicit_index_fills_every_chunk_at_create() {
7401        let path = temp_path("implicit_fill");
7402        {
7403            let file = H5File::create(&path).unwrap();
7404            let ds = file
7405                .new_dataset::<i32>()
7406                .shape([8])
7407                .chunk(&[4])
7408                .early_allocation()
7409                .fill_value(-3i32)
7410                .create("data")
7411                .unwrap();
7412            // Only chunk 0.
7413            let chunk: Vec<u8> = [1i32, 2, 3, 4]
7414                .iter()
7415                .flat_map(|v| v.to_le_bytes())
7416                .collect();
7417            ds.write_chunk(0, &chunk).unwrap();
7418            file.close().unwrap();
7419        }
7420        let file = H5File::open(&path).unwrap();
7421        let ds = file.dataset("data").unwrap();
7422        assert_eq!(
7423            ds.read_raw::<i32>().unwrap(),
7424            vec![1, 2, 3, 4, -3, -3, -3, -3]
7425        );
7426        std::fs::remove_file(&path).ok();
7427    }
7428
7429    /// A reopen has to reconstruct the run's address *and* its length from
7430    /// the layout message alone — there is no index structure to read it
7431    /// back from — or the close would rewrite the dataset as unallocated
7432    /// contiguous storage and drop every byte.
7433    #[test]
7434    fn implicit_index_survives_a_reopen() {
7435        let path = temp_path("implicit_reopen");
7436        {
7437            let file = H5File::create(&path).unwrap();
7438            file.new_dataset::<i32>()
7439                .shape([8])
7440                .chunk(&[4])
7441                .early_allocation()
7442                .create("data")
7443                .unwrap()
7444                .write_raw(&[0i32, 1, 2, 3, 4, 5, 6, 7])
7445                .unwrap();
7446            file.close().unwrap();
7447        }
7448        {
7449            let file = H5File::open_rw(&path).unwrap();
7450            let ds = file.dataset_writer("data").unwrap();
7451            let chunk: Vec<u8> = [10i32, 11, 12, 13]
7452                .iter()
7453                .flat_map(|v| v.to_le_bytes())
7454                .collect();
7455            ds.write_chunk(1, &chunk).unwrap();
7456            file.close().unwrap();
7457        }
7458        let file = H5File::open(&path).unwrap();
7459        assert_eq!(
7460            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
7461            vec![0, 1, 2, 3, 10, 11, 12, 13]
7462        );
7463        std::fs::remove_file(&path).ok();
7464    }
7465
7466    #[test]
7467    fn fill_value_chunked_roundtrip() {
7468        let path = temp_path("fill_value_chunked");
7469        {
7470            let file = H5File::create(&path).unwrap();
7471            let ds = file
7472                .new_dataset::<i32>()
7473                .shape([0])
7474                .chunk(&[4])
7475                .max_shape(&[None])
7476                .fill_value(-7i32)
7477                .create("vals")
7478                .unwrap();
7479            ds.append(&[1i32, 2, 3, 4]).unwrap();
7480            file.close().unwrap();
7481        }
7482        {
7483            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7484            let idx = writer.dataset_index("vals").unwrap();
7485            assert_eq!(
7486                writer.ds(idx).lock().fill_value,
7487                Some((-7i32).to_le_bytes().to_vec())
7488            );
7489        }
7490        std::fs::remove_file(&path).ok();
7491    }
7492
7493    #[test]
7494    fn fill_value_read_missing_chunks() {
7495        // A chunked dataset with chunk 1 left unwritten must read that
7496        // gap back as the user-defined fill value, not zero.
7497        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
7498            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
7499        }
7500        let path = temp_path("fill_value_read_missing");
7501        {
7502            let file = H5File::create(&path).unwrap();
7503            let ds = file
7504                .new_dataset::<i32>()
7505                .shape([0])
7506                .chunk(&[2])
7507                .max_shape(&[None])
7508                .fill_value(-1i32)
7509                .create("vals")
7510                .unwrap();
7511            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
7512            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
7513            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
7514            ds.extend(&[6]).unwrap();
7515            file.close().unwrap();
7516        }
7517        {
7518            let file = H5File::open(&path).unwrap();
7519            let ds = file.dataset("vals").unwrap();
7520            let all = ds.read_raw::<i32>().unwrap();
7521            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
7522        }
7523        std::fs::remove_file(&path).ok();
7524    }
7525
7526    #[test]
7527    fn fill_value_partial_chunk_padded_with_fill() {
7528        // A partial trailing chunk flushed at close must pad its unwritten
7529        // tail with the fill value. That pad sits beyond the logical shape,
7530        // so it is verified by scanning the on-disk chunk bytes directly.
7531        let path = temp_path("fill_value_partial_pad");
7532        {
7533            let file = H5File::create(&path).unwrap();
7534            let ds = file
7535                .new_dataset::<i32>()
7536                .shape([0])
7537                .chunk(&[4])
7538                .max_shape(&[None])
7539                .fill_value(-9i32)
7540                .create("vals")
7541                .unwrap();
7542            // 3 of 4 frames -> flushed as a partial chunk on close.
7543            ds.append(&[1i32, 2, 3]).unwrap();
7544            file.close().unwrap();
7545        }
7546        let bytes = std::fs::read(&path).unwrap();
7547        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
7548        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
7549        let pos = bytes
7550            .windows(needle.len())
7551            .position(|w| w == needle)
7552            .expect("chunk data [1,2,3] not found in file");
7553        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
7554        assert_eq!(
7555            pad,
7556            &(-9i32).to_le_bytes(),
7557            "partial chunk tail must be padded with fill value -9, got {:?}",
7558            pad
7559        );
7560        std::fs::remove_file(&path).ok();
7561    }
7562
7563    #[test]
7564    fn vlen_append_after_reopen_preserves_existing() {
7565        // Reopening and appending into a partially-written vlen chunk must
7566        // read-modify-write: the strings already on disk must survive.
7567        let path = temp_path("vlen_append_reopen");
7568        {
7569            let file = H5File::create(&path).unwrap();
7570            file.create_appendable_vlen_dataset("strs", 4, None)
7571                .unwrap();
7572            // 3 of 4 frames -> flushed as a partial chunk on close.
7573            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
7574            file.close().unwrap();
7575        }
7576        {
7577            // Append a 4th string -> partial-chunk write into chunk 0.
7578            let file = H5File::open_rw(&path).unwrap();
7579            file.append_vlen_strings("strs", &["d"]).unwrap();
7580            file.close().unwrap();
7581        }
7582        {
7583            let file = H5File::open(&path).unwrap();
7584            let ds = file.dataset("strs").unwrap();
7585            let got = ds.read_vlen_strings().unwrap();
7586            assert_eq!(
7587                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
7588                vec!["a", "b", "c", "d"]
7589            );
7590        }
7591        std::fs::remove_file(&path).ok();
7592    }
7593
7594    #[test]
7595    fn fill_value_size_mismatch_errors() {
7596        let path = temp_path("fill_value_mismatch");
7597        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
7598        let dt = <f64 as crate::types::H5Type>::hdf5_type();
7599        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
7600        // f64 element size is 8; a 4-byte fill value must be rejected.
7601        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
7602        // The correct width succeeds.
7603        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
7604        writer.close().unwrap();
7605        std::fs::remove_file(&path).ok();
7606    }
7607
7608    #[test]
7609    fn datatype_exposes_class_sign_and_byteorder() {
7610        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
7611        // from f32 (both 4 bytes). datatype() must report the real class and
7612        // signedness so a reader does not have to guess from element_size.
7613        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
7614
7615        let path = temp_path("datatype_accessor");
7616        {
7617            let file = H5File::create(&path).unwrap();
7618            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
7619            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
7620            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
7621            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
7622            file.close().unwrap();
7623        }
7624
7625        let file = H5File::open(&path).unwrap();
7626
7627        match file.dataset("u8d").unwrap().datatype().unwrap() {
7628            DatatypeMessage::FixedPoint {
7629                size,
7630                signed,
7631                byte_order,
7632                ..
7633            } => {
7634                assert_eq!(size, 1);
7635                assert!(!signed, "u8 must be unsigned");
7636                assert_eq!(byte_order, ByteOrder::LittleEndian);
7637            }
7638            other => panic!("expected FixedPoint for u8, got {other:?}"),
7639        }
7640
7641        match file.dataset("i8d").unwrap().datatype().unwrap() {
7642            DatatypeMessage::FixedPoint { size, signed, .. } => {
7643                assert_eq!(size, 1);
7644                assert!(signed, "i8 must be signed");
7645            }
7646            other => panic!("expected FixedPoint for i8, got {other:?}"),
7647        }
7648
7649        match file.dataset("i32d").unwrap().datatype().unwrap() {
7650            DatatypeMessage::FixedPoint { size, signed, .. } => {
7651                assert_eq!(size, 4);
7652                assert!(signed, "i32 must be signed");
7653            }
7654            other => panic!("expected FixedPoint for i32, got {other:?}"),
7655        }
7656
7657        match file.dataset("f32d").unwrap().datatype().unwrap() {
7658            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
7659            other => panic!("expected FloatingPoint for f32, got {other:?}"),
7660        }
7661
7662        std::fs::remove_file(&path).ok();
7663    }
7664
7665    #[test]
7666    fn datatype_in_write_mode_errors() {
7667        let path = temp_path("datatype_write_mode");
7668        let file = H5File::create(&path).unwrap();
7669        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
7670        assert!(ds.datatype().is_err());
7671        std::fs::remove_file(&path).ok();
7672    }
7673
7674    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
7675
7676    /// Extensible-array path: pre-compress with the dataset's pipeline, write
7677    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
7678    /// the data round-trips through the reader unchanged.
7679    #[cfg(feature = "deflate")]
7680    #[test]
7681    fn write_chunk_raw_ea_roundtrip_mask0() {
7682        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7683        let path = temp_path("wcr_ea_mask0");
7684        let original: Vec<i32> = (0..12).collect();
7685        {
7686            let file = H5File::create(&path).unwrap();
7687            let ds = file
7688                .new_dataset::<i32>()
7689                .shape([0])
7690                .chunk(&[4])
7691                .max_shape(&[None])
7692                .deflate(4)
7693                .create("v")
7694                .unwrap();
7695            assert!(ds.is_chunked());
7696            let pipeline = FilterPipeline::deflate(4);
7697            for c in 0..3usize {
7698                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7699                    .iter()
7700                    .flat_map(|v| v.to_le_bytes())
7701                    .collect();
7702                let compressed = apply_filters(&pipeline, &raw).unwrap();
7703                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7704            }
7705            ds.set_extent(&[12]).unwrap();
7706            file.close().unwrap();
7707        }
7708        {
7709            let file = H5File::open(&path).unwrap();
7710            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7711            assert_eq!(v, original);
7712        }
7713        std::fs::remove_file(&path).ok();
7714    }
7715
7716    /// Fixed-array path (all dimensions bounded): same verbatim write through
7717    /// the linear-index dispatch, round-tripped through the reader.
7718    #[cfg(feature = "deflate")]
7719    #[test]
7720    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
7721        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7722        let path = temp_path("wcr_fa_mask0");
7723        let original: Vec<i32> = (0..12).collect();
7724        {
7725            let file = H5File::create(&path).unwrap();
7726            let ds = file
7727                .new_dataset::<i32>()
7728                .shape([12])
7729                .chunk(&[4])
7730                .deflate(4)
7731                .create("v")
7732                .unwrap();
7733            assert!(ds.is_chunked());
7734            let pipeline = FilterPipeline::deflate(4);
7735            for c in 0..3usize {
7736                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
7737                    .iter()
7738                    .flat_map(|v| v.to_le_bytes())
7739                    .collect();
7740                let compressed = apply_filters(&pipeline, &raw).unwrap();
7741                ds.write_chunk_raw(c, &compressed, 0).unwrap();
7742            }
7743            file.close().unwrap();
7744        }
7745        {
7746            let file = H5File::open(&path).unwrap();
7747            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7748            assert_eq!(v, original);
7749        }
7750        std::fs::remove_file(&path).ok();
7751    }
7752
7753    /// The caller-supplied filter_mask must reach the on-disk filtered index
7754    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
7755    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
7756    /// the extensible-array filtered entry to read the mask back at the format
7757    /// level (independent of the data reader's mask handling).
7758    #[cfg(feature = "deflate")]
7759    #[test]
7760    fn write_chunk_raw_records_filter_mask() {
7761        let path = temp_path("wcr_records_mask");
7762        let raw: Vec<u8> = [10i32, 20, 30, 40]
7763            .iter()
7764            .flat_map(|v| v.to_le_bytes())
7765            .collect();
7766        assert_eq!(raw.len(), 16);
7767        {
7768            let file = H5File::create(&path).unwrap();
7769            let ds = file
7770                .new_dataset::<i32>()
7771                .shape([0])
7772                .chunk(&[4])
7773                .max_shape(&[None])
7774                .deflate(4)
7775                .create("v")
7776                .unwrap();
7777            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
7778            // chunk is stored uncompressed (its raw bytes).
7779            ds.write_chunk_raw(0, &raw, 1).unwrap();
7780            ds.set_extent(&[4]).unwrap();
7781            file.close().unwrap();
7782        }
7783        // Reopen the writer; open_append decodes the filtered index block from
7784        // disk, so the entry reflects exactly what was committed.
7785        {
7786            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
7787            let idx = w.dataset_index("v").unwrap();
7788            let ds = w.ds(idx);
7789            let m = ds.lock();
7790            let entry = &m
7791                .chunked
7792                .as_ref()
7793                .unwrap()
7794                .filt_iblk
7795                .as_ref()
7796                .unwrap()
7797                .elements[0];
7798            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
7799            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
7800        }
7801        std::fs::remove_file(&path).ok();
7802    }
7803
7804    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
7805    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
7806    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
7807    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
7808    /// A chunk whose stored stream decodes to less than its image places no
7809    /// run at all: the whole chunk reads as the fill value, whether the read
7810    /// laid the fill down first (a plan that leaves output uncovered — here the
7811    /// unallocated middle chunk) or fills only what nothing wrote. The decode
7812    /// writes into the output image itself, so the bytes a short image leaves
7813    /// behind are the ones this covers.
7814    #[cfg(feature = "deflate")]
7815    #[test]
7816    fn a_chunk_that_decodes_short_of_its_image_reads_as_fill() {
7817        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7818        let path = temp_path("short_chunk_image_is_fill");
7819        let pipeline = FilterPipeline::deflate(4);
7820        {
7821            let file = H5File::create(&path).unwrap();
7822            let ds = file
7823                .new_dataset::<i32>()
7824                .shape([0])
7825                .chunk(&[4])
7826                .max_shape(&[None])
7827                .fill_value(-1i32)
7828                .deflate(4)
7829                .create("v")
7830                .unwrap();
7831            // Chunk 0 carries two elements where its image wants four.
7832            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7833            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7834                .unwrap();
7835            // Chunk 1 is never written; chunk 2 is whole.
7836            let whole: Vec<u8> = [9i32, 10, 11, 12]
7837                .iter()
7838                .flat_map(|v| v.to_le_bytes())
7839                .collect();
7840            ds.write_chunk_raw(2, &apply_filters(&pipeline, &whole).unwrap(), 0)
7841                .unwrap();
7842            ds.set_extent(&[12]).unwrap();
7843            file.close().unwrap();
7844        }
7845        {
7846            let file = H5File::open(&path).unwrap();
7847            let ds = file.dataset("v").unwrap();
7848            assert_eq!(
7849                ds.read_raw::<i32>().unwrap(),
7850                vec![-1, -1, -1, -1, -1, -1, -1, -1, 9, 10, 11, 12]
7851            );
7852            // The same verdict when the plan covers every output byte, so no
7853            // fill goes down first: chunks 0 and 2 alone.
7854            assert_eq!(ds.read_slice::<i32>(&[0], &[4]).unwrap(), vec![-1; 4]);
7855            assert_eq!(
7856                ds.read_slice::<i32>(&[8], &[4]).unwrap(),
7857                vec![9, 10, 11, 12]
7858            );
7859        }
7860        std::fs::remove_file(&path).ok();
7861    }
7862
7863    /// The staged spelling of the case above: a selection that takes only part
7864    /// of the short chunk decodes it into a buffer sized from the layout, and
7865    /// what that buffer holds past the stream is cut off rather than kept — a
7866    /// run inside the decoded bytes is real data, a run reaching past them is
7867    /// fill.
7868    #[cfg(feature = "deflate")]
7869    #[test]
7870    fn a_staged_chunk_carries_only_what_its_stream_decoded() {
7871        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7872        let path = temp_path("short_chunk_image_staged");
7873        let pipeline = FilterPipeline::deflate(4);
7874        {
7875            let file = H5File::create(&path).unwrap();
7876            let ds = file
7877                .new_dataset::<i32>()
7878                .shape([0])
7879                .chunk(&[4])
7880                .max_shape(&[None])
7881                .fill_value(-1i32)
7882                .deflate(4)
7883                .create("v")
7884                .unwrap();
7885            let short: Vec<u8> = [7i32, 8].iter().flat_map(|v| v.to_le_bytes()).collect();
7886            ds.write_chunk_raw(0, &apply_filters(&pipeline, &short).unwrap(), 0)
7887                .unwrap();
7888            ds.set_extent(&[4]).unwrap();
7889            file.close().unwrap();
7890        }
7891        {
7892            let file = H5File::open(&path).unwrap();
7893            let ds = file.dataset("v").unwrap();
7894            // Inside the decoded bytes.
7895            assert_eq!(ds.read_slice::<i32>(&[0], &[2]).unwrap(), vec![7, 8]);
7896            // Straddling their end: the run is not placed at all.
7897            assert_eq!(ds.read_slice::<i32>(&[1], &[2]).unwrap(), vec![-1, -1]);
7898        }
7899        std::fs::remove_file(&path).ok();
7900    }
7901
7902    #[cfg(feature = "deflate")]
7903    #[test]
7904    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
7905        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7906        let path = temp_path("wcr_ea_per_chunk_mask");
7907        let original: Vec<i32> = (0..8).collect();
7908        let pipeline = FilterPipeline::deflate(4);
7909        {
7910            let file = H5File::create(&path).unwrap();
7911            let ds = file
7912                .new_dataset::<i32>()
7913                .shape([0])
7914                .chunk(&[4])
7915                .max_shape(&[None])
7916                .deflate(4)
7917                .create("v")
7918                .unwrap();
7919            let raw0: Vec<u8> = original[0..4]
7920                .iter()
7921                .flat_map(|v| v.to_le_bytes())
7922                .collect();
7923            // chunk 0: compressed through the pipeline, mask 0.
7924            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7925                .unwrap();
7926            let raw1: Vec<u8> = original[4..8]
7927                .iter()
7928                .flat_map(|v| v.to_le_bytes())
7929                .collect();
7930            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
7931            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7932            ds.set_extent(&[8]).unwrap();
7933            file.close().unwrap();
7934        }
7935        {
7936            let file = H5File::open(&path).unwrap();
7937            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7938            assert_eq!(v, original);
7939        }
7940        std::fs::remove_file(&path).ok();
7941    }
7942
7943    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
7944    /// compressed/raw chunks as the EA case, through the fixed-array index.
7945    #[cfg(feature = "deflate")]
7946    #[test]
7947    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
7948        use crate::format::messages::filter::{apply_filters, FilterPipeline};
7949        let path = temp_path("wcr_fa_per_chunk_mask");
7950        let original: Vec<i32> = (0..8).collect();
7951        let pipeline = FilterPipeline::deflate(4);
7952        {
7953            let file = H5File::create(&path).unwrap();
7954            let ds = file
7955                .new_dataset::<i32>()
7956                .shape([8])
7957                .chunk(&[4])
7958                .deflate(4)
7959                .create("v")
7960                .unwrap();
7961            let raw0: Vec<u8> = original[0..4]
7962                .iter()
7963                .flat_map(|v| v.to_le_bytes())
7964                .collect();
7965            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
7966                .unwrap();
7967            let raw1: Vec<u8> = original[4..8]
7968                .iter()
7969                .flat_map(|v| v.to_le_bytes())
7970                .collect();
7971            ds.write_chunk_raw(1, &raw1, 1).unwrap();
7972            file.close().unwrap();
7973        }
7974        {
7975            let file = H5File::open(&path).unwrap();
7976            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
7977            assert_eq!(v, original);
7978        }
7979        std::fs::remove_file(&path).ok();
7980    }
7981
7982    /// An unfiltered chunk index has no slot for a stored size or mask, so a
7983    /// direct chunk write must be rejected rather than silently dropping them.
7984    #[test]
7985    fn write_chunk_raw_rejects_unfiltered() {
7986        let path = temp_path("wcr_unfiltered");
7987        let file = H5File::create(&path).unwrap();
7988        let ds = file
7989            .new_dataset::<i32>()
7990            .shape([0])
7991            .chunk(&[4])
7992            .max_shape(&[None])
7993            .create("v")
7994            .unwrap();
7995        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
7996        assert!(
7997            err.to_string().contains("filtered dataset"),
7998            "expected a filtered-dataset error, got: {err}"
7999        );
8000        std::fs::remove_file(&path).ok();
8001    }
8002
8003    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
8004    /// index to mean anything against, so the linear entry point points the
8005    /// caller at the coordinate-addressed one rather than guessing a grid.
8006    #[test]
8007    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
8008        let path = temp_path("wcr_btree2");
8009        let file = H5File::create(&path).unwrap();
8010        let ds = file
8011            .new_dataset::<i32>()
8012            .shape([0, 0])
8013            .chunk(&[2, 2])
8014            .max_shape(&[None, None])
8015            .create("grid")
8016            .unwrap();
8017        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
8018        assert!(
8019            err.to_string().contains("write_chunk_raw_at"),
8020            "expected a pointer to the coordinate form, got: {err}"
8021        );
8022        std::fs::remove_file(&path).ok();
8023    }
8024
8025    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
8026    /// and the type-11 record carries their size and the caller's mask, so a
8027    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
8028    /// bytes while one written compressed (mask 0) is decompressed.
8029    #[cfg(feature = "deflate")]
8030    #[test]
8031    fn write_chunk_raw_at_round_trips_on_btree_v2() {
8032        use crate::format::messages::filter::{apply_filters, FilterPipeline};
8033
8034        let path = temp_path("wcr_at_btree2");
8035        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
8036        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
8037        {
8038            let file = H5File::create(&path).unwrap();
8039            let ds = file
8040                .new_dataset::<i32>()
8041                .shape([0, 0])
8042                .chunk(&[2, 2])
8043                .max_shape(&[None, None])
8044                .deflate(6)
8045                .create("grid")
8046                .unwrap();
8047            let pipeline = FilterPipeline::deflate(6);
8048            // Chunk (0,0): pipeline already applied upstream, mask 0.
8049            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
8050                .unwrap();
8051            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
8052            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
8053            file.close().unwrap();
8054        }
8055        let file = H5File::open(&path).unwrap();
8056        let ds = file.dataset("grid").unwrap();
8057        assert_eq!(ds.shape(), vec![4, 4]);
8058        let all = ds.read_raw::<i32>().unwrap();
8059        // Chunk (0,0) occupies rows 0..2, columns 0..2.
8060        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
8061        // Chunk (1,1) occupies rows 2..4, columns 2..4.
8062        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
8063        drop(file);
8064        std::fs::remove_file(&path).ok();
8065    }
8066
8067    /// The coordinate form is not BT2-only: it addresses an extensible- or
8068    /// fixed-array dataset's grid just as well, and records the same mask.
8069    #[cfg(feature = "deflate")]
8070    #[test]
8071    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
8072        for (label, max_shape) in [
8073            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
8074            ("wcr_at_fa", None),
8075        ] {
8076            let path = temp_path(label);
8077            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
8078            {
8079                let file = H5File::create(&path).unwrap();
8080                let mut b = file
8081                    .new_dataset::<i32>()
8082                    .shape([4usize, 4])
8083                    .chunk(&[2, 4])
8084                    .deflate(6);
8085                if let Some(ref ms) = max_shape {
8086                    b = b.max_shape(ms);
8087                }
8088                let ds = b.create("grid").unwrap();
8089                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
8090                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
8091                file.close().unwrap();
8092            }
8093            let file = H5File::open(&path).unwrap();
8094            let ds = file.dataset("grid").unwrap();
8095            let all = ds.read_raw::<i32>().unwrap();
8096            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
8097            drop(file);
8098            std::fs::remove_file(&path).ok();
8099        }
8100    }
8101
8102    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
8103    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
8104    /// chunk gives chunk_size_len = 2 (max 65535).
8105    #[cfg(feature = "deflate")]
8106    #[test]
8107    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
8108        let path = temp_path("wcr_at_oversized");
8109        let file = H5File::create(&path).unwrap();
8110        let ds = file
8111            .new_dataset::<i32>()
8112            .shape([0, 0])
8113            .chunk(&[1, 1])
8114            .max_shape(&[None, None])
8115            .deflate(4)
8116            .create("grid")
8117            .unwrap();
8118        let err = ds
8119            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
8120            .unwrap_err();
8121        assert!(
8122            err.to_string().contains("does not fit"),
8123            "expected a chunk-size-field overflow error, got: {err}"
8124        );
8125        std::fs::remove_file(&path).ok();
8126    }
8127
8128    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
8129    /// the same reason the array indexes reject a direct write.
8130    #[test]
8131    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
8132        let path = temp_path("wcr_at_unfiltered");
8133        let file = H5File::create(&path).unwrap();
8134        let ds = file
8135            .new_dataset::<i32>()
8136            .shape([0, 0])
8137            .chunk(&[2, 2])
8138            .max_shape(&[None, None])
8139            .create("grid")
8140            .unwrap();
8141        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
8142        assert!(
8143            err.to_string().contains("filtered dataset"),
8144            "expected a filtered-dataset error, got: {err}"
8145        );
8146        std::fs::remove_file(&path).ok();
8147    }
8148
8149    /// A stored size that does not fit the index's chunk-size field must error
8150    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
8151    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
8152    /// a 70000-byte stored chunk overflows it.
8153    #[cfg(feature = "deflate")]
8154    #[test]
8155    fn write_chunk_raw_rejects_oversized_chunk() {
8156        let path = temp_path("wcr_oversized");
8157        let file = H5File::create(&path).unwrap();
8158        let ds = file
8159            .new_dataset::<i32>()
8160            .shape([0])
8161            .chunk(&[1])
8162            .max_shape(&[None])
8163            .deflate(4)
8164            .create("v")
8165            .unwrap();
8166        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
8167        assert!(
8168            err.to_string().contains("does not fit"),
8169            "expected a chunk-size-field overflow error, got: {err}"
8170        );
8171        std::fs::remove_file(&path).ok();
8172    }
8173
8174    // ---- issue #5: runtime-width fixed-string reading ----------------------
8175
8176    use crate::format::messages::datatype::{CompoundMember, DatatypeMessage};
8177
8178    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
8179    /// element images, optionally chunked and deflated.
8180    fn write_fixed_string_dataset(
8181        path: &std::path::Path,
8182        dt: DatatypeMessage,
8183        width: usize,
8184        elems: &[&[u8]],
8185        compressed: bool,
8186    ) {
8187        let mut raw = Vec::with_capacity(elems.len() * width);
8188        for e in elems {
8189            assert!(e.len() <= width);
8190            raw.extend_from_slice(e);
8191            raw.resize(raw.len() + (width - e.len()), 0);
8192        }
8193        let file = H5File::create(path).unwrap();
8194        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
8195        if compressed {
8196            b = b.chunk(&[2]).deflate(6);
8197        }
8198        let ds = b.create("labels").unwrap();
8199        ds.write_raw_bytes(&raw).unwrap();
8200        file.close().unwrap();
8201    }
8202
8203    /// The width is whatever the file says, so one call reads a 24-byte label
8204    /// column and a 100-byte one. Producers like VASP pick it per dataset.
8205    #[test]
8206    fn read_strings_handles_any_fixed_width() {
8207        for width in [4usize, 24, 100] {
8208            let path = temp_path(&format!("fixed_str_{width}"));
8209            write_fixed_string_dataset(
8210                &path,
8211                DatatypeMessage::fixed_string(width as u32),
8212                width,
8213                &[b"ab", b"cde", b""],
8214                false,
8215            );
8216            let file = H5File::open(&path).unwrap();
8217            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8218            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
8219            std::fs::remove_file(&path).ok();
8220        }
8221    }
8222
8223    /// Each padding rule decides where the value ends. Null-terminated and
8224    /// null-padded both stop at the first NUL and ignore the bytes after it;
8225    /// space-padded strips only a tail of spaces, so an embedded NUL is
8226    /// content there.
8227    ///
8228    /// Checked against libhdf5 1.14.6: reading this same `"ab\0X\0\0"`
8229    /// null-padded element into a wider null-terminated destination gives
8230    /// `"ab"`, and reading a space-padded `"a\0b     "` gives `"a\0b"` —
8231    /// `H5T__conv_s_s` runs the same `!s[nchars]` loop for both null rules.
8232    #[test]
8233    fn read_strings_honors_every_padding_rule() {
8234        // "ab" then a NUL then trailing junk that both null rules must drop.
8235        let elem: &[u8] = b"ab\0X\0\0";
8236        for (padding, want) in [(0u8, "ab"), (1, "ab")] {
8237            let path = temp_path(&format!("fixed_pad_{padding}"));
8238            write_fixed_string_dataset(
8239                &path,
8240                DatatypeMessage::FixedString {
8241                    size: 6,
8242                    padding,
8243                    charset: 0,
8244                },
8245                6,
8246                &[elem],
8247                false,
8248            );
8249            let file = H5File::open(&path).unwrap();
8250            let got = file.dataset("labels").unwrap().read_strings().unwrap();
8251            assert_eq!(got, vec![want.to_string()], "padding {padding}");
8252            std::fs::remove_file(&path).ok();
8253        }
8254        // Space-padded keeps interior spaces and strips only the tail.
8255        let path = temp_path("fixed_pad_2");
8256        write_fixed_string_dataset(
8257            &path,
8258            DatatypeMessage::FixedString {
8259                size: 8,
8260                padding: 2,
8261                charset: 0,
8262            },
8263            8,
8264            &[b"a b     "],
8265            false,
8266        );
8267        let file = H5File::open(&path).unwrap();
8268        assert_eq!(
8269            file.dataset("labels").unwrap().read_strings().unwrap(),
8270            vec!["a b".to_string()]
8271        );
8272        std::fs::remove_file(&path).ok();
8273
8274        // ... and an embedded NUL, which no space rule marks as an end.
8275        let path = temp_path("fixed_pad_2_nul");
8276        write_fixed_string_dataset(
8277            &path,
8278            DatatypeMessage::FixedString {
8279                size: 8,
8280                padding: 2,
8281                charset: 0,
8282            },
8283            8,
8284            &[b"a\0b     "],
8285            false,
8286        );
8287        let file = H5File::open(&path).unwrap();
8288        assert_eq!(
8289            file.dataset("labels").unwrap().read_strings().unwrap(),
8290            vec!["a\0b".to_string()]
8291        );
8292        std::fs::remove_file(&path).ok();
8293    }
8294
8295    /// A reserved padding or character-set code is an error naming the element,
8296    /// not a guess.
8297    #[test]
8298    fn read_strings_rejects_reserved_datatype_codes() {
8299        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
8300            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
8301            write_fixed_string_dataset(
8302                &path,
8303                DatatypeMessage::FixedString {
8304                    size: 4,
8305                    padding,
8306                    charset,
8307                },
8308                4,
8309                &[b"ab"],
8310                false,
8311            );
8312            let file = H5File::open(&path).unwrap();
8313            let err = file
8314                .dataset("labels")
8315                .unwrap()
8316                .read_strings()
8317                .unwrap_err()
8318                .to_string();
8319            assert!(err.contains(want), "got: {err}");
8320            std::fs::remove_file(&path).ok();
8321        }
8322    }
8323
8324    /// The typed read paths reinterpret the element image, so the stored order
8325    /// has to be the host's first. A scalar is swapped; a composite cannot be
8326    /// (its members have their own orders and offsets) and is refused.
8327    #[test]
8328    fn to_host_byte_order_converts_scalars_and_refuses_composites() {
8329        use crate::dataset::{to_host_byte_order, HOST_BYTE_ORDER};
8330        use crate::format::messages::datatype::ByteOrder;
8331
8332        let foreign = match HOST_BYTE_ORDER {
8333            ByteOrder::LittleEndian => ByteOrder::BigEndian,
8334            ByteOrder::BigEndian => ByteOrder::LittleEndian,
8335        };
8336        let int = |order, size| DatatypeMessage::FixedPoint {
8337            size,
8338            byte_order: order,
8339            signed: false,
8340            bit_offset: 0,
8341            bit_precision: (size * 8) as u16,
8342        };
8343
8344        // Foreign order: each element is reversed, elementwise.
8345        let mut buf = [1u8, 2, 3, 4, 5, 6, 7, 8];
8346        to_host_byte_order(&mut buf, &int(foreign, 4), 4).unwrap();
8347        assert_eq!(buf, [4, 3, 2, 1, 8, 7, 6, 5]);
8348
8349        // Host order: untouched.
8350        let mut buf = [1u8, 2, 3, 4];
8351        to_host_byte_order(&mut buf, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8352        assert_eq!(buf, [1, 2, 3, 4]);
8353
8354        // One byte wide: no order to convert.
8355        let mut buf = [1u8, 2, 3, 4];
8356        to_host_byte_order(&mut buf, &int(foreign, 1), 1).unwrap();
8357        assert_eq!(buf, [1, 2, 3, 4]);
8358
8359        // An enum stores its values in its base type's order.
8360        let mut buf = [1u8, 2];
8361        let enumeration = DatatypeMessage::Enum {
8362            base: Box::new(int(foreign, 2)),
8363            members: Vec::new(),
8364        };
8365        to_host_byte_order(&mut buf, &enumeration, 2).unwrap();
8366        assert_eq!(buf, [2, 1]);
8367
8368        // A string has no byte order at all.
8369        let mut buf = *b"abcd";
8370        to_host_byte_order(&mut buf, &DatatypeMessage::fixed_string(4), 4).unwrap();
8371        assert_eq!(&buf, b"abcd");
8372
8373        // A compound whose members are all host-order is reinterpretable.
8374        let compound = |order| DatatypeMessage::Compound {
8375            size: 4,
8376            members: vec![CompoundMember {
8377                name: "x".into(),
8378                offset: 0,
8379                datatype: int(order, 4),
8380            }],
8381        };
8382        let mut buf = [1u8, 2, 3, 4];
8383        to_host_byte_order(&mut buf, &compound(HOST_BYTE_ORDER), 4).unwrap();
8384        assert_eq!(buf, [1, 2, 3, 4]);
8385
8386        // One that is not says so, rather than handing back the raw bytes.
8387        let mut buf = [1u8, 2, 3, 4];
8388        let err = to_host_byte_order(&mut buf, &compound(foreign), 4)
8389            .expect_err("a foreign-order compound was reinterpreted")
8390            .to_string();
8391        assert!(err.contains("read_raw_bytes"), "got: {err}");
8392        assert_eq!(buf, [1, 2, 3, 4], "the refused image is left alone");
8393    }
8394
8395    /// The write direction answers for exactly the types the read direction
8396    /// does — same classifier — and borrows the caller's bytes whenever the
8397    /// declared order is already the host's.
8398    #[test]
8399    fn to_stored_byte_order_converts_scalars_and_refuses_composites() {
8400        use crate::dataset::{to_stored_byte_order, FOREIGN_BYTE_ORDER, HOST_BYTE_ORDER};
8401        use std::borrow::Cow;
8402
8403        let int = |order, size| DatatypeMessage::FixedPoint {
8404            size,
8405            byte_order: order,
8406            signed: false,
8407            bit_offset: 0,
8408            bit_precision: (size * 8) as u16,
8409        };
8410
8411        // Declared foreign: each element is reversed on the way out.
8412        let host = [1u8, 2, 3, 4, 5, 6, 7, 8];
8413        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 4), 4).unwrap();
8414        assert_eq!(&*stored, &[4, 3, 2, 1, 8, 7, 6, 5]);
8415        assert!(matches!(stored, Cow::Owned(_)), "a swap needs its own copy");
8416
8417        // Declared host order: handed through without a copy.
8418        let stored = to_stored_byte_order(&host, &int(HOST_BYTE_ORDER, 4), 4).unwrap();
8419        assert!(matches!(stored, Cow::Borrowed(_)), "no copy without a swap");
8420        assert_eq!(&*stored, &host);
8421
8422        // One byte wide: no order to lay out.
8423        let stored = to_stored_byte_order(&host, &int(FOREIGN_BYTE_ORDER, 1), 1).unwrap();
8424        assert_eq!(&*stored, &host);
8425
8426        // An enum stores its values in its base type's order.
8427        let enumeration = DatatypeMessage::Enum {
8428            base: Box::new(int(FOREIGN_BYTE_ORDER, 2)),
8429            members: Vec::new(),
8430        };
8431        let stored = to_stored_byte_order(&[1u8, 2], &enumeration, 2).unwrap();
8432        assert_eq!(&*stored, &[2, 1]);
8433
8434        // A compound cannot be laid out as a unit; one that declares the
8435        // foreign order for a member is refused, not written host-order.
8436        let compound = |order| DatatypeMessage::Compound {
8437            size: 4,
8438            members: vec![CompoundMember {
8439                name: "x".into(),
8440                offset: 0,
8441                datatype: int(order, 4),
8442            }],
8443        };
8444        let stored = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(HOST_BYTE_ORDER), 4).unwrap();
8445        assert_eq!(&*stored, &[1, 2, 3, 4]);
8446        let err = to_stored_byte_order(&[1u8, 2, 3, 4], &compound(FOREIGN_BYTE_ORDER), 4)
8447            .expect_err("a foreign-order compound was written from host bytes")
8448            .to_string();
8449        assert!(err.contains("write_raw_bytes"), "got: {err}");
8450    }
8451
8452    /// The declared character set is enforced: a byte that cannot be decoded is
8453    /// an error naming the element, and the lossy call is what accepts the file
8454    /// instead of a silent substitution here.
8455    #[test]
8456    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
8457        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
8458        // in one that declares UTF-8.
8459        for (charset, bytes, want) in [
8460            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
8461            (1, b"a\xff".as_slice(), "not valid UTF-8"),
8462        ] {
8463            let path = temp_path(&format!("fixed_charset_{charset}"));
8464            write_fixed_string_dataset(
8465                &path,
8466                DatatypeMessage::FixedString {
8467                    size: 6,
8468                    padding: 1,
8469                    charset,
8470                },
8471                6,
8472                &[b"ok", bytes],
8473                false,
8474            );
8475            let file = H5File::open(&path).unwrap();
8476            let ds = file.dataset("labels").unwrap();
8477            let err = ds.read_strings().unwrap_err().to_string();
8478            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
8479            let lossy = ds.read_strings_lossy().unwrap();
8480            assert_eq!(lossy[0], "ok");
8481            assert_eq!(
8482                lossy[1].chars().next().unwrap(),
8483                if charset == 0 { 'c' } else { 'a' }
8484            );
8485            std::fs::remove_file(&path).ok();
8486        }
8487    }
8488
8489    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
8490    /// split a character.
8491    #[test]
8492    fn read_strings_reads_utf8_fixed_strings() {
8493        let path = temp_path("fixed_utf8");
8494        write_fixed_string_dataset(
8495            &path,
8496            DatatypeMessage::fixed_string_utf8(12),
8497            12,
8498            &["héllo".as_bytes(), "안녕".as_bytes()],
8499            false,
8500        );
8501        let file = H5File::open(&path).unwrap();
8502        assert_eq!(
8503            file.dataset("labels").unwrap().read_strings().unwrap(),
8504            vec!["héllo".to_string(), "안녕".to_string()]
8505        );
8506        std::fs::remove_file(&path).ok();
8507    }
8508
8509    /// The decode sits on the decoded raw-data path, so a chunked and deflated
8510    /// dataset reads the same as a contiguous one.
8511    #[cfg(feature = "deflate")]
8512    #[test]
8513    fn read_strings_reads_a_compressed_fixed_string_dataset() {
8514        let path = temp_path("fixed_str_deflate");
8515        write_fixed_string_dataset(
8516            &path,
8517            DatatypeMessage::fixed_string(16),
8518            16,
8519            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
8520            true,
8521        );
8522        let file = H5File::open(&path).unwrap();
8523        assert_eq!(
8524            file.dataset("labels").unwrap().read_strings().unwrap(),
8525            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
8526        );
8527        std::fs::remove_file(&path).ok();
8528    }
8529
8530    /// One call covers both string datatypes, so a caller need not branch on
8531    /// which one the file used.
8532    #[test]
8533    fn read_strings_also_reads_variable_length_strings() {
8534        let path = temp_path("read_strings_vlen");
8535        {
8536            let file = H5File::create(&path).unwrap();
8537            file.write_vlen_strings("names", &["alpha", "", "안녕"])
8538                .unwrap();
8539            file.close().unwrap();
8540        }
8541        let file = H5File::open(&path).unwrap();
8542        assert_eq!(
8543            file.dataset("names").unwrap().read_strings().unwrap(),
8544            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
8545        );
8546        std::fs::remove_file(&path).ok();
8547    }
8548
8549    /// A file declaring a zero-width fixed string is an error, not the panic
8550    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
8551    /// test patches the width in the encoded datatype message down to zero and
8552    /// re-stamps the object header's checksum over the result.
8553    #[test]
8554    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
8555        use crate::format::checksum::checksum_metadata;
8556        use crate::format::object_header::OHDR_SIGNATURE;
8557
8558        let path = temp_path("fixed_str_zero_width");
8559        write_fixed_string_dataset(
8560            &path,
8561            DatatypeMessage::fixed_string(37),
8562            37,
8563            &[b"ab", b"cd"],
8564            false,
8565        );
8566
8567        // Version 1 string datatype: class|version, padding|charset, two
8568        // reserved bytes, then the width as a little-endian u32. The width is
8569        // 37 so the eight bytes occur once in the file.
8570        let mut bytes = std::fs::read(&path).unwrap();
8571        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
8572        let at = bytes
8573            .windows(needle.len())
8574            .position(|w| w == needle)
8575            .expect("encoded fixed-string datatype message");
8576        assert!(
8577            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
8578            "the datatype message pattern is not unique in the file"
8579        );
8580
8581        // The enclosing v2 object header ends in a checksum over everything
8582        // from its signature onwards; find the offset where the stored value
8583        // still agrees, so the patched header can be re-stamped there.
8584        let ohdr = bytes[..at]
8585            .windows(4)
8586            .rposition(|w| w == OHDR_SIGNATURE)
8587            .expect("enclosing object header");
8588        let cksum_at = (at + needle.len()..bytes.len() - 4)
8589            .find(|&e| {
8590                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
8591                    == checksum_metadata(&bytes[ohdr..e])
8592            })
8593            .expect("object header checksum");
8594
8595        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
8596        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
8597        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
8598        std::fs::write(&path, &bytes).unwrap();
8599
8600        let file = H5File::open(&path).unwrap();
8601        let err = file
8602            .dataset("labels")
8603            .unwrap()
8604            .read_strings()
8605            .unwrap_err()
8606            .to_string();
8607        assert!(err.contains("zero width"), "got: {err}");
8608        std::fs::remove_file(&path).ok();
8609    }
8610
8611    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
8612    #[test]
8613    fn read_strings_rejects_a_non_string_dataset() {
8614        let path = temp_path("read_strings_numeric");
8615        {
8616            let file = H5File::create(&path).unwrap();
8617            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8618            ds.write_raw(&[1i32, 2, 3]).unwrap();
8619            file.close().unwrap();
8620        }
8621        let file = H5File::open(&path).unwrap();
8622        let err = file
8623            .dataset("nums")
8624            .unwrap()
8625            .read_strings()
8626            .unwrap_err()
8627            .to_string();
8628        assert!(err.contains("only for string datasets"), "got: {err}");
8629        std::fs::remove_file(&path).ok();
8630    }
8631
8632    // ---- issue #6: random updates to vlen string datasets ------------------
8633
8634    /// One element changes; the extent and every other element stay as they
8635    /// were, on a contiguous vlen dataset.
8636    #[test]
8637    fn write_vlen_strings_slice_replaces_one_element() {
8638        let path = temp_path("vlen_slice_contig");
8639        {
8640            let file = H5File::create(&path).unwrap();
8641            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8642                .unwrap();
8643            file.close().unwrap();
8644        }
8645        {
8646            let file = H5File::open_rw(&path).unwrap();
8647            file.dataset_writer("notes")
8648                .unwrap()
8649                .write_vlen_strings_slice(1, &["replacement"])
8650                .unwrap();
8651            file.close().unwrap();
8652        }
8653        let file = H5File::open(&path).unwrap();
8654        let ds = file.dataset("notes").unwrap();
8655        assert_eq!(ds.shape(), vec![4]);
8656        assert_eq!(
8657            ds.read_vlen_strings().unwrap(),
8658            vec!["a", "replacement", "c", "d"]
8659        );
8660        std::fs::remove_file(&path).ok();
8661    }
8662
8663    /// The same on an appendable chunked dataset, across a reopen, over a range
8664    /// that spans a chunk boundary.
8665    #[test]
8666    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
8667        let path = temp_path("vlen_slice_chunked");
8668        {
8669            let file = H5File::create(&path).unwrap();
8670            file.create_appendable_vlen_dataset("notes", 2, None)
8671                .unwrap();
8672            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
8673            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
8674            file.append_vlen_strings("notes", &refs).unwrap();
8675            file.close().unwrap();
8676        }
8677        {
8678            // Elements 1..4 cross the 2-element chunk boundary twice.
8679            let file = H5File::open_rw(&path).unwrap();
8680            file.dataset_writer("notes")
8681                .unwrap()
8682                .write_vlen_strings_slice(1, &["x", "y", "z"])
8683                .unwrap();
8684            file.close().unwrap();
8685        }
8686        let file = H5File::open(&path).unwrap();
8687        let ds = file.dataset("notes").unwrap();
8688        assert_eq!(ds.shape(), vec![6]);
8689        assert_eq!(
8690            ds.read_vlen_strings().unwrap(),
8691            vec!["v0", "x", "y", "z", "v4", "v5"]
8692        );
8693        std::fs::remove_file(&path).ok();
8694    }
8695
8696    /// Elements the append buffer still holds are not on disk yet; the
8697    /// update flushes them to their chunks first, so the flush at close has
8698    /// nothing left to write the pre-update reference over.
8699    #[test]
8700    fn write_vlen_strings_slice_updates_buffered_elements() {
8701        let path = temp_path("vlen_slice_buffered");
8702        {
8703            let file = H5File::create(&path).unwrap();
8704            file.create_appendable_vlen_dataset("notes", 4, None)
8705                .unwrap();
8706            // 3 of a 4-element chunk: all three stay in the append buffer.
8707            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
8708            file.dataset_writer("notes")
8709                .unwrap()
8710                .write_vlen_strings_slice(1, &["patched"])
8711                .unwrap();
8712            file.close().unwrap();
8713        }
8714        let file = H5File::open(&path).unwrap();
8715        assert_eq!(
8716            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8717            vec!["a", "patched", "c"]
8718        );
8719        std::fs::remove_file(&path).ok();
8720    }
8721
8722    /// A range past the end is rejected before anything is written, and an
8723    /// empty batch costs the file nothing — without the early return it would
8724    /// still allocate and write an empty global-heap collection.
8725    #[test]
8726    fn write_vlen_strings_slice_checks_its_range() {
8727        let build = |name: &str, empty_call: bool| {
8728            let path = temp_path(name);
8729            let file = H5File::create(&path).unwrap();
8730            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8731            let ds = file.dataset_writer("notes").unwrap();
8732            let err = ds
8733                .write_vlen_strings_slice(1, &["x", "y"])
8734                .unwrap_err()
8735                .to_string();
8736            assert!(
8737                err.contains("outside the dataset's 2 elements"),
8738                "got: {err}"
8739            );
8740            if empty_call {
8741                ds.write_vlen_strings_slice(0, &[]).unwrap();
8742            }
8743            file.close().unwrap();
8744            path
8745        };
8746
8747        let with_empty = build("vlen_slice_range", true);
8748        let control = build("vlen_slice_range_control", false);
8749        assert_eq!(
8750            std::fs::metadata(&with_empty).unwrap().len(),
8751            std::fs::metadata(&control).unwrap().len(),
8752            "the rejected and empty calls must leave the file untouched"
8753        );
8754
8755        let file = H5File::open(&with_empty).unwrap();
8756        assert_eq!(
8757            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8758            vec!["a", "b"]
8759        );
8760        std::fs::remove_file(&with_empty).ok();
8761        std::fs::remove_file(&control).ok();
8762    }
8763
8764    /// The element offset is one-dimensional, so a multi-dimensional dataset is
8765    /// rejected rather than silently indexed along the first axis.
8766    #[test]
8767    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
8768        let path = temp_path("vlen_slice_2d");
8769        let file = H5File::create(&path).unwrap();
8770        let ds = file
8771            .new_dataset::<u8>()
8772            .datatype(DatatypeMessage::vlen_string_utf8())
8773            .shape([2, 3])
8774            .create("grid")
8775            .unwrap();
8776        let err = ds
8777            .write_vlen_strings_slice(0, &["x"])
8778            .unwrap_err()
8779            .to_string();
8780        assert!(err.contains("1-dimension datasets"), "got: {err}");
8781        file.close().unwrap();
8782        std::fs::remove_file(&path).ok();
8783    }
8784
8785    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
8786    /// declares the ASCII character set would mislabel the bytes.
8787    #[test]
8788    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
8789        let path = temp_path("vlen_slice_ascii");
8790        let file = H5File::create(&path).unwrap();
8791        let ds = file
8792            .new_dataset::<u8>()
8793            .datatype(DatatypeMessage::vlen_string_ascii())
8794            .shape([3])
8795            .create("notes")
8796            .unwrap();
8797        let err = ds
8798            .write_vlen_strings_slice(0, &["ok", "안녕"])
8799            .unwrap_err()
8800            .to_string();
8801        assert!(
8802            err.contains("string 1") && err.contains("is not ASCII"),
8803            "got: {err}"
8804        );
8805        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
8806        file.close().unwrap();
8807        std::fs::remove_file(&path).ok();
8808    }
8809
8810    /// A numeric dataset is rejected: its elements are not vlen references and
8811    /// writing one would corrupt the column.
8812    #[test]
8813    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
8814        let path = temp_path("vlen_slice_numeric");
8815        let file = H5File::create(&path).unwrap();
8816        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
8817        ds.write_raw(&[1i32, 2, 3]).unwrap();
8818        let err = ds
8819            .write_vlen_strings_slice(0, &["x"])
8820            .unwrap_err()
8821            .to_string();
8822        assert!(
8823            err.contains("only for variable-length string datasets"),
8824            "got: {err}"
8825        );
8826        file.close().unwrap();
8827        std::fs::remove_file(&path).ok();
8828    }
8829
8830    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
8831
8832    /// Repeatedly replacing the same element must not grow the file per
8833    /// update: the collection each update supersedes is freed and the next
8834    /// update's collection lands in that block. Without the release every
8835    /// update costs another `H5HG_MINALLOC` (4096) bytes.
8836    #[test]
8837    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
8838        let size_after = |updates: usize| {
8839            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
8840            let file = H5File::create(&path).unwrap();
8841            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
8842            let ds = file.dataset_writer("notes").unwrap();
8843            for i in 0..updates {
8844                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
8845                    .unwrap();
8846            }
8847            file.close().unwrap();
8848            let n = std::fs::metadata(&path).unwrap().len();
8849            let read = H5File::open(&path).unwrap();
8850            assert_eq!(
8851                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8852                vec![format!("update {}", updates - 1), "b".to_string()]
8853            );
8854            drop(read);
8855            std::fs::remove_file(&path).ok();
8856            n
8857        };
8858
8859        // The allocator settles once a freed block is available to reuse, so
8860        // every count past that produces the same file.
8861        let settled = size_after(3);
8862        assert_eq!(size_after(20), settled, "20 updates against 3");
8863        assert_eq!(size_after(50), settled, "50 updates against 3");
8864    }
8865
8866    /// An empty string is stored as a real heap object under a reference whose
8867    /// sequence length is zero, so the release must go by the address, not the
8868    /// length — a length test strands the object and its collection forever.
8869    #[test]
8870    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
8871        let size_after = |updates: usize| {
8872            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
8873            let file = H5File::create(&path).unwrap();
8874            file.write_vlen_strings("notes", &["", "b"]).unwrap();
8875            let ds = file.dataset_writer("notes").unwrap();
8876            for _ in 0..updates {
8877                ds.write_vlen_strings_slice(0, &[""]).unwrap();
8878            }
8879            file.close().unwrap();
8880            let n = std::fs::metadata(&path).unwrap().len();
8881            let read = H5File::open(&path).unwrap();
8882            assert_eq!(
8883                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8884                vec!["".to_string(), "b".to_string()]
8885            );
8886            drop(read);
8887            std::fs::remove_file(&path).ok();
8888            n
8889        };
8890
8891        let settled = size_after(3);
8892        assert_eq!(size_after(20), settled, "20 empty updates against 3");
8893        assert_eq!(size_after(50), settled, "50 empty updates against 3");
8894    }
8895
8896    /// The elements the update does not name keep their strings, so freeing
8897    /// the superseded objects must not disturb the collection's survivors.
8898    #[test]
8899    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
8900        let path = temp_path("vlen_slice_heap_survivors");
8901        {
8902            let file = H5File::create(&path).unwrap();
8903            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
8904                .unwrap();
8905            let ds = file.dataset_writer("notes").unwrap();
8906            // Two updates inside the one collection the create wrote, so the
8907            // second reads a collection the first already rewrote.
8908            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
8909            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
8910            file.close().unwrap();
8911        }
8912        let file = H5File::open(&path).unwrap();
8913        assert_eq!(
8914            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8915            vec!["a", "B", "c", "D"]
8916        );
8917        std::fs::remove_file(&path).ok();
8918    }
8919
8920    /// Replacing every element of a chunked dataset empties the collection the
8921    /// append wrote, and the file must still read back correctly after its
8922    /// block goes to the allocator.
8923    #[test]
8924    fn write_vlen_strings_slice_frees_an_emptied_collection() {
8925        let path = temp_path("vlen_slice_heap_emptied");
8926        {
8927            let file = H5File::create(&path).unwrap();
8928            file.create_appendable_vlen_dataset("notes", 2, None)
8929                .unwrap();
8930            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
8931                .unwrap();
8932            file.close().unwrap();
8933        }
8934        {
8935            let file = H5File::open_rw(&path).unwrap();
8936            file.dataset_writer("notes")
8937                .unwrap()
8938                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
8939                .unwrap();
8940            file.close().unwrap();
8941        }
8942        let file = H5File::open(&path).unwrap();
8943        assert_eq!(
8944            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8945            vec!["w", "x", "y", "z"]
8946        );
8947        std::fs::remove_file(&path).ok();
8948    }
8949
8950    /// A collection larger than the 4096-byte minimum must keep its size when
8951    /// an object leaves it. Re-encoding at the natural size instead shrinks
8952    /// what the header declares, so the block's tail stops being part of the
8953    /// collection and the eventual free returns less than was allocated —
8954    /// stranding the difference on every cycle.
8955    #[test]
8956    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
8957        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
8958        let size_after = |cycles: usize| {
8959            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
8960            let file = H5File::create(&path).unwrap();
8961            let seed: Vec<String> = "abcd".chars().map(big).collect();
8962            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
8963            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
8964            // is one collection well above it.
8965            file.write_vlen_strings("notes", &refs).unwrap();
8966            let ds = file.dataset_writer("notes").unwrap();
8967            for _ in 0..cycles {
8968                // Partially empty the collection, then finish it off: the
8969                // block is freed only after it has been rewritten once.
8970                let head = big('x');
8971                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
8972                let tail: Vec<String> = "yzw".chars().map(big).collect();
8973                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
8974                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
8975            }
8976            file.close().unwrap();
8977            let n = std::fs::metadata(&path).unwrap().len();
8978            let read = H5File::open(&path).unwrap();
8979            assert_eq!(
8980                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
8981                vec![big('x'), big('y'), big('z'), big('w')]
8982            );
8983            drop(read);
8984            std::fs::remove_file(&path).ok();
8985            n
8986        };
8987
8988        let settled = size_after(4);
8989        assert_eq!(size_after(30), settled, "30 cycles against 4");
8990    }
8991
8992    /// An element still in the append buffer has never been on disk, so its
8993    /// superseded object has to be found in the buffer or it is stranded.
8994    #[test]
8995    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
8996        let path = temp_path("vlen_slice_heap_buffered");
8997        let file = H5File::create(&path).unwrap();
8998        file.create_appendable_vlen_dataset("notes", 4, None)
8999            .unwrap();
9000        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
9001        let ds = file.dataset_writer("notes").unwrap();
9002        for i in 0..20 {
9003            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
9004                .unwrap();
9005        }
9006        file.close().unwrap();
9007
9008        let file = H5File::open(&path).unwrap();
9009        assert_eq!(
9010            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
9011            vec!["a", "patch 19", "c"]
9012        );
9013        let size = std::fs::metadata(&path).unwrap().len();
9014        std::fs::remove_file(&path).ok();
9015        assert!(
9016            size < 20 * 4096,
9017            "20 buffered updates left {size} bytes, one collection per update"
9018        );
9019    }
9020
9021    /// Regression: a typed `write_slice` into rows the append buffer still
9022    /// held wrote the chunks, and the flush at close wrote the stale buffered
9023    /// rows back over it — write 99, read 50. The slice now flushes the
9024    /// buffer first, making the chunks the single authority for those rows.
9025    #[test]
9026    fn write_slice_into_the_buffered_tail_survives_close() {
9027        let path = temp_path("slice_into_buffered_tail");
9028        {
9029            let file = H5File::create(&path).unwrap();
9030            let ds = file
9031                .new_dataset::<i32>()
9032                .shape([0])
9033                .chunk(&[4])
9034                .max_shape(&[None])
9035                .create("d")
9036                .unwrap();
9037            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
9038            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
9039            ds.write_slice(&[4], &[1], &[99]).unwrap();
9040            file.close().unwrap();
9041        }
9042        {
9043            let file = H5File::open(&path).unwrap();
9044            let ds = file.dataset("d").unwrap();
9045            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
9046        }
9047        std::fs::remove_file(&path).ok();
9048    }
9049
9050    /// Extending a dataset while appends sit in the buffer must not move
9051    /// them: the buffer records the absolute row its frames belong to, so
9052    /// the flush at close lands them there, and the grown region reads as
9053    /// fill.
9054    #[test]
9055    fn extend_does_not_move_buffered_appends() {
9056        let path = temp_path("extend_keeps_buffered_rows");
9057        {
9058            let file = H5File::create(&path).unwrap();
9059            let ds = file
9060                .new_dataset::<i32>()
9061                .shape([0])
9062                .chunk(&[4])
9063                .max_shape(&[None])
9064                .create("d")
9065                .unwrap();
9066            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
9067            ds.extend(&[10]).unwrap();
9068            file.close().unwrap();
9069        }
9070        {
9071            let file = H5File::open(&path).unwrap();
9072            let ds = file.dataset("d").unwrap();
9073            assert_eq!(
9074                ds.read_raw::<i32>().unwrap(),
9075                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
9076            );
9077        }
9078        std::fs::remove_file(&path).ok();
9079    }
9080
9081    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
9082    /// dimensions) buffered fine but close() failed "not a chunked dataset"
9083    /// and lost the buffered rows — the append's chunk writes required the
9084    /// extensible-array index. They now go through the index-generic
9085    /// hyperslab engine.
9086    #[test]
9087    fn append_to_a_btree_v2_dataset_survives_close() {
9088        let path = temp_path("append_bt2_close");
9089        {
9090            let file = H5File::create(&path).unwrap();
9091            let ds = file
9092                .new_dataset::<i32>()
9093                .shape([0, 3])
9094                .chunk(&[4, 3])
9095                .max_shape(&[None, None])
9096                .create("d")
9097                .unwrap();
9098            // One buffered row, then a batch that crosses the chunk
9099            // boundary: 4 rows fill chunk band 0, one row stays buffered
9100            // for the flush at close.
9101            ds.append(&[1, 2, 3]).unwrap();
9102            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
9103            file.close().unwrap();
9104        }
9105        {
9106            let file = H5File::open(&path).unwrap();
9107            let ds = file.dataset("d").unwrap();
9108            assert_eq!(ds.shape(), vec![5, 3]);
9109            assert_eq!(
9110                ds.read_raw::<i32>().unwrap(),
9111                (1..=15).collect::<Vec<i32>>()
9112            );
9113        }
9114        std::fs::remove_file(&path).ok();
9115    }
9116
9117    /// A chunk row narrower than the frame row is legal geometry (libhdf5
9118    /// creates it); appended frames must be scattered across the row's
9119    /// tiles at the chunk stride, not packed at the frame stride.
9120    #[test]
9121    fn append_scatters_frames_across_narrow_chunk_tiles() {
9122        let path = temp_path("append_narrow_chunks");
9123        {
9124            let file = H5File::create(&path).unwrap();
9125            let ds = file
9126                .new_dataset::<i32>()
9127                .shape([0, 8])
9128                .chunk(&[2, 4])
9129                .max_shape(&[None, Some(8)])
9130                .create("d")
9131                .unwrap();
9132            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
9133            // row 2 is flushed partial at close.
9134            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
9135            file.close().unwrap();
9136        }
9137        {
9138            let file = H5File::open(&path).unwrap();
9139            let ds = file.dataset("d").unwrap();
9140            assert_eq!(ds.shape(), vec![3, 8]);
9141            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
9142        }
9143        std::fs::remove_file(&path).ok();
9144    }
9145
9146    /// A fixed-array dataset has no room to grow: appending must surface an
9147    /// error naming the chunk grid, not lose rows silently. (Before the
9148    /// index-generic append it failed as "not a chunked dataset".)
9149    #[test]
9150    fn append_to_a_full_fixed_array_dataset_errors() {
9151        let path = temp_path("append_fa_errors");
9152        let file = H5File::create(&path).unwrap();
9153        let ds = file
9154            .new_dataset::<i32>()
9155            .shape([4, 3])
9156            .chunk(&[2, 3])
9157            .create("d")
9158            .unwrap();
9159        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
9160        assert!(
9161            err.to_string().contains("chunk grid"),
9162            "unexpected error: {err}"
9163        );
9164        file.close().unwrap();
9165        std::fs::remove_file(&path).ok();
9166    }
9167
9168    /// A finite max_shape above the current shape used to be dropped on the
9169    /// fixed-array path: the array was sized from the current dims and the
9170    /// stored dataspace had no maximum, so growth failed. The array is now
9171    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
9172    /// fixed-max dataset appends up to its maximum and roundtrips.
9173    #[test]
9174    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
9175        let path = temp_path("fa_growable_dim0");
9176        {
9177            let file = H5File::create(&path).unwrap();
9178            let ds = file
9179                .new_dataset::<i32>()
9180                .shape([4, 3])
9181                .chunk(&[2, 3])
9182                .max_shape(&[Some(10), Some(3)])
9183                .create("d")
9184                .unwrap();
9185            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9186            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
9187            file.close().unwrap();
9188        }
9189        {
9190            let file = H5File::open(&path).unwrap();
9191            let ds = file.dataset("d").unwrap();
9192            assert_eq!(ds.shape(), vec![6, 3]);
9193            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
9194        }
9195        std::fs::remove_file(&path).ok();
9196    }
9197
9198    /// The multiplier-dimension boundary: growing a dimension other than 0
9199    /// changes the current chunk grid but not the index grid. Chunk slots
9200    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
9201    /// chunks written before the extend are looked up under different
9202    /// indices after it.
9203    #[test]
9204    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
9205        let path = temp_path("fa_growable_dim1");
9206        {
9207            let file = H5File::create(&path).unwrap();
9208            let ds = file
9209                .new_dataset::<i32>()
9210                .shape([4, 3])
9211                .chunk(&[2, 3])
9212                .max_shape(&[Some(4), Some(9)])
9213                .create("d")
9214                .unwrap();
9215            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
9216            ds.extend(&[4, 6]).unwrap();
9217            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
9218                .unwrap();
9219            file.close().unwrap();
9220        }
9221        {
9222            let file = H5File::open(&path).unwrap();
9223            let ds = file.dataset("d").unwrap();
9224            assert_eq!(ds.shape(), vec![4, 6]);
9225            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
9226            // then [12 + r*3 ..) from the second.
9227            let mut expect = Vec::new();
9228            for r in 0i32..4 {
9229                expect.extend((r * 3)..(r * 3 + 3));
9230                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
9231            }
9232            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
9233        }
9234        std::fs::remove_file(&path).ok();
9235    }
9236
9237    /// Growth boundaries: past the stored maximum is rejected, and a dataset
9238    /// without a stored maximum is fixed at its extent (libhdf5 defaults
9239    /// maxdims to dims at creation).
9240    #[test]
9241    fn extend_beyond_the_maximum_is_rejected() {
9242        let path = temp_path("extend_beyond_max");
9243        let file = H5File::create(&path).unwrap();
9244        let ds = file
9245            .new_dataset::<i32>()
9246            .shape([4, 3])
9247            .chunk(&[2, 3])
9248            .max_shape(&[Some(6), Some(3)])
9249            .create("d")
9250            .unwrap();
9251        ds.extend(&[6, 3]).unwrap();
9252        let err = ds.extend(&[8, 3]).unwrap_err();
9253        assert!(
9254            err.to_string().contains("exceeds the maximum"),
9255            "unexpected error: {err}"
9256        );
9257        file.close().unwrap();
9258        std::fs::remove_file(&path).ok();
9259    }
9260
9261    /// An unlimited dimension other than 0 has no fixed linear slot without
9262    /// libhdf5's extensible-array swizzling; `chunk_grid::linear_index` now
9263    /// implements that swizzle for any dimension, so this creates cleanly
9264    /// and every extend keeps writing new chunks to new slots, never
9265    /// re-addressing one already on disk.
9266    #[test]
9267    fn builder_accepts_an_unlimited_inner_dimension() {
9268        let path = temp_path("unlimited_inner_dim");
9269        let file = H5File::create(&path).unwrap();
9270        let ds = file
9271            .new_dataset::<i32>()
9272            .shape([4, 0])
9273            .chunk(&[2, 2])
9274            .max_shape(&[Some(4), None])
9275            .create("d")
9276            .unwrap();
9277        assert_eq!(ds.shape(), vec![4, 0]);
9278
9279        // Write, then extend and write again: if the linear index were
9280        // recomputed from the *current* extent instead of the maximum one,
9281        // the second extend would shift every slot number and the first
9282        // write's chunks would decode under the wrong coordinates below.
9283        ds.extend(&[4, 2]).unwrap();
9284        ds.write_slice(&[0, 0], &[4, 2], &[1, 2, 3, 4, 5, 6, 7, 8])
9285            .unwrap();
9286        ds.extend(&[4, 4]).unwrap();
9287        ds.write_slice(&[0, 2], &[4, 2], &[9, 10, 11, 12, 13, 14, 15, 16])
9288            .unwrap();
9289
9290        file.close().unwrap();
9291        let file = H5File::open(&path).unwrap();
9292        let ds = file.dataset("d").unwrap();
9293        assert_eq!(
9294            ds.read_slice::<i32>(&[0, 0], &[4, 4]).unwrap(),
9295            vec![1, 2, 9, 10, 3, 4, 11, 12, 5, 6, 13, 14, 7, 8, 15, 16]
9296        );
9297        std::fs::remove_file(&path).ok();
9298    }
9299
9300    /// Regression: a chunk wider than a fixed max dimension used to be
9301    /// accepted, and appends then packed rows at the chunk stride — writing
9302    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
9303    /// geometry at create (`H5D__chunk_construct`); so do we now.
9304    #[test]
9305    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
9306        let path = temp_path("builder_chunk_wider_than_max");
9307        let file = H5File::create(&path).unwrap();
9308        let err = match file
9309            .new_dataset::<i32>()
9310            .shape([0, 2])
9311            .chunk(&[2, 4])
9312            .max_shape(&[None, Some(2)])
9313            .create("v5")
9314        {
9315            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
9316            Err(e) => e,
9317        };
9318        assert!(
9319            err.to_string().contains("maximum dimension size"),
9320            "unexpected error: {err}"
9321        );
9322        file.close().unwrap();
9323        std::fs::remove_file(&path).ok();
9324    }
9325
9326    /// Boundary: `fits in T` vs `does not fit in T`, for both the too-large
9327    /// (u64::MAX → i64) and the negative-to-unsigned (−1 → u32) directions.
9328    #[test]
9329    fn numeric_int_checked_conversion_boundaries() {
9330        let path = temp_path("numeric_int_bounds");
9331        {
9332            let file = H5File::create(&path).unwrap();
9333            let ds = file.new_dataset::<u64>().shape([2]).create("u").unwrap();
9334            ds.write_raw(&[1u64, u64::MAX]).unwrap();
9335            let ds = file.new_dataset::<i32>().shape([2]).create("i").unwrap();
9336            ds.write_raw(&[-1i32, 5]).unwrap();
9337            file.close().unwrap();
9338        }
9339        let file = H5File::open(&path).unwrap();
9340
9341        let u = file.dataset("u").unwrap();
9342        assert_eq!(u.read_numeric_as::<u64>().unwrap(), vec![1, u64::MAX]);
9343        assert_eq!(
9344            u.read_numeric_as::<i128>().unwrap(),
9345            vec![1, i128::from(u64::MAX)]
9346        );
9347        let err = u.read_numeric_as::<i64>().unwrap_err();
9348        assert!(
9349            err.to_string()
9350                .contains("value 18446744073709551615 at element 1 does not fit in i64"),
9351            "unexpected error: {err}"
9352        );
9353
9354        let i = file.dataset("i").unwrap();
9355        assert_eq!(i.read_numeric_as::<i64>().unwrap(), vec![-1, 5]);
9356        let err = i.read_numeric_as::<u32>().unwrap_err();
9357        assert!(
9358            err.to_string()
9359                .contains("value -1 at element 0 does not fit in u32"),
9360            "unexpected error: {err}"
9361        );
9362        std::fs::remove_file(&path).ok();
9363    }
9364
9365    /// Boundary: f32 → f64 is exact widening; f64 → f32 is rejected.
9366    #[test]
9367    fn numeric_float_widening_only() {
9368        let path = temp_path("numeric_float");
9369        {
9370            let file = H5File::create(&path).unwrap();
9371            let ds = file.new_dataset::<f32>().shape([2]).create("f4").unwrap();
9372            ds.write_raw(&[1.5f32, -2.25]).unwrap();
9373            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9374            ds.write_raw(&[3.75f64]).unwrap();
9375            file.close().unwrap();
9376        }
9377        let file = H5File::open(&path).unwrap();
9378
9379        let f4 = file.dataset("f4").unwrap();
9380        assert_eq!(f4.read_numeric_as::<f32>().unwrap(), vec![1.5, -2.25]);
9381        assert_eq!(f4.read_numeric_as::<f64>().unwrap(), vec![1.5, -2.25]);
9382
9383        let f8 = file.dataset("f8").unwrap();
9384        assert_eq!(f8.read_numeric_as::<f64>().unwrap(), vec![3.75]);
9385        let err = f8.read_numeric_as::<f32>().unwrap_err();
9386        assert!(
9387            err.to_string().contains("narrowing"),
9388            "unexpected error: {err}"
9389        );
9390        std::fs::remove_file(&path).ok();
9391    }
9392
9393    /// Boundary: cross-class conversions (float ↔ integer) are rejected in
9394    /// both directions, and a non-numeric datatype is rejected at classify.
9395    #[test]
9396    fn numeric_cross_class_and_non_numeric_rejected() {
9397        let path = temp_path("numeric_cross_class");
9398        {
9399            let file = H5File::create(&path).unwrap();
9400            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
9401            ds.write_raw(&[1.0f64]).unwrap();
9402            let ds = file.new_dataset::<i32>().shape([1]).create("i4").unwrap();
9403            ds.write_raw(&[7i32]).unwrap();
9404            file.write_vlen_strings("s", &["a", "b"]).unwrap();
9405            file.close().unwrap();
9406        }
9407        let file = H5File::open(&path).unwrap();
9408
9409        let err = file
9410            .dataset("f8")
9411            .unwrap()
9412            .read_numeric_as::<i64>()
9413            .unwrap_err();
9414        assert!(
9415            err.to_string().contains("floating-point dataset as i64"),
9416            "unexpected error: {err}"
9417        );
9418        let err = file
9419            .dataset("i4")
9420            .unwrap()
9421            .read_numeric_as::<f64>()
9422            .unwrap_err();
9423        assert!(
9424            err.to_string().contains("integer dataset as f64"),
9425            "unexpected error: {err}"
9426        );
9427        let err = file
9428            .dataset("s")
9429            .unwrap()
9430            .read_numeric_as::<i64>()
9431            .unwrap_err();
9432        assert!(
9433            err.to_string().contains("is not numeric"),
9434            "unexpected error: {err}"
9435        );
9436        std::fs::remove_file(&path).ok();
9437    }
9438
9439    /// Boundary: big-endian sources decode per the datatype's byte order.
9440    /// Unit-level (the writer only emits little-endian): feed `convert` a
9441    /// big-endian datatype plus big-endian bytes directly.
9442    #[test]
9443    fn numeric_big_endian_decode() {
9444        use super::numeric;
9445        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
9446
9447        let dt = DatatypeMessage::FixedPoint {
9448            size: 4,
9449            byte_order: ByteOrder::BigEndian,
9450            signed: true,
9451            bit_offset: 0,
9452            bit_precision: 32,
9453        };
9454        let mut raw = Vec::new();
9455        raw.extend_from_slice(&(-2i32).to_be_bytes());
9456        raw.extend_from_slice(&(100_000i32).to_be_bytes());
9457        let kind = numeric::classify(&dt).unwrap();
9458        assert_eq!(
9459            numeric::convert::<i64>(kind, &raw).unwrap(),
9460            vec![-2, 100_000]
9461        );
9462
9463        let dt = DatatypeMessage::FloatingPoint {
9464            size: 8,
9465            byte_order: ByteOrder::BigEndian,
9466            sign_location: 63,
9467            bit_offset: 0,
9468            bit_precision: 64,
9469            exponent_location: 52,
9470            exponent_size: 11,
9471            mantissa_location: 0,
9472            mantissa_size: 52,
9473            exponent_bias: 1023,
9474        };
9475        let raw = (-2.25f64).to_be_bytes();
9476        let kind = numeric::classify(&dt).unwrap();
9477        assert_eq!(numeric::convert::<f64>(kind, &raw).unwrap(), vec![-2.25]);
9478    }
9479
9480    /// `H5Attribute::read_numeric` validates the stored datatype before
9481    /// reinterpreting bytes: cross-width, cross-class, and non-numeric
9482    /// attributes error instead of returning bit-garbage, while the exact
9483    /// type and the HBool / complex-compound paths keep working.
9484    #[test]
9485    fn attr_read_numeric_validates_datatype() {
9486        use crate::types::{Complex64, HBool, VarLenUnicode};
9487        let path = temp_path("attr_read_numeric_validate");
9488        {
9489            let file = H5File::create(&path).unwrap();
9490            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9491            ds.write_raw(&[1.0f32; 2]).unwrap();
9492            let a = ds.new_attr::<f64>().shape(()).create("f8").unwrap();
9493            a.write_numeric(&1.5f64).unwrap();
9494            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9495            a.write_numeric(&-7i32).unwrap();
9496            let a = ds.new_attr::<HBool>().shape(()).create("b").unwrap();
9497            a.write_numeric(&HBool::from(true)).unwrap();
9498            let a = ds.new_attr::<Complex64>().shape(()).create("z").unwrap();
9499            a.write_numeric(&Complex64 { re: 1.0, im: -2.0 }).unwrap();
9500            let a = ds
9501                .new_attr::<VarLenUnicode>()
9502                .shape(())
9503                .create("s")
9504                .unwrap();
9505            a.write_scalar(&VarLenUnicode("text".into())).unwrap();
9506            file.close().unwrap();
9507        }
9508        let file = H5File::open(&path).unwrap();
9509        let ds = file.dataset("d").unwrap();
9510
9511        let f8 = ds.attr("f8").unwrap();
9512        assert_eq!(f8.read_numeric::<f64>().unwrap(), 1.5);
9513        // Previously returned the low half of the f64 image as an f32.
9514        let err = f8.read_numeric::<f32>().unwrap_err();
9515        assert!(
9516            err.to_string().contains("read_numeric_as"),
9517            "unexpected error: {err}"
9518        );
9519        assert!(f8.read_numeric::<i64>().is_err());
9520
9521        let i4 = ds.attr("i4").unwrap();
9522        assert_eq!(i4.read_numeric::<i32>().unwrap(), -7);
9523        assert!(i4.read_numeric::<u32>().is_err());
9524
9525        assert!(bool::from(
9526            ds.attr("b").unwrap().read_numeric::<HBool>().unwrap()
9527        ));
9528        let z = ds.attr("z").unwrap().read_numeric::<Complex64>().unwrap();
9529        assert_eq!((z.re, z.im), (1.0, -2.0));
9530
9531        // A vlen string attribute: read_numeric used to transmute the heap
9532        // reference bytes into the requested type.
9533        let s = ds.attr("s").unwrap();
9534        assert!(s.read_numeric::<f64>().is_err());
9535        assert!(s.read_numeric_as::<f64>().is_err());
9536        std::fs::remove_file(&path).ok();
9537    }
9538
9539    /// The attribute conversion read applies the dataset rules: checked
9540    /// int → int naming index and value on overflow, widening-only floats,
9541    /// cross-class rejected; an array attribute converts every element.
9542    #[test]
9543    fn attr_read_numeric_as_converts() {
9544        let path = temp_path("attr_read_numeric_as");
9545        {
9546            let file = H5File::create(&path).unwrap();
9547            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
9548            ds.write_raw(&[1.0f32; 2]).unwrap();
9549            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
9550            a.write_numeric(&-7i32).unwrap();
9551            let a = ds.new_attr::<u64>().shape(()).create("u8max").unwrap();
9552            a.write_numeric(&u64::MAX).unwrap();
9553            let a = ds.new_attr::<i16>().shape([3]).create("arr").unwrap();
9554            a.write_array(&[1i16, -2, 3]).unwrap();
9555            file.close().unwrap();
9556        }
9557        let file = H5File::open(&path).unwrap();
9558        let ds = file.dataset("d").unwrap();
9559        assert_eq!(
9560            ds.attr("i4").unwrap().read_numeric_as::<i64>().unwrap(),
9561            vec![-7]
9562        );
9563        let err = ds
9564            .attr("u8max")
9565            .unwrap()
9566            .read_numeric_as::<i64>()
9567            .unwrap_err();
9568        assert!(
9569            err.to_string().contains("does not fit in i64"),
9570            "unexpected error: {err}"
9571        );
9572        assert_eq!(
9573            ds.attr("arr").unwrap().read_numeric_as::<i32>().unwrap(),
9574            vec![1, -2, 3]
9575        );
9576        assert!(ds.attr("i4").unwrap().read_numeric_as::<f64>().is_err());
9577        std::fs::remove_file(&path).ok();
9578    }
9579
9580    /// The hyperslab variant applies the same conversion to a sub-selection.
9581    #[test]
9582    fn numeric_slice_conversion() {
9583        let path = temp_path("numeric_slice");
9584        {
9585            let file = H5File::create(&path).unwrap();
9586            let ds = file.new_dataset::<i16>().shape([2, 3]).create("m").unwrap();
9587            ds.write_raw(&[1i16, 2, 3, 4, 5, 6]).unwrap();
9588            file.close().unwrap();
9589        }
9590        let file = H5File::open(&path).unwrap();
9591        let m = file.dataset("m").unwrap();
9592        assert_eq!(
9593            m.read_numeric_slice_as::<i32>(&[0, 1], &[2, 2]).unwrap(),
9594            vec![2, 3, 5, 6]
9595        );
9596        std::fs::remove_file(&path).ok();
9597    }
9598
9599    /// A NULL dataspace round-trips through the public API as `is_null() ==
9600    /// true`, `shape() == []`, and `read_raw_bytes()` empty — and stays
9601    /// distinguishable from a scalar dataset, which shares the same empty
9602    /// `shape()` but holds exactly one element.
9603    #[test]
9604    fn null_dataspace_distinct_from_scalar() {
9605        let path = temp_path("null_vs_scalar");
9606        {
9607            let file = H5File::create(&path).unwrap();
9608            file.new_dataset::<i32>().null().create("empty").unwrap();
9609            let scalar = file.new_dataset::<i32>().scalar().create("scalar").unwrap();
9610            scalar.write_raw(&[42i32]).unwrap();
9611            file.close().unwrap();
9612        }
9613        let file = H5File::open(&path).unwrap();
9614
9615        let empty = file.dataset("empty").unwrap();
9616        assert!(empty.is_null());
9617        assert_eq!(empty.shape(), Vec::<usize>::new());
9618        assert_eq!(empty.total_elements(), 0);
9619        assert_eq!(empty.read_raw_bytes().unwrap(), Vec::<u8>::new());
9620
9621        let scalar = file.dataset("scalar").unwrap();
9622        assert!(!scalar.is_null());
9623        assert_eq!(scalar.shape(), Vec::<usize>::new());
9624        assert_eq!(scalar.total_elements(), 1);
9625        assert_eq!(scalar.read_raw::<i32>().unwrap(), vec![42]);
9626
9627        std::fs::remove_file(&path).ok();
9628    }
9629
9630    /// A NULL dataspace dataset rejects writes outright — there is nothing
9631    /// to write into — rather than silently accepting a scalar-shaped
9632    /// write against unallocated storage.
9633    #[test]
9634    fn null_dataspace_rejects_writes() {
9635        let path = temp_path("null_write_rejected");
9636        let file = H5File::create(&path).unwrap();
9637        let ds = file.new_dataset::<i32>().null().create("empty").unwrap();
9638        assert!(ds.write_raw(&[1i32]).is_err());
9639        assert!(ds.write_raw_bytes(&[0u8; 4]).is_err());
9640        file.close().unwrap();
9641        std::fs::remove_file(&path).ok();
9642    }
9643
9644    /// `.null()` combined with `.chunk()` or a fill value is rejected at
9645    /// `create()` rather than silently dropping the conflicting option —
9646    /// a NULL dataspace can never be chunked or filtered upstream.
9647    #[test]
9648    fn null_dataspace_rejects_chunking_and_fill_value() {
9649        let path = temp_path("null_chunk_rejected");
9650        let file = H5File::create(&path).unwrap();
9651        assert!(file
9652            .new_dataset::<i32>()
9653            .null()
9654            .chunk(&[4])
9655            .create("a")
9656            .is_err());
9657        assert!(file
9658            .new_dataset::<i32>()
9659            .null()
9660            .fill_value(7i32)
9661            .create("b")
9662            .is_err());
9663        file.close().unwrap();
9664        std::fs::remove_file(&path).ok();
9665    }
9666
9667    /// A committed type is resolved before the dataset is created, so a name
9668    /// that is not one — or one paired with object references, which would
9669    /// make the stored type disagree with the payload — leaves no dataset
9670    /// behind.
9671    #[test]
9672    fn a_committed_type_that_cannot_be_shared_creates_no_dataset() {
9673        use crate::format::messages::datatype::DatatypeMessage;
9674
9675        let path = temp_path("committed_refused");
9676        let file = H5File::create(&path).unwrap();
9677        file.commit_datatype("t", DatatypeMessage::i32_type())
9678            .unwrap();
9679
9680        assert!(file
9681            .new_dataset::<i32>()
9682            .committed_type("absent")
9683            .shape([2usize])
9684            .create("a")
9685            .is_err());
9686        assert!(file
9687            .new_dataset::<u64>()
9688            .committed_type("t")
9689            .object_references()
9690            .shape([2usize])
9691            .create("b")
9692            .is_err());
9693        // A dataset already exists under that name, so the type cannot take
9694        // it either.
9695        file.new_dataset::<i32>()
9696            .shape([2usize])
9697            .create("taken")
9698            .unwrap();
9699        assert!(file
9700            .commit_datatype("taken", DatatypeMessage::i32_type())
9701            .is_err());
9702        assert!(file
9703            .commit_datatype("t", DatatypeMessage::f64_type())
9704            .is_err());
9705
9706        assert_eq!(file.dataset_names(), vec!["taken".to_string()]);
9707        assert_eq!(file.named_datatype_names(), vec!["t".to_string()]);
9708        file.close().unwrap();
9709        std::fs::remove_file(&path).ok();
9710    }
9711
9712    /// Deleting the group that named a committed datatype takes the name with
9713    /// it: nothing in the file reaches the type, so it is not written, the
9714    /// name is free again, and it can no longer be shared by that name.
9715    #[test]
9716    fn deleting_a_group_takes_the_committed_datatypes_it_named() {
9717        use crate::format::messages::datatype::DatatypeMessage;
9718
9719        let path = temp_path("committed_group_deleted");
9720        {
9721            let file = H5File::create(&path).unwrap();
9722            let types = file.create_group("types").unwrap();
9723            types
9724                .commit_datatype("t", DatatypeMessage::i32_type())
9725                .unwrap();
9726            assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9727
9728            file.delete_group("types").unwrap();
9729            assert!(file.named_datatype_names().is_empty());
9730            assert!(file
9731                .new_dataset::<i32>()
9732                .committed_type("types/t")
9733                .shape([2usize])
9734                .create("d")
9735                .is_err());
9736
9737            // The name is free, so a new group may take it back.
9738            let types = file.create_group("types").unwrap();
9739            types
9740                .commit_datatype("t", DatatypeMessage::f64_type())
9741                .unwrap();
9742            file.close().unwrap();
9743        }
9744        let file = H5File::open(&path).unwrap();
9745        assert_eq!(file.named_datatype_names(), vec!["types/t".to_string()]);
9746        assert_eq!(
9747            file.named_datatype("types/t").unwrap().datatype().unwrap(),
9748            crate::format::messages::datatype::DatatypeMessage::f64_type()
9749        );
9750        drop(file);
9751        std::fs::remove_file(&path).ok();
9752    }
9753
9754    /// The layout class the reader sees for `name`, plus the image a compact
9755    /// layout carries — what distinguishes compact storage from contiguous
9756    /// storage that happens to hold the same bytes.
9757    fn compact_image(path: &std::path::Path, name: &str) -> Option<Vec<u8>> {
9758        use crate::format::messages::data_layout::DataLayoutMessage;
9759        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9760        match &reader.dataset_info(name).unwrap().layout {
9761            DataLayoutMessage::Compact { data } => Some(data.clone()),
9762            other => panic!("{name}: expected a compact layout, got {other:?}"),
9763        }
9764    }
9765
9766    /// `.compact()` puts the raw data inside the data layout message: the
9767    /// dataset has no data block of its own, and the image the layout carries
9768    /// is what a read returns.
9769    #[test]
9770    fn a_compact_dataset_stores_its_data_in_the_layout_message() {
9771        let path = temp_path("compact_roundtrip");
9772        let values: Vec<i32> = (0..16).collect();
9773        {
9774            let file = H5File::create(&path).unwrap();
9775            file.new_dataset::<i32>()
9776                .shape([16usize])
9777                .compact()
9778                .create("d")
9779                .unwrap()
9780                .write_raw(&values)
9781                .unwrap();
9782            file.close().unwrap();
9783        }
9784
9785        let image = compact_image(&path, "d").unwrap();
9786        assert_eq!(
9787            image,
9788            values
9789                .iter()
9790                .flat_map(|v| v.to_le_bytes())
9791                .collect::<Vec<_>>()
9792        );
9793
9794        let file = H5File::open(&path).unwrap();
9795        let ds = file.dataset("d").unwrap();
9796        assert_eq!(ds.shape(), vec![16]);
9797        assert_eq!(ds.read_raw::<i32>().unwrap(), values);
9798        std::fs::remove_file(&path).ok();
9799    }
9800
9801    /// A compact dataset created inside a group is linked from that group,
9802    /// not from the root: its create path goes through the same parent
9803    /// resolution every other layout uses.
9804    #[test]
9805    fn a_compact_dataset_lands_in_its_group() {
9806        let path = temp_path("compact_in_group");
9807        {
9808            let file = H5File::create(&path).unwrap();
9809            let g = file.root_group().create_group("g").unwrap();
9810            g.new_dataset::<u8>()
9811                .shape([4usize])
9812                .compact()
9813                .create("d")
9814                .unwrap()
9815                .write_raw(&[1u8, 2, 3, 4])
9816                .unwrap();
9817            file.close().unwrap();
9818        }
9819        assert_eq!(compact_image(&path, "/g/d").unwrap(), vec![1u8, 2, 3, 4]);
9820
9821        let file = H5File::open(&path).unwrap();
9822        assert_eq!(
9823            file.dataset("/g/d").unwrap().read_raw::<u8>().unwrap(),
9824            vec![1u8, 2, 3, 4]
9825        );
9826        std::fs::remove_file(&path).ok();
9827    }
9828
9829    /// A compact dataset's storage is the image itself, so the fill value has
9830    /// to be tiled into it at create — `H5D__compact_fill`'s job. An unwritten
9831    /// element must read back as the fill value, not as zero.
9832    #[test]
9833    fn an_unwritten_compact_dataset_reads_back_as_its_fill_value() {
9834        let path = temp_path("compact_fill");
9835        {
9836            let file = H5File::create(&path).unwrap();
9837            file.new_dataset::<i32>()
9838                .shape([4usize])
9839                .compact()
9840                .fill_value(-7i32)
9841                .create("d")
9842                .unwrap();
9843            file.close().unwrap();
9844        }
9845        let file = H5File::open(&path).unwrap();
9846        assert_eq!(
9847            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9848            vec![-7i32; 4]
9849        );
9850        std::fs::remove_file(&path).ok();
9851    }
9852
9853    /// The image is the layout message, so anything that rewrites the header
9854    /// rewrites the data with it. Reopening and attaching an attribute makes
9855    /// the header stale; the rebuilt one must still carry the image rather
9856    /// than fall back to an unallocated contiguous layout.
9857    #[test]
9858    fn a_reopened_compact_dataset_keeps_its_image() {
9859        let path = temp_path("compact_reopen");
9860        let values: Vec<i32> = (100..108).collect();
9861        {
9862            let file = H5File::create(&path).unwrap();
9863            file.new_dataset::<i32>()
9864                .shape([8usize])
9865                .compact()
9866                .create("d")
9867                .unwrap()
9868                .write_raw(&values)
9869                .unwrap();
9870            file.close().unwrap();
9871        }
9872        {
9873            let file = H5File::open_rw(&path).unwrap();
9874            file.dataset_writer("d")
9875                .unwrap()
9876                .new_attr::<i32>()
9877                .shape(())
9878                .create("note")
9879                .unwrap()
9880                .write_numeric(&1i32)
9881                .unwrap();
9882            file.close().unwrap();
9883        }
9884        assert_eq!(
9885            compact_image(&path, "d").unwrap(),
9886            values
9887                .iter()
9888                .flat_map(|v| v.to_le_bytes())
9889                .collect::<Vec<_>>()
9890        );
9891
9892        let file = H5File::open(&path).unwrap();
9893        assert_eq!(
9894            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
9895            values
9896        );
9897        std::fs::remove_file(&path).ok();
9898    }
9899
9900    /// The ceiling is what a data layout message can hold, so it is checked
9901    /// in bytes and names them: the largest image that fits is accepted and
9902    /// one element more is refused.
9903    #[test]
9904    fn the_compact_ceiling_is_checked_in_bytes() {
9905        let path = temp_path("compact_ceiling");
9906        let file = H5File::create(&path).unwrap();
9907
9908        let fits = crate::MAX_COMPACT_DATA / 4;
9909        file.new_dataset::<i32>()
9910            .shape([fits])
9911            .compact()
9912            .create("fits")
9913            .unwrap();
9914
9915        let over = fits + 1;
9916        let err = match file
9917            .new_dataset::<i32>()
9918            .shape([over])
9919            .compact()
9920            .create("over")
9921        {
9922            Err(e) => e.to_string(),
9923            Ok(_) => panic!("an image {} bytes wide must be refused", over * 4),
9924        };
9925        assert!(
9926            err.contains(&(over * 4).to_string())
9927                && err.contains(&crate::MAX_COMPACT_DATA.to_string()),
9928            "the error must name both sizes: {err}"
9929        );
9930
9931        file.close().unwrap();
9932        std::fs::remove_file(&path).ok();
9933    }
9934
9935    /// Compact storage has no chunk grid to filter and no room to grow, so
9936    /// each conflicting option is refused at `create()` rather than silently
9937    /// overriding the layout the way `H5Pset_chunk` does.
9938    #[test]
9939    fn compact_rejects_chunking_filters_growth_and_a_null_dataspace() {
9940        let path = temp_path("compact_rejects");
9941        let file = H5File::create(&path).unwrap();
9942        assert!(file
9943            .new_dataset::<i32>()
9944            .shape([4usize])
9945            .compact()
9946            .chunk(&[4])
9947            .create("a")
9948            .is_err());
9949        assert!(file
9950            .new_dataset::<i32>()
9951            .shape([4usize])
9952            .compact()
9953            .deflate(4)
9954            .create("b")
9955            .is_err());
9956        assert!(file
9957            .new_dataset::<i32>()
9958            .shape([4usize])
9959            .compact()
9960            .max_shape(&[None])
9961            .create("c")
9962            .is_err());
9963        assert!(file
9964            .new_dataset::<i32>()
9965            .shape([4usize])
9966            .compact()
9967            .max_shape(&[Some(8)])
9968            .create("d")
9969            .is_err());
9970        assert!(file
9971            .new_dataset::<i32>()
9972            .null()
9973            .compact()
9974            .create("e")
9975            .is_err());
9976        file.close().unwrap();
9977        std::fs::remove_file(&path).ok();
9978    }
9979
9980    /// The filter pipeline the reader decodes from `name`'s header.
9981    fn stored_pipeline(
9982        path: &std::path::Path,
9983        name: &str,
9984    ) -> crate::format::messages::filter::FilterPipeline {
9985        let mut reader = crate::io::reader::Hdf5Reader::open(path).unwrap();
9986        reader
9987            .dataset_info(name)
9988            .unwrap()
9989            .filter_pipeline
9990            .clone()
9991            .unwrap_or_else(|| panic!("{name}: no filter pipeline"))
9992    }
9993
9994    /// Shuffle is a permutation, not a compressor, so it is a pipeline on its
9995    /// own — `H5Pset_shuffle` with nothing behind it. Its stage must reach the
9996    /// header, and the reader must unpermute what it wrote.
9997    #[test]
9998    fn shuffle_alone_is_a_filter_pipeline() {
9999        use crate::format::messages::filter::{FilterPipeline, FILTER_SHUFFLE};
10000        let path = temp_path("shuffle_alone");
10001        let values: Vec<i32> = (0..64).collect();
10002        {
10003            let file = H5File::create(&path).unwrap();
10004            file.new_dataset::<i32>()
10005                .shape([64usize])
10006                .chunk(&[16])
10007                .shuffle()
10008                .create("d")
10009                .unwrap()
10010                .write_raw(&values)
10011                .unwrap();
10012            file.close().unwrap();
10013        }
10014        let pipeline = stored_pipeline(&path, "d");
10015        assert_eq!(pipeline, FilterPipeline::shuffle(4));
10016        assert_eq!(pipeline.filters[0].id, FILTER_SHUFFLE);
10017
10018        let file = H5File::open(&path).unwrap();
10019        assert_eq!(
10020            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
10021            values
10022        );
10023        std::fs::remove_file(&path).ok();
10024    }
10025
10026    /// `.shuffle()` and `.deflate()` are separate stages that compose, and
10027    /// `.shuffle_deflate()` is the shorthand for both — one pipeline, built
10028    /// once, whichever way it was asked for.
10029    #[cfg(feature = "deflate")]
10030    #[test]
10031    fn shuffle_composes_with_deflate() {
10032        let path = temp_path("shuffle_then_deflate");
10033        let values: Vec<i32> = (0..64).collect();
10034        {
10035            let file = H5File::create(&path).unwrap();
10036            for (name, ds) in [
10037                ("split", file.new_dataset::<i32>().shuffle().deflate(6)),
10038                ("combined", file.new_dataset::<i32>().shuffle_deflate(6)),
10039            ] {
10040                ds.shape([64usize])
10041                    .chunk(&[16])
10042                    .create(name)
10043                    .unwrap()
10044                    .write_raw(&values)
10045                    .unwrap();
10046            }
10047            file.close().unwrap();
10048        }
10049        assert_eq!(
10050            stored_pipeline(&path, "split"),
10051            crate::format::messages::filter::FilterPipeline::shuffle_deflate(4, 6)
10052        );
10053        assert_eq!(
10054            stored_pipeline(&path, "split"),
10055            stored_pipeline(&path, "combined")
10056        );
10057
10058        let file = H5File::open(&path).unwrap();
10059        for name in ["split", "combined"] {
10060            assert_eq!(
10061                file.dataset(name).unwrap().read_raw::<i32>().unwrap(),
10062                values,
10063                "{name}"
10064            );
10065        }
10066        std::fs::remove_file(&path).ok();
10067    }
10068
10069    /// The width shuffle permutes by is the stored element's, which a
10070    /// `datatype` override moves away from the carrier type `T`: recording
10071    /// `T`'s width would permute a 4-byte element as four 1-byte ones and
10072    /// hand libhdf5 a chunk it unshuffles into different bytes.
10073    #[test]
10074    fn shuffle_records_the_stored_element_width() {
10075        use crate::format::messages::datatype::DatatypeMessage;
10076        use crate::format::messages::filter::FilterPipeline;
10077        let path = temp_path("shuffle_override_width");
10078        let bytes: Vec<u8> = (0..16u8).collect();
10079        {
10080            let file = H5File::create(&path).unwrap();
10081            file.new_dataset::<u8>()
10082                .datatype(DatatypeMessage::i32_type())
10083                .shape([4usize])
10084                .chunk(&[4])
10085                .shuffle()
10086                .create("d")
10087                .unwrap()
10088                .write_raw_bytes(&bytes)
10089                .unwrap();
10090            file.close().unwrap();
10091        }
10092        assert_eq!(stored_pipeline(&path, "d"), FilterPipeline::shuffle(4));
10093
10094        let file = H5File::open(&path).unwrap();
10095        assert_eq!(
10096            file.dataset("d").unwrap().read_raw::<i32>().unwrap(),
10097            bytes
10098                .chunks(4)
10099                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
10100                .collect::<Vec<_>>()
10101        );
10102        std::fs::remove_file(&path).ok();
10103    }
10104
10105    /// A write into a virtual dataset is refused by name rather than landing
10106    /// somewhere no reader would look: libhdf5 pushes such a write through
10107    /// the mapping into the source dataset (`H5D__virtual_write`), which this
10108    /// writer does not do.
10109    #[test]
10110    fn a_virtual_dataset_refuses_every_write() {
10111        use crate::Selection;
10112        let path = temp_path("vds_write_refused");
10113        let file = H5File::create(&path).unwrap();
10114        let ds = file
10115            .new_dataset::<i32>()
10116            .shape([16usize])
10117            .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10118            .create("vds")
10119            .unwrap();
10120        for err in [
10121            ds.write_raw(&(0..16i32).collect::<Vec<_>>()).unwrap_err(),
10122            ds.write_slice(&[0], &[2], &[1i32, 2]).unwrap_err(),
10123        ] {
10124            let msg = err.to_string();
10125            assert!(msg.contains("virtual dataset"), "{msg}");
10126        }
10127        file.close().unwrap();
10128        std::fs::remove_file(&path).ok();
10129    }
10130
10131    /// A `%b` substitution only means something when the virtual selection is
10132    /// unlimited and the source selection is not: that is the shape where
10133    /// each block draws from a different source dataset. On any other mapping
10134    /// there is only one block, so `H5D_virtual_check_mapping_post` refuses
10135    /// the specifier — and an illegal conversion is refused wherever it
10136    /// appears.
10137    #[test]
10138    fn a_printf_source_name_needs_the_mapping_shape_that_uses_it() {
10139        use crate::Selection;
10140        let path = temp_path("vds_printf");
10141        let file = H5File::create(&path).unwrap();
10142        for (f, d) in [("src_%b.h5", "src"), ("src.h5", "block_%b")] {
10143            let err = match file
10144                .new_dataset::<i32>()
10145                .shape([16usize])
10146                .virtual_mapping(Selection::All, f, d, Selection::All)
10147                .create("vds")
10148            {
10149                Ok(_) => panic!("a bounded mapping has one block, so %b names nothing"),
10150                Err(e) => e.to_string(),
10151            };
10152            assert!(err.contains("printf specifier"), "{err}");
10153        }
10154        // `%z` is not a conversion libhdf5 has, in any mapping shape.
10155        let err = match file
10156            .new_dataset::<i32>()
10157            .shape([1usize, 2])
10158            .max_shape(&[None, Some(2)])
10159            .virtual_mapping(unlimited_rows(), "src_%z.h5", "src", Selection::All)
10160            .create("vds_bad")
10161        {
10162            Ok(_) => panic!("%z is not a legal conversion"),
10163            Err(e) => e.to_string(),
10164        };
10165        assert!(err.contains("invalid format specifier"), "{err}");
10166        file.close().unwrap();
10167        std::fs::remove_file(&path).ok();
10168    }
10169
10170    /// A printf mapping stitches one source dataset per block of its
10171    /// unlimited virtual selection, and the extent stops at the first block
10172    /// with no source (`H5D__virtual_set_extent_unlim`'s printf arm, at the
10173    /// default `H5D_VDS_LAST_AVAILABLE` view and `printf_gap` 0).
10174    #[test]
10175    fn a_printf_mapping_stitches_one_source_per_block() {
10176        use crate::Selection;
10177        let path = temp_path("vds_printf_blocks");
10178        {
10179            let file = H5File::create(&path).unwrap();
10180            for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10181                file.new_dataset::<i32>()
10182                    .shape([2usize])
10183                    .create(&format!("block{b}"))
10184                    .unwrap()
10185                    .write_raw(&[base, base + 1])
10186                    .unwrap();
10187            }
10188            file.new_dataset::<i32>()
10189                .shape([1usize, 2])
10190                .max_shape(&[None, Some(2)])
10191                .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10192                .create("vds")
10193                .unwrap();
10194            file.close().unwrap();
10195        }
10196        let file = H5File::open(&path).unwrap();
10197        let ds = file.dataset("vds").unwrap();
10198        // `block2` is missing, so `block3` is never reached: two rows.
10199        assert_eq!(ds.shape(), vec![2, 2]);
10200        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10201        drop(file);
10202        std::fs::remove_file(&path).ok();
10203    }
10204
10205    /// A mapping whose source cannot be opened reads back as the fill value,
10206    /// not as an error: `H5D__virtual_open_source_dset` accepts a null source
10207    /// file and clears the error stack for a missing source dataset
10208    /// (H5Dvirtual.c:877-909), so `H5D__virtual_read_one` finds no projected
10209    /// memory space and reads nothing for it (H5Dvirtual.c:2661-2665).
10210    #[test]
10211    fn a_source_that_cannot_be_opened_reads_as_the_fill_value() {
10212        use crate::{Hyperslab, HyperslabBlock, Selection};
10213        let block = |start: u64, end: u64| Selection::Hyperslab {
10214            rank: 1,
10215            form: Hyperslab::Blocks(vec![HyperslabBlock {
10216                start: vec![start],
10217                end: vec![end],
10218            }]),
10219        };
10220        let path = temp_path("vds_absent_source");
10221        {
10222            let file = H5File::create(&path).unwrap();
10223            file.new_dataset::<i32>()
10224                .shape([4usize])
10225                .create("here")
10226                .unwrap()
10227                .write_raw(&[1i32, 2, 3, 4])
10228                .unwrap();
10229            file.new_dataset::<i32>()
10230                .shape([12usize])
10231                .fill_value(-3i32)
10232                .virtual_mapping(block(0, 3), ".", "here", block(0, 3))
10233                // A dataset that is not in this file.
10234                .virtual_mapping(block(4, 7), ".", "absent", block(0, 3))
10235                // A file that does not exist beside this one.
10236                .virtual_mapping(block(8, 11), "no_such_vds_source.h5", "src", block(0, 3))
10237                .create("vds")
10238                .unwrap();
10239            file.close().unwrap();
10240        }
10241        let file = H5File::open(&path).unwrap();
10242        let ds = file.dataset("vds").unwrap();
10243        assert_eq!(
10244            ds.read_raw::<i32>().unwrap(),
10245            vec![1, 2, 3, 4, -3, -3, -3, -3, -3, -3, -3, -3]
10246        );
10247        // The same rule on the slice path, which stitches the whole image
10248        // before extracting the region.
10249        assert_eq!(
10250            ds.read_slice::<i32>(&[2], &[6]).unwrap(),
10251            vec![3, 4, -3, -3, -3, -3]
10252        );
10253        drop(file);
10254        std::fs::remove_file(&path).ok();
10255    }
10256
10257    /// `%%` is an escaped literal `%`, not a substitution: the mapping is an
10258    /// ordinary bounded one, and the source it resolves against is the name
10259    /// with a single `%` in it.
10260    #[test]
10261    fn an_escaped_percent_is_a_literal_in_a_source_name() {
10262        use crate::Selection;
10263        let path = temp_path("vds_escaped_pct");
10264        {
10265            let file = H5File::create(&path).unwrap();
10266            file.new_dataset::<i32>()
10267                .shape([4usize])
10268                .create("od%d")
10269                .unwrap()
10270                .write_raw(&[5i32, 6, 7, 8])
10271                .unwrap();
10272            file.new_dataset::<i32>()
10273                .shape([4usize])
10274                .virtual_mapping(Selection::All, ".", "od%%d", Selection::All)
10275                .create("vds")
10276                .unwrap();
10277            file.close().unwrap();
10278        }
10279        let file = H5File::open(&path).unwrap();
10280        let ds = file.dataset("vds").unwrap();
10281        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![5, 6, 7, 8]);
10282        // The stored name keeps its escape; only the resolution unescapes.
10283        assert_eq!(ds.virtual_mappings().unwrap()[0].source_dset_name, "od%%d");
10284        drop(file);
10285        std::fs::remove_file(&path).ok();
10286    }
10287
10288    /// An unlimited mapping takes its extent from the source it names, so a
10289    /// virtual dataset written with one reports the source's rows, not the
10290    /// seed extent its dataspace message stores
10291    /// (`H5D__virtual_set_extent_unlim`). Same file, so the resolution runs
10292    /// without opening another one.
10293    #[test]
10294    fn an_unlimited_mapping_takes_its_extent_from_its_source() {
10295        let path = temp_path("vds_unlimited");
10296        {
10297            let file = H5File::create(&path).unwrap();
10298            file.new_dataset::<i32>()
10299                .shape([5usize, 2])
10300                .chunk(&[5, 2])
10301                .max_shape(&[None, Some(2)])
10302                .create("src")
10303                .unwrap()
10304                .write_raw(&(0..10i32).collect::<Vec<_>>())
10305                .unwrap();
10306            file.new_dataset::<i32>()
10307                .shape([1usize, 2])
10308                .max_shape(&[None, Some(2)])
10309                .virtual_mapping(unlimited_rows(), ".", "src", unlimited_rows())
10310                .create("vds")
10311                .unwrap();
10312            file.close().unwrap();
10313        }
10314        let file = H5File::open(&path).unwrap();
10315        let ds = file.dataset("vds").unwrap();
10316        assert_eq!(ds.shape(), vec![5, 2]);
10317        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..10).collect::<Vec<i32>>());
10318        drop(file);
10319        std::fs::remove_file(&path).ok();
10320    }
10321
10322    /// The blocks-0/1/3 printf file every dataset-access test below reads,
10323    /// laid out exactly like `a_printf_mapping_stitches_one_source_per_block`
10324    /// so the gap is the only thing that changes.
10325    fn printf_gap_file(tag: &str) -> std::path::PathBuf {
10326        use crate::Selection;
10327        let path = temp_path(tag);
10328        let file = H5File::create(&path).unwrap();
10329        for (b, base) in [(0, 0i32), (1, 100), (3, 300)] {
10330            file.new_dataset::<i32>()
10331                .shape([2usize])
10332                .create(&format!("block{b}"))
10333                .unwrap()
10334                .write_raw(&[base, base + 1])
10335                .unwrap();
10336        }
10337        file.new_dataset::<i32>()
10338            .shape([1usize, 2])
10339            .max_shape(&[None, Some(2)])
10340            .fill_value(-7i32)
10341            .virtual_mapping(unlimited_rows(), ".", "block%b", Selection::All)
10342            .create("vds")
10343            .unwrap();
10344        file.close().unwrap();
10345        path
10346    }
10347
10348    /// `H5Pset_virtual_printf_gap` lets the block scan look past a missing
10349    /// source, and the blocks it looked past stay inside the extent reading
10350    /// as the fill value (H5Dvirtual.c:1519-1614, :2661-2665). Measured
10351    /// against libhdf5 1.14.6 through h5py's `h5p.PropDAID`: gap 0 gives
10352    /// two rows, gap 1 and gap 2 both give four with row 2 filled.
10353    #[test]
10354    fn a_printf_gap_looks_past_the_missing_block() {
10355        use crate::DatasetAccess;
10356        let path = printf_gap_file("vds_printf_gap");
10357        let file = H5File::open(&path).unwrap();
10358        for (gap, shape, data) in [
10359            (0u64, vec![2usize, 2], vec![0i32, 1, 100, 101]),
10360            (1, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10361            (2, vec![4, 2], vec![0, 1, 100, 101, -7, -7, 300, 301]),
10362        ] {
10363            let ds = file
10364                .dataset_with("vds", DatasetAccess::new().virtual_printf_gap(gap))
10365                .unwrap();
10366            assert_eq!(ds.shape(), shape, "gap {gap}");
10367            assert_eq!(ds.read_raw::<i32>().unwrap(), data, "gap {gap}");
10368        }
10369        // Back to the default: the extent follows the properties the open
10370        // names, in both directions.
10371        let ds = file.dataset("vds").unwrap();
10372        assert_eq!(ds.shape(), vec![2, 2]);
10373        assert_eq!(ds.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10374        drop(file);
10375        std::fs::remove_file(&path).ok();
10376    }
10377
10378    /// A relatively-named source is found next to the virtual dataset even
10379    /// when the process is somewhere else entirely: `H5F_prefix_open_file`
10380    /// tries the primary file's `H5F_EXTPATH` — the directory it was opened
10381    /// from — before the bare relative name against the working directory
10382    /// (H5Fint.c:952-977). Measured against libhdf5 1.14.6 through h5py: a
10383    /// `VirtualSource("src.h5", ...)` beside its VDS reads its data with
10384    /// `HDF5_VDS_PREFIX` unset and the working directory elsewhere; before
10385    /// the reader took that step it read back all fill value.
10386    #[test]
10387    fn a_relative_source_resolves_next_to_the_virtual_dataset() {
10388        use crate::Selection;
10389        let dir = std::env::temp_dir().join(format!(
10390            "rust_hdf5_vds_beside_{}_{:?}",
10391            std::process::id(),
10392            std::thread::current().id()
10393        ));
10394        std::fs::create_dir_all(&dir).unwrap();
10395        {
10396            let file = H5File::create(dir.join("src.h5")).unwrap();
10397            file.new_dataset::<i32>()
10398                .shape([2usize, 4])
10399                .create("data")
10400                .unwrap()
10401                .write_raw(&(0..8i32).collect::<Vec<_>>())
10402                .unwrap();
10403            file.close().unwrap();
10404        }
10405        {
10406            let file = H5File::create(dir.join("v.h5")).unwrap();
10407            file.new_dataset::<i32>()
10408                .shape([2usize, 4])
10409                .fill_value(-9i32)
10410                // Named relatively, as h5py's `VirtualSource("src.h5", ...)`
10411                // stores it — nothing in the file says where it lives.
10412                .virtual_mapping(Selection::All, "src.h5", "data", Selection::All)
10413                .create("v")
10414                .unwrap();
10415            file.close().unwrap();
10416        }
10417        // The working directory is the crate root under `cargo test`, not
10418        // `dir`, so only the extpath step can find `src.h5`.
10419        assert_ne!(std::env::current_dir().unwrap(), dir);
10420        let file = H5File::open(dir.join("v.h5")).unwrap();
10421        let ds = file.dataset("v").unwrap();
10422        assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8i32).collect::<Vec<_>>());
10423        drop(file);
10424        std::fs::remove_dir_all(&dir).ok();
10425    }
10426
10427    /// The first open of a virtual dataset fixes its access properties for
10428    /// every open that overlaps it: only the open that finds no shared info
10429    /// in `H5FO_opened` runs `H5D__open_oid(dataset, dapl_id)`, and a later
10430    /// one just points at that shared info without ever reading its own dapl
10431    /// (H5Dint.c:1496-1500, :1523-1528) — the view and the gap live in the
10432    /// shared layout storage `H5D__virtual_init` filled from that first dapl
10433    /// (H5Dvirtual.c:2178-2188). Measured against libhdf5 1.14.6 and 2.0.0
10434    /// through `h5d.open(..., dapl=...)` on the printf-gap VDS below: opening
10435    /// gap 0 then gap 1 gives both handles two rows; with every handle closed,
10436    /// opening gap 1 then gap 0 gives both four rows and the gap row reads as
10437    /// fill; with every handle closed again, gap 0 alone is back to two rows.
10438    #[test]
10439    fn the_first_open_of_a_virtual_dataset_fixes_the_properties_for_later_opens() {
10440        use crate::DatasetAccess;
10441        let path = printf_gap_file("vds_printf_first_open");
10442        let file = H5File::open(&path).unwrap();
10443        let gap = |g: u64| DatasetAccess::new().virtual_printf_gap(g);
10444        {
10445            let a = file.dataset_with("vds", gap(0)).unwrap();
10446            let b = file.dataset_with("vds", gap(1)).unwrap();
10447            assert_eq!(a.shape(), vec![2, 2]);
10448            assert_eq!(b.shape(), vec![2, 2]);
10449            assert_eq!(b.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10450        }
10451        {
10452            let c = file.dataset_with("vds", gap(1)).unwrap();
10453            let d = file.dataset_with("vds", gap(0)).unwrap();
10454            assert_eq!(c.shape(), vec![4, 2]);
10455            assert_eq!(d.shape(), vec![4, 2]);
10456            assert_eq!(
10457                d.read_raw::<i32>().unwrap(),
10458                vec![0, 1, 100, 101, -7, -7, 300, 301]
10459            );
10460        }
10461        let e = file.dataset_with("vds", gap(0)).unwrap();
10462        assert_eq!(e.shape(), vec![2, 2]);
10463        assert_eq!(e.read_raw::<i32>().unwrap(), vec![0, 1, 100, 101]);
10464        drop(e);
10465        drop(file);
10466        std::fs::remove_file(&path).ok();
10467    }
10468
10469    /// `H5D_VDS_FIRST_MISSING` ignores the printf gap: `H5D__virtual_init`
10470    /// reads the gap property only under `H5D_VDS_LAST_AVAILABLE` and forces
10471    /// it to 0 otherwise (H5Dvirtual.c:2182-2188). Measured against libhdf5
10472    /// 1.14.6: gap 2 under this view still gives two rows.
10473    #[test]
10474    fn the_first_missing_view_ignores_the_printf_gap() {
10475        use crate::{DatasetAccess, VirtualView};
10476        let path = printf_gap_file("vds_printf_first_missing");
10477        let file = H5File::open(&path).unwrap();
10478        for gap in [0u64, 2] {
10479            let ds = file
10480                .dataset_with(
10481                    "vds",
10482                    DatasetAccess::new()
10483                        .virtual_view(VirtualView::FirstMissing)
10484                        .virtual_printf_gap(gap),
10485                )
10486                .unwrap();
10487            assert_eq!(ds.shape(), vec![2, 2], "gap {gap}");
10488            assert_eq!(
10489                ds.read_raw::<i32>().unwrap(),
10490                vec![0, 1, 100, 101],
10491                "gap {gap}"
10492            );
10493        }
10494        drop(file);
10495        std::fs::remove_file(&path).ok();
10496    }
10497
10498    /// On a mapping unlimited on both sides the view is
10499    /// `H5S_hyper_get_clip_extent_match`'s `incl_trail`
10500    /// (H5Dvirtual.c:1447-1451): with a stride wider than its block, the
10501    /// extent under `H5D_VDS_LAST_AVAILABLE` ends at the last mapped row,
10502    /// and under `H5D_VDS_FIRST_MISSING` it runs on to where the next block
10503    /// would start. Measured against libhdf5 1.14.6 over a three-row source
10504    /// with stride 3 and block 2: two rows and three rows.
10505    #[test]
10506    fn the_view_decides_whether_a_trailing_gap_is_inside_the_extent() {
10507        use crate::format::selection::UNLIMITED;
10508        use crate::{DatasetAccess, Hyperslab, RegularHyperslab, Selection, VirtualView};
10509        let strided = || Selection::Hyperslab {
10510            rank: 2,
10511            form: Hyperslab::Regular(RegularHyperslab {
10512                start: vec![0, 0],
10513                stride: vec![3, 1],
10514                count: vec![UNLIMITED, 1],
10515                block: vec![2, 2],
10516            }),
10517        };
10518        let path = temp_path("vds_view_trail");
10519        {
10520            let file = H5File::create(&path).unwrap();
10521            file.new_dataset::<i32>()
10522                .shape([3usize, 2])
10523                .max_shape(&[None, Some(2)])
10524                .chunk(&[1, 2])
10525                .create("src")
10526                .unwrap()
10527                .write_raw(&(0..6i32).collect::<Vec<_>>())
10528                .unwrap();
10529            file.new_dataset::<i32>()
10530                .shape([1usize, 2])
10531                .max_shape(&[None, Some(2)])
10532                .fill_value(-9i32)
10533                .virtual_mapping(strided(), ".", "src", strided())
10534                .create("vds")
10535                .unwrap();
10536            file.close().unwrap();
10537        }
10538        let file = H5File::open(&path).unwrap();
10539        {
10540            let last = file.dataset("vds").unwrap();
10541            assert_eq!(last.shape(), vec![2, 2]);
10542            assert_eq!(last.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3]);
10543        }
10544        // The handle above is gone, so this open is the one that resolves.
10545        let first = file
10546            .dataset_with(
10547                "vds",
10548                DatasetAccess::new().virtual_view(VirtualView::FirstMissing),
10549            )
10550            .unwrap();
10551        assert_eq!(first.shape(), vec![3, 2]);
10552        assert_eq!(first.read_raw::<i32>().unwrap(), vec![0, 1, 2, 3, -9, -9]);
10553        drop(file);
10554        std::fs::remove_file(&path).ok();
10555    }
10556
10557    /// The property list reads back what was set (`H5Pget_virtual_view`,
10558    /// `H5Pget_virtual_printf_gap`), and the gap `HSIZE_UNDEF` that
10559    /// `H5Pset_virtual_printf_gap` refuses is refused by the open that would
10560    /// have used it.
10561    #[test]
10562    fn the_access_property_list_reads_back_and_refuses_hsize_undef() {
10563        use crate::{DatasetAccess, VirtualView};
10564        let plist = DatasetAccess::new();
10565        assert_eq!(plist.view(), VirtualView::LastAvailable);
10566        assert_eq!(plist.printf_gap(), 0);
10567        let set = plist
10568            .virtual_view(VirtualView::FirstMissing)
10569            .virtual_printf_gap(4);
10570        assert_eq!(set.view(), VirtualView::FirstMissing);
10571        // The *property* keeps what was set even though the resolution under
10572        // this view scans with 0.
10573        assert_eq!(set.printf_gap(), 4);
10574
10575        let path = printf_gap_file("vds_printf_gap_undef");
10576        let file = H5File::open(&path).unwrap();
10577        let err = match file.dataset_with("vds", DatasetAccess::new().virtual_printf_gap(u64::MAX))
10578        {
10579            Ok(_) => panic!("HSIZE_UNDEF is not a valid printf gap size"),
10580            Err(e) => e.to_string(),
10581        };
10582        assert!(err.contains("HSIZE_UNDEF"), "{err}");
10583        drop(file);
10584        std::fs::remove_file(&path).ok();
10585    }
10586
10587    /// The rank-2 `count = (H5S_UNLIMITED, 1)`, `block = (1, 2)` selection
10588    /// both sides of an unlimited row-wise mapping use.
10589    fn unlimited_rows() -> crate::Selection {
10590        use crate::format::selection::UNLIMITED;
10591        use crate::{Hyperslab, RegularHyperslab, Selection};
10592        Selection::Hyperslab {
10593            rank: 2,
10594            form: Hyperslab::Regular(RegularHyperslab {
10595                start: vec![0, 0],
10596                stride: vec![1, 1],
10597                count: vec![UNLIMITED, 1],
10598                block: vec![1, 2],
10599            }),
10600        }
10601    }
10602
10603    /// An unlimited virtual selection over a *limited* source selection is
10604    /// the printf shape, and without a `%b` in a source name there is no
10605    /// second dataset to fill the second block —
10606    /// `H5D_virtual_check_mapping_post` refuses it, and so does this.
10607    #[test]
10608    fn an_unlimited_virtual_selection_over_a_limited_source_is_refused() {
10609        use crate::Selection;
10610        let path = temp_path("vds_unlim_limited_src");
10611        let file = H5File::create(&path).unwrap();
10612        let err = match file
10613            .new_dataset::<i32>()
10614            .shape([1usize, 2])
10615            .max_shape(&[None, Some(2)])
10616            .virtual_mapping(unlimited_rows(), "src.h5", "src", Selection::All)
10617            .create("vds")
10618        {
10619            Ok(_) => panic!("no printf substitution names the mapping's later blocks"),
10620            Err(e) => e.to_string(),
10621        };
10622        assert!(err.contains("printf"), "{err}");
10623        file.close().unwrap();
10624        std::fs::remove_file(&path).ok();
10625    }
10626
10627    /// A virtual dataset stores nothing of its own, so it cannot also be one
10628    /// of the storage classes that do.
10629    #[test]
10630    fn a_virtual_dataset_cannot_also_be_chunked_or_external() {
10631        use crate::Selection;
10632        let path = temp_path("vds_exclusive");
10633        let file = H5File::create(&path).unwrap();
10634        let builder = || {
10635            file.new_dataset::<i32>().shape([16usize]).virtual_mapping(
10636                Selection::All,
10637                "src.h5",
10638                "src",
10639                Selection::All,
10640            )
10641        };
10642        for (which, res) in [
10643            ("chunked", builder().chunk(&[4]).create("a")),
10644            ("compact", builder().compact().create("b")),
10645            (
10646                "external",
10647                builder().external(&[("x.raw", 0, 64)]).create("c"),
10648            ),
10649            ("references", builder().object_references().create("d")),
10650        ] {
10651            match res {
10652                Ok(_) => panic!("a virtual dataset cannot also be {which}"),
10653                Err(e) => assert!(e.to_string().contains("virtual dataset"), "{which}: {e}"),
10654            }
10655        }
10656        file.close().unwrap();
10657        std::fs::remove_file(&path).ok();
10658    }
10659
10660    /// Deleting a virtual dataset frees the global heap object its mapping
10661    /// list lived in — `H5D__virtual_delete` reaches `H5HG_remove` — so the
10662    /// next dataset's own heap object reuses that space instead of the file
10663    /// growing by a whole collection per deleted virtual dataset.
10664    #[test]
10665    fn deleting_a_virtual_dataset_frees_its_mapping_list() {
10666        use crate::Selection;
10667        let path = temp_path("vds_delete");
10668        let baseline = {
10669            let file = H5File::create(&path).unwrap();
10670            file.new_dataset::<i32>()
10671                .shape([16usize])
10672                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10673                .create("vds")
10674                .unwrap();
10675            file.delete_dataset("vds").unwrap();
10676            file.close().unwrap();
10677            std::fs::metadata(&path).unwrap().len()
10678        };
10679        std::fs::remove_file(&path).ok();
10680
10681        // Ten more create-then-delete rounds must land on the same file size:
10682        // each round's heap object is removed, its collection becomes empty
10683        // and returns to the allocator, and the next round takes it back.
10684        let file = H5File::create(&path).unwrap();
10685        for i in 0..10 {
10686            let name = format!("vds{i}");
10687            file.new_dataset::<i32>()
10688                .shape([16usize])
10689                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
10690                .create(&name)
10691                .unwrap();
10692            file.delete_dataset(&name).unwrap();
10693        }
10694        file.close().unwrap();
10695        assert_eq!(std::fs::metadata(&path).unwrap().len(), baseline);
10696        std::fs::remove_file(&path).ok();
10697    }
10698
10699    /// [`H5Dataset::storage_layout`] tells the four classes apart — the
10700    /// negative case for any one class is simply that it is not another.
10701    #[test]
10702    fn storage_layout_reports_each_class() {
10703        use crate::StorageLayout;
10704        let path = temp_path("storage_layout");
10705        {
10706            let file = H5File::create(&path).unwrap();
10707            file.new_dataset::<i32>()
10708                .shape([4usize])
10709                .create("contig")
10710                .unwrap();
10711            file.new_dataset::<i32>()
10712                .shape([4usize])
10713                .compact()
10714                .create("compact")
10715                .unwrap();
10716            file.new_dataset::<i32>()
10717                .shape([8usize])
10718                .chunk(&[4])
10719                .create("chunked")
10720                .unwrap();
10721            file.close().unwrap();
10722        }
10723        let file = H5File::open(&path).unwrap();
10724        assert_eq!(
10725            file.dataset("contig").unwrap().storage_layout().unwrap(),
10726            StorageLayout::Contiguous
10727        );
10728        assert_eq!(
10729            file.dataset("compact").unwrap().storage_layout().unwrap(),
10730            StorageLayout::Compact
10731        );
10732        assert_eq!(
10733            file.dataset("chunked").unwrap().storage_layout().unwrap(),
10734            StorageLayout::Chunked
10735        );
10736    }
10737
10738    /// Read-mode-only accessor, matching `datatype()`'s own contract.
10739    #[test]
10740    fn storage_layout_errors_in_write_mode() {
10741        let path = temp_path("storage_layout_write_mode");
10742        let file = H5File::create(&path).unwrap();
10743        let ds = file
10744            .new_dataset::<i32>()
10745            .shape([4usize])
10746            .create("data")
10747            .unwrap();
10748        assert!(ds.storage_layout().is_err());
10749        file.close().unwrap();
10750    }
10751
10752    /// [`H5Dataset::chunk_index`] reports the real on-disk index kind
10753    /// (extensible array for one unlimited dimension, version-1 B-tree
10754    /// under a legacy libver bound) and `None` for an unchunked dataset —
10755    /// the negative case.
10756    #[test]
10757    fn chunk_index_reports_the_stored_kind() {
10758        use crate::ChunkIndex;
10759        let path = temp_path("chunk_index");
10760        {
10761            let file = H5File::create(&path).unwrap();
10762            file.new_dataset::<i32>()
10763                .shape([4usize])
10764                .create("contig")
10765                .unwrap();
10766            file.new_dataset::<i32>()
10767                .shape([16usize])
10768                .chunk(&[4])
10769                .max_shape(&[None])
10770                .create("earray")
10771                .unwrap();
10772            file.set_libver_latest(false).unwrap();
10773            file.new_dataset::<i32>()
10774                .shape([8usize])
10775                .chunk(&[4])
10776                .max_shape(&[None])
10777                .create("btree1")
10778                .unwrap();
10779            file.close().unwrap();
10780        }
10781        let file = H5File::open(&path).unwrap();
10782        assert_eq!(file.dataset("contig").unwrap().chunk_index().unwrap(), None);
10783        assert_eq!(
10784            file.dataset("earray").unwrap().chunk_index().unwrap(),
10785            Some(ChunkIndex::ExtensibleArray)
10786        );
10787        assert_eq!(
10788            file.dataset("btree1").unwrap().chunk_index().unwrap(),
10789            Some(ChunkIndex::BtreeV1)
10790        );
10791    }
10792
10793    /// [`H5Dataset::filters`] reports the stored pipeline in order — and
10794    /// the negative case: an unfiltered dataset reports an empty pipeline,
10795    /// not an error.
10796    #[test]
10797    fn filters_reports_the_stored_pipeline() {
10798        use crate::format::messages::filter::{FILTER_DEFLATE, FILTER_SHUFFLE, FLAG_OPTIONAL};
10799        let path = temp_path("filters");
10800        {
10801            let file = H5File::create(&path).unwrap();
10802            file.new_dataset::<i32>()
10803                .shape([16usize])
10804                .create("unfiltered")
10805                .unwrap();
10806            file.new_dataset::<i32>()
10807                .shape([16usize])
10808                .chunk(&[4])
10809                .shuffle()
10810                .deflate(6)
10811                .create("filtered")
10812                .unwrap();
10813            file.close().unwrap();
10814        }
10815        let file = H5File::open(&path).unwrap();
10816        assert_eq!(
10817            file.dataset("unfiltered").unwrap().filters().unwrap(),
10818            Vec::new()
10819        );
10820        let filters = file.dataset("filtered").unwrap().filters().unwrap();
10821        assert_eq!(filters.len(), 2);
10822        assert_eq!(filters[0].id, FILTER_SHUFFLE);
10823        assert_eq!(filters[0].flags, FLAG_OPTIONAL);
10824        assert_eq!(filters[1].id, FILTER_DEFLATE);
10825        assert_eq!(filters[1].cd_values, vec![6]);
10826    }
10827
10828    /// [`H5Dataset::fill_value`] reports the explicit bytes for a dataset
10829    /// created with `.fill_value(...)`, and the negative case: a dataset
10830    /// with no fill value set reports [`FillValue::Default`], not an error.
10831    /// `FillValue::Undefined` has no constructor on either this crate's
10832    /// writer or h5py's public API, so it is not exercised here.
10833    #[test]
10834    fn fill_value_reports_the_stored_value() {
10835        use crate::FillValue;
10836        let path = temp_path("fill_value");
10837        {
10838            let file = H5File::create(&path).unwrap();
10839            file.new_dataset::<i32>()
10840                .shape([4usize])
10841                .create("unset")
10842                .unwrap();
10843            file.new_dataset::<i32>()
10844                .shape([4usize])
10845                .fill_value(-7i32)
10846                .create("set")
10847                .unwrap();
10848            file.close().unwrap();
10849        }
10850        let file = H5File::open(&path).unwrap();
10851        assert_eq!(
10852            file.dataset("unset").unwrap().fill_value().unwrap(),
10853            FillValue::Default
10854        );
10855        assert_eq!(
10856            file.dataset("set").unwrap().fill_value().unwrap(),
10857            FillValue::UserDefined((-7i32).to_le_bytes().to_vec())
10858        );
10859    }
10860
10861    /// The three `H5D__efl_construct` / `H5Pset_external` rules an
10862    /// `H5O_EFL_UNLIMITED` slot lives inside: it may only be the last slot,
10863    /// an unlimited dataspace must have one, and only the first dimension
10864    /// may be extendible.
10865    #[test]
10866    fn the_unlimited_external_slot_keeps_its_three_rules() {
10867        use crate::format::messages::external_file_list::UNLIMITED;
10868        let path = temp_path("efl_unlim_rules");
10869        let file = H5File::create(&path).unwrap();
10870
10871        // "previous file size is unlimited": nothing behind an unlimited slot
10872        // could ever be reached.
10873        let err = match file
10874            .new_dataset::<i32>()
10875            .shape([8usize])
10876            .external(&[("a.raw", 0, UNLIMITED), ("b.raw", 0, 32)])
10877            .create("mid")
10878        {
10879            Ok(_) => panic!("an unlimited slot absorbs everything behind it"),
10880            Err(e) => e.to_string(),
10881        };
10882        assert!(err.contains("only be the last"), "{err}");
10883
10884        // "unlimited dataspace but finite storage".
10885        let err = match file
10886            .new_dataset::<i32>()
10887            .shape([8usize])
10888            .max_shape(&[None])
10889            .external(&[("a.raw", 0, 32)])
10890            .create("finite")
10891        {
10892            Ok(_) => panic!("no finite reservation covers an unlimited extent"),
10893            Err(e) => e.to_string(),
10894        };
10895        assert!(err.contains("unlimited dataspace"), "{err}");
10896
10897        // "only the first dimension can be extendible".
10898        let err = match file
10899            .new_dataset::<i32>()
10900            .shape([2usize, 4])
10901            .max_shape(&[Some(2), None])
10902            .external(&[("a.raw", 0, UNLIMITED)])
10903            .create("dim1")
10904        {
10905            Ok(_) => panic!("only the slowest-varying dimension may be extendible"),
10906            Err(e) => e.to_string(),
10907        };
10908        assert!(err.contains("only the first dimension"), "{err}");
10909
10910        // And the legal shape: an unlimited last slot under an unlimited
10911        // first dimension.
10912        file.new_dataset::<i32>()
10913            .shape([8usize])
10914            .max_shape(&[None])
10915            .external(&[("a.raw", 0, 16), ("b.raw", 0, UNLIMITED)])
10916            .create("ok")
10917            .unwrap()
10918            .write_raw(&(0..8i32).collect::<Vec<_>>())
10919            .unwrap();
10920        file.close().unwrap();
10921        std::fs::remove_file(&path).ok();
10922        for raw in ["a.raw", "b.raw"] {
10923            std::fs::remove_file(raw).ok();
10924        }
10925    }
10926
10927    /// An unlimited slot reserves nothing, so a read of it is bounded by the
10928    /// dataset's extent and by what the file physically holds: the tail past
10929    /// the end of a short raw file reads back as zero, exactly as
10930    /// `H5D__efl_read` fills it.
10931    #[test]
10932    fn an_unlimited_external_slot_reads_zero_past_the_end_of_its_file() {
10933        use crate::format::messages::external_file_list::UNLIMITED;
10934        let dir = std::env::temp_dir().join(format!("rh5_efl_short_{}", std::process::id()));
10935        std::fs::create_dir_all(&dir).unwrap();
10936        let path = dir.join("f.h5");
10937        let raw = dir.join("short.raw");
10938        // Four elements' worth of bytes for an eight-element dataset.
10939        std::fs::write(&raw, [0u8; 16]).unwrap();
10940        {
10941            let file = H5File::create(&path).unwrap();
10942            file.new_dataset::<i32>()
10943                .shape([8usize])
10944                .max_shape(&[None])
10945                .external(&[(raw.to_str().unwrap(), 0, UNLIMITED)])
10946                .create("data")
10947                .unwrap();
10948            file.close().unwrap();
10949        }
10950        let file = H5File::open(&path).unwrap();
10951        assert_eq!(
10952            file.dataset("data").unwrap().read_raw::<i32>().unwrap(),
10953            vec![0i32; 8]
10954        );
10955        drop(file);
10956        std::fs::remove_dir_all(&dir).ok();
10957    }
10958
10959    /// [`H5Dataset::external_files`] reports the stored segment list in
10960    /// order, and the negative case: a dataset whose data lives in this
10961    /// file reports an empty list, not an error.
10962    #[test]
10963    fn external_files_reports_the_stored_segments() {
10964        let path = temp_path("external_files");
10965        {
10966            let file = H5File::create(&path).unwrap();
10967            file.new_dataset::<i32>()
10968                .shape([4usize])
10969                .create("contig")
10970                .unwrap();
10971            file.new_dataset::<i32>()
10972                .shape([16usize])
10973                .external(&[("a.raw", 0, 32), ("b.raw", 8, 32)])
10974                .create("external")
10975                .unwrap();
10976            file.close().unwrap();
10977        }
10978        let file = H5File::open(&path).unwrap();
10979        assert_eq!(
10980            file.dataset("contig").unwrap().external_files().unwrap(),
10981            Vec::new()
10982        );
10983        let segments = file.dataset("external").unwrap().external_files().unwrap();
10984        assert_eq!(segments.len(), 2);
10985        assert_eq!(segments[0].name, "a.raw");
10986        assert_eq!(segments[0].offset, 0);
10987        assert_eq!(segments[0].size, 32);
10988        assert_eq!(segments[1].name, "b.raw");
10989        assert_eq!(segments[1].offset, 8);
10990        assert_eq!(segments[1].size, 32);
10991    }
10992
10993    /// [`H5Dataset::max_shape`] reports an unlimited axis as `None` and a
10994    /// fixed one as its current size — and the negative case: a dataset
10995    /// with no maximum-dimensions message reports max == current, not an
10996    /// error.
10997    #[test]
10998    fn max_shape_reports_unlimited_and_fixed_axes() {
10999        let path = temp_path("max_shape");
11000        {
11001            let file = H5File::create(&path).unwrap();
11002            file.new_dataset::<i32>()
11003                .shape([4usize, 8])
11004                .create("fixed")
11005                .unwrap();
11006            file.new_dataset::<i32>()
11007                .shape([4usize, 8])
11008                .chunk(&[2, 8])
11009                .max_shape(&[None, Some(8)])
11010                .create("unlimited")
11011                .unwrap();
11012            file.close().unwrap();
11013        }
11014        let file = H5File::open(&path).unwrap();
11015        assert_eq!(
11016            file.dataset("fixed").unwrap().max_shape().unwrap(),
11017            vec![Some(4), Some(8)]
11018        );
11019        assert_eq!(
11020            file.dataset("unlimited").unwrap().max_shape().unwrap(),
11021            vec![None, Some(8)]
11022        );
11023    }
11024
11025    /// [`H5Dataset::virtual_mappings`] reports the stored source/virtual
11026    /// mapping list in order, and the negative case: a dataset with no
11027    /// virtual layout reports an empty list, not an error.
11028    #[test]
11029    fn virtual_mappings_reports_the_stored_mappings() {
11030        use crate::Selection;
11031        let path = temp_path("virtual_mappings");
11032        {
11033            let file = H5File::create(&path).unwrap();
11034            file.new_dataset::<i32>()
11035                .shape([4usize])
11036                .create("plain")
11037                .unwrap();
11038            file.new_dataset::<i32>()
11039                .shape([16usize])
11040                .virtual_mapping(Selection::All, "src.h5", "src", Selection::All)
11041                .create("vds")
11042                .unwrap();
11043            file.close().unwrap();
11044        }
11045        let file = H5File::open(&path).unwrap();
11046        assert_eq!(
11047            file.dataset("plain").unwrap().virtual_mappings().unwrap(),
11048            Vec::new()
11049        );
11050        let mappings = file.dataset("vds").unwrap().virtual_mappings().unwrap();
11051        assert_eq!(mappings.len(), 1);
11052        assert_eq!(mappings[0].source_file_name, "src.h5");
11053        assert_eq!(mappings[0].source_dset_name, "src");
11054        assert_eq!(mappings[0].source_selection, Selection::All);
11055        assert_eq!(mappings[0].virtual_selection, Selection::All);
11056    }
11057}