Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use crate::attribute::AttrBuilder;
8use crate::error::{Hdf5Error, Result};
9use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
10use crate::format::messages::datatype::DatatypeMessage;
11use crate::types::H5Type;
12
13// ---------------------------------------------------------------------------
14// DatasetBuilder
15// ---------------------------------------------------------------------------
16
17/// A fluent builder for creating datasets.
18///
19/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
20///
21/// ```no_run
22/// # use rust_hdf5::H5File;
23/// let file = H5File::create("builder.h5").unwrap();
24/// let ds = file.new_dataset::<f32>()
25///     .shape(&[10, 20])
26///     .create("temperatures")
27///     .unwrap();
28/// ```
29pub struct DatasetBuilder<T: H5Type> {
30    file_inner: SharedInner,
31    shape: Option<Vec<usize>>,
32    chunk_dims: Option<Vec<usize>>,
33    max_shape: Option<Vec<Option<usize>>>,
34    deflate_level: Option<u32>,
35    shuffle_deflate_level: Option<u32>,
36    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
37    group_path: Option<String>,
38    fill_value: Option<Vec<u8>>,
39    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
40    _marker: std::marker::PhantomData<T>,
41}
42
43impl<T: H5Type> DatasetBuilder<T> {
44    pub(crate) fn new(file_inner: SharedInner) -> Self {
45        Self {
46            file_inner,
47            shape: None,
48            chunk_dims: None,
49            max_shape: None,
50            deflate_level: None,
51            shuffle_deflate_level: None,
52            custom_pipeline: None,
53            group_path: None,
54            fill_value: None,
55            datatype_override: None,
56            _marker: std::marker::PhantomData,
57        }
58    }
59
60    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
61        Self {
62            file_inner,
63            shape: None,
64            chunk_dims: None,
65            max_shape: None,
66            deflate_level: None,
67            shuffle_deflate_level: None,
68            custom_pipeline: None,
69            group_path: Some(group_path),
70            fill_value: None,
71            datatype_override: None,
72            _marker: std::marker::PhantomData,
73        }
74    }
75
76    /// Set the dataset dimensions.
77    ///
78    /// This is required before calling [`create`](Self::create).
79    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
80    #[must_use]
81    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
82        self.shape = Some(dims.as_ref().to_vec());
83        self
84    }
85
86    /// Create a scalar (0-dimensional) dataset holding a single value.
87    #[must_use]
88    pub fn scalar(mut self) -> Self {
89        self.shape = Some(vec![]);
90        self
91    }
92
93    /// Set chunk dimensions for chunked storage.
94    ///
95    /// When set, the dataset uses chunked storage with the extensible array
96    /// index. You should also call [`max_shape`](Self::max_shape) or
97    /// [`resizable`](Self::resizable) to allow extending.
98    #[must_use]
99    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
100        self.chunk_dims = Some(chunk_dims.to_vec());
101        self
102    }
103
104    /// Make all dimensions unlimited (resizable).
105    ///
106    /// This sets max_dims to u64::MAX for all dimensions.
107    #[must_use]
108    pub fn resizable(mut self) -> Self {
109        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
110        self
111    }
112
113    /// Set maximum dimensions. `None` means unlimited for that dimension.
114    #[must_use]
115    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
116        self.max_shape = Some(max.to_vec());
117        self
118    }
119
120    /// Enable deflate (gzip) compression with the given level (0-9).
121    ///
122    /// Requires chunked storage (call `.chunk()` before `.create()`).
123    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
124    #[must_use]
125    pub fn deflate(mut self, level: u32) -> Self {
126        self.deflate_level = Some(level);
127        self
128    }
129
130    /// Enable shuffle + deflate compression.
131    ///
132    /// Shuffle reorders bytes by position within elements before compression,
133    /// which typically improves compression ratios for numeric data.
134    /// Requires chunked storage.
135    #[must_use]
136    pub fn shuffle_deflate(mut self, level: u32) -> Self {
137        self.shuffle_deflate_level = Some(level);
138        self
139    }
140
141    /// Enable Zstandard compression with the given level (1-22, default 3).
142    ///
143    /// Requires chunked storage (call `.chunk()` before `.create()`).
144    #[must_use]
145    pub fn zstd(mut self, level: u32) -> Self {
146        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
147        self
148    }
149
150    /// Set a custom filter pipeline for compression.
151    ///
152    /// This takes precedence over [`deflate`](Self::deflate) and
153    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
154    #[must_use]
155    pub fn filter_pipeline(
156        mut self,
157        pipeline: crate::format::messages::filter::FilterPipeline,
158    ) -> Self {
159        self.custom_pipeline = Some(pipeline);
160        self
161    }
162
163    /// Override the stored element datatype.
164    ///
165    /// By default the dataset is created with the datatype derived from the
166    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
167    /// different on-disk datatype than the in-memory element type — for
168    /// example a reduced-precision fixed-point type that matches an N-bit
169    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
170    /// override must equal `T::element_size()`; the N-bit filter packs the
171    /// significant bits within that fixed footprint.
172    ///
173    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
174    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
175    #[must_use]
176    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
177        self.datatype_override = Some(dt);
178        self
179    }
180
181    /// Set a user-defined fill value for unwritten elements.
182    ///
183    /// Without this, datasets use the HDF5 default zero-fill. When set,
184    /// the value is written into the dataset's fill-value message
185    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
186    /// unwritten regions as this value rather than zero.
187    ///
188    /// ```no_run
189    /// # use rust_hdf5::H5File;
190    /// let file = H5File::create("fv.h5").unwrap();
191    /// let ds = file.new_dataset::<f32>()
192    ///     .shape(&[100])
193    ///     .fill_value(f32::NAN)
194    ///     .create("data")
195    ///     .unwrap();
196    /// ```
197    #[must_use]
198    pub fn fill_value(mut self, value: T) -> Self {
199        let es = T::element_size();
200        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
201        // well-defined byte representation; `element_size()` matches
202        // `size_of::<T>()`. The slice borrows `value` only for this call.
203        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
204        self.fill_value = Some(raw.to_vec());
205        self
206    }
207
208    /// Finalize and create the dataset with the given `name`.
209    ///
210    /// The name is the link name within the root group (e.g. `"data"` or
211    /// `"group1/data"` once nested groups are supported).
212    pub fn create(self, name: &str) -> Result<H5Dataset> {
213        let shape = self.shape.ok_or_else(|| {
214            Hdf5Error::InvalidState("shape must be set before calling create()".into())
215        })?;
216
217        // Build the full name: if created within a group, prefix with group path
218        let full_name = if let Some(ref gp) = self.group_path {
219            if gp == "/" {
220                name.to_string()
221            } else {
222                let trimmed = gp.trim_start_matches('/');
223                format!("{}/{}", trimmed, name)
224            }
225        } else {
226            name.to_string()
227        };
228        let group_path = self.group_path.clone();
229        let fill_value = self.fill_value.clone();
230
231        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
232        let datatype = self.datatype_override.clone().unwrap_or_else(T::hdf5_type);
233        // Size one element from the on-disk datatype, not the carrier `T`. For
234        // the default path this equals `T::element_size()`; when a `datatype()`
235        // override is set (N-bit, or a runtime `CompoundType`), the stored type
236        // — not `T` — defines the element width, so the dataspace, the raw
237        // allocation, and the `write_raw` length check all agree with the bytes
238        // libhdf5/h5py will read.
239        let element_size = datatype.element_size() as usize;
240
241        // A filter pipeline requires chunked storage. When a filter is
242        // requested without explicit chunk dimensions, store the whole
243        // dataset as a single chunk instead of silently dropping the filter
244        // on the contiguous path. (This is one whole-dataset chunk, not
245        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
246        // for large datasets.)
247        let wants_filter = self.custom_pipeline.is_some()
248            || self.shuffle_deflate_level.is_some()
249            || self.deflate_level.is_some();
250        let auto_chunk: Option<Vec<usize>> =
251            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
252                Some(shape.iter().map(|&d| d.max(1)).collect())
253            } else {
254                None
255            };
256
257        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
258            // Chunked dataset
259            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
260            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
261                max.iter()
262                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
263                    .collect()
264            } else {
265                // Default: max = current
266                dims_u64.clone()
267            };
268
269            // libhdf5 selects the chunk index from the dataspace: a v2
270            // B-tree for two or more unlimited dimensions, an extensible
271            // array for exactly one, and a fixed array when there are none.
272            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
273            let is_btree2 = n_unlimited >= 2;
274            let is_fixed_array = n_unlimited == 0;
275
276            let index = {
277                let inner = borrow_inner(&self.file_inner);
278                match &*inner {
279                    H5FileInner::Writer(writer) => {
280                        // The requested filter pipeline, if any. Both index
281                        // types that take one explicitly (fixed array and v2
282                        // B-tree) build it the same way, so resolve it once.
283                        let explicit_pipeline = || {
284                            if let Some(p) = self.custom_pipeline.clone() {
285                                p
286                            } else if let Some(level) = self.shuffle_deflate_level {
287                                crate::format::messages::filter::FilterPipeline::shuffle_deflate(
288                                    T::element_size() as u32,
289                                    level,
290                                )
291                            } else {
292                                // deflate_level (checked by wants_filter).
293                                crate::format::messages::filter::FilterPipeline::deflate(
294                                    self.deflate_level.unwrap(),
295                                )
296                            }
297                        };
298                        let idx = if is_btree2 {
299                            // Two or more unlimited dimensions: a v2 B-tree,
300                            // whose records carry the stored size and filter
301                            // mask when the dataset is compressed (libhdf5
302                            // H5D_BT2_FILT).
303                            if wants_filter {
304                                writer.create_btree_v2_dataset_with_pipeline(
305                                    &full_name,
306                                    datatype,
307                                    &dims_u64,
308                                    &max_u64,
309                                    &chunk_u64,
310                                    explicit_pipeline(),
311                                )?
312                            } else {
313                                writer.create_btree_v2_dataset(
314                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
315                                )?
316                            }
317                        } else if is_fixed_array {
318                            // A chunked dataset with no unlimited dimension
319                            // must use the fixed-array index — libhdf5
320                            // rejects an extensible-array index here. A
321                            // compressed fixed-shape dataset uses a *filtered*
322                            // fixed array (FA client id 1). The maximum shape
323                            // sizes the array, so a finite max above the
324                            // current shape stays growable.
325                            let pipeline = wants_filter.then(explicit_pipeline);
326                            writer.create_fixed_array_dataset_with_max(
327                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
328                            )?
329                        } else if let Some(pipeline) = self.custom_pipeline {
330                            writer.create_chunked_dataset_with_pipeline(
331                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
332                            )?
333                        } else if let Some(level) = self.shuffle_deflate_level {
334                            let pipeline =
335                                crate::format::messages::filter::FilterPipeline::shuffle_deflate(
336                                    T::element_size() as u32,
337                                    level,
338                                );
339                            writer.create_chunked_dataset_with_pipeline(
340                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
341                            )?
342                        } else if let Some(level) = self.deflate_level {
343                            writer.create_chunked_dataset_compressed(
344                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, level,
345                            )?
346                        } else {
347                            writer.create_chunked_dataset(
348                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
349                            )?
350                        };
351                        if let Some(ref gp) = group_path {
352                            if gp != "/" {
353                                writer.assign_dataset_to_group(gp, idx)?;
354                            }
355                        }
356                        if let Some(ref fv) = fill_value {
357                            writer.set_dataset_fill_value(idx, fv.clone())?;
358                        }
359                        idx
360                    }
361                    H5FileInner::Reader(_) => {
362                        return Err(Hdf5Error::InvalidState(
363                            "cannot create a dataset in read mode".into(),
364                        ));
365                    }
366                    H5FileInner::Closed => {
367                        return Err(Hdf5Error::InvalidState("file is closed".into()));
368                    }
369                }
370            };
371
372            Ok(H5Dataset {
373                file_inner: clone_inner(&self.file_inner),
374                info: DatasetInfo::Writer {
375                    index,
376                    shape,
377                    element_size,
378                    chunked: true,
379                    btree2: is_btree2,
380                    fixed_array: is_fixed_array,
381                },
382            })
383        } else {
384            // Contiguous dataset (original path)
385            let index = {
386                let inner = borrow_inner(&self.file_inner);
387                match &*inner {
388                    H5FileInner::Writer(writer) => {
389                        let idx = writer.create_dataset(&full_name, datatype, &dims_u64)?;
390                        if let Some(ref gp) = group_path {
391                            if gp != "/" {
392                                writer.assign_dataset_to_group(gp, idx)?;
393                            }
394                        }
395                        if let Some(ref fv) = fill_value {
396                            writer.set_dataset_fill_value(idx, fv.clone())?;
397                        }
398                        idx
399                    }
400                    H5FileInner::Reader(_) => {
401                        return Err(Hdf5Error::InvalidState(
402                            "cannot create a dataset in read mode".into(),
403                        ));
404                    }
405                    H5FileInner::Closed => {
406                        return Err(Hdf5Error::InvalidState("file is closed".into()));
407                    }
408                }
409            };
410
411            Ok(H5Dataset {
412                file_inner: clone_inner(&self.file_inner),
413                info: DatasetInfo::Writer {
414                    index,
415                    shape,
416                    element_size,
417                    chunked: false,
418                    btree2: false,
419                    fixed_array: false,
420                },
421            })
422        }
423    }
424}
425
426// ---------------------------------------------------------------------------
427// DatasetInfo
428// ---------------------------------------------------------------------------
429
430/// Internal metadata about a dataset handle.
431enum DatasetInfo {
432    /// A dataset created via `new_dataset().create()` in write mode.
433    Writer {
434        /// Index into the writer's dataset list.
435        index: usize,
436        /// Shape (current dimensions).
437        shape: Vec<usize>,
438        /// Size of one element in bytes.
439        element_size: usize,
440        /// Whether this is a chunked dataset.
441        chunked: bool,
442        /// Whether the chunk index is a v2 B-tree (multiple unlimited dims).
443        btree2: bool,
444        /// Whether the chunk index is a Fixed Array (no unlimited dims).
445        fixed_array: bool,
446    },
447    /// A dataset opened by name in read mode.
448    Reader {
449        /// The link name of the dataset.
450        name: String,
451        /// Shape (current dimensions).
452        shape: Vec<usize>,
453        /// Size of one element in bytes.
454        element_size: usize,
455    },
456}
457
458// ---------------------------------------------------------------------------
459// H5Dataset
460// ---------------------------------------------------------------------------
461
462/// A handle to an HDF5 dataset, supporting typed read and write operations.
463///
464/// The dataset holds a shared reference to the file's I/O backend, so it
465/// remains valid even if the originating [`H5File`](crate::file::H5File) is
466/// moved or dropped (they share ownership via `Rc`).
467pub struct H5Dataset {
468    file_inner: SharedInner,
469    info: DatasetInfo,
470}
471
472/// One chunk's bytes on the way to the file, and who filtered them.
473///
474/// This is what separates a normal chunk write from a direct one; everything
475/// else about placing a chunk is identical, so the two share a single dispatch.
476#[derive(Clone, Copy)]
477enum ChunkBytes<'a> {
478    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
479    /// are stored.
480    Unfiltered(&'a [u8]),
481    /// Bytes already in their stored form, with `filter_mask` naming the
482    /// filters that were skipped.
483    Prefiltered { data: &'a [u8], filter_mask: u32 },
484}
485
486/// Strip a fixed-string element's padding, leaving the bytes that carry the
487/// value.
488///
489/// The three padding rules are the HDF5 datatype message's: null-terminated
490/// stops at the first NUL and says nothing about the bytes after it,
491/// null-padded and space-padded fill the tail with that byte. `index` names
492/// the element in the error a reserved padding rule produces.
493fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
494    let end = match padding {
495        // Null-terminated.
496        0 => elem.iter().position(|&b| b == 0).unwrap_or(elem.len()),
497        // Null-padded / space-padded: the tail of that byte is padding.
498        1 => elem.iter().rposition(|&b| b != 0).map_or(0, |i| i + 1),
499        2 => elem.iter().rposition(|&b| b != b' ').map_or(0, |i| i + 1),
500        other => {
501            return Err(Hdf5Error::InvalidState(format!(
502                "string {index} uses padding rule {other}, which the format reserves"
503            )))
504        }
505    };
506    Ok(&elem[..end])
507}
508
509/// Decode one string element's bytes under the datatype's character set.
510///
511/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
512/// `index` names the element in the error otherwise.
513fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
514    if lossy {
515        return Ok(String::from_utf8_lossy(bytes).into_owned());
516    }
517    match charset {
518        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
519        0 => match bytes.iter().position(|&b| b >= 0x80) {
520            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
521            Some(at) => Err(Hdf5Error::InvalidState(format!(
522                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
523                bytes[at]
524            ))),
525        },
526        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
527            Hdf5Error::InvalidState(format!(
528                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
529            ))
530        }),
531        other => Err(Hdf5Error::InvalidState(format!(
532            "string {index} uses character set {other}, which the format reserves"
533        ))),
534    }
535}
536
537impl H5Dataset {
538    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
539    pub(crate) fn new_reader(
540        file_inner: SharedInner,
541        name: String,
542        shape: Vec<usize>,
543        element_size: usize,
544    ) -> Self {
545        Self {
546            file_inner,
547            info: DatasetInfo::Reader {
548                name,
549                shape,
550                element_size,
551            },
552        }
553    }
554
555    /// Create a writer-mode dataset handle for an already-created dataset
556    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
557    ///
558    /// Reconstructs the same handle `new_dataset().create()` returns, so the
559    /// reopened dataset supports attribute writes and chunk appends.
560    pub(crate) fn new_writer(
561        file_inner: SharedInner,
562        index: usize,
563        shape: Vec<usize>,
564        element_size: usize,
565        chunked: bool,
566        btree2: bool,
567        fixed_array: bool,
568    ) -> Self {
569        Self {
570            file_inner,
571            info: DatasetInfo::Writer {
572                index,
573                shape,
574                element_size,
575                chunked,
576                btree2,
577                fixed_array,
578            },
579        }
580    }
581
582    /// Return the dataset dimensions.
583    pub fn shape(&self) -> Vec<usize> {
584        match &self.info {
585            DatasetInfo::Writer { shape, .. } => shape.clone(),
586            DatasetInfo::Reader { shape, .. } => shape.clone(),
587        }
588    }
589
590    /// Return the number of dimensions (rank) of the dataset.
591    pub fn ndims(&self) -> usize {
592        match &self.info {
593            DatasetInfo::Writer { shape, .. } => shape.len(),
594            DatasetInfo::Reader { shape, .. } => shape.len(),
595        }
596    }
597
598    /// Return the total number of elements in the dataset.
599    pub fn total_elements(&self) -> usize {
600        match &self.info {
601            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
602            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
603        }
604    }
605
606    /// Return the size of one element in bytes.
607    pub fn element_size(&self) -> usize {
608        match &self.info {
609            DatasetInfo::Writer { element_size, .. } => *element_size,
610            DatasetInfo::Reader { element_size, .. } => *element_size,
611        }
612    }
613
614    /// Return the element datatype as parsed from the file (read mode only).
615    ///
616    /// Unlike [`element_size`](Self::element_size), which reports only the
617    /// byte width, this exposes the full datatype: its class (integer vs
618    /// floating-point vs string vs compound …), signedness, byte order and
619    /// bit precision. Callers that must reconstruct the exact stored type —
620    /// for example to map it to a NumPy / Arrow dtype — should use this
621    /// instead of inferring a type from the byte width, which cannot
622    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
623    /// bytes).
624    ///
625    /// # Errors
626    ///
627    /// Returns an error if the file is in write mode, or if the dataset can
628    /// no longer be found in the reader's metadata.
629    ///
630    /// ```no_run
631    /// # use rust_hdf5::{H5File, DatatypeMessage};
632    /// let file = H5File::open("data.h5").unwrap();
633    /// let ds = file.dataset("image").unwrap();
634    /// match ds.datatype().unwrap() {
635    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
636    ///         println!("integer: {} bytes, signed={}", size, signed);
637    ///     }
638    ///     DatatypeMessage::FloatingPoint { size, .. } => {
639    ///         println!("float: {} bytes", size);
640    ///     }
641    ///     other => println!("other type: {other}"),
642    /// }
643    /// ```
644    pub fn datatype(&self) -> Result<DatatypeMessage> {
645        match &self.info {
646            DatasetInfo::Reader { name, .. } => {
647                let inner = borrow_inner(&self.file_inner);
648                match &*inner {
649                    H5FileInner::Reader(reader) => reader
650                        .dataset_info(name)
651                        .map(|info| info.datatype.clone())
652                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
653                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
654                }
655            }
656            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
657                "datatype() is only available in read mode".into(),
658            )),
659        }
660    }
661
662    /// Return the chunk dimensions, if this is a chunked dataset.
663    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
664        match &self.info {
665            DatasetInfo::Reader { name, .. } => {
666                let inner = borrow_inner(&self.file_inner);
667                if let H5FileInner::Reader(reader) = &*inner {
668                    if let Some(info) = reader.dataset_info(name) {
669                        use crate::format::messages::data_layout::DataLayoutMessage;
670                        let chunk_dims = match &info.layout {
671                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
672                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
673                            _ => None,
674                        };
675                        if let Some(chunk_dims) = chunk_dims {
676                            // Strip trailing element-size dimension
677                            return Some(
678                                chunk_dims[..chunk_dims.len() - 1]
679                                    .iter()
680                                    .map(|&d| d as usize)
681                                    .collect(),
682                            );
683                        }
684                    }
685                }
686                None
687            }
688            DatasetInfo::Writer { .. } => None,
689        }
690    }
691
692    /// Return whether this is a chunked dataset.
693    pub fn is_chunked(&self) -> bool {
694        match &self.info {
695            DatasetInfo::Writer { chunked, .. } => *chunked,
696            DatasetInfo::Reader { name, .. } => {
697                let inner = borrow_inner(&self.file_inner);
698                match &*inner {
699                    H5FileInner::Reader(reader) => {
700                        if let Some(info) = reader.dataset_info(name) {
701                            use crate::format::messages::data_layout::DataLayoutMessage;
702                            matches!(
703                                info.layout,
704                                DataLayoutMessage::ChunkedV4 { .. }
705                                    | DataLayoutMessage::ChunkedV3 { .. }
706                            )
707                        } else {
708                            false
709                        }
710                    }
711                    _ => false,
712                }
713            }
714        }
715    }
716
717    /// Return the names of all attributes on this dataset (read mode only).
718    pub fn attr_names(&self) -> Result<Vec<String>> {
719        match &self.info {
720            DatasetInfo::Reader { name, .. } => {
721                let inner = borrow_inner(&self.file_inner);
722                match &*inner {
723                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
724                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
725                }
726            }
727            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
728                "attr_names not available in write mode".into(),
729            )),
730        }
731    }
732
733    /// Open an attribute by name (read mode only).
734    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
735        match &self.info {
736            DatasetInfo::Reader { name, .. } => {
737                let inner = borrow_inner(&self.file_inner);
738                match &*inner {
739                    H5FileInner::Reader(reader) => {
740                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
741                        Ok(crate::attribute::H5Attribute::new_reader(
742                            clone_inner(&self.file_inner),
743                            attr_msg,
744                        ))
745                    }
746                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
747                }
748            }
749            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
750                "attr() not available in write mode".into(),
751            )),
752        }
753    }
754
755    /// Start building a new attribute on this dataset.
756    ///
757    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
758    /// and `.create("name")` to finalize.
759    ///
760    /// # Example
761    ///
762    /// ```no_run
763    /// # use rust_hdf5::H5File;
764    /// # use rust_hdf5::types::VarLenUnicode;
765    /// let file = H5File::create("attr.h5").unwrap();
766    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
767    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
768    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
769    /// ```
770    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
771        let ds_index = match &self.info {
772            DatasetInfo::Writer { index, .. } => *index,
773            DatasetInfo::Reader { .. } => {
774                // Reader mode: we'll return a builder that will error on create.
775                // Using usize::MAX as sentinel.
776                usize::MAX
777            }
778        };
779        AttrBuilder::new(&self.file_inner, ds_index)
780    }
781
782    /// Write a typed slice holding the dataset's whole image.
783    ///
784    /// The slice length must match the total number of elements declared by
785    /// the dataset shape. The data is reinterpreted as raw bytes and written
786    /// to the file: to the contiguous data block, or — for a chunked dataset —
787    /// scattered across its chunk grid, through the filter pipeline if one is
788    /// set. To write only part of a dataset, use
789    /// [`write_slice`](Self::write_slice).
790    ///
791    /// # Errors
792    ///
793    /// Returns an error if:
794    /// - The file is in read mode.
795    /// - The data length does not match the declared shape.
796    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
797        match &self.info {
798            DatasetInfo::Writer {
799                index,
800                shape,
801                element_size,
802                chunked,
803                btree2,
804                fixed_array,
805            } => {
806                let total_elements: usize = shape.iter().product();
807                if data.len() != total_elements {
808                    return Err(Hdf5Error::InvalidState(format!(
809                        "data length {} does not match dataset size {}",
810                        data.len(),
811                        total_elements,
812                    )));
813                }
814
815                // Verify element size matches
816                if T::element_size() != *element_size {
817                    return Err(Hdf5Error::TypeMismatch(format!(
818                        "write type has element size {} but dataset expects {}",
819                        T::element_size(),
820                        element_size,
821                    )));
822                }
823
824                // Safety: T: Copy + 'static (numeric primitive) with well-defined
825                // byte representation. The resulting slice borrows `data` and
826                // lives only as long as this block.
827                let byte_len = data.len() * T::element_size();
828                let raw =
829                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
830
831                if *chunked {
832                    // A chunked dataset has no contiguous data block; scatter
833                    // the full row-major image into its chunk grid and write
834                    // each chunk through the dataset's filter pipeline.
835                    return self.write_full_image_chunked(
836                        *index,
837                        *btree2,
838                        *fixed_array,
839                        raw,
840                        *element_size,
841                    );
842                }
843
844                let inner = borrow_inner(&self.file_inner);
845                match &*inner {
846                    H5FileInner::Writer(writer) => {
847                        writer.write_dataset_raw(*index, raw)?;
848                        Ok(())
849                    }
850                    _ => Err(Hdf5Error::InvalidState(
851                        "file is no longer in write mode".into(),
852                    )),
853                }
854            }
855            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
856                "cannot write to a dataset opened in read mode".into(),
857            )),
858        }
859    }
860
861    /// Write the raw byte image of the whole dataset directly.
862    ///
863    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
864    /// data block, or a chunk grid the image is scattered across.
865    ///
866    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
867    /// `H5Type` carrier, so it works for element types that have no matching
868    /// Rust primitive — in particular a runtime
869    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
870    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
871    /// `product(shape) * element_size`, where `element_size` is taken from the
872    /// dataset's on-disk datatype.
873    ///
874    /// ```no_run
875    /// # use rust_hdf5::H5File;
876    /// # use rust_hdf5::types::{CompoundType, H5Type};
877    /// let file = H5File::create("c.h5").unwrap();
878    /// let ct = CompoundType {
879    ///     members: vec![
880    ///         ("id".to_string(), i32::hdf5_type(), 0),
881    ///         ("val".to_string(), f64::hdf5_type(), 4),
882    ///     ],
883    ///     total_size: 12,
884    /// };
885    /// let ds = file
886    ///     .new_dataset::<u8>()
887    ///     .datatype(ct.to_datatype())
888    ///     .shape(&[2])
889    ///     .create("records")
890    ///     .unwrap();
891    /// let mut bytes = Vec::new();
892    /// bytes.extend_from_slice(&1i32.to_le_bytes());
893    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
894    /// bytes.extend_from_slice(&2i32.to_le_bytes());
895    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
896    /// ds.write_raw_bytes(&bytes).unwrap();
897    /// ```
898    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
899        match &self.info {
900            DatasetInfo::Writer {
901                index,
902                shape,
903                element_size,
904                chunked,
905                btree2,
906                fixed_array,
907            } => {
908                let expected: usize = shape.iter().product::<usize>() * *element_size;
909                if bytes.len() != expected {
910                    return Err(Hdf5Error::InvalidState(format!(
911                        "raw byte length {} does not match dataset size {} \
912                         (product(shape) * element_size {})",
913                        bytes.len(),
914                        expected,
915                        element_size,
916                    )));
917                }
918                if *chunked {
919                    // Scatter the full row-major image into the chunk grid
920                    // (same path as write_raw, carrier-agnostic bytes).
921                    return self.write_full_image_chunked(
922                        *index,
923                        *btree2,
924                        *fixed_array,
925                        bytes,
926                        *element_size,
927                    );
928                }
929                let inner = borrow_inner(&self.file_inner);
930                match &*inner {
931                    H5FileInner::Writer(writer) => {
932                        writer.write_dataset_raw(*index, bytes)?;
933                        Ok(())
934                    }
935                    _ => Err(Hdf5Error::InvalidState(
936                        "file is no longer in write mode".into(),
937                    )),
938                }
939            }
940            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
941                "cannot write to a dataset opened in read mode".into(),
942            )),
943        }
944    }
945
946    /// Scatter a full row-major dataset image into its chunk grid, writing
947    /// every chunk through the dataset's filter pipeline.
948    ///
949    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
950    /// — it is how [`write_raw`](Self::write_raw) and
951    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
952    /// (including the single auto-chunk created when a filter is set without
953    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
954    /// chunk footprint, exactly as libhdf5 stores them.
955    fn write_full_image_chunked(
956        &self,
957        index: usize,
958        btree2: bool,
959        fixed_array: bool,
960        bytes: &[u8],
961        element_size: usize,
962    ) -> Result<()> {
963        let inner = borrow_inner(&self.file_inner);
964        let writer = match &*inner {
965            H5FileInner::Writer(w) => w,
966            _ => {
967                return Err(Hdf5Error::InvalidState(
968                    "file is no longer in write mode".into(),
969                ))
970            }
971        };
972        // Whole-operation guard: the flush, the grid snapshot and the chunk
973        // writes below must not interleave with a concurrent same-dataset
974        // operation.
975        let cell = writer.ds(index);
976        let _op = cell.op.lock();
977        // A buffered append tail would flush over the image at close; hand
978        // it to the chunks first, the image below overwrites everything.
979        writer.flush_append_buffer(index)?;
980        let chunk_dims = writer
981            .dataset_chunk_dims(index)
982            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
983            .to_vec();
984        let dims = writer.dataset_dims(index).to_vec();
985        let rank = dims.len();
986
987        // Chunk grid: number of chunks along each dimension (row-major).
988        let mut grid = vec![0u64; rank];
989        for d in 0..rank {
990            grid[d] = if chunk_dims[d] > 0 {
991                dims[d].div_ceil(chunk_dims[d])
992            } else {
993                0
994            };
995        }
996        let total_chunks: u64 = grid.iter().product();
997
998        // Decode the iteration counter into row-major coordinates over the
999        // *current* image's chunk grid. This is only an odometer over the
1000        // chunks the image spans — the slot a chunk is recorded under comes
1001        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
1002        // extent decides.
1003        let coords_of = |linear: u64| -> Vec<u64> {
1004            let mut rem = linear;
1005            let mut coords = vec![0u64; rank];
1006            for d in (0..rank).rev() {
1007                coords[d] = rem % grid[d];
1008                rem /= grid[d];
1009            }
1010            coords
1011        };
1012
1013        if btree2 {
1014            // B-tree v2 stores chunks verbatim (unfiltered only in this
1015            // codebase), so there is no compression to parallelize; write one
1016            // chunk at a time.
1017            for linear in 0..total_chunks {
1018                let coords = coords_of(linear);
1019                let chunk_buf =
1020                    Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
1021                writer.write_chunk_btree_v2_inner(index, &coords, &chunk_buf)?;
1022            }
1023        } else {
1024            // Extensible array and fixed array both compress each chunk through
1025            // the filter pipeline. Gather chunks and write them through the
1026            // per-index batch path so the pipeline compresses them in parallel
1027            // (with the `parallel` feature). A fixed-size window bounds peak
1028            // memory instead of materializing every chunk at once; 256 keeps
1029            // every rayon worker fed while capping the transient buffers to
1030            // window * chunk bytes. The two indexes differ only in how a chunk
1031            // is addressed: EA by its linear grid index, FA by grid coordinates.
1032            const BATCH_WINDOW: u64 = 256;
1033            let mut start = 0u64;
1034            while start < total_chunks {
1035                let end = (start + BATCH_WINDOW).min(total_chunks);
1036                let items: Vec<(Vec<u64>, Vec<u8>)> = (start..end)
1037                    .map(|counter| {
1038                        let coords = coords_of(counter);
1039                        let buf =
1040                            Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
1041                        (coords, buf)
1042                    })
1043                    .collect();
1044                if fixed_array {
1045                    let pairs: Vec<(&[u64], &[u8])> = items
1046                        .iter()
1047                        .map(|(c, d)| (c.as_slice(), d.as_slice()))
1048                        .collect();
1049                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
1050                } else {
1051                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
1052                    for (c, d) in &items {
1053                        pairs.push((writer.chunk_slot(index, c)?, d.as_slice()));
1054                    }
1055                    writer.write_chunks_batch_inner(index, &pairs)?;
1056                }
1057                start = end;
1058            }
1059        }
1060        Ok(())
1061    }
1062
1063    /// Gather one chunk's bytes from a row-major full-dataset image.
1064    ///
1065    /// `coords` are the chunk's grid coordinates. The returned buffer is
1066    /// exactly `product(chunk_dims) * element_size` bytes, zero-padded where
1067    /// the chunk extends past the dataset edge.
1068    fn gather_chunk(
1069        source: &[u8],
1070        dims: &[u64],
1071        chunk_dims: &[u64],
1072        coords: &[u64],
1073        element_size: usize,
1074    ) -> Vec<u8> {
1075        let rank = dims.len();
1076        let chunk_elems: u64 = chunk_dims.iter().product();
1077        let mut out = vec![0u8; chunk_elems as usize * element_size];
1078        if rank == 0 {
1079            // Scalar dataset: a single element, no chunking dimension.
1080            if source.len() >= element_size {
1081                out[..element_size].copy_from_slice(&source[..element_size]);
1082            }
1083            return out;
1084        }
1085
1086        // Actual extent of this chunk along each dimension (edge chunks are
1087        // smaller than the nominal chunk shape).
1088        let mut extent = vec![0u64; rank];
1089        for d in 0..rank {
1090            let start = coords[d] * chunk_dims[d];
1091            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
1092            extent[d] = end.saturating_sub(start);
1093        }
1094        if extent.contains(&0) {
1095            return out; // nothing of the dataset falls in this chunk
1096        }
1097
1098        // Row-major strides (in elements) for the source (over `dims`) and the
1099        // destination chunk buffer (over `chunk_dims`).
1100        let mut src_stride = vec![1u64; rank];
1101        let mut dst_stride = vec![1u64; rank];
1102        for d in (0..rank - 1).rev() {
1103            src_stride[d] = src_stride[d + 1] * dims[d + 1];
1104            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
1105        }
1106
1107        // Copy one contiguous run along the last axis per outer multi-index.
1108        let last = rank - 1;
1109        let run = extent[last] as usize * element_size;
1110        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
1111        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
1112        for _ in 0..outer {
1113            let mut src_off = 0u64;
1114            let mut dst_off = 0u64;
1115            for d in 0..rank {
1116                let global = coords[d] * chunk_dims[d] + idx[d];
1117                src_off += global * src_stride[d];
1118                dst_off += idx[d] * dst_stride[d];
1119            }
1120            let s = src_off as usize * element_size;
1121            let dpos = dst_off as usize * element_size;
1122            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
1123
1124            // Advance the multi-index over axes [0..last); the last axis is the
1125            // contiguous run handled above.
1126            let mut d = last;
1127            while d > 0 {
1128                d -= 1;
1129                idx[d] += 1;
1130                if idx[d] < extent[d] {
1131                    break;
1132                }
1133                idx[d] = 0;
1134            }
1135        }
1136        out
1137    }
1138
1139    /// Write a single chunk to a chunked dataset.
1140    ///
1141    /// `chunk_idx` is the linear chunk index (typically the frame number for
1142    /// streaming datasets). `data` is the raw byte data for one chunk.
1143    ///
1144    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
1145    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
1146    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
1147        match &self.info {
1148            DatasetInfo::Writer {
1149                index,
1150                chunked,
1151                btree2,
1152                fixed_array,
1153                ..
1154            } => {
1155                if !*chunked {
1156                    return Err(Hdf5Error::InvalidState(
1157                        "write_chunk is only for chunked datasets".into(),
1158                    ));
1159                }
1160                if *btree2 {
1161                    return Err(Hdf5Error::InvalidState(
1162                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
1163                         with the chunk's grid coordinates"
1164                            .into(),
1165                    ));
1166                }
1167
1168                let inner = borrow_inner(&self.file_inner);
1169                match &*inner {
1170                    H5FileInner::Writer(writer) => {
1171                        // One op: the slot decode and the write see the same
1172                        // extents.
1173                        let cell = writer.ds(*index);
1174                        let _op = cell.op.lock();
1175                        if *fixed_array {
1176                            // Fixed-array dataset: decode the index-grid slot
1177                            // into row-major grid coordinates.
1178                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
1179                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?;
1180                        } else {
1181                            writer.write_chunk_inner(*index, chunk_idx as u64, data)?;
1182                        }
1183                        Ok(())
1184                    }
1185                    _ => Err(Hdf5Error::InvalidState(
1186                        "file is no longer in write mode".into(),
1187                    )),
1188                }
1189            }
1190            DatasetInfo::Reader { .. } => {
1191                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1192            }
1193        }
1194    }
1195
1196    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
1197    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
1198    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
1199    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
1200    ///
1201    /// `chunk_idx` is the linear chunk index (the frame number for streaming
1202    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
1203    /// the already-filtered bytes of one chunk — its length is the *stored*
1204    /// (compressed) size, not the uncompressed chunk size.
1205    ///
1206    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
1207    /// dataset's pipeline was **not** applied to this chunk and must be skipped
1208    /// on read. Pass 0 when the full pipeline was already applied upstream (the
1209    /// common case: a codec plugin handed you compressed frames).
1210    ///
1211    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
1212    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
1213    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
1214    /// against, so address its chunks with
1215    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
1216    ///
1217    /// # Reading back
1218    ///
1219    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
1220    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
1221    /// the reader skipping exactly the filters the mask marks as not applied.
1222    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
1223        match &self.info {
1224            DatasetInfo::Writer {
1225                index,
1226                chunked,
1227                btree2,
1228                fixed_array,
1229                ..
1230            } => {
1231                if !*chunked {
1232                    return Err(Hdf5Error::InvalidState(
1233                        "write_chunk_raw is only for chunked datasets".into(),
1234                    ));
1235                }
1236                if *btree2 {
1237                    return Err(Hdf5Error::InvalidState(
1238                        "this dataset uses a v2 B-tree chunk index; use \
1239                         write_chunk_raw_at with the chunk's grid coordinates"
1240                            .into(),
1241                    ));
1242                }
1243
1244                let inner = borrow_inner(&self.file_inner);
1245                match &*inner {
1246                    H5FileInner::Writer(writer) => {
1247                        // One op: the slot decode and the write see the same
1248                        // extents.
1249                        let cell = writer.ds(*index);
1250                        let _op = cell.op.lock();
1251                        if *fixed_array {
1252                            // Fixed-array dataset: decode the index-grid slot
1253                            // into row-major grid coordinates.
1254                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
1255                            writer.write_compressed_chunk_fixed_array_inner(
1256                                *index,
1257                                &coords,
1258                                data,
1259                                filter_mask,
1260                            )?;
1261                        } else {
1262                            writer.write_compressed_chunk_inner(
1263                                *index,
1264                                chunk_idx as u64,
1265                                data,
1266                                filter_mask,
1267                            )?;
1268                        }
1269                        Ok(())
1270                    }
1271                    _ => Err(Hdf5Error::InvalidState(
1272                        "file is no longer in write mode".into(),
1273                    )),
1274                }
1275            }
1276            DatasetInfo::Reader { .. } => {
1277                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1278            }
1279        }
1280    }
1281
1282    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
1283    /// chunk-grid coordinates (one per dimension).
1284    ///
1285    /// This is the entry point for datasets with two or more unlimited
1286    /// dimensions. The dataset's logical dimensions are extended to cover
1287    /// the written chunk. `data` is the raw bytes of one full chunk.
1288    ///
1289    /// ```no_run
1290    /// # use rust_hdf5::H5File;
1291    /// let file = H5File::create("bt2.h5").unwrap();
1292    /// let ds = file.new_dataset::<i32>()
1293    ///     .shape(&[0, 0])
1294    ///     .chunk(&[2, 2])
1295    ///     .max_shape(&[None, None])
1296    ///     .create("grid")
1297    ///     .unwrap();
1298    /// let chunk = [0i32, 1, 2, 3];
1299    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
1300    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
1301    /// ```
1302    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
1303        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
1304    }
1305
1306    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
1307    /// addressed by its chunk-grid coordinates.
1308    ///
1309    /// The coordinate-addressed twin of
1310    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
1311    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
1312    /// there is no fixed chunk grid for a linear index to mean anything against.
1313    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
1314    /// to cover the written chunk.
1315    ///
1316    /// `data` is the already-filtered bytes of one chunk — its length is the
1317    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
1318    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
1319    /// full pipeline already ran upstream.
1320    ///
1321    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
1322    /// has no slot to record a stored size or mask.
1323    pub fn write_chunk_raw_at(
1324        &self,
1325        chunk_coords: &[usize],
1326        data: &[u8],
1327        filter_mask: u32,
1328    ) -> Result<()> {
1329        self.write_chunk_at_inner(
1330            chunk_coords,
1331            ChunkBytes::Prefiltered { data, filter_mask },
1332            "write_chunk_raw_at",
1333        )
1334    }
1335
1336    /// The single owner of coordinate-addressed chunk writes: validates the
1337    /// coordinates, grows the dataspace to cover them, and routes the bytes to
1338    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
1339    /// here or already ran upstream is carried by `bytes`, not by a second copy
1340    /// of this dispatch.
1341    fn write_chunk_at_inner(
1342        &self,
1343        chunk_coords: &[usize],
1344        bytes: ChunkBytes<'_>,
1345        what: &str,
1346    ) -> Result<()> {
1347        match &self.info {
1348            DatasetInfo::Writer {
1349                index,
1350                chunked,
1351                btree2,
1352                fixed_array,
1353                ..
1354            } => {
1355                if !*chunked {
1356                    return Err(Hdf5Error::InvalidState(format!(
1357                        "{what} is only for chunked datasets"
1358                    )));
1359                }
1360                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
1361                let btree2 = *btree2;
1362                let fixed_array = *fixed_array;
1363                let inner = borrow_inner(&self.file_inner);
1364                let writer = match &*inner {
1365                    H5FileInner::Writer(w) => w,
1366                    _ => {
1367                        return Err(Hdf5Error::InvalidState(
1368                            "file is no longer in write mode".into(),
1369                        ))
1370                    }
1371                };
1372                // Whole-operation guard: the dims snapshot, the chunk write
1373                // and the extend below must not interleave with a concurrent
1374                // same-dataset operation.
1375                let cell = writer.ds(*index);
1376                let _op = cell.op.lock();
1377                let chunk_dims = writer
1378                    .dataset_chunk_dims(*index)
1379                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1380                    .to_vec();
1381                let dims = writer.dataset_dims(*index).to_vec();
1382                if coords.len() != dims.len() {
1383                    return Err(Hdf5Error::InvalidState(format!(
1384                        "chunk_coords has {} entries but the dataset has {} dimensions",
1385                        coords.len(),
1386                        dims.len()
1387                    )));
1388                }
1389                if chunk_dims.len() != dims.len() {
1390                    return Err(Hdf5Error::InvalidState(format!(
1391                        "dataset chunk shape has {} dimensions but the dataspace has {}",
1392                        chunk_dims.len(),
1393                        dims.len()
1394                    )));
1395                }
1396
1397                // Validate coordinates and compute the grown dimensions
1398                // up-front, before any chunk is written, so an overflowing
1399                // coordinate cannot leave an orphaned chunk in the file.
1400                let mut new_dims = dims.clone();
1401                for d in 0..dims.len() {
1402                    let needed = coords[d]
1403                        .checked_add(1)
1404                        .and_then(|c| c.checked_mul(chunk_dims[d]))
1405                        .ok_or_else(|| {
1406                            Hdf5Error::InvalidState(format!(
1407                                "chunk coordinate {} in dimension {} is too large",
1408                                coords[d], d
1409                            ))
1410                        })?;
1411                    if needed > new_dims[d] {
1412                        new_dims[d] = needed;
1413                    }
1414                }
1415
1416                if fixed_array {
1417                    // Fixed-array (fixed-shape) dataset: no dimension growth.
1418                    match bytes {
1419                        ChunkBytes::Unfiltered(data) => {
1420                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
1421                        }
1422                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1423                            .write_compressed_chunk_fixed_array_inner(
1424                                *index,
1425                                &coords,
1426                                data,
1427                                filter_mask,
1428                            )?,
1429                    }
1430                    return Ok(());
1431                }
1432
1433                if btree2 {
1434                    match bytes {
1435                        ChunkBytes::Unfiltered(data) => {
1436                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
1437                        }
1438                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1439                            .write_compressed_chunk_btree_v2_inner(
1440                                *index,
1441                                &coords,
1442                                data,
1443                                filter_mask,
1444                            )?,
1445                    }
1446                } else {
1447                    // Extensible array: the chunk's index-grid slot (row-major
1448                    // against the maximum extent).
1449                    let linear = writer.chunk_slot(*index, &coords)?;
1450                    match bytes {
1451                        ChunkBytes::Unfiltered(data) => {
1452                            writer.write_chunk_inner(*index, linear, data)?
1453                        }
1454                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1455                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
1456                    }
1457                }
1458
1459                if new_dims != dims {
1460                    writer.extend_dataset_inner(*index, &new_dims)?;
1461                }
1462                Ok(())
1463            }
1464            DatasetInfo::Reader { .. } => {
1465                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1466            }
1467        }
1468    }
1469
1470    /// Write multiple chunks in a batch, optionally compressing in parallel.
1471    ///
1472    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
1473    /// pipeline is configured and the `parallel` feature is enabled, all
1474    /// chunks are compressed concurrently via rayon.
1475    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
1476        match &self.info {
1477            DatasetInfo::Writer { index, chunked, .. } => {
1478                if !*chunked {
1479                    return Err(Hdf5Error::InvalidState(
1480                        "write_chunks_batch is only for chunked datasets".into(),
1481                    ));
1482                }
1483                let pairs: Vec<(u64, &[u8])> = chunks
1484                    .iter()
1485                    .map(|(idx, data)| (*idx as u64, *data))
1486                    .collect();
1487                let inner = borrow_inner(&self.file_inner);
1488                match &*inner {
1489                    H5FileInner::Writer(writer) => {
1490                        writer.write_chunks_batch(*index, &pairs)?;
1491                        Ok(())
1492                    }
1493                    _ => Err(Hdf5Error::InvalidState(
1494                        "file is no longer in write mode".into(),
1495                    )),
1496                }
1497            }
1498            DatasetInfo::Reader { .. } => {
1499                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1500            }
1501        }
1502    }
1503
1504    /// Append data along the first dimension of a chunked dataset.
1505    ///
1506    /// `data` must contain a whole number of "frames" — slices along
1507    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
1508    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
1509    /// multiple of `H * W`.
1510    ///
1511    /// This method writes the necessary chunks and extends the dataset
1512    /// shape automatically.
1513    ///
1514    /// ```no_run
1515    /// # use rust_hdf5::H5File;
1516    /// let file = H5File::create("append.h5").unwrap();
1517    /// let ds = file.new_dataset::<f64>()
1518    ///     .shape(&[0, 3])
1519    ///     .chunk(&[1, 3])
1520    ///     .max_shape(&[None, Some(3)])
1521    ///     .create("data")
1522    ///     .unwrap();
1523    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
1524    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
1525    /// ```
1526    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
1527        match &self.info {
1528            DatasetInfo::Writer {
1529                index,
1530                element_size,
1531                chunked,
1532                ..
1533            } => {
1534                if !*chunked {
1535                    return Err(Hdf5Error::InvalidState(
1536                        "append is only for chunked datasets".into(),
1537                    ));
1538                }
1539                if T::element_size() != *element_size {
1540                    return Err(Hdf5Error::TypeMismatch(format!(
1541                        "append type has element size {} but dataset expects {}",
1542                        T::element_size(),
1543                        element_size,
1544                    )));
1545                }
1546
1547                let ds_index = *index;
1548                let es = *element_size;
1549
1550                let inner = borrow_inner(&self.file_inner);
1551                let writer = match &*inner {
1552                    H5FileInner::Writer(w) => w,
1553                    _ => {
1554                        return Err(Hdf5Error::InvalidState(
1555                            "file is no longer in write mode".into(),
1556                        ))
1557                    }
1558                };
1559
1560                // Whole-operation guard: the buffer take, the frame writes,
1561                // the re-buffer and the extend below are separate slot
1562                // acquisitions that a concurrent same-dataset append must not
1563                // interleave with.
1564                let cell = writer.ds(ds_index);
1565                let _op = cell.op.lock();
1566
1567                let chunk_dims = writer
1568                    .dataset_chunk_dims(ds_index)
1569                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1570                    .to_vec();
1571                let dims = writer.dataset_dims(ds_index).to_vec();
1572
1573                // Frame size = product of dims[1..]
1574                let frame_elems: usize = if dims.len() > 1 {
1575                    dims[1..].iter().map(|&d| d as usize).product()
1576                } else {
1577                    1
1578                };
1579
1580                if frame_elems == 0 {
1581                    return Err(Hdf5Error::InvalidState(
1582                        "cannot append to dataset with zero-size trailing dimensions".into(),
1583                    ));
1584                }
1585
1586                if !data.len().is_multiple_of(frame_elems) {
1587                    return Err(Hdf5Error::InvalidState(format!(
1588                        "data length {} is not a multiple of frame size {}",
1589                        data.len(),
1590                        frame_elems,
1591                    )));
1592                }
1593
1594                let n_new_frames = data.len() / frame_elems;
1595                let current_dim0 = dims[0] as usize;
1596
1597                // Chunk size along first dimension
1598                let chunk_dim0 = chunk_dims[0] as usize;
1599                let frame_bytes = frame_elems * es;
1600
1601                let raw = unsafe {
1602                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
1603                };
1604
1605                // Merge the buffer with the new frames when it is the
1606                // dataset's tail; a buffer left mid-extent (the extent moved
1607                // past it) keeps its recorded place — flush it and start
1608                // fresh at the current end.
1609                let taken = { writer.ds(ds_index).lock().append.take() };
1610                let (base_dim0, buffered_frames, mut combined) = match taken {
1611                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
1612                        (b.base as usize, b.frames as usize, b.bytes)
1613                    }
1614                    Some(b) => {
1615                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
1616                        (current_dim0, 0, Vec::new())
1617                    }
1618                    None => (current_dim0, 0, Vec::new()),
1619                };
1620                combined.extend_from_slice(raw);
1621
1622                let total_frames = buffered_frames + n_new_frames;
1623
1624                // Rows up to the last chunk boundary are written now; the
1625                // tail that does not complete a chunk goes back in the
1626                // buffer for the next append (or the flush at close). The
1627                // boundary can precede `base_dim0` — a reopened file's
1628                // flushed partial chunk leaves the base mid-chunk — in
1629                // which case everything is tail.
1630                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
1631                let write_frames = last_boundary.saturating_sub(base_dim0);
1632                let tail_frames = total_frames - write_frames;
1633                if write_frames > 0 {
1634                    writer.write_append_frames(
1635                        ds_index,
1636                        base_dim0 as u64,
1637                        write_frames as u64,
1638                        &combined[..write_frames * frame_bytes],
1639                    )?;
1640                }
1641                if tail_frames > 0 {
1642                    let ds = writer.ds(ds_index);
1643                    let mut m = ds.lock();
1644                    m.append = Some(crate::io::writer::AppendBuffer {
1645                        base: (base_dim0 + write_frames) as u64,
1646                        frames: tail_frames as u64,
1647                        bytes: combined[write_frames * frame_bytes..].to_vec(),
1648                    });
1649                }
1650
1651                // Extend dims to include all frames (buffered + new)
1652                let logical_dim0 = base_dim0 + total_frames;
1653                let mut new_dims: Vec<u64> = dims;
1654                new_dims[0] = logical_dim0 as u64;
1655                writer.extend_dataset_inner(ds_index, &new_dims)?;
1656
1657                Ok(())
1658            }
1659            DatasetInfo::Reader { .. } => {
1660                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
1661            }
1662        }
1663    }
1664
1665    /// Extend the dimensions of a chunked dataset.
1666    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
1667        match &self.info {
1668            DatasetInfo::Writer { index, chunked, .. } => {
1669                if !*chunked {
1670                    return Err(Hdf5Error::InvalidState(
1671                        "extend is only for chunked datasets".into(),
1672                    ));
1673                }
1674
1675                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1676                let inner = borrow_inner(&self.file_inner);
1677                match &*inner {
1678                    H5FileInner::Writer(writer) => {
1679                        writer.extend_dataset(*index, &dims_u64)?;
1680                        Ok(())
1681                    }
1682                    _ => Err(Hdf5Error::InvalidState(
1683                        "file is no longer in write mode".into(),
1684                    )),
1685                }
1686            }
1687            DatasetInfo::Reader { .. } => {
1688                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
1689            }
1690        }
1691    }
1692
1693    /// Set the logical extent of a chunked dataset, growing **or
1694    /// shrinking** any dimension.
1695    ///
1696    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
1697    /// dimension — for example to correct an over-extended frame count
1698    /// after writing a partial multi-frame chunk. Shrinking prunes the
1699    /// stored chunks the way libhdf5's `H5Dset_extent` does: a chunk
1700    /// entirely beyond the new extent is removed from the chunk index and
1701    /// its storage freed for reuse, and a chunk the new extent cuts
1702    /// through has its out-of-extent region overwritten with the fill
1703    /// value — so growing the extent back exposes fill values, not the
1704    /// old data. The new extent must not exceed the dataset's maximum
1705    /// dimensions.
1706    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
1707        match &self.info {
1708            DatasetInfo::Writer { index, .. } => {
1709                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1710                let inner = borrow_inner(&self.file_inner);
1711                match &*inner {
1712                    H5FileInner::Writer(writer) => {
1713                        writer.set_dataset_extent(*index, &dims_u64)?;
1714                        Ok(())
1715                    }
1716                    _ => Err(Hdf5Error::InvalidState(
1717                        "file is no longer in write mode".into(),
1718                    )),
1719                }
1720            }
1721            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
1722                "cannot set extent in read mode".into(),
1723            )),
1724        }
1725    }
1726
1727    /// Flush a chunked dataset's index structures to disk.
1728    pub fn flush(&self) -> Result<()> {
1729        match &self.info {
1730            DatasetInfo::Writer { index, .. } => {
1731                let inner = borrow_inner(&self.file_inner);
1732                match &*inner {
1733                    H5FileInner::Writer(writer) => {
1734                        writer.flush_dataset(*index)?;
1735                        Ok(())
1736                    }
1737                    _ => Ok(()),
1738                }
1739            }
1740            DatasetInfo::Reader { .. } => Ok(()),
1741        }
1742    }
1743
1744    /// Read a slice (hyperslab) of the dataset as a typed vector.
1745    ///
1746    /// `starts` and `counts` define the N-dimensional selection:
1747    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
1748    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
1749        match &self.info {
1750            DatasetInfo::Reader {
1751                name, element_size, ..
1752            } => {
1753                if T::element_size() != *element_size {
1754                    return Err(Hdf5Error::TypeMismatch(format!(
1755                        "read type has element size {} but dataset has element size {}",
1756                        T::element_size(),
1757                        element_size,
1758                    )));
1759                }
1760                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1761                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1762
1763                let raw = {
1764                    let mut inner = borrow_inner_mut(&self.file_inner);
1765                    match &mut *inner {
1766                        H5FileInner::Reader(reader) => {
1767                            reader.read_slice(name, &starts_u64, &counts_u64)?
1768                        }
1769                        _ => {
1770                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
1771                        }
1772                    }
1773                };
1774
1775                if raw.len() % T::element_size() != 0 {
1776                    return Err(Hdf5Error::TypeMismatch(format!(
1777                        "raw data size {} is not a multiple of element size {}",
1778                        raw.len(),
1779                        T::element_size(),
1780                    )));
1781                }
1782
1783                let count = raw.len() / T::element_size();
1784                let mut result = Vec::<T>::with_capacity(count);
1785                unsafe {
1786                    std::ptr::copy_nonoverlapping(
1787                        raw.as_ptr(),
1788                        result.as_mut_ptr() as *mut u8,
1789                        raw.len(),
1790                    );
1791                    result.set_len(count);
1792                }
1793                Ok(result)
1794            }
1795            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1796                "cannot read_slice from a dataset in write mode".into(),
1797            )),
1798        }
1799    }
1800
1801    /// Write a typed slice to a sub-region of the dataset.
1802    ///
1803    /// `starts` and `counts` define the N-dimensional selection, which must lie
1804    /// inside the dataset's current extent.
1805    ///
1806    /// Works for both contiguous and chunked datasets. For a chunked dataset
1807    /// only the chunks the selection touches are rewritten — a partially
1808    /// covered chunk is read back, patched, and written again, so updating one
1809    /// row of an appendable dataset costs the chunks that row crosses rather
1810    /// than the whole dataset. Elements of a touched chunk that the selection
1811    /// does not cover keep their stored value, or the dataset's fill value if
1812    /// the chunk did not exist yet.
1813    pub fn write_slice<T: H5Type>(
1814        &self,
1815        starts: &[usize],
1816        counts: &[usize],
1817        data: &[T],
1818    ) -> Result<()> {
1819        match &self.info {
1820            DatasetInfo::Writer {
1821                index,
1822                element_size,
1823                ..
1824            } => {
1825                if T::element_size() != *element_size {
1826                    return Err(Hdf5Error::TypeMismatch(format!(
1827                        "write type has element size {} but dataset expects {}",
1828                        T::element_size(),
1829                        element_size,
1830                    )));
1831                }
1832
1833                let expected: usize = counts.iter().product();
1834                if data.len() != expected {
1835                    return Err(Hdf5Error::InvalidState(format!(
1836                        "data length {} does not match slice size {}",
1837                        data.len(),
1838                        expected,
1839                    )));
1840                }
1841
1842                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1843                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1844
1845                let byte_len = data.len() * T::element_size();
1846                let raw =
1847                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
1848
1849                let inner = borrow_inner(&self.file_inner);
1850                match &*inner {
1851                    H5FileInner::Writer(writer) => {
1852                        writer.write_slice(*index, &starts_u64, &counts_u64, raw)?;
1853                        Ok(())
1854                    }
1855                    _ => Err(Hdf5Error::InvalidState(
1856                        "file is no longer in write mode".into(),
1857                    )),
1858                }
1859            }
1860            DatasetInfo::Reader { .. } => {
1861                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1862            }
1863        }
1864    }
1865
1866    /// Replace elements `start .. start + strings.len()` of a 1-D
1867    /// variable-length string dataset.
1868    ///
1869    /// The extent and every element outside the range are left alone, and the
1870    /// cost is the new strings plus the chunks holding their references — not
1871    /// the column. The dataset's character set is enforced: a non-ASCII
1872    /// replacement in a dataset that declares ASCII is rejected rather than
1873    /// stored under a datatype that misdescribes it.
1874    ///
1875    /// The global heap objects the replaced references pointed at are freed —
1876    /// the same reclaim libhdf5 performs on an overwrite — so updating one
1877    /// element repeatedly reuses space rather than growing the file. A
1878    /// collection emptied by the update returns its block to the allocator.
1879    /// Under SWMR nothing is freed, because a reader may still be following
1880    /// those references.
1881    ///
1882    /// ```no_run
1883    /// # use rust_hdf5::H5File;
1884    /// let file = H5File::open_rw("meta.h5").unwrap();
1885    /// let ds = file.dataset_writer("notes").unwrap();
1886    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
1887    /// file.close().unwrap();
1888    /// ```
1889    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
1890        match &self.info {
1891            DatasetInfo::Writer { index, .. } => {
1892                let inner = borrow_inner(&self.file_inner);
1893                match &*inner {
1894                    H5FileInner::Writer(writer) => {
1895                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
1896                        Ok(())
1897                    }
1898                    _ => Err(Hdf5Error::InvalidState(
1899                        "file is no longer in write mode".into(),
1900                    )),
1901                }
1902            }
1903            DatasetInfo::Reader { .. } => {
1904                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1905            }
1906        }
1907    }
1908
1909    /// Read variable-length strings from a dataset.
1910    ///
1911    /// This handles h5py-style vlen string datasets that store strings
1912    /// as global heap references. Returns one String per element.
1913    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
1914        match &self.info {
1915            DatasetInfo::Reader { name, .. } => {
1916                let mut inner = borrow_inner_mut(&self.file_inner);
1917                match &mut *inner {
1918                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
1919                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1920                }
1921            }
1922            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1923                "cannot read vlen strings from a dataset in write mode".into(),
1924            )),
1925        }
1926    }
1927
1928    /// Read variable-length byte arrays from a dataset.
1929    ///
1930    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
1931    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
1932    /// that store each element as a global heap reference. Returns one
1933    /// `Vec<u8>` per element.
1934    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
1935        match &self.info {
1936            DatasetInfo::Reader { name, .. } => {
1937                let mut inner = borrow_inner_mut(&self.file_inner);
1938                match &mut *inner {
1939                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
1940                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1941                }
1942            }
1943            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1944                "cannot read vlen bytes from a dataset in write mode".into(),
1945            )),
1946        }
1947    }
1948
1949    /// Read a string dataset, fixed-width or variable-length, as one `String`
1950    /// per element.
1951    ///
1952    /// The width of a `FixedString` dataset is whatever the file says, so a
1953    /// 24-byte label column and a 100-byte one are read by the same call. The
1954    /// padding rule the datatype declares decides where each element ends —
1955    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
1956    /// character set decides how the remaining bytes are decoded: ASCII (0)
1957    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
1958    /// violates either is an error naming the element, not a silent
1959    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
1960    /// call that accepts such a file, replacing what it cannot decode.
1961    ///
1962    /// ```no_run
1963    /// # use rust_hdf5::H5File;
1964    /// let file = H5File::open("labels.h5").unwrap();
1965    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
1966    /// ```
1967    pub fn read_strings(&self) -> Result<Vec<String>> {
1968        self.read_strings_inner(false)
1969    }
1970
1971    /// [`read_strings`](Self::read_strings), but bytes that do not decode
1972    /// under the dataset's character set become U+FFFD instead of an error.
1973    ///
1974    /// Producers do mislabel the character set — a file that declares ASCII
1975    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
1976    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
1977        self.read_strings_inner(true)
1978    }
1979
1980    /// The single owner of string decoding for both string datatypes: the
1981    /// element bytes are found differently, the padding and character-set
1982    /// rules that turn them into a `String` are the same.
1983    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
1984        if matches!(self.info, DatasetInfo::Writer { .. }) {
1985            return Err(Hdf5Error::InvalidState(
1986                "cannot read strings from a dataset in write mode".into(),
1987            ));
1988        }
1989        match self.datatype()? {
1990            DatatypeMessage::VarLenString { charset } => self
1991                .read_vlen_bytes()?
1992                .iter()
1993                .enumerate()
1994                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
1995                .collect(),
1996            DatatypeMessage::FixedString {
1997                size,
1998                padding,
1999                charset,
2000            } => {
2001                let width = size as usize;
2002                if width == 0 {
2003                    // A corrupt file can declare it; `chunks_exact(0)` panics.
2004                    return Err(Hdf5Error::InvalidState(
2005                        "fixed-string datatype has zero width".into(),
2006                    ));
2007                }
2008                // `read_raw_bytes` returns `product(dims) * width` bytes, so
2009                // `chunks_exact` leaves no remainder.
2010                let raw = self.read_raw_bytes()?;
2011                raw.chunks_exact(width)
2012                    .enumerate()
2013                    .map(|(i, elem)| {
2014                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
2015                    })
2016                    .collect()
2017            }
2018            other => Err(Hdf5Error::InvalidState(format!(
2019                "read_strings is only for string datasets, this one is {other:?}"
2020            ))),
2021        }
2022    }
2023
2024    /// Read the entire dataset as a typed vector.
2025    ///
2026    /// The raw bytes are read from the file and reinterpreted as `T`. The
2027    /// caller must ensure that `T` matches the datatype used when the dataset
2028    /// was written.
2029    ///
2030    /// # Errors
2031    ///
2032    /// Returns an error if:
2033    /// - The file is in write mode.
2034    /// - The raw data size is not a multiple of `T::element_size()`.
2035    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
2036        match &self.info {
2037            DatasetInfo::Reader {
2038                name, element_size, ..
2039            } => {
2040                if T::element_size() != *element_size {
2041                    return Err(Hdf5Error::TypeMismatch(format!(
2042                        "read type has element size {} but dataset has element size {}",
2043                        T::element_size(),
2044                        element_size,
2045                    )));
2046                }
2047
2048                let raw = {
2049                    let mut inner = borrow_inner_mut(&self.file_inner);
2050                    match &mut *inner {
2051                        H5FileInner::Reader(reader) => reader.read_dataset_raw(name)?,
2052                        _ => {
2053                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
2054                        }
2055                    }
2056                };
2057
2058                if raw.len() % T::element_size() != 0 {
2059                    return Err(Hdf5Error::TypeMismatch(format!(
2060                        "raw data size {} is not a multiple of element size {}",
2061                        raw.len(),
2062                        T::element_size(),
2063                    )));
2064                }
2065
2066                let count = raw.len() / T::element_size();
2067                let mut result = Vec::<T>::with_capacity(count);
2068
2069                // Safety: T is Copy + 'static (required by H5Type). We verified
2070                // the byte count matches count * size_of::<T>() above.
2071                // copy_nonoverlapping fills the memory with valid bit patterns
2072                // for all H5Type implementors (numeric primitives).
2073                // We call set_len AFTER the copy so that if an unexpected panic
2074                // occurs, uninitialized memory is never exposed.
2075                unsafe {
2076                    std::ptr::copy_nonoverlapping(
2077                        raw.as_ptr(),
2078                        result.as_mut_ptr() as *mut u8,
2079                        raw.len(),
2080                    );
2081                    result.set_len(count);
2082                }
2083
2084                Ok(result)
2085            }
2086            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2087                "cannot read from a dataset in write mode".into(),
2088            )),
2089        }
2090    }
2091
2092    /// Read the raw byte image of a dataset without an `H5Type` carrier.
2093    ///
2094    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
2095    /// the element bytes verbatim regardless of the on-disk element type, so a
2096    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
2097    /// no matching Rust primitive can be read back and decoded by the caller.
2098    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
2099        match &self.info {
2100            DatasetInfo::Reader { name, .. } => {
2101                let mut inner = borrow_inner_mut(&self.file_inner);
2102                match &mut *inner {
2103                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
2104                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2105                }
2106            }
2107            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2108                "cannot read from a dataset in write mode".into(),
2109            )),
2110        }
2111    }
2112
2113    /// Read a numeric dataset as `T`, converting each element from the
2114    /// on-disk datatype.
2115    ///
2116    /// Unlike [`read_raw`](Self::read_raw), which requires `T`'s size to match
2117    /// the stored element size exactly, this inspects the dataset's datatype
2118    /// message — class, signedness, byte order, width — and converts per
2119    /// element:
2120    ///
2121    /// - integer → integer: checked; a stored value that does not fit in `T`
2122    ///   is an error naming the element index and value, never a silent wrap.
2123    /// - `f32` source → `f64`: exact widening.
2124    /// - `f64` source → `f32`, float → integer, and integer → float are
2125    ///   rejected as [`TypeMismatch`](Hdf5Error::TypeMismatch).
2126    ///
2127    /// Big-endian sources are decoded according to the datatype's byte order,
2128    /// which [`read_raw`](Self::read_raw)'s size-only check would misread.
2129    ///
2130    /// ```no_run
2131    /// # use rust_hdf5::H5File;
2132    /// let file = H5File::open("data.h5").unwrap();
2133    /// let ds = file.dataset("counts").unwrap(); // stored as e.g. i16
2134    /// let counts = ds.read_numeric_as::<i64>().unwrap();
2135    /// ```
2136    pub fn read_numeric_as<T: ReadNumeric>(&self) -> Result<Vec<T>> {
2137        match &self.info {
2138            DatasetInfo::Reader { name, .. } => {
2139                let (kind, raw) = {
2140                    let mut inner = borrow_inner_mut(&self.file_inner);
2141                    match &mut *inner {
2142                        H5FileInner::Reader(reader) => {
2143                            let info = reader
2144                                .dataset_info(name)
2145                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
2146                            let kind = numeric::classify(&info.datatype)?;
2147                            (kind, reader.read_dataset_raw(name)?)
2148                        }
2149                        _ => {
2150                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
2151                        }
2152                    }
2153                };
2154                numeric::convert(kind, &raw)
2155            }
2156            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2157                "cannot read from a dataset in write mode".into(),
2158            )),
2159        }
2160    }
2161
2162    /// Read a slice (hyperslab) of a numeric dataset as `T`, with the same
2163    /// per-element datatype conversion as
2164    /// [`read_numeric_as`](Self::read_numeric_as).
2165    ///
2166    /// `starts` and `counts` define the N-dimensional selection exactly as in
2167    /// [`read_slice`](Self::read_slice).
2168    pub fn read_numeric_slice_as<T: ReadNumeric>(
2169        &self,
2170        starts: &[usize],
2171        counts: &[usize],
2172    ) -> Result<Vec<T>> {
2173        match &self.info {
2174            DatasetInfo::Reader { name, .. } => {
2175                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
2176                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
2177                let (kind, raw) = {
2178                    let mut inner = borrow_inner_mut(&self.file_inner);
2179                    match &mut *inner {
2180                        H5FileInner::Reader(reader) => {
2181                            let info = reader
2182                                .dataset_info(name)
2183                                .ok_or_else(|| Hdf5Error::NotFound(name.clone()))?;
2184                            let kind = numeric::classify(&info.datatype)?;
2185                            (kind, reader.read_slice(name, &starts_u64, &counts_u64)?)
2186                        }
2187                        _ => {
2188                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
2189                        }
2190                    }
2191                };
2192                numeric::convert(kind, &raw)
2193            }
2194            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2195                "cannot read_slice from a dataset in write mode".into(),
2196            )),
2197        }
2198    }
2199
2200    /// Read the whole dataset into a caller-provided buffer, with no allocation.
2201    ///
2202    /// `out` must have exactly `product(dims)` elements (the dataset's element
2203    /// count) and `T::element_size()` must match the dataset's on-disk element
2204    /// size, otherwise an error is returned and `out` is left unspecified. The
2205    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
2206    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
2207    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
2208    /// without the extra staging copy a `read_raw` + copy-into-pinned would
2209    /// incur. Works for every layout (contiguous, compact, and chunked under
2210    /// any index); for chunked data each decoded chunk is scattered directly
2211    /// into `out`.
2212    ///
2213    /// ```no_run
2214    /// # use rust_hdf5::H5File;
2215    /// let file = H5File::open("data.h5").unwrap();
2216    /// let ds = file.dataset("frames").unwrap();
2217    /// let n: usize = ds.shape().iter().product();
2218    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
2219    /// ds.read_raw_into(&mut buf).unwrap();
2220    /// ```
2221    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
2222        match &self.info {
2223            DatasetInfo::Reader {
2224                name, element_size, ..
2225            } => {
2226                if T::element_size() != *element_size {
2227                    return Err(Hdf5Error::TypeMismatch(format!(
2228                        "read type has element size {} but dataset has element size {}",
2229                        T::element_size(),
2230                        element_size,
2231                    )));
2232                }
2233                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
2234                // byte representation; every bit pattern the read writes is a
2235                // valid `T`. The byte view borrows `out` exclusively for this
2236                // call, and `out.len() * element_size` cannot overflow because
2237                // it is the byte length of an existing slice (<= isize::MAX).
2238                let byte_len = out.len() * T::element_size();
2239                let bytes = unsafe {
2240                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
2241                };
2242                let mut inner = borrow_inner_mut(&self.file_inner);
2243                match &mut *inner {
2244                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw_into(name, bytes)?),
2245                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2246                }
2247            }
2248            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2249                "cannot read from a dataset in write mode".into(),
2250            )),
2251        }
2252    }
2253
2254    /// Read a hyperslab into a caller-provided buffer, with no allocation.
2255    ///
2256    /// `out` must have exactly `product(counts)` elements and
2257    /// `T::element_size()` must match the dataset's element size. The zero-copy
2258    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
2259    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
2260    /// selection are read, and the selected bytes land directly in `out` — the
2261    /// entry point for reading one frame / block straight into a pinned host
2262    /// buffer for an H2D transfer.
2263    ///
2264    /// ```no_run
2265    /// # use rust_hdf5::H5File;
2266    /// let file = H5File::open("vol.h5").unwrap();
2267    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
2268    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
2269    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
2270    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
2271    /// ```
2272    pub fn read_slice_into<T: H5Type>(
2273        &self,
2274        out: &mut [T],
2275        starts: &[usize],
2276        counts: &[usize],
2277    ) -> Result<()> {
2278        match &self.info {
2279            DatasetInfo::Reader {
2280                name, element_size, ..
2281            } => {
2282                if T::element_size() != *element_size {
2283                    return Err(Hdf5Error::TypeMismatch(format!(
2284                        "read type has element size {} but dataset has element size {}",
2285                        T::element_size(),
2286                        element_size,
2287                    )));
2288                }
2289                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
2290                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
2291                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
2292                // borrow of `out`, byte length within bounds.
2293                let byte_len = out.len() * T::element_size();
2294                let bytes = unsafe {
2295                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
2296                };
2297                let mut inner = borrow_inner_mut(&self.file_inner);
2298                match &mut *inner {
2299                    H5FileInner::Reader(reader) => {
2300                        Ok(reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?)
2301                    }
2302                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2303                }
2304            }
2305            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2306                "cannot read from a dataset in write mode".into(),
2307            )),
2308        }
2309    }
2310}
2311
2312// ---------------------------------------------------------------------------
2313// Datatype-aware numeric conversion (read_numeric_as)
2314// ---------------------------------------------------------------------------
2315
2316/// Marker trait for the Rust types [`H5Dataset::read_numeric_as`] can convert
2317/// into: the integer primitives (checked, never wrapping) plus `f32`/`f64`
2318/// (widening only).
2319///
2320/// Sealed — the conversion policy is part of the library contract, so the
2321/// trait cannot be implemented outside this crate.
2322pub trait ReadNumeric: numeric::Sealed {}
2323impl<T: numeric::Sealed> ReadNumeric for T {}
2324
2325pub(crate) mod numeric {
2326    //! Per-element decode + checked conversion for `read_numeric_as` (and the
2327    //! attribute counterpart `H5Attribute::read_numeric_as`).
2328
2329    use crate::error::{Hdf5Error, Result};
2330    use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
2331
2332    /// A source element, normalized: every standard integer width — u64::MAX
2333    /// included — fits in `i128` without loss.
2334    pub enum NumericSource {
2335        Int(i128),
2336        F32(f32),
2337        F64(f64),
2338    }
2339
2340    /// The on-disk element shape `classify` accepted.
2341    #[derive(Clone, Copy)]
2342    pub enum SourceKind {
2343        Int {
2344            size: usize,
2345            signed: bool,
2346            byte_order: ByteOrder,
2347        },
2348        F32(ByteOrder),
2349        F64(ByteOrder),
2350    }
2351
2352    impl SourceKind {
2353        fn element_size(self) -> usize {
2354            match self {
2355                SourceKind::Int { size, .. } => size,
2356                SourceKind::F32(_) => 4,
2357                SourceKind::F64(_) => 8,
2358            }
2359        }
2360    }
2361
2362    /// Map a datatype message to a supported numeric source shape.
2363    ///
2364    /// Accepts standard-width integers (1/2/4/8 bytes, full precision, zero
2365    /// bit offset) and IEEE binary32/binary64 floats; everything else is a
2366    /// `TypeMismatch` naming what was found.
2367    pub fn classify(dt: &DatatypeMessage) -> Result<SourceKind> {
2368        match *dt {
2369            DatatypeMessage::FixedPoint {
2370                size,
2371                byte_order,
2372                signed,
2373                bit_offset,
2374                bit_precision,
2375            } => {
2376                if !matches!(size, 1 | 2 | 4 | 8)
2377                    || bit_offset != 0
2378                    || u32::from(bit_precision) != size * 8
2379                {
2380                    return Err(Hdf5Error::TypeMismatch(format!(
2381                        "fixed-point datatype (size {size}, bit offset {bit_offset}, \
2382                         precision {bit_precision}) is not a standard-width integer",
2383                    )));
2384                }
2385                Ok(SourceKind::Int {
2386                    size: size as usize,
2387                    signed,
2388                    byte_order,
2389                })
2390            }
2391            DatatypeMessage::FloatingPoint {
2392                size,
2393                byte_order,
2394                exponent_size,
2395                mantissa_size,
2396                ..
2397            } => match (size, exponent_size, mantissa_size) {
2398                (4, 8, 23) => Ok(SourceKind::F32(byte_order)),
2399                (8, 11, 52) => Ok(SourceKind::F64(byte_order)),
2400                _ => Err(Hdf5Error::TypeMismatch(format!(
2401                    "floating-point datatype (size {size}, exponent {exponent_size} bits, \
2402                     mantissa {mantissa_size} bits) is not IEEE binary32 or binary64",
2403                ))),
2404            },
2405            ref other => Err(Hdf5Error::TypeMismatch(format!(
2406                "dataset datatype '{other}' is not numeric",
2407            ))),
2408        }
2409    }
2410
2411    fn decode_element(kind: SourceKind, bytes: &[u8]) -> NumericSource {
2412        match kind {
2413            SourceKind::Int {
2414                size,
2415                signed,
2416                byte_order,
2417            } => {
2418                let mut le = [0u8; 8];
2419                match byte_order {
2420                    ByteOrder::LittleEndian => le[..size].copy_from_slice(bytes),
2421                    ByteOrder::BigEndian => {
2422                        for (dst, src) in le[..size].iter_mut().zip(bytes.iter().rev()) {
2423                            *dst = *src;
2424                        }
2425                    }
2426                }
2427                let zero_extended = u64::from_le_bytes(le);
2428                let value = if signed {
2429                    // Arithmetic right shift sign-extends the low `size` bytes.
2430                    let shift = 64 - 8 * size as u32;
2431                    i128::from(((zero_extended as i64) << shift) >> shift)
2432                } else {
2433                    i128::from(zero_extended)
2434                };
2435                NumericSource::Int(value)
2436            }
2437            SourceKind::F32(byte_order) => {
2438                let arr: [u8; 4] = bytes.try_into().unwrap();
2439                NumericSource::F32(match byte_order {
2440                    ByteOrder::LittleEndian => f32::from_le_bytes(arr),
2441                    ByteOrder::BigEndian => f32::from_be_bytes(arr),
2442                })
2443            }
2444            SourceKind::F64(byte_order) => {
2445                let arr: [u8; 8] = bytes.try_into().unwrap();
2446                NumericSource::F64(match byte_order {
2447                    ByteOrder::LittleEndian => f64::from_le_bytes(arr),
2448                    ByteOrder::BigEndian => f64::from_be_bytes(arr),
2449                })
2450            }
2451        }
2452    }
2453
2454    /// Decode and convert every element of `raw` into `T`.
2455    pub fn convert<T: Sealed>(kind: SourceKind, raw: &[u8]) -> Result<Vec<T>> {
2456        let size = kind.element_size();
2457        if !raw.len().is_multiple_of(size) {
2458            return Err(Hdf5Error::TypeMismatch(format!(
2459                "raw data size {} is not a multiple of element size {size}",
2460                raw.len(),
2461            )));
2462        }
2463        raw.chunks_exact(size)
2464            .enumerate()
2465            .map(|(index, bytes)| T::from_source(decode_element(kind, bytes), index))
2466            .collect()
2467    }
2468
2469    /// The sealed half of `ReadNumeric`: how one normalized source element
2470    /// becomes a `Self`, or a `TypeMismatch` explaining why it cannot.
2471    pub trait Sealed: Sized {
2472        fn from_source(src: NumericSource, index: usize) -> Result<Self>;
2473    }
2474
2475    macro_rules! int_targets {
2476        ($($t:ty),* $(,)?) => {$(
2477            impl Sealed for $t {
2478                fn from_source(src: NumericSource, index: usize) -> Result<Self> {
2479                    match src {
2480                        NumericSource::Int(v) => <$t>::try_from(v).map_err(|_| {
2481                            Hdf5Error::TypeMismatch(format!(
2482                                concat!(
2483                                    "value {} at element {} does not fit in ",
2484                                    stringify!($t),
2485                                ),
2486                                v, index,
2487                            ))
2488                        }),
2489                        NumericSource::F32(_) | NumericSource::F64(_) => {
2490                            Err(Hdf5Error::TypeMismatch(
2491                                concat!(
2492                                    "cannot read a floating-point dataset as ",
2493                                    stringify!($t),
2494                                    "; read as f64 and convert explicitly",
2495                                )
2496                                .into(),
2497                            ))
2498                        }
2499                    }
2500                }
2501            }
2502        )*};
2503    }
2504    int_targets!(i8, i16, i32, i64, u8, u16, u32, u64, u128);
2505
2506    // Not in the macro: `i128::try_from(i128)` is infallible, which trips
2507    // clippy::unnecessary_fallible_conversions.
2508    impl Sealed for i128 {
2509        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
2510            match src {
2511                NumericSource::Int(v) => Ok(v),
2512                NumericSource::F32(_) | NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
2513                    "cannot read a floating-point dataset as i128; read as f64 and \
2514                     convert explicitly"
2515                        .into(),
2516                )),
2517            }
2518        }
2519    }
2520
2521    impl Sealed for f32 {
2522        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
2523            match src {
2524                NumericSource::F32(v) => Ok(v),
2525                NumericSource::F64(_) => Err(Hdf5Error::TypeMismatch(
2526                    "narrowing an f64 dataset to f32 loses precision; read as f64".into(),
2527                )),
2528                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
2529                    "cannot read an integer dataset as f32; read as an integer type and \
2530                     convert explicitly"
2531                        .into(),
2532                )),
2533            }
2534        }
2535    }
2536
2537    impl Sealed for f64 {
2538        fn from_source(src: NumericSource, _index: usize) -> Result<Self> {
2539            match src {
2540                NumericSource::F64(v) => Ok(v),
2541                // Every f32 is exactly representable as f64.
2542                NumericSource::F32(v) => Ok(f64::from(v)),
2543                NumericSource::Int(_) => Err(Hdf5Error::TypeMismatch(
2544                    "cannot read an integer dataset as f64; integers above 2^53 lose \
2545                     precision — read as an integer type and convert explicitly"
2546                        .into(),
2547                )),
2548            }
2549        }
2550    }
2551}
2552
2553#[cfg(test)]
2554mod tests {
2555    use crate::H5File;
2556    use std::path::PathBuf;
2557
2558    fn temp_path(name: &str) -> PathBuf {
2559        // Include PID + a per-call atomic counter so that concurrent
2560        // cargo invocations and any kernel-level "lock not yet
2561        // released" races between sequential opens cannot collide.
2562        use std::sync::atomic::{AtomicU64, Ordering};
2563        static COUNTER: AtomicU64 = AtomicU64::new(0);
2564        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
2565        std::env::temp_dir().join(format!(
2566            "hdf5_dataset_test_{}_{}_{}.h5",
2567            name,
2568            std::process::id(),
2569            n
2570        ))
2571    }
2572
2573    #[test]
2574    fn runtime_compound_via_datatype_override_and_raw_bytes() {
2575        use crate::format::messages::datatype::DatatypeMessage;
2576        use crate::types::{CompoundType, H5Type};
2577
2578        let path = temp_path("compound_raw");
2579        // A 12-byte packed compound with NO matching Rust primitive carrier,
2580        // so it can only be written through the datatype() override +
2581        // write_raw_bytes path (the runtime-CompoundType use case).
2582        let ct = CompoundType {
2583            members: vec![
2584                ("id".to_string(), i32::hdf5_type(), 0),
2585                ("val".to_string(), f64::hdf5_type(), 4),
2586            ],
2587            total_size: 12,
2588        };
2589        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
2590        let mut bytes = Vec::new();
2591        for (id, val) in recs {
2592            bytes.extend_from_slice(&id.to_le_bytes());
2593            bytes.extend_from_slice(&val.to_le_bytes());
2594        }
2595
2596        {
2597            let file = H5File::create(&path).unwrap();
2598            let ds = file
2599                .new_dataset::<u8>()
2600                .datatype(ct.to_datatype())
2601                .shape([recs.len()])
2602                .create("records")
2603                .unwrap();
2604            ds.write_raw_bytes(&bytes).unwrap();
2605            file.close().unwrap();
2606        }
2607        {
2608            let file = H5File::open(&path).unwrap();
2609            let ds = file.dataset("records").unwrap();
2610            // The on-disk element type is the compound we specified (size 12),
2611            // not the u8 carrier.
2612            match ds.datatype().unwrap() {
2613                DatatypeMessage::Compound { size, members } => {
2614                    assert_eq!(size, 12);
2615                    assert_eq!(members.len(), 2);
2616                    assert_eq!(members[0].name, "id");
2617                    assert_eq!(members[0].offset, 0);
2618                    assert_eq!(members[1].name, "val");
2619                    assert_eq!(members[1].offset, 4);
2620                }
2621                other => panic!("expected compound datatype, got {other:?}"),
2622            }
2623            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
2624        }
2625        std::fs::remove_file(&path).ok();
2626    }
2627
2628    #[test]
2629    fn builder_requires_shape() {
2630        let path = temp_path("no_shape");
2631        let file = H5File::create(&path).unwrap();
2632        let result = file.new_dataset::<u8>().create("data");
2633        assert!(result.is_err());
2634        std::fs::remove_file(&path).ok();
2635    }
2636
2637    #[test]
2638    fn write_raw_size_mismatch() {
2639        let path = temp_path("size_mismatch");
2640        let file = H5File::create(&path).unwrap();
2641        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
2642        // Provide 3 elements instead of 4
2643        let result = ds.write_raw(&[1u8, 2, 3]);
2644        assert!(result.is_err());
2645        std::fs::remove_file(&path).ok();
2646    }
2647
2648    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
2649    // dataset = one chunk) rather than silently drop the filter on the
2650    // contiguous path. write_raw then populates that single chunk.
2651    #[cfg(feature = "deflate")]
2652    #[test]
2653    fn filter_without_chunk_autochunks_and_roundtrips() {
2654        let path = temp_path("autochunk_filter");
2655        let data: Vec<i32> = (0..8).collect();
2656        {
2657            let file = H5File::create(&path).unwrap();
2658            let ds = file
2659                .new_dataset::<i32>()
2660                .deflate(6)
2661                .shape([8])
2662                .create("seq")
2663                .unwrap();
2664            ds.write_raw(&data).unwrap();
2665            file.close().unwrap();
2666        }
2667        {
2668            let file = H5File::open(&path).unwrap();
2669            let ds = file.dataset("seq").unwrap();
2670            // The filter forced chunked storage: a single whole-dataset chunk.
2671            assert!(
2672                ds.is_chunked(),
2673                "auto-chunk did not produce chunked storage"
2674            );
2675            assert_eq!(ds.chunk_dims(), Some(vec![8]));
2676            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2677        }
2678        std::fs::remove_file(&path).ok();
2679    }
2680
2681    // B: write_raw on an explicitly chunked + compressed dataset scatters the
2682    // full row-major image across a multi-chunk grid, including edge chunks
2683    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
2684    #[cfg(feature = "deflate")]
2685    #[test]
2686    fn write_raw_multichunk_edge_roundtrips() {
2687        let path = temp_path("multichunk_edge");
2688        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
2689        {
2690            let file = H5File::create(&path).unwrap();
2691            let ds = file
2692                .new_dataset::<i32>()
2693                .shape([7, 5])
2694                .chunk(&[3, 2])
2695                .deflate(4)
2696                .create("grid")
2697                .unwrap();
2698            ds.write_raw(&data).unwrap();
2699            file.close().unwrap();
2700        }
2701        {
2702            let file = H5File::open(&path).unwrap();
2703            let ds = file.dataset("grid").unwrap();
2704            assert_eq!(ds.shape(), vec![7, 5]);
2705            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2706            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2707        }
2708        std::fs::remove_file(&path).ok();
2709    }
2710
2711    // B: write_raw on an unfiltered chunked dataset (previously rejected with
2712    // "use write_chunk for chunked datasets") now gathers and round-trips.
2713    #[test]
2714    fn write_raw_unfiltered_chunked_roundtrips() {
2715        let path = temp_path("chunked_unfiltered");
2716        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
2717        {
2718            let file = H5File::create(&path).unwrap();
2719            let ds = file
2720                .new_dataset::<f64>()
2721                .shape([4, 3])
2722                .chunk(&[2, 2])
2723                .create("m")
2724                .unwrap();
2725            ds.write_raw(&data).unwrap();
2726            file.close().unwrap();
2727        }
2728        {
2729            let file = H5File::open(&path).unwrap();
2730            let ds = file.dataset("m").unwrap();
2731            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
2732            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
2733        }
2734        std::fs::remove_file(&path).ok();
2735    }
2736
2737    // write_raw on an extensible-array (unlimited first dim) compressed dataset
2738    // drives write_full_image_chunked's EA branch, which gathers chunks and
2739    // compresses them through the windowed batch path. Round-trips the full
2740    // image, including a partial edge chunk along the unlimited dimension.
2741    #[cfg(feature = "deflate")]
2742    #[test]
2743    fn write_raw_ea_compressed_roundtrips() {
2744        let path = temp_path("write_raw_ea_deflate");
2745        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
2746        {
2747            let file = H5File::create(&path).unwrap();
2748            let ds = file
2749                .new_dataset::<i32>()
2750                .shape([5, 4])
2751                .chunk(&[2, 4])
2752                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
2753                .deflate(5)
2754                .create("stream")
2755                .unwrap();
2756            ds.write_raw(&data).unwrap();
2757            file.close().unwrap();
2758        }
2759        {
2760            let file = H5File::open(&path).unwrap();
2761            let ds = file.dataset("stream").unwrap();
2762            assert_eq!(ds.shape(), vec![5, 4]);
2763            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2764        }
2765        std::fs::remove_file(&path).ok();
2766    }
2767
2768    // 3D chunked Full read with partial-edge chunks in every dimension. This
2769    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
2770    // dimensions, exercising the nested outer-coordinate carry and the
2771    // last-axis edge clamp (chunks hang off the high edge in all three axes).
2772    #[cfg(feature = "deflate")]
2773    #[test]
2774    fn read_full_3d_chunked_edge_roundtrips() {
2775        let path = temp_path("full_3d_chunked_edge");
2776        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
2777        // along each axis is partial (1, 1, and 1 element respectively).
2778        let total: usize = 5 * 4 * 3;
2779        let data: Vec<i32> = (0..total as i32).collect();
2780        {
2781            let file = H5File::create(&path).unwrap();
2782            let ds = file
2783                .new_dataset::<i32>()
2784                .shape([5, 4, 3])
2785                .chunk(&[2, 3, 2])
2786                .deflate(4)
2787                .create("vol")
2788                .unwrap();
2789            ds.write_raw(&data).unwrap();
2790            file.close().unwrap();
2791        }
2792        {
2793            let file = H5File::open(&path).unwrap();
2794            let ds = file.dataset("vol").unwrap();
2795            assert_eq!(ds.shape(), vec![5, 4, 3]);
2796            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
2797            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2798        }
2799        std::fs::remove_file(&path).ok();
2800    }
2801
2802    #[test]
2803    fn roundtrip_u8_1d() {
2804        let path = temp_path("rt_u8_1d");
2805        let data: Vec<u8> = (0..10).collect();
2806
2807        {
2808            let file = H5File::create(&path).unwrap();
2809            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
2810            ds.write_raw(&data).unwrap();
2811            file.close().unwrap();
2812        }
2813
2814        {
2815            let file = H5File::open(&path).unwrap();
2816            let ds = file.dataset("seq").unwrap();
2817            assert_eq!(ds.shape(), vec![10]);
2818            let readback = ds.read_raw::<u8>().unwrap();
2819            assert_eq!(readback, data);
2820        }
2821
2822        std::fs::remove_file(&path).ok();
2823    }
2824
2825    #[test]
2826    fn roundtrip_i32_2d() {
2827        let path = temp_path("rt_i32_2d");
2828        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
2829
2830        {
2831            let file = H5File::create(&path).unwrap();
2832            let ds = file
2833                .new_dataset::<i32>()
2834                .shape([2, 3])
2835                .create("matrix")
2836                .unwrap();
2837            ds.write_raw(&data).unwrap();
2838            file.close().unwrap();
2839        }
2840
2841        {
2842            let file = H5File::open(&path).unwrap();
2843            let ds = file.dataset("matrix").unwrap();
2844            assert_eq!(ds.shape(), vec![2, 3]);
2845            let readback = ds.read_raw::<i32>().unwrap();
2846            assert_eq!(readback, data);
2847        }
2848
2849        std::fs::remove_file(&path).ok();
2850    }
2851
2852    #[test]
2853    fn roundtrip_f64_3d() {
2854        let path = temp_path("rt_f64_3d");
2855        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
2856
2857        {
2858            let file = H5File::create(&path).unwrap();
2859            let ds = file
2860                .new_dataset::<f64>()
2861                .shape([2, 3, 4])
2862                .create("cube")
2863                .unwrap();
2864            ds.write_raw(&data).unwrap();
2865            file.close().unwrap();
2866        }
2867
2868        {
2869            let file = H5File::open(&path).unwrap();
2870            let ds = file.dataset("cube").unwrap();
2871            assert_eq!(ds.shape(), vec![2, 3, 4]);
2872            let readback = ds.read_raw::<f64>().unwrap();
2873            assert_eq!(readback, data);
2874        }
2875
2876        std::fs::remove_file(&path).ok();
2877    }
2878
2879    #[test]
2880    fn cannot_read_in_write_mode() {
2881        let path = temp_path("no_read_write");
2882        let file = H5File::create(&path).unwrap();
2883        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2884        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2885        let result = ds.read_raw::<u8>();
2886        assert!(result.is_err());
2887        std::fs::remove_file(&path).ok();
2888    }
2889
2890    #[test]
2891    fn cannot_write_in_read_mode() {
2892        let path = temp_path("no_write_read");
2893
2894        {
2895            let file = H5File::create(&path).unwrap();
2896            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2897            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2898            file.close().unwrap();
2899        }
2900
2901        {
2902            let file = H5File::open(&path).unwrap();
2903            let ds = file.dataset("x").unwrap();
2904            let result = ds.write_raw(&[5u8, 6, 7, 8]);
2905            assert!(result.is_err());
2906        }
2907
2908        std::fs::remove_file(&path).ok();
2909    }
2910
2911    #[test]
2912    fn numeric_attr_roundtrip() {
2913        let path = temp_path("num_attr");
2914        {
2915            let file = H5File::create(&path).unwrap();
2916            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2917            ds.write_raw(&[1.0f32; 4]).unwrap();
2918
2919            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
2920            a1.write_numeric(&1.2345f64).unwrap();
2921
2922            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
2923            a2.write_numeric(&42i32).unwrap();
2924
2925            file.close().unwrap();
2926        }
2927        {
2928            let file = H5File::open(&path).unwrap();
2929            let ds = file.dataset("data").unwrap();
2930
2931            let scale = ds.attr("scale").unwrap();
2932            let val: f64 = scale.read_numeric().unwrap();
2933            assert!((val - 1.2345).abs() < 1e-10);
2934
2935            let count = ds.attr("count").unwrap();
2936            let val: i32 = count.read_numeric().unwrap();
2937            assert_eq!(val, 42);
2938        }
2939        std::fs::remove_file(&path).ok();
2940    }
2941
2942    #[test]
2943    fn array_attr_roundtrip() {
2944        let path = temp_path("array_attr");
2945        let offsets = [10i32, -20, 30];
2946        {
2947            let file = H5File::create(&path).unwrap();
2948            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2949            ds.write_raw(&[1.0f32; 4]).unwrap();
2950
2951            // 1-D int32 array attribute (NDArrayDimOffset-style).
2952            let a = ds
2953                .new_attr::<i32>()
2954                .shape([3])
2955                .create("dim_offset")
2956                .unwrap();
2957            a.write_array(&offsets).unwrap();
2958
2959            // Wrong element count is rejected.
2960            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
2961            assert!(bad.write_array(&[1i32, 2]).is_err());
2962
2963            file.close().unwrap();
2964        }
2965        {
2966            let file = H5File::open(&path).unwrap();
2967            let ds = file.dataset("data").unwrap();
2968            let a = ds.attr("dim_offset").unwrap();
2969            let raw = a.read_raw().unwrap();
2970            assert_eq!(raw.len(), 3 * 4);
2971            let got: Vec<i32> = raw
2972                .chunks_exact(4)
2973                .map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
2974                .collect();
2975            assert_eq!(got, offsets);
2976        }
2977        std::fs::remove_file(&path).ok();
2978    }
2979
2980    #[test]
2981    fn attr_datatype_exposes_class_and_sign() {
2982        // H5Attribute::datatype() must report the stored datatype class and
2983        // signedness so a generic attr->metadata mapper need not infer it from
2984        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
2985        use crate::format::messages::datatype::DatatypeMessage;
2986
2987        let path = temp_path("attr_datatype");
2988        {
2989            let file = H5File::create(&path).unwrap();
2990            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2991            ds.new_attr::<f64>()
2992                .shape(())
2993                .create("scale")
2994                .unwrap()
2995                .write_numeric(&1.5f64)
2996                .unwrap();
2997            ds.new_attr::<i32>()
2998                .shape(())
2999                .create("count")
3000                .unwrap()
3001                .write_numeric(&7i32)
3002                .unwrap();
3003            file.close().unwrap();
3004        }
3005        {
3006            let file = H5File::open(&path).unwrap();
3007            let ds = file.dataset("data").unwrap();
3008
3009            match ds.attr("scale").unwrap().datatype().unwrap() {
3010                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
3011                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
3012            }
3013
3014            match ds.attr("count").unwrap().datatype().unwrap() {
3015                DatatypeMessage::FixedPoint { size, signed, .. } => {
3016                    assert_eq!(size, 4);
3017                    assert!(signed, "i32 attr must be signed");
3018                }
3019                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
3020            }
3021        }
3022        std::fs::remove_file(&path).ok();
3023    }
3024
3025    #[test]
3026    fn attr_datatype_in_write_mode_errors() {
3027        let path = temp_path("attr_datatype_write_mode");
3028        let file = H5File::create(&path).unwrap();
3029        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
3030        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
3031        assert!(attr.datatype().is_err());
3032        std::fs::remove_file(&path).ok();
3033    }
3034
3035    #[test]
3036    fn cannot_create_dataset_in_read_mode() {
3037        let path = temp_path("no_create_read");
3038
3039        {
3040            let _file = H5File::create(&path).unwrap();
3041        }
3042
3043        {
3044            let file = H5File::open(&path).unwrap();
3045            let result = file.new_dataset::<u8>().shape([4]).create("x");
3046            assert!(result.is_err());
3047        }
3048
3049        std::fs::remove_file(&path).ok();
3050    }
3051
3052    #[test]
3053    fn shape_accessor() {
3054        let path = temp_path("shape_acc");
3055
3056        let file = H5File::create(&path).unwrap();
3057        let ds = file
3058            .new_dataset::<f32>()
3059            .shape([5, 10, 3])
3060            .create("tensor")
3061            .unwrap();
3062        assert_eq!(ds.shape(), vec![5, 10, 3]);
3063
3064        std::fs::remove_file(&path).ok();
3065    }
3066
3067    #[test]
3068    fn slice_roundtrip_2d() {
3069        let path = temp_path("slice_2d");
3070
3071        // Create a 4x5 dataset, write full, then read a slice
3072        let data: Vec<i32> = (0..20).collect();
3073        {
3074            let file = H5File::create(&path).unwrap();
3075            let ds = file
3076                .new_dataset::<i32>()
3077                .shape([4, 5])
3078                .create("mat")
3079                .unwrap();
3080            ds.write_raw(&data).unwrap();
3081            file.close().unwrap();
3082        }
3083        {
3084            let file = H5File::open(&path).unwrap();
3085            let ds = file.dataset("mat").unwrap();
3086            // Read rows 1..3, cols 2..4 (2x2 slice)
3087            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
3088            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
3089            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
3090            assert_eq!(slice, vec![7, 8, 12, 13]);
3091        }
3092
3093        std::fs::remove_file(&path).ok();
3094    }
3095
3096    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
3097    // caller-provided buffer and MUST produce byte-for-byte the same data as
3098    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
3099    // creatable layout, since both now share one buffer-filling core.
3100    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
3101    where
3102        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
3103    {
3104        let n: usize = ds.shape().iter().product();
3105        let want_full = ds.read_raw::<T>().unwrap();
3106        let mut got_full = vec![T::default(); n];
3107        ds.read_raw_into::<T>(&mut got_full).unwrap();
3108        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
3109
3110        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
3111        let sn: usize = counts.iter().product();
3112        let mut got_slice = vec![T::default(); sn];
3113        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
3114            .unwrap();
3115        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
3116    }
3117
3118    #[test]
3119    fn read_into_matches_vec_contiguous() {
3120        let path = temp_path("into_contig");
3121        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
3122        {
3123            let file = H5File::create(&path).unwrap();
3124            let ds = file
3125                .new_dataset::<i32>()
3126                .shape([4, 5])
3127                .create("mat")
3128                .unwrap();
3129            ds.write_raw(&data).unwrap();
3130            file.close().unwrap();
3131        }
3132        {
3133            let file = H5File::open(&path).unwrap();
3134            let ds = file.dataset("mat").unwrap();
3135            assert_eq!(ds.chunk_dims(), None);
3136            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
3137        }
3138        std::fs::remove_file(&path).ok();
3139    }
3140
3141    #[test]
3142    fn read_into_matches_vec_chunked_unfiltered() {
3143        let path = temp_path("into_chunk");
3144        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
3145        {
3146            let file = H5File::create(&path).unwrap();
3147            let ds = file
3148                .new_dataset::<f64>()
3149                .shape([7, 5])
3150                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
3151                .create("grid")
3152                .unwrap();
3153            ds.write_raw(&data).unwrap();
3154            file.close().unwrap();
3155        }
3156        {
3157            let file = H5File::open(&path).unwrap();
3158            let ds = file.dataset("grid").unwrap();
3159            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
3160            // Slice spans multiple chunks (rows 2..5, cols 1..4).
3161            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
3162        }
3163        std::fs::remove_file(&path).ok();
3164    }
3165
3166    #[test]
3167    fn read_into_matches_vec_single_chunk() {
3168        let path = temp_path("into_single_chunk");
3169        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
3170        {
3171            let file = H5File::create(&path).unwrap();
3172            let ds = file
3173                .new_dataset::<i32>()
3174                .shape([3, 4])
3175                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
3176                .create("g")
3177                .unwrap();
3178            ds.write_raw(&data).unwrap();
3179            file.close().unwrap();
3180        }
3181        {
3182            let file = H5File::open(&path).unwrap();
3183            let ds = file.dataset("g").unwrap();
3184            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
3185            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
3186        }
3187        std::fs::remove_file(&path).ok();
3188    }
3189
3190    #[cfg(feature = "deflate")]
3191    #[test]
3192    fn read_into_matches_vec_chunked_deflate() {
3193        let path = temp_path("into_chunk_deflate");
3194        let data: Vec<i32> = (0..35).collect(); // 7 x 5
3195        {
3196            let file = H5File::create(&path).unwrap();
3197            let ds = file
3198                .new_dataset::<i32>()
3199                .shape([7, 5])
3200                .chunk(&[3, 2])
3201                .deflate(4)
3202                .create("grid")
3203                .unwrap();
3204            ds.write_raw(&data).unwrap();
3205            file.close().unwrap();
3206        }
3207        {
3208            let file = H5File::open(&path).unwrap();
3209            let ds = file.dataset("grid").unwrap();
3210            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
3211            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
3212        }
3213        std::fs::remove_file(&path).ok();
3214    }
3215
3216    #[test]
3217    fn read_into_wrong_buffer_size_rejected() {
3218        let path = temp_path("into_badlen");
3219        let data: Vec<i32> = (0..20).collect(); // 4 x 5
3220        {
3221            let file = H5File::create(&path).unwrap();
3222            let ds = file
3223                .new_dataset::<i32>()
3224                .shape([4, 5])
3225                .create("mat")
3226                .unwrap();
3227            ds.write_raw(&data).unwrap();
3228            file.close().unwrap();
3229        }
3230        {
3231            let file = H5File::open(&path).unwrap();
3232            let ds = file.dataset("mat").unwrap();
3233
3234            // Too small / too large full-read buffers are both rejected.
3235            let mut small = vec![0i32; 19];
3236            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
3237            let mut large = vec![0i32; 21];
3238            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
3239
3240            // Slice buffer must be exactly product(counts) = 4.
3241            let mut bad_slice = vec![0i32; 3];
3242            assert!(ds
3243                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
3244                .is_err());
3245            // The correctly sized slice buffer succeeds.
3246            let mut ok_slice = vec![0i32; 4];
3247            assert!(ds
3248                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
3249                .is_ok());
3250        }
3251        std::fs::remove_file(&path).ok();
3252    }
3253
3254    #[test]
3255    fn read_into_wrong_element_size_rejected() {
3256        let path = temp_path("into_badtype");
3257        let data: Vec<i32> = (0..20).collect(); // element size 4
3258        {
3259            let file = H5File::create(&path).unwrap();
3260            let ds = file
3261                .new_dataset::<i32>()
3262                .shape([4, 5])
3263                .create("mat")
3264                .unwrap();
3265            ds.write_raw(&data).unwrap();
3266            file.close().unwrap();
3267        }
3268        {
3269            let file = H5File::open(&path).unwrap();
3270            let ds = file.dataset("mat").unwrap();
3271            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
3272            // element size -> TypeMismatch, even with a "correctly sized" Vec.
3273            let mut as_u8 = vec![0u8; 20];
3274            assert!(matches!(
3275                ds.read_raw_into::<u8>(&mut as_u8),
3276                Err(crate::Hdf5Error::TypeMismatch(_))
3277            ));
3278            let mut as_i64 = vec![0i64; 20];
3279            assert!(matches!(
3280                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
3281                Err(crate::Hdf5Error::TypeMismatch(_))
3282            ));
3283        }
3284        std::fs::remove_file(&path).ok();
3285    }
3286
3287    #[test]
3288    fn write_slice_2d() {
3289        let path = temp_path("write_slice_2d");
3290
3291        {
3292            let file = H5File::create(&path).unwrap();
3293            let ds = file
3294                .new_dataset::<f32>()
3295                .shape([3, 4])
3296                .create("data")
3297                .unwrap();
3298            ds.write_raw(&[0.0f32; 12]).unwrap();
3299            // Overwrite a 2x2 sub-region
3300            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
3301                .unwrap();
3302            file.close().unwrap();
3303        }
3304        {
3305            let file = H5File::open(&path).unwrap();
3306            let ds = file.dataset("data").unwrap();
3307            let full = ds.read_raw::<f32>().unwrap();
3308            // Row 0: [0,0,0,0]
3309            // Row 1: [0,10,20,0]
3310            // Row 2: [0,30,40,0]
3311            assert_eq!(
3312                full,
3313                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
3314            );
3315        }
3316
3317        std::fs::remove_file(&path).ok();
3318    }
3319
3320    /// One 2x4 i32 chunk whose every element is `v`.
3321    fn chunk_of(v: i32) -> Vec<u8> {
3322        (0..8).flat_map(|_| v.to_le_bytes()).collect()
3323    }
3324
3325    /// Write chunk (0,0) `rewrites` times — each time with a different value,
3326    /// so no write can be skipped — and return the closed file's size along
3327    /// with what the chunk reads back as.
3328    fn rewrite_chunk(
3329        tag: &str,
3330        rewrites: i32,
3331        build: impl Fn(&H5File) -> crate::H5Dataset,
3332    ) -> (u64, i32) {
3333        let path = temp_path(tag);
3334        {
3335            let file = H5File::create(&path).unwrap();
3336            let ds = build(&file);
3337            for v in 1..=rewrites {
3338                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
3339            }
3340            file.close().unwrap();
3341        }
3342        let size = std::fs::metadata(&path).unwrap().len();
3343        let first = {
3344            let file = H5File::open(&path).unwrap();
3345            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
3346        };
3347        std::fs::remove_file(&path).ok();
3348        (size, first)
3349    }
3350
3351    // An unfiltered chunk's stored size is fixed by the chunk shape, so
3352    // rewriting it must overwrite the block it already occupies rather than
3353    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
3354    // leaves must_alloc false for exactly this case). The file must therefore
3355    // be byte-identical in size no matter how many times the chunk is written.
3356    #[test]
3357    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
3358        let build = |f: &H5File| {
3359            f.new_dataset::<i32>()
3360                .shape([2, 4])
3361                .chunk(&[2, 4])
3362                .max_shape(&[None, Some(4)])
3363                .create("d")
3364                .unwrap()
3365        };
3366        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
3367        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
3368        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3369        assert_eq!(last, 8, "the last write must be the one that survives");
3370    }
3371
3372    #[test]
3373    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
3374        let build = |f: &H5File| {
3375            f.new_dataset::<i32>()
3376                .shape([2, 4])
3377                .chunk(&[2, 4])
3378                .create("d")
3379                .unwrap()
3380        };
3381        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
3382        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
3383        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3384        assert_eq!(last, 8);
3385    }
3386
3387    #[test]
3388    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
3389        let build = |f: &H5File| {
3390            f.new_dataset::<i32>()
3391                .shape([2, 4])
3392                .chunk(&[2, 4])
3393                .max_shape(&[None, None])
3394                .create("d")
3395                .unwrap()
3396        };
3397        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
3398        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
3399        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3400        assert_eq!(last, 8);
3401    }
3402
3403    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
3404    // pool. Every node is the same size, so the blocks already on disk are
3405    // reused and repeated flushes cost nothing; sizing the root to its record
3406    // count instead would relocate it each time and orphan the block it left.
3407    #[test]
3408    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
3409        let flush_n = |label: &str, flushes: usize| -> u64 {
3410            let path = temp_path(label);
3411            {
3412                let file = H5File::create(&path).unwrap();
3413                let ds = file
3414                    .new_dataset::<i32>()
3415                    .shape([2, 4])
3416                    .chunk(&[2, 4])
3417                    .max_shape(&[None, None])
3418                    .create("d")
3419                    .unwrap();
3420                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
3421                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3422                for _ in 0..flushes {
3423                    ds.flush().unwrap();
3424                }
3425                file.close().unwrap();
3426            }
3427            let size = std::fs::metadata(&path).unwrap().len();
3428            // The data must survive every rewrite of the index.
3429            {
3430                let file = H5File::open(&path).unwrap();
3431                let ds = file.dataset("d").unwrap();
3432                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
3433            }
3434            std::fs::remove_file(&path).ok();
3435            size
3436        };
3437        assert_eq!(
3438            flush_n("bt2_flush_8", 8),
3439            flush_n("bt2_flush_1", 1),
3440            "8 index flushes grew the file past a single one"
3441        );
3442    }
3443
3444    // A filtered chunk whose compressed size changes cannot stay put, so it
3445    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
3446    // H5MF_xfree). Alternating between two payloads of different compressed
3447    // size must therefore keep reusing the same two blocks instead of
3448    // appending a fresh one each time.
3449    #[cfg(feature = "deflate")]
3450    #[test]
3451    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
3452        // All-equal elements deflate to far fewer bytes than a varied payload,
3453        // so the two writes below land at different stored sizes.
3454        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
3455        let varied: Vec<u8> = (0..8i32)
3456            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
3457            .collect();
3458
3459        let sizes: Vec<u64> = [1usize, 8]
3460            .iter()
3461            .map(|&rounds| {
3462                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
3463                {
3464                    let file = H5File::create(&path).unwrap();
3465                    let ds = file
3466                        .new_dataset::<i32>()
3467                        .shape([2, 4])
3468                        .chunk(&[2, 4])
3469                        .max_shape(&[None, Some(4)])
3470                        .deflate(6)
3471                        .create("d")
3472                        .unwrap();
3473                    for _ in 0..rounds {
3474                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
3475                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
3476                    }
3477                    file.close().unwrap();
3478                }
3479                let size = std::fs::metadata(&path).unwrap().len();
3480                {
3481                    let file = H5File::open(&path).unwrap();
3482                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
3483                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
3484                    assert_eq!(got, want, "the last write must survive the round trip");
3485                }
3486                std::fs::remove_file(&path).ok();
3487                size
3488            })
3489            .collect();
3490
3491        assert_eq!(
3492            sizes[1], sizes[0],
3493            "8 alternating rewrites grew the file past a single pair"
3494        );
3495    }
3496
3497    #[test]
3498    fn write_slice_out_of_bounds_rejected() {
3499        let path = temp_path("write_slice_oob");
3500        let file = H5File::create(&path).unwrap();
3501        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
3502        ds.write_raw(&[0i32; 4]).unwrap();
3503        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
3504        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
3505        // An in-bounds slice still works.
3506        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
3507        std::fs::remove_file(&path).ok();
3508    }
3509
3510    #[test]
3511    fn duplicate_dataset_name_rejected() {
3512        let path = temp_path("dup_name");
3513        let file = H5File::create(&path).unwrap();
3514        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
3515        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
3516        std::fs::remove_file(&path).ok();
3517    }
3518
3519    #[test]
3520    fn extend_cannot_shrink() {
3521        let path = temp_path("extend_shrink");
3522        let file = H5File::create(&path).unwrap();
3523        let ds = file
3524            .new_dataset::<i32>()
3525            .shape([0])
3526            .chunk(&[2])
3527            .max_shape(&[None])
3528            .create("d")
3529            .unwrap();
3530        ds.append(&[1i32, 2, 3, 4]).unwrap();
3531        // Shrinking below the written extent must be rejected.
3532        assert!(ds.extend(&[2]).is_err());
3533        // Growing is fine.
3534        assert!(ds.extend(&[6]).is_ok());
3535        std::fs::remove_file(&path).ok();
3536    }
3537
3538    #[test]
3539    fn attr_read_roundtrip() {
3540        use crate::types::VarLenUnicode;
3541        let path = temp_path("attr_read");
3542
3543        {
3544            let file = H5File::create(&path).unwrap();
3545            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
3546            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3547            let a1 = ds
3548                .new_attr::<VarLenUnicode>()
3549                .shape(())
3550                .create("units")
3551                .unwrap();
3552            a1.write_string("meters").unwrap();
3553            let a2 = ds
3554                .new_attr::<VarLenUnicode>()
3555                .shape(())
3556                .create("desc")
3557                .unwrap();
3558            a2.write_string("test data").unwrap();
3559            file.close().unwrap();
3560        }
3561        {
3562            let file = H5File::open(&path).unwrap();
3563            let ds = file.dataset("data").unwrap();
3564
3565            let names = ds.attr_names().unwrap();
3566            assert!(names.contains(&"units".to_string()));
3567            assert!(names.contains(&"desc".to_string()));
3568
3569            let units = ds.attr("units").unwrap();
3570            assert_eq!(units.read_string().unwrap(), "meters");
3571
3572            let desc = ds.attr("desc").unwrap();
3573            assert_eq!(desc.read_string().unwrap(), "test data");
3574        }
3575
3576        std::fs::remove_file(&path).ok();
3577    }
3578
3579    #[test]
3580    fn type_mismatch_element_size() {
3581        let path = temp_path("type_mismatch");
3582
3583        {
3584            let file = H5File::create(&path).unwrap();
3585            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
3586            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
3587            file.close().unwrap();
3588        }
3589
3590        {
3591            let file = H5File::open(&path).unwrap();
3592            let ds = file.dataset("data").unwrap();
3593            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
3594            let result = ds.read_raw::<u8>();
3595            assert!(result.is_err());
3596        }
3597
3598        std::fs::remove_file(&path).ok();
3599    }
3600
3601    #[test]
3602    fn dataset_survives_file_move() {
3603        let path = temp_path("ds_survives");
3604
3605        let ds = {
3606            let file = H5File::create(&path).unwrap();
3607            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
3608        };
3609        // file is dropped here, but ds still holds Rc to the inner state
3610        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3611        // The writer will finalize on drop of the last Rc
3612
3613        std::fs::remove_file(&path).ok();
3614    }
3615
3616    #[test]
3617    fn new_attr_scalar_string() {
3618        use crate::types::VarLenUnicode;
3619
3620        let path = temp_path("attr_scalar_string");
3621        {
3622            let file = H5File::create(&path).unwrap();
3623            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
3624            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3625
3626            let attr = ds
3627                .new_attr::<VarLenUnicode>()
3628                .shape(())
3629                .create("name")
3630                .unwrap();
3631            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
3632                .unwrap();
3633
3634            file.close().unwrap();
3635        }
3636
3637        // Verify the file is still valid and readable
3638        {
3639            use crate::format::messages::datatype::DatatypeMessage;
3640            let file = H5File::open(&path).unwrap();
3641            let ds = file.dataset("data").unwrap();
3642            assert_eq!(ds.shape(), vec![4]);
3643            let readback = ds.read_raw::<u8>().unwrap();
3644            assert_eq!(readback, vec![1u8, 2, 3, 4]);
3645
3646            // The string attribute is stored as a true variable-length string
3647            // (not fixed-length) and round-trips its value.
3648            let attr = ds.attr("name").unwrap();
3649            assert!(
3650                matches!(
3651                    attr.datatype().unwrap(),
3652                    DatatypeMessage::VarLenString { .. }
3653                ),
3654                "string attribute should have a variable-length string datatype"
3655            );
3656            assert_eq!(attr.read_string().unwrap(), "test_value");
3657        }
3658
3659        std::fs::remove_file(&path).ok();
3660    }
3661
3662    #[test]
3663    fn all_numeric_types_roundtrip() {
3664        let path = temp_path("all_types");
3665
3666        {
3667            let file = H5File::create(&path).unwrap();
3668
3669            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
3670            ds.write_raw(&[1u8, 2]).unwrap();
3671
3672            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
3673            ds.write_raw(&[-1i8, 1]).unwrap();
3674
3675            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
3676            ds.write_raw(&[100u16, 200]).unwrap();
3677
3678            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
3679            ds.write_raw(&[-100i16, 100]).unwrap();
3680
3681            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
3682            ds.write_raw(&[1000u32, 2000]).unwrap();
3683
3684            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
3685            ds.write_raw(&[-1000i32, 1000]).unwrap();
3686
3687            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
3688            ds.write_raw(&[10000u64, 20000]).unwrap();
3689
3690            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
3691            ds.write_raw(&[-10000i64, 10000]).unwrap();
3692
3693            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
3694            ds.write_raw(&[1.5f32, 2.5]).unwrap();
3695
3696            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
3697            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
3698
3699            file.close().unwrap();
3700        }
3701
3702        {
3703            let file = H5File::open(&path).unwrap();
3704
3705            assert_eq!(
3706                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
3707                vec![1u8, 2]
3708            );
3709            assert_eq!(
3710                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
3711                vec![-1i8, 1]
3712            );
3713            assert_eq!(
3714                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
3715                vec![100u16, 200]
3716            );
3717            assert_eq!(
3718                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
3719                vec![-100i16, 100]
3720            );
3721            assert_eq!(
3722                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
3723                vec![1000u32, 2000]
3724            );
3725            assert_eq!(
3726                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
3727                vec![-1000i32, 1000]
3728            );
3729            assert_eq!(
3730                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
3731                vec![10000u64, 20000]
3732            );
3733            assert_eq!(
3734                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
3735                vec![-10000i64, 10000]
3736            );
3737            assert_eq!(
3738                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
3739                vec![1.5f32, 2.5]
3740            );
3741            assert_eq!(
3742                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
3743                vec![1.23456f64, 7.89012]
3744            );
3745        }
3746
3747        std::fs::remove_file(&path).ok();
3748    }
3749
3750    #[test]
3751    fn append_chunked_roundtrip() {
3752        let path = temp_path("append_chunked");
3753
3754        {
3755            let file = H5File::create(&path).unwrap();
3756            let ds = file
3757                .new_dataset::<f64>()
3758                .shape([0, 3])
3759                .chunk(&[1, 3])
3760                .max_shape(&[None, Some(3)])
3761                .create("data")
3762                .unwrap();
3763
3764            // Append one frame
3765            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
3766            // Append two frames at once
3767            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
3768
3769            file.close().unwrap();
3770        }
3771
3772        {
3773            let file = H5File::open(&path).unwrap();
3774            let ds = file.dataset("data").unwrap();
3775            assert_eq!(ds.shape(), vec![3, 3]);
3776            let all = ds.read_raw::<f64>().unwrap();
3777            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
3778        }
3779
3780        std::fs::remove_file(&path).ok();
3781    }
3782
3783    #[test]
3784    fn append_1d_chunked() {
3785        let path = temp_path("append_1d");
3786
3787        {
3788            let file = H5File::create(&path).unwrap();
3789            let ds = file
3790                .new_dataset::<i32>()
3791                .shape([0])
3792                .chunk(&[4])
3793                .max_shape(&[None])
3794                .create("values")
3795                .unwrap();
3796
3797            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
3798            ds.append(&[40i32]).unwrap(); // fills chunk boundary
3799            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
3800
3801            file.close().unwrap();
3802        }
3803
3804        {
3805            let file = H5File::open(&path).unwrap();
3806            let ds = file.dataset("values").unwrap();
3807            assert_eq!(ds.shape(), vec![8]);
3808            let all = ds.read_raw::<i32>().unwrap();
3809            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
3810        }
3811
3812        std::fs::remove_file(&path).ok();
3813    }
3814
3815    #[test]
3816    fn append_partial_chunk_flushed_on_close() {
3817        let path = temp_path("append_partial_close");
3818
3819        {
3820            let file = H5File::create(&path).unwrap();
3821            let ds = file
3822                .new_dataset::<f64>()
3823                .shape([0])
3824                .chunk(&[4])
3825                .max_shape(&[None])
3826                .create("vals")
3827                .unwrap();
3828
3829            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
3830            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
3831            file.close().unwrap();
3832        }
3833
3834        {
3835            let file = H5File::open(&path).unwrap();
3836            let ds = file.dataset("vals").unwrap();
3837            assert_eq!(ds.shape(), vec![5]);
3838            let all = ds.read_raw::<f64>().unwrap();
3839            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
3840            // read_raw reads total shape elements
3841            assert_eq!(all.len(), 5);
3842            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
3843        }
3844
3845        std::fs::remove_file(&path).ok();
3846    }
3847
3848    /// An append that leaves its chunk partial is buffered until close, and
3849    /// the flush has to keep the frames that chunk already holds. It built a
3850    /// fresh fill-value chunk around the buffered frame instead, so reopening
3851    /// a file and appending one row erased every earlier row of that chunk
3852    /// (issue #3). Four sessions: the second lands beside an existing row, the
3853    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
3854    /// the third left in chunk 1.
3855    #[test]
3856    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
3857        let path = temp_path("append_reopen_partial");
3858
3859        {
3860            let file = H5File::create(&path).unwrap();
3861            let ds = file
3862                .new_dataset::<i32>()
3863                .shape([0, 3])
3864                .chunk(&[4, 3])
3865                .max_shape(&[None, Some(3)])
3866                .create("values")
3867                .unwrap();
3868            ds.append(&[1, 2, 3]).unwrap();
3869            file.close().unwrap();
3870        }
3871        for rows in [
3872            vec![4, 5, 6],
3873            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
3874            vec![16, 17, 18],
3875        ] {
3876            let file = H5File::open_rw(&path).unwrap();
3877            file.dataset_writer("values")
3878                .unwrap()
3879                .append(&rows)
3880                .unwrap();
3881            file.close().unwrap();
3882        }
3883
3884        let file = H5File::open(&path).unwrap();
3885        let ds = file.dataset("values").unwrap();
3886        assert_eq!(ds.shape(), vec![6, 3]);
3887        assert_eq!(
3888            ds.read_raw::<i32>().unwrap(),
3889            (1..=18).collect::<Vec<i32>>()
3890        );
3891        std::fs::remove_file(&path).ok();
3892    }
3893
3894    #[cfg(feature = "deflate")]
3895    #[test]
3896    fn vlen_append_after_reopen_filtered() {
3897        // Reopen + append into a partially-written *compressed* vlen chunk
3898        // (index-block chunk). Exercises filtered-index-block reconstruction
3899        // in open_append plus filtered read-modify-write.
3900        let path = temp_path("vlen_reopen_filtered");
3901        {
3902            let file = H5File::create(&path).unwrap();
3903            file.create_appendable_vlen_dataset(
3904                "strs",
3905                4,
3906                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
3907            )
3908            .unwrap();
3909            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
3910                .unwrap();
3911            file.close().unwrap();
3912        }
3913        {
3914            let file = H5File::open_rw(&path).unwrap();
3915            file.append_vlen_strings("strs", &["delta"]).unwrap();
3916            file.close().unwrap();
3917        }
3918        {
3919            let file = H5File::open(&path).unwrap();
3920            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3921            assert_eq!(
3922                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
3923                vec!["alpha", "beta", "gamma", "delta"]
3924            );
3925        }
3926        std::fs::remove_file(&path).ok();
3927    }
3928
3929    #[test]
3930    fn vlen_append_after_reopen_data_block() {
3931        // Reopen + append into a partial chunk that lives in an extensible-
3932        // array *data block* (chunk index >= idx_blk_elmts). Exercises
3933        // data-block resolution in read_chunk_if_present and write_chunk.
3934        let path = temp_path("vlen_reopen_datablk");
3935        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
3936        {
3937            let file = H5File::create(&path).unwrap();
3938            file.create_appendable_vlen_dataset("strs", 2, None)
3939                .unwrap();
3940            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3941            file.append_vlen_strings("strs", &refs).unwrap();
3942            file.close().unwrap();
3943        }
3944        {
3945            let file = H5File::open_rw(&path).unwrap();
3946            file.append_vlen_strings("strs", &["s9"]).unwrap();
3947            file.close().unwrap();
3948        }
3949        {
3950            let file = H5File::open(&path).unwrap();
3951            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3952            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
3953            assert_eq!(got, want);
3954        }
3955        std::fs::remove_file(&path).ok();
3956    }
3957
3958    #[test]
3959    fn vlen_append_after_reopen_super_block() {
3960        // Reopen + append into a partial chunk whose index falls in an
3961        // extensible-array *super block* (chunk index 244 with the default
3962        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
3963        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
3964        // index block or its direct data blocks, so chunk 244 is reached
3965        // via a super block read from disk). Exercises the ViaSblk branch
3966        // of read_chunk_if_present.
3967        let path = temp_path("vlen_reopen_super");
3968        // 489 strings, chunk size 2 -> chunk 244 holds one string only
3969        // (partially filled) and is flushed to disk on close.
3970        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
3971        {
3972            let file = H5File::create(&path).unwrap();
3973            file.create_appendable_vlen_dataset("strs", 2, None)
3974                .unwrap();
3975            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3976            file.append_vlen_strings("strs", &refs).unwrap();
3977            file.close().unwrap();
3978        }
3979        {
3980            let file = H5File::open_rw(&path).unwrap();
3981            file.append_vlen_strings("strs", &["v489"]).unwrap();
3982            file.close().unwrap();
3983        }
3984        {
3985            let file = H5File::open(&path).unwrap();
3986            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3987            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
3988            assert_eq!(got, want);
3989        }
3990        std::fs::remove_file(&path).ok();
3991    }
3992
3993    #[cfg(feature = "deflate")]
3994    #[test]
3995    fn vlen_append_after_reopen_filtered_data_block() {
3996        // The hardest path: compressed + chunk in a data block + partial
3997        // read-modify-write across a reopen.
3998        let path = temp_path("vlen_reopen_filt_datablk");
3999        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
4000        {
4001            let file = H5File::create(&path).unwrap();
4002            file.create_appendable_vlen_dataset(
4003                "strs",
4004                2,
4005                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
4006            )
4007            .unwrap();
4008            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
4009            file.append_vlen_strings("strs", &refs).unwrap();
4010            file.close().unwrap();
4011        }
4012        {
4013            let file = H5File::open_rw(&path).unwrap();
4014            file.append_vlen_strings("strs", &["item09"]).unwrap();
4015            file.close().unwrap();
4016        }
4017        {
4018            let file = H5File::open(&path).unwrap();
4019            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
4020            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
4021            assert_eq!(got, want);
4022        }
4023        std::fs::remove_file(&path).ok();
4024    }
4025
4026    #[test]
4027    fn group_nx_class_attribute_roundtrip() {
4028        // Non-root groups carry attributes (NeXus `NX_class`) in their
4029        // own object header, and the reader reads them back by path.
4030        let path = temp_path("group_nx_class");
4031        {
4032            let file = H5File::create(&path).unwrap();
4033            let entry = file.create_group("entry").unwrap();
4034            entry.set_attr_string("NX_class", "NXentry").unwrap();
4035            let det = entry.create_group("detector").unwrap();
4036            det.set_attr_string("NX_class", "NXdetector").unwrap();
4037            det.set_attr_numeric("frame_count", &7i32).unwrap();
4038            det.new_dataset::<f32>()
4039                .shape([4])
4040                .create("data")
4041                .unwrap()
4042                .write_raw(&[1.0f32; 4])
4043                .unwrap();
4044            file.close().unwrap();
4045        }
4046        {
4047            let file = H5File::open(&path).unwrap();
4048            let entry = file.root_group().group("entry").unwrap();
4049            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
4050            let det = entry.group("detector").unwrap();
4051            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
4052            let names = det.attr_names().unwrap();
4053            assert!(names.contains(&"NX_class".to_string()));
4054            assert!(names.contains(&"frame_count".to_string()));
4055        }
4056        std::fs::remove_file(&path).ok();
4057    }
4058
4059    #[test]
4060    fn ea_super_block_roundtrip() {
4061        // 2000 chunks span several extensible-array super blocks. Before
4062        // super-block support the writer errored at chunk index 228.
4063        let path = temp_path("ea_super_rt");
4064        {
4065            let file = H5File::create(&path).unwrap();
4066            let ds = file
4067                .new_dataset::<i32>()
4068                .shape([0])
4069                .chunk(&[1])
4070                .max_shape(&[None])
4071                .create("v")
4072                .unwrap();
4073            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
4074            file.close().unwrap();
4075        }
4076        {
4077            let file = H5File::open(&path).unwrap();
4078            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4079            assert_eq!(v.len(), 2000);
4080            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
4081        }
4082        std::fs::remove_file(&path).ok();
4083    }
4084
4085    #[cfg(feature = "deflate")]
4086    #[test]
4087    fn ea_filtered_super_block_roundtrip() {
4088        // Compressed chunks across super blocks.
4089        let path = temp_path("ea_filt_super");
4090        {
4091            let file = H5File::create(&path).unwrap();
4092            let ds = file
4093                .new_dataset::<i32>()
4094                .shape([0])
4095                .chunk(&[1])
4096                .max_shape(&[None])
4097                .deflate(4)
4098                .create("v")
4099                .unwrap();
4100            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
4101            file.close().unwrap();
4102        }
4103        {
4104            let file = H5File::open(&path).unwrap();
4105            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4106            assert_eq!(v, (0..600).collect::<Vec<i32>>());
4107        }
4108        std::fs::remove_file(&path).ok();
4109    }
4110
4111    #[test]
4112    fn ea_super_block_open_append() {
4113        // Reopen a dataset and append chunks that fall in super blocks.
4114        let path = temp_path("ea_super_append");
4115        {
4116            let file = H5File::create(&path).unwrap();
4117            let ds = file
4118                .new_dataset::<i32>()
4119                .shape([0])
4120                .chunk(&[1])
4121                .max_shape(&[None])
4122                .create("v")
4123                .unwrap();
4124            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
4125            file.close().unwrap();
4126        }
4127        {
4128            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
4129            let idx = w.dataset_index("v").unwrap();
4130            for c in 300..900u64 {
4131                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
4132            }
4133            w.extend_dataset(idx, &[900]).unwrap();
4134            w.close().unwrap();
4135        }
4136        {
4137            let file = H5File::open(&path).unwrap();
4138            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4139            assert_eq!(v.len(), 900);
4140            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
4141        }
4142        std::fs::remove_file(&path).ok();
4143    }
4144
4145    // Two or more unlimited dimensions select the v2 B-tree index; with a
4146    // filter its records become type 11, carrying each chunk's stored size and
4147    // mask. The payload is highly compressible, so the chunks really are
4148    // stored smaller than the extent — the file would be at least
4149    // 6*8*4 = 192 bytes of raw chunk data otherwise.
4150    #[cfg(feature = "deflate")]
4151    #[test]
4152    fn compressed_multi_unlimited_dataset_roundtrips() {
4153        let path = temp_path("bt2_filtered");
4154        {
4155            let file = H5File::create(&path).unwrap();
4156            let ds = file
4157                .new_dataset::<i32>()
4158                .shape([6, 8])
4159                .chunk(&[2, 4])
4160                .max_shape(&[None, None])
4161                .deflate(6)
4162                .create("d")
4163                .unwrap();
4164            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
4165            // A partial write forces a decompress-patch-recompress of one
4166            // chunk, whose new compressed size may not fit its old block.
4167            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
4168            file.close().unwrap();
4169        }
4170        {
4171            let file = H5File::open(&path).unwrap();
4172            let ds = file.dataset("d").unwrap();
4173            assert_eq!(ds.shape(), vec![6, 8]);
4174            let mut want = vec![7i32; 48];
4175            want[9] = 1;
4176            want[10] = 2;
4177            want[17] = 3;
4178            want[18] = 4;
4179            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
4180        }
4181        std::fs::remove_file(&path).ok();
4182    }
4183
4184    #[test]
4185    fn btree_v2_multi_unlimited_roundtrip() {
4186        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
4187        // index; chunks are written by grid coordinates with write_chunk_at.
4188        let path = temp_path("bt2_multi");
4189        {
4190            let file = H5File::create(&path).unwrap();
4191            let ds = file
4192                .new_dataset::<i32>()
4193                .shape([0, 0])
4194                .chunk(&[2, 2])
4195                .max_shape(&[None, None])
4196                .create("grid")
4197                .unwrap();
4198            assert!(ds.is_chunked());
4199            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
4200            for cr in 0..2usize {
4201                for cc in 0..2usize {
4202                    let mut bytes = Vec::new();
4203                    for i in 0..2usize {
4204                        for j in 0..2usize {
4205                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
4206                            bytes.extend_from_slice(&v.to_le_bytes());
4207                        }
4208                    }
4209                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
4210                }
4211            }
4212            file.close().unwrap();
4213        }
4214        {
4215            let file = H5File::open(&path).unwrap();
4216            let ds = file.dataset("grid").unwrap();
4217            assert_eq!(ds.shape(), vec![4, 4]);
4218            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
4219        }
4220        std::fs::remove_file(&path).ok();
4221    }
4222
4223    #[test]
4224    fn subframe_chunking_roundtrip() {
4225        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
4226        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
4227        // the chunk-grid coordinates.
4228        let path = temp_path("subframe");
4229        {
4230            let file = H5File::create(&path).unwrap();
4231            let ds = file
4232                .new_dataset::<i32>()
4233                .shape([0, 8, 8])
4234                .chunk(&[1, 4, 4])
4235                .max_shape(&[None, Some(8), Some(8)])
4236                .create("v")
4237                .unwrap();
4238            for f in 0..3usize {
4239                for cr in 0..2usize {
4240                    for cc in 0..2usize {
4241                        let mut bytes = Vec::new();
4242                        for i in 0..4usize {
4243                            for j in 0..4usize {
4244                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
4245                                bytes.extend_from_slice(&v.to_le_bytes());
4246                            }
4247                        }
4248                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
4249                    }
4250                }
4251            }
4252            file.close().unwrap();
4253        }
4254        {
4255            let file = H5File::open(&path).unwrap();
4256            let ds = file.dataset("v").unwrap();
4257            assert_eq!(ds.shape(), vec![3, 8, 8]);
4258            assert_eq!(
4259                ds.read_raw::<i32>().unwrap(),
4260                (0..192).collect::<Vec<i32>>()
4261            );
4262        }
4263        std::fs::remove_file(&path).ok();
4264    }
4265
4266    #[test]
4267    fn fill_value_contiguous_roundtrip() {
4268        let path = temp_path("fill_value_contig");
4269        {
4270            let file = H5File::create(&path).unwrap();
4271            let ds = file
4272                .new_dataset::<f32>()
4273                .shape([4])
4274                .fill_value(2.5f32)
4275                .create("data")
4276                .unwrap();
4277            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
4278            file.close().unwrap();
4279        }
4280        // open_append decodes the fill-value message back from the header.
4281        {
4282            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
4283            let idx = writer.dataset_index("data").unwrap();
4284            assert_eq!(
4285                writer.ds(idx).lock().fill_value,
4286                Some(2.5f32.to_le_bytes().to_vec())
4287            );
4288        }
4289        // Data still reads back correctly.
4290        {
4291            let file = H5File::open(&path).unwrap();
4292            let ds = file.dataset("data").unwrap();
4293            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
4294        }
4295        std::fs::remove_file(&path).ok();
4296    }
4297
4298    #[test]
4299    fn fill_value_chunked_roundtrip() {
4300        let path = temp_path("fill_value_chunked");
4301        {
4302            let file = H5File::create(&path).unwrap();
4303            let ds = file
4304                .new_dataset::<i32>()
4305                .shape([0])
4306                .chunk(&[4])
4307                .max_shape(&[None])
4308                .fill_value(-7i32)
4309                .create("vals")
4310                .unwrap();
4311            ds.append(&[1i32, 2, 3, 4]).unwrap();
4312            file.close().unwrap();
4313        }
4314        {
4315            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
4316            let idx = writer.dataset_index("vals").unwrap();
4317            assert_eq!(
4318                writer.ds(idx).lock().fill_value,
4319                Some((-7i32).to_le_bytes().to_vec())
4320            );
4321        }
4322        std::fs::remove_file(&path).ok();
4323    }
4324
4325    #[test]
4326    fn fill_value_read_missing_chunks() {
4327        // A chunked dataset with chunk 1 left unwritten must read that
4328        // gap back as the user-defined fill value, not zero.
4329        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
4330            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
4331        }
4332        let path = temp_path("fill_value_read_missing");
4333        {
4334            let file = H5File::create(&path).unwrap();
4335            let ds = file
4336                .new_dataset::<i32>()
4337                .shape([0])
4338                .chunk(&[2])
4339                .max_shape(&[None])
4340                .fill_value(-1i32)
4341                .create("vals")
4342                .unwrap();
4343            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
4344            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
4345            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
4346            ds.extend(&[6]).unwrap();
4347            file.close().unwrap();
4348        }
4349        {
4350            let file = H5File::open(&path).unwrap();
4351            let ds = file.dataset("vals").unwrap();
4352            let all = ds.read_raw::<i32>().unwrap();
4353            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
4354        }
4355        std::fs::remove_file(&path).ok();
4356    }
4357
4358    #[test]
4359    fn fill_value_partial_chunk_padded_with_fill() {
4360        // A partial trailing chunk flushed at close must pad its unwritten
4361        // tail with the fill value. That pad sits beyond the logical shape,
4362        // so it is verified by scanning the on-disk chunk bytes directly.
4363        let path = temp_path("fill_value_partial_pad");
4364        {
4365            let file = H5File::create(&path).unwrap();
4366            let ds = file
4367                .new_dataset::<i32>()
4368                .shape([0])
4369                .chunk(&[4])
4370                .max_shape(&[None])
4371                .fill_value(-9i32)
4372                .create("vals")
4373                .unwrap();
4374            // 3 of 4 frames -> flushed as a partial chunk on close.
4375            ds.append(&[1i32, 2, 3]).unwrap();
4376            file.close().unwrap();
4377        }
4378        let bytes = std::fs::read(&path).unwrap();
4379        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
4380        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
4381        let pos = bytes
4382            .windows(needle.len())
4383            .position(|w| w == needle)
4384            .expect("chunk data [1,2,3] not found in file");
4385        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
4386        assert_eq!(
4387            pad,
4388            &(-9i32).to_le_bytes(),
4389            "partial chunk tail must be padded with fill value -9, got {:?}",
4390            pad
4391        );
4392        std::fs::remove_file(&path).ok();
4393    }
4394
4395    #[test]
4396    fn vlen_append_after_reopen_preserves_existing() {
4397        // Reopening and appending into a partially-written vlen chunk must
4398        // read-modify-write: the strings already on disk must survive.
4399        let path = temp_path("vlen_append_reopen");
4400        {
4401            let file = H5File::create(&path).unwrap();
4402            file.create_appendable_vlen_dataset("strs", 4, None)
4403                .unwrap();
4404            // 3 of 4 frames -> flushed as a partial chunk on close.
4405            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
4406            file.close().unwrap();
4407        }
4408        {
4409            // Append a 4th string -> partial-chunk write into chunk 0.
4410            let file = H5File::open_rw(&path).unwrap();
4411            file.append_vlen_strings("strs", &["d"]).unwrap();
4412            file.close().unwrap();
4413        }
4414        {
4415            let file = H5File::open(&path).unwrap();
4416            let ds = file.dataset("strs").unwrap();
4417            let got = ds.read_vlen_strings().unwrap();
4418            assert_eq!(
4419                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
4420                vec!["a", "b", "c", "d"]
4421            );
4422        }
4423        std::fs::remove_file(&path).ok();
4424    }
4425
4426    #[test]
4427    fn fill_value_size_mismatch_errors() {
4428        let path = temp_path("fill_value_mismatch");
4429        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
4430        let dt = <f64 as crate::types::H5Type>::hdf5_type();
4431        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
4432        // f64 element size is 8; a 4-byte fill value must be rejected.
4433        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
4434        // The correct width succeeds.
4435        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
4436        writer.close().unwrap();
4437        std::fs::remove_file(&path).ok();
4438    }
4439
4440    #[test]
4441    fn datatype_exposes_class_sign_and_byteorder() {
4442        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
4443        // from f32 (both 4 bytes). datatype() must report the real class and
4444        // signedness so a reader does not have to guess from element_size.
4445        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
4446
4447        let path = temp_path("datatype_accessor");
4448        {
4449            let file = H5File::create(&path).unwrap();
4450            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
4451            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
4452            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
4453            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
4454            file.close().unwrap();
4455        }
4456
4457        let file = H5File::open(&path).unwrap();
4458
4459        match file.dataset("u8d").unwrap().datatype().unwrap() {
4460            DatatypeMessage::FixedPoint {
4461                size,
4462                signed,
4463                byte_order,
4464                ..
4465            } => {
4466                assert_eq!(size, 1);
4467                assert!(!signed, "u8 must be unsigned");
4468                assert_eq!(byte_order, ByteOrder::LittleEndian);
4469            }
4470            other => panic!("expected FixedPoint for u8, got {other:?}"),
4471        }
4472
4473        match file.dataset("i8d").unwrap().datatype().unwrap() {
4474            DatatypeMessage::FixedPoint { size, signed, .. } => {
4475                assert_eq!(size, 1);
4476                assert!(signed, "i8 must be signed");
4477            }
4478            other => panic!("expected FixedPoint for i8, got {other:?}"),
4479        }
4480
4481        match file.dataset("i32d").unwrap().datatype().unwrap() {
4482            DatatypeMessage::FixedPoint { size, signed, .. } => {
4483                assert_eq!(size, 4);
4484                assert!(signed, "i32 must be signed");
4485            }
4486            other => panic!("expected FixedPoint for i32, got {other:?}"),
4487        }
4488
4489        match file.dataset("f32d").unwrap().datatype().unwrap() {
4490            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
4491            other => panic!("expected FloatingPoint for f32, got {other:?}"),
4492        }
4493
4494        std::fs::remove_file(&path).ok();
4495    }
4496
4497    #[test]
4498    fn datatype_in_write_mode_errors() {
4499        let path = temp_path("datatype_write_mode");
4500        let file = H5File::create(&path).unwrap();
4501        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
4502        assert!(ds.datatype().is_err());
4503        std::fs::remove_file(&path).ok();
4504    }
4505
4506    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
4507
4508    /// Extensible-array path: pre-compress with the dataset's pipeline, write
4509    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
4510    /// the data round-trips through the reader unchanged.
4511    #[cfg(feature = "deflate")]
4512    #[test]
4513    fn write_chunk_raw_ea_roundtrip_mask0() {
4514        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4515        let path = temp_path("wcr_ea_mask0");
4516        let original: Vec<i32> = (0..12).collect();
4517        {
4518            let file = H5File::create(&path).unwrap();
4519            let ds = file
4520                .new_dataset::<i32>()
4521                .shape([0])
4522                .chunk(&[4])
4523                .max_shape(&[None])
4524                .deflate(4)
4525                .create("v")
4526                .unwrap();
4527            assert!(ds.is_chunked());
4528            let pipeline = FilterPipeline::deflate(4);
4529            for c in 0..3usize {
4530                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
4531                    .iter()
4532                    .flat_map(|v| v.to_le_bytes())
4533                    .collect();
4534                let compressed = apply_filters(&pipeline, &raw).unwrap();
4535                ds.write_chunk_raw(c, &compressed, 0).unwrap();
4536            }
4537            ds.set_extent(&[12]).unwrap();
4538            file.close().unwrap();
4539        }
4540        {
4541            let file = H5File::open(&path).unwrap();
4542            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4543            assert_eq!(v, original);
4544        }
4545        std::fs::remove_file(&path).ok();
4546    }
4547
4548    /// Fixed-array path (all dimensions bounded): same verbatim write through
4549    /// the linear-index dispatch, round-tripped through the reader.
4550    #[cfg(feature = "deflate")]
4551    #[test]
4552    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
4553        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4554        let path = temp_path("wcr_fa_mask0");
4555        let original: Vec<i32> = (0..12).collect();
4556        {
4557            let file = H5File::create(&path).unwrap();
4558            let ds = file
4559                .new_dataset::<i32>()
4560                .shape([12])
4561                .chunk(&[4])
4562                .deflate(4)
4563                .create("v")
4564                .unwrap();
4565            assert!(ds.is_chunked());
4566            let pipeline = FilterPipeline::deflate(4);
4567            for c in 0..3usize {
4568                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
4569                    .iter()
4570                    .flat_map(|v| v.to_le_bytes())
4571                    .collect();
4572                let compressed = apply_filters(&pipeline, &raw).unwrap();
4573                ds.write_chunk_raw(c, &compressed, 0).unwrap();
4574            }
4575            file.close().unwrap();
4576        }
4577        {
4578            let file = H5File::open(&path).unwrap();
4579            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4580            assert_eq!(v, original);
4581        }
4582        std::fs::remove_file(&path).ok();
4583    }
4584
4585    /// The caller-supplied filter_mask must reach the on-disk filtered index
4586    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
4587    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
4588    /// the extensible-array filtered entry to read the mask back at the format
4589    /// level (independent of the data reader's mask handling).
4590    #[cfg(feature = "deflate")]
4591    #[test]
4592    fn write_chunk_raw_records_filter_mask() {
4593        let path = temp_path("wcr_records_mask");
4594        let raw: Vec<u8> = [10i32, 20, 30, 40]
4595            .iter()
4596            .flat_map(|v| v.to_le_bytes())
4597            .collect();
4598        assert_eq!(raw.len(), 16);
4599        {
4600            let file = H5File::create(&path).unwrap();
4601            let ds = file
4602                .new_dataset::<i32>()
4603                .shape([0])
4604                .chunk(&[4])
4605                .max_shape(&[None])
4606                .deflate(4)
4607                .create("v")
4608                .unwrap();
4609            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
4610            // chunk is stored uncompressed (its raw bytes).
4611            ds.write_chunk_raw(0, &raw, 1).unwrap();
4612            ds.set_extent(&[4]).unwrap();
4613            file.close().unwrap();
4614        }
4615        // Reopen the writer; open_append decodes the filtered index block from
4616        // disk, so the entry reflects exactly what was committed.
4617        {
4618            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
4619            let idx = w.dataset_index("v").unwrap();
4620            let ds = w.ds(idx);
4621            let m = ds.lock();
4622            let entry = &m
4623                .chunked
4624                .as_ref()
4625                .unwrap()
4626                .filt_iblk
4627                .as_ref()
4628                .unwrap()
4629                .elements[0];
4630            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
4631            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
4632        }
4633        std::fs::remove_file(&path).ok();
4634    }
4635
4636    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
4637    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
4638    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
4639    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
4640    #[cfg(feature = "deflate")]
4641    #[test]
4642    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
4643        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4644        let path = temp_path("wcr_ea_per_chunk_mask");
4645        let original: Vec<i32> = (0..8).collect();
4646        let pipeline = FilterPipeline::deflate(4);
4647        {
4648            let file = H5File::create(&path).unwrap();
4649            let ds = file
4650                .new_dataset::<i32>()
4651                .shape([0])
4652                .chunk(&[4])
4653                .max_shape(&[None])
4654                .deflate(4)
4655                .create("v")
4656                .unwrap();
4657            let raw0: Vec<u8> = original[0..4]
4658                .iter()
4659                .flat_map(|v| v.to_le_bytes())
4660                .collect();
4661            // chunk 0: compressed through the pipeline, mask 0.
4662            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
4663                .unwrap();
4664            let raw1: Vec<u8> = original[4..8]
4665                .iter()
4666                .flat_map(|v| v.to_le_bytes())
4667                .collect();
4668            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
4669            ds.write_chunk_raw(1, &raw1, 1).unwrap();
4670            ds.set_extent(&[8]).unwrap();
4671            file.close().unwrap();
4672        }
4673        {
4674            let file = H5File::open(&path).unwrap();
4675            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4676            assert_eq!(v, original);
4677        }
4678        std::fs::remove_file(&path).ok();
4679    }
4680
4681    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
4682    /// compressed/raw chunks as the EA case, through the fixed-array index.
4683    #[cfg(feature = "deflate")]
4684    #[test]
4685    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
4686        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4687        let path = temp_path("wcr_fa_per_chunk_mask");
4688        let original: Vec<i32> = (0..8).collect();
4689        let pipeline = FilterPipeline::deflate(4);
4690        {
4691            let file = H5File::create(&path).unwrap();
4692            let ds = file
4693                .new_dataset::<i32>()
4694                .shape([8])
4695                .chunk(&[4])
4696                .deflate(4)
4697                .create("v")
4698                .unwrap();
4699            let raw0: Vec<u8> = original[0..4]
4700                .iter()
4701                .flat_map(|v| v.to_le_bytes())
4702                .collect();
4703            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
4704                .unwrap();
4705            let raw1: Vec<u8> = original[4..8]
4706                .iter()
4707                .flat_map(|v| v.to_le_bytes())
4708                .collect();
4709            ds.write_chunk_raw(1, &raw1, 1).unwrap();
4710            file.close().unwrap();
4711        }
4712        {
4713            let file = H5File::open(&path).unwrap();
4714            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4715            assert_eq!(v, original);
4716        }
4717        std::fs::remove_file(&path).ok();
4718    }
4719
4720    /// An unfiltered chunk index has no slot for a stored size or mask, so a
4721    /// direct chunk write must be rejected rather than silently dropping them.
4722    #[test]
4723    fn write_chunk_raw_rejects_unfiltered() {
4724        let path = temp_path("wcr_unfiltered");
4725        let file = H5File::create(&path).unwrap();
4726        let ds = file
4727            .new_dataset::<i32>()
4728            .shape([0])
4729            .chunk(&[4])
4730            .max_shape(&[None])
4731            .create("v")
4732            .unwrap();
4733        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
4734        assert!(
4735            err.to_string().contains("filtered dataset"),
4736            "expected a filtered-dataset error, got: {err}"
4737        );
4738        std::fs::remove_file(&path).ok();
4739    }
4740
4741    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
4742    /// index to mean anything against, so the linear entry point points the
4743    /// caller at the coordinate-addressed one rather than guessing a grid.
4744    #[test]
4745    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
4746        let path = temp_path("wcr_btree2");
4747        let file = H5File::create(&path).unwrap();
4748        let ds = file
4749            .new_dataset::<i32>()
4750            .shape([0, 0])
4751            .chunk(&[2, 2])
4752            .max_shape(&[None, None])
4753            .create("grid")
4754            .unwrap();
4755        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
4756        assert!(
4757            err.to_string().contains("write_chunk_raw_at"),
4758            "expected a pointer to the coordinate form, got: {err}"
4759        );
4760        std::fs::remove_file(&path).ok();
4761    }
4762
4763    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
4764    /// and the type-11 record carries their size and the caller's mask, so a
4765    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
4766    /// bytes while one written compressed (mask 0) is decompressed.
4767    #[cfg(feature = "deflate")]
4768    #[test]
4769    fn write_chunk_raw_at_round_trips_on_btree_v2() {
4770        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4771
4772        let path = temp_path("wcr_at_btree2");
4773        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
4774        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
4775        {
4776            let file = H5File::create(&path).unwrap();
4777            let ds = file
4778                .new_dataset::<i32>()
4779                .shape([0, 0])
4780                .chunk(&[2, 2])
4781                .max_shape(&[None, None])
4782                .deflate(6)
4783                .create("grid")
4784                .unwrap();
4785            let pipeline = FilterPipeline::deflate(6);
4786            // Chunk (0,0): pipeline already applied upstream, mask 0.
4787            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
4788                .unwrap();
4789            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
4790            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
4791            file.close().unwrap();
4792        }
4793        let file = H5File::open(&path).unwrap();
4794        let ds = file.dataset("grid").unwrap();
4795        assert_eq!(ds.shape(), vec![4, 4]);
4796        let all = ds.read_raw::<i32>().unwrap();
4797        // Chunk (0,0) occupies rows 0..2, columns 0..2.
4798        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
4799        // Chunk (1,1) occupies rows 2..4, columns 2..4.
4800        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
4801        drop(file);
4802        std::fs::remove_file(&path).ok();
4803    }
4804
4805    /// The coordinate form is not BT2-only: it addresses an extensible- or
4806    /// fixed-array dataset's grid just as well, and records the same mask.
4807    #[cfg(feature = "deflate")]
4808    #[test]
4809    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
4810        for (label, max_shape) in [
4811            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
4812            ("wcr_at_fa", None),
4813        ] {
4814            let path = temp_path(label);
4815            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
4816            {
4817                let file = H5File::create(&path).unwrap();
4818                let mut b = file
4819                    .new_dataset::<i32>()
4820                    .shape([4usize, 4])
4821                    .chunk(&[2, 4])
4822                    .deflate(6);
4823                if let Some(ref ms) = max_shape {
4824                    b = b.max_shape(ms);
4825                }
4826                let ds = b.create("grid").unwrap();
4827                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
4828                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
4829                file.close().unwrap();
4830            }
4831            let file = H5File::open(&path).unwrap();
4832            let ds = file.dataset("grid").unwrap();
4833            let all = ds.read_raw::<i32>().unwrap();
4834            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
4835            drop(file);
4836            std::fs::remove_file(&path).ok();
4837        }
4838    }
4839
4840    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
4841    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
4842    /// chunk gives chunk_size_len = 2 (max 65535).
4843    #[cfg(feature = "deflate")]
4844    #[test]
4845    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
4846        let path = temp_path("wcr_at_oversized");
4847        let file = H5File::create(&path).unwrap();
4848        let ds = file
4849            .new_dataset::<i32>()
4850            .shape([0, 0])
4851            .chunk(&[1, 1])
4852            .max_shape(&[None, None])
4853            .deflate(4)
4854            .create("grid")
4855            .unwrap();
4856        let err = ds
4857            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
4858            .unwrap_err();
4859        assert!(
4860            err.to_string().contains("does not fit"),
4861            "expected a chunk-size-field overflow error, got: {err}"
4862        );
4863        std::fs::remove_file(&path).ok();
4864    }
4865
4866    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
4867    /// the same reason the array indexes reject a direct write.
4868    #[test]
4869    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
4870        let path = temp_path("wcr_at_unfiltered");
4871        let file = H5File::create(&path).unwrap();
4872        let ds = file
4873            .new_dataset::<i32>()
4874            .shape([0, 0])
4875            .chunk(&[2, 2])
4876            .max_shape(&[None, None])
4877            .create("grid")
4878            .unwrap();
4879        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
4880        assert!(
4881            err.to_string().contains("filtered dataset"),
4882            "expected a filtered-dataset error, got: {err}"
4883        );
4884        std::fs::remove_file(&path).ok();
4885    }
4886
4887    /// A stored size that does not fit the index's chunk-size field must error
4888    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
4889    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
4890    /// a 70000-byte stored chunk overflows it.
4891    #[cfg(feature = "deflate")]
4892    #[test]
4893    fn write_chunk_raw_rejects_oversized_chunk() {
4894        let path = temp_path("wcr_oversized");
4895        let file = H5File::create(&path).unwrap();
4896        let ds = file
4897            .new_dataset::<i32>()
4898            .shape([0])
4899            .chunk(&[1])
4900            .max_shape(&[None])
4901            .deflate(4)
4902            .create("v")
4903            .unwrap();
4904        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
4905        assert!(
4906            err.to_string().contains("does not fit"),
4907            "expected a chunk-size-field overflow error, got: {err}"
4908        );
4909        std::fs::remove_file(&path).ok();
4910    }
4911
4912    // ---- issue #5: runtime-width fixed-string reading ----------------------
4913
4914    use crate::format::messages::datatype::DatatypeMessage;
4915
4916    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
4917    /// element images, optionally chunked and deflated.
4918    fn write_fixed_string_dataset(
4919        path: &std::path::Path,
4920        dt: DatatypeMessage,
4921        width: usize,
4922        elems: &[&[u8]],
4923        compressed: bool,
4924    ) {
4925        let mut raw = Vec::with_capacity(elems.len() * width);
4926        for e in elems {
4927            assert!(e.len() <= width);
4928            raw.extend_from_slice(e);
4929            raw.resize(raw.len() + (width - e.len()), 0);
4930        }
4931        let file = H5File::create(path).unwrap();
4932        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
4933        if compressed {
4934            b = b.chunk(&[2]).deflate(6);
4935        }
4936        let ds = b.create("labels").unwrap();
4937        ds.write_raw_bytes(&raw).unwrap();
4938        file.close().unwrap();
4939    }
4940
4941    /// The width is whatever the file says, so one call reads a 24-byte label
4942    /// column and a 100-byte one. Producers like VASP pick it per dataset.
4943    #[test]
4944    fn read_strings_handles_any_fixed_width() {
4945        for width in [4usize, 24, 100] {
4946            let path = temp_path(&format!("fixed_str_{width}"));
4947            write_fixed_string_dataset(
4948                &path,
4949                DatatypeMessage::fixed_string(width as u32),
4950                width,
4951                &[b"ab", b"cde", b""],
4952                false,
4953            );
4954            let file = H5File::open(&path).unwrap();
4955            let got = file.dataset("labels").unwrap().read_strings().unwrap();
4956            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
4957            std::fs::remove_file(&path).ok();
4958        }
4959    }
4960
4961    /// Each padding rule decides where the value ends. Null-terminated stops at
4962    /// the first NUL and ignores the bytes after it; the two pad rules strip a
4963    /// tail of that byte and keep everything before it.
4964    #[test]
4965    fn read_strings_honors_every_padding_rule() {
4966        // "ab" then a NUL then trailing junk a null-terminated read must drop
4967        // and a null-padded read must keep.
4968        let elem: &[u8] = b"ab\0X\0\0";
4969        for (padding, want) in [(0u8, "ab"), (1, "ab\0X")] {
4970            let path = temp_path(&format!("fixed_pad_{padding}"));
4971            write_fixed_string_dataset(
4972                &path,
4973                DatatypeMessage::FixedString {
4974                    size: 6,
4975                    padding,
4976                    charset: 0,
4977                },
4978                6,
4979                &[elem],
4980                false,
4981            );
4982            let file = H5File::open(&path).unwrap();
4983            let got = file.dataset("labels").unwrap().read_strings().unwrap();
4984            assert_eq!(got, vec![want.to_string()], "padding {padding}");
4985            std::fs::remove_file(&path).ok();
4986        }
4987        // Space-padded keeps interior spaces and strips only the tail.
4988        let path = temp_path("fixed_pad_2");
4989        write_fixed_string_dataset(
4990            &path,
4991            DatatypeMessage::FixedString {
4992                size: 8,
4993                padding: 2,
4994                charset: 0,
4995            },
4996            8,
4997            &[b"a b     "],
4998            false,
4999        );
5000        let file = H5File::open(&path).unwrap();
5001        assert_eq!(
5002            file.dataset("labels").unwrap().read_strings().unwrap(),
5003            vec!["a b".to_string()]
5004        );
5005        std::fs::remove_file(&path).ok();
5006    }
5007
5008    /// A reserved padding or character-set code is an error naming the element,
5009    /// not a guess.
5010    #[test]
5011    fn read_strings_rejects_reserved_datatype_codes() {
5012        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
5013            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
5014            write_fixed_string_dataset(
5015                &path,
5016                DatatypeMessage::FixedString {
5017                    size: 4,
5018                    padding,
5019                    charset,
5020                },
5021                4,
5022                &[b"ab"],
5023                false,
5024            );
5025            let file = H5File::open(&path).unwrap();
5026            let err = file
5027                .dataset("labels")
5028                .unwrap()
5029                .read_strings()
5030                .unwrap_err()
5031                .to_string();
5032            assert!(err.contains(want), "got: {err}");
5033            std::fs::remove_file(&path).ok();
5034        }
5035    }
5036
5037    /// The declared character set is enforced: a byte that cannot be decoded is
5038    /// an error naming the element, and the lossy call is what accepts the file
5039    /// instead of a silent substitution here.
5040    #[test]
5041    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
5042        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
5043        // in one that declares UTF-8.
5044        for (charset, bytes, want) in [
5045            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
5046            (1, b"a\xff".as_slice(), "not valid UTF-8"),
5047        ] {
5048            let path = temp_path(&format!("fixed_charset_{charset}"));
5049            write_fixed_string_dataset(
5050                &path,
5051                DatatypeMessage::FixedString {
5052                    size: 6,
5053                    padding: 1,
5054                    charset,
5055                },
5056                6,
5057                &[b"ok", bytes],
5058                false,
5059            );
5060            let file = H5File::open(&path).unwrap();
5061            let ds = file.dataset("labels").unwrap();
5062            let err = ds.read_strings().unwrap_err().to_string();
5063            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
5064            let lossy = ds.read_strings_lossy().unwrap();
5065            assert_eq!(lossy[0], "ok");
5066            assert_eq!(
5067                lossy[1].chars().next().unwrap(),
5068                if charset == 0 { 'c' } else { 'a' }
5069            );
5070            std::fs::remove_file(&path).ok();
5071        }
5072    }
5073
5074    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
5075    /// split a character.
5076    #[test]
5077    fn read_strings_reads_utf8_fixed_strings() {
5078        let path = temp_path("fixed_utf8");
5079        write_fixed_string_dataset(
5080            &path,
5081            DatatypeMessage::fixed_string_utf8(12),
5082            12,
5083            &["héllo".as_bytes(), "안녕".as_bytes()],
5084            false,
5085        );
5086        let file = H5File::open(&path).unwrap();
5087        assert_eq!(
5088            file.dataset("labels").unwrap().read_strings().unwrap(),
5089            vec!["héllo".to_string(), "안녕".to_string()]
5090        );
5091        std::fs::remove_file(&path).ok();
5092    }
5093
5094    /// The decode sits on the decoded raw-data path, so a chunked and deflated
5095    /// dataset reads the same as a contiguous one.
5096    #[cfg(feature = "deflate")]
5097    #[test]
5098    fn read_strings_reads_a_compressed_fixed_string_dataset() {
5099        let path = temp_path("fixed_str_deflate");
5100        write_fixed_string_dataset(
5101            &path,
5102            DatatypeMessage::fixed_string(16),
5103            16,
5104            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
5105            true,
5106        );
5107        let file = H5File::open(&path).unwrap();
5108        assert_eq!(
5109            file.dataset("labels").unwrap().read_strings().unwrap(),
5110            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
5111        );
5112        std::fs::remove_file(&path).ok();
5113    }
5114
5115    /// One call covers both string datatypes, so a caller need not branch on
5116    /// which one the file used.
5117    #[test]
5118    fn read_strings_also_reads_variable_length_strings() {
5119        let path = temp_path("read_strings_vlen");
5120        {
5121            let file = H5File::create(&path).unwrap();
5122            file.write_vlen_strings("names", &["alpha", "", "안녕"])
5123                .unwrap();
5124            file.close().unwrap();
5125        }
5126        let file = H5File::open(&path).unwrap();
5127        assert_eq!(
5128            file.dataset("names").unwrap().read_strings().unwrap(),
5129            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
5130        );
5131        std::fs::remove_file(&path).ok();
5132    }
5133
5134    /// A file declaring a zero-width fixed string is an error, not the panic
5135    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
5136    /// test patches the width in the encoded datatype message down to zero and
5137    /// re-stamps the object header's checksum over the result.
5138    #[test]
5139    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
5140        use crate::format::checksum::checksum_metadata;
5141        use crate::format::object_header::OHDR_SIGNATURE;
5142
5143        let path = temp_path("fixed_str_zero_width");
5144        write_fixed_string_dataset(
5145            &path,
5146            DatatypeMessage::fixed_string(37),
5147            37,
5148            &[b"ab", b"cd"],
5149            false,
5150        );
5151
5152        // Version 1 string datatype: class|version, padding|charset, two
5153        // reserved bytes, then the width as a little-endian u32. The width is
5154        // 37 so the eight bytes occur once in the file.
5155        let mut bytes = std::fs::read(&path).unwrap();
5156        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
5157        let at = bytes
5158            .windows(needle.len())
5159            .position(|w| w == needle)
5160            .expect("encoded fixed-string datatype message");
5161        assert!(
5162            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
5163            "the datatype message pattern is not unique in the file"
5164        );
5165
5166        // The enclosing v2 object header ends in a checksum over everything
5167        // from its signature onwards; find the offset where the stored value
5168        // still agrees, so the patched header can be re-stamped there.
5169        let ohdr = bytes[..at]
5170            .windows(4)
5171            .rposition(|w| w == OHDR_SIGNATURE)
5172            .expect("enclosing object header");
5173        let cksum_at = (at + needle.len()..bytes.len() - 4)
5174            .find(|&e| {
5175                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
5176                    == checksum_metadata(&bytes[ohdr..e])
5177            })
5178            .expect("object header checksum");
5179
5180        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
5181        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
5182        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
5183        std::fs::write(&path, &bytes).unwrap();
5184
5185        let file = H5File::open(&path).unwrap();
5186        let err = file
5187            .dataset("labels")
5188            .unwrap()
5189            .read_strings()
5190            .unwrap_err()
5191            .to_string();
5192        assert!(err.contains("zero width"), "got: {err}");
5193        std::fs::remove_file(&path).ok();
5194    }
5195
5196    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
5197    #[test]
5198    fn read_strings_rejects_a_non_string_dataset() {
5199        let path = temp_path("read_strings_numeric");
5200        {
5201            let file = H5File::create(&path).unwrap();
5202            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
5203            ds.write_raw(&[1i32, 2, 3]).unwrap();
5204            file.close().unwrap();
5205        }
5206        let file = H5File::open(&path).unwrap();
5207        let err = file
5208            .dataset("nums")
5209            .unwrap()
5210            .read_strings()
5211            .unwrap_err()
5212            .to_string();
5213        assert!(err.contains("only for string datasets"), "got: {err}");
5214        std::fs::remove_file(&path).ok();
5215    }
5216
5217    // ---- issue #6: random updates to vlen string datasets ------------------
5218
5219    /// One element changes; the extent and every other element stay as they
5220    /// were, on a contiguous vlen dataset.
5221    #[test]
5222    fn write_vlen_strings_slice_replaces_one_element() {
5223        let path = temp_path("vlen_slice_contig");
5224        {
5225            let file = H5File::create(&path).unwrap();
5226            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
5227                .unwrap();
5228            file.close().unwrap();
5229        }
5230        {
5231            let file = H5File::open_rw(&path).unwrap();
5232            file.dataset_writer("notes")
5233                .unwrap()
5234                .write_vlen_strings_slice(1, &["replacement"])
5235                .unwrap();
5236            file.close().unwrap();
5237        }
5238        let file = H5File::open(&path).unwrap();
5239        let ds = file.dataset("notes").unwrap();
5240        assert_eq!(ds.shape(), vec![4]);
5241        assert_eq!(
5242            ds.read_vlen_strings().unwrap(),
5243            vec!["a", "replacement", "c", "d"]
5244        );
5245        std::fs::remove_file(&path).ok();
5246    }
5247
5248    /// The same on an appendable chunked dataset, across a reopen, over a range
5249    /// that spans a chunk boundary.
5250    #[test]
5251    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
5252        let path = temp_path("vlen_slice_chunked");
5253        {
5254            let file = H5File::create(&path).unwrap();
5255            file.create_appendable_vlen_dataset("notes", 2, None)
5256                .unwrap();
5257            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
5258            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
5259            file.append_vlen_strings("notes", &refs).unwrap();
5260            file.close().unwrap();
5261        }
5262        {
5263            // Elements 1..4 cross the 2-element chunk boundary twice.
5264            let file = H5File::open_rw(&path).unwrap();
5265            file.dataset_writer("notes")
5266                .unwrap()
5267                .write_vlen_strings_slice(1, &["x", "y", "z"])
5268                .unwrap();
5269            file.close().unwrap();
5270        }
5271        let file = H5File::open(&path).unwrap();
5272        let ds = file.dataset("notes").unwrap();
5273        assert_eq!(ds.shape(), vec![6]);
5274        assert_eq!(
5275            ds.read_vlen_strings().unwrap(),
5276            vec!["v0", "x", "y", "z", "v4", "v5"]
5277        );
5278        std::fs::remove_file(&path).ok();
5279    }
5280
5281    /// Elements the append buffer still holds are not on disk yet; the
5282    /// update flushes them to their chunks first, so the flush at close has
5283    /// nothing left to write the pre-update reference over.
5284    #[test]
5285    fn write_vlen_strings_slice_updates_buffered_elements() {
5286        let path = temp_path("vlen_slice_buffered");
5287        {
5288            let file = H5File::create(&path).unwrap();
5289            file.create_appendable_vlen_dataset("notes", 4, None)
5290                .unwrap();
5291            // 3 of a 4-element chunk: all three stay in the append buffer.
5292            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
5293            file.dataset_writer("notes")
5294                .unwrap()
5295                .write_vlen_strings_slice(1, &["patched"])
5296                .unwrap();
5297            file.close().unwrap();
5298        }
5299        let file = H5File::open(&path).unwrap();
5300        assert_eq!(
5301            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5302            vec!["a", "patched", "c"]
5303        );
5304        std::fs::remove_file(&path).ok();
5305    }
5306
5307    /// A range past the end is rejected before anything is written, and an
5308    /// empty batch costs the file nothing — without the early return it would
5309    /// still allocate and write an empty global-heap collection.
5310    #[test]
5311    fn write_vlen_strings_slice_checks_its_range() {
5312        let build = |name: &str, empty_call: bool| {
5313            let path = temp_path(name);
5314            let file = H5File::create(&path).unwrap();
5315            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
5316            let ds = file.dataset_writer("notes").unwrap();
5317            let err = ds
5318                .write_vlen_strings_slice(1, &["x", "y"])
5319                .unwrap_err()
5320                .to_string();
5321            assert!(
5322                err.contains("outside the dataset's 2 elements"),
5323                "got: {err}"
5324            );
5325            if empty_call {
5326                ds.write_vlen_strings_slice(0, &[]).unwrap();
5327            }
5328            file.close().unwrap();
5329            path
5330        };
5331
5332        let with_empty = build("vlen_slice_range", true);
5333        let control = build("vlen_slice_range_control", false);
5334        assert_eq!(
5335            std::fs::metadata(&with_empty).unwrap().len(),
5336            std::fs::metadata(&control).unwrap().len(),
5337            "the rejected and empty calls must leave the file untouched"
5338        );
5339
5340        let file = H5File::open(&with_empty).unwrap();
5341        assert_eq!(
5342            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5343            vec!["a", "b"]
5344        );
5345        std::fs::remove_file(&with_empty).ok();
5346        std::fs::remove_file(&control).ok();
5347    }
5348
5349    /// The element offset is one-dimensional, so a multi-dimensional dataset is
5350    /// rejected rather than silently indexed along the first axis.
5351    #[test]
5352    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
5353        let path = temp_path("vlen_slice_2d");
5354        let file = H5File::create(&path).unwrap();
5355        let ds = file
5356            .new_dataset::<u8>()
5357            .datatype(DatatypeMessage::vlen_string_utf8())
5358            .shape([2, 3])
5359            .create("grid")
5360            .unwrap();
5361        let err = ds
5362            .write_vlen_strings_slice(0, &["x"])
5363            .unwrap_err()
5364            .to_string();
5365        assert!(err.contains("1-dimension datasets"), "got: {err}");
5366        file.close().unwrap();
5367        std::fs::remove_file(&path).ok();
5368    }
5369
5370    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
5371    /// declares the ASCII character set would mislabel the bytes.
5372    #[test]
5373    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
5374        let path = temp_path("vlen_slice_ascii");
5375        let file = H5File::create(&path).unwrap();
5376        let ds = file
5377            .new_dataset::<u8>()
5378            .datatype(DatatypeMessage::vlen_string_ascii())
5379            .shape([3])
5380            .create("notes")
5381            .unwrap();
5382        let err = ds
5383            .write_vlen_strings_slice(0, &["ok", "안녕"])
5384            .unwrap_err()
5385            .to_string();
5386        assert!(
5387            err.contains("string 1") && err.contains("is not ASCII"),
5388            "got: {err}"
5389        );
5390        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
5391        file.close().unwrap();
5392        std::fs::remove_file(&path).ok();
5393    }
5394
5395    /// A numeric dataset is rejected: its elements are not vlen references and
5396    /// writing one would corrupt the column.
5397    #[test]
5398    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
5399        let path = temp_path("vlen_slice_numeric");
5400        let file = H5File::create(&path).unwrap();
5401        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
5402        ds.write_raw(&[1i32, 2, 3]).unwrap();
5403        let err = ds
5404            .write_vlen_strings_slice(0, &["x"])
5405            .unwrap_err()
5406            .to_string();
5407        assert!(
5408            err.contains("only for variable-length string datasets"),
5409            "got: {err}"
5410        );
5411        file.close().unwrap();
5412        std::fs::remove_file(&path).ok();
5413    }
5414
5415    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
5416
5417    /// Repeatedly replacing the same element must not grow the file per
5418    /// update: the collection each update supersedes is freed and the next
5419    /// update's collection lands in that block. Without the release every
5420    /// update costs another `H5HG_MINALLOC` (4096) bytes.
5421    #[test]
5422    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
5423        let size_after = |updates: usize| {
5424            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
5425            let file = H5File::create(&path).unwrap();
5426            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
5427            let ds = file.dataset_writer("notes").unwrap();
5428            for i in 0..updates {
5429                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
5430                    .unwrap();
5431            }
5432            file.close().unwrap();
5433            let n = std::fs::metadata(&path).unwrap().len();
5434            let read = H5File::open(&path).unwrap();
5435            assert_eq!(
5436                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5437                vec![format!("update {}", updates - 1), "b".to_string()]
5438            );
5439            drop(read);
5440            std::fs::remove_file(&path).ok();
5441            n
5442        };
5443
5444        // The allocator settles once a freed block is available to reuse, so
5445        // every count past that produces the same file.
5446        let settled = size_after(3);
5447        assert_eq!(size_after(20), settled, "20 updates against 3");
5448        assert_eq!(size_after(50), settled, "50 updates against 3");
5449    }
5450
5451    /// An empty string is stored as a real heap object under a reference whose
5452    /// sequence length is zero, so the release must go by the address, not the
5453    /// length — a length test strands the object and its collection forever.
5454    #[test]
5455    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
5456        let size_after = |updates: usize| {
5457            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
5458            let file = H5File::create(&path).unwrap();
5459            file.write_vlen_strings("notes", &["", "b"]).unwrap();
5460            let ds = file.dataset_writer("notes").unwrap();
5461            for _ in 0..updates {
5462                ds.write_vlen_strings_slice(0, &[""]).unwrap();
5463            }
5464            file.close().unwrap();
5465            let n = std::fs::metadata(&path).unwrap().len();
5466            let read = H5File::open(&path).unwrap();
5467            assert_eq!(
5468                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5469                vec!["".to_string(), "b".to_string()]
5470            );
5471            drop(read);
5472            std::fs::remove_file(&path).ok();
5473            n
5474        };
5475
5476        let settled = size_after(3);
5477        assert_eq!(size_after(20), settled, "20 empty updates against 3");
5478        assert_eq!(size_after(50), settled, "50 empty updates against 3");
5479    }
5480
5481    /// The elements the update does not name keep their strings, so freeing
5482    /// the superseded objects must not disturb the collection's survivors.
5483    #[test]
5484    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
5485        let path = temp_path("vlen_slice_heap_survivors");
5486        {
5487            let file = H5File::create(&path).unwrap();
5488            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
5489                .unwrap();
5490            let ds = file.dataset_writer("notes").unwrap();
5491            // Two updates inside the one collection the create wrote, so the
5492            // second reads a collection the first already rewrote.
5493            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
5494            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
5495            file.close().unwrap();
5496        }
5497        let file = H5File::open(&path).unwrap();
5498        assert_eq!(
5499            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5500            vec!["a", "B", "c", "D"]
5501        );
5502        std::fs::remove_file(&path).ok();
5503    }
5504
5505    /// Replacing every element of a chunked dataset empties the collection the
5506    /// append wrote, and the file must still read back correctly after its
5507    /// block goes to the allocator.
5508    #[test]
5509    fn write_vlen_strings_slice_frees_an_emptied_collection() {
5510        let path = temp_path("vlen_slice_heap_emptied");
5511        {
5512            let file = H5File::create(&path).unwrap();
5513            file.create_appendable_vlen_dataset("notes", 2, None)
5514                .unwrap();
5515            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
5516                .unwrap();
5517            file.close().unwrap();
5518        }
5519        {
5520            let file = H5File::open_rw(&path).unwrap();
5521            file.dataset_writer("notes")
5522                .unwrap()
5523                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
5524                .unwrap();
5525            file.close().unwrap();
5526        }
5527        let file = H5File::open(&path).unwrap();
5528        assert_eq!(
5529            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5530            vec!["w", "x", "y", "z"]
5531        );
5532        std::fs::remove_file(&path).ok();
5533    }
5534
5535    /// A collection larger than the 4096-byte minimum must keep its size when
5536    /// an object leaves it. Re-encoding at the natural size instead shrinks
5537    /// what the header declares, so the block's tail stops being part of the
5538    /// collection and the eventual free returns less than was allocated —
5539    /// stranding the difference on every cycle.
5540    #[test]
5541    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
5542        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
5543        let size_after = |cycles: usize| {
5544            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
5545            let file = H5File::create(&path).unwrap();
5546            let seed: Vec<String> = "abcd".chars().map(big).collect();
5547            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
5548            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
5549            // is one collection well above it.
5550            file.write_vlen_strings("notes", &refs).unwrap();
5551            let ds = file.dataset_writer("notes").unwrap();
5552            for _ in 0..cycles {
5553                // Partially empty the collection, then finish it off: the
5554                // block is freed only after it has been rewritten once.
5555                let head = big('x');
5556                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
5557                let tail: Vec<String> = "yzw".chars().map(big).collect();
5558                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
5559                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
5560            }
5561            file.close().unwrap();
5562            let n = std::fs::metadata(&path).unwrap().len();
5563            let read = H5File::open(&path).unwrap();
5564            assert_eq!(
5565                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5566                vec![big('x'), big('y'), big('z'), big('w')]
5567            );
5568            drop(read);
5569            std::fs::remove_file(&path).ok();
5570            n
5571        };
5572
5573        let settled = size_after(4);
5574        assert_eq!(size_after(30), settled, "30 cycles against 4");
5575    }
5576
5577    /// An element still in the append buffer has never been on disk, so its
5578    /// superseded object has to be found in the buffer or it is stranded.
5579    #[test]
5580    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
5581        let path = temp_path("vlen_slice_heap_buffered");
5582        let file = H5File::create(&path).unwrap();
5583        file.create_appendable_vlen_dataset("notes", 4, None)
5584            .unwrap();
5585        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
5586        let ds = file.dataset_writer("notes").unwrap();
5587        for i in 0..20 {
5588            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
5589                .unwrap();
5590        }
5591        file.close().unwrap();
5592
5593        let file = H5File::open(&path).unwrap();
5594        assert_eq!(
5595            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5596            vec!["a", "patch 19", "c"]
5597        );
5598        let size = std::fs::metadata(&path).unwrap().len();
5599        std::fs::remove_file(&path).ok();
5600        assert!(
5601            size < 20 * 4096,
5602            "20 buffered updates left {size} bytes, one collection per update"
5603        );
5604    }
5605
5606    /// Regression: a typed `write_slice` into rows the append buffer still
5607    /// held wrote the chunks, and the flush at close wrote the stale buffered
5608    /// rows back over it — write 99, read 50. The slice now flushes the
5609    /// buffer first, making the chunks the single authority for those rows.
5610    #[test]
5611    fn write_slice_into_the_buffered_tail_survives_close() {
5612        let path = temp_path("slice_into_buffered_tail");
5613        {
5614            let file = H5File::create(&path).unwrap();
5615            let ds = file
5616                .new_dataset::<i32>()
5617                .shape([0])
5618                .chunk(&[4])
5619                .max_shape(&[None])
5620                .create("d")
5621                .unwrap();
5622            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
5623            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
5624            ds.write_slice(&[4], &[1], &[99]).unwrap();
5625            file.close().unwrap();
5626        }
5627        {
5628            let file = H5File::open(&path).unwrap();
5629            let ds = file.dataset("d").unwrap();
5630            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
5631        }
5632        std::fs::remove_file(&path).ok();
5633    }
5634
5635    /// Extending a dataset while appends sit in the buffer must not move
5636    /// them: the buffer records the absolute row its frames belong to, so
5637    /// the flush at close lands them there, and the grown region reads as
5638    /// fill.
5639    #[test]
5640    fn extend_does_not_move_buffered_appends() {
5641        let path = temp_path("extend_keeps_buffered_rows");
5642        {
5643            let file = H5File::create(&path).unwrap();
5644            let ds = file
5645                .new_dataset::<i32>()
5646                .shape([0])
5647                .chunk(&[4])
5648                .max_shape(&[None])
5649                .create("d")
5650                .unwrap();
5651            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
5652            ds.extend(&[10]).unwrap();
5653            file.close().unwrap();
5654        }
5655        {
5656            let file = H5File::open(&path).unwrap();
5657            let ds = file.dataset("d").unwrap();
5658            assert_eq!(
5659                ds.read_raw::<i32>().unwrap(),
5660                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
5661            );
5662        }
5663        std::fs::remove_file(&path).ok();
5664    }
5665
5666    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
5667    /// dimensions) buffered fine but close() failed "not a chunked dataset"
5668    /// and lost the buffered rows — the append's chunk writes required the
5669    /// extensible-array index. They now go through the index-generic
5670    /// hyperslab engine.
5671    #[test]
5672    fn append_to_a_btree_v2_dataset_survives_close() {
5673        let path = temp_path("append_bt2_close");
5674        {
5675            let file = H5File::create(&path).unwrap();
5676            let ds = file
5677                .new_dataset::<i32>()
5678                .shape([0, 3])
5679                .chunk(&[4, 3])
5680                .max_shape(&[None, None])
5681                .create("d")
5682                .unwrap();
5683            // One buffered row, then a batch that crosses the chunk
5684            // boundary: 4 rows fill chunk band 0, one row stays buffered
5685            // for the flush at close.
5686            ds.append(&[1, 2, 3]).unwrap();
5687            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
5688            file.close().unwrap();
5689        }
5690        {
5691            let file = H5File::open(&path).unwrap();
5692            let ds = file.dataset("d").unwrap();
5693            assert_eq!(ds.shape(), vec![5, 3]);
5694            assert_eq!(
5695                ds.read_raw::<i32>().unwrap(),
5696                (1..=15).collect::<Vec<i32>>()
5697            );
5698        }
5699        std::fs::remove_file(&path).ok();
5700    }
5701
5702    /// A chunk row narrower than the frame row is legal geometry (libhdf5
5703    /// creates it); appended frames must be scattered across the row's
5704    /// tiles at the chunk stride, not packed at the frame stride.
5705    #[test]
5706    fn append_scatters_frames_across_narrow_chunk_tiles() {
5707        let path = temp_path("append_narrow_chunks");
5708        {
5709            let file = H5File::create(&path).unwrap();
5710            let ds = file
5711                .new_dataset::<i32>()
5712                .shape([0, 8])
5713                .chunk(&[2, 4])
5714                .max_shape(&[None, Some(8)])
5715                .create("d")
5716                .unwrap();
5717            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
5718            // row 2 is flushed partial at close.
5719            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
5720            file.close().unwrap();
5721        }
5722        {
5723            let file = H5File::open(&path).unwrap();
5724            let ds = file.dataset("d").unwrap();
5725            assert_eq!(ds.shape(), vec![3, 8]);
5726            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
5727        }
5728        std::fs::remove_file(&path).ok();
5729    }
5730
5731    /// A fixed-array dataset has no room to grow: appending must surface an
5732    /// error naming the chunk grid, not lose rows silently. (Before the
5733    /// index-generic append it failed as "not a chunked dataset".)
5734    #[test]
5735    fn append_to_a_full_fixed_array_dataset_errors() {
5736        let path = temp_path("append_fa_errors");
5737        let file = H5File::create(&path).unwrap();
5738        let ds = file
5739            .new_dataset::<i32>()
5740            .shape([4, 3])
5741            .chunk(&[2, 3])
5742            .create("d")
5743            .unwrap();
5744        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
5745        assert!(
5746            err.to_string().contains("chunk grid"),
5747            "unexpected error: {err}"
5748        );
5749        file.close().unwrap();
5750        std::fs::remove_file(&path).ok();
5751    }
5752
5753    /// A finite max_shape above the current shape used to be dropped on the
5754    /// fixed-array path: the array was sized from the current dims and the
5755    /// stored dataspace had no maximum, so growth failed. The array is now
5756    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
5757    /// fixed-max dataset appends up to its maximum and roundtrips.
5758    #[test]
5759    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
5760        let path = temp_path("fa_growable_dim0");
5761        {
5762            let file = H5File::create(&path).unwrap();
5763            let ds = file
5764                .new_dataset::<i32>()
5765                .shape([4, 3])
5766                .chunk(&[2, 3])
5767                .max_shape(&[Some(10), Some(3)])
5768                .create("d")
5769                .unwrap();
5770            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
5771            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
5772            file.close().unwrap();
5773        }
5774        {
5775            let file = H5File::open(&path).unwrap();
5776            let ds = file.dataset("d").unwrap();
5777            assert_eq!(ds.shape(), vec![6, 3]);
5778            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
5779        }
5780        std::fs::remove_file(&path).ok();
5781    }
5782
5783    /// The multiplier-dimension boundary: growing a dimension other than 0
5784    /// changes the current chunk grid but not the index grid. Chunk slots
5785    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
5786    /// chunks written before the extend are looked up under different
5787    /// indices after it.
5788    #[test]
5789    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
5790        let path = temp_path("fa_growable_dim1");
5791        {
5792            let file = H5File::create(&path).unwrap();
5793            let ds = file
5794                .new_dataset::<i32>()
5795                .shape([4, 3])
5796                .chunk(&[2, 3])
5797                .max_shape(&[Some(4), Some(9)])
5798                .create("d")
5799                .unwrap();
5800            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
5801            ds.extend(&[4, 6]).unwrap();
5802            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
5803                .unwrap();
5804            file.close().unwrap();
5805        }
5806        {
5807            let file = H5File::open(&path).unwrap();
5808            let ds = file.dataset("d").unwrap();
5809            assert_eq!(ds.shape(), vec![4, 6]);
5810            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
5811            // then [12 + r*3 ..) from the second.
5812            let mut expect = Vec::new();
5813            for r in 0i32..4 {
5814                expect.extend((r * 3)..(r * 3 + 3));
5815                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
5816            }
5817            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
5818        }
5819        std::fs::remove_file(&path).ok();
5820    }
5821
5822    /// Growth boundaries: past the stored maximum is rejected, and a dataset
5823    /// without a stored maximum is fixed at its extent (libhdf5 defaults
5824    /// maxdims to dims at creation).
5825    #[test]
5826    fn extend_beyond_the_maximum_is_rejected() {
5827        let path = temp_path("extend_beyond_max");
5828        let file = H5File::create(&path).unwrap();
5829        let ds = file
5830            .new_dataset::<i32>()
5831            .shape([4, 3])
5832            .chunk(&[2, 3])
5833            .max_shape(&[Some(6), Some(3)])
5834            .create("d")
5835            .unwrap();
5836        ds.extend(&[6, 3]).unwrap();
5837        let err = ds.extend(&[8, 3]).unwrap_err();
5838        assert!(
5839            err.to_string().contains("exceeds the maximum"),
5840            "unexpected error: {err}"
5841        );
5842        file.close().unwrap();
5843        std::fs::remove_file(&path).ok();
5844    }
5845
5846    /// An unlimited dimension other than 0 has no fixed linear slot without
5847    /// libhdf5's extensible-array swizzling, which is not implemented;
5848    /// creating the geometry silently re-indexed chunks on every extend, so
5849    /// it is rejected at create.
5850    #[test]
5851    fn builder_rejects_an_unlimited_inner_dimension() {
5852        let path = temp_path("unlimited_inner_dim");
5853        let file = H5File::create(&path).unwrap();
5854        let err = match file
5855            .new_dataset::<i32>()
5856            .shape([4, 0])
5857            .chunk(&[2, 2])
5858            .max_shape(&[Some(4), None])
5859            .create("d")
5860        {
5861            Ok(_) => panic!("create accepted an unlimited inner dimension"),
5862            Err(e) => e,
5863        };
5864        assert!(
5865            err.to_string().contains("not the first"),
5866            "unexpected error: {err}"
5867        );
5868        file.close().unwrap();
5869        std::fs::remove_file(&path).ok();
5870    }
5871
5872    /// Regression: a chunk wider than a fixed max dimension used to be
5873    /// accepted, and appends then packed rows at the chunk stride — writing
5874    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
5875    /// geometry at create (`H5D__chunk_construct`); so do we now.
5876    #[test]
5877    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
5878        let path = temp_path("builder_chunk_wider_than_max");
5879        let file = H5File::create(&path).unwrap();
5880        let err = match file
5881            .new_dataset::<i32>()
5882            .shape([0, 2])
5883            .chunk(&[2, 4])
5884            .max_shape(&[None, Some(2)])
5885            .create("v5")
5886        {
5887            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
5888            Err(e) => e,
5889        };
5890        assert!(
5891            err.to_string().contains("maximum dimension size"),
5892            "unexpected error: {err}"
5893        );
5894        file.close().unwrap();
5895        std::fs::remove_file(&path).ok();
5896    }
5897
5898    /// Boundary: `fits in T` vs `does not fit in T`, for both the too-large
5899    /// (u64::MAX → i64) and the negative-to-unsigned (−1 → u32) directions.
5900    #[test]
5901    fn numeric_int_checked_conversion_boundaries() {
5902        let path = temp_path("numeric_int_bounds");
5903        {
5904            let file = H5File::create(&path).unwrap();
5905            let ds = file.new_dataset::<u64>().shape([2]).create("u").unwrap();
5906            ds.write_raw(&[1u64, u64::MAX]).unwrap();
5907            let ds = file.new_dataset::<i32>().shape([2]).create("i").unwrap();
5908            ds.write_raw(&[-1i32, 5]).unwrap();
5909            file.close().unwrap();
5910        }
5911        let file = H5File::open(&path).unwrap();
5912
5913        let u = file.dataset("u").unwrap();
5914        assert_eq!(u.read_numeric_as::<u64>().unwrap(), vec![1, u64::MAX]);
5915        assert_eq!(
5916            u.read_numeric_as::<i128>().unwrap(),
5917            vec![1, i128::from(u64::MAX)]
5918        );
5919        let err = u.read_numeric_as::<i64>().unwrap_err();
5920        assert!(
5921            err.to_string()
5922                .contains("value 18446744073709551615 at element 1 does not fit in i64"),
5923            "unexpected error: {err}"
5924        );
5925
5926        let i = file.dataset("i").unwrap();
5927        assert_eq!(i.read_numeric_as::<i64>().unwrap(), vec![-1, 5]);
5928        let err = i.read_numeric_as::<u32>().unwrap_err();
5929        assert!(
5930            err.to_string()
5931                .contains("value -1 at element 0 does not fit in u32"),
5932            "unexpected error: {err}"
5933        );
5934        std::fs::remove_file(&path).ok();
5935    }
5936
5937    /// Boundary: f32 → f64 is exact widening; f64 → f32 is rejected.
5938    #[test]
5939    fn numeric_float_widening_only() {
5940        let path = temp_path("numeric_float");
5941        {
5942            let file = H5File::create(&path).unwrap();
5943            let ds = file.new_dataset::<f32>().shape([2]).create("f4").unwrap();
5944            ds.write_raw(&[1.5f32, -2.25]).unwrap();
5945            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
5946            ds.write_raw(&[3.75f64]).unwrap();
5947            file.close().unwrap();
5948        }
5949        let file = H5File::open(&path).unwrap();
5950
5951        let f4 = file.dataset("f4").unwrap();
5952        assert_eq!(f4.read_numeric_as::<f32>().unwrap(), vec![1.5, -2.25]);
5953        assert_eq!(f4.read_numeric_as::<f64>().unwrap(), vec![1.5, -2.25]);
5954
5955        let f8 = file.dataset("f8").unwrap();
5956        assert_eq!(f8.read_numeric_as::<f64>().unwrap(), vec![3.75]);
5957        let err = f8.read_numeric_as::<f32>().unwrap_err();
5958        assert!(
5959            err.to_string().contains("narrowing"),
5960            "unexpected error: {err}"
5961        );
5962        std::fs::remove_file(&path).ok();
5963    }
5964
5965    /// Boundary: cross-class conversions (float ↔ integer) are rejected in
5966    /// both directions, and a non-numeric datatype is rejected at classify.
5967    #[test]
5968    fn numeric_cross_class_and_non_numeric_rejected() {
5969        let path = temp_path("numeric_cross_class");
5970        {
5971            let file = H5File::create(&path).unwrap();
5972            let ds = file.new_dataset::<f64>().shape([1]).create("f8").unwrap();
5973            ds.write_raw(&[1.0f64]).unwrap();
5974            let ds = file.new_dataset::<i32>().shape([1]).create("i4").unwrap();
5975            ds.write_raw(&[7i32]).unwrap();
5976            file.write_vlen_strings("s", &["a", "b"]).unwrap();
5977            file.close().unwrap();
5978        }
5979        let file = H5File::open(&path).unwrap();
5980
5981        let err = file
5982            .dataset("f8")
5983            .unwrap()
5984            .read_numeric_as::<i64>()
5985            .unwrap_err();
5986        assert!(
5987            err.to_string().contains("floating-point dataset as i64"),
5988            "unexpected error: {err}"
5989        );
5990        let err = file
5991            .dataset("i4")
5992            .unwrap()
5993            .read_numeric_as::<f64>()
5994            .unwrap_err();
5995        assert!(
5996            err.to_string().contains("integer dataset as f64"),
5997            "unexpected error: {err}"
5998        );
5999        let err = file
6000            .dataset("s")
6001            .unwrap()
6002            .read_numeric_as::<i64>()
6003            .unwrap_err();
6004        assert!(
6005            err.to_string().contains("is not numeric"),
6006            "unexpected error: {err}"
6007        );
6008        std::fs::remove_file(&path).ok();
6009    }
6010
6011    /// Boundary: big-endian sources decode per the datatype's byte order.
6012    /// Unit-level (the writer only emits little-endian): feed `convert` a
6013    /// big-endian datatype plus big-endian bytes directly.
6014    #[test]
6015    fn numeric_big_endian_decode() {
6016        use super::numeric;
6017        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
6018
6019        let dt = DatatypeMessage::FixedPoint {
6020            size: 4,
6021            byte_order: ByteOrder::BigEndian,
6022            signed: true,
6023            bit_offset: 0,
6024            bit_precision: 32,
6025        };
6026        let mut raw = Vec::new();
6027        raw.extend_from_slice(&(-2i32).to_be_bytes());
6028        raw.extend_from_slice(&(100_000i32).to_be_bytes());
6029        let kind = numeric::classify(&dt).unwrap();
6030        assert_eq!(
6031            numeric::convert::<i64>(kind, &raw).unwrap(),
6032            vec![-2, 100_000]
6033        );
6034
6035        let dt = DatatypeMessage::FloatingPoint {
6036            size: 8,
6037            byte_order: ByteOrder::BigEndian,
6038            sign_location: 63,
6039            bit_offset: 0,
6040            bit_precision: 64,
6041            exponent_location: 52,
6042            exponent_size: 11,
6043            mantissa_location: 0,
6044            mantissa_size: 52,
6045            exponent_bias: 1023,
6046        };
6047        let raw = (-2.25f64).to_be_bytes();
6048        let kind = numeric::classify(&dt).unwrap();
6049        assert_eq!(numeric::convert::<f64>(kind, &raw).unwrap(), vec![-2.25]);
6050    }
6051
6052    /// `H5Attribute::read_numeric` validates the stored datatype before
6053    /// reinterpreting bytes: cross-width, cross-class, and non-numeric
6054    /// attributes error instead of returning bit-garbage, while the exact
6055    /// type and the HBool / complex-compound paths keep working.
6056    #[test]
6057    fn attr_read_numeric_validates_datatype() {
6058        use crate::types::{Complex64, HBool, VarLenUnicode};
6059        let path = temp_path("attr_read_numeric_validate");
6060        {
6061            let file = H5File::create(&path).unwrap();
6062            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
6063            ds.write_raw(&[1.0f32; 2]).unwrap();
6064            let a = ds.new_attr::<f64>().shape(()).create("f8").unwrap();
6065            a.write_numeric(&1.5f64).unwrap();
6066            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
6067            a.write_numeric(&-7i32).unwrap();
6068            let a = ds.new_attr::<HBool>().shape(()).create("b").unwrap();
6069            a.write_numeric(&HBool::from(true)).unwrap();
6070            let a = ds.new_attr::<Complex64>().shape(()).create("z").unwrap();
6071            a.write_numeric(&Complex64 { re: 1.0, im: -2.0 }).unwrap();
6072            let a = ds
6073                .new_attr::<VarLenUnicode>()
6074                .shape(())
6075                .create("s")
6076                .unwrap();
6077            a.write_scalar(&VarLenUnicode("text".into())).unwrap();
6078            file.close().unwrap();
6079        }
6080        let file = H5File::open(&path).unwrap();
6081        let ds = file.dataset("d").unwrap();
6082
6083        let f8 = ds.attr("f8").unwrap();
6084        assert_eq!(f8.read_numeric::<f64>().unwrap(), 1.5);
6085        // Previously returned the low half of the f64 image as an f32.
6086        let err = f8.read_numeric::<f32>().unwrap_err();
6087        assert!(
6088            err.to_string().contains("read_numeric_as"),
6089            "unexpected error: {err}"
6090        );
6091        assert!(f8.read_numeric::<i64>().is_err());
6092
6093        let i4 = ds.attr("i4").unwrap();
6094        assert_eq!(i4.read_numeric::<i32>().unwrap(), -7);
6095        assert!(i4.read_numeric::<u32>().is_err());
6096
6097        assert!(bool::from(
6098            ds.attr("b").unwrap().read_numeric::<HBool>().unwrap()
6099        ));
6100        let z = ds.attr("z").unwrap().read_numeric::<Complex64>().unwrap();
6101        assert_eq!((z.re, z.im), (1.0, -2.0));
6102
6103        // A vlen string attribute: read_numeric used to transmute the heap
6104        // reference bytes into the requested type.
6105        let s = ds.attr("s").unwrap();
6106        assert!(s.read_numeric::<f64>().is_err());
6107        assert!(s.read_numeric_as::<f64>().is_err());
6108        std::fs::remove_file(&path).ok();
6109    }
6110
6111    /// The attribute conversion read applies the dataset rules: checked
6112    /// int → int naming index and value on overflow, widening-only floats,
6113    /// cross-class rejected; an array attribute converts every element.
6114    #[test]
6115    fn attr_read_numeric_as_converts() {
6116        let path = temp_path("attr_read_numeric_as");
6117        {
6118            let file = H5File::create(&path).unwrap();
6119            let ds = file.new_dataset::<f32>().shape([2]).create("d").unwrap();
6120            ds.write_raw(&[1.0f32; 2]).unwrap();
6121            let a = ds.new_attr::<i32>().shape(()).create("i4").unwrap();
6122            a.write_numeric(&-7i32).unwrap();
6123            let a = ds.new_attr::<u64>().shape(()).create("u8max").unwrap();
6124            a.write_numeric(&u64::MAX).unwrap();
6125            let a = ds.new_attr::<i16>().shape([3]).create("arr").unwrap();
6126            a.write_array(&[1i16, -2, 3]).unwrap();
6127            file.close().unwrap();
6128        }
6129        let file = H5File::open(&path).unwrap();
6130        let ds = file.dataset("d").unwrap();
6131        assert_eq!(
6132            ds.attr("i4").unwrap().read_numeric_as::<i64>().unwrap(),
6133            vec![-7]
6134        );
6135        let err = ds
6136            .attr("u8max")
6137            .unwrap()
6138            .read_numeric_as::<i64>()
6139            .unwrap_err();
6140        assert!(
6141            err.to_string().contains("does not fit in i64"),
6142            "unexpected error: {err}"
6143        );
6144        assert_eq!(
6145            ds.attr("arr").unwrap().read_numeric_as::<i32>().unwrap(),
6146            vec![1, -2, 3]
6147        );
6148        assert!(ds.attr("i4").unwrap().read_numeric_as::<f64>().is_err());
6149        std::fs::remove_file(&path).ok();
6150    }
6151
6152    /// The hyperslab variant applies the same conversion to a sub-selection.
6153    #[test]
6154    fn numeric_slice_conversion() {
6155        let path = temp_path("numeric_slice");
6156        {
6157            let file = H5File::create(&path).unwrap();
6158            let ds = file.new_dataset::<i16>().shape([2, 3]).create("m").unwrap();
6159            ds.write_raw(&[1i16, 2, 3, 4, 5, 6]).unwrap();
6160            file.close().unwrap();
6161        }
6162        let file = H5File::open(&path).unwrap();
6163        let m = file.dataset("m").unwrap();
6164        assert_eq!(
6165            m.read_numeric_slice_as::<i32>(&[0, 1], &[2, 2]).unwrap(),
6166            vec![2, 3, 5, 6]
6167        );
6168        std::fs::remove_file(&path).ok();
6169    }
6170}