Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use crate::attribute::AttrBuilder;
8use crate::error::{Hdf5Error, Result};
9use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
10use crate::format::messages::datatype::DatatypeMessage;
11use crate::types::H5Type;
12
13// ---------------------------------------------------------------------------
14// DatasetBuilder
15// ---------------------------------------------------------------------------
16
17/// A fluent builder for creating datasets.
18///
19/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
20///
21/// ```no_run
22/// # use rust_hdf5::H5File;
23/// let file = H5File::create("builder.h5").unwrap();
24/// let ds = file.new_dataset::<f32>()
25///     .shape(&[10, 20])
26///     .create("temperatures")
27///     .unwrap();
28/// ```
29pub struct DatasetBuilder<T: H5Type> {
30    file_inner: SharedInner,
31    shape: Option<Vec<usize>>,
32    chunk_dims: Option<Vec<usize>>,
33    max_shape: Option<Vec<Option<usize>>>,
34    deflate_level: Option<u32>,
35    shuffle_deflate_level: Option<u32>,
36    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
37    group_path: Option<String>,
38    fill_value: Option<Vec<u8>>,
39    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
40    _marker: std::marker::PhantomData<T>,
41}
42
43impl<T: H5Type> DatasetBuilder<T> {
44    pub(crate) fn new(file_inner: SharedInner) -> Self {
45        Self {
46            file_inner,
47            shape: None,
48            chunk_dims: None,
49            max_shape: None,
50            deflate_level: None,
51            shuffle_deflate_level: None,
52            custom_pipeline: None,
53            group_path: None,
54            fill_value: None,
55            datatype_override: None,
56            _marker: std::marker::PhantomData,
57        }
58    }
59
60    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
61        Self {
62            file_inner,
63            shape: None,
64            chunk_dims: None,
65            max_shape: None,
66            deflate_level: None,
67            shuffle_deflate_level: None,
68            custom_pipeline: None,
69            group_path: Some(group_path),
70            fill_value: None,
71            datatype_override: None,
72            _marker: std::marker::PhantomData,
73        }
74    }
75
76    /// Set the dataset dimensions.
77    ///
78    /// This is required before calling [`create`](Self::create).
79    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
80    #[must_use]
81    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
82        self.shape = Some(dims.as_ref().to_vec());
83        self
84    }
85
86    /// Create a scalar (0-dimensional) dataset holding a single value.
87    #[must_use]
88    pub fn scalar(mut self) -> Self {
89        self.shape = Some(vec![]);
90        self
91    }
92
93    /// Set chunk dimensions for chunked storage.
94    ///
95    /// When set, the dataset uses chunked storage with the extensible array
96    /// index. You should also call [`max_shape`](Self::max_shape) or
97    /// [`resizable`](Self::resizable) to allow extending.
98    #[must_use]
99    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
100        self.chunk_dims = Some(chunk_dims.to_vec());
101        self
102    }
103
104    /// Make all dimensions unlimited (resizable).
105    ///
106    /// This sets max_dims to u64::MAX for all dimensions.
107    #[must_use]
108    pub fn resizable(mut self) -> Self {
109        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
110        self
111    }
112
113    /// Set maximum dimensions. `None` means unlimited for that dimension.
114    #[must_use]
115    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
116        self.max_shape = Some(max.to_vec());
117        self
118    }
119
120    /// Enable deflate (gzip) compression with the given level (0-9).
121    ///
122    /// Requires chunked storage (call `.chunk()` before `.create()`).
123    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
124    #[must_use]
125    pub fn deflate(mut self, level: u32) -> Self {
126        self.deflate_level = Some(level);
127        self
128    }
129
130    /// Enable shuffle + deflate compression.
131    ///
132    /// Shuffle reorders bytes by position within elements before compression,
133    /// which typically improves compression ratios for numeric data.
134    /// Requires chunked storage.
135    #[must_use]
136    pub fn shuffle_deflate(mut self, level: u32) -> Self {
137        self.shuffle_deflate_level = Some(level);
138        self
139    }
140
141    /// Enable Zstandard compression with the given level (1-22, default 3).
142    ///
143    /// Requires chunked storage (call `.chunk()` before `.create()`).
144    #[must_use]
145    pub fn zstd(mut self, level: u32) -> Self {
146        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
147        self
148    }
149
150    /// Set a custom filter pipeline for compression.
151    ///
152    /// This takes precedence over [`deflate`](Self::deflate) and
153    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
154    #[must_use]
155    pub fn filter_pipeline(
156        mut self,
157        pipeline: crate::format::messages::filter::FilterPipeline,
158    ) -> Self {
159        self.custom_pipeline = Some(pipeline);
160        self
161    }
162
163    /// Override the stored element datatype.
164    ///
165    /// By default the dataset is created with the datatype derived from the
166    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
167    /// different on-disk datatype than the in-memory element type — for
168    /// example a reduced-precision fixed-point type that matches an N-bit
169    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
170    /// override must equal `T::element_size()`; the N-bit filter packs the
171    /// significant bits within that fixed footprint.
172    ///
173    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
174    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
175    #[must_use]
176    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
177        self.datatype_override = Some(dt);
178        self
179    }
180
181    /// Set a user-defined fill value for unwritten elements.
182    ///
183    /// Without this, datasets use the HDF5 default zero-fill. When set,
184    /// the value is written into the dataset's fill-value message
185    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
186    /// unwritten regions as this value rather than zero.
187    ///
188    /// ```no_run
189    /// # use rust_hdf5::H5File;
190    /// let file = H5File::create("fv.h5").unwrap();
191    /// let ds = file.new_dataset::<f32>()
192    ///     .shape(&[100])
193    ///     .fill_value(f32::NAN)
194    ///     .create("data")
195    ///     .unwrap();
196    /// ```
197    #[must_use]
198    pub fn fill_value(mut self, value: T) -> Self {
199        let es = T::element_size();
200        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
201        // well-defined byte representation; `element_size()` matches
202        // `size_of::<T>()`. The slice borrows `value` only for this call.
203        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
204        self.fill_value = Some(raw.to_vec());
205        self
206    }
207
208    /// Finalize and create the dataset with the given `name`.
209    ///
210    /// The name is the link name within the root group (e.g. `"data"` or
211    /// `"group1/data"` once nested groups are supported).
212    pub fn create(self, name: &str) -> Result<H5Dataset> {
213        let shape = self.shape.ok_or_else(|| {
214            Hdf5Error::InvalidState("shape must be set before calling create()".into())
215        })?;
216
217        // Build the full name: if created within a group, prefix with group path
218        let full_name = if let Some(ref gp) = self.group_path {
219            if gp == "/" {
220                name.to_string()
221            } else {
222                let trimmed = gp.trim_start_matches('/');
223                format!("{}/{}", trimmed, name)
224            }
225        } else {
226            name.to_string()
227        };
228        let group_path = self.group_path.clone();
229        let fill_value = self.fill_value.clone();
230
231        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
232        let datatype = self.datatype_override.clone().unwrap_or_else(T::hdf5_type);
233        // Size one element from the on-disk datatype, not the carrier `T`. For
234        // the default path this equals `T::element_size()`; when a `datatype()`
235        // override is set (N-bit, or a runtime `CompoundType`), the stored type
236        // — not `T` — defines the element width, so the dataspace, the raw
237        // allocation, and the `write_raw` length check all agree with the bytes
238        // libhdf5/h5py will read.
239        let element_size = datatype.element_size() as usize;
240
241        // A filter pipeline requires chunked storage. When a filter is
242        // requested without explicit chunk dimensions, store the whole
243        // dataset as a single chunk instead of silently dropping the filter
244        // on the contiguous path. (This is one whole-dataset chunk, not
245        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
246        // for large datasets.)
247        let wants_filter = self.custom_pipeline.is_some()
248            || self.shuffle_deflate_level.is_some()
249            || self.deflate_level.is_some();
250        let auto_chunk: Option<Vec<usize>> =
251            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
252                Some(shape.iter().map(|&d| d.max(1)).collect())
253            } else {
254                None
255            };
256
257        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
258            // Chunked dataset
259            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
260            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
261                max.iter()
262                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
263                    .collect()
264            } else {
265                // Default: max = current
266                dims_u64.clone()
267            };
268
269            // libhdf5 selects the chunk index from the dataspace: a v2
270            // B-tree for two or more unlimited dimensions, an extensible
271            // array for exactly one, and a fixed array when there are none.
272            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
273            let is_btree2 = n_unlimited >= 2;
274            let is_fixed_array = n_unlimited == 0;
275
276            let index = {
277                let inner = borrow_inner(&self.file_inner);
278                match &*inner {
279                    H5FileInner::Writer(writer) => {
280                        // The requested filter pipeline, if any. Both index
281                        // types that take one explicitly (fixed array and v2
282                        // B-tree) build it the same way, so resolve it once.
283                        let explicit_pipeline = || {
284                            if let Some(p) = self.custom_pipeline.clone() {
285                                p
286                            } else if let Some(level) = self.shuffle_deflate_level {
287                                crate::format::messages::filter::FilterPipeline::shuffle_deflate(
288                                    T::element_size() as u32,
289                                    level,
290                                )
291                            } else {
292                                // deflate_level (checked by wants_filter).
293                                crate::format::messages::filter::FilterPipeline::deflate(
294                                    self.deflate_level.unwrap(),
295                                )
296                            }
297                        };
298                        let idx = if is_btree2 {
299                            // Two or more unlimited dimensions: a v2 B-tree,
300                            // whose records carry the stored size and filter
301                            // mask when the dataset is compressed (libhdf5
302                            // H5D_BT2_FILT).
303                            if wants_filter {
304                                writer.create_btree_v2_dataset_with_pipeline(
305                                    &full_name,
306                                    datatype,
307                                    &dims_u64,
308                                    &max_u64,
309                                    &chunk_u64,
310                                    explicit_pipeline(),
311                                )?
312                            } else {
313                                writer.create_btree_v2_dataset(
314                                    &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
315                                )?
316                            }
317                        } else if is_fixed_array {
318                            // A chunked dataset with no unlimited dimension
319                            // must use the fixed-array index — libhdf5
320                            // rejects an extensible-array index here. A
321                            // compressed fixed-shape dataset uses a *filtered*
322                            // fixed array (FA client id 1). The maximum shape
323                            // sizes the array, so a finite max above the
324                            // current shape stays growable.
325                            let pipeline = wants_filter.then(explicit_pipeline);
326                            writer.create_fixed_array_dataset_with_max(
327                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
328                            )?
329                        } else if let Some(pipeline) = self.custom_pipeline {
330                            writer.create_chunked_dataset_with_pipeline(
331                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
332                            )?
333                        } else if let Some(level) = self.shuffle_deflate_level {
334                            let pipeline =
335                                crate::format::messages::filter::FilterPipeline::shuffle_deflate(
336                                    T::element_size() as u32,
337                                    level,
338                                );
339                            writer.create_chunked_dataset_with_pipeline(
340                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
341                            )?
342                        } else if let Some(level) = self.deflate_level {
343                            writer.create_chunked_dataset_compressed(
344                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, level,
345                            )?
346                        } else {
347                            writer.create_chunked_dataset(
348                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
349                            )?
350                        };
351                        if let Some(ref gp) = group_path {
352                            if gp != "/" {
353                                writer.assign_dataset_to_group(gp, idx)?;
354                            }
355                        }
356                        if let Some(ref fv) = fill_value {
357                            writer.set_dataset_fill_value(idx, fv.clone())?;
358                        }
359                        idx
360                    }
361                    H5FileInner::Reader(_) => {
362                        return Err(Hdf5Error::InvalidState(
363                            "cannot create a dataset in read mode".into(),
364                        ));
365                    }
366                    H5FileInner::Closed => {
367                        return Err(Hdf5Error::InvalidState("file is closed".into()));
368                    }
369                }
370            };
371
372            Ok(H5Dataset {
373                file_inner: clone_inner(&self.file_inner),
374                info: DatasetInfo::Writer {
375                    index,
376                    shape,
377                    element_size,
378                    chunked: true,
379                    btree2: is_btree2,
380                    fixed_array: is_fixed_array,
381                },
382            })
383        } else {
384            // Contiguous dataset (original path)
385            let index = {
386                let inner = borrow_inner(&self.file_inner);
387                match &*inner {
388                    H5FileInner::Writer(writer) => {
389                        let idx = writer.create_dataset(&full_name, datatype, &dims_u64)?;
390                        if let Some(ref gp) = group_path {
391                            if gp != "/" {
392                                writer.assign_dataset_to_group(gp, idx)?;
393                            }
394                        }
395                        if let Some(ref fv) = fill_value {
396                            writer.set_dataset_fill_value(idx, fv.clone())?;
397                        }
398                        idx
399                    }
400                    H5FileInner::Reader(_) => {
401                        return Err(Hdf5Error::InvalidState(
402                            "cannot create a dataset in read mode".into(),
403                        ));
404                    }
405                    H5FileInner::Closed => {
406                        return Err(Hdf5Error::InvalidState("file is closed".into()));
407                    }
408                }
409            };
410
411            Ok(H5Dataset {
412                file_inner: clone_inner(&self.file_inner),
413                info: DatasetInfo::Writer {
414                    index,
415                    shape,
416                    element_size,
417                    chunked: false,
418                    btree2: false,
419                    fixed_array: false,
420                },
421            })
422        }
423    }
424}
425
426// ---------------------------------------------------------------------------
427// DatasetInfo
428// ---------------------------------------------------------------------------
429
430/// Internal metadata about a dataset handle.
431enum DatasetInfo {
432    /// A dataset created via `new_dataset().create()` in write mode.
433    Writer {
434        /// Index into the writer's dataset list.
435        index: usize,
436        /// Shape (current dimensions).
437        shape: Vec<usize>,
438        /// Size of one element in bytes.
439        element_size: usize,
440        /// Whether this is a chunked dataset.
441        chunked: bool,
442        /// Whether the chunk index is a v2 B-tree (multiple unlimited dims).
443        btree2: bool,
444        /// Whether the chunk index is a Fixed Array (no unlimited dims).
445        fixed_array: bool,
446    },
447    /// A dataset opened by name in read mode.
448    Reader {
449        /// The link name of the dataset.
450        name: String,
451        /// Shape (current dimensions).
452        shape: Vec<usize>,
453        /// Size of one element in bytes.
454        element_size: usize,
455    },
456}
457
458// ---------------------------------------------------------------------------
459// H5Dataset
460// ---------------------------------------------------------------------------
461
462/// A handle to an HDF5 dataset, supporting typed read and write operations.
463///
464/// The dataset holds a shared reference to the file's I/O backend, so it
465/// remains valid even if the originating [`H5File`](crate::file::H5File) is
466/// moved or dropped (they share ownership via `Rc`).
467pub struct H5Dataset {
468    file_inner: SharedInner,
469    info: DatasetInfo,
470}
471
472/// One chunk's bytes on the way to the file, and who filtered them.
473///
474/// This is what separates a normal chunk write from a direct one; everything
475/// else about placing a chunk is identical, so the two share a single dispatch.
476#[derive(Clone, Copy)]
477enum ChunkBytes<'a> {
478    /// The chunk's raw bytes; the dataset's filter pipeline runs before they
479    /// are stored.
480    Unfiltered(&'a [u8]),
481    /// Bytes already in their stored form, with `filter_mask` naming the
482    /// filters that were skipped.
483    Prefiltered { data: &'a [u8], filter_mask: u32 },
484}
485
486/// Strip a fixed-string element's padding, leaving the bytes that carry the
487/// value.
488///
489/// The three padding rules are the HDF5 datatype message's: null-terminated
490/// stops at the first NUL and says nothing about the bytes after it,
491/// null-padded and space-padded fill the tail with that byte. `index` names
492/// the element in the error a reserved padding rule produces.
493fn trim_fixed_string(elem: &[u8], padding: u8, index: usize) -> Result<&[u8]> {
494    let end = match padding {
495        // Null-terminated.
496        0 => elem.iter().position(|&b| b == 0).unwrap_or(elem.len()),
497        // Null-padded / space-padded: the tail of that byte is padding.
498        1 => elem.iter().rposition(|&b| b != 0).map_or(0, |i| i + 1),
499        2 => elem.iter().rposition(|&b| b != b' ').map_or(0, |i| i + 1),
500        other => {
501            return Err(Hdf5Error::InvalidState(format!(
502                "string {index} uses padding rule {other}, which the format reserves"
503            )))
504        }
505    };
506    Ok(&elem[..end])
507}
508
509/// Decode one string element's bytes under the datatype's character set.
510///
511/// `lossy` replaces what it cannot decode with U+FFFD instead of failing;
512/// `index` names the element in the error otherwise.
513fn decode_string(bytes: &[u8], charset: u8, lossy: bool, index: usize) -> Result<String> {
514    if lossy {
515        return Ok(String::from_utf8_lossy(bytes).into_owned());
516    }
517    match charset {
518        // ASCII. Bytes are 7-bit, which makes them UTF-8 as well.
519        0 => match bytes.iter().position(|&b| b >= 0x80) {
520            None => Ok(String::from_utf8_lossy(bytes).into_owned()),
521            Some(at) => Err(Hdf5Error::InvalidState(format!(
522                "string {index} declares the ASCII character set but byte {at} is {:#04x}",
523                bytes[at]
524            ))),
525        },
526        1 => String::from_utf8(bytes.to_vec()).map_err(|e| {
527            Hdf5Error::InvalidState(format!(
528                "string {index} declares UTF-8 but is not valid UTF-8: {e}"
529            ))
530        }),
531        other => Err(Hdf5Error::InvalidState(format!(
532            "string {index} uses character set {other}, which the format reserves"
533        ))),
534    }
535}
536
537impl H5Dataset {
538    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
539    pub(crate) fn new_reader(
540        file_inner: SharedInner,
541        name: String,
542        shape: Vec<usize>,
543        element_size: usize,
544    ) -> Self {
545        Self {
546            file_inner,
547            info: DatasetInfo::Reader {
548                name,
549                shape,
550                element_size,
551            },
552        }
553    }
554
555    /// Create a writer-mode dataset handle for an already-created dataset
556    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
557    ///
558    /// Reconstructs the same handle `new_dataset().create()` returns, so the
559    /// reopened dataset supports attribute writes and chunk appends.
560    pub(crate) fn new_writer(
561        file_inner: SharedInner,
562        index: usize,
563        shape: Vec<usize>,
564        element_size: usize,
565        chunked: bool,
566        btree2: bool,
567        fixed_array: bool,
568    ) -> Self {
569        Self {
570            file_inner,
571            info: DatasetInfo::Writer {
572                index,
573                shape,
574                element_size,
575                chunked,
576                btree2,
577                fixed_array,
578            },
579        }
580    }
581
582    /// Return the dataset dimensions.
583    pub fn shape(&self) -> Vec<usize> {
584        match &self.info {
585            DatasetInfo::Writer { shape, .. } => shape.clone(),
586            DatasetInfo::Reader { shape, .. } => shape.clone(),
587        }
588    }
589
590    /// Return the number of dimensions (rank) of the dataset.
591    pub fn ndims(&self) -> usize {
592        match &self.info {
593            DatasetInfo::Writer { shape, .. } => shape.len(),
594            DatasetInfo::Reader { shape, .. } => shape.len(),
595        }
596    }
597
598    /// Return the total number of elements in the dataset.
599    pub fn total_elements(&self) -> usize {
600        match &self.info {
601            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
602            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
603        }
604    }
605
606    /// Return the size of one element in bytes.
607    pub fn element_size(&self) -> usize {
608        match &self.info {
609            DatasetInfo::Writer { element_size, .. } => *element_size,
610            DatasetInfo::Reader { element_size, .. } => *element_size,
611        }
612    }
613
614    /// Return the element datatype as parsed from the file (read mode only).
615    ///
616    /// Unlike [`element_size`](Self::element_size), which reports only the
617    /// byte width, this exposes the full datatype: its class (integer vs
618    /// floating-point vs string vs compound …), signedness, byte order and
619    /// bit precision. Callers that must reconstruct the exact stored type —
620    /// for example to map it to a NumPy / Arrow dtype — should use this
621    /// instead of inferring a type from the byte width, which cannot
622    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
623    /// bytes).
624    ///
625    /// # Errors
626    ///
627    /// Returns an error if the file is in write mode, or if the dataset can
628    /// no longer be found in the reader's metadata.
629    ///
630    /// ```no_run
631    /// # use rust_hdf5::{H5File, DatatypeMessage};
632    /// let file = H5File::open("data.h5").unwrap();
633    /// let ds = file.dataset("image").unwrap();
634    /// match ds.datatype().unwrap() {
635    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
636    ///         println!("integer: {} bytes, signed={}", size, signed);
637    ///     }
638    ///     DatatypeMessage::FloatingPoint { size, .. } => {
639    ///         println!("float: {} bytes", size);
640    ///     }
641    ///     other => println!("other type: {other}"),
642    /// }
643    /// ```
644    pub fn datatype(&self) -> Result<DatatypeMessage> {
645        match &self.info {
646            DatasetInfo::Reader { name, .. } => {
647                let inner = borrow_inner(&self.file_inner);
648                match &*inner {
649                    H5FileInner::Reader(reader) => reader
650                        .dataset_info(name)
651                        .map(|info| info.datatype.clone())
652                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
653                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
654                }
655            }
656            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
657                "datatype() is only available in read mode".into(),
658            )),
659        }
660    }
661
662    /// Return the chunk dimensions, if this is a chunked dataset.
663    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
664        match &self.info {
665            DatasetInfo::Reader { name, .. } => {
666                let inner = borrow_inner(&self.file_inner);
667                if let H5FileInner::Reader(reader) = &*inner {
668                    if let Some(info) = reader.dataset_info(name) {
669                        use crate::format::messages::data_layout::DataLayoutMessage;
670                        let chunk_dims = match &info.layout {
671                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
672                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
673                            _ => None,
674                        };
675                        if let Some(chunk_dims) = chunk_dims {
676                            // Strip trailing element-size dimension
677                            return Some(
678                                chunk_dims[..chunk_dims.len() - 1]
679                                    .iter()
680                                    .map(|&d| d as usize)
681                                    .collect(),
682                            );
683                        }
684                    }
685                }
686                None
687            }
688            DatasetInfo::Writer { .. } => None,
689        }
690    }
691
692    /// Return whether this is a chunked dataset.
693    pub fn is_chunked(&self) -> bool {
694        match &self.info {
695            DatasetInfo::Writer { chunked, .. } => *chunked,
696            DatasetInfo::Reader { name, .. } => {
697                let inner = borrow_inner(&self.file_inner);
698                match &*inner {
699                    H5FileInner::Reader(reader) => {
700                        if let Some(info) = reader.dataset_info(name) {
701                            use crate::format::messages::data_layout::DataLayoutMessage;
702                            matches!(
703                                info.layout,
704                                DataLayoutMessage::ChunkedV4 { .. }
705                                    | DataLayoutMessage::ChunkedV3 { .. }
706                            )
707                        } else {
708                            false
709                        }
710                    }
711                    _ => false,
712                }
713            }
714        }
715    }
716
717    /// Return the names of all attributes on this dataset (read mode only).
718    pub fn attr_names(&self) -> Result<Vec<String>> {
719        match &self.info {
720            DatasetInfo::Reader { name, .. } => {
721                let inner = borrow_inner(&self.file_inner);
722                match &*inner {
723                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
724                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
725                }
726            }
727            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
728                "attr_names not available in write mode".into(),
729            )),
730        }
731    }
732
733    /// Open an attribute by name (read mode only).
734    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
735        match &self.info {
736            DatasetInfo::Reader { name, .. } => {
737                let inner = borrow_inner(&self.file_inner);
738                match &*inner {
739                    H5FileInner::Reader(reader) => {
740                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
741                        Ok(crate::attribute::H5Attribute::new_reader(
742                            clone_inner(&self.file_inner),
743                            attr_msg,
744                        ))
745                    }
746                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
747                }
748            }
749            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
750                "attr() not available in write mode".into(),
751            )),
752        }
753    }
754
755    /// Start building a new attribute on this dataset.
756    ///
757    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
758    /// and `.create("name")` to finalize.
759    ///
760    /// # Example
761    ///
762    /// ```no_run
763    /// # use rust_hdf5::H5File;
764    /// # use rust_hdf5::types::VarLenUnicode;
765    /// let file = H5File::create("attr.h5").unwrap();
766    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
767    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
768    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
769    /// ```
770    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
771        let ds_index = match &self.info {
772            DatasetInfo::Writer { index, .. } => *index,
773            DatasetInfo::Reader { .. } => {
774                // Reader mode: we'll return a builder that will error on create.
775                // Using usize::MAX as sentinel.
776                usize::MAX
777            }
778        };
779        AttrBuilder::new(&self.file_inner, ds_index)
780    }
781
782    /// Write a typed slice holding the dataset's whole image.
783    ///
784    /// The slice length must match the total number of elements declared by
785    /// the dataset shape. The data is reinterpreted as raw bytes and written
786    /// to the file: to the contiguous data block, or — for a chunked dataset —
787    /// scattered across its chunk grid, through the filter pipeline if one is
788    /// set. To write only part of a dataset, use
789    /// [`write_slice`](Self::write_slice).
790    ///
791    /// # Errors
792    ///
793    /// Returns an error if:
794    /// - The file is in read mode.
795    /// - The data length does not match the declared shape.
796    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
797        match &self.info {
798            DatasetInfo::Writer {
799                index,
800                shape,
801                element_size,
802                chunked,
803                btree2,
804                fixed_array,
805            } => {
806                let total_elements: usize = shape.iter().product();
807                if data.len() != total_elements {
808                    return Err(Hdf5Error::InvalidState(format!(
809                        "data length {} does not match dataset size {}",
810                        data.len(),
811                        total_elements,
812                    )));
813                }
814
815                // Verify element size matches
816                if T::element_size() != *element_size {
817                    return Err(Hdf5Error::TypeMismatch(format!(
818                        "write type has element size {} but dataset expects {}",
819                        T::element_size(),
820                        element_size,
821                    )));
822                }
823
824                // Safety: T: Copy + 'static (numeric primitive) with well-defined
825                // byte representation. The resulting slice borrows `data` and
826                // lives only as long as this block.
827                let byte_len = data.len() * T::element_size();
828                let raw =
829                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
830
831                if *chunked {
832                    // A chunked dataset has no contiguous data block; scatter
833                    // the full row-major image into its chunk grid and write
834                    // each chunk through the dataset's filter pipeline.
835                    return self.write_full_image_chunked(
836                        *index,
837                        *btree2,
838                        *fixed_array,
839                        raw,
840                        *element_size,
841                    );
842                }
843
844                let inner = borrow_inner(&self.file_inner);
845                match &*inner {
846                    H5FileInner::Writer(writer) => {
847                        writer.write_dataset_raw(*index, raw)?;
848                        Ok(())
849                    }
850                    _ => Err(Hdf5Error::InvalidState(
851                        "file is no longer in write mode".into(),
852                    )),
853                }
854            }
855            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
856                "cannot write to a dataset opened in read mode".into(),
857            )),
858        }
859    }
860
861    /// Write the raw byte image of the whole dataset directly.
862    ///
863    /// Takes the same layouts as [`write_raw`](Self::write_raw): a contiguous
864    /// data block, or a chunk grid the image is scattered across.
865    ///
866    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
867    /// `H5Type` carrier, so it works for element types that have no matching
868    /// Rust primitive — in particular a runtime
869    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
870    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
871    /// `product(shape) * element_size`, where `element_size` is taken from the
872    /// dataset's on-disk datatype.
873    ///
874    /// ```no_run
875    /// # use rust_hdf5::H5File;
876    /// # use rust_hdf5::types::{CompoundType, H5Type};
877    /// let file = H5File::create("c.h5").unwrap();
878    /// let ct = CompoundType {
879    ///     members: vec![
880    ///         ("id".to_string(), i32::hdf5_type(), 0),
881    ///         ("val".to_string(), f64::hdf5_type(), 4),
882    ///     ],
883    ///     total_size: 12,
884    /// };
885    /// let ds = file
886    ///     .new_dataset::<u8>()
887    ///     .datatype(ct.to_datatype())
888    ///     .shape(&[2])
889    ///     .create("records")
890    ///     .unwrap();
891    /// let mut bytes = Vec::new();
892    /// bytes.extend_from_slice(&1i32.to_le_bytes());
893    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
894    /// bytes.extend_from_slice(&2i32.to_le_bytes());
895    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
896    /// ds.write_raw_bytes(&bytes).unwrap();
897    /// ```
898    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
899        match &self.info {
900            DatasetInfo::Writer {
901                index,
902                shape,
903                element_size,
904                chunked,
905                btree2,
906                fixed_array,
907            } => {
908                let expected: usize = shape.iter().product::<usize>() * *element_size;
909                if bytes.len() != expected {
910                    return Err(Hdf5Error::InvalidState(format!(
911                        "raw byte length {} does not match dataset size {} \
912                         (product(shape) * element_size {})",
913                        bytes.len(),
914                        expected,
915                        element_size,
916                    )));
917                }
918                if *chunked {
919                    // Scatter the full row-major image into the chunk grid
920                    // (same path as write_raw, carrier-agnostic bytes).
921                    return self.write_full_image_chunked(
922                        *index,
923                        *btree2,
924                        *fixed_array,
925                        bytes,
926                        *element_size,
927                    );
928                }
929                let inner = borrow_inner(&self.file_inner);
930                match &*inner {
931                    H5FileInner::Writer(writer) => {
932                        writer.write_dataset_raw(*index, bytes)?;
933                        Ok(())
934                    }
935                    _ => Err(Hdf5Error::InvalidState(
936                        "file is no longer in write mode".into(),
937                    )),
938                }
939            }
940            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
941                "cannot write to a dataset opened in read mode".into(),
942            )),
943        }
944    }
945
946    /// Scatter a full row-major dataset image into its chunk grid, writing
947    /// every chunk through the dataset's filter pipeline.
948    ///
949    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
950    /// — it is how [`write_raw`](Self::write_raw) and
951    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
952    /// (including the single auto-chunk created when a filter is set without
953    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
954    /// chunk footprint, exactly as libhdf5 stores them.
955    fn write_full_image_chunked(
956        &self,
957        index: usize,
958        btree2: bool,
959        fixed_array: bool,
960        bytes: &[u8],
961        element_size: usize,
962    ) -> Result<()> {
963        let inner = borrow_inner(&self.file_inner);
964        let writer = match &*inner {
965            H5FileInner::Writer(w) => w,
966            _ => {
967                return Err(Hdf5Error::InvalidState(
968                    "file is no longer in write mode".into(),
969                ))
970            }
971        };
972        // Whole-operation guard: the flush, the grid snapshot and the chunk
973        // writes below must not interleave with a concurrent same-dataset
974        // operation.
975        let cell = writer.ds(index);
976        let _op = cell.op.lock();
977        // A buffered append tail would flush over the image at close; hand
978        // it to the chunks first, the image below overwrites everything.
979        writer.flush_append_buffer(index)?;
980        let chunk_dims = writer
981            .dataset_chunk_dims(index)
982            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
983            .to_vec();
984        let dims = writer.dataset_dims(index).to_vec();
985        let rank = dims.len();
986
987        // Chunk grid: number of chunks along each dimension (row-major).
988        let mut grid = vec![0u64; rank];
989        for d in 0..rank {
990            grid[d] = if chunk_dims[d] > 0 {
991                dims[d].div_ceil(chunk_dims[d])
992            } else {
993                0
994            };
995        }
996        let total_chunks: u64 = grid.iter().product();
997
998        // Decode the iteration counter into row-major coordinates over the
999        // *current* image's chunk grid. This is only an odometer over the
1000        // chunks the image spans — the slot a chunk is recorded under comes
1001        // from the index grid (`Hdf5Writer::chunk_slot`), which the maximum
1002        // extent decides.
1003        let coords_of = |linear: u64| -> Vec<u64> {
1004            let mut rem = linear;
1005            let mut coords = vec![0u64; rank];
1006            for d in (0..rank).rev() {
1007                coords[d] = rem % grid[d];
1008                rem /= grid[d];
1009            }
1010            coords
1011        };
1012
1013        if btree2 {
1014            // B-tree v2 stores chunks verbatim (unfiltered only in this
1015            // codebase), so there is no compression to parallelize; write one
1016            // chunk at a time.
1017            for linear in 0..total_chunks {
1018                let coords = coords_of(linear);
1019                let chunk_buf =
1020                    Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
1021                writer.write_chunk_btree_v2_inner(index, &coords, &chunk_buf)?;
1022            }
1023        } else {
1024            // Extensible array and fixed array both compress each chunk through
1025            // the filter pipeline. Gather chunks and write them through the
1026            // per-index batch path so the pipeline compresses them in parallel
1027            // (with the `parallel` feature). A fixed-size window bounds peak
1028            // memory instead of materializing every chunk at once; 256 keeps
1029            // every rayon worker fed while capping the transient buffers to
1030            // window * chunk bytes. The two indexes differ only in how a chunk
1031            // is addressed: EA by its linear grid index, FA by grid coordinates.
1032            const BATCH_WINDOW: u64 = 256;
1033            let mut start = 0u64;
1034            while start < total_chunks {
1035                let end = (start + BATCH_WINDOW).min(total_chunks);
1036                let items: Vec<(Vec<u64>, Vec<u8>)> = (start..end)
1037                    .map(|counter| {
1038                        let coords = coords_of(counter);
1039                        let buf =
1040                            Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
1041                        (coords, buf)
1042                    })
1043                    .collect();
1044                if fixed_array {
1045                    let pairs: Vec<(&[u64], &[u8])> = items
1046                        .iter()
1047                        .map(|(c, d)| (c.as_slice(), d.as_slice()))
1048                        .collect();
1049                    writer.write_chunks_fixed_array_batch_inner(index, &pairs)?;
1050                } else {
1051                    let mut pairs: Vec<(u64, &[u8])> = Vec::with_capacity(items.len());
1052                    for (c, d) in &items {
1053                        pairs.push((writer.chunk_slot(index, c)?, d.as_slice()));
1054                    }
1055                    writer.write_chunks_batch_inner(index, &pairs)?;
1056                }
1057                start = end;
1058            }
1059        }
1060        Ok(())
1061    }
1062
1063    /// Gather one chunk's bytes from a row-major full-dataset image.
1064    ///
1065    /// `coords` are the chunk's grid coordinates. The returned buffer is
1066    /// exactly `product(chunk_dims) * element_size` bytes, zero-padded where
1067    /// the chunk extends past the dataset edge.
1068    fn gather_chunk(
1069        source: &[u8],
1070        dims: &[u64],
1071        chunk_dims: &[u64],
1072        coords: &[u64],
1073        element_size: usize,
1074    ) -> Vec<u8> {
1075        let rank = dims.len();
1076        let chunk_elems: u64 = chunk_dims.iter().product();
1077        let mut out = vec![0u8; chunk_elems as usize * element_size];
1078        if rank == 0 {
1079            // Scalar dataset: a single element, no chunking dimension.
1080            if source.len() >= element_size {
1081                out[..element_size].copy_from_slice(&source[..element_size]);
1082            }
1083            return out;
1084        }
1085
1086        // Actual extent of this chunk along each dimension (edge chunks are
1087        // smaller than the nominal chunk shape).
1088        let mut extent = vec![0u64; rank];
1089        for d in 0..rank {
1090            let start = coords[d] * chunk_dims[d];
1091            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
1092            extent[d] = end.saturating_sub(start);
1093        }
1094        if extent.contains(&0) {
1095            return out; // nothing of the dataset falls in this chunk
1096        }
1097
1098        // Row-major strides (in elements) for the source (over `dims`) and the
1099        // destination chunk buffer (over `chunk_dims`).
1100        let mut src_stride = vec![1u64; rank];
1101        let mut dst_stride = vec![1u64; rank];
1102        for d in (0..rank - 1).rev() {
1103            src_stride[d] = src_stride[d + 1] * dims[d + 1];
1104            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
1105        }
1106
1107        // Copy one contiguous run along the last axis per outer multi-index.
1108        let last = rank - 1;
1109        let run = extent[last] as usize * element_size;
1110        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
1111        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
1112        for _ in 0..outer {
1113            let mut src_off = 0u64;
1114            let mut dst_off = 0u64;
1115            for d in 0..rank {
1116                let global = coords[d] * chunk_dims[d] + idx[d];
1117                src_off += global * src_stride[d];
1118                dst_off += idx[d] * dst_stride[d];
1119            }
1120            let s = src_off as usize * element_size;
1121            let dpos = dst_off as usize * element_size;
1122            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
1123
1124            // Advance the multi-index over axes [0..last); the last axis is the
1125            // contiguous run handled above.
1126            let mut d = last;
1127            while d > 0 {
1128                d -= 1;
1129                idx[d] += 1;
1130                if idx[d] < extent[d] {
1131                    break;
1132                }
1133                idx[d] = 0;
1134            }
1135        }
1136        out
1137    }
1138
1139    /// Write a single chunk to a chunked dataset.
1140    ///
1141    /// `chunk_idx` is the linear chunk index (typically the frame number for
1142    /// streaming datasets). `data` is the raw byte data for one chunk.
1143    ///
1144    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
1145    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
1146    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
1147        match &self.info {
1148            DatasetInfo::Writer {
1149                index,
1150                chunked,
1151                btree2,
1152                fixed_array,
1153                ..
1154            } => {
1155                if !*chunked {
1156                    return Err(Hdf5Error::InvalidState(
1157                        "write_chunk is only for chunked datasets".into(),
1158                    ));
1159                }
1160                if *btree2 {
1161                    return Err(Hdf5Error::InvalidState(
1162                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
1163                         with the chunk's grid coordinates"
1164                            .into(),
1165                    ));
1166                }
1167
1168                let inner = borrow_inner(&self.file_inner);
1169                match &*inner {
1170                    H5FileInner::Writer(writer) => {
1171                        // One op: the slot decode and the write see the same
1172                        // extents.
1173                        let cell = writer.ds(*index);
1174                        let _op = cell.op.lock();
1175                        if *fixed_array {
1176                            // Fixed-array dataset: decode the index-grid slot
1177                            // into row-major grid coordinates.
1178                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
1179                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?;
1180                        } else {
1181                            writer.write_chunk_inner(*index, chunk_idx as u64, data)?;
1182                        }
1183                        Ok(())
1184                    }
1185                    _ => Err(Hdf5Error::InvalidState(
1186                        "file is no longer in write mode".into(),
1187                    )),
1188                }
1189            }
1190            DatasetInfo::Reader { .. } => {
1191                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1192            }
1193        }
1194    }
1195
1196    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
1197    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
1198    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
1199    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
1200    ///
1201    /// `chunk_idx` is the linear chunk index (the frame number for streaming
1202    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
1203    /// the already-filtered bytes of one chunk — its length is the *stored*
1204    /// (compressed) size, not the uncompressed chunk size.
1205    ///
1206    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
1207    /// dataset's pipeline was **not** applied to this chunk and must be skipped
1208    /// on read. Pass 0 when the full pipeline was already applied upstream (the
1209    /// common case: a codec plugin handed you compressed frames).
1210    ///
1211    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
1212    /// has no slot to record a stored size or mask. A v2-B-tree-indexed dataset
1213    /// (two or more unlimited dimensions) has no fixed chunk grid to linearize
1214    /// against, so address its chunks with
1215    /// [`write_chunk_raw_at`](Self::write_chunk_raw_at) instead.
1216    ///
1217    /// # Reading back
1218    ///
1219    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
1220    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
1221    /// the reader skipping exactly the filters the mask marks as not applied.
1222    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
1223        match &self.info {
1224            DatasetInfo::Writer {
1225                index,
1226                chunked,
1227                btree2,
1228                fixed_array,
1229                ..
1230            } => {
1231                if !*chunked {
1232                    return Err(Hdf5Error::InvalidState(
1233                        "write_chunk_raw is only for chunked datasets".into(),
1234                    ));
1235                }
1236                if *btree2 {
1237                    return Err(Hdf5Error::InvalidState(
1238                        "this dataset uses a v2 B-tree chunk index; use \
1239                         write_chunk_raw_at with the chunk's grid coordinates"
1240                            .into(),
1241                    ));
1242                }
1243
1244                let inner = borrow_inner(&self.file_inner);
1245                match &*inner {
1246                    H5FileInner::Writer(writer) => {
1247                        // One op: the slot decode and the write see the same
1248                        // extents.
1249                        let cell = writer.ds(*index);
1250                        let _op = cell.op.lock();
1251                        if *fixed_array {
1252                            // Fixed-array dataset: decode the index-grid slot
1253                            // into row-major grid coordinates.
1254                            let coords = writer.chunk_coords_from_slot(*index, chunk_idx as u64)?;
1255                            writer.write_compressed_chunk_fixed_array_inner(
1256                                *index,
1257                                &coords,
1258                                data,
1259                                filter_mask,
1260                            )?;
1261                        } else {
1262                            writer.write_compressed_chunk_inner(
1263                                *index,
1264                                chunk_idx as u64,
1265                                data,
1266                                filter_mask,
1267                            )?;
1268                        }
1269                        Ok(())
1270                    }
1271                    _ => Err(Hdf5Error::InvalidState(
1272                        "file is no longer in write mode".into(),
1273                    )),
1274                }
1275            }
1276            DatasetInfo::Reader { .. } => {
1277                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1278            }
1279        }
1280    }
1281
1282    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
1283    /// chunk-grid coordinates (one per dimension).
1284    ///
1285    /// This is the entry point for datasets with two or more unlimited
1286    /// dimensions. The dataset's logical dimensions are extended to cover
1287    /// the written chunk. `data` is the raw bytes of one full chunk.
1288    ///
1289    /// ```no_run
1290    /// # use rust_hdf5::H5File;
1291    /// let file = H5File::create("bt2.h5").unwrap();
1292    /// let ds = file.new_dataset::<i32>()
1293    ///     .shape(&[0, 0])
1294    ///     .chunk(&[2, 2])
1295    ///     .max_shape(&[None, None])
1296    ///     .create("grid")
1297    ///     .unwrap();
1298    /// let chunk = [0i32, 1, 2, 3];
1299    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
1300    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
1301    /// ```
1302    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
1303        self.write_chunk_at_inner(chunk_coords, ChunkBytes::Unfiltered(data), "write_chunk_at")
1304    }
1305
1306    /// Write an already-filtered chunk **verbatim** to a chunked dataset,
1307    /// addressed by its chunk-grid coordinates.
1308    ///
1309    /// The coordinate-addressed twin of
1310    /// [`write_chunk_raw`](Self::write_chunk_raw), and the form a
1311    /// v2-B-tree-indexed dataset needs: with two or more unlimited dimensions
1312    /// there is no fixed chunk grid for a linear index to mean anything against.
1313    /// As with `write_chunk_at`, the dataset's logical dimensions are extended
1314    /// to cover the written chunk.
1315    ///
1316    /// `data` is the already-filtered bytes of one chunk — its length is the
1317    /// *stored* size — and `filter_mask` bit *i* set means filter *i* of the
1318    /// pipeline was **not** applied and must be skipped on read. Pass 0 when the
1319    /// full pipeline already ran upstream.
1320    ///
1321    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
1322    /// has no slot to record a stored size or mask.
1323    pub fn write_chunk_raw_at(
1324        &self,
1325        chunk_coords: &[usize],
1326        data: &[u8],
1327        filter_mask: u32,
1328    ) -> Result<()> {
1329        self.write_chunk_at_inner(
1330            chunk_coords,
1331            ChunkBytes::Prefiltered { data, filter_mask },
1332            "write_chunk_raw_at",
1333        )
1334    }
1335
1336    /// The single owner of coordinate-addressed chunk writes: validates the
1337    /// coordinates, grows the dataspace to cover them, and routes the bytes to
1338    /// whichever chunk index the dataset uses. Whether the filter pipeline runs
1339    /// here or already ran upstream is carried by `bytes`, not by a second copy
1340    /// of this dispatch.
1341    fn write_chunk_at_inner(
1342        &self,
1343        chunk_coords: &[usize],
1344        bytes: ChunkBytes<'_>,
1345        what: &str,
1346    ) -> Result<()> {
1347        match &self.info {
1348            DatasetInfo::Writer {
1349                index,
1350                chunked,
1351                btree2,
1352                fixed_array,
1353                ..
1354            } => {
1355                if !*chunked {
1356                    return Err(Hdf5Error::InvalidState(format!(
1357                        "{what} is only for chunked datasets"
1358                    )));
1359                }
1360                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
1361                let btree2 = *btree2;
1362                let fixed_array = *fixed_array;
1363                let inner = borrow_inner(&self.file_inner);
1364                let writer = match &*inner {
1365                    H5FileInner::Writer(w) => w,
1366                    _ => {
1367                        return Err(Hdf5Error::InvalidState(
1368                            "file is no longer in write mode".into(),
1369                        ))
1370                    }
1371                };
1372                // Whole-operation guard: the dims snapshot, the chunk write
1373                // and the extend below must not interleave with a concurrent
1374                // same-dataset operation.
1375                let cell = writer.ds(*index);
1376                let _op = cell.op.lock();
1377                let chunk_dims = writer
1378                    .dataset_chunk_dims(*index)
1379                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1380                    .to_vec();
1381                let dims = writer.dataset_dims(*index).to_vec();
1382                if coords.len() != dims.len() {
1383                    return Err(Hdf5Error::InvalidState(format!(
1384                        "chunk_coords has {} entries but the dataset has {} dimensions",
1385                        coords.len(),
1386                        dims.len()
1387                    )));
1388                }
1389                if chunk_dims.len() != dims.len() {
1390                    return Err(Hdf5Error::InvalidState(format!(
1391                        "dataset chunk shape has {} dimensions but the dataspace has {}",
1392                        chunk_dims.len(),
1393                        dims.len()
1394                    )));
1395                }
1396
1397                // Validate coordinates and compute the grown dimensions
1398                // up-front, before any chunk is written, so an overflowing
1399                // coordinate cannot leave an orphaned chunk in the file.
1400                let mut new_dims = dims.clone();
1401                for d in 0..dims.len() {
1402                    let needed = coords[d]
1403                        .checked_add(1)
1404                        .and_then(|c| c.checked_mul(chunk_dims[d]))
1405                        .ok_or_else(|| {
1406                            Hdf5Error::InvalidState(format!(
1407                                "chunk coordinate {} in dimension {} is too large",
1408                                coords[d], d
1409                            ))
1410                        })?;
1411                    if needed > new_dims[d] {
1412                        new_dims[d] = needed;
1413                    }
1414                }
1415
1416                if fixed_array {
1417                    // Fixed-array (fixed-shape) dataset: no dimension growth.
1418                    match bytes {
1419                        ChunkBytes::Unfiltered(data) => {
1420                            writer.write_chunk_fixed_array_inner(*index, &coords, data)?
1421                        }
1422                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1423                            .write_compressed_chunk_fixed_array_inner(
1424                                *index,
1425                                &coords,
1426                                data,
1427                                filter_mask,
1428                            )?,
1429                    }
1430                    return Ok(());
1431                }
1432
1433                if btree2 {
1434                    match bytes {
1435                        ChunkBytes::Unfiltered(data) => {
1436                            writer.write_chunk_btree_v2_inner(*index, &coords, data)?
1437                        }
1438                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1439                            .write_compressed_chunk_btree_v2_inner(
1440                                *index,
1441                                &coords,
1442                                data,
1443                                filter_mask,
1444                            )?,
1445                    }
1446                } else {
1447                    // Extensible array: the chunk's index-grid slot (row-major
1448                    // against the maximum extent).
1449                    let linear = writer.chunk_slot(*index, &coords)?;
1450                    match bytes {
1451                        ChunkBytes::Unfiltered(data) => {
1452                            writer.write_chunk_inner(*index, linear, data)?
1453                        }
1454                        ChunkBytes::Prefiltered { data, filter_mask } => writer
1455                            .write_compressed_chunk_inner(*index, linear, data, filter_mask)?,
1456                    }
1457                }
1458
1459                if new_dims != dims {
1460                    writer.extend_dataset_inner(*index, &new_dims)?;
1461                }
1462                Ok(())
1463            }
1464            DatasetInfo::Reader { .. } => {
1465                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1466            }
1467        }
1468    }
1469
1470    /// Write multiple chunks in a batch, optionally compressing in parallel.
1471    ///
1472    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
1473    /// pipeline is configured and the `parallel` feature is enabled, all
1474    /// chunks are compressed concurrently via rayon.
1475    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
1476        match &self.info {
1477            DatasetInfo::Writer { index, chunked, .. } => {
1478                if !*chunked {
1479                    return Err(Hdf5Error::InvalidState(
1480                        "write_chunks_batch is only for chunked datasets".into(),
1481                    ));
1482                }
1483                let pairs: Vec<(u64, &[u8])> = chunks
1484                    .iter()
1485                    .map(|(idx, data)| (*idx as u64, *data))
1486                    .collect();
1487                let inner = borrow_inner(&self.file_inner);
1488                match &*inner {
1489                    H5FileInner::Writer(writer) => {
1490                        writer.write_chunks_batch(*index, &pairs)?;
1491                        Ok(())
1492                    }
1493                    _ => Err(Hdf5Error::InvalidState(
1494                        "file is no longer in write mode".into(),
1495                    )),
1496                }
1497            }
1498            DatasetInfo::Reader { .. } => {
1499                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1500            }
1501        }
1502    }
1503
1504    /// Append data along the first dimension of a chunked dataset.
1505    ///
1506    /// `data` must contain a whole number of "frames" — slices along
1507    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
1508    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
1509    /// multiple of `H * W`.
1510    ///
1511    /// This method writes the necessary chunks and extends the dataset
1512    /// shape automatically.
1513    ///
1514    /// ```no_run
1515    /// # use rust_hdf5::H5File;
1516    /// let file = H5File::create("append.h5").unwrap();
1517    /// let ds = file.new_dataset::<f64>()
1518    ///     .shape(&[0, 3])
1519    ///     .chunk(&[1, 3])
1520    ///     .max_shape(&[None, Some(3)])
1521    ///     .create("data")
1522    ///     .unwrap();
1523    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
1524    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
1525    /// ```
1526    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
1527        match &self.info {
1528            DatasetInfo::Writer {
1529                index,
1530                element_size,
1531                chunked,
1532                ..
1533            } => {
1534                if !*chunked {
1535                    return Err(Hdf5Error::InvalidState(
1536                        "append is only for chunked datasets".into(),
1537                    ));
1538                }
1539                if T::element_size() != *element_size {
1540                    return Err(Hdf5Error::TypeMismatch(format!(
1541                        "append type has element size {} but dataset expects {}",
1542                        T::element_size(),
1543                        element_size,
1544                    )));
1545                }
1546
1547                let ds_index = *index;
1548                let es = *element_size;
1549
1550                let inner = borrow_inner(&self.file_inner);
1551                let writer = match &*inner {
1552                    H5FileInner::Writer(w) => w,
1553                    _ => {
1554                        return Err(Hdf5Error::InvalidState(
1555                            "file is no longer in write mode".into(),
1556                        ))
1557                    }
1558                };
1559
1560                // Whole-operation guard: the buffer take, the frame writes,
1561                // the re-buffer and the extend below are separate slot
1562                // acquisitions that a concurrent same-dataset append must not
1563                // interleave with.
1564                let cell = writer.ds(ds_index);
1565                let _op = cell.op.lock();
1566
1567                let chunk_dims = writer
1568                    .dataset_chunk_dims(ds_index)
1569                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1570                    .to_vec();
1571                let dims = writer.dataset_dims(ds_index).to_vec();
1572
1573                // Frame size = product of dims[1..]
1574                let frame_elems: usize = if dims.len() > 1 {
1575                    dims[1..].iter().map(|&d| d as usize).product()
1576                } else {
1577                    1
1578                };
1579
1580                if frame_elems == 0 {
1581                    return Err(Hdf5Error::InvalidState(
1582                        "cannot append to dataset with zero-size trailing dimensions".into(),
1583                    ));
1584                }
1585
1586                if !data.len().is_multiple_of(frame_elems) {
1587                    return Err(Hdf5Error::InvalidState(format!(
1588                        "data length {} is not a multiple of frame size {}",
1589                        data.len(),
1590                        frame_elems,
1591                    )));
1592                }
1593
1594                let n_new_frames = data.len() / frame_elems;
1595                let current_dim0 = dims[0] as usize;
1596
1597                // Chunk size along first dimension
1598                let chunk_dim0 = chunk_dims[0] as usize;
1599                let frame_bytes = frame_elems * es;
1600
1601                let raw = unsafe {
1602                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
1603                };
1604
1605                // Merge the buffer with the new frames when it is the
1606                // dataset's tail; a buffer left mid-extent (the extent moved
1607                // past it) keeps its recorded place — flush it and start
1608                // fresh at the current end.
1609                let taken = { writer.ds(ds_index).lock().append.take() };
1610                let (base_dim0, buffered_frames, mut combined) = match taken {
1611                    Some(b) if b.base + b.frames == current_dim0 as u64 => {
1612                        (b.base as usize, b.frames as usize, b.bytes)
1613                    }
1614                    Some(b) => {
1615                        writer.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
1616                        (current_dim0, 0, Vec::new())
1617                    }
1618                    None => (current_dim0, 0, Vec::new()),
1619                };
1620                combined.extend_from_slice(raw);
1621
1622                let total_frames = buffered_frames + n_new_frames;
1623
1624                // Rows up to the last chunk boundary are written now; the
1625                // tail that does not complete a chunk goes back in the
1626                // buffer for the next append (or the flush at close). The
1627                // boundary can precede `base_dim0` — a reopened file's
1628                // flushed partial chunk leaves the base mid-chunk — in
1629                // which case everything is tail.
1630                let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
1631                let write_frames = last_boundary.saturating_sub(base_dim0);
1632                let tail_frames = total_frames - write_frames;
1633                if write_frames > 0 {
1634                    writer.write_append_frames(
1635                        ds_index,
1636                        base_dim0 as u64,
1637                        write_frames as u64,
1638                        &combined[..write_frames * frame_bytes],
1639                    )?;
1640                }
1641                if tail_frames > 0 {
1642                    let ds = writer.ds(ds_index);
1643                    let mut m = ds.lock();
1644                    m.append = Some(crate::io::writer::AppendBuffer {
1645                        base: (base_dim0 + write_frames) as u64,
1646                        frames: tail_frames as u64,
1647                        bytes: combined[write_frames * frame_bytes..].to_vec(),
1648                    });
1649                }
1650
1651                // Extend dims to include all frames (buffered + new)
1652                let logical_dim0 = base_dim0 + total_frames;
1653                let mut new_dims: Vec<u64> = dims;
1654                new_dims[0] = logical_dim0 as u64;
1655                writer.extend_dataset_inner(ds_index, &new_dims)?;
1656
1657                Ok(())
1658            }
1659            DatasetInfo::Reader { .. } => {
1660                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
1661            }
1662        }
1663    }
1664
1665    /// Extend the dimensions of a chunked dataset.
1666    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
1667        match &self.info {
1668            DatasetInfo::Writer { index, chunked, .. } => {
1669                if !*chunked {
1670                    return Err(Hdf5Error::InvalidState(
1671                        "extend is only for chunked datasets".into(),
1672                    ));
1673                }
1674
1675                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1676                let inner = borrow_inner(&self.file_inner);
1677                match &*inner {
1678                    H5FileInner::Writer(writer) => {
1679                        writer.extend_dataset(*index, &dims_u64)?;
1680                        Ok(())
1681                    }
1682                    _ => Err(Hdf5Error::InvalidState(
1683                        "file is no longer in write mode".into(),
1684                    )),
1685                }
1686            }
1687            DatasetInfo::Reader { .. } => {
1688                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
1689            }
1690        }
1691    }
1692
1693    /// Set the logical extent of a chunked dataset, growing **or
1694    /// shrinking** any dimension.
1695    ///
1696    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
1697    /// dimension — for example to correct an over-extended frame count
1698    /// after writing a partial multi-frame chunk. Shrinking changes the
1699    /// logical dataspace only: data in chunks beyond the new extent stays
1700    /// in the file but is no longer visible on read, exactly as libhdf5's
1701    /// `H5Dset_extent` behaves. The new extent must not exceed the
1702    /// dataset's maximum dimensions.
1703    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
1704        match &self.info {
1705            DatasetInfo::Writer { index, .. } => {
1706                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1707                let inner = borrow_inner(&self.file_inner);
1708                match &*inner {
1709                    H5FileInner::Writer(writer) => {
1710                        writer.set_dataset_extent(*index, &dims_u64)?;
1711                        Ok(())
1712                    }
1713                    _ => Err(Hdf5Error::InvalidState(
1714                        "file is no longer in write mode".into(),
1715                    )),
1716                }
1717            }
1718            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
1719                "cannot set extent in read mode".into(),
1720            )),
1721        }
1722    }
1723
1724    /// Flush a chunked dataset's index structures to disk.
1725    pub fn flush(&self) -> Result<()> {
1726        match &self.info {
1727            DatasetInfo::Writer { index, .. } => {
1728                let inner = borrow_inner(&self.file_inner);
1729                match &*inner {
1730                    H5FileInner::Writer(writer) => {
1731                        writer.flush_dataset(*index)?;
1732                        Ok(())
1733                    }
1734                    _ => Ok(()),
1735                }
1736            }
1737            DatasetInfo::Reader { .. } => Ok(()),
1738        }
1739    }
1740
1741    /// Read a slice (hyperslab) of the dataset as a typed vector.
1742    ///
1743    /// `starts` and `counts` define the N-dimensional selection:
1744    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
1745    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
1746        match &self.info {
1747            DatasetInfo::Reader {
1748                name, element_size, ..
1749            } => {
1750                if T::element_size() != *element_size {
1751                    return Err(Hdf5Error::TypeMismatch(format!(
1752                        "read type has element size {} but dataset has element size {}",
1753                        T::element_size(),
1754                        element_size,
1755                    )));
1756                }
1757                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1758                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1759
1760                let raw = {
1761                    let mut inner = borrow_inner_mut(&self.file_inner);
1762                    match &mut *inner {
1763                        H5FileInner::Reader(reader) => {
1764                            reader.read_slice(name, &starts_u64, &counts_u64)?
1765                        }
1766                        _ => {
1767                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
1768                        }
1769                    }
1770                };
1771
1772                if raw.len() % T::element_size() != 0 {
1773                    return Err(Hdf5Error::TypeMismatch(format!(
1774                        "raw data size {} is not a multiple of element size {}",
1775                        raw.len(),
1776                        T::element_size(),
1777                    )));
1778                }
1779
1780                let count = raw.len() / T::element_size();
1781                let mut result = Vec::<T>::with_capacity(count);
1782                unsafe {
1783                    std::ptr::copy_nonoverlapping(
1784                        raw.as_ptr(),
1785                        result.as_mut_ptr() as *mut u8,
1786                        raw.len(),
1787                    );
1788                    result.set_len(count);
1789                }
1790                Ok(result)
1791            }
1792            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1793                "cannot read_slice from a dataset in write mode".into(),
1794            )),
1795        }
1796    }
1797
1798    /// Write a typed slice to a sub-region of the dataset.
1799    ///
1800    /// `starts` and `counts` define the N-dimensional selection, which must lie
1801    /// inside the dataset's current extent.
1802    ///
1803    /// Works for both contiguous and chunked datasets. For a chunked dataset
1804    /// only the chunks the selection touches are rewritten — a partially
1805    /// covered chunk is read back, patched, and written again, so updating one
1806    /// row of an appendable dataset costs the chunks that row crosses rather
1807    /// than the whole dataset. Elements of a touched chunk that the selection
1808    /// does not cover keep their stored value, or the dataset's fill value if
1809    /// the chunk did not exist yet.
1810    pub fn write_slice<T: H5Type>(
1811        &self,
1812        starts: &[usize],
1813        counts: &[usize],
1814        data: &[T],
1815    ) -> Result<()> {
1816        match &self.info {
1817            DatasetInfo::Writer {
1818                index,
1819                element_size,
1820                ..
1821            } => {
1822                if T::element_size() != *element_size {
1823                    return Err(Hdf5Error::TypeMismatch(format!(
1824                        "write type has element size {} but dataset expects {}",
1825                        T::element_size(),
1826                        element_size,
1827                    )));
1828                }
1829
1830                let expected: usize = counts.iter().product();
1831                if data.len() != expected {
1832                    return Err(Hdf5Error::InvalidState(format!(
1833                        "data length {} does not match slice size {}",
1834                        data.len(),
1835                        expected,
1836                    )));
1837                }
1838
1839                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1840                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1841
1842                let byte_len = data.len() * T::element_size();
1843                let raw =
1844                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
1845
1846                let inner = borrow_inner(&self.file_inner);
1847                match &*inner {
1848                    H5FileInner::Writer(writer) => {
1849                        writer.write_slice(*index, &starts_u64, &counts_u64, raw)?;
1850                        Ok(())
1851                    }
1852                    _ => Err(Hdf5Error::InvalidState(
1853                        "file is no longer in write mode".into(),
1854                    )),
1855                }
1856            }
1857            DatasetInfo::Reader { .. } => {
1858                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1859            }
1860        }
1861    }
1862
1863    /// Replace elements `start .. start + strings.len()` of a 1-D
1864    /// variable-length string dataset.
1865    ///
1866    /// The extent and every element outside the range are left alone, and the
1867    /// cost is the new strings plus the chunks holding their references — not
1868    /// the column. The dataset's character set is enforced: a non-ASCII
1869    /// replacement in a dataset that declares ASCII is rejected rather than
1870    /// stored under a datatype that misdescribes it.
1871    ///
1872    /// The global heap objects the replaced references pointed at are freed —
1873    /// the same reclaim libhdf5 performs on an overwrite — so updating one
1874    /// element repeatedly reuses space rather than growing the file. A
1875    /// collection emptied by the update returns its block to the allocator.
1876    /// Under SWMR nothing is freed, because a reader may still be following
1877    /// those references.
1878    ///
1879    /// ```no_run
1880    /// # use rust_hdf5::H5File;
1881    /// let file = H5File::open_rw("meta.h5").unwrap();
1882    /// let ds = file.dataset_writer("notes").unwrap();
1883    /// ds.write_vlen_strings_slice(42, &["replacement"]).unwrap();
1884    /// file.close().unwrap();
1885    /// ```
1886    pub fn write_vlen_strings_slice(&self, start: usize, strings: &[&str]) -> Result<()> {
1887        match &self.info {
1888            DatasetInfo::Writer { index, .. } => {
1889                let inner = borrow_inner(&self.file_inner);
1890                match &*inner {
1891                    H5FileInner::Writer(writer) => {
1892                        writer.write_vlen_strings_slice(*index, start as u64, strings)?;
1893                        Ok(())
1894                    }
1895                    _ => Err(Hdf5Error::InvalidState(
1896                        "file is no longer in write mode".into(),
1897                    )),
1898                }
1899            }
1900            DatasetInfo::Reader { .. } => {
1901                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1902            }
1903        }
1904    }
1905
1906    /// Read variable-length strings from a dataset.
1907    ///
1908    /// This handles h5py-style vlen string datasets that store strings
1909    /// as global heap references. Returns one String per element.
1910    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
1911        match &self.info {
1912            DatasetInfo::Reader { name, .. } => {
1913                let mut inner = borrow_inner_mut(&self.file_inner);
1914                match &mut *inner {
1915                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
1916                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1917                }
1918            }
1919            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1920                "cannot read vlen strings from a dataset in write mode".into(),
1921            )),
1922        }
1923    }
1924
1925    /// Read variable-length byte arrays from a dataset.
1926    ///
1927    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
1928    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
1929    /// that store each element as a global heap reference. Returns one
1930    /// `Vec<u8>` per element.
1931    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
1932        match &self.info {
1933            DatasetInfo::Reader { name, .. } => {
1934                let mut inner = borrow_inner_mut(&self.file_inner);
1935                match &mut *inner {
1936                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
1937                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1938                }
1939            }
1940            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1941                "cannot read vlen bytes from a dataset in write mode".into(),
1942            )),
1943        }
1944    }
1945
1946    /// Read a string dataset, fixed-width or variable-length, as one `String`
1947    /// per element.
1948    ///
1949    /// The width of a `FixedString` dataset is whatever the file says, so a
1950    /// 24-byte label column and a 100-byte one are read by the same call. The
1951    /// padding rule the datatype declares decides where each element ends —
1952    /// null-terminated (0), null-padded (1) or space-padded (2) — and its
1953    /// character set decides how the remaining bytes are decoded: ASCII (0)
1954    /// requires 7-bit bytes, UTF-8 (1) requires valid UTF-8. An element that
1955    /// violates either is an error naming the element, not a silent
1956    /// substitution; [`read_strings_lossy`](Self::read_strings_lossy) is the
1957    /// call that accepts such a file, replacing what it cannot decode.
1958    ///
1959    /// ```no_run
1960    /// # use rust_hdf5::H5File;
1961    /// let file = H5File::open("labels.h5").unwrap();
1962    /// let labels = file.dataset("names").unwrap().read_strings().unwrap();
1963    /// ```
1964    pub fn read_strings(&self) -> Result<Vec<String>> {
1965        self.read_strings_inner(false)
1966    }
1967
1968    /// [`read_strings`](Self::read_strings), but bytes that do not decode
1969    /// under the dataset's character set become U+FFFD instead of an error.
1970    ///
1971    /// Producers do mislabel the character set — a file that declares ASCII
1972    /// while storing Latin-1 or UTF-8 bytes reads here and not there.
1973    pub fn read_strings_lossy(&self) -> Result<Vec<String>> {
1974        self.read_strings_inner(true)
1975    }
1976
1977    /// The single owner of string decoding for both string datatypes: the
1978    /// element bytes are found differently, the padding and character-set
1979    /// rules that turn them into a `String` are the same.
1980    fn read_strings_inner(&self, lossy: bool) -> Result<Vec<String>> {
1981        if matches!(self.info, DatasetInfo::Writer { .. }) {
1982            return Err(Hdf5Error::InvalidState(
1983                "cannot read strings from a dataset in write mode".into(),
1984            ));
1985        }
1986        match self.datatype()? {
1987            DatatypeMessage::VarLenString { charset } => self
1988                .read_vlen_bytes()?
1989                .iter()
1990                .enumerate()
1991                .map(|(i, bytes)| decode_string(bytes, charset, lossy, i))
1992                .collect(),
1993            DatatypeMessage::FixedString {
1994                size,
1995                padding,
1996                charset,
1997            } => {
1998                let width = size as usize;
1999                if width == 0 {
2000                    // A corrupt file can declare it; `chunks_exact(0)` panics.
2001                    return Err(Hdf5Error::InvalidState(
2002                        "fixed-string datatype has zero width".into(),
2003                    ));
2004                }
2005                // `read_raw_bytes` returns `product(dims) * width` bytes, so
2006                // `chunks_exact` leaves no remainder.
2007                let raw = self.read_raw_bytes()?;
2008                raw.chunks_exact(width)
2009                    .enumerate()
2010                    .map(|(i, elem)| {
2011                        decode_string(trim_fixed_string(elem, padding, i)?, charset, lossy, i)
2012                    })
2013                    .collect()
2014            }
2015            other => Err(Hdf5Error::InvalidState(format!(
2016                "read_strings is only for string datasets, this one is {other:?}"
2017            ))),
2018        }
2019    }
2020
2021    /// Read the entire dataset as a typed vector.
2022    ///
2023    /// The raw bytes are read from the file and reinterpreted as `T`. The
2024    /// caller must ensure that `T` matches the datatype used when the dataset
2025    /// was written.
2026    ///
2027    /// # Errors
2028    ///
2029    /// Returns an error if:
2030    /// - The file is in write mode.
2031    /// - The raw data size is not a multiple of `T::element_size()`.
2032    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
2033        match &self.info {
2034            DatasetInfo::Reader {
2035                name, element_size, ..
2036            } => {
2037                if T::element_size() != *element_size {
2038                    return Err(Hdf5Error::TypeMismatch(format!(
2039                        "read type has element size {} but dataset has element size {}",
2040                        T::element_size(),
2041                        element_size,
2042                    )));
2043                }
2044
2045                let raw = {
2046                    let mut inner = borrow_inner_mut(&self.file_inner);
2047                    match &mut *inner {
2048                        H5FileInner::Reader(reader) => reader.read_dataset_raw(name)?,
2049                        _ => {
2050                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
2051                        }
2052                    }
2053                };
2054
2055                if raw.len() % T::element_size() != 0 {
2056                    return Err(Hdf5Error::TypeMismatch(format!(
2057                        "raw data size {} is not a multiple of element size {}",
2058                        raw.len(),
2059                        T::element_size(),
2060                    )));
2061                }
2062
2063                let count = raw.len() / T::element_size();
2064                let mut result = Vec::<T>::with_capacity(count);
2065
2066                // Safety: T is Copy + 'static (required by H5Type). We verified
2067                // the byte count matches count * size_of::<T>() above.
2068                // copy_nonoverlapping fills the memory with valid bit patterns
2069                // for all H5Type implementors (numeric primitives).
2070                // We call set_len AFTER the copy so that if an unexpected panic
2071                // occurs, uninitialized memory is never exposed.
2072                unsafe {
2073                    std::ptr::copy_nonoverlapping(
2074                        raw.as_ptr(),
2075                        result.as_mut_ptr() as *mut u8,
2076                        raw.len(),
2077                    );
2078                    result.set_len(count);
2079                }
2080
2081                Ok(result)
2082            }
2083            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2084                "cannot read from a dataset in write mode".into(),
2085            )),
2086        }
2087    }
2088
2089    /// Read the raw byte image of a dataset without an `H5Type` carrier.
2090    ///
2091    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
2092    /// the element bytes verbatim regardless of the on-disk element type, so a
2093    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
2094    /// no matching Rust primitive can be read back and decoded by the caller.
2095    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
2096        match &self.info {
2097            DatasetInfo::Reader { name, .. } => {
2098                let mut inner = borrow_inner_mut(&self.file_inner);
2099                match &mut *inner {
2100                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
2101                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2102                }
2103            }
2104            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2105                "cannot read from a dataset in write mode".into(),
2106            )),
2107        }
2108    }
2109
2110    /// Read the whole dataset into a caller-provided buffer, with no allocation.
2111    ///
2112    /// `out` must have exactly `product(dims)` elements (the dataset's element
2113    /// count) and `T::element_size()` must match the dataset's on-disk element
2114    /// size, otherwise an error is returned and `out` is left unspecified. The
2115    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
2116    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
2117    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
2118    /// without the extra staging copy a `read_raw` + copy-into-pinned would
2119    /// incur. Works for every layout (contiguous, compact, and chunked under
2120    /// any index); for chunked data each decoded chunk is scattered directly
2121    /// into `out`.
2122    ///
2123    /// ```no_run
2124    /// # use rust_hdf5::H5File;
2125    /// let file = H5File::open("data.h5").unwrap();
2126    /// let ds = file.dataset("frames").unwrap();
2127    /// let n: usize = ds.shape().iter().product();
2128    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
2129    /// ds.read_raw_into(&mut buf).unwrap();
2130    /// ```
2131    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
2132        match &self.info {
2133            DatasetInfo::Reader {
2134                name, element_size, ..
2135            } => {
2136                if T::element_size() != *element_size {
2137                    return Err(Hdf5Error::TypeMismatch(format!(
2138                        "read type has element size {} but dataset has element size {}",
2139                        T::element_size(),
2140                        element_size,
2141                    )));
2142                }
2143                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
2144                // byte representation; every bit pattern the read writes is a
2145                // valid `T`. The byte view borrows `out` exclusively for this
2146                // call, and `out.len() * element_size` cannot overflow because
2147                // it is the byte length of an existing slice (<= isize::MAX).
2148                let byte_len = out.len() * T::element_size();
2149                let bytes = unsafe {
2150                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
2151                };
2152                let mut inner = borrow_inner_mut(&self.file_inner);
2153                match &mut *inner {
2154                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw_into(name, bytes)?),
2155                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2156                }
2157            }
2158            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2159                "cannot read from a dataset in write mode".into(),
2160            )),
2161        }
2162    }
2163
2164    /// Read a hyperslab into a caller-provided buffer, with no allocation.
2165    ///
2166    /// `out` must have exactly `product(counts)` elements and
2167    /// `T::element_size()` must match the dataset's element size. The zero-copy
2168    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
2169    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
2170    /// selection are read, and the selected bytes land directly in `out` — the
2171    /// entry point for reading one frame / block straight into a pinned host
2172    /// buffer for an H2D transfer.
2173    ///
2174    /// ```no_run
2175    /// # use rust_hdf5::H5File;
2176    /// let file = H5File::open("vol.h5").unwrap();
2177    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
2178    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
2179    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
2180    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
2181    /// ```
2182    pub fn read_slice_into<T: H5Type>(
2183        &self,
2184        out: &mut [T],
2185        starts: &[usize],
2186        counts: &[usize],
2187    ) -> Result<()> {
2188        match &self.info {
2189            DatasetInfo::Reader {
2190                name, element_size, ..
2191            } => {
2192                if T::element_size() != *element_size {
2193                    return Err(Hdf5Error::TypeMismatch(format!(
2194                        "read type has element size {} but dataset has element size {}",
2195                        T::element_size(),
2196                        element_size,
2197                    )));
2198                }
2199                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
2200                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
2201                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
2202                // borrow of `out`, byte length within bounds.
2203                let byte_len = out.len() * T::element_size();
2204                let bytes = unsafe {
2205                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
2206                };
2207                let mut inner = borrow_inner_mut(&self.file_inner);
2208                match &mut *inner {
2209                    H5FileInner::Reader(reader) => {
2210                        Ok(reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?)
2211                    }
2212                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2213                }
2214            }
2215            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2216                "cannot read from a dataset in write mode".into(),
2217            )),
2218        }
2219    }
2220}
2221
2222#[cfg(test)]
2223mod tests {
2224    use crate::H5File;
2225    use std::path::PathBuf;
2226
2227    fn temp_path(name: &str) -> PathBuf {
2228        // Include PID + a per-call atomic counter so that concurrent
2229        // cargo invocations and any kernel-level "lock not yet
2230        // released" races between sequential opens cannot collide.
2231        use std::sync::atomic::{AtomicU64, Ordering};
2232        static COUNTER: AtomicU64 = AtomicU64::new(0);
2233        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
2234        std::env::temp_dir().join(format!(
2235            "hdf5_dataset_test_{}_{}_{}.h5",
2236            name,
2237            std::process::id(),
2238            n
2239        ))
2240    }
2241
2242    #[test]
2243    fn runtime_compound_via_datatype_override_and_raw_bytes() {
2244        use crate::format::messages::datatype::DatatypeMessage;
2245        use crate::types::{CompoundType, H5Type};
2246
2247        let path = temp_path("compound_raw");
2248        // A 12-byte packed compound with NO matching Rust primitive carrier,
2249        // so it can only be written through the datatype() override +
2250        // write_raw_bytes path (the runtime-CompoundType use case).
2251        let ct = CompoundType {
2252            members: vec![
2253                ("id".to_string(), i32::hdf5_type(), 0),
2254                ("val".to_string(), f64::hdf5_type(), 4),
2255            ],
2256            total_size: 12,
2257        };
2258        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
2259        let mut bytes = Vec::new();
2260        for (id, val) in recs {
2261            bytes.extend_from_slice(&id.to_le_bytes());
2262            bytes.extend_from_slice(&val.to_le_bytes());
2263        }
2264
2265        {
2266            let file = H5File::create(&path).unwrap();
2267            let ds = file
2268                .new_dataset::<u8>()
2269                .datatype(ct.to_datatype())
2270                .shape([recs.len()])
2271                .create("records")
2272                .unwrap();
2273            ds.write_raw_bytes(&bytes).unwrap();
2274            file.close().unwrap();
2275        }
2276        {
2277            let file = H5File::open(&path).unwrap();
2278            let ds = file.dataset("records").unwrap();
2279            // The on-disk element type is the compound we specified (size 12),
2280            // not the u8 carrier.
2281            match ds.datatype().unwrap() {
2282                DatatypeMessage::Compound { size, members } => {
2283                    assert_eq!(size, 12);
2284                    assert_eq!(members.len(), 2);
2285                    assert_eq!(members[0].name, "id");
2286                    assert_eq!(members[0].offset, 0);
2287                    assert_eq!(members[1].name, "val");
2288                    assert_eq!(members[1].offset, 4);
2289                }
2290                other => panic!("expected compound datatype, got {other:?}"),
2291            }
2292            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
2293        }
2294        std::fs::remove_file(&path).ok();
2295    }
2296
2297    #[test]
2298    fn builder_requires_shape() {
2299        let path = temp_path("no_shape");
2300        let file = H5File::create(&path).unwrap();
2301        let result = file.new_dataset::<u8>().create("data");
2302        assert!(result.is_err());
2303        std::fs::remove_file(&path).ok();
2304    }
2305
2306    #[test]
2307    fn write_raw_size_mismatch() {
2308        let path = temp_path("size_mismatch");
2309        let file = H5File::create(&path).unwrap();
2310        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
2311        // Provide 3 elements instead of 4
2312        let result = ds.write_raw(&[1u8, 2, 3]);
2313        assert!(result.is_err());
2314        std::fs::remove_file(&path).ok();
2315    }
2316
2317    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
2318    // dataset = one chunk) rather than silently drop the filter on the
2319    // contiguous path. write_raw then populates that single chunk.
2320    #[cfg(feature = "deflate")]
2321    #[test]
2322    fn filter_without_chunk_autochunks_and_roundtrips() {
2323        let path = temp_path("autochunk_filter");
2324        let data: Vec<i32> = (0..8).collect();
2325        {
2326            let file = H5File::create(&path).unwrap();
2327            let ds = file
2328                .new_dataset::<i32>()
2329                .deflate(6)
2330                .shape([8])
2331                .create("seq")
2332                .unwrap();
2333            ds.write_raw(&data).unwrap();
2334            file.close().unwrap();
2335        }
2336        {
2337            let file = H5File::open(&path).unwrap();
2338            let ds = file.dataset("seq").unwrap();
2339            // The filter forced chunked storage: a single whole-dataset chunk.
2340            assert!(
2341                ds.is_chunked(),
2342                "auto-chunk did not produce chunked storage"
2343            );
2344            assert_eq!(ds.chunk_dims(), Some(vec![8]));
2345            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2346        }
2347        std::fs::remove_file(&path).ok();
2348    }
2349
2350    // B: write_raw on an explicitly chunked + compressed dataset scatters the
2351    // full row-major image across a multi-chunk grid, including edge chunks
2352    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
2353    #[cfg(feature = "deflate")]
2354    #[test]
2355    fn write_raw_multichunk_edge_roundtrips() {
2356        let path = temp_path("multichunk_edge");
2357        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
2358        {
2359            let file = H5File::create(&path).unwrap();
2360            let ds = file
2361                .new_dataset::<i32>()
2362                .shape([7, 5])
2363                .chunk(&[3, 2])
2364                .deflate(4)
2365                .create("grid")
2366                .unwrap();
2367            ds.write_raw(&data).unwrap();
2368            file.close().unwrap();
2369        }
2370        {
2371            let file = H5File::open(&path).unwrap();
2372            let ds = file.dataset("grid").unwrap();
2373            assert_eq!(ds.shape(), vec![7, 5]);
2374            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2375            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2376        }
2377        std::fs::remove_file(&path).ok();
2378    }
2379
2380    // B: write_raw on an unfiltered chunked dataset (previously rejected with
2381    // "use write_chunk for chunked datasets") now gathers and round-trips.
2382    #[test]
2383    fn write_raw_unfiltered_chunked_roundtrips() {
2384        let path = temp_path("chunked_unfiltered");
2385        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
2386        {
2387            let file = H5File::create(&path).unwrap();
2388            let ds = file
2389                .new_dataset::<f64>()
2390                .shape([4, 3])
2391                .chunk(&[2, 2])
2392                .create("m")
2393                .unwrap();
2394            ds.write_raw(&data).unwrap();
2395            file.close().unwrap();
2396        }
2397        {
2398            let file = H5File::open(&path).unwrap();
2399            let ds = file.dataset("m").unwrap();
2400            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
2401            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
2402        }
2403        std::fs::remove_file(&path).ok();
2404    }
2405
2406    // write_raw on an extensible-array (unlimited first dim) compressed dataset
2407    // drives write_full_image_chunked's EA branch, which gathers chunks and
2408    // compresses them through the windowed batch path. Round-trips the full
2409    // image, including a partial edge chunk along the unlimited dimension.
2410    #[cfg(feature = "deflate")]
2411    #[test]
2412    fn write_raw_ea_compressed_roundtrips() {
2413        let path = temp_path("write_raw_ea_deflate");
2414        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
2415        {
2416            let file = H5File::create(&path).unwrap();
2417            let ds = file
2418                .new_dataset::<i32>()
2419                .shape([5, 4])
2420                .chunk(&[2, 4])
2421                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
2422                .deflate(5)
2423                .create("stream")
2424                .unwrap();
2425            ds.write_raw(&data).unwrap();
2426            file.close().unwrap();
2427        }
2428        {
2429            let file = H5File::open(&path).unwrap();
2430            let ds = file.dataset("stream").unwrap();
2431            assert_eq!(ds.shape(), vec![5, 4]);
2432            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2433        }
2434        std::fs::remove_file(&path).ok();
2435    }
2436
2437    // 3D chunked Full read with partial-edge chunks in every dimension. This
2438    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
2439    // dimensions, exercising the nested outer-coordinate carry and the
2440    // last-axis edge clamp (chunks hang off the high edge in all three axes).
2441    #[cfg(feature = "deflate")]
2442    #[test]
2443    fn read_full_3d_chunked_edge_roundtrips() {
2444        let path = temp_path("full_3d_chunked_edge");
2445        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
2446        // along each axis is partial (1, 1, and 1 element respectively).
2447        let total: usize = 5 * 4 * 3;
2448        let data: Vec<i32> = (0..total as i32).collect();
2449        {
2450            let file = H5File::create(&path).unwrap();
2451            let ds = file
2452                .new_dataset::<i32>()
2453                .shape([5, 4, 3])
2454                .chunk(&[2, 3, 2])
2455                .deflate(4)
2456                .create("vol")
2457                .unwrap();
2458            ds.write_raw(&data).unwrap();
2459            file.close().unwrap();
2460        }
2461        {
2462            let file = H5File::open(&path).unwrap();
2463            let ds = file.dataset("vol").unwrap();
2464            assert_eq!(ds.shape(), vec![5, 4, 3]);
2465            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
2466            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2467        }
2468        std::fs::remove_file(&path).ok();
2469    }
2470
2471    #[test]
2472    fn roundtrip_u8_1d() {
2473        let path = temp_path("rt_u8_1d");
2474        let data: Vec<u8> = (0..10).collect();
2475
2476        {
2477            let file = H5File::create(&path).unwrap();
2478            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
2479            ds.write_raw(&data).unwrap();
2480            file.close().unwrap();
2481        }
2482
2483        {
2484            let file = H5File::open(&path).unwrap();
2485            let ds = file.dataset("seq").unwrap();
2486            assert_eq!(ds.shape(), vec![10]);
2487            let readback = ds.read_raw::<u8>().unwrap();
2488            assert_eq!(readback, data);
2489        }
2490
2491        std::fs::remove_file(&path).ok();
2492    }
2493
2494    #[test]
2495    fn roundtrip_i32_2d() {
2496        let path = temp_path("rt_i32_2d");
2497        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
2498
2499        {
2500            let file = H5File::create(&path).unwrap();
2501            let ds = file
2502                .new_dataset::<i32>()
2503                .shape([2, 3])
2504                .create("matrix")
2505                .unwrap();
2506            ds.write_raw(&data).unwrap();
2507            file.close().unwrap();
2508        }
2509
2510        {
2511            let file = H5File::open(&path).unwrap();
2512            let ds = file.dataset("matrix").unwrap();
2513            assert_eq!(ds.shape(), vec![2, 3]);
2514            let readback = ds.read_raw::<i32>().unwrap();
2515            assert_eq!(readback, data);
2516        }
2517
2518        std::fs::remove_file(&path).ok();
2519    }
2520
2521    #[test]
2522    fn roundtrip_f64_3d() {
2523        let path = temp_path("rt_f64_3d");
2524        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
2525
2526        {
2527            let file = H5File::create(&path).unwrap();
2528            let ds = file
2529                .new_dataset::<f64>()
2530                .shape([2, 3, 4])
2531                .create("cube")
2532                .unwrap();
2533            ds.write_raw(&data).unwrap();
2534            file.close().unwrap();
2535        }
2536
2537        {
2538            let file = H5File::open(&path).unwrap();
2539            let ds = file.dataset("cube").unwrap();
2540            assert_eq!(ds.shape(), vec![2, 3, 4]);
2541            let readback = ds.read_raw::<f64>().unwrap();
2542            assert_eq!(readback, data);
2543        }
2544
2545        std::fs::remove_file(&path).ok();
2546    }
2547
2548    #[test]
2549    fn cannot_read_in_write_mode() {
2550        let path = temp_path("no_read_write");
2551        let file = H5File::create(&path).unwrap();
2552        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2553        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2554        let result = ds.read_raw::<u8>();
2555        assert!(result.is_err());
2556        std::fs::remove_file(&path).ok();
2557    }
2558
2559    #[test]
2560    fn cannot_write_in_read_mode() {
2561        let path = temp_path("no_write_read");
2562
2563        {
2564            let file = H5File::create(&path).unwrap();
2565            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2566            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2567            file.close().unwrap();
2568        }
2569
2570        {
2571            let file = H5File::open(&path).unwrap();
2572            let ds = file.dataset("x").unwrap();
2573            let result = ds.write_raw(&[5u8, 6, 7, 8]);
2574            assert!(result.is_err());
2575        }
2576
2577        std::fs::remove_file(&path).ok();
2578    }
2579
2580    #[test]
2581    fn numeric_attr_roundtrip() {
2582        let path = temp_path("num_attr");
2583        {
2584            let file = H5File::create(&path).unwrap();
2585            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2586            ds.write_raw(&[1.0f32; 4]).unwrap();
2587
2588            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
2589            a1.write_numeric(&1.2345f64).unwrap();
2590
2591            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
2592            a2.write_numeric(&42i32).unwrap();
2593
2594            file.close().unwrap();
2595        }
2596        {
2597            let file = H5File::open(&path).unwrap();
2598            let ds = file.dataset("data").unwrap();
2599
2600            let scale = ds.attr("scale").unwrap();
2601            let val: f64 = scale.read_numeric().unwrap();
2602            assert!((val - 1.2345).abs() < 1e-10);
2603
2604            let count = ds.attr("count").unwrap();
2605            let val: i32 = count.read_numeric().unwrap();
2606            assert_eq!(val, 42);
2607        }
2608        std::fs::remove_file(&path).ok();
2609    }
2610
2611    #[test]
2612    fn array_attr_roundtrip() {
2613        let path = temp_path("array_attr");
2614        let offsets = [10i32, -20, 30];
2615        {
2616            let file = H5File::create(&path).unwrap();
2617            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2618            ds.write_raw(&[1.0f32; 4]).unwrap();
2619
2620            // 1-D int32 array attribute (NDArrayDimOffset-style).
2621            let a = ds
2622                .new_attr::<i32>()
2623                .shape([3])
2624                .create("dim_offset")
2625                .unwrap();
2626            a.write_array(&offsets).unwrap();
2627
2628            // Wrong element count is rejected.
2629            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
2630            assert!(bad.write_array(&[1i32, 2]).is_err());
2631
2632            file.close().unwrap();
2633        }
2634        {
2635            let file = H5File::open(&path).unwrap();
2636            let ds = file.dataset("data").unwrap();
2637            let a = ds.attr("dim_offset").unwrap();
2638            let raw = a.read_raw().unwrap();
2639            assert_eq!(raw.len(), 3 * 4);
2640            let got: Vec<i32> = raw
2641                .chunks_exact(4)
2642                .map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
2643                .collect();
2644            assert_eq!(got, offsets);
2645        }
2646        std::fs::remove_file(&path).ok();
2647    }
2648
2649    #[test]
2650    fn attr_datatype_exposes_class_and_sign() {
2651        // H5Attribute::datatype() must report the stored datatype class and
2652        // signedness so a generic attr->metadata mapper need not infer it from
2653        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
2654        use crate::format::messages::datatype::DatatypeMessage;
2655
2656        let path = temp_path("attr_datatype");
2657        {
2658            let file = H5File::create(&path).unwrap();
2659            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2660            ds.new_attr::<f64>()
2661                .shape(())
2662                .create("scale")
2663                .unwrap()
2664                .write_numeric(&1.5f64)
2665                .unwrap();
2666            ds.new_attr::<i32>()
2667                .shape(())
2668                .create("count")
2669                .unwrap()
2670                .write_numeric(&7i32)
2671                .unwrap();
2672            file.close().unwrap();
2673        }
2674        {
2675            let file = H5File::open(&path).unwrap();
2676            let ds = file.dataset("data").unwrap();
2677
2678            match ds.attr("scale").unwrap().datatype().unwrap() {
2679                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
2680                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
2681            }
2682
2683            match ds.attr("count").unwrap().datatype().unwrap() {
2684                DatatypeMessage::FixedPoint { size, signed, .. } => {
2685                    assert_eq!(size, 4);
2686                    assert!(signed, "i32 attr must be signed");
2687                }
2688                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
2689            }
2690        }
2691        std::fs::remove_file(&path).ok();
2692    }
2693
2694    #[test]
2695    fn attr_datatype_in_write_mode_errors() {
2696        let path = temp_path("attr_datatype_write_mode");
2697        let file = H5File::create(&path).unwrap();
2698        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2699        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
2700        assert!(attr.datatype().is_err());
2701        std::fs::remove_file(&path).ok();
2702    }
2703
2704    #[test]
2705    fn cannot_create_dataset_in_read_mode() {
2706        let path = temp_path("no_create_read");
2707
2708        {
2709            let _file = H5File::create(&path).unwrap();
2710        }
2711
2712        {
2713            let file = H5File::open(&path).unwrap();
2714            let result = file.new_dataset::<u8>().shape([4]).create("x");
2715            assert!(result.is_err());
2716        }
2717
2718        std::fs::remove_file(&path).ok();
2719    }
2720
2721    #[test]
2722    fn shape_accessor() {
2723        let path = temp_path("shape_acc");
2724
2725        let file = H5File::create(&path).unwrap();
2726        let ds = file
2727            .new_dataset::<f32>()
2728            .shape([5, 10, 3])
2729            .create("tensor")
2730            .unwrap();
2731        assert_eq!(ds.shape(), vec![5, 10, 3]);
2732
2733        std::fs::remove_file(&path).ok();
2734    }
2735
2736    #[test]
2737    fn slice_roundtrip_2d() {
2738        let path = temp_path("slice_2d");
2739
2740        // Create a 4x5 dataset, write full, then read a slice
2741        let data: Vec<i32> = (0..20).collect();
2742        {
2743            let file = H5File::create(&path).unwrap();
2744            let ds = file
2745                .new_dataset::<i32>()
2746                .shape([4, 5])
2747                .create("mat")
2748                .unwrap();
2749            ds.write_raw(&data).unwrap();
2750            file.close().unwrap();
2751        }
2752        {
2753            let file = H5File::open(&path).unwrap();
2754            let ds = file.dataset("mat").unwrap();
2755            // Read rows 1..3, cols 2..4 (2x2 slice)
2756            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
2757            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
2758            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
2759            assert_eq!(slice, vec![7, 8, 12, 13]);
2760        }
2761
2762        std::fs::remove_file(&path).ok();
2763    }
2764
2765    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
2766    // caller-provided buffer and MUST produce byte-for-byte the same data as
2767    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
2768    // creatable layout, since both now share one buffer-filling core.
2769    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
2770    where
2771        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
2772    {
2773        let n: usize = ds.shape().iter().product();
2774        let want_full = ds.read_raw::<T>().unwrap();
2775        let mut got_full = vec![T::default(); n];
2776        ds.read_raw_into::<T>(&mut got_full).unwrap();
2777        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
2778
2779        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
2780        let sn: usize = counts.iter().product();
2781        let mut got_slice = vec![T::default(); sn];
2782        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
2783            .unwrap();
2784        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
2785    }
2786
2787    #[test]
2788    fn read_into_matches_vec_contiguous() {
2789        let path = temp_path("into_contig");
2790        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
2791        {
2792            let file = H5File::create(&path).unwrap();
2793            let ds = file
2794                .new_dataset::<i32>()
2795                .shape([4, 5])
2796                .create("mat")
2797                .unwrap();
2798            ds.write_raw(&data).unwrap();
2799            file.close().unwrap();
2800        }
2801        {
2802            let file = H5File::open(&path).unwrap();
2803            let ds = file.dataset("mat").unwrap();
2804            assert_eq!(ds.chunk_dims(), None);
2805            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
2806        }
2807        std::fs::remove_file(&path).ok();
2808    }
2809
2810    #[test]
2811    fn read_into_matches_vec_chunked_unfiltered() {
2812        let path = temp_path("into_chunk");
2813        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
2814        {
2815            let file = H5File::create(&path).unwrap();
2816            let ds = file
2817                .new_dataset::<f64>()
2818                .shape([7, 5])
2819                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
2820                .create("grid")
2821                .unwrap();
2822            ds.write_raw(&data).unwrap();
2823            file.close().unwrap();
2824        }
2825        {
2826            let file = H5File::open(&path).unwrap();
2827            let ds = file.dataset("grid").unwrap();
2828            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2829            // Slice spans multiple chunks (rows 2..5, cols 1..4).
2830            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
2831        }
2832        std::fs::remove_file(&path).ok();
2833    }
2834
2835    #[test]
2836    fn read_into_matches_vec_single_chunk() {
2837        let path = temp_path("into_single_chunk");
2838        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
2839        {
2840            let file = H5File::create(&path).unwrap();
2841            let ds = file
2842                .new_dataset::<i32>()
2843                .shape([3, 4])
2844                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
2845                .create("g")
2846                .unwrap();
2847            ds.write_raw(&data).unwrap();
2848            file.close().unwrap();
2849        }
2850        {
2851            let file = H5File::open(&path).unwrap();
2852            let ds = file.dataset("g").unwrap();
2853            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
2854            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
2855        }
2856        std::fs::remove_file(&path).ok();
2857    }
2858
2859    #[cfg(feature = "deflate")]
2860    #[test]
2861    fn read_into_matches_vec_chunked_deflate() {
2862        let path = temp_path("into_chunk_deflate");
2863        let data: Vec<i32> = (0..35).collect(); // 7 x 5
2864        {
2865            let file = H5File::create(&path).unwrap();
2866            let ds = file
2867                .new_dataset::<i32>()
2868                .shape([7, 5])
2869                .chunk(&[3, 2])
2870                .deflate(4)
2871                .create("grid")
2872                .unwrap();
2873            ds.write_raw(&data).unwrap();
2874            file.close().unwrap();
2875        }
2876        {
2877            let file = H5File::open(&path).unwrap();
2878            let ds = file.dataset("grid").unwrap();
2879            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2880            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
2881        }
2882        std::fs::remove_file(&path).ok();
2883    }
2884
2885    #[test]
2886    fn read_into_wrong_buffer_size_rejected() {
2887        let path = temp_path("into_badlen");
2888        let data: Vec<i32> = (0..20).collect(); // 4 x 5
2889        {
2890            let file = H5File::create(&path).unwrap();
2891            let ds = file
2892                .new_dataset::<i32>()
2893                .shape([4, 5])
2894                .create("mat")
2895                .unwrap();
2896            ds.write_raw(&data).unwrap();
2897            file.close().unwrap();
2898        }
2899        {
2900            let file = H5File::open(&path).unwrap();
2901            let ds = file.dataset("mat").unwrap();
2902
2903            // Too small / too large full-read buffers are both rejected.
2904            let mut small = vec![0i32; 19];
2905            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
2906            let mut large = vec![0i32; 21];
2907            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
2908
2909            // Slice buffer must be exactly product(counts) = 4.
2910            let mut bad_slice = vec![0i32; 3];
2911            assert!(ds
2912                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
2913                .is_err());
2914            // The correctly sized slice buffer succeeds.
2915            let mut ok_slice = vec![0i32; 4];
2916            assert!(ds
2917                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
2918                .is_ok());
2919        }
2920        std::fs::remove_file(&path).ok();
2921    }
2922
2923    #[test]
2924    fn read_into_wrong_element_size_rejected() {
2925        let path = temp_path("into_badtype");
2926        let data: Vec<i32> = (0..20).collect(); // element size 4
2927        {
2928            let file = H5File::create(&path).unwrap();
2929            let ds = file
2930                .new_dataset::<i32>()
2931                .shape([4, 5])
2932                .create("mat")
2933                .unwrap();
2934            ds.write_raw(&data).unwrap();
2935            file.close().unwrap();
2936        }
2937        {
2938            let file = H5File::open(&path).unwrap();
2939            let ds = file.dataset("mat").unwrap();
2940            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
2941            // element size -> TypeMismatch, even with a "correctly sized" Vec.
2942            let mut as_u8 = vec![0u8; 20];
2943            assert!(matches!(
2944                ds.read_raw_into::<u8>(&mut as_u8),
2945                Err(crate::Hdf5Error::TypeMismatch(_))
2946            ));
2947            let mut as_i64 = vec![0i64; 20];
2948            assert!(matches!(
2949                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
2950                Err(crate::Hdf5Error::TypeMismatch(_))
2951            ));
2952        }
2953        std::fs::remove_file(&path).ok();
2954    }
2955
2956    #[test]
2957    fn write_slice_2d() {
2958        let path = temp_path("write_slice_2d");
2959
2960        {
2961            let file = H5File::create(&path).unwrap();
2962            let ds = file
2963                .new_dataset::<f32>()
2964                .shape([3, 4])
2965                .create("data")
2966                .unwrap();
2967            ds.write_raw(&[0.0f32; 12]).unwrap();
2968            // Overwrite a 2x2 sub-region
2969            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
2970                .unwrap();
2971            file.close().unwrap();
2972        }
2973        {
2974            let file = H5File::open(&path).unwrap();
2975            let ds = file.dataset("data").unwrap();
2976            let full = ds.read_raw::<f32>().unwrap();
2977            // Row 0: [0,0,0,0]
2978            // Row 1: [0,10,20,0]
2979            // Row 2: [0,30,40,0]
2980            assert_eq!(
2981                full,
2982                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
2983            );
2984        }
2985
2986        std::fs::remove_file(&path).ok();
2987    }
2988
2989    /// One 2x4 i32 chunk whose every element is `v`.
2990    fn chunk_of(v: i32) -> Vec<u8> {
2991        (0..8).flat_map(|_| v.to_le_bytes()).collect()
2992    }
2993
2994    /// Write chunk (0,0) `rewrites` times — each time with a different value,
2995    /// so no write can be skipped — and return the closed file's size along
2996    /// with what the chunk reads back as.
2997    fn rewrite_chunk(
2998        tag: &str,
2999        rewrites: i32,
3000        build: impl Fn(&H5File) -> crate::H5Dataset,
3001    ) -> (u64, i32) {
3002        let path = temp_path(tag);
3003        {
3004            let file = H5File::create(&path).unwrap();
3005            let ds = build(&file);
3006            for v in 1..=rewrites {
3007                ds.write_chunk_at(&[0, 0], &chunk_of(v)).unwrap();
3008            }
3009            file.close().unwrap();
3010        }
3011        let size = std::fs::metadata(&path).unwrap().len();
3012        let first = {
3013            let file = H5File::open(&path).unwrap();
3014            file.dataset("d").unwrap().read_raw::<i32>().unwrap()[0]
3015        };
3016        std::fs::remove_file(&path).ok();
3017        (size, first)
3018    }
3019
3020    // An unfiltered chunk's stored size is fixed by the chunk shape, so
3021    // rewriting it must overwrite the block it already occupies rather than
3022    // abandoning it and appending a new one (libhdf5 H5D__chunk_flush_entry
3023    // leaves must_alloc false for exactly this case). The file must therefore
3024    // be byte-identical in size no matter how many times the chunk is written.
3025    #[test]
3026    fn rewriting_an_unfiltered_extensible_array_chunk_stays_in_place() {
3027        let build = |f: &H5File| {
3028            f.new_dataset::<i32>()
3029                .shape([2, 4])
3030                .chunk(&[2, 4])
3031                .max_shape(&[None, Some(4)])
3032                .create("d")
3033                .unwrap()
3034        };
3035        let (once, _) = rewrite_chunk("rewrite_ea_1", 1, build);
3036        let (many, last) = rewrite_chunk("rewrite_ea_8", 8, build);
3037        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3038        assert_eq!(last, 8, "the last write must be the one that survives");
3039    }
3040
3041    #[test]
3042    fn rewriting_an_unfiltered_fixed_array_chunk_stays_in_place() {
3043        let build = |f: &H5File| {
3044            f.new_dataset::<i32>()
3045                .shape([2, 4])
3046                .chunk(&[2, 4])
3047                .create("d")
3048                .unwrap()
3049        };
3050        let (once, _) = rewrite_chunk("rewrite_fa_1", 1, build);
3051        let (many, last) = rewrite_chunk("rewrite_fa_8", 8, build);
3052        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3053        assert_eq!(last, 8);
3054    }
3055
3056    #[test]
3057    fn rewriting_an_unfiltered_btree_v2_chunk_stays_in_place() {
3058        let build = |f: &H5File| {
3059            f.new_dataset::<i32>()
3060                .shape([2, 4])
3061                .chunk(&[2, 4])
3062                .max_shape(&[None, None])
3063                .create("d")
3064                .unwrap()
3065        };
3066        let (once, _) = rewrite_chunk("rewrite_bt2_1", 1, build);
3067        let (many, last) = rewrite_chunk("rewrite_bt2_8", 8, build);
3068        assert_eq!(many, once, "8 rewrites grew the file past a single write");
3069        assert_eq!(last, 8);
3070    }
3071
3072    // A flush re-serializes the whole v2 B-tree over the dataset's node-block
3073    // pool. Every node is the same size, so the blocks already on disk are
3074    // reused and repeated flushes cost nothing; sizing the root to its record
3075    // count instead would relocate it each time and orphan the block it left.
3076    #[test]
3077    fn repeated_flushes_do_not_grow_a_btree_v2_index() {
3078        let flush_n = |label: &str, flushes: usize| -> u64 {
3079            let path = temp_path(label);
3080            {
3081                let file = H5File::create(&path).unwrap();
3082                let ds = file
3083                    .new_dataset::<i32>()
3084                    .shape([2, 4])
3085                    .chunk(&[2, 4])
3086                    .max_shape(&[None, None])
3087                    .create("d")
3088                    .unwrap();
3089                let bytes: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
3090                ds.write_chunk_at(&[0, 0], &bytes).unwrap();
3091                for _ in 0..flushes {
3092                    ds.flush().unwrap();
3093                }
3094                file.close().unwrap();
3095            }
3096            let size = std::fs::metadata(&path).unwrap().len();
3097            // The data must survive every rewrite of the index.
3098            {
3099                let file = H5File::open(&path).unwrap();
3100                let ds = file.dataset("d").unwrap();
3101                assert_eq!(ds.read_raw::<i32>().unwrap(), (0..8).collect::<Vec<i32>>());
3102            }
3103            std::fs::remove_file(&path).ok();
3104            size
3105        };
3106        assert_eq!(
3107            flush_n("bt2_flush_8", 8),
3108            flush_n("bt2_flush_1", 1),
3109            "8 index flushes grew the file past a single one"
3110        );
3111    }
3112
3113    // A filtered chunk whose compressed size changes cannot stay put, so it
3114    // moves and releases its old block (libhdf5 H5D__chunk_file_alloc calls
3115    // H5MF_xfree). Alternating between two payloads of different compressed
3116    // size must therefore keep reusing the same two blocks instead of
3117    // appending a fresh one each time.
3118    #[cfg(feature = "deflate")]
3119    #[test]
3120    fn rewriting_a_filtered_chunk_recycles_the_released_block() {
3121        // All-equal elements deflate to far fewer bytes than a varied payload,
3122        // so the two writes below land at different stored sizes.
3123        let flat: Vec<u8> = (0..8).flat_map(|_| 7i32.to_le_bytes()).collect();
3124        let varied: Vec<u8> = (0..8i32)
3125            .flat_map(|i| i.wrapping_mul(0x5bd1_e995).to_le_bytes())
3126            .collect();
3127
3128        let sizes: Vec<u64> = [1usize, 8]
3129            .iter()
3130            .map(|&rounds| {
3131                let path = temp_path(&format!("rewrite_filtered_{rounds}"));
3132                {
3133                    let file = H5File::create(&path).unwrap();
3134                    let ds = file
3135                        .new_dataset::<i32>()
3136                        .shape([2, 4])
3137                        .chunk(&[2, 4])
3138                        .max_shape(&[None, Some(4)])
3139                        .deflate(6)
3140                        .create("d")
3141                        .unwrap();
3142                    for _ in 0..rounds {
3143                        ds.write_chunk_at(&[0, 0], &flat).unwrap();
3144                        ds.write_chunk_at(&[0, 0], &varied).unwrap();
3145                    }
3146                    file.close().unwrap();
3147                }
3148                let size = std::fs::metadata(&path).unwrap().len();
3149                {
3150                    let file = H5File::open(&path).unwrap();
3151                    let got = file.dataset("d").unwrap().read_raw::<i32>().unwrap();
3152                    let want: Vec<i32> = (0..8i32).map(|i| i.wrapping_mul(0x5bd1_e995)).collect();
3153                    assert_eq!(got, want, "the last write must survive the round trip");
3154                }
3155                std::fs::remove_file(&path).ok();
3156                size
3157            })
3158            .collect();
3159
3160        assert_eq!(
3161            sizes[1], sizes[0],
3162            "8 alternating rewrites grew the file past a single pair"
3163        );
3164    }
3165
3166    #[test]
3167    fn write_slice_out_of_bounds_rejected() {
3168        let path = temp_path("write_slice_oob");
3169        let file = H5File::create(&path).unwrap();
3170        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
3171        ds.write_raw(&[0i32; 4]).unwrap();
3172        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
3173        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
3174        // An in-bounds slice still works.
3175        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
3176        std::fs::remove_file(&path).ok();
3177    }
3178
3179    #[test]
3180    fn duplicate_dataset_name_rejected() {
3181        let path = temp_path("dup_name");
3182        let file = H5File::create(&path).unwrap();
3183        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
3184        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
3185        std::fs::remove_file(&path).ok();
3186    }
3187
3188    #[test]
3189    fn extend_cannot_shrink() {
3190        let path = temp_path("extend_shrink");
3191        let file = H5File::create(&path).unwrap();
3192        let ds = file
3193            .new_dataset::<i32>()
3194            .shape([0])
3195            .chunk(&[2])
3196            .max_shape(&[None])
3197            .create("d")
3198            .unwrap();
3199        ds.append(&[1i32, 2, 3, 4]).unwrap();
3200        // Shrinking below the written extent must be rejected.
3201        assert!(ds.extend(&[2]).is_err());
3202        // Growing is fine.
3203        assert!(ds.extend(&[6]).is_ok());
3204        std::fs::remove_file(&path).ok();
3205    }
3206
3207    #[test]
3208    fn attr_read_roundtrip() {
3209        use crate::types::VarLenUnicode;
3210        let path = temp_path("attr_read");
3211
3212        {
3213            let file = H5File::create(&path).unwrap();
3214            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
3215            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3216            let a1 = ds
3217                .new_attr::<VarLenUnicode>()
3218                .shape(())
3219                .create("units")
3220                .unwrap();
3221            a1.write_string("meters").unwrap();
3222            let a2 = ds
3223                .new_attr::<VarLenUnicode>()
3224                .shape(())
3225                .create("desc")
3226                .unwrap();
3227            a2.write_string("test data").unwrap();
3228            file.close().unwrap();
3229        }
3230        {
3231            let file = H5File::open(&path).unwrap();
3232            let ds = file.dataset("data").unwrap();
3233
3234            let names = ds.attr_names().unwrap();
3235            assert!(names.contains(&"units".to_string()));
3236            assert!(names.contains(&"desc".to_string()));
3237
3238            let units = ds.attr("units").unwrap();
3239            assert_eq!(units.read_string().unwrap(), "meters");
3240
3241            let desc = ds.attr("desc").unwrap();
3242            assert_eq!(desc.read_string().unwrap(), "test data");
3243        }
3244
3245        std::fs::remove_file(&path).ok();
3246    }
3247
3248    #[test]
3249    fn type_mismatch_element_size() {
3250        let path = temp_path("type_mismatch");
3251
3252        {
3253            let file = H5File::create(&path).unwrap();
3254            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
3255            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
3256            file.close().unwrap();
3257        }
3258
3259        {
3260            let file = H5File::open(&path).unwrap();
3261            let ds = file.dataset("data").unwrap();
3262            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
3263            let result = ds.read_raw::<u8>();
3264            assert!(result.is_err());
3265        }
3266
3267        std::fs::remove_file(&path).ok();
3268    }
3269
3270    #[test]
3271    fn dataset_survives_file_move() {
3272        let path = temp_path("ds_survives");
3273
3274        let ds = {
3275            let file = H5File::create(&path).unwrap();
3276            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
3277        };
3278        // file is dropped here, but ds still holds Rc to the inner state
3279        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3280        // The writer will finalize on drop of the last Rc
3281
3282        std::fs::remove_file(&path).ok();
3283    }
3284
3285    #[test]
3286    fn new_attr_scalar_string() {
3287        use crate::types::VarLenUnicode;
3288
3289        let path = temp_path("attr_scalar_string");
3290        {
3291            let file = H5File::create(&path).unwrap();
3292            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
3293            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
3294
3295            let attr = ds
3296                .new_attr::<VarLenUnicode>()
3297                .shape(())
3298                .create("name")
3299                .unwrap();
3300            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
3301                .unwrap();
3302
3303            file.close().unwrap();
3304        }
3305
3306        // Verify the file is still valid and readable
3307        {
3308            use crate::format::messages::datatype::DatatypeMessage;
3309            let file = H5File::open(&path).unwrap();
3310            let ds = file.dataset("data").unwrap();
3311            assert_eq!(ds.shape(), vec![4]);
3312            let readback = ds.read_raw::<u8>().unwrap();
3313            assert_eq!(readback, vec![1u8, 2, 3, 4]);
3314
3315            // The string attribute is stored as a true variable-length string
3316            // (not fixed-length) and round-trips its value.
3317            let attr = ds.attr("name").unwrap();
3318            assert!(
3319                matches!(
3320                    attr.datatype().unwrap(),
3321                    DatatypeMessage::VarLenString { .. }
3322                ),
3323                "string attribute should have a variable-length string datatype"
3324            );
3325            assert_eq!(attr.read_string().unwrap(), "test_value");
3326        }
3327
3328        std::fs::remove_file(&path).ok();
3329    }
3330
3331    #[test]
3332    fn all_numeric_types_roundtrip() {
3333        let path = temp_path("all_types");
3334
3335        {
3336            let file = H5File::create(&path).unwrap();
3337
3338            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
3339            ds.write_raw(&[1u8, 2]).unwrap();
3340
3341            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
3342            ds.write_raw(&[-1i8, 1]).unwrap();
3343
3344            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
3345            ds.write_raw(&[100u16, 200]).unwrap();
3346
3347            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
3348            ds.write_raw(&[-100i16, 100]).unwrap();
3349
3350            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
3351            ds.write_raw(&[1000u32, 2000]).unwrap();
3352
3353            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
3354            ds.write_raw(&[-1000i32, 1000]).unwrap();
3355
3356            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
3357            ds.write_raw(&[10000u64, 20000]).unwrap();
3358
3359            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
3360            ds.write_raw(&[-10000i64, 10000]).unwrap();
3361
3362            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
3363            ds.write_raw(&[1.5f32, 2.5]).unwrap();
3364
3365            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
3366            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
3367
3368            file.close().unwrap();
3369        }
3370
3371        {
3372            let file = H5File::open(&path).unwrap();
3373
3374            assert_eq!(
3375                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
3376                vec![1u8, 2]
3377            );
3378            assert_eq!(
3379                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
3380                vec![-1i8, 1]
3381            );
3382            assert_eq!(
3383                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
3384                vec![100u16, 200]
3385            );
3386            assert_eq!(
3387                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
3388                vec![-100i16, 100]
3389            );
3390            assert_eq!(
3391                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
3392                vec![1000u32, 2000]
3393            );
3394            assert_eq!(
3395                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
3396                vec![-1000i32, 1000]
3397            );
3398            assert_eq!(
3399                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
3400                vec![10000u64, 20000]
3401            );
3402            assert_eq!(
3403                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
3404                vec![-10000i64, 10000]
3405            );
3406            assert_eq!(
3407                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
3408                vec![1.5f32, 2.5]
3409            );
3410            assert_eq!(
3411                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
3412                vec![1.23456f64, 7.89012]
3413            );
3414        }
3415
3416        std::fs::remove_file(&path).ok();
3417    }
3418
3419    #[test]
3420    fn append_chunked_roundtrip() {
3421        let path = temp_path("append_chunked");
3422
3423        {
3424            let file = H5File::create(&path).unwrap();
3425            let ds = file
3426                .new_dataset::<f64>()
3427                .shape([0, 3])
3428                .chunk(&[1, 3])
3429                .max_shape(&[None, Some(3)])
3430                .create("data")
3431                .unwrap();
3432
3433            // Append one frame
3434            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
3435            // Append two frames at once
3436            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
3437
3438            file.close().unwrap();
3439        }
3440
3441        {
3442            let file = H5File::open(&path).unwrap();
3443            let ds = file.dataset("data").unwrap();
3444            assert_eq!(ds.shape(), vec![3, 3]);
3445            let all = ds.read_raw::<f64>().unwrap();
3446            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
3447        }
3448
3449        std::fs::remove_file(&path).ok();
3450    }
3451
3452    #[test]
3453    fn append_1d_chunked() {
3454        let path = temp_path("append_1d");
3455
3456        {
3457            let file = H5File::create(&path).unwrap();
3458            let ds = file
3459                .new_dataset::<i32>()
3460                .shape([0])
3461                .chunk(&[4])
3462                .max_shape(&[None])
3463                .create("values")
3464                .unwrap();
3465
3466            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
3467            ds.append(&[40i32]).unwrap(); // fills chunk boundary
3468            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
3469
3470            file.close().unwrap();
3471        }
3472
3473        {
3474            let file = H5File::open(&path).unwrap();
3475            let ds = file.dataset("values").unwrap();
3476            assert_eq!(ds.shape(), vec![8]);
3477            let all = ds.read_raw::<i32>().unwrap();
3478            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
3479        }
3480
3481        std::fs::remove_file(&path).ok();
3482    }
3483
3484    #[test]
3485    fn append_partial_chunk_flushed_on_close() {
3486        let path = temp_path("append_partial_close");
3487
3488        {
3489            let file = H5File::create(&path).unwrap();
3490            let ds = file
3491                .new_dataset::<f64>()
3492                .shape([0])
3493                .chunk(&[4])
3494                .max_shape(&[None])
3495                .create("vals")
3496                .unwrap();
3497
3498            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
3499            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
3500            file.close().unwrap();
3501        }
3502
3503        {
3504            let file = H5File::open(&path).unwrap();
3505            let ds = file.dataset("vals").unwrap();
3506            assert_eq!(ds.shape(), vec![5]);
3507            let all = ds.read_raw::<f64>().unwrap();
3508            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
3509            // read_raw reads total shape elements
3510            assert_eq!(all.len(), 5);
3511            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
3512        }
3513
3514        std::fs::remove_file(&path).ok();
3515    }
3516
3517    /// An append that leaves its chunk partial is buffered until close, and
3518    /// the flush has to keep the frames that chunk already holds. It built a
3519    /// fresh fill-value chunk around the buffered frame instead, so reopening
3520    /// a file and appending one row erased every earlier row of that chunk
3521    /// (issue #3). Four sessions: the second lands beside an existing row, the
3522    /// third closes chunk 0 and opens chunk 1, the fourth lands beside the row
3523    /// the third left in chunk 1.
3524    #[test]
3525    fn append_after_reopen_keeps_the_partial_chunk_it_lands_in() {
3526        let path = temp_path("append_reopen_partial");
3527
3528        {
3529            let file = H5File::create(&path).unwrap();
3530            let ds = file
3531                .new_dataset::<i32>()
3532                .shape([0, 3])
3533                .chunk(&[4, 3])
3534                .max_shape(&[None, Some(3)])
3535                .create("values")
3536                .unwrap();
3537            ds.append(&[1, 2, 3]).unwrap();
3538            file.close().unwrap();
3539        }
3540        for rows in [
3541            vec![4, 5, 6],
3542            vec![7, 8, 9, 10, 11, 12, 13, 14, 15],
3543            vec![16, 17, 18],
3544        ] {
3545            let file = H5File::open_rw(&path).unwrap();
3546            file.dataset_writer("values")
3547                .unwrap()
3548                .append(&rows)
3549                .unwrap();
3550            file.close().unwrap();
3551        }
3552
3553        let file = H5File::open(&path).unwrap();
3554        let ds = file.dataset("values").unwrap();
3555        assert_eq!(ds.shape(), vec![6, 3]);
3556        assert_eq!(
3557            ds.read_raw::<i32>().unwrap(),
3558            (1..=18).collect::<Vec<i32>>()
3559        );
3560        std::fs::remove_file(&path).ok();
3561    }
3562
3563    #[cfg(feature = "deflate")]
3564    #[test]
3565    fn vlen_append_after_reopen_filtered() {
3566        // Reopen + append into a partially-written *compressed* vlen chunk
3567        // (index-block chunk). Exercises filtered-index-block reconstruction
3568        // in open_append plus filtered read-modify-write.
3569        let path = temp_path("vlen_reopen_filtered");
3570        {
3571            let file = H5File::create(&path).unwrap();
3572            file.create_appendable_vlen_dataset(
3573                "strs",
3574                4,
3575                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
3576            )
3577            .unwrap();
3578            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
3579                .unwrap();
3580            file.close().unwrap();
3581        }
3582        {
3583            let file = H5File::open_rw(&path).unwrap();
3584            file.append_vlen_strings("strs", &["delta"]).unwrap();
3585            file.close().unwrap();
3586        }
3587        {
3588            let file = H5File::open(&path).unwrap();
3589            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3590            assert_eq!(
3591                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
3592                vec!["alpha", "beta", "gamma", "delta"]
3593            );
3594        }
3595        std::fs::remove_file(&path).ok();
3596    }
3597
3598    #[test]
3599    fn vlen_append_after_reopen_data_block() {
3600        // Reopen + append into a partial chunk that lives in an extensible-
3601        // array *data block* (chunk index >= idx_blk_elmts). Exercises
3602        // data-block resolution in read_chunk_if_present and write_chunk.
3603        let path = temp_path("vlen_reopen_datablk");
3604        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
3605        {
3606            let file = H5File::create(&path).unwrap();
3607            file.create_appendable_vlen_dataset("strs", 2, None)
3608                .unwrap();
3609            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3610            file.append_vlen_strings("strs", &refs).unwrap();
3611            file.close().unwrap();
3612        }
3613        {
3614            let file = H5File::open_rw(&path).unwrap();
3615            file.append_vlen_strings("strs", &["s9"]).unwrap();
3616            file.close().unwrap();
3617        }
3618        {
3619            let file = H5File::open(&path).unwrap();
3620            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3621            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
3622            assert_eq!(got, want);
3623        }
3624        std::fs::remove_file(&path).ok();
3625    }
3626
3627    #[test]
3628    fn vlen_append_after_reopen_super_block() {
3629        // Reopen + append into a partial chunk whose index falls in an
3630        // extensible-array *super block* (chunk index 244 with the default
3631        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
3632        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
3633        // index block or its direct data blocks, so chunk 244 is reached
3634        // via a super block read from disk). Exercises the ViaSblk branch
3635        // of read_chunk_if_present.
3636        let path = temp_path("vlen_reopen_super");
3637        // 489 strings, chunk size 2 -> chunk 244 holds one string only
3638        // (partially filled) and is flushed to disk on close.
3639        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
3640        {
3641            let file = H5File::create(&path).unwrap();
3642            file.create_appendable_vlen_dataset("strs", 2, None)
3643                .unwrap();
3644            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3645            file.append_vlen_strings("strs", &refs).unwrap();
3646            file.close().unwrap();
3647        }
3648        {
3649            let file = H5File::open_rw(&path).unwrap();
3650            file.append_vlen_strings("strs", &["v489"]).unwrap();
3651            file.close().unwrap();
3652        }
3653        {
3654            let file = H5File::open(&path).unwrap();
3655            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3656            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
3657            assert_eq!(got, want);
3658        }
3659        std::fs::remove_file(&path).ok();
3660    }
3661
3662    #[cfg(feature = "deflate")]
3663    #[test]
3664    fn vlen_append_after_reopen_filtered_data_block() {
3665        // The hardest path: compressed + chunk in a data block + partial
3666        // read-modify-write across a reopen.
3667        let path = temp_path("vlen_reopen_filt_datablk");
3668        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
3669        {
3670            let file = H5File::create(&path).unwrap();
3671            file.create_appendable_vlen_dataset(
3672                "strs",
3673                2,
3674                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
3675            )
3676            .unwrap();
3677            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3678            file.append_vlen_strings("strs", &refs).unwrap();
3679            file.close().unwrap();
3680        }
3681        {
3682            let file = H5File::open_rw(&path).unwrap();
3683            file.append_vlen_strings("strs", &["item09"]).unwrap();
3684            file.close().unwrap();
3685        }
3686        {
3687            let file = H5File::open(&path).unwrap();
3688            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3689            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
3690            assert_eq!(got, want);
3691        }
3692        std::fs::remove_file(&path).ok();
3693    }
3694
3695    #[test]
3696    fn group_nx_class_attribute_roundtrip() {
3697        // Non-root groups carry attributes (NeXus `NX_class`) in their
3698        // own object header, and the reader reads them back by path.
3699        let path = temp_path("group_nx_class");
3700        {
3701            let file = H5File::create(&path).unwrap();
3702            let entry = file.create_group("entry").unwrap();
3703            entry.set_attr_string("NX_class", "NXentry").unwrap();
3704            let det = entry.create_group("detector").unwrap();
3705            det.set_attr_string("NX_class", "NXdetector").unwrap();
3706            det.set_attr_numeric("frame_count", &7i32).unwrap();
3707            det.new_dataset::<f32>()
3708                .shape([4])
3709                .create("data")
3710                .unwrap()
3711                .write_raw(&[1.0f32; 4])
3712                .unwrap();
3713            file.close().unwrap();
3714        }
3715        {
3716            let file = H5File::open(&path).unwrap();
3717            let entry = file.root_group().group("entry").unwrap();
3718            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
3719            let det = entry.group("detector").unwrap();
3720            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
3721            let names = det.attr_names().unwrap();
3722            assert!(names.contains(&"NX_class".to_string()));
3723            assert!(names.contains(&"frame_count".to_string()));
3724        }
3725        std::fs::remove_file(&path).ok();
3726    }
3727
3728    #[test]
3729    fn ea_super_block_roundtrip() {
3730        // 2000 chunks span several extensible-array super blocks. Before
3731        // super-block support the writer errored at chunk index 228.
3732        let path = temp_path("ea_super_rt");
3733        {
3734            let file = H5File::create(&path).unwrap();
3735            let ds = file
3736                .new_dataset::<i32>()
3737                .shape([0])
3738                .chunk(&[1])
3739                .max_shape(&[None])
3740                .create("v")
3741                .unwrap();
3742            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
3743            file.close().unwrap();
3744        }
3745        {
3746            let file = H5File::open(&path).unwrap();
3747            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3748            assert_eq!(v.len(), 2000);
3749            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
3750        }
3751        std::fs::remove_file(&path).ok();
3752    }
3753
3754    #[cfg(feature = "deflate")]
3755    #[test]
3756    fn ea_filtered_super_block_roundtrip() {
3757        // Compressed chunks across super blocks.
3758        let path = temp_path("ea_filt_super");
3759        {
3760            let file = H5File::create(&path).unwrap();
3761            let ds = file
3762                .new_dataset::<i32>()
3763                .shape([0])
3764                .chunk(&[1])
3765                .max_shape(&[None])
3766                .deflate(4)
3767                .create("v")
3768                .unwrap();
3769            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
3770            file.close().unwrap();
3771        }
3772        {
3773            let file = H5File::open(&path).unwrap();
3774            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3775            assert_eq!(v, (0..600).collect::<Vec<i32>>());
3776        }
3777        std::fs::remove_file(&path).ok();
3778    }
3779
3780    #[test]
3781    fn ea_super_block_open_append() {
3782        // Reopen a dataset and append chunks that fall in super blocks.
3783        let path = temp_path("ea_super_append");
3784        {
3785            let file = H5File::create(&path).unwrap();
3786            let ds = file
3787                .new_dataset::<i32>()
3788                .shape([0])
3789                .chunk(&[1])
3790                .max_shape(&[None])
3791                .create("v")
3792                .unwrap();
3793            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
3794            file.close().unwrap();
3795        }
3796        {
3797            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3798            let idx = w.dataset_index("v").unwrap();
3799            for c in 300..900u64 {
3800                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
3801            }
3802            w.extend_dataset(idx, &[900]).unwrap();
3803            w.close().unwrap();
3804        }
3805        {
3806            let file = H5File::open(&path).unwrap();
3807            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3808            assert_eq!(v.len(), 900);
3809            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
3810        }
3811        std::fs::remove_file(&path).ok();
3812    }
3813
3814    // Two or more unlimited dimensions select the v2 B-tree index; with a
3815    // filter its records become type 11, carrying each chunk's stored size and
3816    // mask. The payload is highly compressible, so the chunks really are
3817    // stored smaller than the extent — the file would be at least
3818    // 6*8*4 = 192 bytes of raw chunk data otherwise.
3819    #[cfg(feature = "deflate")]
3820    #[test]
3821    fn compressed_multi_unlimited_dataset_roundtrips() {
3822        let path = temp_path("bt2_filtered");
3823        {
3824            let file = H5File::create(&path).unwrap();
3825            let ds = file
3826                .new_dataset::<i32>()
3827                .shape([6, 8])
3828                .chunk(&[2, 4])
3829                .max_shape(&[None, None])
3830                .deflate(6)
3831                .create("d")
3832                .unwrap();
3833            ds.write_slice(&[0, 0], &[6, 8], &[7i32; 48]).unwrap();
3834            // A partial write forces a decompress-patch-recompress of one
3835            // chunk, whose new compressed size may not fit its old block.
3836            ds.write_slice(&[1, 1], &[2, 2], &[1i32, 2, 3, 4]).unwrap();
3837            file.close().unwrap();
3838        }
3839        {
3840            let file = H5File::open(&path).unwrap();
3841            let ds = file.dataset("d").unwrap();
3842            assert_eq!(ds.shape(), vec![6, 8]);
3843            let mut want = vec![7i32; 48];
3844            want[9] = 1;
3845            want[10] = 2;
3846            want[17] = 3;
3847            want[18] = 4;
3848            assert_eq!(ds.read_raw::<i32>().unwrap(), want);
3849        }
3850        std::fs::remove_file(&path).ok();
3851    }
3852
3853    #[test]
3854    fn btree_v2_multi_unlimited_roundtrip() {
3855        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
3856        // index; chunks are written by grid coordinates with write_chunk_at.
3857        let path = temp_path("bt2_multi");
3858        {
3859            let file = H5File::create(&path).unwrap();
3860            let ds = file
3861                .new_dataset::<i32>()
3862                .shape([0, 0])
3863                .chunk(&[2, 2])
3864                .max_shape(&[None, None])
3865                .create("grid")
3866                .unwrap();
3867            assert!(ds.is_chunked());
3868            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
3869            for cr in 0..2usize {
3870                for cc in 0..2usize {
3871                    let mut bytes = Vec::new();
3872                    for i in 0..2usize {
3873                        for j in 0..2usize {
3874                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
3875                            bytes.extend_from_slice(&v.to_le_bytes());
3876                        }
3877                    }
3878                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
3879                }
3880            }
3881            file.close().unwrap();
3882        }
3883        {
3884            let file = H5File::open(&path).unwrap();
3885            let ds = file.dataset("grid").unwrap();
3886            assert_eq!(ds.shape(), vec![4, 4]);
3887            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
3888        }
3889        std::fs::remove_file(&path).ok();
3890    }
3891
3892    #[test]
3893    fn subframe_chunking_roundtrip() {
3894        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
3895        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
3896        // the chunk-grid coordinates.
3897        let path = temp_path("subframe");
3898        {
3899            let file = H5File::create(&path).unwrap();
3900            let ds = file
3901                .new_dataset::<i32>()
3902                .shape([0, 8, 8])
3903                .chunk(&[1, 4, 4])
3904                .max_shape(&[None, Some(8), Some(8)])
3905                .create("v")
3906                .unwrap();
3907            for f in 0..3usize {
3908                for cr in 0..2usize {
3909                    for cc in 0..2usize {
3910                        let mut bytes = Vec::new();
3911                        for i in 0..4usize {
3912                            for j in 0..4usize {
3913                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
3914                                bytes.extend_from_slice(&v.to_le_bytes());
3915                            }
3916                        }
3917                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
3918                    }
3919                }
3920            }
3921            file.close().unwrap();
3922        }
3923        {
3924            let file = H5File::open(&path).unwrap();
3925            let ds = file.dataset("v").unwrap();
3926            assert_eq!(ds.shape(), vec![3, 8, 8]);
3927            assert_eq!(
3928                ds.read_raw::<i32>().unwrap(),
3929                (0..192).collect::<Vec<i32>>()
3930            );
3931        }
3932        std::fs::remove_file(&path).ok();
3933    }
3934
3935    #[test]
3936    fn fill_value_contiguous_roundtrip() {
3937        let path = temp_path("fill_value_contig");
3938        {
3939            let file = H5File::create(&path).unwrap();
3940            let ds = file
3941                .new_dataset::<f32>()
3942                .shape([4])
3943                .fill_value(2.5f32)
3944                .create("data")
3945                .unwrap();
3946            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
3947            file.close().unwrap();
3948        }
3949        // open_append decodes the fill-value message back from the header.
3950        {
3951            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3952            let idx = writer.dataset_index("data").unwrap();
3953            assert_eq!(
3954                writer.ds(idx).lock().fill_value,
3955                Some(2.5f32.to_le_bytes().to_vec())
3956            );
3957        }
3958        // Data still reads back correctly.
3959        {
3960            let file = H5File::open(&path).unwrap();
3961            let ds = file.dataset("data").unwrap();
3962            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
3963        }
3964        std::fs::remove_file(&path).ok();
3965    }
3966
3967    #[test]
3968    fn fill_value_chunked_roundtrip() {
3969        let path = temp_path("fill_value_chunked");
3970        {
3971            let file = H5File::create(&path).unwrap();
3972            let ds = file
3973                .new_dataset::<i32>()
3974                .shape([0])
3975                .chunk(&[4])
3976                .max_shape(&[None])
3977                .fill_value(-7i32)
3978                .create("vals")
3979                .unwrap();
3980            ds.append(&[1i32, 2, 3, 4]).unwrap();
3981            file.close().unwrap();
3982        }
3983        {
3984            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3985            let idx = writer.dataset_index("vals").unwrap();
3986            assert_eq!(
3987                writer.ds(idx).lock().fill_value,
3988                Some((-7i32).to_le_bytes().to_vec())
3989            );
3990        }
3991        std::fs::remove_file(&path).ok();
3992    }
3993
3994    #[test]
3995    fn fill_value_read_missing_chunks() {
3996        // A chunked dataset with chunk 1 left unwritten must read that
3997        // gap back as the user-defined fill value, not zero.
3998        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
3999            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
4000        }
4001        let path = temp_path("fill_value_read_missing");
4002        {
4003            let file = H5File::create(&path).unwrap();
4004            let ds = file
4005                .new_dataset::<i32>()
4006                .shape([0])
4007                .chunk(&[2])
4008                .max_shape(&[None])
4009                .fill_value(-1i32)
4010                .create("vals")
4011                .unwrap();
4012            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
4013            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
4014            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
4015            ds.extend(&[6]).unwrap();
4016            file.close().unwrap();
4017        }
4018        {
4019            let file = H5File::open(&path).unwrap();
4020            let ds = file.dataset("vals").unwrap();
4021            let all = ds.read_raw::<i32>().unwrap();
4022            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
4023        }
4024        std::fs::remove_file(&path).ok();
4025    }
4026
4027    #[test]
4028    fn fill_value_partial_chunk_padded_with_fill() {
4029        // A partial trailing chunk flushed at close must pad its unwritten
4030        // tail with the fill value. That pad sits beyond the logical shape,
4031        // so it is verified by scanning the on-disk chunk bytes directly.
4032        let path = temp_path("fill_value_partial_pad");
4033        {
4034            let file = H5File::create(&path).unwrap();
4035            let ds = file
4036                .new_dataset::<i32>()
4037                .shape([0])
4038                .chunk(&[4])
4039                .max_shape(&[None])
4040                .fill_value(-9i32)
4041                .create("vals")
4042                .unwrap();
4043            // 3 of 4 frames -> flushed as a partial chunk on close.
4044            ds.append(&[1i32, 2, 3]).unwrap();
4045            file.close().unwrap();
4046        }
4047        let bytes = std::fs::read(&path).unwrap();
4048        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
4049        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
4050        let pos = bytes
4051            .windows(needle.len())
4052            .position(|w| w == needle)
4053            .expect("chunk data [1,2,3] not found in file");
4054        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
4055        assert_eq!(
4056            pad,
4057            &(-9i32).to_le_bytes(),
4058            "partial chunk tail must be padded with fill value -9, got {:?}",
4059            pad
4060        );
4061        std::fs::remove_file(&path).ok();
4062    }
4063
4064    #[test]
4065    fn vlen_append_after_reopen_preserves_existing() {
4066        // Reopening and appending into a partially-written vlen chunk must
4067        // read-modify-write: the strings already on disk must survive.
4068        let path = temp_path("vlen_append_reopen");
4069        {
4070            let file = H5File::create(&path).unwrap();
4071            file.create_appendable_vlen_dataset("strs", 4, None)
4072                .unwrap();
4073            // 3 of 4 frames -> flushed as a partial chunk on close.
4074            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
4075            file.close().unwrap();
4076        }
4077        {
4078            // Append a 4th string -> partial-chunk write into chunk 0.
4079            let file = H5File::open_rw(&path).unwrap();
4080            file.append_vlen_strings("strs", &["d"]).unwrap();
4081            file.close().unwrap();
4082        }
4083        {
4084            let file = H5File::open(&path).unwrap();
4085            let ds = file.dataset("strs").unwrap();
4086            let got = ds.read_vlen_strings().unwrap();
4087            assert_eq!(
4088                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
4089                vec!["a", "b", "c", "d"]
4090            );
4091        }
4092        std::fs::remove_file(&path).ok();
4093    }
4094
4095    #[test]
4096    fn fill_value_size_mismatch_errors() {
4097        let path = temp_path("fill_value_mismatch");
4098        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
4099        let dt = <f64 as crate::types::H5Type>::hdf5_type();
4100        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
4101        // f64 element size is 8; a 4-byte fill value must be rejected.
4102        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
4103        // The correct width succeeds.
4104        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
4105        writer.close().unwrap();
4106        std::fs::remove_file(&path).ok();
4107    }
4108
4109    #[test]
4110    fn datatype_exposes_class_sign_and_byteorder() {
4111        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
4112        // from f32 (both 4 bytes). datatype() must report the real class and
4113        // signedness so a reader does not have to guess from element_size.
4114        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
4115
4116        let path = temp_path("datatype_accessor");
4117        {
4118            let file = H5File::create(&path).unwrap();
4119            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
4120            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
4121            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
4122            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
4123            file.close().unwrap();
4124        }
4125
4126        let file = H5File::open(&path).unwrap();
4127
4128        match file.dataset("u8d").unwrap().datatype().unwrap() {
4129            DatatypeMessage::FixedPoint {
4130                size,
4131                signed,
4132                byte_order,
4133                ..
4134            } => {
4135                assert_eq!(size, 1);
4136                assert!(!signed, "u8 must be unsigned");
4137                assert_eq!(byte_order, ByteOrder::LittleEndian);
4138            }
4139            other => panic!("expected FixedPoint for u8, got {other:?}"),
4140        }
4141
4142        match file.dataset("i8d").unwrap().datatype().unwrap() {
4143            DatatypeMessage::FixedPoint { size, signed, .. } => {
4144                assert_eq!(size, 1);
4145                assert!(signed, "i8 must be signed");
4146            }
4147            other => panic!("expected FixedPoint for i8, got {other:?}"),
4148        }
4149
4150        match file.dataset("i32d").unwrap().datatype().unwrap() {
4151            DatatypeMessage::FixedPoint { size, signed, .. } => {
4152                assert_eq!(size, 4);
4153                assert!(signed, "i32 must be signed");
4154            }
4155            other => panic!("expected FixedPoint for i32, got {other:?}"),
4156        }
4157
4158        match file.dataset("f32d").unwrap().datatype().unwrap() {
4159            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
4160            other => panic!("expected FloatingPoint for f32, got {other:?}"),
4161        }
4162
4163        std::fs::remove_file(&path).ok();
4164    }
4165
4166    #[test]
4167    fn datatype_in_write_mode_errors() {
4168        let path = temp_path("datatype_write_mode");
4169        let file = H5File::create(&path).unwrap();
4170        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
4171        assert!(ds.datatype().is_err());
4172        std::fs::remove_file(&path).ok();
4173    }
4174
4175    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
4176
4177    /// Extensible-array path: pre-compress with the dataset's pipeline, write
4178    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
4179    /// the data round-trips through the reader unchanged.
4180    #[cfg(feature = "deflate")]
4181    #[test]
4182    fn write_chunk_raw_ea_roundtrip_mask0() {
4183        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4184        let path = temp_path("wcr_ea_mask0");
4185        let original: Vec<i32> = (0..12).collect();
4186        {
4187            let file = H5File::create(&path).unwrap();
4188            let ds = file
4189                .new_dataset::<i32>()
4190                .shape([0])
4191                .chunk(&[4])
4192                .max_shape(&[None])
4193                .deflate(4)
4194                .create("v")
4195                .unwrap();
4196            assert!(ds.is_chunked());
4197            let pipeline = FilterPipeline::deflate(4);
4198            for c in 0..3usize {
4199                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
4200                    .iter()
4201                    .flat_map(|v| v.to_le_bytes())
4202                    .collect();
4203                let compressed = apply_filters(&pipeline, &raw).unwrap();
4204                ds.write_chunk_raw(c, &compressed, 0).unwrap();
4205            }
4206            ds.set_extent(&[12]).unwrap();
4207            file.close().unwrap();
4208        }
4209        {
4210            let file = H5File::open(&path).unwrap();
4211            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4212            assert_eq!(v, original);
4213        }
4214        std::fs::remove_file(&path).ok();
4215    }
4216
4217    /// Fixed-array path (all dimensions bounded): same verbatim write through
4218    /// the linear-index dispatch, round-tripped through the reader.
4219    #[cfg(feature = "deflate")]
4220    #[test]
4221    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
4222        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4223        let path = temp_path("wcr_fa_mask0");
4224        let original: Vec<i32> = (0..12).collect();
4225        {
4226            let file = H5File::create(&path).unwrap();
4227            let ds = file
4228                .new_dataset::<i32>()
4229                .shape([12])
4230                .chunk(&[4])
4231                .deflate(4)
4232                .create("v")
4233                .unwrap();
4234            assert!(ds.is_chunked());
4235            let pipeline = FilterPipeline::deflate(4);
4236            for c in 0..3usize {
4237                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
4238                    .iter()
4239                    .flat_map(|v| v.to_le_bytes())
4240                    .collect();
4241                let compressed = apply_filters(&pipeline, &raw).unwrap();
4242                ds.write_chunk_raw(c, &compressed, 0).unwrap();
4243            }
4244            file.close().unwrap();
4245        }
4246        {
4247            let file = H5File::open(&path).unwrap();
4248            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4249            assert_eq!(v, original);
4250        }
4251        std::fs::remove_file(&path).ok();
4252    }
4253
4254    /// The caller-supplied filter_mask must reach the on-disk filtered index
4255    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
4256    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
4257    /// the extensible-array filtered entry to read the mask back at the format
4258    /// level (independent of the data reader's mask handling).
4259    #[cfg(feature = "deflate")]
4260    #[test]
4261    fn write_chunk_raw_records_filter_mask() {
4262        let path = temp_path("wcr_records_mask");
4263        let raw: Vec<u8> = [10i32, 20, 30, 40]
4264            .iter()
4265            .flat_map(|v| v.to_le_bytes())
4266            .collect();
4267        assert_eq!(raw.len(), 16);
4268        {
4269            let file = H5File::create(&path).unwrap();
4270            let ds = file
4271                .new_dataset::<i32>()
4272                .shape([0])
4273                .chunk(&[4])
4274                .max_shape(&[None])
4275                .deflate(4)
4276                .create("v")
4277                .unwrap();
4278            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
4279            // chunk is stored uncompressed (its raw bytes).
4280            ds.write_chunk_raw(0, &raw, 1).unwrap();
4281            ds.set_extent(&[4]).unwrap();
4282            file.close().unwrap();
4283        }
4284        // Reopen the writer; open_append decodes the filtered index block from
4285        // disk, so the entry reflects exactly what was committed.
4286        {
4287            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
4288            let idx = w.dataset_index("v").unwrap();
4289            let ds = w.ds(idx);
4290            let m = ds.lock();
4291            let entry = &m
4292                .chunked
4293                .as_ref()
4294                .unwrap()
4295                .filt_iblk
4296                .as_ref()
4297                .unwrap()
4298                .elements[0];
4299            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
4300            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
4301        }
4302        std::fs::remove_file(&path).ok();
4303    }
4304
4305    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
4306    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
4307    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
4308    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
4309    #[cfg(feature = "deflate")]
4310    #[test]
4311    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
4312        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4313        let path = temp_path("wcr_ea_per_chunk_mask");
4314        let original: Vec<i32> = (0..8).collect();
4315        let pipeline = FilterPipeline::deflate(4);
4316        {
4317            let file = H5File::create(&path).unwrap();
4318            let ds = file
4319                .new_dataset::<i32>()
4320                .shape([0])
4321                .chunk(&[4])
4322                .max_shape(&[None])
4323                .deflate(4)
4324                .create("v")
4325                .unwrap();
4326            let raw0: Vec<u8> = original[0..4]
4327                .iter()
4328                .flat_map(|v| v.to_le_bytes())
4329                .collect();
4330            // chunk 0: compressed through the pipeline, mask 0.
4331            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
4332                .unwrap();
4333            let raw1: Vec<u8> = original[4..8]
4334                .iter()
4335                .flat_map(|v| v.to_le_bytes())
4336                .collect();
4337            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
4338            ds.write_chunk_raw(1, &raw1, 1).unwrap();
4339            ds.set_extent(&[8]).unwrap();
4340            file.close().unwrap();
4341        }
4342        {
4343            let file = H5File::open(&path).unwrap();
4344            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4345            assert_eq!(v, original);
4346        }
4347        std::fs::remove_file(&path).ok();
4348    }
4349
4350    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
4351    /// compressed/raw chunks as the EA case, through the fixed-array index.
4352    #[cfg(feature = "deflate")]
4353    #[test]
4354    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
4355        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4356        let path = temp_path("wcr_fa_per_chunk_mask");
4357        let original: Vec<i32> = (0..8).collect();
4358        let pipeline = FilterPipeline::deflate(4);
4359        {
4360            let file = H5File::create(&path).unwrap();
4361            let ds = file
4362                .new_dataset::<i32>()
4363                .shape([8])
4364                .chunk(&[4])
4365                .deflate(4)
4366                .create("v")
4367                .unwrap();
4368            let raw0: Vec<u8> = original[0..4]
4369                .iter()
4370                .flat_map(|v| v.to_le_bytes())
4371                .collect();
4372            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
4373                .unwrap();
4374            let raw1: Vec<u8> = original[4..8]
4375                .iter()
4376                .flat_map(|v| v.to_le_bytes())
4377                .collect();
4378            ds.write_chunk_raw(1, &raw1, 1).unwrap();
4379            file.close().unwrap();
4380        }
4381        {
4382            let file = H5File::open(&path).unwrap();
4383            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
4384            assert_eq!(v, original);
4385        }
4386        std::fs::remove_file(&path).ok();
4387    }
4388
4389    /// An unfiltered chunk index has no slot for a stored size or mask, so a
4390    /// direct chunk write must be rejected rather than silently dropping them.
4391    #[test]
4392    fn write_chunk_raw_rejects_unfiltered() {
4393        let path = temp_path("wcr_unfiltered");
4394        let file = H5File::create(&path).unwrap();
4395        let ds = file
4396            .new_dataset::<i32>()
4397            .shape([0])
4398            .chunk(&[4])
4399            .max_shape(&[None])
4400            .create("v")
4401            .unwrap();
4402        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
4403        assert!(
4404            err.to_string().contains("filtered dataset"),
4405            "expected a filtered-dataset error, got: {err}"
4406        );
4407        std::fs::remove_file(&path).ok();
4408    }
4409
4410    /// Two or more unlimited dimensions leave no fixed chunk grid for a linear
4411    /// index to mean anything against, so the linear entry point points the
4412    /// caller at the coordinate-addressed one rather than guessing a grid.
4413    #[test]
4414    fn write_chunk_raw_sends_btree_v2_to_the_coordinate_form() {
4415        let path = temp_path("wcr_btree2");
4416        let file = H5File::create(&path).unwrap();
4417        let ds = file
4418            .new_dataset::<i32>()
4419            .shape([0, 0])
4420            .chunk(&[2, 2])
4421            .max_shape(&[None, None])
4422            .create("grid")
4423            .unwrap();
4424        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
4425        assert!(
4426            err.to_string().contains("write_chunk_raw_at"),
4427            "expected a pointer to the coordinate form, got: {err}"
4428        );
4429        std::fs::remove_file(&path).ok();
4430    }
4431
4432    /// Direct chunk writes on a v2-B-tree index: the bytes are stored verbatim
4433    /// and the type-11 record carries their size and the caller's mask, so a
4434    /// chunk written with the pipeline skipped (mask 1) reads back as the raw
4435    /// bytes while one written compressed (mask 0) is decompressed.
4436    #[cfg(feature = "deflate")]
4437    #[test]
4438    fn write_chunk_raw_at_round_trips_on_btree_v2() {
4439        use crate::format::messages::filter::{apply_filters, FilterPipeline};
4440
4441        let path = temp_path("wcr_at_btree2");
4442        let raw0: Vec<u8> = (0..4i32).flat_map(|v| v.to_le_bytes()).collect();
4443        let raw1: Vec<u8> = (100..104i32).flat_map(|v| v.to_le_bytes()).collect();
4444        {
4445            let file = H5File::create(&path).unwrap();
4446            let ds = file
4447                .new_dataset::<i32>()
4448                .shape([0, 0])
4449                .chunk(&[2, 2])
4450                .max_shape(&[None, None])
4451                .deflate(6)
4452                .create("grid")
4453                .unwrap();
4454            let pipeline = FilterPipeline::deflate(6);
4455            // Chunk (0,0): pipeline already applied upstream, mask 0.
4456            ds.write_chunk_raw_at(&[0, 0], &apply_filters(&pipeline, &raw0).unwrap(), 0)
4457                .unwrap();
4458            // Chunk (1,1): stored uncompressed, mask 1 says filter 0 was skipped.
4459            ds.write_chunk_raw_at(&[1, 1], &raw1, 1).unwrap();
4460            file.close().unwrap();
4461        }
4462        let file = H5File::open(&path).unwrap();
4463        let ds = file.dataset("grid").unwrap();
4464        assert_eq!(ds.shape(), vec![4, 4]);
4465        let all = ds.read_raw::<i32>().unwrap();
4466        // Chunk (0,0) occupies rows 0..2, columns 0..2.
4467        assert_eq!([all[0], all[1], all[4], all[5]], [0, 1, 2, 3]);
4468        // Chunk (1,1) occupies rows 2..4, columns 2..4.
4469        assert_eq!([all[10], all[11], all[14], all[15]], [100, 101, 102, 103]);
4470        drop(file);
4471        std::fs::remove_file(&path).ok();
4472    }
4473
4474    /// The coordinate form is not BT2-only: it addresses an extensible- or
4475    /// fixed-array dataset's grid just as well, and records the same mask.
4476    #[cfg(feature = "deflate")]
4477    #[test]
4478    fn write_chunk_raw_at_round_trips_on_the_array_indexes() {
4479        for (label, max_shape) in [
4480            ("wcr_at_ea", Some(vec![None, Some(4usize)])),
4481            ("wcr_at_fa", None),
4482        ] {
4483            let path = temp_path(label);
4484            let raw: Vec<u8> = (0..8i32).flat_map(|v| v.to_le_bytes()).collect();
4485            {
4486                let file = H5File::create(&path).unwrap();
4487                let mut b = file
4488                    .new_dataset::<i32>()
4489                    .shape([4usize, 4])
4490                    .chunk(&[2, 4])
4491                    .deflate(6);
4492                if let Some(ref ms) = max_shape {
4493                    b = b.max_shape(ms);
4494                }
4495                let ds = b.create("grid").unwrap();
4496                // Row-of-chunks 1, stored uncompressed with filter 0 skipped.
4497                ds.write_chunk_raw_at(&[1, 0], &raw, 1).unwrap();
4498                file.close().unwrap();
4499            }
4500            let file = H5File::open(&path).unwrap();
4501            let ds = file.dataset("grid").unwrap();
4502            let all = ds.read_raw::<i32>().unwrap();
4503            assert_eq!(&all[8..16], &(0..8).collect::<Vec<i32>>()[..], "{label}");
4504            drop(file);
4505            std::fs::remove_file(&path).ok();
4506        }
4507    }
4508
4509    /// A direct write hands over caller-supplied bytes, so the v2 B-tree's
4510    /// chunk-size field can overflow just as the array indexes' can. A 4-byte
4511    /// chunk gives chunk_size_len = 2 (max 65535).
4512    #[cfg(feature = "deflate")]
4513    #[test]
4514    fn write_chunk_raw_at_rejects_an_oversized_btree_v2_chunk() {
4515        let path = temp_path("wcr_at_oversized");
4516        let file = H5File::create(&path).unwrap();
4517        let ds = file
4518            .new_dataset::<i32>()
4519            .shape([0, 0])
4520            .chunk(&[1, 1])
4521            .max_shape(&[None, None])
4522            .deflate(4)
4523            .create("grid")
4524            .unwrap();
4525        let err = ds
4526            .write_chunk_raw_at(&[0, 0], &vec![0u8; 70000], 0)
4527            .unwrap_err();
4528        assert!(
4529            err.to_string().contains("does not fit"),
4530            "expected a chunk-size-field overflow error, got: {err}"
4531        );
4532        std::fs::remove_file(&path).ok();
4533    }
4534
4535    /// An unfiltered v2 B-tree record has no slot for a stored size or mask,
4536    /// the same reason the array indexes reject a direct write.
4537    #[test]
4538    fn write_chunk_raw_at_rejects_an_unfiltered_btree_v2() {
4539        let path = temp_path("wcr_at_unfiltered");
4540        let file = H5File::create(&path).unwrap();
4541        let ds = file
4542            .new_dataset::<i32>()
4543            .shape([0, 0])
4544            .chunk(&[2, 2])
4545            .max_shape(&[None, None])
4546            .create("grid")
4547            .unwrap();
4548        let err = ds.write_chunk_raw_at(&[0, 0], &[0u8; 16], 0).unwrap_err();
4549        assert!(
4550            err.to_string().contains("filtered dataset"),
4551            "expected a filtered-dataset error, got: {err}"
4552        );
4553        std::fs::remove_file(&path).ok();
4554    }
4555
4556    /// A stored size that does not fit the index's chunk-size field must error
4557    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
4558    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
4559    /// a 70000-byte stored chunk overflows it.
4560    #[cfg(feature = "deflate")]
4561    #[test]
4562    fn write_chunk_raw_rejects_oversized_chunk() {
4563        let path = temp_path("wcr_oversized");
4564        let file = H5File::create(&path).unwrap();
4565        let ds = file
4566            .new_dataset::<i32>()
4567            .shape([0])
4568            .chunk(&[1])
4569            .max_shape(&[None])
4570            .deflate(4)
4571            .create("v")
4572            .unwrap();
4573        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
4574        assert!(
4575            err.to_string().contains("does not fit"),
4576            "expected a chunk-size-field overflow error, got: {err}"
4577        );
4578        std::fs::remove_file(&path).ok();
4579    }
4580
4581    // ---- issue #5: runtime-width fixed-string reading ----------------------
4582
4583    use crate::format::messages::datatype::DatatypeMessage;
4584
4585    /// Build a 1-D fixed-string dataset of `width` bytes per element from raw
4586    /// element images, optionally chunked and deflated.
4587    fn write_fixed_string_dataset(
4588        path: &std::path::Path,
4589        dt: DatatypeMessage,
4590        width: usize,
4591        elems: &[&[u8]],
4592        compressed: bool,
4593    ) {
4594        let mut raw = Vec::with_capacity(elems.len() * width);
4595        for e in elems {
4596            assert!(e.len() <= width);
4597            raw.extend_from_slice(e);
4598            raw.resize(raw.len() + (width - e.len()), 0);
4599        }
4600        let file = H5File::create(path).unwrap();
4601        let mut b = file.new_dataset::<u8>().datatype(dt).shape([elems.len()]);
4602        if compressed {
4603            b = b.chunk(&[2]).deflate(6);
4604        }
4605        let ds = b.create("labels").unwrap();
4606        ds.write_raw_bytes(&raw).unwrap();
4607        file.close().unwrap();
4608    }
4609
4610    /// The width is whatever the file says, so one call reads a 24-byte label
4611    /// column and a 100-byte one. Producers like VASP pick it per dataset.
4612    #[test]
4613    fn read_strings_handles_any_fixed_width() {
4614        for width in [4usize, 24, 100] {
4615            let path = temp_path(&format!("fixed_str_{width}"));
4616            write_fixed_string_dataset(
4617                &path,
4618                DatatypeMessage::fixed_string(width as u32),
4619                width,
4620                &[b"ab", b"cde", b""],
4621                false,
4622            );
4623            let file = H5File::open(&path).unwrap();
4624            let got = file.dataset("labels").unwrap().read_strings().unwrap();
4625            assert_eq!(got, vec!["ab", "cde", ""], "width {width}");
4626            std::fs::remove_file(&path).ok();
4627        }
4628    }
4629
4630    /// Each padding rule decides where the value ends. Null-terminated stops at
4631    /// the first NUL and ignores the bytes after it; the two pad rules strip a
4632    /// tail of that byte and keep everything before it.
4633    #[test]
4634    fn read_strings_honors_every_padding_rule() {
4635        // "ab" then a NUL then trailing junk a null-terminated read must drop
4636        // and a null-padded read must keep.
4637        let elem: &[u8] = b"ab\0X\0\0";
4638        for (padding, want) in [(0u8, "ab"), (1, "ab\0X")] {
4639            let path = temp_path(&format!("fixed_pad_{padding}"));
4640            write_fixed_string_dataset(
4641                &path,
4642                DatatypeMessage::FixedString {
4643                    size: 6,
4644                    padding,
4645                    charset: 0,
4646                },
4647                6,
4648                &[elem],
4649                false,
4650            );
4651            let file = H5File::open(&path).unwrap();
4652            let got = file.dataset("labels").unwrap().read_strings().unwrap();
4653            assert_eq!(got, vec![want.to_string()], "padding {padding}");
4654            std::fs::remove_file(&path).ok();
4655        }
4656        // Space-padded keeps interior spaces and strips only the tail.
4657        let path = temp_path("fixed_pad_2");
4658        write_fixed_string_dataset(
4659            &path,
4660            DatatypeMessage::FixedString {
4661                size: 8,
4662                padding: 2,
4663                charset: 0,
4664            },
4665            8,
4666            &[b"a b     "],
4667            false,
4668        );
4669        let file = H5File::open(&path).unwrap();
4670        assert_eq!(
4671            file.dataset("labels").unwrap().read_strings().unwrap(),
4672            vec!["a b".to_string()]
4673        );
4674        std::fs::remove_file(&path).ok();
4675    }
4676
4677    /// A reserved padding or character-set code is an error naming the element,
4678    /// not a guess.
4679    #[test]
4680    fn read_strings_rejects_reserved_datatype_codes() {
4681        for (padding, charset, want) in [(3u8, 0u8, "padding rule 3"), (0, 7, "character set 7")] {
4682            let path = temp_path(&format!("fixed_reserved_{padding}_{charset}"));
4683            write_fixed_string_dataset(
4684                &path,
4685                DatatypeMessage::FixedString {
4686                    size: 4,
4687                    padding,
4688                    charset,
4689                },
4690                4,
4691                &[b"ab"],
4692                false,
4693            );
4694            let file = H5File::open(&path).unwrap();
4695            let err = file
4696                .dataset("labels")
4697                .unwrap()
4698                .read_strings()
4699                .unwrap_err()
4700                .to_string();
4701            assert!(err.contains(want), "got: {err}");
4702            std::fs::remove_file(&path).ok();
4703        }
4704    }
4705
4706    /// The declared character set is enforced: a byte that cannot be decoded is
4707    /// an error naming the element, and the lossy call is what accepts the file
4708    /// instead of a silent substitution here.
4709    #[test]
4710    fn read_strings_enforces_the_character_set_and_lossy_does_not() {
4711        // Latin-1 "é" (0xE9) in a dataset that declares ASCII, and a lone 0xFF
4712        // in one that declares UTF-8.
4713        for (charset, bytes, want) in [
4714            (0u8, b"caf\xe9".as_slice(), "ASCII character set"),
4715            (1, b"a\xff".as_slice(), "not valid UTF-8"),
4716        ] {
4717            let path = temp_path(&format!("fixed_charset_{charset}"));
4718            write_fixed_string_dataset(
4719                &path,
4720                DatatypeMessage::FixedString {
4721                    size: 6,
4722                    padding: 1,
4723                    charset,
4724                },
4725                6,
4726                &[b"ok", bytes],
4727                false,
4728            );
4729            let file = H5File::open(&path).unwrap();
4730            let ds = file.dataset("labels").unwrap();
4731            let err = ds.read_strings().unwrap_err().to_string();
4732            assert!(err.contains(want) && err.contains("string 1"), "got: {err}");
4733            let lossy = ds.read_strings_lossy().unwrap();
4734            assert_eq!(lossy[0], "ok");
4735            assert_eq!(
4736                lossy[1].chars().next().unwrap(),
4737                if charset == 0 { 'c' } else { 'a' }
4738            );
4739            std::fs::remove_file(&path).ok();
4740        }
4741    }
4742
4743    /// Valid multi-byte UTF-8 survives, and the trailing NUL padding does not
4744    /// split a character.
4745    #[test]
4746    fn read_strings_reads_utf8_fixed_strings() {
4747        let path = temp_path("fixed_utf8");
4748        write_fixed_string_dataset(
4749            &path,
4750            DatatypeMessage::fixed_string_utf8(12),
4751            12,
4752            &["héllo".as_bytes(), "안녕".as_bytes()],
4753            false,
4754        );
4755        let file = H5File::open(&path).unwrap();
4756        assert_eq!(
4757            file.dataset("labels").unwrap().read_strings().unwrap(),
4758            vec!["héllo".to_string(), "안녕".to_string()]
4759        );
4760        std::fs::remove_file(&path).ok();
4761    }
4762
4763    /// The decode sits on the decoded raw-data path, so a chunked and deflated
4764    /// dataset reads the same as a contiguous one.
4765    #[cfg(feature = "deflate")]
4766    #[test]
4767    fn read_strings_reads_a_compressed_fixed_string_dataset() {
4768        let path = temp_path("fixed_str_deflate");
4769        write_fixed_string_dataset(
4770            &path,
4771            DatatypeMessage::fixed_string(16),
4772            16,
4773            &[b"alpha", b"beta", b"gamma", b"delta", b"epsilon"],
4774            true,
4775        );
4776        let file = H5File::open(&path).unwrap();
4777        assert_eq!(
4778            file.dataset("labels").unwrap().read_strings().unwrap(),
4779            vec!["alpha", "beta", "gamma", "delta", "epsilon"]
4780        );
4781        std::fs::remove_file(&path).ok();
4782    }
4783
4784    /// One call covers both string datatypes, so a caller need not branch on
4785    /// which one the file used.
4786    #[test]
4787    fn read_strings_also_reads_variable_length_strings() {
4788        let path = temp_path("read_strings_vlen");
4789        {
4790            let file = H5File::create(&path).unwrap();
4791            file.write_vlen_strings("names", &["alpha", "", "안녕"])
4792                .unwrap();
4793            file.close().unwrap();
4794        }
4795        let file = H5File::open(&path).unwrap();
4796        assert_eq!(
4797            file.dataset("names").unwrap().read_strings().unwrap(),
4798            vec!["alpha".to_string(), String::new(), "안녕".to_string()]
4799        );
4800        std::fs::remove_file(&path).ok();
4801    }
4802
4803    /// A file declaring a zero-width fixed string is an error, not the panic
4804    /// `chunks_exact(0)` would raise. Nothing in this crate writes one, so the
4805    /// test patches the width in the encoded datatype message down to zero and
4806    /// re-stamps the object header's checksum over the result.
4807    #[test]
4808    fn read_strings_rejects_a_zero_width_fixed_string_dataset() {
4809        use crate::format::checksum::checksum_metadata;
4810        use crate::format::object_header::OHDR_SIGNATURE;
4811
4812        let path = temp_path("fixed_str_zero_width");
4813        write_fixed_string_dataset(
4814            &path,
4815            DatatypeMessage::fixed_string(37),
4816            37,
4817            &[b"ab", b"cd"],
4818            false,
4819        );
4820
4821        // Version 1 string datatype: class|version, padding|charset, two
4822        // reserved bytes, then the width as a little-endian u32. The width is
4823        // 37 so the eight bytes occur once in the file.
4824        let mut bytes = std::fs::read(&path).unwrap();
4825        let needle = [0x13u8, 0, 0, 0, 37, 0, 0, 0];
4826        let at = bytes
4827            .windows(needle.len())
4828            .position(|w| w == needle)
4829            .expect("encoded fixed-string datatype message");
4830        assert!(
4831            !bytes[at + 1..].windows(needle.len()).any(|w| w == needle),
4832            "the datatype message pattern is not unique in the file"
4833        );
4834
4835        // The enclosing v2 object header ends in a checksum over everything
4836        // from its signature onwards; find the offset where the stored value
4837        // still agrees, so the patched header can be re-stamped there.
4838        let ohdr = bytes[..at]
4839            .windows(4)
4840            .rposition(|w| w == OHDR_SIGNATURE)
4841            .expect("enclosing object header");
4842        let cksum_at = (at + needle.len()..bytes.len() - 4)
4843            .find(|&e| {
4844                u32::from_le_bytes(bytes[e..e + 4].try_into().unwrap())
4845                    == checksum_metadata(&bytes[ohdr..e])
4846            })
4847            .expect("object header checksum");
4848
4849        bytes[at + 4..at + 8].copy_from_slice(&0u32.to_le_bytes());
4850        let fixed = checksum_metadata(&bytes[ohdr..cksum_at]);
4851        bytes[cksum_at..cksum_at + 4].copy_from_slice(&fixed.to_le_bytes());
4852        std::fs::write(&path, &bytes).unwrap();
4853
4854        let file = H5File::open(&path).unwrap();
4855        let err = file
4856            .dataset("labels")
4857            .unwrap()
4858            .read_strings()
4859            .unwrap_err()
4860            .to_string();
4861        assert!(err.contains("zero width"), "got: {err}");
4862        std::fs::remove_file(&path).ok();
4863    }
4864
4865    /// A non-string dataset is an error, not an attempt to reinterpret bytes.
4866    #[test]
4867    fn read_strings_rejects_a_non_string_dataset() {
4868        let path = temp_path("read_strings_numeric");
4869        {
4870            let file = H5File::create(&path).unwrap();
4871            let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
4872            ds.write_raw(&[1i32, 2, 3]).unwrap();
4873            file.close().unwrap();
4874        }
4875        let file = H5File::open(&path).unwrap();
4876        let err = file
4877            .dataset("nums")
4878            .unwrap()
4879            .read_strings()
4880            .unwrap_err()
4881            .to_string();
4882        assert!(err.contains("only for string datasets"), "got: {err}");
4883        std::fs::remove_file(&path).ok();
4884    }
4885
4886    // ---- issue #6: random updates to vlen string datasets ------------------
4887
4888    /// One element changes; the extent and every other element stay as they
4889    /// were, on a contiguous vlen dataset.
4890    #[test]
4891    fn write_vlen_strings_slice_replaces_one_element() {
4892        let path = temp_path("vlen_slice_contig");
4893        {
4894            let file = H5File::create(&path).unwrap();
4895            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
4896                .unwrap();
4897            file.close().unwrap();
4898        }
4899        {
4900            let file = H5File::open_rw(&path).unwrap();
4901            file.dataset_writer("notes")
4902                .unwrap()
4903                .write_vlen_strings_slice(1, &["replacement"])
4904                .unwrap();
4905            file.close().unwrap();
4906        }
4907        let file = H5File::open(&path).unwrap();
4908        let ds = file.dataset("notes").unwrap();
4909        assert_eq!(ds.shape(), vec![4]);
4910        assert_eq!(
4911            ds.read_vlen_strings().unwrap(),
4912            vec!["a", "replacement", "c", "d"]
4913        );
4914        std::fs::remove_file(&path).ok();
4915    }
4916
4917    /// The same on an appendable chunked dataset, across a reopen, over a range
4918    /// that spans a chunk boundary.
4919    #[test]
4920    fn write_vlen_strings_slice_spans_chunks_after_reopen() {
4921        let path = temp_path("vlen_slice_chunked");
4922        {
4923            let file = H5File::create(&path).unwrap();
4924            file.create_appendable_vlen_dataset("notes", 2, None)
4925                .unwrap();
4926            let all: Vec<String> = (0..6).map(|i| format!("v{i}")).collect();
4927            let refs: Vec<&str> = all.iter().map(|s| s.as_str()).collect();
4928            file.append_vlen_strings("notes", &refs).unwrap();
4929            file.close().unwrap();
4930        }
4931        {
4932            // Elements 1..4 cross the 2-element chunk boundary twice.
4933            let file = H5File::open_rw(&path).unwrap();
4934            file.dataset_writer("notes")
4935                .unwrap()
4936                .write_vlen_strings_slice(1, &["x", "y", "z"])
4937                .unwrap();
4938            file.close().unwrap();
4939        }
4940        let file = H5File::open(&path).unwrap();
4941        let ds = file.dataset("notes").unwrap();
4942        assert_eq!(ds.shape(), vec![6]);
4943        assert_eq!(
4944            ds.read_vlen_strings().unwrap(),
4945            vec!["v0", "x", "y", "z", "v4", "v5"]
4946        );
4947        std::fs::remove_file(&path).ok();
4948    }
4949
4950    /// Elements the append buffer still holds are not on disk yet; the
4951    /// update flushes them to their chunks first, so the flush at close has
4952    /// nothing left to write the pre-update reference over.
4953    #[test]
4954    fn write_vlen_strings_slice_updates_buffered_elements() {
4955        let path = temp_path("vlen_slice_buffered");
4956        {
4957            let file = H5File::create(&path).unwrap();
4958            file.create_appendable_vlen_dataset("notes", 4, None)
4959                .unwrap();
4960            // 3 of a 4-element chunk: all three stay in the append buffer.
4961            file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
4962            file.dataset_writer("notes")
4963                .unwrap()
4964                .write_vlen_strings_slice(1, &["patched"])
4965                .unwrap();
4966            file.close().unwrap();
4967        }
4968        let file = H5File::open(&path).unwrap();
4969        assert_eq!(
4970            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
4971            vec!["a", "patched", "c"]
4972        );
4973        std::fs::remove_file(&path).ok();
4974    }
4975
4976    /// A range past the end is rejected before anything is written, and an
4977    /// empty batch costs the file nothing — without the early return it would
4978    /// still allocate and write an empty global-heap collection.
4979    #[test]
4980    fn write_vlen_strings_slice_checks_its_range() {
4981        let build = |name: &str, empty_call: bool| {
4982            let path = temp_path(name);
4983            let file = H5File::create(&path).unwrap();
4984            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
4985            let ds = file.dataset_writer("notes").unwrap();
4986            let err = ds
4987                .write_vlen_strings_slice(1, &["x", "y"])
4988                .unwrap_err()
4989                .to_string();
4990            assert!(
4991                err.contains("outside the dataset's 2 elements"),
4992                "got: {err}"
4993            );
4994            if empty_call {
4995                ds.write_vlen_strings_slice(0, &[]).unwrap();
4996            }
4997            file.close().unwrap();
4998            path
4999        };
5000
5001        let with_empty = build("vlen_slice_range", true);
5002        let control = build("vlen_slice_range_control", false);
5003        assert_eq!(
5004            std::fs::metadata(&with_empty).unwrap().len(),
5005            std::fs::metadata(&control).unwrap().len(),
5006            "the rejected and empty calls must leave the file untouched"
5007        );
5008
5009        let file = H5File::open(&with_empty).unwrap();
5010        assert_eq!(
5011            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5012            vec!["a", "b"]
5013        );
5014        std::fs::remove_file(&with_empty).ok();
5015        std::fs::remove_file(&control).ok();
5016    }
5017
5018    /// The element offset is one-dimensional, so a multi-dimensional dataset is
5019    /// rejected rather than silently indexed along the first axis.
5020    #[test]
5021    fn write_vlen_strings_slice_rejects_a_multidimensional_dataset() {
5022        let path = temp_path("vlen_slice_2d");
5023        let file = H5File::create(&path).unwrap();
5024        let ds = file
5025            .new_dataset::<u8>()
5026            .datatype(DatatypeMessage::vlen_string_utf8())
5027            .shape([2, 3])
5028            .create("grid")
5029            .unwrap();
5030        let err = ds
5031            .write_vlen_strings_slice(0, &["x"])
5032            .unwrap_err()
5033            .to_string();
5034        assert!(err.contains("1-dimension datasets"), "got: {err}");
5035        file.close().unwrap();
5036        std::fs::remove_file(&path).ok();
5037    }
5038
5039    /// A `&str` is UTF-8, so writing a non-ASCII one into a dataset that
5040    /// declares the ASCII character set would mislabel the bytes.
5041    #[test]
5042    fn write_vlen_strings_slice_enforces_the_ascii_character_set() {
5043        let path = temp_path("vlen_slice_ascii");
5044        let file = H5File::create(&path).unwrap();
5045        let ds = file
5046            .new_dataset::<u8>()
5047            .datatype(DatatypeMessage::vlen_string_ascii())
5048            .shape([3])
5049            .create("notes")
5050            .unwrap();
5051        let err = ds
5052            .write_vlen_strings_slice(0, &["ok", "안녕"])
5053            .unwrap_err()
5054            .to_string();
5055        assert!(
5056            err.contains("string 1") && err.contains("is not ASCII"),
5057            "got: {err}"
5058        );
5059        ds.write_vlen_strings_slice(0, &["ok", "fine"]).unwrap();
5060        file.close().unwrap();
5061        std::fs::remove_file(&path).ok();
5062    }
5063
5064    /// A numeric dataset is rejected: its elements are not vlen references and
5065    /// writing one would corrupt the column.
5066    #[test]
5067    fn write_vlen_strings_slice_rejects_a_non_vlen_dataset() {
5068        let path = temp_path("vlen_slice_numeric");
5069        let file = H5File::create(&path).unwrap();
5070        let ds = file.new_dataset::<i32>().shape([3]).create("nums").unwrap();
5071        ds.write_raw(&[1i32, 2, 3]).unwrap();
5072        let err = ds
5073            .write_vlen_strings_slice(0, &["x"])
5074            .unwrap_err()
5075            .to_string();
5076        assert!(
5077            err.contains("only for variable-length string datasets"),
5078            "got: {err}"
5079        );
5080        file.close().unwrap();
5081        std::fs::remove_file(&path).ok();
5082    }
5083
5084    // ---- superseded global heap objects (libhdf5 H5HG_remove parity) -------
5085
5086    /// Repeatedly replacing the same element must not grow the file per
5087    /// update: the collection each update supersedes is freed and the next
5088    /// update's collection lands in that block. Without the release every
5089    /// update costs another `H5HG_MINALLOC` (4096) bytes.
5090    #[test]
5091    fn write_vlen_strings_slice_reuses_the_freed_heap_block() {
5092        let size_after = |updates: usize| {
5093            let path = temp_path(&format!("vlen_slice_heap_reuse_{updates}"));
5094            let file = H5File::create(&path).unwrap();
5095            file.write_vlen_strings("notes", &["a", "b"]).unwrap();
5096            let ds = file.dataset_writer("notes").unwrap();
5097            for i in 0..updates {
5098                ds.write_vlen_strings_slice(0, &[&format!("update {i}")])
5099                    .unwrap();
5100            }
5101            file.close().unwrap();
5102            let n = std::fs::metadata(&path).unwrap().len();
5103            let read = H5File::open(&path).unwrap();
5104            assert_eq!(
5105                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5106                vec![format!("update {}", updates - 1), "b".to_string()]
5107            );
5108            drop(read);
5109            std::fs::remove_file(&path).ok();
5110            n
5111        };
5112
5113        // The allocator settles once a freed block is available to reuse, so
5114        // every count past that produces the same file.
5115        let settled = size_after(3);
5116        assert_eq!(size_after(20), settled, "20 updates against 3");
5117        assert_eq!(size_after(50), settled, "50 updates against 3");
5118    }
5119
5120    /// An empty string is stored as a real heap object under a reference whose
5121    /// sequence length is zero, so the release must go by the address, not the
5122    /// length — a length test strands the object and its collection forever.
5123    #[test]
5124    fn write_vlen_strings_slice_frees_an_empty_strings_object() {
5125        let size_after = |updates: usize| {
5126            let path = temp_path(&format!("vlen_slice_empty_reuse_{updates}"));
5127            let file = H5File::create(&path).unwrap();
5128            file.write_vlen_strings("notes", &["", "b"]).unwrap();
5129            let ds = file.dataset_writer("notes").unwrap();
5130            for _ in 0..updates {
5131                ds.write_vlen_strings_slice(0, &[""]).unwrap();
5132            }
5133            file.close().unwrap();
5134            let n = std::fs::metadata(&path).unwrap().len();
5135            let read = H5File::open(&path).unwrap();
5136            assert_eq!(
5137                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5138                vec!["".to_string(), "b".to_string()]
5139            );
5140            drop(read);
5141            std::fs::remove_file(&path).ok();
5142            n
5143        };
5144
5145        let settled = size_after(3);
5146        assert_eq!(size_after(20), settled, "20 empty updates against 3");
5147        assert_eq!(size_after(50), settled, "50 empty updates against 3");
5148    }
5149
5150    /// The elements the update does not name keep their strings, so freeing
5151    /// the superseded objects must not disturb the collection's survivors.
5152    #[test]
5153    fn write_vlen_strings_slice_keeps_the_untouched_strings_readable() {
5154        let path = temp_path("vlen_slice_heap_survivors");
5155        {
5156            let file = H5File::create(&path).unwrap();
5157            file.write_vlen_strings("notes", &["a", "b", "c", "d"])
5158                .unwrap();
5159            let ds = file.dataset_writer("notes").unwrap();
5160            // Two updates inside the one collection the create wrote, so the
5161            // second reads a collection the first already rewrote.
5162            ds.write_vlen_strings_slice(1, &["B"]).unwrap();
5163            ds.write_vlen_strings_slice(3, &["D"]).unwrap();
5164            file.close().unwrap();
5165        }
5166        let file = H5File::open(&path).unwrap();
5167        assert_eq!(
5168            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5169            vec!["a", "B", "c", "D"]
5170        );
5171        std::fs::remove_file(&path).ok();
5172    }
5173
5174    /// Replacing every element of a chunked dataset empties the collection the
5175    /// append wrote, and the file must still read back correctly after its
5176    /// block goes to the allocator.
5177    #[test]
5178    fn write_vlen_strings_slice_frees_an_emptied_collection() {
5179        let path = temp_path("vlen_slice_heap_emptied");
5180        {
5181            let file = H5File::create(&path).unwrap();
5182            file.create_appendable_vlen_dataset("notes", 2, None)
5183                .unwrap();
5184            file.append_vlen_strings("notes", &["p", "q", "r", "s"])
5185                .unwrap();
5186            file.close().unwrap();
5187        }
5188        {
5189            let file = H5File::open_rw(&path).unwrap();
5190            file.dataset_writer("notes")
5191                .unwrap()
5192                .write_vlen_strings_slice(0, &["w", "x", "y", "z"])
5193                .unwrap();
5194            file.close().unwrap();
5195        }
5196        let file = H5File::open(&path).unwrap();
5197        assert_eq!(
5198            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5199            vec!["w", "x", "y", "z"]
5200        );
5201        std::fs::remove_file(&path).ok();
5202    }
5203
5204    /// A collection larger than the 4096-byte minimum must keep its size when
5205    /// an object leaves it. Re-encoding at the natural size instead shrinks
5206    /// what the header declares, so the block's tail stops being part of the
5207    /// collection and the eventual free returns less than was allocated —
5208    /// stranding the difference on every cycle.
5209    #[test]
5210    fn write_vlen_strings_slice_keeps_an_oversized_collections_block_whole() {
5211        let big = |tag: char| std::iter::repeat_n(tag, 2000).collect::<String>();
5212        let size_after = |cycles: usize| {
5213            let path = temp_path(&format!("vlen_slice_heap_big_{cycles}"));
5214            let file = H5File::create(&path).unwrap();
5215            let seed: Vec<String> = "abcd".chars().map(big).collect();
5216            let refs: Vec<&str> = seed.iter().map(|s| s.as_str()).collect();
5217            // Four 2000-byte strings do not fit the 4096-byte minimum, so this
5218            // is one collection well above it.
5219            file.write_vlen_strings("notes", &refs).unwrap();
5220            let ds = file.dataset_writer("notes").unwrap();
5221            for _ in 0..cycles {
5222                // Partially empty the collection, then finish it off: the
5223                // block is freed only after it has been rewritten once.
5224                let head = big('x');
5225                ds.write_vlen_strings_slice(0, &[&head]).unwrap();
5226                let tail: Vec<String> = "yzw".chars().map(big).collect();
5227                let tail_refs: Vec<&str> = tail.iter().map(|s| s.as_str()).collect();
5228                ds.write_vlen_strings_slice(1, &tail_refs).unwrap();
5229            }
5230            file.close().unwrap();
5231            let n = std::fs::metadata(&path).unwrap().len();
5232            let read = H5File::open(&path).unwrap();
5233            assert_eq!(
5234                read.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5235                vec![big('x'), big('y'), big('z'), big('w')]
5236            );
5237            drop(read);
5238            std::fs::remove_file(&path).ok();
5239            n
5240        };
5241
5242        let settled = size_after(4);
5243        assert_eq!(size_after(30), settled, "30 cycles against 4");
5244    }
5245
5246    /// An element still in the append buffer has never been on disk, so its
5247    /// superseded object has to be found in the buffer or it is stranded.
5248    #[test]
5249    fn write_vlen_strings_slice_releases_a_buffered_elements_object() {
5250        let path = temp_path("vlen_slice_heap_buffered");
5251        let file = H5File::create(&path).unwrap();
5252        file.create_appendable_vlen_dataset("notes", 4, None)
5253            .unwrap();
5254        file.append_vlen_strings("notes", &["a", "b", "c"]).unwrap();
5255        let ds = file.dataset_writer("notes").unwrap();
5256        for i in 0..20 {
5257            ds.write_vlen_strings_slice(1, &[&format!("patch {i}")])
5258                .unwrap();
5259        }
5260        file.close().unwrap();
5261
5262        let file = H5File::open(&path).unwrap();
5263        assert_eq!(
5264            file.dataset("notes").unwrap().read_vlen_strings().unwrap(),
5265            vec!["a", "patch 19", "c"]
5266        );
5267        let size = std::fs::metadata(&path).unwrap().len();
5268        std::fs::remove_file(&path).ok();
5269        assert!(
5270            size < 20 * 4096,
5271            "20 buffered updates left {size} bytes, one collection per update"
5272        );
5273    }
5274
5275    /// Regression: a typed `write_slice` into rows the append buffer still
5276    /// held wrote the chunks, and the flush at close wrote the stale buffered
5277    /// rows back over it — write 99, read 50. The slice now flushes the
5278    /// buffer first, making the chunks the single authority for those rows.
5279    #[test]
5280    fn write_slice_into_the_buffered_tail_survives_close() {
5281        let path = temp_path("slice_into_buffered_tail");
5282        {
5283            let file = H5File::create(&path).unwrap();
5284            let ds = file
5285                .new_dataset::<i32>()
5286                .shape([0])
5287                .chunk(&[4])
5288                .max_shape(&[None])
5289                .create("d")
5290                .unwrap();
5291            // 6 rows: 4 land in chunk 0, rows 4 and 5 stay buffered.
5292            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap();
5293            ds.write_slice(&[4], &[1], &[99]).unwrap();
5294            file.close().unwrap();
5295        }
5296        {
5297            let file = H5File::open(&path).unwrap();
5298            let ds = file.dataset("d").unwrap();
5299            assert_eq!(ds.read_raw::<i32>().unwrap(), vec![10, 11, 12, 13, 99, 51]);
5300        }
5301        std::fs::remove_file(&path).ok();
5302    }
5303
5304    /// Extending a dataset while appends sit in the buffer must not move
5305    /// them: the buffer records the absolute row its frames belong to, so
5306    /// the flush at close lands them there, and the grown region reads as
5307    /// fill.
5308    #[test]
5309    fn extend_does_not_move_buffered_appends() {
5310        let path = temp_path("extend_keeps_buffered_rows");
5311        {
5312            let file = H5File::create(&path).unwrap();
5313            let ds = file
5314                .new_dataset::<i32>()
5315                .shape([0])
5316                .chunk(&[4])
5317                .max_shape(&[None])
5318                .create("d")
5319                .unwrap();
5320            ds.append(&[10, 11, 12, 13, 50, 51]).unwrap(); // rows 4, 5 buffered
5321            ds.extend(&[10]).unwrap();
5322            file.close().unwrap();
5323        }
5324        {
5325            let file = H5File::open(&path).unwrap();
5326            let ds = file.dataset("d").unwrap();
5327            assert_eq!(
5328                ds.read_raw::<i32>().unwrap(),
5329                vec![10, 11, 12, 13, 50, 51, 0, 0, 0, 0]
5330            );
5331        }
5332        std::fs::remove_file(&path).ok();
5333    }
5334
5335    /// Regression: appends to a v2 B-tree indexed dataset (two unlimited
5336    /// dimensions) buffered fine but close() failed "not a chunked dataset"
5337    /// and lost the buffered rows — the append's chunk writes required the
5338    /// extensible-array index. They now go through the index-generic
5339    /// hyperslab engine.
5340    #[test]
5341    fn append_to_a_btree_v2_dataset_survives_close() {
5342        let path = temp_path("append_bt2_close");
5343        {
5344            let file = H5File::create(&path).unwrap();
5345            let ds = file
5346                .new_dataset::<i32>()
5347                .shape([0, 3])
5348                .chunk(&[4, 3])
5349                .max_shape(&[None, None])
5350                .create("d")
5351                .unwrap();
5352            // One buffered row, then a batch that crosses the chunk
5353            // boundary: 4 rows fill chunk band 0, one row stays buffered
5354            // for the flush at close.
5355            ds.append(&[1, 2, 3]).unwrap();
5356            ds.append(&(4..=15).collect::<Vec<i32>>()).unwrap();
5357            file.close().unwrap();
5358        }
5359        {
5360            let file = H5File::open(&path).unwrap();
5361            let ds = file.dataset("d").unwrap();
5362            assert_eq!(ds.shape(), vec![5, 3]);
5363            assert_eq!(
5364                ds.read_raw::<i32>().unwrap(),
5365                (1..=15).collect::<Vec<i32>>()
5366            );
5367        }
5368        std::fs::remove_file(&path).ok();
5369    }
5370
5371    /// A chunk row narrower than the frame row is legal geometry (libhdf5
5372    /// creates it); appended frames must be scattered across the row's
5373    /// tiles at the chunk stride, not packed at the frame stride.
5374    #[test]
5375    fn append_scatters_frames_across_narrow_chunk_tiles() {
5376        let path = temp_path("append_narrow_chunks");
5377        {
5378            let file = H5File::create(&path).unwrap();
5379            let ds = file
5380                .new_dataset::<i32>()
5381                .shape([0, 8])
5382                .chunk(&[2, 4])
5383                .max_shape(&[None, Some(8)])
5384                .create("d")
5385                .unwrap();
5386            // 3 rows of 8: rows 0..2 complete chunk band 0 (two tiles),
5387            // row 2 is flushed partial at close.
5388            ds.append(&(0..24).collect::<Vec<i32>>()).unwrap();
5389            file.close().unwrap();
5390        }
5391        {
5392            let file = H5File::open(&path).unwrap();
5393            let ds = file.dataset("d").unwrap();
5394            assert_eq!(ds.shape(), vec![3, 8]);
5395            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..24).collect::<Vec<i32>>());
5396        }
5397        std::fs::remove_file(&path).ok();
5398    }
5399
5400    /// A fixed-array dataset has no room to grow: appending must surface an
5401    /// error naming the chunk grid, not lose rows silently. (Before the
5402    /// index-generic append it failed as "not a chunked dataset".)
5403    #[test]
5404    fn append_to_a_full_fixed_array_dataset_errors() {
5405        let path = temp_path("append_fa_errors");
5406        let file = H5File::create(&path).unwrap();
5407        let ds = file
5408            .new_dataset::<i32>()
5409            .shape([4, 3])
5410            .chunk(&[2, 3])
5411            .create("d")
5412            .unwrap();
5413        let err = ds.append(&(0..6).collect::<Vec<i32>>()).unwrap_err();
5414        assert!(
5415            err.to_string().contains("chunk grid"),
5416            "unexpected error: {err}"
5417        );
5418        file.close().unwrap();
5419        std::fs::remove_file(&path).ok();
5420    }
5421
5422    /// A finite max_shape above the current shape used to be dropped on the
5423    /// fixed-array path: the array was sized from the current dims and the
5424    /// stored dataspace had no maximum, so growth failed. The array is now
5425    /// sized from the maximum's chunk grid (libhdf5 `max_nchunks`), so a
5426    /// fixed-max dataset appends up to its maximum and roundtrips.
5427    #[test]
5428    fn fixed_array_with_a_larger_max_shape_grows_and_survives_close() {
5429        let path = temp_path("fa_growable_dim0");
5430        {
5431            let file = H5File::create(&path).unwrap();
5432            let ds = file
5433                .new_dataset::<i32>()
5434                .shape([4, 3])
5435                .chunk(&[2, 3])
5436                .max_shape(&[Some(10), Some(3)])
5437                .create("d")
5438                .unwrap();
5439            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
5440            ds.append(&(12..18).collect::<Vec<i32>>()).unwrap();
5441            file.close().unwrap();
5442        }
5443        {
5444            let file = H5File::open(&path).unwrap();
5445            let ds = file.dataset("d").unwrap();
5446            assert_eq!(ds.shape(), vec![6, 3]);
5447            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..18).collect::<Vec<i32>>());
5448        }
5449        std::fs::remove_file(&path).ok();
5450    }
5451
5452    /// The multiplier-dimension boundary: growing a dimension other than 0
5453    /// changes the current chunk grid but not the index grid. Chunk slots
5454    /// must come from the maximum's grid (libhdf5 `max_down_chunks`), or the
5455    /// chunks written before the extend are looked up under different
5456    /// indices after it.
5457    #[test]
5458    fn fixed_array_growable_inner_dimension_keeps_chunk_slots() {
5459        let path = temp_path("fa_growable_dim1");
5460        {
5461            let file = H5File::create(&path).unwrap();
5462            let ds = file
5463                .new_dataset::<i32>()
5464                .shape([4, 3])
5465                .chunk(&[2, 3])
5466                .max_shape(&[Some(4), Some(9)])
5467                .create("d")
5468                .unwrap();
5469            ds.write_raw(&(0..12).collect::<Vec<i32>>()).unwrap();
5470            ds.extend(&[4, 6]).unwrap();
5471            ds.write_slice(&[0, 3], &[4, 3], &(12..24).collect::<Vec<i32>>())
5472                .unwrap();
5473            file.close().unwrap();
5474        }
5475        {
5476            let file = H5File::open(&path).unwrap();
5477            let ds = file.dataset("d").unwrap();
5478            assert_eq!(ds.shape(), vec![4, 6]);
5479            // Row-major [4,6]: row r is [r*3 .. r*3+3) from the first write
5480            // then [12 + r*3 ..) from the second.
5481            let mut expect = Vec::new();
5482            for r in 0i32..4 {
5483                expect.extend((r * 3)..(r * 3 + 3));
5484                expect.extend((12 + r * 3)..(12 + r * 3 + 3));
5485            }
5486            assert_eq!(ds.read_raw::<i32>().unwrap(), expect);
5487        }
5488        std::fs::remove_file(&path).ok();
5489    }
5490
5491    /// Growth boundaries: past the stored maximum is rejected, and a dataset
5492    /// without a stored maximum is fixed at its extent (libhdf5 defaults
5493    /// maxdims to dims at creation).
5494    #[test]
5495    fn extend_beyond_the_maximum_is_rejected() {
5496        let path = temp_path("extend_beyond_max");
5497        let file = H5File::create(&path).unwrap();
5498        let ds = file
5499            .new_dataset::<i32>()
5500            .shape([4, 3])
5501            .chunk(&[2, 3])
5502            .max_shape(&[Some(6), Some(3)])
5503            .create("d")
5504            .unwrap();
5505        ds.extend(&[6, 3]).unwrap();
5506        let err = ds.extend(&[8, 3]).unwrap_err();
5507        assert!(
5508            err.to_string().contains("exceeds the maximum"),
5509            "unexpected error: {err}"
5510        );
5511        file.close().unwrap();
5512        std::fs::remove_file(&path).ok();
5513    }
5514
5515    /// An unlimited dimension other than 0 has no fixed linear slot without
5516    /// libhdf5's extensible-array swizzling, which is not implemented;
5517    /// creating the geometry silently re-indexed chunks on every extend, so
5518    /// it is rejected at create.
5519    #[test]
5520    fn builder_rejects_an_unlimited_inner_dimension() {
5521        let path = temp_path("unlimited_inner_dim");
5522        let file = H5File::create(&path).unwrap();
5523        let err = match file
5524            .new_dataset::<i32>()
5525            .shape([4, 0])
5526            .chunk(&[2, 2])
5527            .max_shape(&[Some(4), None])
5528            .create("d")
5529        {
5530            Ok(_) => panic!("create accepted an unlimited inner dimension"),
5531            Err(e) => e,
5532        };
5533        assert!(
5534            err.to_string().contains("not the first"),
5535            "unexpected error: {err}"
5536        );
5537        file.close().unwrap();
5538        std::fs::remove_file(&path).ok();
5539    }
5540
5541    /// Regression: a chunk wider than a fixed max dimension used to be
5542    /// accepted, and appends then packed rows at the chunk stride — writing
5543    /// [1, 2, 3, 4] and reading back [1, 2, 0, 0]. libhdf5 rejects the
5544    /// geometry at create (`H5D__chunk_construct`); so do we now.
5545    #[test]
5546    fn builder_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
5547        let path = temp_path("builder_chunk_wider_than_max");
5548        let file = H5File::create(&path).unwrap();
5549        let err = match file
5550            .new_dataset::<i32>()
5551            .shape([0, 2])
5552            .chunk(&[2, 4])
5553            .max_shape(&[None, Some(2)])
5554            .create("v5")
5555        {
5556            Ok(_) => panic!("create accepted a chunk wider than the fixed max dimension"),
5557            Err(e) => e,
5558        };
5559        assert!(
5560            err.to_string().contains("maximum dimension size"),
5561            "unexpected error: {err}"
5562        );
5563        file.close().unwrap();
5564        std::fs::remove_file(&path).ok();
5565    }
5566}