Skip to main content

rust_hdf5/
dataset.rs

1//! Dataset creation and I/O.
2//!
3//! Datasets are created via the fluent [`DatasetBuilder`] API obtained from
4//! [`H5File::new_dataset`](crate::file::H5File::new_dataset). Once created,
5//! the [`H5Dataset`] handle can read or write raw typed data.
6
7use crate::attribute::AttrBuilder;
8use crate::error::{Hdf5Error, Result};
9use crate::file::{borrow_inner, borrow_inner_mut, clone_inner, H5FileInner, SharedInner};
10use crate::format::messages::datatype::DatatypeMessage;
11use crate::types::H5Type;
12
13// ---------------------------------------------------------------------------
14// DatasetBuilder
15// ---------------------------------------------------------------------------
16
17/// A fluent builder for creating datasets.
18///
19/// Obtained from [`H5File::new_dataset::<T>()`](crate::file::H5File::new_dataset).
20///
21/// ```no_run
22/// # use rust_hdf5::H5File;
23/// let file = H5File::create("builder.h5").unwrap();
24/// let ds = file.new_dataset::<f32>()
25///     .shape(&[10, 20])
26///     .create("temperatures")
27///     .unwrap();
28/// ```
29pub struct DatasetBuilder<T: H5Type> {
30    file_inner: SharedInner,
31    shape: Option<Vec<usize>>,
32    chunk_dims: Option<Vec<usize>>,
33    max_shape: Option<Vec<Option<usize>>>,
34    deflate_level: Option<u32>,
35    shuffle_deflate_level: Option<u32>,
36    custom_pipeline: Option<crate::format::messages::filter::FilterPipeline>,
37    group_path: Option<String>,
38    fill_value: Option<Vec<u8>>,
39    datatype_override: Option<crate::format::messages::datatype::DatatypeMessage>,
40    _marker: std::marker::PhantomData<T>,
41}
42
43impl<T: H5Type> DatasetBuilder<T> {
44    pub(crate) fn new(file_inner: SharedInner) -> Self {
45        Self {
46            file_inner,
47            shape: None,
48            chunk_dims: None,
49            max_shape: None,
50            deflate_level: None,
51            shuffle_deflate_level: None,
52            custom_pipeline: None,
53            group_path: None,
54            fill_value: None,
55            datatype_override: None,
56            _marker: std::marker::PhantomData,
57        }
58    }
59
60    pub(crate) fn new_in_group(file_inner: SharedInner, group_path: String) -> Self {
61        Self {
62            file_inner,
63            shape: None,
64            chunk_dims: None,
65            max_shape: None,
66            deflate_level: None,
67            shuffle_deflate_level: None,
68            custom_pipeline: None,
69            group_path: Some(group_path),
70            fill_value: None,
71            datatype_override: None,
72            _marker: std::marker::PhantomData,
73        }
74    }
75
76    /// Set the dataset dimensions.
77    ///
78    /// This is required before calling [`create`](Self::create).
79    /// Use an empty slice `&[]` for a scalar (0-dimensional) dataset.
80    #[must_use]
81    pub fn shape<S: AsRef<[usize]>>(mut self, dims: S) -> Self {
82        self.shape = Some(dims.as_ref().to_vec());
83        self
84    }
85
86    /// Create a scalar (0-dimensional) dataset holding a single value.
87    #[must_use]
88    pub fn scalar(mut self) -> Self {
89        self.shape = Some(vec![]);
90        self
91    }
92
93    /// Set chunk dimensions for chunked storage.
94    ///
95    /// When set, the dataset uses chunked storage with the extensible array
96    /// index. You should also call [`max_shape`](Self::max_shape) or
97    /// [`resizable`](Self::resizable) to allow extending.
98    #[must_use]
99    pub fn chunk(mut self, chunk_dims: &[usize]) -> Self {
100        self.chunk_dims = Some(chunk_dims.to_vec());
101        self
102    }
103
104    /// Make all dimensions unlimited (resizable).
105    ///
106    /// This sets max_dims to u64::MAX for all dimensions.
107    #[must_use]
108    pub fn resizable(mut self) -> Self {
109        self.max_shape = Some(vec![None; self.shape.as_ref().map_or(0, |s| s.len())]);
110        self
111    }
112
113    /// Set maximum dimensions. `None` means unlimited for that dimension.
114    #[must_use]
115    pub fn max_shape(mut self, max: &[Option<usize>]) -> Self {
116        self.max_shape = Some(max.to_vec());
117        self
118    }
119
120    /// Enable deflate (gzip) compression with the given level (0-9).
121    ///
122    /// Requires chunked storage (call `.chunk()` before `.create()`).
123    /// Level 0 = no compression, 9 = maximum compression. Default is 6.
124    #[must_use]
125    pub fn deflate(mut self, level: u32) -> Self {
126        self.deflate_level = Some(level);
127        self
128    }
129
130    /// Enable shuffle + deflate compression.
131    ///
132    /// Shuffle reorders bytes by position within elements before compression,
133    /// which typically improves compression ratios for numeric data.
134    /// Requires chunked storage.
135    #[must_use]
136    pub fn shuffle_deflate(mut self, level: u32) -> Self {
137        self.shuffle_deflate_level = Some(level);
138        self
139    }
140
141    /// Enable Zstandard compression with the given level (1-22, default 3).
142    ///
143    /// Requires chunked storage (call `.chunk()` before `.create()`).
144    #[must_use]
145    pub fn zstd(mut self, level: u32) -> Self {
146        self.custom_pipeline = Some(crate::format::messages::filter::FilterPipeline::zstd(level));
147        self
148    }
149
150    /// Set a custom filter pipeline for compression.
151    ///
152    /// This takes precedence over [`deflate`](Self::deflate) and
153    /// [`shuffle_deflate`](Self::shuffle_deflate). Requires chunked storage.
154    #[must_use]
155    pub fn filter_pipeline(
156        mut self,
157        pipeline: crate::format::messages::filter::FilterPipeline,
158    ) -> Self {
159        self.custom_pipeline = Some(pipeline);
160        self
161    }
162
163    /// Override the stored element datatype.
164    ///
165    /// By default the dataset is created with the datatype derived from the
166    /// Rust type parameter `T` ([`H5Type::hdf5_type`]). Use this to store a
167    /// different on-disk datatype than the in-memory element type — for
168    /// example a reduced-precision fixed-point type that matches an N-bit
169    /// filter (see [`FilterPipeline::nbit`]). The element *byte* size of the
170    /// override must equal `T::element_size()`; the N-bit filter packs the
171    /// significant bits within that fixed footprint.
172    ///
173    /// [`H5Type::hdf5_type`]: crate::H5Type::hdf5_type
174    /// [`FilterPipeline::nbit`]: crate::FilterPipeline::nbit
175    #[must_use]
176    pub fn datatype(mut self, dt: crate::format::messages::datatype::DatatypeMessage) -> Self {
177        self.datatype_override = Some(dt);
178        self
179    }
180
181    /// Set a user-defined fill value for unwritten elements.
182    ///
183    /// Without this, datasets use the HDF5 default zero-fill. When set,
184    /// the value is written into the dataset's fill-value message
185    /// (`fill_defined = 2`), so HDF5 readers treat unallocated chunks and
186    /// unwritten regions as this value rather than zero.
187    ///
188    /// ```no_run
189    /// # use rust_hdf5::H5File;
190    /// let file = H5File::create("fv.h5").unwrap();
191    /// let ds = file.new_dataset::<f32>()
192    ///     .shape(&[100])
193    ///     .fill_value(f32::NAN)
194    ///     .create("data")
195    ///     .unwrap();
196    /// ```
197    #[must_use]
198    pub fn fill_value(mut self, value: T) -> Self {
199        let es = T::element_size();
200        // Safety: `T: H5Type` is a `Copy` numeric primitive with a
201        // well-defined byte representation; `element_size()` matches
202        // `size_of::<T>()`. The slice borrows `value` only for this call.
203        let raw = unsafe { std::slice::from_raw_parts(&value as *const T as *const u8, es) };
204        self.fill_value = Some(raw.to_vec());
205        self
206    }
207
208    /// Finalize and create the dataset with the given `name`.
209    ///
210    /// The name is the link name within the root group (e.g. `"data"` or
211    /// `"group1/data"` once nested groups are supported).
212    pub fn create(self, name: &str) -> Result<H5Dataset> {
213        let shape = self.shape.ok_or_else(|| {
214            Hdf5Error::InvalidState("shape must be set before calling create()".into())
215        })?;
216
217        // Build the full name: if created within a group, prefix with group path
218        let full_name = if let Some(ref gp) = self.group_path {
219            if gp == "/" {
220                name.to_string()
221            } else {
222                let trimmed = gp.trim_start_matches('/');
223                format!("{}/{}", trimmed, name)
224            }
225        } else {
226            name.to_string()
227        };
228        let group_path = self.group_path.clone();
229        let fill_value = self.fill_value.clone();
230
231        let dims_u64: Vec<u64> = shape.iter().map(|&d| d as u64).collect();
232        let datatype = self.datatype_override.clone().unwrap_or_else(T::hdf5_type);
233        // Size one element from the on-disk datatype, not the carrier `T`. For
234        // the default path this equals `T::element_size()`; when a `datatype()`
235        // override is set (N-bit, or a runtime `CompoundType`), the stored type
236        // — not `T` — defines the element width, so the dataspace, the raw
237        // allocation, and the `write_raw` length check all agree with the bytes
238        // libhdf5/h5py will read.
239        let element_size = datatype.element_size() as usize;
240
241        // A filter pipeline requires chunked storage. When a filter is
242        // requested without explicit chunk dimensions, store the whole
243        // dataset as a single chunk instead of silently dropping the filter
244        // on the contiguous path. (This is one whole-dataset chunk, not
245        // h5py's ~1 MiB chunk-size heuristic; pass explicit chunk dimensions
246        // for large datasets.)
247        let wants_filter = self.custom_pipeline.is_some()
248            || self.shuffle_deflate_level.is_some()
249            || self.deflate_level.is_some();
250        let auto_chunk: Option<Vec<usize>> =
251            if self.chunk_dims.is_none() && wants_filter && !shape.is_empty() {
252                Some(shape.iter().map(|&d| d.max(1)).collect())
253            } else {
254                None
255            };
256
257        if let Some(chunk_dims) = self.chunk_dims.as_ref().or(auto_chunk.as_ref()) {
258            // Chunked dataset
259            let chunk_u64: Vec<u64> = chunk_dims.iter().map(|&d| d as u64).collect();
260            let max_u64: Vec<u64> = if let Some(ref max) = self.max_shape {
261                max.iter()
262                    .map(|m| m.map_or(u64::MAX, |v| v as u64))
263                    .collect()
264            } else {
265                // Default: max = current
266                dims_u64.clone()
267            };
268
269            // libhdf5 selects the chunk index from the dataspace: a v2
270            // B-tree for two or more unlimited dimensions, an extensible
271            // array for exactly one, and a fixed array when there are none.
272            let n_unlimited = max_u64.iter().filter(|&&m| m == u64::MAX).count();
273            let is_btree2 = n_unlimited >= 2;
274            let is_fixed_array = n_unlimited == 0;
275
276            let index = {
277                let inner = borrow_inner(&self.file_inner);
278                match &*inner {
279                    H5FileInner::Writer(writer) => {
280                        let idx = if is_btree2 {
281                            if wants_filter {
282                                return Err(Hdf5Error::InvalidState(
283                                    "compression of v2 B-tree (multi-unlimited-dimension) \
284                                     datasets is not yet supported"
285                                        .into(),
286                                ));
287                            }
288                            writer.create_btree_v2_dataset(
289                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
290                            )?
291                        } else if is_fixed_array {
292                            // A chunked dataset with no unlimited dimension
293                            // must use the fixed-array index — libhdf5
294                            // rejects an extensible-array index here. A
295                            // compressed fixed-shape dataset uses a *filtered*
296                            // fixed array (FA client id 1).
297                            if wants_filter {
298                                let pipeline = if let Some(p) = self.custom_pipeline {
299                                    p
300                                } else if let Some(level) = self.shuffle_deflate_level {
301                                    crate::format::messages::filter::FilterPipeline::shuffle_deflate(
302                                        T::element_size() as u32,
303                                        level,
304                                    )
305                                } else {
306                                    // deflate_level (checked by wants_filter).
307                                    crate::format::messages::filter::FilterPipeline::deflate(
308                                        self.deflate_level.unwrap(),
309                                    )
310                                };
311                                writer.create_fixed_array_dataset_with_pipeline(
312                                    &full_name, datatype, &dims_u64, &chunk_u64, pipeline,
313                                )?
314                            } else {
315                                writer.create_fixed_array_dataset(
316                                    &full_name, datatype, &dims_u64, &chunk_u64,
317                                )?
318                            }
319                        } else if let Some(pipeline) = self.custom_pipeline {
320                            writer.create_chunked_dataset_with_pipeline(
321                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
322                            )?
323                        } else if let Some(level) = self.shuffle_deflate_level {
324                            let pipeline =
325                                crate::format::messages::filter::FilterPipeline::shuffle_deflate(
326                                    T::element_size() as u32,
327                                    level,
328                                );
329                            writer.create_chunked_dataset_with_pipeline(
330                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, pipeline,
331                            )?
332                        } else if let Some(level) = self.deflate_level {
333                            writer.create_chunked_dataset_compressed(
334                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64, level,
335                            )?
336                        } else {
337                            writer.create_chunked_dataset(
338                                &full_name, datatype, &dims_u64, &max_u64, &chunk_u64,
339                            )?
340                        };
341                        if let Some(ref gp) = group_path {
342                            if gp != "/" {
343                                writer.assign_dataset_to_group(gp, idx)?;
344                            }
345                        }
346                        if let Some(ref fv) = fill_value {
347                            writer.set_dataset_fill_value(idx, fv.clone())?;
348                        }
349                        idx
350                    }
351                    H5FileInner::Reader(_) => {
352                        return Err(Hdf5Error::InvalidState(
353                            "cannot create a dataset in read mode".into(),
354                        ));
355                    }
356                    H5FileInner::Closed => {
357                        return Err(Hdf5Error::InvalidState("file is closed".into()));
358                    }
359                }
360            };
361
362            Ok(H5Dataset {
363                file_inner: clone_inner(&self.file_inner),
364                info: DatasetInfo::Writer {
365                    index,
366                    shape,
367                    element_size,
368                    chunked: true,
369                    btree2: is_btree2,
370                    fixed_array: is_fixed_array,
371                },
372            })
373        } else {
374            // Contiguous dataset (original path)
375            let index = {
376                let inner = borrow_inner(&self.file_inner);
377                match &*inner {
378                    H5FileInner::Writer(writer) => {
379                        let idx = writer.create_dataset(&full_name, datatype, &dims_u64)?;
380                        if let Some(ref gp) = group_path {
381                            if gp != "/" {
382                                writer.assign_dataset_to_group(gp, idx)?;
383                            }
384                        }
385                        if let Some(ref fv) = fill_value {
386                            writer.set_dataset_fill_value(idx, fv.clone())?;
387                        }
388                        idx
389                    }
390                    H5FileInner::Reader(_) => {
391                        return Err(Hdf5Error::InvalidState(
392                            "cannot create a dataset in read mode".into(),
393                        ));
394                    }
395                    H5FileInner::Closed => {
396                        return Err(Hdf5Error::InvalidState("file is closed".into()));
397                    }
398                }
399            };
400
401            Ok(H5Dataset {
402                file_inner: clone_inner(&self.file_inner),
403                info: DatasetInfo::Writer {
404                    index,
405                    shape,
406                    element_size,
407                    chunked: false,
408                    btree2: false,
409                    fixed_array: false,
410                },
411            })
412        }
413    }
414}
415
416// ---------------------------------------------------------------------------
417// DatasetInfo
418// ---------------------------------------------------------------------------
419
420/// Internal metadata about a dataset handle.
421enum DatasetInfo {
422    /// A dataset created via `new_dataset().create()` in write mode.
423    Writer {
424        /// Index into the writer's dataset list.
425        index: usize,
426        /// Shape (current dimensions).
427        shape: Vec<usize>,
428        /// Size of one element in bytes.
429        element_size: usize,
430        /// Whether this is a chunked dataset.
431        chunked: bool,
432        /// Whether the chunk index is a v2 B-tree (multiple unlimited dims).
433        btree2: bool,
434        /// Whether the chunk index is a Fixed Array (no unlimited dims).
435        fixed_array: bool,
436    },
437    /// A dataset opened by name in read mode.
438    Reader {
439        /// The link name of the dataset.
440        name: String,
441        /// Shape (current dimensions).
442        shape: Vec<usize>,
443        /// Size of one element in bytes.
444        element_size: usize,
445    },
446}
447
448// ---------------------------------------------------------------------------
449// H5Dataset
450// ---------------------------------------------------------------------------
451
452/// A handle to an HDF5 dataset, supporting typed read and write operations.
453///
454/// The dataset holds a shared reference to the file's I/O backend, so it
455/// remains valid even if the originating [`H5File`](crate::file::H5File) is
456/// moved or dropped (they share ownership via `Rc`).
457pub struct H5Dataset {
458    file_inner: SharedInner,
459    info: DatasetInfo,
460}
461
462impl H5Dataset {
463    /// Create a reader-mode dataset handle (called internally by `H5File::dataset`).
464    pub(crate) fn new_reader(
465        file_inner: SharedInner,
466        name: String,
467        shape: Vec<usize>,
468        element_size: usize,
469    ) -> Self {
470        Self {
471            file_inner,
472            info: DatasetInfo::Reader {
473                name,
474                shape,
475                element_size,
476            },
477        }
478    }
479
480    /// Create a writer-mode dataset handle for an already-created dataset
481    /// (called internally by [`H5File::dataset_writer`](crate::file::H5File::dataset_writer)).
482    ///
483    /// Reconstructs the same handle `new_dataset().create()` returns, so the
484    /// reopened dataset supports attribute writes and chunk appends.
485    pub(crate) fn new_writer(
486        file_inner: SharedInner,
487        index: usize,
488        shape: Vec<usize>,
489        element_size: usize,
490        chunked: bool,
491        btree2: bool,
492        fixed_array: bool,
493    ) -> Self {
494        Self {
495            file_inner,
496            info: DatasetInfo::Writer {
497                index,
498                shape,
499                element_size,
500                chunked,
501                btree2,
502                fixed_array,
503            },
504        }
505    }
506
507    /// Return the dataset dimensions.
508    pub fn shape(&self) -> Vec<usize> {
509        match &self.info {
510            DatasetInfo::Writer { shape, .. } => shape.clone(),
511            DatasetInfo::Reader { shape, .. } => shape.clone(),
512        }
513    }
514
515    /// Return the number of dimensions (rank) of the dataset.
516    pub fn ndims(&self) -> usize {
517        match &self.info {
518            DatasetInfo::Writer { shape, .. } => shape.len(),
519            DatasetInfo::Reader { shape, .. } => shape.len(),
520        }
521    }
522
523    /// Return the total number of elements in the dataset.
524    pub fn total_elements(&self) -> usize {
525        match &self.info {
526            DatasetInfo::Writer { shape, .. } => shape.iter().product(),
527            DatasetInfo::Reader { shape, .. } => shape.iter().product(),
528        }
529    }
530
531    /// Return the size of one element in bytes.
532    pub fn element_size(&self) -> usize {
533        match &self.info {
534            DatasetInfo::Writer { element_size, .. } => *element_size,
535            DatasetInfo::Reader { element_size, .. } => *element_size,
536        }
537    }
538
539    /// Return the element datatype as parsed from the file (read mode only).
540    ///
541    /// Unlike [`element_size`](Self::element_size), which reports only the
542    /// byte width, this exposes the full datatype: its class (integer vs
543    /// floating-point vs string vs compound …), signedness, byte order and
544    /// bit precision. Callers that must reconstruct the exact stored type —
545    /// for example to map it to a NumPy / Arrow dtype — should use this
546    /// instead of inferring a type from the byte width, which cannot
547    /// distinguish `u8` from `i8` (both 1 byte) or `i32` from `f32` (both 4
548    /// bytes).
549    ///
550    /// # Errors
551    ///
552    /// Returns an error if the file is in write mode, or if the dataset can
553    /// no longer be found in the reader's metadata.
554    ///
555    /// ```no_run
556    /// # use rust_hdf5::{H5File, DatatypeMessage};
557    /// let file = H5File::open("data.h5").unwrap();
558    /// let ds = file.dataset("image").unwrap();
559    /// match ds.datatype().unwrap() {
560    ///     DatatypeMessage::FixedPoint { size, signed, .. } => {
561    ///         println!("integer: {} bytes, signed={}", size, signed);
562    ///     }
563    ///     DatatypeMessage::FloatingPoint { size, .. } => {
564    ///         println!("float: {} bytes", size);
565    ///     }
566    ///     other => println!("other type: {other}"),
567    /// }
568    /// ```
569    pub fn datatype(&self) -> Result<DatatypeMessage> {
570        match &self.info {
571            DatasetInfo::Reader { name, .. } => {
572                let inner = borrow_inner(&self.file_inner);
573                match &*inner {
574                    H5FileInner::Reader(reader) => reader
575                        .dataset_info(name)
576                        .map(|info| info.datatype.clone())
577                        .ok_or_else(|| Hdf5Error::NotFound(name.clone())),
578                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
579                }
580            }
581            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
582                "datatype() is only available in read mode".into(),
583            )),
584        }
585    }
586
587    /// Return the chunk dimensions, if this is a chunked dataset.
588    pub fn chunk_dims(&self) -> Option<Vec<usize>> {
589        match &self.info {
590            DatasetInfo::Reader { name, .. } => {
591                let inner = borrow_inner(&self.file_inner);
592                if let H5FileInner::Reader(reader) = &*inner {
593                    if let Some(info) = reader.dataset_info(name) {
594                        use crate::format::messages::data_layout::DataLayoutMessage;
595                        let chunk_dims = match &info.layout {
596                            DataLayoutMessage::ChunkedV4 { chunk_dims, .. }
597                            | DataLayoutMessage::ChunkedV3 { chunk_dims, .. } => Some(chunk_dims),
598                            _ => None,
599                        };
600                        if let Some(chunk_dims) = chunk_dims {
601                            // Strip trailing element-size dimension
602                            return Some(
603                                chunk_dims[..chunk_dims.len() - 1]
604                                    .iter()
605                                    .map(|&d| d as usize)
606                                    .collect(),
607                            );
608                        }
609                    }
610                }
611                None
612            }
613            DatasetInfo::Writer { .. } => None,
614        }
615    }
616
617    /// Return whether this is a chunked dataset.
618    pub fn is_chunked(&self) -> bool {
619        match &self.info {
620            DatasetInfo::Writer { chunked, .. } => *chunked,
621            DatasetInfo::Reader { name, .. } => {
622                let inner = borrow_inner(&self.file_inner);
623                match &*inner {
624                    H5FileInner::Reader(reader) => {
625                        if let Some(info) = reader.dataset_info(name) {
626                            use crate::format::messages::data_layout::DataLayoutMessage;
627                            matches!(
628                                info.layout,
629                                DataLayoutMessage::ChunkedV4 { .. }
630                                    | DataLayoutMessage::ChunkedV3 { .. }
631                            )
632                        } else {
633                            false
634                        }
635                    }
636                    _ => false,
637                }
638            }
639        }
640    }
641
642    /// Return the names of all attributes on this dataset (read mode only).
643    pub fn attr_names(&self) -> Result<Vec<String>> {
644        match &self.info {
645            DatasetInfo::Reader { name, .. } => {
646                let inner = borrow_inner(&self.file_inner);
647                match &*inner {
648                    H5FileInner::Reader(reader) => Ok(reader.dataset_attr_names(name)?),
649                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
650                }
651            }
652            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
653                "attr_names not available in write mode".into(),
654            )),
655        }
656    }
657
658    /// Open an attribute by name (read mode only).
659    pub fn attr(&self, attr_name: &str) -> Result<crate::attribute::H5Attribute> {
660        match &self.info {
661            DatasetInfo::Reader { name, .. } => {
662                let inner = borrow_inner(&self.file_inner);
663                match &*inner {
664                    H5FileInner::Reader(reader) => {
665                        let attr_msg = reader.dataset_attr(name, attr_name)?.clone();
666                        Ok(crate::attribute::H5Attribute::new_reader(
667                            clone_inner(&self.file_inner),
668                            attr_msg,
669                        ))
670                    }
671                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
672                }
673            }
674            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
675                "attr() not available in write mode".into(),
676            )),
677        }
678    }
679
680    /// Start building a new attribute on this dataset.
681    ///
682    /// Returns a fluent builder. Call `.shape(())` for a scalar attribute
683    /// and `.create("name")` to finalize.
684    ///
685    /// # Example
686    ///
687    /// ```no_run
688    /// # use rust_hdf5::H5File;
689    /// # use rust_hdf5::types::VarLenUnicode;
690    /// let file = H5File::create("attr.h5").unwrap();
691    /// let ds = file.new_dataset::<f32>().shape(&[10]).create("data").unwrap();
692    /// let attr = ds.new_attr::<VarLenUnicode>().shape(()).create("units").unwrap();
693    /// attr.write_scalar(&VarLenUnicode("meters".to_string())).unwrap();
694    /// ```
695    pub fn new_attr<T: 'static>(&self) -> AttrBuilder<'_, T> {
696        let ds_index = match &self.info {
697            DatasetInfo::Writer { index, .. } => *index,
698            DatasetInfo::Reader { .. } => {
699                // Reader mode: we'll return a builder that will error on create.
700                // Using usize::MAX as sentinel.
701                usize::MAX
702            }
703        };
704        AttrBuilder::new(&self.file_inner, ds_index)
705    }
706
707    /// Write a typed slice to the dataset (contiguous datasets only).
708    ///
709    /// The slice length must match the total number of elements declared by
710    /// the dataset shape. The data is reinterpreted as raw bytes and written
711    /// to the file.
712    ///
713    /// # Errors
714    ///
715    /// Returns an error if:
716    /// - The file is in read mode.
717    /// - The data length does not match the declared shape.
718    pub fn write_raw<T: H5Type>(&self, data: &[T]) -> Result<()> {
719        match &self.info {
720            DatasetInfo::Writer {
721                index,
722                shape,
723                element_size,
724                chunked,
725                btree2,
726                fixed_array,
727            } => {
728                let total_elements: usize = shape.iter().product();
729                if data.len() != total_elements {
730                    return Err(Hdf5Error::InvalidState(format!(
731                        "data length {} does not match dataset size {}",
732                        data.len(),
733                        total_elements,
734                    )));
735                }
736
737                // Verify element size matches
738                if T::element_size() != *element_size {
739                    return Err(Hdf5Error::TypeMismatch(format!(
740                        "write type has element size {} but dataset expects {}",
741                        T::element_size(),
742                        element_size,
743                    )));
744                }
745
746                // Safety: T: Copy + 'static (numeric primitive) with well-defined
747                // byte representation. The resulting slice borrows `data` and
748                // lives only as long as this block.
749                let byte_len = data.len() * T::element_size();
750                let raw =
751                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
752
753                if *chunked {
754                    // A chunked dataset has no contiguous data block; scatter
755                    // the full row-major image into its chunk grid and write
756                    // each chunk through the dataset's filter pipeline.
757                    return self.write_full_image_chunked(
758                        *index,
759                        *btree2,
760                        *fixed_array,
761                        raw,
762                        *element_size,
763                    );
764                }
765
766                let inner = borrow_inner(&self.file_inner);
767                match &*inner {
768                    H5FileInner::Writer(writer) => {
769                        writer.write_dataset_raw(*index, raw)?;
770                        Ok(())
771                    }
772                    _ => Err(Hdf5Error::InvalidState(
773                        "file is no longer in write mode".into(),
774                    )),
775                }
776            }
777            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
778                "cannot write to a dataset opened in read mode".into(),
779            )),
780        }
781    }
782
783    /// Write the raw byte image of a contiguous dataset directly.
784    ///
785    /// Unlike [`write_raw`](Self::write_raw), this is not generic over an
786    /// `H5Type` carrier, so it works for element types that have no matching
787    /// Rust primitive — in particular a runtime
788    /// [`CompoundType`](crate::types::CompoundType) of arbitrary size set via
789    /// [`DatasetBuilder::datatype`]. `bytes.len()` must equal
790    /// `product(shape) * element_size`, where `element_size` is taken from the
791    /// dataset's on-disk datatype.
792    ///
793    /// ```no_run
794    /// # use rust_hdf5::H5File;
795    /// # use rust_hdf5::types::{CompoundType, H5Type};
796    /// let file = H5File::create("c.h5").unwrap();
797    /// let ct = CompoundType {
798    ///     members: vec![
799    ///         ("id".to_string(), i32::hdf5_type(), 0),
800    ///         ("val".to_string(), f64::hdf5_type(), 4),
801    ///     ],
802    ///     total_size: 12,
803    /// };
804    /// let ds = file
805    ///     .new_dataset::<u8>()
806    ///     .datatype(ct.to_datatype())
807    ///     .shape(&[2])
808    ///     .create("records")
809    ///     .unwrap();
810    /// let mut bytes = Vec::new();
811    /// bytes.extend_from_slice(&1i32.to_le_bytes());
812    /// bytes.extend_from_slice(&2.5f64.to_le_bytes());
813    /// bytes.extend_from_slice(&2i32.to_le_bytes());
814    /// bytes.extend_from_slice(&3.5f64.to_le_bytes());
815    /// ds.write_raw_bytes(&bytes).unwrap();
816    /// ```
817    pub fn write_raw_bytes(&self, bytes: &[u8]) -> Result<()> {
818        match &self.info {
819            DatasetInfo::Writer {
820                index,
821                shape,
822                element_size,
823                chunked,
824                btree2,
825                fixed_array,
826            } => {
827                let expected: usize = shape.iter().product::<usize>() * *element_size;
828                if bytes.len() != expected {
829                    return Err(Hdf5Error::InvalidState(format!(
830                        "raw byte length {} does not match dataset size {} \
831                         (product(shape) * element_size {})",
832                        bytes.len(),
833                        expected,
834                        element_size,
835                    )));
836                }
837                if *chunked {
838                    // Scatter the full row-major image into the chunk grid
839                    // (same path as write_raw, carrier-agnostic bytes).
840                    return self.write_full_image_chunked(
841                        *index,
842                        *btree2,
843                        *fixed_array,
844                        bytes,
845                        *element_size,
846                    );
847                }
848                let inner = borrow_inner(&self.file_inner);
849                match &*inner {
850                    H5FileInner::Writer(writer) => {
851                        writer.write_dataset_raw(*index, bytes)?;
852                        Ok(())
853                    }
854                    _ => Err(Hdf5Error::InvalidState(
855                        "file is no longer in write mode".into(),
856                    )),
857                }
858            }
859            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
860                "cannot write to a dataset opened in read mode".into(),
861            )),
862        }
863    }
864
865    /// Scatter a full row-major dataset image into its chunk grid, writing
866    /// every chunk through the dataset's filter pipeline.
867    ///
868    /// This is the chunked counterpart of a single contiguous `write_dataset_raw`
869    /// — it is how [`write_raw`](Self::write_raw) and
870    /// [`write_raw_bytes`](Self::write_raw_bytes) populate a chunked dataset
871    /// (including the single auto-chunk created when a filter is set without
872    /// explicit chunk dimensions). Edge chunks are zero-padded to the full
873    /// chunk footprint, exactly as libhdf5 stores them.
874    fn write_full_image_chunked(
875        &self,
876        index: usize,
877        btree2: bool,
878        fixed_array: bool,
879        bytes: &[u8],
880        element_size: usize,
881    ) -> Result<()> {
882        let inner = borrow_inner(&self.file_inner);
883        let writer = match &*inner {
884            H5FileInner::Writer(w) => w,
885            _ => {
886                return Err(Hdf5Error::InvalidState(
887                    "file is no longer in write mode".into(),
888                ))
889            }
890        };
891        let chunk_dims = writer
892            .dataset_chunk_dims(index)
893            .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
894            .to_vec();
895        let dims = writer.dataset_dims(index).to_vec();
896        let rank = dims.len();
897
898        // Chunk grid: number of chunks along each dimension (row-major).
899        let mut grid = vec![0u64; rank];
900        for d in 0..rank {
901            grid[d] = if chunk_dims[d] > 0 {
902                dims[d].div_ceil(chunk_dims[d])
903            } else {
904                0
905            };
906        }
907        let total_chunks: u64 = grid.iter().product();
908
909        // Decode a linear chunk index into row-major grid coordinates.
910        let coords_of = |linear: u64| -> Vec<u64> {
911            let mut rem = linear;
912            let mut coords = vec![0u64; rank];
913            for d in (0..rank).rev() {
914                coords[d] = rem % grid[d];
915                rem /= grid[d];
916            }
917            coords
918        };
919
920        if btree2 {
921            // B-tree v2 stores chunks verbatim (unfiltered only in this
922            // codebase), so there is no compression to parallelize; write one
923            // chunk at a time.
924            for linear in 0..total_chunks {
925                let coords = coords_of(linear);
926                let chunk_buf =
927                    Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
928                writer.write_chunk_btree_v2(index, &coords, &chunk_buf)?;
929            }
930        } else {
931            // Extensible array and fixed array both compress each chunk through
932            // the filter pipeline. Gather chunks and write them through the
933            // per-index batch path so the pipeline compresses them in parallel
934            // (with the `parallel` feature). A fixed-size window bounds peak
935            // memory instead of materializing every chunk at once; 256 keeps
936            // every rayon worker fed while capping the transient buffers to
937            // window * chunk bytes. The two indexes differ only in how a chunk
938            // is addressed: EA by its linear grid index, FA by grid coordinates.
939            const BATCH_WINDOW: u64 = 256;
940            let mut start = 0u64;
941            while start < total_chunks {
942                let end = (start + BATCH_WINDOW).min(total_chunks);
943                let items: Vec<(u64, Vec<u64>, Vec<u8>)> = (start..end)
944                    .map(|linear| {
945                        let coords = coords_of(linear);
946                        let buf =
947                            Self::gather_chunk(bytes, &dims, &chunk_dims, &coords, element_size);
948                        (linear, coords, buf)
949                    })
950                    .collect();
951                if fixed_array {
952                    let pairs: Vec<(&[u64], &[u8])> = items
953                        .iter()
954                        .map(|(_, c, d)| (c.as_slice(), d.as_slice()))
955                        .collect();
956                    writer.write_chunks_fixed_array_batch(index, &pairs)?;
957                } else {
958                    let pairs: Vec<(u64, &[u8])> =
959                        items.iter().map(|(l, _, d)| (*l, d.as_slice())).collect();
960                    writer.write_chunks_batch(index, &pairs)?;
961                }
962                start = end;
963            }
964        }
965        Ok(())
966    }
967
968    /// Gather one chunk's bytes from a row-major full-dataset image.
969    ///
970    /// `coords` are the chunk's grid coordinates. The returned buffer is
971    /// exactly `product(chunk_dims) * element_size` bytes, zero-padded where
972    /// the chunk extends past the dataset edge.
973    fn gather_chunk(
974        source: &[u8],
975        dims: &[u64],
976        chunk_dims: &[u64],
977        coords: &[u64],
978        element_size: usize,
979    ) -> Vec<u8> {
980        let rank = dims.len();
981        let chunk_elems: u64 = chunk_dims.iter().product();
982        let mut out = vec![0u8; chunk_elems as usize * element_size];
983        if rank == 0 {
984            // Scalar dataset: a single element, no chunking dimension.
985            if source.len() >= element_size {
986                out[..element_size].copy_from_slice(&source[..element_size]);
987            }
988            return out;
989        }
990
991        // Actual extent of this chunk along each dimension (edge chunks are
992        // smaller than the nominal chunk shape).
993        let mut extent = vec![0u64; rank];
994        for d in 0..rank {
995            let start = coords[d] * chunk_dims[d];
996            let end = ((coords[d] + 1) * chunk_dims[d]).min(dims[d]);
997            extent[d] = end.saturating_sub(start);
998        }
999        if extent.contains(&0) {
1000            return out; // nothing of the dataset falls in this chunk
1001        }
1002
1003        // Row-major strides (in elements) for the source (over `dims`) and the
1004        // destination chunk buffer (over `chunk_dims`).
1005        let mut src_stride = vec![1u64; rank];
1006        let mut dst_stride = vec![1u64; rank];
1007        for d in (0..rank - 1).rev() {
1008            src_stride[d] = src_stride[d + 1] * dims[d + 1];
1009            dst_stride[d] = dst_stride[d + 1] * chunk_dims[d + 1];
1010        }
1011
1012        // Copy one contiguous run along the last axis per outer multi-index.
1013        let last = rank - 1;
1014        let run = extent[last] as usize * element_size;
1015        let outer: u64 = extent[..last].iter().product::<u64>().max(1);
1016        let mut idx = vec![0u64; rank]; // local indices within the chunk extent
1017        for _ in 0..outer {
1018            let mut src_off = 0u64;
1019            let mut dst_off = 0u64;
1020            for d in 0..rank {
1021                let global = coords[d] * chunk_dims[d] + idx[d];
1022                src_off += global * src_stride[d];
1023                dst_off += idx[d] * dst_stride[d];
1024            }
1025            let s = src_off as usize * element_size;
1026            let dpos = dst_off as usize * element_size;
1027            out[dpos..dpos + run].copy_from_slice(&source[s..s + run]);
1028
1029            // Advance the multi-index over axes [0..last); the last axis is the
1030            // contiguous run handled above.
1031            let mut d = last;
1032            while d > 0 {
1033                d -= 1;
1034                idx[d] += 1;
1035                if idx[d] < extent[d] {
1036                    break;
1037                }
1038                idx[d] = 0;
1039            }
1040        }
1041        out
1042    }
1043
1044    /// Write a single chunk to a chunked dataset.
1045    ///
1046    /// `chunk_idx` is the linear chunk index (typically the frame number for
1047    /// streaming datasets). `data` is the raw byte data for one chunk.
1048    ///
1049    /// For datasets with two or more unlimited dimensions (v2 B-tree index),
1050    /// use [`write_chunk_at`](Self::write_chunk_at) instead.
1051    pub fn write_chunk(&self, chunk_idx: usize, data: &[u8]) -> Result<()> {
1052        match &self.info {
1053            DatasetInfo::Writer {
1054                index,
1055                chunked,
1056                btree2,
1057                fixed_array,
1058                ..
1059            } => {
1060                if !*chunked {
1061                    return Err(Hdf5Error::InvalidState(
1062                        "write_chunk is only for chunked datasets".into(),
1063                    ));
1064                }
1065                if *btree2 {
1066                    return Err(Hdf5Error::InvalidState(
1067                        "this dataset uses a v2 B-tree chunk index; use write_chunk_at \
1068                         with the chunk's grid coordinates"
1069                            .into(),
1070                    ));
1071                }
1072
1073                let inner = borrow_inner(&self.file_inner);
1074                match &*inner {
1075                    H5FileInner::Writer(writer) => {
1076                        if *fixed_array {
1077                            // Fixed-array dataset: convert the linear chunk
1078                            // index into row-major grid coordinates.
1079                            let chunk_dims = writer
1080                                .dataset_chunk_dims(*index)
1081                                .ok_or_else(|| {
1082                                    Hdf5Error::InvalidState("dataset has no chunk info".into())
1083                                })?
1084                                .to_vec();
1085                            let dims = writer.dataset_dims(*index).to_vec();
1086                            let mut grid = vec![0u64; dims.len()];
1087                            for d in 0..dims.len() {
1088                                grid[d] = if chunk_dims[d] > 0 {
1089                                    dims[d].div_ceil(chunk_dims[d])
1090                                } else {
1091                                    1
1092                                };
1093                            }
1094                            // A zero-extent dimension yields a grid of 0
1095                            // chunks — there is no chunk to write.
1096                            if grid.contains(&0) {
1097                                return Err(Hdf5Error::InvalidState(
1098                                    "dataset has a zero-extent dimension and no chunks".into(),
1099                                ));
1100                            }
1101                            let mut rem = chunk_idx as u64;
1102                            let mut coords = vec![0u64; dims.len()];
1103                            for d in (0..dims.len()).rev() {
1104                                coords[d] = rem % grid[d];
1105                                rem /= grid[d];
1106                            }
1107                            // A leftover means chunk_idx exceeded the grid.
1108                            if rem != 0 {
1109                                return Err(Hdf5Error::InvalidState(format!(
1110                                    "chunk index {chunk_idx} is out of range for this dataset"
1111                                )));
1112                            }
1113                            writer.write_chunk_fixed_array(*index, &coords, data)?;
1114                        } else {
1115                            writer.write_chunk(*index, chunk_idx as u64, data)?;
1116                        }
1117                        Ok(())
1118                    }
1119                    _ => Err(Hdf5Error::InvalidState(
1120                        "file is no longer in write mode".into(),
1121                    )),
1122                }
1123            }
1124            DatasetInfo::Reader { .. } => {
1125                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1126            }
1127        }
1128    }
1129
1130    /// Write an already-filtered (pre-compressed) chunk **verbatim**, recording
1131    /// the caller-supplied `filter_mask`. The bytes are stored as-is without
1132    /// running the dataset's filter pipeline — the HDF5 "direct chunk write"
1133    /// (`H5Dwrite_chunk`, formerly `H5DOwrite_chunk`) operation.
1134    ///
1135    /// `chunk_idx` is the linear chunk index (the frame number for streaming
1136    /// datasets), exactly as for [`write_chunk`](Self::write_chunk). `data` is
1137    /// the already-filtered bytes of one chunk — its length is the *stored*
1138    /// (compressed) size, not the uncompressed chunk size.
1139    ///
1140    /// `filter_mask` is a bitfield: bit *i* set means filter *i* of the
1141    /// dataset's pipeline was **not** applied to this chunk and must be skipped
1142    /// on read. Pass 0 when the full pipeline was already applied upstream (the
1143    /// common case: a codec plugin handed you compressed frames).
1144    ///
1145    /// The dataset must be chunked **and** filtered; an unfiltered chunk index
1146    /// has no slot to record a stored size or mask. For a v2-B-tree-indexed
1147    /// dataset (two or more unlimited dimensions) direct chunk writes are not
1148    /// supported.
1149    ///
1150    /// # Reading back
1151    ///
1152    /// Both this crate's reader and libhdf5/h5py honor the per-chunk
1153    /// `filter_mask`: a chunk written with any mask round-trips correctly, with
1154    /// the reader skipping exactly the filters the mask marks as not applied.
1155    pub fn write_chunk_raw(&self, chunk_idx: usize, data: &[u8], filter_mask: u32) -> Result<()> {
1156        match &self.info {
1157            DatasetInfo::Writer {
1158                index,
1159                chunked,
1160                btree2,
1161                fixed_array,
1162                ..
1163            } => {
1164                if !*chunked {
1165                    return Err(Hdf5Error::InvalidState(
1166                        "write_chunk_raw is only for chunked datasets".into(),
1167                    ));
1168                }
1169                if *btree2 {
1170                    return Err(Hdf5Error::InvalidState(
1171                        "direct chunk writes are not supported for v2-B-tree-indexed \
1172                         datasets (two or more unlimited dimensions)"
1173                            .into(),
1174                    ));
1175                }
1176
1177                let inner = borrow_inner(&self.file_inner);
1178                match &*inner {
1179                    H5FileInner::Writer(writer) => {
1180                        if *fixed_array {
1181                            // Fixed-array dataset: convert the linear chunk
1182                            // index into row-major grid coordinates.
1183                            let chunk_dims = writer
1184                                .dataset_chunk_dims(*index)
1185                                .ok_or_else(|| {
1186                                    Hdf5Error::InvalidState("dataset has no chunk info".into())
1187                                })?
1188                                .to_vec();
1189                            let dims = writer.dataset_dims(*index).to_vec();
1190                            let mut grid = vec![0u64; dims.len()];
1191                            for d in 0..dims.len() {
1192                                grid[d] = if chunk_dims[d] > 0 {
1193                                    dims[d].div_ceil(chunk_dims[d])
1194                                } else {
1195                                    1
1196                                };
1197                            }
1198                            // A zero-extent dimension yields a grid of 0
1199                            // chunks — there is no chunk to write.
1200                            if grid.contains(&0) {
1201                                return Err(Hdf5Error::InvalidState(
1202                                    "dataset has a zero-extent dimension and no chunks".into(),
1203                                ));
1204                            }
1205                            let mut rem = chunk_idx as u64;
1206                            let mut coords = vec![0u64; dims.len()];
1207                            for d in (0..dims.len()).rev() {
1208                                coords[d] = rem % grid[d];
1209                                rem /= grid[d];
1210                            }
1211                            // A leftover means chunk_idx exceeded the grid.
1212                            if rem != 0 {
1213                                return Err(Hdf5Error::InvalidState(format!(
1214                                    "chunk index {chunk_idx} is out of range for this dataset"
1215                                )));
1216                            }
1217                            writer.write_compressed_chunk_fixed_array(
1218                                *index,
1219                                &coords,
1220                                data,
1221                                filter_mask,
1222                            )?;
1223                        } else {
1224                            writer.write_compressed_chunk(
1225                                *index,
1226                                chunk_idx as u64,
1227                                data,
1228                                filter_mask,
1229                            )?;
1230                        }
1231                        Ok(())
1232                    }
1233                    _ => Err(Hdf5Error::InvalidState(
1234                        "file is no longer in write mode".into(),
1235                    )),
1236                }
1237            }
1238            DatasetInfo::Reader { .. } => {
1239                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1240            }
1241        }
1242    }
1243
1244    /// Write a single chunk to a v2-B-tree-indexed dataset, addressed by its
1245    /// chunk-grid coordinates (one per dimension).
1246    ///
1247    /// This is the entry point for datasets with two or more unlimited
1248    /// dimensions. The dataset's logical dimensions are extended to cover
1249    /// the written chunk. `data` is the raw bytes of one full chunk.
1250    ///
1251    /// ```no_run
1252    /// # use rust_hdf5::H5File;
1253    /// let file = H5File::create("bt2.h5").unwrap();
1254    /// let ds = file.new_dataset::<i32>()
1255    ///     .shape(&[0, 0])
1256    ///     .chunk(&[2, 2])
1257    ///     .max_shape(&[None, None])
1258    ///     .create("grid")
1259    ///     .unwrap();
1260    /// let chunk = [0i32, 1, 2, 3];
1261    /// let bytes: Vec<u8> = chunk.iter().flat_map(|v| v.to_le_bytes()).collect();
1262    /// ds.write_chunk_at(&[0, 0], &bytes).unwrap();
1263    /// ```
1264    pub fn write_chunk_at(&self, chunk_coords: &[usize], data: &[u8]) -> Result<()> {
1265        match &self.info {
1266            DatasetInfo::Writer {
1267                index,
1268                chunked,
1269                btree2,
1270                fixed_array,
1271                ..
1272            } => {
1273                if !*chunked {
1274                    return Err(Hdf5Error::InvalidState(
1275                        "write_chunk_at is only for chunked datasets".into(),
1276                    ));
1277                }
1278                let coords: Vec<u64> = chunk_coords.iter().map(|&c| c as u64).collect();
1279                let btree2 = *btree2;
1280                let fixed_array = *fixed_array;
1281                let inner = borrow_inner(&self.file_inner);
1282                let writer = match &*inner {
1283                    H5FileInner::Writer(w) => w,
1284                    _ => {
1285                        return Err(Hdf5Error::InvalidState(
1286                            "file is no longer in write mode".into(),
1287                        ))
1288                    }
1289                };
1290                let chunk_dims = writer
1291                    .dataset_chunk_dims(*index)
1292                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1293                    .to_vec();
1294                let dims = writer.dataset_dims(*index).to_vec();
1295                if coords.len() != dims.len() {
1296                    return Err(Hdf5Error::InvalidState(format!(
1297                        "chunk_coords has {} entries but the dataset has {} dimensions",
1298                        coords.len(),
1299                        dims.len()
1300                    )));
1301                }
1302                if chunk_dims.len() != dims.len() {
1303                    return Err(Hdf5Error::InvalidState(format!(
1304                        "dataset chunk shape has {} dimensions but the dataspace has {}",
1305                        chunk_dims.len(),
1306                        dims.len()
1307                    )));
1308                }
1309
1310                // Validate coordinates and compute the grown dimensions
1311                // up-front, before any chunk is written, so an overflowing
1312                // coordinate cannot leave an orphaned chunk in the file.
1313                let mut new_dims = dims.clone();
1314                for d in 0..dims.len() {
1315                    let needed = coords[d]
1316                        .checked_add(1)
1317                        .and_then(|c| c.checked_mul(chunk_dims[d]))
1318                        .ok_or_else(|| {
1319                            Hdf5Error::InvalidState(format!(
1320                                "chunk coordinate {} in dimension {} is too large",
1321                                coords[d], d
1322                            ))
1323                        })?;
1324                    if needed > new_dims[d] {
1325                        new_dims[d] = needed;
1326                    }
1327                }
1328
1329                if fixed_array {
1330                    // Fixed-array (fixed-shape) dataset: no dimension growth.
1331                    writer.write_chunk_fixed_array(*index, &coords, data)?;
1332                    return Ok(());
1333                }
1334
1335                if btree2 {
1336                    writer.write_chunk_btree_v2(*index, &coords, data)?;
1337                } else {
1338                    // Extensible array: linearize the chunk-grid coordinates
1339                    // (row-major) into the array's chunk index.
1340                    let mut linear = 0u64;
1341                    for d in 0..dims.len() {
1342                        let grid = if chunk_dims[d] > 0 {
1343                            dims[d].div_ceil(chunk_dims[d])
1344                        } else {
1345                            1
1346                        };
1347                        linear = linear
1348                            .checked_mul(grid)
1349                            .and_then(|l| l.checked_add(coords[d]))
1350                            .ok_or_else(|| {
1351                                Hdf5Error::InvalidState(
1352                                    "chunk coordinates overflow the array index".into(),
1353                                )
1354                            })?;
1355                    }
1356                    writer.write_chunk(*index, linear, data)?;
1357                }
1358
1359                if new_dims != dims {
1360                    writer.extend_dataset(*index, &new_dims)?;
1361                }
1362                Ok(())
1363            }
1364            DatasetInfo::Reader { .. } => {
1365                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1366            }
1367        }
1368    }
1369
1370    /// Write multiple chunks in a batch, optionally compressing in parallel.
1371    ///
1372    /// `chunks` is a slice of `(chunk_index, raw_data)` pairs. When a filter
1373    /// pipeline is configured and the `parallel` feature is enabled, all
1374    /// chunks are compressed concurrently via rayon.
1375    pub fn write_chunks_batch(&self, chunks: &[(usize, &[u8])]) -> Result<()> {
1376        match &self.info {
1377            DatasetInfo::Writer { index, chunked, .. } => {
1378                if !*chunked {
1379                    return Err(Hdf5Error::InvalidState(
1380                        "write_chunks_batch is only for chunked datasets".into(),
1381                    ));
1382                }
1383                let pairs: Vec<(u64, &[u8])> = chunks
1384                    .iter()
1385                    .map(|(idx, data)| (*idx as u64, *data))
1386                    .collect();
1387                let inner = borrow_inner(&self.file_inner);
1388                match &*inner {
1389                    H5FileInner::Writer(writer) => {
1390                        writer.write_chunks_batch(*index, &pairs)?;
1391                        Ok(())
1392                    }
1393                    _ => Err(Hdf5Error::InvalidState(
1394                        "file is no longer in write mode".into(),
1395                    )),
1396                }
1397            }
1398            DatasetInfo::Reader { .. } => {
1399                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1400            }
1401        }
1402    }
1403
1404    /// Append data along the first dimension of a chunked dataset.
1405    ///
1406    /// `data` must contain a whole number of "frames" — slices along
1407    /// dimension 0. For example, if the dataset has shape `[N, H, W]`
1408    /// and `chunk_dims = [1, H, W]`, then `data.len()` must be a
1409    /// multiple of `H * W`.
1410    ///
1411    /// This method writes the necessary chunks and extends the dataset
1412    /// shape automatically.
1413    ///
1414    /// ```no_run
1415    /// # use rust_hdf5::H5File;
1416    /// let file = H5File::create("append.h5").unwrap();
1417    /// let ds = file.new_dataset::<f64>()
1418    ///     .shape(&[0, 3])
1419    ///     .chunk(&[1, 3])
1420    ///     .max_shape(&[None, Some(3)])
1421    ///     .create("data")
1422    ///     .unwrap();
1423    /// ds.append(&[1.0, 2.0, 3.0]).unwrap();       // shape becomes [1, 3]
1424    /// ds.append(&[4.0, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap(); // shape becomes [3, 3]
1425    /// ```
1426    pub fn append<T: H5Type>(&self, data: &[T]) -> Result<()> {
1427        match &self.info {
1428            DatasetInfo::Writer {
1429                index,
1430                element_size,
1431                chunked,
1432                ..
1433            } => {
1434                if !*chunked {
1435                    return Err(Hdf5Error::InvalidState(
1436                        "append is only for chunked datasets".into(),
1437                    ));
1438                }
1439                if T::element_size() != *element_size {
1440                    return Err(Hdf5Error::TypeMismatch(format!(
1441                        "append type has element size {} but dataset expects {}",
1442                        T::element_size(),
1443                        element_size,
1444                    )));
1445                }
1446
1447                let ds_index = *index;
1448                let es = *element_size;
1449
1450                let inner = borrow_inner(&self.file_inner);
1451                let writer = match &*inner {
1452                    H5FileInner::Writer(w) => w,
1453                    _ => {
1454                        return Err(Hdf5Error::InvalidState(
1455                            "file is no longer in write mode".into(),
1456                        ))
1457                    }
1458                };
1459
1460                let chunk_dims = writer
1461                    .dataset_chunk_dims(ds_index)
1462                    .ok_or_else(|| Hdf5Error::InvalidState("dataset has no chunk info".into()))?
1463                    .to_vec();
1464                let dims = writer.dataset_dims(ds_index).to_vec();
1465
1466                // Frame size = product of dims[1..]
1467                let frame_elems: usize = if dims.len() > 1 {
1468                    dims[1..].iter().map(|&d| d as usize).product()
1469                } else {
1470                    1
1471                };
1472
1473                if frame_elems == 0 {
1474                    return Err(Hdf5Error::InvalidState(
1475                        "cannot append to dataset with zero-size trailing dimensions".into(),
1476                    ));
1477                }
1478
1479                if !data.len().is_multiple_of(frame_elems) {
1480                    return Err(Hdf5Error::InvalidState(format!(
1481                        "data length {} is not a multiple of frame size {}",
1482                        data.len(),
1483                        frame_elems,
1484                    )));
1485                }
1486
1487                let n_new_frames = data.len() / frame_elems;
1488                let current_dim0 = dims[0] as usize;
1489
1490                // Chunk size along first dimension
1491                let chunk_dim0 = chunk_dims[0] as usize;
1492                let frame_bytes = frame_elems * es;
1493
1494                let raw = unsafe {
1495                    std::slice::from_raw_parts(data.as_ptr() as *const u8, data.len() * es)
1496                };
1497
1498                // Merge buffered data with new data. Scope the slot guard: the
1499                // loop below calls `write_chunk`, which re-locks the same slot.
1500                let (buffered_frames, mut combined) = {
1501                    let ds = writer.ds(ds_index);
1502                    let mut m = ds.lock();
1503                    let buffered_frames = m.append_buffered_frames as usize;
1504                    let combined = std::mem::take(&mut m.append_buffer);
1505                    m.append_buffered_frames = 0;
1506                    (buffered_frames, combined)
1507                };
1508                combined.extend_from_slice(raw);
1509
1510                let total_frames = buffered_frames + n_new_frames;
1511                let total_bytes = combined.len();
1512
1513                // Base chunk index: account for buffered frames
1514                let base_dim0 = current_dim0 - buffered_frames;
1515                let mut byte_pos = 0usize;
1516                let mut frame_pos = 0usize;
1517
1518                while frame_pos < total_frames {
1519                    let abs_frame = base_dim0 + frame_pos;
1520                    let chunk_idx = abs_frame / chunk_dim0;
1521                    let remaining_frames = total_frames - frame_pos;
1522                    let frames_to_fill = chunk_dim0 - (abs_frame % chunk_dim0);
1523
1524                    if remaining_frames >= frames_to_fill {
1525                        // Full chunk — write
1526                        let end = byte_pos + frames_to_fill * frame_bytes;
1527                        if frames_to_fill == chunk_dim0 {
1528                            writer.write_chunk(
1529                                ds_index,
1530                                chunk_idx as u64,
1531                                &combined[byte_pos..end],
1532                            )?;
1533                        } else {
1534                            // Partial-chunk write: this branch only runs with
1535                            // offset_in_chunk > 0, meaning the chunk already
1536                            // holds earlier frames on disk. Read-modify-write
1537                            // so those frames survive — a fresh fill buffer
1538                            // would erase them.
1539                            let offset_in_chunk = (abs_frame % chunk_dim0) * frame_bytes;
1540                            let mut chunk_buf =
1541                                match writer.read_chunk_if_present(ds_index, chunk_idx as u64)? {
1542                                    Some(existing) => existing,
1543                                    None => {
1544                                        return Err(Hdf5Error::InvalidState(format!(
1545                                            "cannot append into partially-written chunk {}: \
1546                                         its existing content was not found in the chunk \
1547                                         index (the file may be inconsistent)",
1548                                            chunk_idx
1549                                        )));
1550                                    }
1551                                };
1552                            chunk_buf
1553                                [offset_in_chunk..offset_in_chunk + frames_to_fill * frame_bytes]
1554                                .copy_from_slice(&combined[byte_pos..end]);
1555                            writer.write_chunk(ds_index, chunk_idx as u64, &chunk_buf)?;
1556                        }
1557                        byte_pos = end;
1558                        frame_pos += frames_to_fill;
1559                    } else {
1560                        // Partial chunk — buffer for next append
1561                        let ds = writer.ds(ds_index);
1562                        let mut m = ds.lock();
1563                        m.append_buffer = combined[byte_pos..total_bytes].to_vec();
1564                        m.append_buffered_frames = remaining_frames as u64;
1565                        frame_pos = total_frames;
1566                    }
1567                }
1568
1569                // Extend dims to include all frames (buffered + new)
1570                let logical_dim0 = base_dim0 + total_frames;
1571                let mut new_dims: Vec<u64> = dims;
1572                new_dims[0] = logical_dim0 as u64;
1573                writer.extend_dataset(ds_index, &new_dims)?;
1574
1575                Ok(())
1576            }
1577            DatasetInfo::Reader { .. } => {
1578                Err(Hdf5Error::InvalidState("cannot append in read mode".into()))
1579            }
1580        }
1581    }
1582
1583    /// Extend the dimensions of a chunked dataset.
1584    pub fn extend(&self, new_dims: &[usize]) -> Result<()> {
1585        match &self.info {
1586            DatasetInfo::Writer { index, chunked, .. } => {
1587                if !*chunked {
1588                    return Err(Hdf5Error::InvalidState(
1589                        "extend is only for chunked datasets".into(),
1590                    ));
1591                }
1592
1593                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1594                let inner = borrow_inner(&self.file_inner);
1595                match &*inner {
1596                    H5FileInner::Writer(writer) => {
1597                        writer.extend_dataset(*index, &dims_u64)?;
1598                        Ok(())
1599                    }
1600                    _ => Err(Hdf5Error::InvalidState(
1601                        "file is no longer in write mode".into(),
1602                    )),
1603                }
1604            }
1605            DatasetInfo::Reader { .. } => {
1606                Err(Hdf5Error::InvalidState("cannot extend in read mode".into()))
1607            }
1608        }
1609    }
1610
1611    /// Set the logical extent of a chunked dataset, growing **or
1612    /// shrinking** any dimension.
1613    ///
1614    /// Unlike [`extend`](Self::extend), which only grows, this can reduce a
1615    /// dimension — for example to correct an over-extended frame count
1616    /// after writing a partial multi-frame chunk. Shrinking changes the
1617    /// logical dataspace only: data in chunks beyond the new extent stays
1618    /// in the file but is no longer visible on read, exactly as libhdf5's
1619    /// `H5Dset_extent` behaves. The new extent must not exceed the
1620    /// dataset's maximum dimensions.
1621    pub fn set_extent(&self, new_dims: &[usize]) -> Result<()> {
1622        match &self.info {
1623            DatasetInfo::Writer { index, .. } => {
1624                let dims_u64: Vec<u64> = new_dims.iter().map(|&d| d as u64).collect();
1625                let inner = borrow_inner(&self.file_inner);
1626                match &*inner {
1627                    H5FileInner::Writer(writer) => {
1628                        writer.set_dataset_extent(*index, &dims_u64)?;
1629                        Ok(())
1630                    }
1631                    _ => Err(Hdf5Error::InvalidState(
1632                        "file is no longer in write mode".into(),
1633                    )),
1634                }
1635            }
1636            DatasetInfo::Reader { .. } => Err(Hdf5Error::InvalidState(
1637                "cannot set extent in read mode".into(),
1638            )),
1639        }
1640    }
1641
1642    /// Flush a chunked dataset's index structures to disk.
1643    pub fn flush(&self) -> Result<()> {
1644        match &self.info {
1645            DatasetInfo::Writer { index, .. } => {
1646                let inner = borrow_inner(&self.file_inner);
1647                match &*inner {
1648                    H5FileInner::Writer(writer) => {
1649                        writer.flush_dataset(*index)?;
1650                        Ok(())
1651                    }
1652                    _ => Ok(()),
1653                }
1654            }
1655            DatasetInfo::Reader { .. } => Ok(()),
1656        }
1657    }
1658
1659    /// Read a slice (hyperslab) of the dataset as a typed vector.
1660    ///
1661    /// `starts` and `counts` define the N-dimensional selection:
1662    /// `starts[d]` = first index along dim d, `counts[d]` = how many elements.
1663    pub fn read_slice<T: H5Type>(&self, starts: &[usize], counts: &[usize]) -> Result<Vec<T>> {
1664        match &self.info {
1665            DatasetInfo::Reader {
1666                name, element_size, ..
1667            } => {
1668                if T::element_size() != *element_size {
1669                    return Err(Hdf5Error::TypeMismatch(format!(
1670                        "read type has element size {} but dataset has element size {}",
1671                        T::element_size(),
1672                        element_size,
1673                    )));
1674                }
1675                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1676                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1677
1678                let raw = {
1679                    let mut inner = borrow_inner_mut(&self.file_inner);
1680                    match &mut *inner {
1681                        H5FileInner::Reader(reader) => {
1682                            reader.read_slice(name, &starts_u64, &counts_u64)?
1683                        }
1684                        _ => {
1685                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()))
1686                        }
1687                    }
1688                };
1689
1690                if raw.len() % T::element_size() != 0 {
1691                    return Err(Hdf5Error::TypeMismatch(format!(
1692                        "raw data size {} is not a multiple of element size {}",
1693                        raw.len(),
1694                        T::element_size(),
1695                    )));
1696                }
1697
1698                let count = raw.len() / T::element_size();
1699                let mut result = Vec::<T>::with_capacity(count);
1700                unsafe {
1701                    std::ptr::copy_nonoverlapping(
1702                        raw.as_ptr(),
1703                        result.as_mut_ptr() as *mut u8,
1704                        raw.len(),
1705                    );
1706                    result.set_len(count);
1707                }
1708                Ok(result)
1709            }
1710            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1711                "cannot read_slice from a dataset in write mode".into(),
1712            )),
1713        }
1714    }
1715
1716    /// Write a typed slice to a sub-region of a contiguous dataset.
1717    ///
1718    /// `starts` and `counts` define the N-dimensional selection.
1719    pub fn write_slice<T: H5Type>(
1720        &self,
1721        starts: &[usize],
1722        counts: &[usize],
1723        data: &[T],
1724    ) -> Result<()> {
1725        match &self.info {
1726            DatasetInfo::Writer {
1727                index,
1728                element_size,
1729                chunked,
1730                ..
1731            } => {
1732                if *chunked {
1733                    return Err(Hdf5Error::InvalidState(
1734                        "write_slice is only for contiguous datasets".into(),
1735                    ));
1736                }
1737                if T::element_size() != *element_size {
1738                    return Err(Hdf5Error::TypeMismatch(format!(
1739                        "write type has element size {} but dataset expects {}",
1740                        T::element_size(),
1741                        element_size,
1742                    )));
1743                }
1744
1745                let expected: usize = counts.iter().product();
1746                if data.len() != expected {
1747                    return Err(Hdf5Error::InvalidState(format!(
1748                        "data length {} does not match slice size {}",
1749                        data.len(),
1750                        expected,
1751                    )));
1752                }
1753
1754                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1755                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1756
1757                let byte_len = data.len() * T::element_size();
1758                let raw =
1759                    unsafe { std::slice::from_raw_parts(data.as_ptr() as *const u8, byte_len) };
1760
1761                let inner = borrow_inner(&self.file_inner);
1762                match &*inner {
1763                    H5FileInner::Writer(writer) => {
1764                        writer.write_slice(*index, &starts_u64, &counts_u64, raw)?;
1765                        Ok(())
1766                    }
1767                    _ => Err(Hdf5Error::InvalidState(
1768                        "file is no longer in write mode".into(),
1769                    )),
1770                }
1771            }
1772            DatasetInfo::Reader { .. } => {
1773                Err(Hdf5Error::InvalidState("cannot write in read mode".into()))
1774            }
1775        }
1776    }
1777
1778    /// Read variable-length strings from a dataset.
1779    ///
1780    /// This handles h5py-style vlen string datasets that store strings
1781    /// as global heap references. Returns one String per element.
1782    pub fn read_vlen_strings(&self) -> Result<Vec<String>> {
1783        match &self.info {
1784            DatasetInfo::Reader { name, .. } => {
1785                let mut inner = borrow_inner_mut(&self.file_inner);
1786                match &mut *inner {
1787                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_strings(name)?),
1788                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1789                }
1790            }
1791            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1792                "cannot read vlen strings from a dataset in write mode".into(),
1793            )),
1794        }
1795    }
1796
1797    /// Read variable-length byte arrays from a dataset.
1798    ///
1799    /// This handles vlen byte-array datasets (a vlen sequence of `u8`, e.g.
1800    /// those written by [`write_vlen_bytes`](crate::H5File::write_vlen_bytes))
1801    /// that store each element as a global heap reference. Returns one
1802    /// `Vec<u8>` per element.
1803    pub fn read_vlen_bytes(&self) -> Result<Vec<Vec<u8>>> {
1804        match &self.info {
1805            DatasetInfo::Reader { name, .. } => {
1806                let mut inner = borrow_inner_mut(&self.file_inner);
1807                match &mut *inner {
1808                    H5FileInner::Reader(reader) => Ok(reader.read_vlen_bytes(name)?),
1809                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1810                }
1811            }
1812            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1813                "cannot read vlen bytes from a dataset in write mode".into(),
1814            )),
1815        }
1816    }
1817
1818    /// Read the entire dataset as a typed vector.
1819    ///
1820    /// The raw bytes are read from the file and reinterpreted as `T`. The
1821    /// caller must ensure that `T` matches the datatype used when the dataset
1822    /// was written.
1823    ///
1824    /// # Errors
1825    ///
1826    /// Returns an error if:
1827    /// - The file is in write mode.
1828    /// - The raw data size is not a multiple of `T::element_size()`.
1829    pub fn read_raw<T: H5Type>(&self) -> Result<Vec<T>> {
1830        match &self.info {
1831            DatasetInfo::Reader {
1832                name, element_size, ..
1833            } => {
1834                if T::element_size() != *element_size {
1835                    return Err(Hdf5Error::TypeMismatch(format!(
1836                        "read type has element size {} but dataset has element size {}",
1837                        T::element_size(),
1838                        element_size,
1839                    )));
1840                }
1841
1842                let raw = {
1843                    let mut inner = borrow_inner_mut(&self.file_inner);
1844                    match &mut *inner {
1845                        H5FileInner::Reader(reader) => reader.read_dataset_raw(name)?,
1846                        _ => {
1847                            return Err(Hdf5Error::InvalidState("file is not in read mode".into()));
1848                        }
1849                    }
1850                };
1851
1852                if raw.len() % T::element_size() != 0 {
1853                    return Err(Hdf5Error::TypeMismatch(format!(
1854                        "raw data size {} is not a multiple of element size {}",
1855                        raw.len(),
1856                        T::element_size(),
1857                    )));
1858                }
1859
1860                let count = raw.len() / T::element_size();
1861                let mut result = Vec::<T>::with_capacity(count);
1862
1863                // Safety: T is Copy + 'static (required by H5Type). We verified
1864                // the byte count matches count * size_of::<T>() above.
1865                // copy_nonoverlapping fills the memory with valid bit patterns
1866                // for all H5Type implementors (numeric primitives).
1867                // We call set_len AFTER the copy so that if an unexpected panic
1868                // occurs, uninitialized memory is never exposed.
1869                unsafe {
1870                    std::ptr::copy_nonoverlapping(
1871                        raw.as_ptr(),
1872                        result.as_mut_ptr() as *mut u8,
1873                        raw.len(),
1874                    );
1875                    result.set_len(count);
1876                }
1877
1878                Ok(result)
1879            }
1880            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1881                "cannot read from a dataset in write mode".into(),
1882            )),
1883        }
1884    }
1885
1886    /// Read the raw byte image of a dataset without an `H5Type` carrier.
1887    ///
1888    /// The counterpart to [`write_raw_bytes`](Self::write_raw_bytes): returns
1889    /// the element bytes verbatim regardless of the on-disk element type, so a
1890    /// runtime [`CompoundType`](crate::types::CompoundType) whose records have
1891    /// no matching Rust primitive can be read back and decoded by the caller.
1892    pub fn read_raw_bytes(&self) -> Result<Vec<u8>> {
1893        match &self.info {
1894            DatasetInfo::Reader { name, .. } => {
1895                let mut inner = borrow_inner_mut(&self.file_inner);
1896                match &mut *inner {
1897                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw(name)?),
1898                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1899                }
1900            }
1901            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1902                "cannot read from a dataset in write mode".into(),
1903            )),
1904        }
1905    }
1906
1907    /// Read the whole dataset into a caller-provided buffer, with no allocation.
1908    ///
1909    /// `out` must have exactly `product(dims)` elements (the dataset's element
1910    /// count) and `T::element_size()` must match the dataset's on-disk element
1911    /// size, otherwise an error is returned and `out` is left unspecified. The
1912    /// zero-copy counterpart of [`read_raw`](Self::read_raw): the bytes are read
1913    /// straight into `out` rather than into a fresh `Vec`, so a pinned /
1914    /// page-locked host buffer can be filled in one pass and DMA'd to a GPU
1915    /// without the extra staging copy a `read_raw` + copy-into-pinned would
1916    /// incur. Works for every layout (contiguous, compact, and chunked under
1917    /// any index); for chunked data each decoded chunk is scattered directly
1918    /// into `out`.
1919    ///
1920    /// ```no_run
1921    /// # use rust_hdf5::H5File;
1922    /// let file = H5File::open("data.h5").unwrap();
1923    /// let ds = file.dataset("frames").unwrap();
1924    /// let n: usize = ds.shape().iter().product();
1925    /// let mut buf = vec![0u16; n];           // or a pinned host allocation
1926    /// ds.read_raw_into(&mut buf).unwrap();
1927    /// ```
1928    pub fn read_raw_into<T: H5Type>(&self, out: &mut [T]) -> Result<()> {
1929        match &self.info {
1930            DatasetInfo::Reader {
1931                name, element_size, ..
1932            } => {
1933                if T::element_size() != *element_size {
1934                    return Err(Hdf5Error::TypeMismatch(format!(
1935                        "read type has element size {} but dataset has element size {}",
1936                        T::element_size(),
1937                        element_size,
1938                    )));
1939                }
1940                // Safety: `T: H5Type` is a `Copy` POD numeric with a defined
1941                // byte representation; every bit pattern the read writes is a
1942                // valid `T`. The byte view borrows `out` exclusively for this
1943                // call, and `out.len() * element_size` cannot overflow because
1944                // it is the byte length of an existing slice (<= isize::MAX).
1945                let byte_len = out.len() * T::element_size();
1946                let bytes = unsafe {
1947                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
1948                };
1949                let mut inner = borrow_inner_mut(&self.file_inner);
1950                match &mut *inner {
1951                    H5FileInner::Reader(reader) => Ok(reader.read_dataset_raw_into(name, bytes)?),
1952                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
1953                }
1954            }
1955            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
1956                "cannot read from a dataset in write mode".into(),
1957            )),
1958        }
1959    }
1960
1961    /// Read a hyperslab into a caller-provided buffer, with no allocation.
1962    ///
1963    /// `out` must have exactly `product(counts)` elements and
1964    /// `T::element_size()` must match the dataset's element size. The zero-copy
1965    /// counterpart of [`read_slice`](Self::read_slice) and the slice analogue of
1966    /// [`read_raw_into`](Self::read_raw_into): only chunks overlapping the
1967    /// selection are read, and the selected bytes land directly in `out` — the
1968    /// entry point for reading one frame / block straight into a pinned host
1969    /// buffer for an H2D transfer.
1970    ///
1971    /// ```no_run
1972    /// # use rust_hdf5::H5File;
1973    /// let file = H5File::open("vol.h5").unwrap();
1974    /// let ds = file.dataset("vol").unwrap();   // shape [nz, ny, nx]
1975    /// let (ny, nx) = (ds.shape()[1], ds.shape()[2]);
1976    /// let mut frame = vec![0f32; ny * nx];     // or a pinned host allocation
1977    /// ds.read_slice_into(&mut frame, &[5, 0, 0], &[1, ny, nx]).unwrap();
1978    /// ```
1979    pub fn read_slice_into<T: H5Type>(
1980        &self,
1981        out: &mut [T],
1982        starts: &[usize],
1983        counts: &[usize],
1984    ) -> Result<()> {
1985        match &self.info {
1986            DatasetInfo::Reader {
1987                name, element_size, ..
1988            } => {
1989                if T::element_size() != *element_size {
1990                    return Err(Hdf5Error::TypeMismatch(format!(
1991                        "read type has element size {} but dataset has element size {}",
1992                        T::element_size(),
1993                        element_size,
1994                    )));
1995                }
1996                let starts_u64: Vec<u64> = starts.iter().map(|&s| s as u64).collect();
1997                let counts_u64: Vec<u64> = counts.iter().map(|&c| c as u64).collect();
1998                // Safety: see `read_raw_into` — `T: H5Type` POD, exclusive
1999                // borrow of `out`, byte length within bounds.
2000                let byte_len = out.len() * T::element_size();
2001                let bytes = unsafe {
2002                    std::slice::from_raw_parts_mut(out.as_mut_ptr() as *mut u8, byte_len)
2003                };
2004                let mut inner = borrow_inner_mut(&self.file_inner);
2005                match &mut *inner {
2006                    H5FileInner::Reader(reader) => {
2007                        Ok(reader.read_slice_into(name, &starts_u64, &counts_u64, bytes)?)
2008                    }
2009                    _ => Err(Hdf5Error::InvalidState("file is not in read mode".into())),
2010                }
2011            }
2012            DatasetInfo::Writer { .. } => Err(Hdf5Error::InvalidState(
2013                "cannot read from a dataset in write mode".into(),
2014            )),
2015        }
2016    }
2017}
2018
2019#[cfg(test)]
2020mod tests {
2021    use crate::H5File;
2022    use std::path::PathBuf;
2023
2024    fn temp_path(name: &str) -> PathBuf {
2025        // Include PID + a per-call atomic counter so that concurrent
2026        // cargo invocations and any kernel-level "lock not yet
2027        // released" races between sequential opens cannot collide.
2028        use std::sync::atomic::{AtomicU64, Ordering};
2029        static COUNTER: AtomicU64 = AtomicU64::new(0);
2030        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
2031        std::env::temp_dir().join(format!(
2032            "hdf5_dataset_test_{}_{}_{}.h5",
2033            name,
2034            std::process::id(),
2035            n
2036        ))
2037    }
2038
2039    #[test]
2040    fn runtime_compound_via_datatype_override_and_raw_bytes() {
2041        use crate::format::messages::datatype::DatatypeMessage;
2042        use crate::types::{CompoundType, H5Type};
2043
2044        let path = temp_path("compound_raw");
2045        // A 12-byte packed compound with NO matching Rust primitive carrier,
2046        // so it can only be written through the datatype() override +
2047        // write_raw_bytes path (the runtime-CompoundType use case).
2048        let ct = CompoundType {
2049            members: vec![
2050                ("id".to_string(), i32::hdf5_type(), 0),
2051                ("val".to_string(), f64::hdf5_type(), 4),
2052            ],
2053            total_size: 12,
2054        };
2055        let recs: [(i32, f64); 3] = [(1, 2.5), (2, 3.5), (3, -4.0)];
2056        let mut bytes = Vec::new();
2057        for (id, val) in recs {
2058            bytes.extend_from_slice(&id.to_le_bytes());
2059            bytes.extend_from_slice(&val.to_le_bytes());
2060        }
2061
2062        {
2063            let file = H5File::create(&path).unwrap();
2064            let ds = file
2065                .new_dataset::<u8>()
2066                .datatype(ct.to_datatype())
2067                .shape([recs.len()])
2068                .create("records")
2069                .unwrap();
2070            ds.write_raw_bytes(&bytes).unwrap();
2071            file.close().unwrap();
2072        }
2073        {
2074            let file = H5File::open(&path).unwrap();
2075            let ds = file.dataset("records").unwrap();
2076            // The on-disk element type is the compound we specified (size 12),
2077            // not the u8 carrier.
2078            match ds.datatype().unwrap() {
2079                DatatypeMessage::Compound { size, members } => {
2080                    assert_eq!(size, 12);
2081                    assert_eq!(members.len(), 2);
2082                    assert_eq!(members[0].name, "id");
2083                    assert_eq!(members[0].offset, 0);
2084                    assert_eq!(members[1].name, "val");
2085                    assert_eq!(members[1].offset, 4);
2086                }
2087                other => panic!("expected compound datatype, got {other:?}"),
2088            }
2089            assert_eq!(ds.read_raw_bytes().unwrap(), bytes);
2090        }
2091        std::fs::remove_file(&path).ok();
2092    }
2093
2094    #[test]
2095    fn builder_requires_shape() {
2096        let path = temp_path("no_shape");
2097        let file = H5File::create(&path).unwrap();
2098        let result = file.new_dataset::<u8>().create("data");
2099        assert!(result.is_err());
2100        std::fs::remove_file(&path).ok();
2101    }
2102
2103    #[test]
2104    fn write_raw_size_mismatch() {
2105        let path = temp_path("size_mismatch");
2106        let file = H5File::create(&path).unwrap();
2107        let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
2108        // Provide 3 elements instead of 4
2109        let result = ds.write_raw(&[1u8, 2, 3]);
2110        assert!(result.is_err());
2111        std::fs::remove_file(&path).ok();
2112    }
2113
2114    // A: a filter set without explicit chunk dimensions must auto-chunk (whole
2115    // dataset = one chunk) rather than silently drop the filter on the
2116    // contiguous path. write_raw then populates that single chunk.
2117    #[cfg(feature = "deflate")]
2118    #[test]
2119    fn filter_without_chunk_autochunks_and_roundtrips() {
2120        let path = temp_path("autochunk_filter");
2121        let data: Vec<i32> = (0..8).collect();
2122        {
2123            let file = H5File::create(&path).unwrap();
2124            let ds = file
2125                .new_dataset::<i32>()
2126                .deflate(6)
2127                .shape([8])
2128                .create("seq")
2129                .unwrap();
2130            ds.write_raw(&data).unwrap();
2131            file.close().unwrap();
2132        }
2133        {
2134            let file = H5File::open(&path).unwrap();
2135            let ds = file.dataset("seq").unwrap();
2136            // The filter forced chunked storage: a single whole-dataset chunk.
2137            assert!(
2138                ds.is_chunked(),
2139                "auto-chunk did not produce chunked storage"
2140            );
2141            assert_eq!(ds.chunk_dims(), Some(vec![8]));
2142            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2143        }
2144        std::fs::remove_file(&path).ok();
2145    }
2146
2147    // B: write_raw on an explicitly chunked + compressed dataset scatters the
2148    // full row-major image across a multi-chunk grid, including edge chunks
2149    // (7/3 -> 3,3,1 along dim0; 5/2 -> 2,2,1 along dim1).
2150    #[cfg(feature = "deflate")]
2151    #[test]
2152    fn write_raw_multichunk_edge_roundtrips() {
2153        let path = temp_path("multichunk_edge");
2154        let data: Vec<i32> = (0..35).collect(); // 7 x 5 row-major
2155        {
2156            let file = H5File::create(&path).unwrap();
2157            let ds = file
2158                .new_dataset::<i32>()
2159                .shape([7, 5])
2160                .chunk(&[3, 2])
2161                .deflate(4)
2162                .create("grid")
2163                .unwrap();
2164            ds.write_raw(&data).unwrap();
2165            file.close().unwrap();
2166        }
2167        {
2168            let file = H5File::open(&path).unwrap();
2169            let ds = file.dataset("grid").unwrap();
2170            assert_eq!(ds.shape(), vec![7, 5]);
2171            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2172            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2173        }
2174        std::fs::remove_file(&path).ok();
2175    }
2176
2177    // B: write_raw on an unfiltered chunked dataset (previously rejected with
2178    // "use write_chunk for chunked datasets") now gathers and round-trips.
2179    #[test]
2180    fn write_raw_unfiltered_chunked_roundtrips() {
2181        let path = temp_path("chunked_unfiltered");
2182        let data: Vec<f64> = (0..12).map(|i| i as f64 * 1.5).collect(); // 4 x 3
2183        {
2184            let file = H5File::create(&path).unwrap();
2185            let ds = file
2186                .new_dataset::<f64>()
2187                .shape([4, 3])
2188                .chunk(&[2, 2])
2189                .create("m")
2190                .unwrap();
2191            ds.write_raw(&data).unwrap();
2192            file.close().unwrap();
2193        }
2194        {
2195            let file = H5File::open(&path).unwrap();
2196            let ds = file.dataset("m").unwrap();
2197            assert_eq!(ds.chunk_dims(), Some(vec![2, 2]));
2198            assert_eq!(ds.read_raw::<f64>().unwrap(), data);
2199        }
2200        std::fs::remove_file(&path).ok();
2201    }
2202
2203    // write_raw on an extensible-array (unlimited first dim) compressed dataset
2204    // drives write_full_image_chunked's EA branch, which gathers chunks and
2205    // compresses them through the windowed batch path. Round-trips the full
2206    // image, including a partial edge chunk along the unlimited dimension.
2207    #[cfg(feature = "deflate")]
2208    #[test]
2209    fn write_raw_ea_compressed_roundtrips() {
2210        let path = temp_path("write_raw_ea_deflate");
2211        let data: Vec<i32> = (0..20).collect(); // 5 x 4 row-major
2212        {
2213            let file = H5File::create(&path).unwrap();
2214            let ds = file
2215                .new_dataset::<i32>()
2216                .shape([5, 4])
2217                .chunk(&[2, 4])
2218                .max_shape(&[None, Some(4)]) // unlimited dim 0 -> extensible array
2219                .deflate(5)
2220                .create("stream")
2221                .unwrap();
2222            ds.write_raw(&data).unwrap();
2223            file.close().unwrap();
2224        }
2225        {
2226            let file = H5File::open(&path).unwrap();
2227            let ds = file.dataset("stream").unwrap();
2228            assert_eq!(ds.shape(), vec![5, 4]);
2229            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2230        }
2231        std::fs::remove_file(&path).ok();
2232    }
2233
2234    // 3D chunked Full read with partial-edge chunks in every dimension. This
2235    // drives copy_chunk_to_output's multi-dim run-memcpy path with two outer
2236    // dimensions, exercising the nested outer-coordinate carry and the
2237    // last-axis edge clamp (chunks hang off the high edge in all three axes).
2238    #[cfg(feature = "deflate")]
2239    #[test]
2240    fn read_full_3d_chunked_edge_roundtrips() {
2241        let path = temp_path("full_3d_chunked_edge");
2242        // shape 5x4x3, chunk 2x3x2 -> ceil gives 3x2x2 chunks; the last chunk
2243        // along each axis is partial (1, 1, and 1 element respectively).
2244        let total: usize = 5 * 4 * 3;
2245        let data: Vec<i32> = (0..total as i32).collect();
2246        {
2247            let file = H5File::create(&path).unwrap();
2248            let ds = file
2249                .new_dataset::<i32>()
2250                .shape([5, 4, 3])
2251                .chunk(&[2, 3, 2])
2252                .deflate(4)
2253                .create("vol")
2254                .unwrap();
2255            ds.write_raw(&data).unwrap();
2256            file.close().unwrap();
2257        }
2258        {
2259            let file = H5File::open(&path).unwrap();
2260            let ds = file.dataset("vol").unwrap();
2261            assert_eq!(ds.shape(), vec![5, 4, 3]);
2262            assert_eq!(ds.chunk_dims(), Some(vec![2, 3, 2]));
2263            assert_eq!(ds.read_raw::<i32>().unwrap(), data);
2264        }
2265        std::fs::remove_file(&path).ok();
2266    }
2267
2268    #[test]
2269    fn roundtrip_u8_1d() {
2270        let path = temp_path("rt_u8_1d");
2271        let data: Vec<u8> = (0..10).collect();
2272
2273        {
2274            let file = H5File::create(&path).unwrap();
2275            let ds = file.new_dataset::<u8>().shape([10]).create("seq").unwrap();
2276            ds.write_raw(&data).unwrap();
2277            file.close().unwrap();
2278        }
2279
2280        {
2281            let file = H5File::open(&path).unwrap();
2282            let ds = file.dataset("seq").unwrap();
2283            assert_eq!(ds.shape(), vec![10]);
2284            let readback = ds.read_raw::<u8>().unwrap();
2285            assert_eq!(readback, data);
2286        }
2287
2288        std::fs::remove_file(&path).ok();
2289    }
2290
2291    #[test]
2292    fn roundtrip_i32_2d() {
2293        let path = temp_path("rt_i32_2d");
2294        let data: Vec<i32> = vec![-1, 0, 1, 2, 3, 4];
2295
2296        {
2297            let file = H5File::create(&path).unwrap();
2298            let ds = file
2299                .new_dataset::<i32>()
2300                .shape([2, 3])
2301                .create("matrix")
2302                .unwrap();
2303            ds.write_raw(&data).unwrap();
2304            file.close().unwrap();
2305        }
2306
2307        {
2308            let file = H5File::open(&path).unwrap();
2309            let ds = file.dataset("matrix").unwrap();
2310            assert_eq!(ds.shape(), vec![2, 3]);
2311            let readback = ds.read_raw::<i32>().unwrap();
2312            assert_eq!(readback, data);
2313        }
2314
2315        std::fs::remove_file(&path).ok();
2316    }
2317
2318    #[test]
2319    fn roundtrip_f64_3d() {
2320        let path = temp_path("rt_f64_3d");
2321        let data: Vec<f64> = (0..24).map(|i| i as f64 * 0.5).collect();
2322
2323        {
2324            let file = H5File::create(&path).unwrap();
2325            let ds = file
2326                .new_dataset::<f64>()
2327                .shape([2, 3, 4])
2328                .create("cube")
2329                .unwrap();
2330            ds.write_raw(&data).unwrap();
2331            file.close().unwrap();
2332        }
2333
2334        {
2335            let file = H5File::open(&path).unwrap();
2336            let ds = file.dataset("cube").unwrap();
2337            assert_eq!(ds.shape(), vec![2, 3, 4]);
2338            let readback = ds.read_raw::<f64>().unwrap();
2339            assert_eq!(readback, data);
2340        }
2341
2342        std::fs::remove_file(&path).ok();
2343    }
2344
2345    #[test]
2346    fn cannot_read_in_write_mode() {
2347        let path = temp_path("no_read_write");
2348        let file = H5File::create(&path).unwrap();
2349        let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2350        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2351        let result = ds.read_raw::<u8>();
2352        assert!(result.is_err());
2353        std::fs::remove_file(&path).ok();
2354    }
2355
2356    #[test]
2357    fn cannot_write_in_read_mode() {
2358        let path = temp_path("no_write_read");
2359
2360        {
2361            let file = H5File::create(&path).unwrap();
2362            let ds = file.new_dataset::<u8>().shape([4]).create("x").unwrap();
2363            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2364            file.close().unwrap();
2365        }
2366
2367        {
2368            let file = H5File::open(&path).unwrap();
2369            let ds = file.dataset("x").unwrap();
2370            let result = ds.write_raw(&[5u8, 6, 7, 8]);
2371            assert!(result.is_err());
2372        }
2373
2374        std::fs::remove_file(&path).ok();
2375    }
2376
2377    #[test]
2378    fn numeric_attr_roundtrip() {
2379        let path = temp_path("num_attr");
2380        {
2381            let file = H5File::create(&path).unwrap();
2382            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2383            ds.write_raw(&[1.0f32; 4]).unwrap();
2384
2385            let a1 = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
2386            a1.write_numeric(&1.2345f64).unwrap();
2387
2388            let a2 = ds.new_attr::<i32>().shape(()).create("count").unwrap();
2389            a2.write_numeric(&42i32).unwrap();
2390
2391            file.close().unwrap();
2392        }
2393        {
2394            let file = H5File::open(&path).unwrap();
2395            let ds = file.dataset("data").unwrap();
2396
2397            let scale = ds.attr("scale").unwrap();
2398            let val: f64 = scale.read_numeric().unwrap();
2399            assert!((val - 1.2345).abs() < 1e-10);
2400
2401            let count = ds.attr("count").unwrap();
2402            let val: i32 = count.read_numeric().unwrap();
2403            assert_eq!(val, 42);
2404        }
2405        std::fs::remove_file(&path).ok();
2406    }
2407
2408    #[test]
2409    fn array_attr_roundtrip() {
2410        let path = temp_path("array_attr");
2411        let offsets = [10i32, -20, 30];
2412        {
2413            let file = H5File::create(&path).unwrap();
2414            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2415            ds.write_raw(&[1.0f32; 4]).unwrap();
2416
2417            // 1-D int32 array attribute (NDArrayDimOffset-style).
2418            let a = ds
2419                .new_attr::<i32>()
2420                .shape([3])
2421                .create("dim_offset")
2422                .unwrap();
2423            a.write_array(&offsets).unwrap();
2424
2425            // Wrong element count is rejected.
2426            let bad = ds.new_attr::<i32>().shape([3]).create("bad").unwrap();
2427            assert!(bad.write_array(&[1i32, 2]).is_err());
2428
2429            file.close().unwrap();
2430        }
2431        {
2432            let file = H5File::open(&path).unwrap();
2433            let ds = file.dataset("data").unwrap();
2434            let a = ds.attr("dim_offset").unwrap();
2435            let raw = a.read_raw().unwrap();
2436            assert_eq!(raw.len(), 3 * 4);
2437            let got: Vec<i32> = raw
2438                .chunks_exact(4)
2439                .map(|b| i32::from_le_bytes([b[0], b[1], b[2], b[3]]))
2440                .collect();
2441            assert_eq!(got, offsets);
2442        }
2443        std::fs::remove_file(&path).ok();
2444    }
2445
2446    #[test]
2447    fn attr_datatype_exposes_class_and_sign() {
2448        // H5Attribute::datatype() must report the stored datatype class and
2449        // signedness so a generic attr->metadata mapper need not infer it from
2450        // the byte width (the HDF5-L1 adapter blocker this accessor unblocks).
2451        use crate::format::messages::datatype::DatatypeMessage;
2452
2453        let path = temp_path("attr_datatype");
2454        {
2455            let file = H5File::create(&path).unwrap();
2456            let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2457            ds.new_attr::<f64>()
2458                .shape(())
2459                .create("scale")
2460                .unwrap()
2461                .write_numeric(&1.5f64)
2462                .unwrap();
2463            ds.new_attr::<i32>()
2464                .shape(())
2465                .create("count")
2466                .unwrap()
2467                .write_numeric(&7i32)
2468                .unwrap();
2469            file.close().unwrap();
2470        }
2471        {
2472            let file = H5File::open(&path).unwrap();
2473            let ds = file.dataset("data").unwrap();
2474
2475            match ds.attr("scale").unwrap().datatype().unwrap() {
2476                DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 8),
2477                other => panic!("expected FloatingPoint for f64 attr, got {other:?}"),
2478            }
2479
2480            match ds.attr("count").unwrap().datatype().unwrap() {
2481                DatatypeMessage::FixedPoint { size, signed, .. } => {
2482                    assert_eq!(size, 4);
2483                    assert!(signed, "i32 attr must be signed");
2484                }
2485                other => panic!("expected FixedPoint for i32 attr, got {other:?}"),
2486            }
2487        }
2488        std::fs::remove_file(&path).ok();
2489    }
2490
2491    #[test]
2492    fn attr_datatype_in_write_mode_errors() {
2493        let path = temp_path("attr_datatype_write_mode");
2494        let file = H5File::create(&path).unwrap();
2495        let ds = file.new_dataset::<f32>().shape([4]).create("data").unwrap();
2496        let attr = ds.new_attr::<f64>().shape(()).create("scale").unwrap();
2497        assert!(attr.datatype().is_err());
2498        std::fs::remove_file(&path).ok();
2499    }
2500
2501    #[test]
2502    fn cannot_create_dataset_in_read_mode() {
2503        let path = temp_path("no_create_read");
2504
2505        {
2506            let _file = H5File::create(&path).unwrap();
2507        }
2508
2509        {
2510            let file = H5File::open(&path).unwrap();
2511            let result = file.new_dataset::<u8>().shape([4]).create("x");
2512            assert!(result.is_err());
2513        }
2514
2515        std::fs::remove_file(&path).ok();
2516    }
2517
2518    #[test]
2519    fn shape_accessor() {
2520        let path = temp_path("shape_acc");
2521
2522        let file = H5File::create(&path).unwrap();
2523        let ds = file
2524            .new_dataset::<f32>()
2525            .shape([5, 10, 3])
2526            .create("tensor")
2527            .unwrap();
2528        assert_eq!(ds.shape(), vec![5, 10, 3]);
2529
2530        std::fs::remove_file(&path).ok();
2531    }
2532
2533    #[test]
2534    fn slice_roundtrip_2d() {
2535        let path = temp_path("slice_2d");
2536
2537        // Create a 4x5 dataset, write full, then read a slice
2538        let data: Vec<i32> = (0..20).collect();
2539        {
2540            let file = H5File::create(&path).unwrap();
2541            let ds = file
2542                .new_dataset::<i32>()
2543                .shape([4, 5])
2544                .create("mat")
2545                .unwrap();
2546            ds.write_raw(&data).unwrap();
2547            file.close().unwrap();
2548        }
2549        {
2550            let file = H5File::open(&path).unwrap();
2551            let ds = file.dataset("mat").unwrap();
2552            // Read rows 1..3, cols 2..4 (2x2 slice)
2553            let slice = ds.read_slice::<i32>(&[1, 2], &[2, 2]).unwrap();
2554            // Row 1: [5,6,7,8,9] -> cols 2..4 = [7,8]
2555            // Row 2: [10,11,12,13,14] -> cols 2..4 = [12,13]
2556            assert_eq!(slice, vec![7, 8, 12, 13]);
2557        }
2558
2559        std::fs::remove_file(&path).ok();
2560    }
2561
2562    // H2D zero-alloc reads. `read_raw_into` / `read_slice_into` fill a
2563    // caller-provided buffer and MUST produce byte-for-byte the same data as
2564    // their Vec-returning counterparts (`read_raw` / `read_slice`) on every
2565    // creatable layout, since both now share one buffer-filling core.
2566    fn assert_into_matches<T>(ds: &super::H5Dataset, starts: &[usize], counts: &[usize])
2567    where
2568        T: crate::types::H5Type + Copy + std::fmt::Debug + PartialEq + Default,
2569    {
2570        let n: usize = ds.shape().iter().product();
2571        let want_full = ds.read_raw::<T>().unwrap();
2572        let mut got_full = vec![T::default(); n];
2573        ds.read_raw_into::<T>(&mut got_full).unwrap();
2574        assert_eq!(got_full, want_full, "read_raw_into != read_raw");
2575
2576        let want_slice = ds.read_slice::<T>(starts, counts).unwrap();
2577        let sn: usize = counts.iter().product();
2578        let mut got_slice = vec![T::default(); sn];
2579        ds.read_slice_into::<T>(&mut got_slice, starts, counts)
2580            .unwrap();
2581        assert_eq!(got_slice, want_slice, "read_slice_into != read_slice");
2582    }
2583
2584    #[test]
2585    fn read_into_matches_vec_contiguous() {
2586        let path = temp_path("into_contig");
2587        let data: Vec<i32> = (0..20).collect(); // 4 x 5 contiguous
2588        {
2589            let file = H5File::create(&path).unwrap();
2590            let ds = file
2591                .new_dataset::<i32>()
2592                .shape([4, 5])
2593                .create("mat")
2594                .unwrap();
2595            ds.write_raw(&data).unwrap();
2596            file.close().unwrap();
2597        }
2598        {
2599            let file = H5File::open(&path).unwrap();
2600            let ds = file.dataset("mat").unwrap();
2601            assert_eq!(ds.chunk_dims(), None);
2602            assert_into_matches::<i32>(&ds, &[1, 2], &[2, 2]);
2603        }
2604        std::fs::remove_file(&path).ok();
2605    }
2606
2607    #[test]
2608    fn read_into_matches_vec_chunked_unfiltered() {
2609        let path = temp_path("into_chunk");
2610        let data: Vec<f64> = (0..35).map(|i| i as f64 * 1.5).collect(); // 7 x 5
2611        {
2612            let file = H5File::create(&path).unwrap();
2613            let ds = file
2614                .new_dataset::<f64>()
2615                .shape([7, 5])
2616                .chunk(&[3, 2]) // multi-chunk grid with edge chunks
2617                .create("grid")
2618                .unwrap();
2619            ds.write_raw(&data).unwrap();
2620            file.close().unwrap();
2621        }
2622        {
2623            let file = H5File::open(&path).unwrap();
2624            let ds = file.dataset("grid").unwrap();
2625            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2626            // Slice spans multiple chunks (rows 2..5, cols 1..4).
2627            assert_into_matches::<f64>(&ds, &[2, 1], &[3, 3]);
2628        }
2629        std::fs::remove_file(&path).ok();
2630    }
2631
2632    #[test]
2633    fn read_into_matches_vec_single_chunk() {
2634        let path = temp_path("into_single_chunk");
2635        let data: Vec<i32> = (0..12).collect(); // 3 x 4, one chunk covers all
2636        {
2637            let file = H5File::create(&path).unwrap();
2638            let ds = file
2639                .new_dataset::<i32>()
2640                .shape([3, 4])
2641                .chunk(&[3, 4]) // chunk == shape -> SingleChunk index
2642                .create("g")
2643                .unwrap();
2644            ds.write_raw(&data).unwrap();
2645            file.close().unwrap();
2646        }
2647        {
2648            let file = H5File::open(&path).unwrap();
2649            let ds = file.dataset("g").unwrap();
2650            assert_eq!(ds.chunk_dims(), Some(vec![3, 4]));
2651            assert_into_matches::<i32>(&ds, &[1, 1], &[2, 2]);
2652        }
2653        std::fs::remove_file(&path).ok();
2654    }
2655
2656    #[cfg(feature = "deflate")]
2657    #[test]
2658    fn read_into_matches_vec_chunked_deflate() {
2659        let path = temp_path("into_chunk_deflate");
2660        let data: Vec<i32> = (0..35).collect(); // 7 x 5
2661        {
2662            let file = H5File::create(&path).unwrap();
2663            let ds = file
2664                .new_dataset::<i32>()
2665                .shape([7, 5])
2666                .chunk(&[3, 2])
2667                .deflate(4)
2668                .create("grid")
2669                .unwrap();
2670            ds.write_raw(&data).unwrap();
2671            file.close().unwrap();
2672        }
2673        {
2674            let file = H5File::open(&path).unwrap();
2675            let ds = file.dataset("grid").unwrap();
2676            assert_eq!(ds.chunk_dims(), Some(vec![3, 2]));
2677            assert_into_matches::<i32>(&ds, &[2, 1], &[3, 3]);
2678        }
2679        std::fs::remove_file(&path).ok();
2680    }
2681
2682    #[test]
2683    fn read_into_wrong_buffer_size_rejected() {
2684        let path = temp_path("into_badlen");
2685        let data: Vec<i32> = (0..20).collect(); // 4 x 5
2686        {
2687            let file = H5File::create(&path).unwrap();
2688            let ds = file
2689                .new_dataset::<i32>()
2690                .shape([4, 5])
2691                .create("mat")
2692                .unwrap();
2693            ds.write_raw(&data).unwrap();
2694            file.close().unwrap();
2695        }
2696        {
2697            let file = H5File::open(&path).unwrap();
2698            let ds = file.dataset("mat").unwrap();
2699
2700            // Too small / too large full-read buffers are both rejected.
2701            let mut small = vec![0i32; 19];
2702            assert!(ds.read_raw_into::<i32>(&mut small).is_err());
2703            let mut large = vec![0i32; 21];
2704            assert!(ds.read_raw_into::<i32>(&mut large).is_err());
2705
2706            // Slice buffer must be exactly product(counts) = 4.
2707            let mut bad_slice = vec![0i32; 3];
2708            assert!(ds
2709                .read_slice_into::<i32>(&mut bad_slice, &[1, 2], &[2, 2])
2710                .is_err());
2711            // The correctly sized slice buffer succeeds.
2712            let mut ok_slice = vec![0i32; 4];
2713            assert!(ds
2714                .read_slice_into::<i32>(&mut ok_slice, &[1, 2], &[2, 2])
2715                .is_ok());
2716        }
2717        std::fs::remove_file(&path).ok();
2718    }
2719
2720    #[test]
2721    fn read_into_wrong_element_size_rejected() {
2722        let path = temp_path("into_badtype");
2723        let data: Vec<i32> = (0..20).collect(); // element size 4
2724        {
2725            let file = H5File::create(&path).unwrap();
2726            let ds = file
2727                .new_dataset::<i32>()
2728                .shape([4, 5])
2729                .create("mat")
2730                .unwrap();
2731            ds.write_raw(&data).unwrap();
2732            file.close().unwrap();
2733        }
2734        {
2735            let file = H5File::open(&path).unwrap();
2736            let ds = file.dataset("mat").unwrap();
2737            // u8 (size 1) and i64 (size 8) mismatch the dataset's 4-byte
2738            // element size -> TypeMismatch, even with a "correctly sized" Vec.
2739            let mut as_u8 = vec![0u8; 20];
2740            assert!(matches!(
2741                ds.read_raw_into::<u8>(&mut as_u8),
2742                Err(crate::Hdf5Error::TypeMismatch(_))
2743            ));
2744            let mut as_i64 = vec![0i64; 20];
2745            assert!(matches!(
2746                ds.read_slice_into::<i64>(&mut as_i64, &[0, 0], &[4, 5]),
2747                Err(crate::Hdf5Error::TypeMismatch(_))
2748            ));
2749        }
2750        std::fs::remove_file(&path).ok();
2751    }
2752
2753    #[test]
2754    fn write_slice_2d() {
2755        let path = temp_path("write_slice_2d");
2756
2757        {
2758            let file = H5File::create(&path).unwrap();
2759            let ds = file
2760                .new_dataset::<f32>()
2761                .shape([3, 4])
2762                .create("data")
2763                .unwrap();
2764            ds.write_raw(&[0.0f32; 12]).unwrap();
2765            // Overwrite a 2x2 sub-region
2766            ds.write_slice(&[1, 1], &[2, 2], &[10.0f32, 20.0, 30.0, 40.0])
2767                .unwrap();
2768            file.close().unwrap();
2769        }
2770        {
2771            let file = H5File::open(&path).unwrap();
2772            let ds = file.dataset("data").unwrap();
2773            let full = ds.read_raw::<f32>().unwrap();
2774            // Row 0: [0,0,0,0]
2775            // Row 1: [0,10,20,0]
2776            // Row 2: [0,30,40,0]
2777            assert_eq!(
2778                full,
2779                vec![0.0, 0.0, 0.0, 0.0, 0.0, 10.0, 20.0, 0.0, 0.0, 30.0, 40.0, 0.0,]
2780            );
2781        }
2782
2783        std::fs::remove_file(&path).ok();
2784    }
2785
2786    #[test]
2787    fn write_slice_out_of_bounds_rejected() {
2788        let path = temp_path("write_slice_oob");
2789        let file = H5File::create(&path).unwrap();
2790        let ds = file.new_dataset::<i32>().shape([4]).create("d").unwrap();
2791        ds.write_raw(&[0i32; 4]).unwrap();
2792        // start 2 + count 6 = 8 > extent 4 -> must error, not corrupt.
2793        assert!(ds.write_slice(&[2], &[6], &[9i32; 6]).is_err());
2794        // An in-bounds slice still works.
2795        assert!(ds.write_slice(&[1], &[2], &[7i32, 8]).is_ok());
2796        std::fs::remove_file(&path).ok();
2797    }
2798
2799    #[test]
2800    fn duplicate_dataset_name_rejected() {
2801        let path = temp_path("dup_name");
2802        let file = H5File::create(&path).unwrap();
2803        let _ = file.new_dataset::<i32>().shape([2]).create("d").unwrap();
2804        assert!(file.new_dataset::<i32>().shape([2]).create("d").is_err());
2805        std::fs::remove_file(&path).ok();
2806    }
2807
2808    #[test]
2809    fn extend_cannot_shrink() {
2810        let path = temp_path("extend_shrink");
2811        let file = H5File::create(&path).unwrap();
2812        let ds = file
2813            .new_dataset::<i32>()
2814            .shape([0])
2815            .chunk(&[2])
2816            .max_shape(&[None])
2817            .create("d")
2818            .unwrap();
2819        ds.append(&[1i32, 2, 3, 4]).unwrap();
2820        // Shrinking below the written extent must be rejected.
2821        assert!(ds.extend(&[2]).is_err());
2822        // Growing is fine.
2823        assert!(ds.extend(&[6]).is_ok());
2824        std::fs::remove_file(&path).ok();
2825    }
2826
2827    #[test]
2828    fn attr_read_roundtrip() {
2829        use crate::types::VarLenUnicode;
2830        let path = temp_path("attr_read");
2831
2832        {
2833            let file = H5File::create(&path).unwrap();
2834            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
2835            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2836            let a1 = ds
2837                .new_attr::<VarLenUnicode>()
2838                .shape(())
2839                .create("units")
2840                .unwrap();
2841            a1.write_string("meters").unwrap();
2842            let a2 = ds
2843                .new_attr::<VarLenUnicode>()
2844                .shape(())
2845                .create("desc")
2846                .unwrap();
2847            a2.write_string("test data").unwrap();
2848            file.close().unwrap();
2849        }
2850        {
2851            let file = H5File::open(&path).unwrap();
2852            let ds = file.dataset("data").unwrap();
2853
2854            let names = ds.attr_names().unwrap();
2855            assert!(names.contains(&"units".to_string()));
2856            assert!(names.contains(&"desc".to_string()));
2857
2858            let units = ds.attr("units").unwrap();
2859            assert_eq!(units.read_string().unwrap(), "meters");
2860
2861            let desc = ds.attr("desc").unwrap();
2862            assert_eq!(desc.read_string().unwrap(), "test data");
2863        }
2864
2865        std::fs::remove_file(&path).ok();
2866    }
2867
2868    #[test]
2869    fn type_mismatch_element_size() {
2870        let path = temp_path("type_mismatch");
2871
2872        {
2873            let file = H5File::create(&path).unwrap();
2874            let ds = file.new_dataset::<f64>().shape([4]).create("data").unwrap();
2875            ds.write_raw(&[1.0f64, 2.0, 3.0, 4.0]).unwrap();
2876            file.close().unwrap();
2877        }
2878
2879        {
2880            let file = H5File::open(&path).unwrap();
2881            let ds = file.dataset("data").unwrap();
2882            // Try to read as u8 (element_size = 1) from a f64 dataset (element_size = 8)
2883            let result = ds.read_raw::<u8>();
2884            assert!(result.is_err());
2885        }
2886
2887        std::fs::remove_file(&path).ok();
2888    }
2889
2890    #[test]
2891    fn dataset_survives_file_move() {
2892        let path = temp_path("ds_survives");
2893
2894        let ds = {
2895            let file = H5File::create(&path).unwrap();
2896            file.new_dataset::<u8>().shape([4]).create("x").unwrap()
2897        };
2898        // file is dropped here, but ds still holds Rc to the inner state
2899        ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2900        // The writer will finalize on drop of the last Rc
2901
2902        std::fs::remove_file(&path).ok();
2903    }
2904
2905    #[test]
2906    fn new_attr_scalar_string() {
2907        use crate::types::VarLenUnicode;
2908
2909        let path = temp_path("attr_scalar_string");
2910        {
2911            let file = H5File::create(&path).unwrap();
2912            let ds = file.new_dataset::<u8>().shape([4]).create("data").unwrap();
2913            ds.write_raw(&[1u8, 2, 3, 4]).unwrap();
2914
2915            let attr = ds
2916                .new_attr::<VarLenUnicode>()
2917                .shape(())
2918                .create("name")
2919                .unwrap();
2920            attr.write_scalar(&VarLenUnicode("test_value".to_string()))
2921                .unwrap();
2922
2923            file.close().unwrap();
2924        }
2925
2926        // Verify the file is still valid and readable
2927        {
2928            use crate::format::messages::datatype::DatatypeMessage;
2929            let file = H5File::open(&path).unwrap();
2930            let ds = file.dataset("data").unwrap();
2931            assert_eq!(ds.shape(), vec![4]);
2932            let readback = ds.read_raw::<u8>().unwrap();
2933            assert_eq!(readback, vec![1u8, 2, 3, 4]);
2934
2935            // The string attribute is stored as a true variable-length string
2936            // (not fixed-length) and round-trips its value.
2937            let attr = ds.attr("name").unwrap();
2938            assert!(
2939                matches!(
2940                    attr.datatype().unwrap(),
2941                    DatatypeMessage::VarLenString { .. }
2942                ),
2943                "string attribute should have a variable-length string datatype"
2944            );
2945            assert_eq!(attr.read_string().unwrap(), "test_value");
2946        }
2947
2948        std::fs::remove_file(&path).ok();
2949    }
2950
2951    #[test]
2952    fn all_numeric_types_roundtrip() {
2953        let path = temp_path("all_types");
2954
2955        {
2956            let file = H5File::create(&path).unwrap();
2957
2958            let ds = file.new_dataset::<u8>().shape([2]).create("u8").unwrap();
2959            ds.write_raw(&[1u8, 2]).unwrap();
2960
2961            let ds = file.new_dataset::<i8>().shape([2]).create("i8").unwrap();
2962            ds.write_raw(&[-1i8, 1]).unwrap();
2963
2964            let ds = file.new_dataset::<u16>().shape([2]).create("u16").unwrap();
2965            ds.write_raw(&[100u16, 200]).unwrap();
2966
2967            let ds = file.new_dataset::<i16>().shape([2]).create("i16").unwrap();
2968            ds.write_raw(&[-100i16, 100]).unwrap();
2969
2970            let ds = file.new_dataset::<u32>().shape([2]).create("u32").unwrap();
2971            ds.write_raw(&[1000u32, 2000]).unwrap();
2972
2973            let ds = file.new_dataset::<i32>().shape([2]).create("i32").unwrap();
2974            ds.write_raw(&[-1000i32, 1000]).unwrap();
2975
2976            let ds = file.new_dataset::<u64>().shape([2]).create("u64").unwrap();
2977            ds.write_raw(&[10000u64, 20000]).unwrap();
2978
2979            let ds = file.new_dataset::<i64>().shape([2]).create("i64").unwrap();
2980            ds.write_raw(&[-10000i64, 10000]).unwrap();
2981
2982            let ds = file.new_dataset::<f32>().shape([2]).create("f32").unwrap();
2983            ds.write_raw(&[1.5f32, 2.5]).unwrap();
2984
2985            let ds = file.new_dataset::<f64>().shape([2]).create("f64").unwrap();
2986            ds.write_raw(&[1.23456f64, 7.89012]).unwrap();
2987
2988            file.close().unwrap();
2989        }
2990
2991        {
2992            let file = H5File::open(&path).unwrap();
2993
2994            assert_eq!(
2995                file.dataset("u8").unwrap().read_raw::<u8>().unwrap(),
2996                vec![1u8, 2]
2997            );
2998            assert_eq!(
2999                file.dataset("i8").unwrap().read_raw::<i8>().unwrap(),
3000                vec![-1i8, 1]
3001            );
3002            assert_eq!(
3003                file.dataset("u16").unwrap().read_raw::<u16>().unwrap(),
3004                vec![100u16, 200]
3005            );
3006            assert_eq!(
3007                file.dataset("i16").unwrap().read_raw::<i16>().unwrap(),
3008                vec![-100i16, 100]
3009            );
3010            assert_eq!(
3011                file.dataset("u32").unwrap().read_raw::<u32>().unwrap(),
3012                vec![1000u32, 2000]
3013            );
3014            assert_eq!(
3015                file.dataset("i32").unwrap().read_raw::<i32>().unwrap(),
3016                vec![-1000i32, 1000]
3017            );
3018            assert_eq!(
3019                file.dataset("u64").unwrap().read_raw::<u64>().unwrap(),
3020                vec![10000u64, 20000]
3021            );
3022            assert_eq!(
3023                file.dataset("i64").unwrap().read_raw::<i64>().unwrap(),
3024                vec![-10000i64, 10000]
3025            );
3026            assert_eq!(
3027                file.dataset("f32").unwrap().read_raw::<f32>().unwrap(),
3028                vec![1.5f32, 2.5]
3029            );
3030            assert_eq!(
3031                file.dataset("f64").unwrap().read_raw::<f64>().unwrap(),
3032                vec![1.23456f64, 7.89012]
3033            );
3034        }
3035
3036        std::fs::remove_file(&path).ok();
3037    }
3038
3039    #[test]
3040    fn append_chunked_roundtrip() {
3041        let path = temp_path("append_chunked");
3042
3043        {
3044            let file = H5File::create(&path).unwrap();
3045            let ds = file
3046                .new_dataset::<f64>()
3047                .shape([0, 3])
3048                .chunk(&[1, 3])
3049                .max_shape(&[None, Some(3)])
3050                .create("data")
3051                .unwrap();
3052
3053            // Append one frame
3054            ds.append(&[1.0f64, 2.0, 3.0]).unwrap();
3055            // Append two frames at once
3056            ds.append(&[4.0f64, 5.0, 6.0, 7.0, 8.0, 9.0]).unwrap();
3057
3058            file.close().unwrap();
3059        }
3060
3061        {
3062            let file = H5File::open(&path).unwrap();
3063            let ds = file.dataset("data").unwrap();
3064            assert_eq!(ds.shape(), vec![3, 3]);
3065            let all = ds.read_raw::<f64>().unwrap();
3066            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0, 6.0, 7.0, 8.0, 9.0]);
3067        }
3068
3069        std::fs::remove_file(&path).ok();
3070    }
3071
3072    #[test]
3073    fn append_1d_chunked() {
3074        let path = temp_path("append_1d");
3075
3076        {
3077            let file = H5File::create(&path).unwrap();
3078            let ds = file
3079                .new_dataset::<i32>()
3080                .shape([0])
3081                .chunk(&[4])
3082                .max_shape(&[None])
3083                .create("values")
3084                .unwrap();
3085
3086            ds.append(&[10i32, 20, 30]).unwrap(); // partial chunk
3087            ds.append(&[40i32]).unwrap(); // fills chunk boundary
3088            ds.append(&[50i32, 60, 70, 80]).unwrap(); // full chunk
3089
3090            file.close().unwrap();
3091        }
3092
3093        {
3094            let file = H5File::open(&path).unwrap();
3095            let ds = file.dataset("values").unwrap();
3096            assert_eq!(ds.shape(), vec![8]);
3097            let all = ds.read_raw::<i32>().unwrap();
3098            assert_eq!(all, vec![10, 20, 30, 40, 50, 60, 70, 80]);
3099        }
3100
3101        std::fs::remove_file(&path).ok();
3102    }
3103
3104    #[test]
3105    fn append_partial_chunk_flushed_on_close() {
3106        let path = temp_path("append_partial_close");
3107
3108        {
3109            let file = H5File::create(&path).unwrap();
3110            let ds = file
3111                .new_dataset::<f64>()
3112                .shape([0])
3113                .chunk(&[4])
3114                .max_shape(&[None])
3115                .create("vals")
3116                .unwrap();
3117
3118            // Append 5 elements: chunk 0 = full [1,2,3,4], chunk 1 = partial [5,0,0,0]
3119            ds.append(&[1.0f64, 2.0, 3.0, 4.0, 5.0]).unwrap();
3120            file.close().unwrap();
3121        }
3122
3123        {
3124            let file = H5File::open(&path).unwrap();
3125            let ds = file.dataset("vals").unwrap();
3126            assert_eq!(ds.shape(), vec![5]);
3127            let all = ds.read_raw::<f64>().unwrap();
3128            // The full dataset is 2 chunks * 4 = 8 elements; shape says 5
3129            // read_raw reads total shape elements
3130            assert_eq!(all.len(), 5);
3131            assert_eq!(all, vec![1.0, 2.0, 3.0, 4.0, 5.0]);
3132        }
3133
3134        std::fs::remove_file(&path).ok();
3135    }
3136
3137    #[cfg(feature = "deflate")]
3138    #[test]
3139    fn vlen_append_after_reopen_filtered() {
3140        // Reopen + append into a partially-written *compressed* vlen chunk
3141        // (index-block chunk). Exercises filtered-index-block reconstruction
3142        // in open_append plus filtered read-modify-write.
3143        let path = temp_path("vlen_reopen_filtered");
3144        {
3145            let file = H5File::create(&path).unwrap();
3146            file.create_appendable_vlen_dataset(
3147                "strs",
3148                4,
3149                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
3150            )
3151            .unwrap();
3152            file.append_vlen_strings("strs", &["alpha", "beta", "gamma"])
3153                .unwrap();
3154            file.close().unwrap();
3155        }
3156        {
3157            let file = H5File::open_rw(&path).unwrap();
3158            file.append_vlen_strings("strs", &["delta"]).unwrap();
3159            file.close().unwrap();
3160        }
3161        {
3162            let file = H5File::open(&path).unwrap();
3163            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3164            assert_eq!(
3165                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
3166                vec!["alpha", "beta", "gamma", "delta"]
3167            );
3168        }
3169        std::fs::remove_file(&path).ok();
3170    }
3171
3172    #[test]
3173    fn vlen_append_after_reopen_data_block() {
3174        // Reopen + append into a partial chunk that lives in an extensible-
3175        // array *data block* (chunk index >= idx_blk_elmts). Exercises
3176        // data-block resolution in read_chunk_if_present and write_chunk.
3177        let path = temp_path("vlen_reopen_datablk");
3178        let labels: Vec<String> = (0..9).map(|i| format!("s{i}")).collect();
3179        {
3180            let file = H5File::create(&path).unwrap();
3181            file.create_appendable_vlen_dataset("strs", 2, None)
3182                .unwrap();
3183            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3184            file.append_vlen_strings("strs", &refs).unwrap();
3185            file.close().unwrap();
3186        }
3187        {
3188            let file = H5File::open_rw(&path).unwrap();
3189            file.append_vlen_strings("strs", &["s9"]).unwrap();
3190            file.close().unwrap();
3191        }
3192        {
3193            let file = H5File::open(&path).unwrap();
3194            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3195            let want: Vec<String> = (0..10).map(|i| format!("s{i}")).collect();
3196            assert_eq!(got, want);
3197        }
3198        std::fs::remove_file(&path).ok();
3199    }
3200
3201    #[test]
3202    fn vlen_append_after_reopen_super_block() {
3203        // Reopen + append into a partial chunk whose index falls in an
3204        // extensible-array *super block* (chunk index 244 with the default
3205        // EA geometry: idx_blk_elmts=4, data_blk_min_elmts=16,
3206        // sup_blk_min_data_ptrs=4 -> chunks 0..=243 are reached via the
3207        // index block or its direct data blocks, so chunk 244 is reached
3208        // via a super block read from disk). Exercises the ViaSblk branch
3209        // of read_chunk_if_present.
3210        let path = temp_path("vlen_reopen_super");
3211        // 489 strings, chunk size 2 -> chunk 244 holds one string only
3212        // (partially filled) and is flushed to disk on close.
3213        let labels: Vec<String> = (0..489).map(|i| format!("v{i}")).collect();
3214        {
3215            let file = H5File::create(&path).unwrap();
3216            file.create_appendable_vlen_dataset("strs", 2, None)
3217                .unwrap();
3218            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3219            file.append_vlen_strings("strs", &refs).unwrap();
3220            file.close().unwrap();
3221        }
3222        {
3223            let file = H5File::open_rw(&path).unwrap();
3224            file.append_vlen_strings("strs", &["v489"]).unwrap();
3225            file.close().unwrap();
3226        }
3227        {
3228            let file = H5File::open(&path).unwrap();
3229            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3230            let want: Vec<String> = (0..490).map(|i| format!("v{i}")).collect();
3231            assert_eq!(got, want);
3232        }
3233        std::fs::remove_file(&path).ok();
3234    }
3235
3236    #[cfg(feature = "deflate")]
3237    #[test]
3238    fn vlen_append_after_reopen_filtered_data_block() {
3239        // The hardest path: compressed + chunk in a data block + partial
3240        // read-modify-write across a reopen.
3241        let path = temp_path("vlen_reopen_filt_datablk");
3242        let labels: Vec<String> = (0..9).map(|i| format!("item{i:02}")).collect();
3243        {
3244            let file = H5File::create(&path).unwrap();
3245            file.create_appendable_vlen_dataset(
3246                "strs",
3247                2,
3248                Some(crate::format::messages::filter::FilterPipeline::deflate(6)),
3249            )
3250            .unwrap();
3251            let refs: Vec<&str> = labels.iter().map(|s| s.as_str()).collect();
3252            file.append_vlen_strings("strs", &refs).unwrap();
3253            file.close().unwrap();
3254        }
3255        {
3256            let file = H5File::open_rw(&path).unwrap();
3257            file.append_vlen_strings("strs", &["item09"]).unwrap();
3258            file.close().unwrap();
3259        }
3260        {
3261            let file = H5File::open(&path).unwrap();
3262            let got = file.dataset("strs").unwrap().read_vlen_strings().unwrap();
3263            let want: Vec<String> = (0..10).map(|i| format!("item{i:02}")).collect();
3264            assert_eq!(got, want);
3265        }
3266        std::fs::remove_file(&path).ok();
3267    }
3268
3269    #[test]
3270    fn group_nx_class_attribute_roundtrip() {
3271        // Non-root groups carry attributes (NeXus `NX_class`) in their
3272        // own object header, and the reader reads them back by path.
3273        let path = temp_path("group_nx_class");
3274        {
3275            let file = H5File::create(&path).unwrap();
3276            let entry = file.create_group("entry").unwrap();
3277            entry.set_attr_string("NX_class", "NXentry").unwrap();
3278            let det = entry.create_group("detector").unwrap();
3279            det.set_attr_string("NX_class", "NXdetector").unwrap();
3280            det.set_attr_numeric("frame_count", &7i32).unwrap();
3281            det.new_dataset::<f32>()
3282                .shape([4])
3283                .create("data")
3284                .unwrap()
3285                .write_raw(&[1.0f32; 4])
3286                .unwrap();
3287            file.close().unwrap();
3288        }
3289        {
3290            let file = H5File::open(&path).unwrap();
3291            let entry = file.root_group().group("entry").unwrap();
3292            assert_eq!(entry.attr_string("NX_class").unwrap(), "NXentry");
3293            let det = entry.group("detector").unwrap();
3294            assert_eq!(det.attr_string("NX_class").unwrap(), "NXdetector");
3295            let names = det.attr_names().unwrap();
3296            assert!(names.contains(&"NX_class".to_string()));
3297            assert!(names.contains(&"frame_count".to_string()));
3298        }
3299        std::fs::remove_file(&path).ok();
3300    }
3301
3302    #[test]
3303    fn ea_super_block_roundtrip() {
3304        // 2000 chunks span several extensible-array super blocks. Before
3305        // super-block support the writer errored at chunk index 228.
3306        let path = temp_path("ea_super_rt");
3307        {
3308            let file = H5File::create(&path).unwrap();
3309            let ds = file
3310                .new_dataset::<i32>()
3311                .shape([0])
3312                .chunk(&[1])
3313                .max_shape(&[None])
3314                .create("v")
3315                .unwrap();
3316            ds.append(&(0..2000).collect::<Vec<i32>>()).unwrap();
3317            file.close().unwrap();
3318        }
3319        {
3320            let file = H5File::open(&path).unwrap();
3321            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3322            assert_eq!(v.len(), 2000);
3323            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
3324        }
3325        std::fs::remove_file(&path).ok();
3326    }
3327
3328    #[cfg(feature = "deflate")]
3329    #[test]
3330    fn ea_filtered_super_block_roundtrip() {
3331        // Compressed chunks across super blocks.
3332        let path = temp_path("ea_filt_super");
3333        {
3334            let file = H5File::create(&path).unwrap();
3335            let ds = file
3336                .new_dataset::<i32>()
3337                .shape([0])
3338                .chunk(&[1])
3339                .max_shape(&[None])
3340                .deflate(4)
3341                .create("v")
3342                .unwrap();
3343            ds.append(&(0..600).collect::<Vec<i32>>()).unwrap();
3344            file.close().unwrap();
3345        }
3346        {
3347            let file = H5File::open(&path).unwrap();
3348            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3349            assert_eq!(v, (0..600).collect::<Vec<i32>>());
3350        }
3351        std::fs::remove_file(&path).ok();
3352    }
3353
3354    #[test]
3355    fn ea_super_block_open_append() {
3356        // Reopen a dataset and append chunks that fall in super blocks.
3357        let path = temp_path("ea_super_append");
3358        {
3359            let file = H5File::create(&path).unwrap();
3360            let ds = file
3361                .new_dataset::<i32>()
3362                .shape([0])
3363                .chunk(&[1])
3364                .max_shape(&[None])
3365                .create("v")
3366                .unwrap();
3367            ds.append(&(0..300).collect::<Vec<i32>>()).unwrap();
3368            file.close().unwrap();
3369        }
3370        {
3371            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3372            let idx = w.dataset_index("v").unwrap();
3373            for c in 300..900u64 {
3374                w.write_chunk(idx, c, &(c as i32).to_le_bytes()).unwrap();
3375            }
3376            w.extend_dataset(idx, &[900]).unwrap();
3377            w.close().unwrap();
3378        }
3379        {
3380            let file = H5File::open(&path).unwrap();
3381            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3382            assert_eq!(v.len(), 900);
3383            assert!(v.iter().enumerate().all(|(i, &x)| x == i as i32));
3384        }
3385        std::fs::remove_file(&path).ok();
3386    }
3387
3388    #[test]
3389    fn btree_v2_multi_unlimited_roundtrip() {
3390        // A dataset with two unlimited dimensions uses the v2 B-tree chunk
3391        // index; chunks are written by grid coordinates with write_chunk_at.
3392        let path = temp_path("bt2_multi");
3393        {
3394            let file = H5File::create(&path).unwrap();
3395            let ds = file
3396                .new_dataset::<i32>()
3397                .shape([0, 0])
3398                .chunk(&[2, 2])
3399                .max_shape(&[None, None])
3400                .create("grid")
3401                .unwrap();
3402            assert!(ds.is_chunked());
3403            // 4x4 logical grid, value[r][c] = r*4 + c, in 2x2 chunks.
3404            for cr in 0..2usize {
3405                for cc in 0..2usize {
3406                    let mut bytes = Vec::new();
3407                    for i in 0..2usize {
3408                        for j in 0..2usize {
3409                            let v = ((cr * 2 + i) * 4 + (cc * 2 + j)) as i32;
3410                            bytes.extend_from_slice(&v.to_le_bytes());
3411                        }
3412                    }
3413                    ds.write_chunk_at(&[cr, cc], &bytes).unwrap();
3414                }
3415            }
3416            file.close().unwrap();
3417        }
3418        {
3419            let file = H5File::open(&path).unwrap();
3420            let ds = file.dataset("grid").unwrap();
3421            assert_eq!(ds.shape(), vec![4, 4]);
3422            assert_eq!(ds.read_raw::<i32>().unwrap(), (0..16).collect::<Vec<i32>>());
3423        }
3424        std::fs::remove_file(&path).ok();
3425    }
3426
3427    #[test]
3428    fn subframe_chunking_roundtrip() {
3429        // A chunk smaller than a frame: shape [N,8,8], chunk [1,4,4], so each
3430        // frame is tiled into a 2x2 grid of 4x4 chunks. write_chunk_at takes
3431        // the chunk-grid coordinates.
3432        let path = temp_path("subframe");
3433        {
3434            let file = H5File::create(&path).unwrap();
3435            let ds = file
3436                .new_dataset::<i32>()
3437                .shape([0, 8, 8])
3438                .chunk(&[1, 4, 4])
3439                .max_shape(&[None, Some(8), Some(8)])
3440                .create("v")
3441                .unwrap();
3442            for f in 0..3usize {
3443                for cr in 0..2usize {
3444                    for cc in 0..2usize {
3445                        let mut bytes = Vec::new();
3446                        for i in 0..4usize {
3447                            for j in 0..4usize {
3448                                let v = (f * 64 + (cr * 4 + i) * 8 + (cc * 4 + j)) as i32;
3449                                bytes.extend_from_slice(&v.to_le_bytes());
3450                            }
3451                        }
3452                        ds.write_chunk_at(&[f, cr, cc], &bytes).unwrap();
3453                    }
3454                }
3455            }
3456            file.close().unwrap();
3457        }
3458        {
3459            let file = H5File::open(&path).unwrap();
3460            let ds = file.dataset("v").unwrap();
3461            assert_eq!(ds.shape(), vec![3, 8, 8]);
3462            assert_eq!(
3463                ds.read_raw::<i32>().unwrap(),
3464                (0..192).collect::<Vec<i32>>()
3465            );
3466        }
3467        std::fs::remove_file(&path).ok();
3468    }
3469
3470    #[test]
3471    fn fill_value_contiguous_roundtrip() {
3472        let path = temp_path("fill_value_contig");
3473        {
3474            let file = H5File::create(&path).unwrap();
3475            let ds = file
3476                .new_dataset::<f32>()
3477                .shape([4])
3478                .fill_value(2.5f32)
3479                .create("data")
3480                .unwrap();
3481            ds.write_raw(&[1.0f32, 2.0, 3.0, 4.0]).unwrap();
3482            file.close().unwrap();
3483        }
3484        // open_append decodes the fill-value message back from the header.
3485        {
3486            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3487            let idx = writer.dataset_index("data").unwrap();
3488            assert_eq!(
3489                writer.ds(idx).lock().fill_value,
3490                Some(2.5f32.to_le_bytes().to_vec())
3491            );
3492        }
3493        // Data still reads back correctly.
3494        {
3495            let file = H5File::open(&path).unwrap();
3496            let ds = file.dataset("data").unwrap();
3497            assert_eq!(ds.read_raw::<f32>().unwrap(), vec![1.0, 2.0, 3.0, 4.0]);
3498        }
3499        std::fs::remove_file(&path).ok();
3500    }
3501
3502    #[test]
3503    fn fill_value_chunked_roundtrip() {
3504        let path = temp_path("fill_value_chunked");
3505        {
3506            let file = H5File::create(&path).unwrap();
3507            let ds = file
3508                .new_dataset::<i32>()
3509                .shape([0])
3510                .chunk(&[4])
3511                .max_shape(&[None])
3512                .fill_value(-7i32)
3513                .create("vals")
3514                .unwrap();
3515            ds.append(&[1i32, 2, 3, 4]).unwrap();
3516            file.close().unwrap();
3517        }
3518        {
3519            let writer = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3520            let idx = writer.dataset_index("vals").unwrap();
3521            assert_eq!(
3522                writer.ds(idx).lock().fill_value,
3523                Some((-7i32).to_le_bytes().to_vec())
3524            );
3525        }
3526        std::fs::remove_file(&path).ok();
3527    }
3528
3529    #[test]
3530    fn fill_value_read_missing_chunks() {
3531        // A chunked dataset with chunk 1 left unwritten must read that
3532        // gap back as the user-defined fill value, not zero.
3533        fn i32_bytes(vals: &[i32]) -> Vec<u8> {
3534            vals.iter().flat_map(|v| v.to_le_bytes()).collect()
3535        }
3536        let path = temp_path("fill_value_read_missing");
3537        {
3538            let file = H5File::create(&path).unwrap();
3539            let ds = file
3540                .new_dataset::<i32>()
3541                .shape([0])
3542                .chunk(&[2])
3543                .max_shape(&[None])
3544                .fill_value(-1i32)
3545                .create("vals")
3546                .unwrap();
3547            // chunk 0 = [10,20]; chunk 1 unwritten; chunk 2 = [50,60].
3548            ds.write_chunk(0, &i32_bytes(&[10, 20])).unwrap();
3549            ds.write_chunk(2, &i32_bytes(&[50, 60])).unwrap();
3550            ds.extend(&[6]).unwrap();
3551            file.close().unwrap();
3552        }
3553        {
3554            let file = H5File::open(&path).unwrap();
3555            let ds = file.dataset("vals").unwrap();
3556            let all = ds.read_raw::<i32>().unwrap();
3557            assert_eq!(all, vec![10, 20, -1, -1, 50, 60]);
3558        }
3559        std::fs::remove_file(&path).ok();
3560    }
3561
3562    #[test]
3563    fn fill_value_partial_chunk_padded_with_fill() {
3564        // A partial trailing chunk flushed at close must pad its unwritten
3565        // tail with the fill value. That pad sits beyond the logical shape,
3566        // so it is verified by scanning the on-disk chunk bytes directly.
3567        let path = temp_path("fill_value_partial_pad");
3568        {
3569            let file = H5File::create(&path).unwrap();
3570            let ds = file
3571                .new_dataset::<i32>()
3572                .shape([0])
3573                .chunk(&[4])
3574                .max_shape(&[None])
3575                .fill_value(-9i32)
3576                .create("vals")
3577                .unwrap();
3578            // 3 of 4 frames -> flushed as a partial chunk on close.
3579            ds.append(&[1i32, 2, 3]).unwrap();
3580            file.close().unwrap();
3581        }
3582        let bytes = std::fs::read(&path).unwrap();
3583        // Locate the chunk: i32 LE of [1, 2, 3] written contiguously.
3584        let needle: Vec<u8> = [1i32, 2, 3].iter().flat_map(|v| v.to_le_bytes()).collect();
3585        let pos = bytes
3586            .windows(needle.len())
3587            .position(|w| w == needle)
3588            .expect("chunk data [1,2,3] not found in file");
3589        let pad = &bytes[pos + needle.len()..pos + needle.len() + 4];
3590        assert_eq!(
3591            pad,
3592            &(-9i32).to_le_bytes(),
3593            "partial chunk tail must be padded with fill value -9, got {:?}",
3594            pad
3595        );
3596        std::fs::remove_file(&path).ok();
3597    }
3598
3599    #[test]
3600    fn vlen_append_after_reopen_preserves_existing() {
3601        // Reopening and appending into a partially-written vlen chunk must
3602        // read-modify-write: the strings already on disk must survive.
3603        let path = temp_path("vlen_append_reopen");
3604        {
3605            let file = H5File::create(&path).unwrap();
3606            file.create_appendable_vlen_dataset("strs", 4, None)
3607                .unwrap();
3608            // 3 of 4 frames -> flushed as a partial chunk on close.
3609            file.append_vlen_strings("strs", &["a", "b", "c"]).unwrap();
3610            file.close().unwrap();
3611        }
3612        {
3613            // Append a 4th string -> partial-chunk write into chunk 0.
3614            let file = H5File::open_rw(&path).unwrap();
3615            file.append_vlen_strings("strs", &["d"]).unwrap();
3616            file.close().unwrap();
3617        }
3618        {
3619            let file = H5File::open(&path).unwrap();
3620            let ds = file.dataset("strs").unwrap();
3621            let got = ds.read_vlen_strings().unwrap();
3622            assert_eq!(
3623                got.iter().map(|s| s.as_str()).collect::<Vec<_>>(),
3624                vec!["a", "b", "c", "d"]
3625            );
3626        }
3627        std::fs::remove_file(&path).ok();
3628    }
3629
3630    #[test]
3631    fn fill_value_size_mismatch_errors() {
3632        let path = temp_path("fill_value_mismatch");
3633        let writer = crate::io::writer::Hdf5Writer::create(&path).unwrap();
3634        let dt = <f64 as crate::types::H5Type>::hdf5_type();
3635        let idx = writer.create_dataset("d", dt, &[4u64]).unwrap();
3636        // f64 element size is 8; a 4-byte fill value must be rejected.
3637        assert!(writer.set_dataset_fill_value(idx, vec![0u8; 4]).is_err());
3638        // The correct width succeeds.
3639        writer.set_dataset_fill_value(idx, vec![0u8; 8]).unwrap();
3640        writer.close().unwrap();
3641        std::fs::remove_file(&path).ok();
3642    }
3643
3644    #[test]
3645    fn datatype_exposes_class_sign_and_byteorder() {
3646        // The byte width alone cannot tell u8 from i8 (both 1 byte) or i32
3647        // from f32 (both 4 bytes). datatype() must report the real class and
3648        // signedness so a reader does not have to guess from element_size.
3649        use crate::format::messages::datatype::{ByteOrder, DatatypeMessage};
3650
3651        let path = temp_path("datatype_accessor");
3652        {
3653            let file = H5File::create(&path).unwrap();
3654            file.new_dataset::<u8>().shape([3]).create("u8d").unwrap();
3655            file.new_dataset::<i8>().shape([3]).create("i8d").unwrap();
3656            file.new_dataset::<i32>().shape([3]).create("i32d").unwrap();
3657            file.new_dataset::<f32>().shape([3]).create("f32d").unwrap();
3658            file.close().unwrap();
3659        }
3660
3661        let file = H5File::open(&path).unwrap();
3662
3663        match file.dataset("u8d").unwrap().datatype().unwrap() {
3664            DatatypeMessage::FixedPoint {
3665                size,
3666                signed,
3667                byte_order,
3668                ..
3669            } => {
3670                assert_eq!(size, 1);
3671                assert!(!signed, "u8 must be unsigned");
3672                assert_eq!(byte_order, ByteOrder::LittleEndian);
3673            }
3674            other => panic!("expected FixedPoint for u8, got {other:?}"),
3675        }
3676
3677        match file.dataset("i8d").unwrap().datatype().unwrap() {
3678            DatatypeMessage::FixedPoint { size, signed, .. } => {
3679                assert_eq!(size, 1);
3680                assert!(signed, "i8 must be signed");
3681            }
3682            other => panic!("expected FixedPoint for i8, got {other:?}"),
3683        }
3684
3685        match file.dataset("i32d").unwrap().datatype().unwrap() {
3686            DatatypeMessage::FixedPoint { size, signed, .. } => {
3687                assert_eq!(size, 4);
3688                assert!(signed, "i32 must be signed");
3689            }
3690            other => panic!("expected FixedPoint for i32, got {other:?}"),
3691        }
3692
3693        match file.dataset("f32d").unwrap().datatype().unwrap() {
3694            DatatypeMessage::FloatingPoint { size, .. } => assert_eq!(size, 4),
3695            other => panic!("expected FloatingPoint for f32, got {other:?}"),
3696        }
3697
3698        std::fs::remove_file(&path).ok();
3699    }
3700
3701    #[test]
3702    fn datatype_in_write_mode_errors() {
3703        let path = temp_path("datatype_write_mode");
3704        let file = H5File::create(&path).unwrap();
3705        let ds = file.new_dataset::<f32>().shape([4]).create("d").unwrap();
3706        assert!(ds.datatype().is_err());
3707        std::fs::remove_file(&path).ok();
3708    }
3709
3710    // --- write_chunk_raw (HDF5 direct chunk write) ---------------------------
3711
3712    /// Extensible-array path: pre-compress with the dataset's pipeline, write
3713    /// the bytes verbatim via write_chunk_raw (filter_mask = 0), and confirm
3714    /// the data round-trips through the reader unchanged.
3715    #[cfg(feature = "deflate")]
3716    #[test]
3717    fn write_chunk_raw_ea_roundtrip_mask0() {
3718        use crate::format::messages::filter::{apply_filters, FilterPipeline};
3719        let path = temp_path("wcr_ea_mask0");
3720        let original: Vec<i32> = (0..12).collect();
3721        {
3722            let file = H5File::create(&path).unwrap();
3723            let ds = file
3724                .new_dataset::<i32>()
3725                .shape([0])
3726                .chunk(&[4])
3727                .max_shape(&[None])
3728                .deflate(4)
3729                .create("v")
3730                .unwrap();
3731            assert!(ds.is_chunked());
3732            let pipeline = FilterPipeline::deflate(4);
3733            for c in 0..3usize {
3734                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
3735                    .iter()
3736                    .flat_map(|v| v.to_le_bytes())
3737                    .collect();
3738                let compressed = apply_filters(&pipeline, &raw).unwrap();
3739                ds.write_chunk_raw(c, &compressed, 0).unwrap();
3740            }
3741            ds.set_extent(&[12]).unwrap();
3742            file.close().unwrap();
3743        }
3744        {
3745            let file = H5File::open(&path).unwrap();
3746            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3747            assert_eq!(v, original);
3748        }
3749        std::fs::remove_file(&path).ok();
3750    }
3751
3752    /// Fixed-array path (all dimensions bounded): same verbatim write through
3753    /// the linear-index dispatch, round-tripped through the reader.
3754    #[cfg(feature = "deflate")]
3755    #[test]
3756    fn write_chunk_raw_fixed_array_roundtrip_mask0() {
3757        use crate::format::messages::filter::{apply_filters, FilterPipeline};
3758        let path = temp_path("wcr_fa_mask0");
3759        let original: Vec<i32> = (0..12).collect();
3760        {
3761            let file = H5File::create(&path).unwrap();
3762            let ds = file
3763                .new_dataset::<i32>()
3764                .shape([12])
3765                .chunk(&[4])
3766                .deflate(4)
3767                .create("v")
3768                .unwrap();
3769            assert!(ds.is_chunked());
3770            let pipeline = FilterPipeline::deflate(4);
3771            for c in 0..3usize {
3772                let raw: Vec<u8> = original[c * 4..c * 4 + 4]
3773                    .iter()
3774                    .flat_map(|v| v.to_le_bytes())
3775                    .collect();
3776                let compressed = apply_filters(&pipeline, &raw).unwrap();
3777                ds.write_chunk_raw(c, &compressed, 0).unwrap();
3778            }
3779            file.close().unwrap();
3780        }
3781        {
3782            let file = H5File::open(&path).unwrap();
3783            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3784            assert_eq!(v, original);
3785        }
3786        std::fs::remove_file(&path).ok();
3787    }
3788
3789    /// The caller-supplied filter_mask must reach the on-disk filtered index
3790    /// entry (not be hardcoded to 0). Store one chunk uncompressed in a
3791    /// filtered dataset with mask = 1 (deflate skipped), then reopen and decode
3792    /// the extensible-array filtered entry to read the mask back at the format
3793    /// level (independent of the data reader's mask handling).
3794    #[cfg(feature = "deflate")]
3795    #[test]
3796    fn write_chunk_raw_records_filter_mask() {
3797        let path = temp_path("wcr_records_mask");
3798        let raw: Vec<u8> = [10i32, 20, 30, 40]
3799            .iter()
3800            .flat_map(|v| v.to_le_bytes())
3801            .collect();
3802        assert_eq!(raw.len(), 16);
3803        {
3804            let file = H5File::create(&path).unwrap();
3805            let ds = file
3806                .new_dataset::<i32>()
3807                .shape([0])
3808                .chunk(&[4])
3809                .max_shape(&[None])
3810                .deflate(4)
3811                .create("v")
3812                .unwrap();
3813            // mask = 1: bit 0 set => filter 0 (deflate) was skipped, so the
3814            // chunk is stored uncompressed (its raw bytes).
3815            ds.write_chunk_raw(0, &raw, 1).unwrap();
3816            ds.set_extent(&[4]).unwrap();
3817            file.close().unwrap();
3818        }
3819        // Reopen the writer; open_append decodes the filtered index block from
3820        // disk, so the entry reflects exactly what was committed.
3821        {
3822            let w = crate::io::writer::Hdf5Writer::open_append(&path).unwrap();
3823            let idx = w.dataset_index("v").unwrap();
3824            let ds = w.ds(idx);
3825            let m = ds.lock();
3826            let entry = &m
3827                .chunked
3828                .as_ref()
3829                .unwrap()
3830                .filt_iblk
3831                .as_ref()
3832                .unwrap()
3833                .elements[0];
3834            assert_eq!(entry.filter_mask, 1, "filter_mask must round-trip to disk");
3835            assert_eq!(entry.nbytes, 16, "uncompressed chunk stored verbatim");
3836        }
3837        std::fs::remove_file(&path).ok();
3838    }
3839
3840    /// Reader honors a per-chunk filter_mask (EA): one chunk is stored
3841    /// compressed (mask 0), the next stored raw with deflate skipped (mask 1),
3842    /// in the same dataset. A correct reader skips deflate for chunk 1 only;
3843    /// ignoring the mask would feed raw bytes through inflate and corrupt them.
3844    #[cfg(feature = "deflate")]
3845    #[test]
3846    fn write_chunk_raw_ea_per_chunk_mask_roundtrip() {
3847        use crate::format::messages::filter::{apply_filters, FilterPipeline};
3848        let path = temp_path("wcr_ea_per_chunk_mask");
3849        let original: Vec<i32> = (0..8).collect();
3850        let pipeline = FilterPipeline::deflate(4);
3851        {
3852            let file = H5File::create(&path).unwrap();
3853            let ds = file
3854                .new_dataset::<i32>()
3855                .shape([0])
3856                .chunk(&[4])
3857                .max_shape(&[None])
3858                .deflate(4)
3859                .create("v")
3860                .unwrap();
3861            let raw0: Vec<u8> = original[0..4]
3862                .iter()
3863                .flat_map(|v| v.to_le_bytes())
3864                .collect();
3865            // chunk 0: compressed through the pipeline, mask 0.
3866            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
3867                .unwrap();
3868            let raw1: Vec<u8> = original[4..8]
3869                .iter()
3870                .flat_map(|v| v.to_le_bytes())
3871                .collect();
3872            // chunk 1: stored uncompressed, mask 1 (deflate skipped).
3873            ds.write_chunk_raw(1, &raw1, 1).unwrap();
3874            ds.set_extent(&[8]).unwrap();
3875            file.close().unwrap();
3876        }
3877        {
3878            let file = H5File::open(&path).unwrap();
3879            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3880            assert_eq!(v, original);
3881        }
3882        std::fs::remove_file(&path).ok();
3883    }
3884
3885    /// Reader honors a per-chunk filter_mask (fixed array): same mixed
3886    /// compressed/raw chunks as the EA case, through the fixed-array index.
3887    #[cfg(feature = "deflate")]
3888    #[test]
3889    fn write_chunk_raw_fixed_array_per_chunk_mask_roundtrip() {
3890        use crate::format::messages::filter::{apply_filters, FilterPipeline};
3891        let path = temp_path("wcr_fa_per_chunk_mask");
3892        let original: Vec<i32> = (0..8).collect();
3893        let pipeline = FilterPipeline::deflate(4);
3894        {
3895            let file = H5File::create(&path).unwrap();
3896            let ds = file
3897                .new_dataset::<i32>()
3898                .shape([8])
3899                .chunk(&[4])
3900                .deflate(4)
3901                .create("v")
3902                .unwrap();
3903            let raw0: Vec<u8> = original[0..4]
3904                .iter()
3905                .flat_map(|v| v.to_le_bytes())
3906                .collect();
3907            ds.write_chunk_raw(0, &apply_filters(&pipeline, &raw0).unwrap(), 0)
3908                .unwrap();
3909            let raw1: Vec<u8> = original[4..8]
3910                .iter()
3911                .flat_map(|v| v.to_le_bytes())
3912                .collect();
3913            ds.write_chunk_raw(1, &raw1, 1).unwrap();
3914            file.close().unwrap();
3915        }
3916        {
3917            let file = H5File::open(&path).unwrap();
3918            let v = file.dataset("v").unwrap().read_raw::<i32>().unwrap();
3919            assert_eq!(v, original);
3920        }
3921        std::fs::remove_file(&path).ok();
3922    }
3923
3924    /// An unfiltered chunk index has no slot for a stored size or mask, so a
3925    /// direct chunk write must be rejected rather than silently dropping them.
3926    #[test]
3927    fn write_chunk_raw_rejects_unfiltered() {
3928        let path = temp_path("wcr_unfiltered");
3929        let file = H5File::create(&path).unwrap();
3930        let ds = file
3931            .new_dataset::<i32>()
3932            .shape([0])
3933            .chunk(&[4])
3934            .max_shape(&[None])
3935            .create("v")
3936            .unwrap();
3937        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
3938        assert!(
3939            err.to_string().contains("filtered dataset"),
3940            "expected a filtered-dataset error, got: {err}"
3941        );
3942        std::fs::remove_file(&path).ok();
3943    }
3944
3945    /// v2-B-tree-indexed datasets (two or more unlimited dimensions) do not
3946    /// support direct chunk writes.
3947    #[test]
3948    fn write_chunk_raw_rejects_btree_v2() {
3949        let path = temp_path("wcr_btree2");
3950        let file = H5File::create(&path).unwrap();
3951        let ds = file
3952            .new_dataset::<i32>()
3953            .shape([0, 0])
3954            .chunk(&[2, 2])
3955            .max_shape(&[None, None])
3956            .create("grid")
3957            .unwrap();
3958        let err = ds.write_chunk_raw(0, &[0u8; 16], 0).unwrap_err();
3959        assert!(
3960            err.to_string().contains("v2-B-tree"),
3961            "expected a v2-B-tree rejection, got: {err}"
3962        );
3963        std::fs::remove_file(&path).ok();
3964    }
3965
3966    /// A stored size that does not fit the index's chunk-size field must error
3967    /// (libhdf5 H5D_CHUNK_ENCODE_SIZE_CHECK) instead of truncating silently.
3968    /// A 4-byte chunk (chunk[1] of i32) has chunk_size_len = 2 (max 65535), so
3969    /// a 70000-byte stored chunk overflows it.
3970    #[cfg(feature = "deflate")]
3971    #[test]
3972    fn write_chunk_raw_rejects_oversized_chunk() {
3973        let path = temp_path("wcr_oversized");
3974        let file = H5File::create(&path).unwrap();
3975        let ds = file
3976            .new_dataset::<i32>()
3977            .shape([0])
3978            .chunk(&[1])
3979            .max_shape(&[None])
3980            .deflate(4)
3981            .create("v")
3982            .unwrap();
3983        let err = ds.write_chunk_raw(0, &vec![0u8; 70000], 0).unwrap_err();
3984        assert!(
3985            err.to_string().contains("does not fit"),
3986            "expected a chunk-size-field overflow error, got: {err}"
3987        );
3988        std::fs::remove_file(&path).ok();
3989    }
3990}