Skip to main content

rust_hdf5/io/
reader.rs

1//! HDF5 file reader.
2//!
3//! Opens an HDF5 file, parses the superblock and root group, and provides
4//! access to dataset metadata and raw data.
5//!
6//! Supports both legacy (v0/v1 superblock, v1 object headers, symbol tables)
7//! and modern (v2/v3 superblock, v2 object headers, link messages) formats.
8
9use std::path::{Path, PathBuf};
10
11use crate::dataset::{DatasetAccess, VirtualView};
12use crate::format::btree_v1::{BTreeV1Config, BTreeV1Node, ChunkBTreeV1Node};
13use crate::format::bytes::read_le_uint as read_uint;
14use crate::format::creation_order::CreationOrder;
15use crate::format::fractal_heap::{self, FractalHeapHeader};
16use crate::format::global_heap::{
17    decode_vlen_reference, vlen_reference_size, GlobalHeapCollection,
18};
19use crate::format::local_heap::{local_heap_get_string, LocalHeapHeader};
20use crate::format::messages::attr_info::AttributeInfoMessage;
21use crate::format::messages::attribute::{AttributeEntry, AttributeMessage};
22use crate::format::messages::data_layout::{self, DataLayoutMessage};
23use crate::format::messages::dataspace::DataspaceMessage;
24use crate::format::messages::datatype::{DatatypeMessage, OldReferenceKind, ReferenceEncoding};
25use crate::format::messages::external_file_list::ExternalFileListMessage;
26use crate::format::messages::fill_value::{
27    try_tiled_fill, FillValueMessage, ALLOC_TIME_LATE, FILL_TIME_IFSET,
28};
29use crate::format::messages::filter::{self, FilterPipeline};
30use crate::format::messages::link::LinkMessage;
31use crate::format::messages::link::LinkTarget;
32use crate::format::messages::link_info::LinkInfoMessage;
33use crate::format::messages::shared::{MessageStorage, MSG_FLAG_SHARED};
34use crate::format::messages::superblock_ext::{
35    BtreeKMessage, DriverInfoMessage, FileSpaceInfoMessage, SharedMessageTableMessage,
36};
37use crate::format::messages::virtual_mapping::{
38    parse_source_name, VirtualMapping, VirtualMappingList,
39};
40use crate::format::messages::*;
41use crate::format::object_header::ObjectHeader;
42use crate::format::reference::{
43    decode_object_element, decode_region_element, decode_region_heap_object, decode_revised_body,
44    decode_revised_element, DecodedReference, Reference, ReferenceTarget, RevisedElement,
45};
46use crate::format::selection::{
47    Hyperslab, PointSelection, RegularHyperslab, ResolvedSelection, Selection,
48};
49use crate::format::sohm::SohmMasterTable;
50use crate::format::storage_kind::{AttributeStorage, LinkStorage};
51use crate::format::superblock::{
52    detect_superblock_version, SuperblockV0V1, SuperblockV2V3, SymbolTableCache,
53};
54use crate::format::symbol_table::SymbolTableNode;
55use crate::format::{BlockReader, FormatContext, UNDEF_ADDR};
56
57use crate::format::selection::check_hyperslab;
58use crate::io::file_handle::{FileHandle, ReadDst};
59use crate::io::hyperslab::{compute_strides, for_each_contiguous_run};
60use crate::io::locking::FileLocking;
61use crate::io::{FileMeta, IoResult};
62
63/// The version-4 chunk-index descriptor pulled from a data-layout message:
64/// the index kind, its address, and the per-kind parameters the reader needs
65/// to walk it. Bundled so the chunked read entry point takes one descriptor
66/// instead of a long parameter list.
67struct ChunkIndexDesc<'a> {
68    /// The kind of chunk index (single chunk, fixed/extensible array, …).
69    index_type: data_layout::ChunkIndexType,
70    /// Address of the chunk index structure (or the chunk itself, for a
71    /// single chunk). `UNDEF_ADDR` when unallocated.
72    index_address: u64,
73    /// Extensible-array parameters (present iff `index_type == ExtensibleArray`).
74    earray_params: Option<&'a data_layout::EarrayParams>,
75    /// Filtered single-chunk parameters (present iff `index_type ==
76    /// SingleChunk` and the layout's filtered flag is set): the chunk's exact
77    /// on-disk size and per-chunk filter mask.
78    single_chunk_filter: Option<data_layout::SingleChunkFilter>,
79}
80
81/// One v2-B-tree chunk-index record, resolved to `(chunk address, on-disk
82/// read size, scaled chunk-grid offsets, filter mask)` — see
83/// [`Hdf5Reader::collect_bt2_chunk_entries`].
84type Bt2ChunkEntry = (u64, usize, Vec<u64>, u32);
85
86/// What a chunked read should produce: the whole dataset, or one hyperslab.
87///
88/// Threaded through every chunked reader so the index walk, raw read, and
89/// filter pipeline are shared between full reads and slice reads. For a
90/// `Slice`, the reader allocates a `counts`-shaped buffer, skips reading any
91/// chunk that does not overlap the selection (the I/O win), and scatters only
92/// the chunk∩selection intersection. `Full` reads and places every chunk.
93#[derive(Clone, Copy)]
94enum ChunkTarget<'a> {
95    Full,
96    Slice {
97        starts: &'a [u64],
98        counts: &'a [u64],
99    },
100}
101
102impl<'a> ChunkTarget<'a> {
103    /// Whether a chunk at chunk-grid `coords` (extent `chunk_dims`) intersects
104    /// the target. `Full` always intersects; a `Slice` intersects iff every
105    /// dimension's chunk span `[origin, origin+chunk_dims)` overlaps the
106    /// selection span `[start, start+count)`.
107    fn overlaps(&self, coords: &[u64], chunk_dims: &[u64]) -> bool {
108        match self {
109            ChunkTarget::Full => true,
110            ChunkTarget::Slice { starts, counts } => coords.iter().enumerate().all(|(d, &c)| {
111                let origin = c.saturating_mul(chunk_dims[d]);
112                let chunk_end = origin.saturating_add(chunk_dims[d]);
113                let sel_end = starts[d].saturating_add(counts[d]);
114                origin < sel_end && starts[d] < chunk_end
115            }),
116        }
117    }
118}
119
120/// What one chunked read wants of every chunk, apart from which dataset it is
121/// reading: the filter pipeline the chunks were stored through, the box of the
122/// dataset it fills, and the fill value every byte no chunk covers takes.
123///
124/// Carried as one value because all four are constant across a read and are
125/// threaded unchanged from the entry point through each index type down to
126/// [`place_chunk_jobs`] — so a new index reader cannot pick up the target and
127/// forget the fill or the destination fact.
128#[derive(Clone, Copy)]
129struct ChunkReadRequest<'a> {
130    pipeline: Option<&'a FilterPipeline>,
131    target: ChunkTarget<'a>,
132    fill_value: Option<&'a [u8]>,
133    /// The caller's destination fact ([`ReadDst`]) for the output buffer,
134    /// which prices the per-run direct reads that land straight in it.
135    /// Whole-chunk reads into a fresh image are unaffected.
136    dst: ReadDst,
137}
138
139/// The dataset extent, chunk shape and element size a chunk-placement call
140/// reads through — constant across every chunk of one read, so
141/// [`for_each_chunk_run`] and the two sinks built on it take this once instead
142/// of the three fields separately, leaving only what actually varies per chunk
143/// (its data and grid coordinates) as their own parameters.
144#[derive(Clone, Copy)]
145struct ChunkOutputGeometry<'a> {
146    dims: &'a [u64],
147    chunk_dims: &'a [u64],
148    element_size: u64,
149}
150
151/// One chunked read's placement constants: the geometry above plus the box in
152/// dataset coordinates the read fills.
153///
154/// A full read fills the whole extent, which is the box `starts = 0`,
155/// `counts = dims` describes, so resolving the target once
156/// ([`ChunkPlacement::resolve`]) leaves the placement code below with one
157/// shape instead of a full-read case and a slice case.
158#[derive(Clone, Copy)]
159struct ChunkPlacement<'a> {
160    geo: ChunkOutputGeometry<'a>,
161    starts: &'a [u64],
162    counts: &'a [u64],
163}
164
165impl ChunkOutputGeometry<'_> {
166    /// Bytes one whole chunk's image holds — the product of the chunk shape,
167    /// element-wide. `None` only for geometry a corrupt file can carry: a
168    /// shape whose product overflows, or an empty image.
169    ///
170    /// The size the layout says a decoded chunk is. Both the destination a
171    /// chunk decodes into and the buffer a staged chunk decodes into come from
172    /// here, so neither has to discover it by growing.
173    fn image_bytes(&self) -> Option<u64> {
174        self.chunk_dims
175            .iter()
176            .copied()
177            .try_fold(1u64, |a, d| a.checked_mul(d))
178            .and_then(|elems| elems.checked_mul(self.element_size))
179            .filter(|b| *b > 0)
180    }
181
182    /// Bytes the chunk at `coords` holds that the dataset extent reaches:
183    /// [`image_bytes`](Self::image_bytes), except at an edge where the extent
184    /// cuts the chunk short. A read covering all of them has taken everything
185    /// that chunk has to give, which is the same clip
186    /// [`ChunkOverlap::of`] applies.
187    fn resident_bytes(&self, coords: &[u64]) -> u64 {
188        let mut elems = 1u64;
189        for (d, &c) in coords.iter().enumerate().take(self.dims.len()) {
190            let origin = c.saturating_mul(self.chunk_dims[d]);
191            let end = origin.saturating_add(self.chunk_dims[d]).min(self.dims[d]);
192            elems = elems.saturating_mul(end.saturating_sub(origin));
193        }
194        elems.saturating_mul(self.element_size)
195    }
196}
197
198impl<'a> ChunkPlacement<'a> {
199    /// Resolve a read target against the geometry it reads through. `zeros`
200    /// lends a full read the origin it fills from and does not carry itself.
201    fn resolve(geo: &ChunkOutputGeometry<'a>, target: ChunkTarget<'a>, zeros: &'a [u64]) -> Self {
202        let (starts, counts) = match target {
203            ChunkTarget::Full => (zeros, geo.dims),
204            ChunkTarget::Slice { starts, counts } => (starts, counts),
205        };
206        ChunkPlacement {
207            geo: *geo,
208            starts,
209            counts,
210        }
211    }
212
213    /// Whether this read leaves part of the chunk at `coords` untouched, so a
214    /// later read can still want the rest — which is what makes that chunk's
215    /// decoded image worth keeping.
216    ///
217    /// What the chunk has to give is the chunk clipped to the dataset extent,
218    /// the same clip [`ChunkOverlap::of`] applies and not the chunk image: an
219    /// edge chunk the extent cuts short is fully consumed by a read that takes
220    /// what is inside the extent. So a whole-dataset read leaves nothing,
221    /// whatever the extent does at the edges.
222    fn leaves_chunk_unconsumed(&self, coords: &[u64]) -> bool {
223        match ChunkOverlap::of(self, coords) {
224            Some(overlap) => overlap.bytes(self.geo.element_size) < self.geo.resident_bytes(coords),
225            None => false,
226        }
227    }
228}
229
230/// One resolved external-file slot (H5O_EFL_ID): the on-disk message
231/// stores each slot's name as an offset into a local heap, so this is that
232/// slot after the heap lookup, in the order the dataset's logical byte
233/// range concatenates them.
234#[derive(Debug, Clone, PartialEq, Eq)]
235pub struct ExternalFileSegment {
236    /// The external file's name, exactly as stored — relative names are
237    /// resolved against `HDF5_EXTFILE_PREFIX` at read time, not here.
238    pub name: String,
239    /// Byte offset within the named file where this slot's reserved
240    /// region begins.
241    pub offset: u64,
242    /// Bytes reserved for this slot. `u64::MAX` (`H5O_EFL_UNLIMITED`)
243    /// marks the last slot as unlimited/growable.
244    pub size: u64,
245}
246
247/// Where a dataset's stored image lies, as far as a zero-copy view of it is
248/// concerned.
249///
250/// [`Hdf5Reader::dataset_view_source`] is the only producer; it reports what
251/// the layout message says and judges nothing.
252#[cfg(feature = "mmap")]
253pub(crate) enum ViewStorage {
254    /// One stretch of this file: `len` bytes at absolute file offset
255    /// `offset`, with the userblock already added, so it indexes the map
256    /// directly.
257    Contiguous { offset: u64, len: u64 },
258    /// No storage is allocated. Every element reads as the fill value, which
259    /// is a property of the header, not bytes anywhere in the file.
260    Unallocated,
261    /// The image is not one stretch of this file. The phrase says what it is
262    /// instead, and lands verbatim in the refusal.
263    Elsewhere(&'static str),
264}
265
266/// Everything a zero-copy view of one dataset rests on, gathered from the
267/// file that owns it.
268#[cfg(feature = "mmap")]
269pub(crate) struct DatasetViewSource {
270    /// The owning file's whole-file map, or `None` when that file is read
271    /// through `pread`.
272    pub map: Option<std::sync::Arc<crate::io::file_handle::LockedMap>>,
273    /// Where the dataset's image lies in that file.
274    pub storage: ViewStorage,
275    /// How one element is stored, which decides whether the stored bytes are
276    /// already the host image of a `T`.
277    pub datatype: DatatypeMessage,
278    /// The dataset's extent, for turning a requested range into a byte run.
279    pub dims: Vec<u64>,
280}
281
282/// Read-side metadata for a single dataset.
283pub struct DatasetReadInfo {
284    /// Dataset name (the link name in the root group).
285    pub name: String,
286    /// Address of the dataset's object header — what an object reference to
287    /// this dataset stores.
288    pub object_header_address: u64,
289    /// Element datatype.
290    pub datatype: DatatypeMessage,
291    /// Dataspace (dimensionality).
292    pub dataspace: DataspaceMessage,
293    /// Data layout (contiguous, compact, or chunked).
294    pub layout: DataLayoutMessage,
295    /// Filter pipeline for compressed chunks (None = uncompressed).
296    pub filter_pipeline: Option<FilterPipeline>,
297    /// Attributes attached to this dataset.
298    pub attributes: ObjectAttributes,
299    /// User-defined fill value bytes (one element wide), decoded from the
300    /// fill-value message when `fill_defined == 2`. `None` => default
301    /// zero-fill. Applied to unallocated chunks and unwritten regions.
302    pub fill_value: Option<Vec<u8>>,
303    /// The fill-value message's own definedness byte
304    /// (`H5D_fill_value_t`/`H5Pfill_value_defined`): 0 = explicitly
305    /// undefined (no fill is ever performed), 1 = default (zero-fill, no
306    /// value stored), 2 = user-defined (`fill_value` carries the bytes). A
307    /// dataset with no fill-value message at all reads as 1, matching a
308    /// fresh dataset creation property list (`FillValueMessage::default`).
309    pub fill_defined: u8,
310    /// The fill-value message's write-time byte (`H5D_fill_time_t`): 0 =
311    /// `H5D_FILL_TIME_ALLOC`, 1 = `H5D_FILL_TIME_NEVER`, 2 =
312    /// `H5D_FILL_TIME_IFSET`. A dataset with no fill-value message at all
313    /// reads as 2, `H5D_CRT_FILL_TIME_DEF` — the same "no message" default
314    /// [`fill_defined`](Self::fill_defined) uses.
315    pub fill_write_time: u8,
316    /// The fill-value message's space-allocation-time byte
317    /// (`H5D_alloc_time_t`): 1 = `H5D_ALLOC_TIME_EARLY`, 2 =
318    /// `H5D_ALLOC_TIME_LATE`, 3 = `H5D_ALLOC_TIME_INCR`. A dataset with no
319    /// fill-value message at all reads as `ALLOC_TIME_LATE`, matching
320    /// [`FillValueMessage::default`]'s "no message" convention.
321    pub alloc_time: u8,
322    /// External raw-data segments (H5O_EFL_ID). Non-empty only when this
323    /// dataset's storage is an External Data Files list instead of a
324    /// normal contiguous block — `layout` still reports `Contiguous` with
325    /// an undefined address in that case (H5Dlayout.c overrides the
326    /// layout's storage ops whenever this message is present).
327    pub external_files: Vec<ExternalFileSegment>,
328    /// Virtual dataset source/virtual mappings (H5D_VIRTUAL), resolved from
329    /// the global heap object `layout`'s `Virtual` variant points at.
330    /// `Some` only when `layout` is `DataLayoutMessage::Virtual` and it
331    /// names a mapping list (`heap_index != 0`); `None` for every other
332    /// layout, and for a virtual dataset that has no mappings yet.
333    pub virtual_mappings: Option<VirtualMappingList>,
334    /// What each mapping in [`virtual_mappings`](Self::virtual_mappings)
335    /// resolved to when the file was opened, in the same order — see
336    /// [`MappingResolution`] and `Hdf5Reader::resolve_virtual_extents`.
337    /// `None` for every non-virtual dataset and for a virtual one with no
338    /// mapping list.
339    pub virtual_resolution: Option<Vec<MappingResolution>>,
340    /// The extent this dataset's dataspace message stores, kept because
341    /// `dataspace.dims` holds the extent the *sources* gave it once
342    /// `resolve_virtual_extents` has run. `H5D__virtual_set_extent_unlim`
343    /// resolves from the space freshly loaded from the object header on
344    /// every `H5Dopen` (H5Dvirtual.c:1386), so a later open under different
345    /// [`DatasetAccess`] must resolve from this, not from its own last
346    /// answer.
347    ///
348    /// `Some` exactly for a virtual dataset whose extent has been resolved;
349    /// `None` for every dataset whose extent is simply its stored one.
350    pub virtual_stored_dims: Option<Vec<u64>>,
351}
352
353/// What one virtual-dataset mapping resolved to at open time —
354/// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c), which libhdf5 runs when
355/// the dataset is opened and which both the dataset's reported extent and
356/// every read of it depend on.
357#[derive(Debug, Clone, PartialEq, Eq)]
358pub enum MappingResolution {
359    /// Neither selection grows: the mapping is already concrete, and the
360    /// dataset's stored extent is its extent.
361    Bounded,
362    /// Both selections are unlimited (`unlim_dim_virtual >= 0` and
363    /// `unlim_dim_source >= 0`). `virtual_clip` is how far this mapping
364    /// reaches in its unlimited virtual dimension, `source_clip` the source
365    /// dataset's own extent in its unlimited source dimension — the two
366    /// values `H5S_hyper_clip_unlim` clips the mapping's selections to.
367    Unlimited { virtual_clip: u64, source_clip: u64 },
368    /// A printf mapping (unlimited virtual selection, limited source
369    /// selection, `%b` in a source name).
370    ///
371    /// `blocks` is upstream's `first_missing`: the scan stops at the first
372    /// block whose source is absent and then looks
373    /// [`DatasetAccess::virtual_printf_gap`] blocks further, so blocks
374    /// `0..blocks` are the ones the extent covers. `present` lists which of
375    /// them actually have a source — with a non-zero gap the others are
376    /// inside the extent but read as the fill value.
377    Printf { blocks: u64, present: Vec<u64> },
378}
379
380/// The class of one link record in a group: what `H5Lget_info` reports,
381/// carrying the value `H5Lget_val` returns for the classes that have one.
382///
383/// Every link a group holds gets one of these, whether or not the object it
384/// names can be opened — a listing is a listing of links, not of objects.
385#[derive(Debug, Clone, PartialEq, Eq)]
386pub enum LinkClass {
387    /// Another name for an object in this file.
388    Hard,
389    /// A path inside this file, resolved when the link is traversed.
390    Soft { path: String },
391    /// A path inside another file. Listed and reported, but not followed.
392    External { file: String, path: String },
393    /// A user-defined link class this reader has no interpreter for; libhdf5
394    /// needs a registered link class for these too.
395    UserDefined { link_type: u8 },
396}
397
398impl LinkClass {
399    pub(crate) fn from_target(target: &LinkTarget) -> Self {
400        match target {
401            LinkTarget::Hard { .. } => Self::Hard,
402            LinkTarget::Soft { target } => Self::Soft {
403                path: target.clone(),
404            },
405            LinkTarget::External { file, path } => Self::External {
406                file: file.clone(),
407                path: path.clone(),
408            },
409            LinkTarget::UserDefined { link_type, .. } => Self::UserDefined {
410                link_type: *link_type,
411            },
412        }
413    }
414}
415
416/// The soft link a traversal crossed, kept so a lookup that finds nothing can
417/// say the link dangles instead of reporting a bare absence.
418struct SoftLinkRef {
419    link: String,
420    target: String,
421}
422
423/// Where a path leaves this file: the external link it crosses, the file that
424/// link names, and the remainder of the path inside that file.
425pub(crate) struct ExternalEdge {
426    pub link: String,
427    pub file: String,
428    pub path: String,
429}
430
431impl ExternalEdge {
432    /// The error for "the link resolved to a file, but the object it names is
433    /// not in it" — the external counterpart of a dangling soft link.
434    fn dangling(&self) -> crate::io::IoError {
435        crate::io::IoError::DanglingLink {
436            link: self.link.clone(),
437            target: format!("{}::{}", self.file, self.path),
438        }
439    }
440}
441
442/// How a reader was opened, carried from the entry point to the per-superblock
443/// constructors: both facts an external link needs later (where to resolve a
444/// relative target from, and under which locking policy to open it) are fixed
445/// at open time and belong together.
446struct Origin {
447    path: PathBuf,
448    locking: crate::io::locking::FileLocking,
449}
450
451/// Bound on how many external links one path resolution may cross, matching
452/// libhdf5's `H5L_NUM_LINKS` (the `H5Pset_nlinks` default). Two files that
453/// link to each other form a cycle whose every hop opens a fresh target, so it
454/// is this count, not target identity, that terminates the walk.
455const MAX_EXTERNAL_HOPS: usize = 16;
456
457/// What path traversal produced.
458enum Traversal {
459    /// A path in this file, after every group hard-link alias and soft link
460    /// on it was followed. `via` names the last soft link crossed, if any.
461    Path {
462        path: String,
463        via: Option<SoftLinkRef>,
464    },
465    /// A component of the path is an external link, so the path leaves this
466    /// file. `path` is the remainder inside `file`.
467    External {
468        link: String,
469        file: String,
470        path: String,
471    },
472}
473
474/// One rewrite a traversal step can apply to the path prefix it matched.
475#[derive(Clone, Copy)]
476enum Rewrite<'a> {
477    /// A group hard link: continue from the group's first-walked path.
478    Alias(&'a str),
479    /// A soft link: continue from its value.
480    Soft(&'a str),
481    /// An external link: stop, the rest of the path is in another file.
482    External { file: &'a str, path: &'a str },
483}
484
485/// Resolve a soft link's value against the group the link lives in, the way
486/// `H5G_traverse` does: a value starting with `/` is absolute, anything else
487/// is relative to that group. `.` and `..` components fold. The result has no
488/// leading `/`.
489fn resolve_link_value(link_path: &str, value: &str) -> String {
490    let mut components: Vec<&str> = Vec::new();
491    if !value.starts_with('/') {
492        // The link's own parent group is everything before its last component.
493        if let Some(parent) = link_path.rsplit_once('/').map(|(p, _)| p) {
494            components.extend(parent.split('/').filter(|c| !c.is_empty()));
495        }
496    }
497    for component in value.split('/') {
498        match component {
499            "" | "." => {}
500            ".." => {
501                components.pop();
502            }
503            c => components.push(c),
504        }
505    }
506    components.join("/")
507}
508
509/// Everything one discovery walk found: the objects, the link records that
510/// name them, and the group metadata the lookup paths need. Carried as one
511/// value so the walk has a single owner rather than a widening tuple, and so
512/// every walk (link-message and symbol-table alike) fills the same fields.
513#[derive(Default)]
514struct Catalog {
515    datasets: Vec<DatasetReadInfo>,
516    /// Dataset-shaped objects this crate cannot read, keyed by path; the
517    /// value names what stopped it. They are listed exactly like readable
518    /// datasets — the name is in the file either way — and refuse typed
519    /// access with that reason.
520    unreadable: std::collections::BTreeMap<String, String>,
521    /// Attributes on non-root groups, keyed by group path.
522    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
523    /// Link storage kind and link creation-order policy of non-root groups,
524    /// keyed by group path.
525    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
526    /// Every non-root group path the walk traversed into.
527    group_paths: std::collections::BTreeSet<String>,
528    /// Group object header address → the first path that reached it (no
529    /// leading `/`), taken from the walk's cycle guard. What turns the
530    /// address an object reference stores back into a name.
531    group_object_paths: std::collections::HashMap<u64, String>,
532    /// Group hard-link aliases: alias path → first-walked path.
533    group_aliases: std::collections::HashMap<String, String>,
534    /// Every link record seen, keyed by its full path (no leading `/`).
535    links: std::collections::BTreeMap<String, LinkClass>,
536    /// Committed (named) datatype objects, keyed by path.
537    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
538    /// Committed datatype object header address → its path (no leading `/`).
539    /// The third object kind needs the same address→name entry groups and
540    /// datasets have, or a path that names one resolves to nothing.
541    datatype_object_paths: std::collections::HashMap<u64, String>,
542}
543
544impl Catalog {
545    /// The address→absolute-path catalog this walk implies, with `root_addr`
546    /// named `/` whether or not the walk itself reached it.
547    ///
548    /// The single owner of the catalog: file open and the SWMR
549    /// [`Hdf5Reader::refresh`] rescan both build it here, so a dataset that
550    /// appears after open is resolvable exactly as one present at open is.
551    fn object_paths(&self, root_addr: u64) -> std::collections::HashMap<u64, String> {
552        let mut paths = std::collections::HashMap::new();
553        paths.insert(root_addr, "/".to_string());
554        for (addr, path) in &self.group_object_paths {
555            paths.insert(*addr, absolute_path(path));
556        }
557        for ds in &self.datasets {
558            paths.insert(ds.object_header_address, absolute_path(&ds.name));
559        }
560        for (addr, path) in &self.datatype_object_paths {
561            paths.insert(*addr, absolute_path(path));
562        }
563        paths
564    }
565}
566
567/// The state one catalog walk threads through every group it visits.
568///
569/// A group stores its children either as `Link` messages in its object header
570/// (with dense overflow in a fractal heap) or in the legacy symbol-table
571/// B-tree plus local heap — and *which* it uses is a property of that group
572/// alone. One file mixes the two freely: writing a single link that the old
573/// format cannot express (an external link, a creation-order-tracked group)
574/// migrates just that group, leaving its parent and its children where they
575/// were. Walking a group in the format its *parent* used therefore finds no
576/// children at all and reports an empty group, which is why the two storages
577/// share one walker here: [`CatalogWalk::group`] asks each object header what
578/// it declares, and [`CatalogWalk::child`] is the single place a child is
579/// classified, recorded and descended into.
580struct CatalogWalk<'a> {
581    handle: &'a mut FileHandle,
582    meta: &'a FileMeta,
583    catalog: Catalog,
584    /// Object headers already descended into, keyed to the first path that
585    /// reached them: a later path to the same header is a group hard link,
586    /// recorded in `group_aliases` so lookups resolve through it instead of
587    /// walking (and cycling) a second time.
588    visited: std::collections::HashMap<u64, String>,
589}
590
591impl<'a> CatalogWalk<'a> {
592    /// Bound group nesting on a hostile or corrupt file.
593    const MAX_DEPTH: usize = 256;
594
595    /// Start a walk at the root group's object header address, seeded so a
596    /// hard link cycling back to the root is not descended into again.
597    fn new(handle: &'a mut FileHandle, meta: &'a FileMeta, root_addr: u64) -> Self {
598        let mut visited = std::collections::HashMap::new();
599        visited.insert(root_addr, String::new());
600        Self {
601            handle,
602            meta,
603            catalog: Catalog::default(),
604            visited,
605        }
606    }
607
608    /// The address/length widths, which most of the walk needs and `meta`
609    /// carries alongside the file's B-tree ranks and shared-message table.
610    fn ctx(&self) -> &FormatContext {
611        &self.meta.ctx
612    }
613
614    fn finish(mut self) -> Catalog {
615        self.catalog.group_object_paths = self.visited;
616        self.catalog
617    }
618
619    /// Enumerate one group's children, choosing the storage from what this
620    /// group's own header declares.
621    ///
622    /// `stab` is the symbol-table scratch-pad copy from the entry that named
623    /// this group (the superblock's root entry, or the parent's symbol-table
624    /// entry), which is the only source of those addresses when the header
625    /// itself did not decode. `header` is `None` in exactly that case.
626    fn group(
627        &mut self,
628        header: Option<&ObjectHeader>,
629        prefix: &str,
630        depth: usize,
631        stab: Option<(u64, u64)>,
632    ) -> IoResult<()> {
633        if depth > Self::MAX_DEPTH {
634            return Ok(());
635        }
636        let link_storage = header.filter(|h| header_declares_link_storage(h));
637        if let Some(h) = link_storage {
638            return self.links(h, prefix, depth);
639        }
640        // Symbol-table storage: the scratch-pad copy wins when it is set,
641        // otherwise the addresses come from the group's own `stab` message.
642        let (btree_addr, heap_addr) = match stab {
643            Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
644            _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
645                Hdf5Reader::stab_from_header(h, self.ctx())
646            }),
647        };
648        if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
649            self.btree(btree_addr, heap_addr, prefix, depth)?;
650        }
651        Ok(())
652    }
653
654    /// Enumerate a group that stores its children as `Link` messages.
655    fn links(&mut self, header: &ObjectHeader, prefix: &str, depth: usize) -> IoResult<()> {
656        // Collect every link in this group: inline `Link` messages plus, for
657        // groups using dense storage, links held in a fractal heap referenced
658        // by the `Link Info` message.
659        //
660        // A link message that does not decode has no name to report the
661        // failure against, and a `Link Info` message that does not decode
662        // hides a whole group's dense storage. Either way the listing would
663        // come back silently short, so both are errors: a listing this
664        // reader cannot complete must not present itself as complete.
665        let mut links: Vec<LinkMessage> = Vec::new();
666        for msg in &header.messages {
667            if msg.msg_type == MSG_LINK {
668                let (link, _) = LinkMessage::decode(&msg.data, self.ctx())?;
669                links.push(link);
670            } else if msg.msg_type == MSG_LINK_INFO {
671                let (info, _) = LinkInfoMessage::decode(&msg.data, self.ctx())?;
672                if info.fractal_heap_address != UNDEF_ADDR {
673                    let ctx = self.meta.ctx;
674                    let dense =
675                        Hdf5Reader::read_dense_links(self.handle, &ctx, info.fractal_heap_address)?;
676                    links.extend(dense);
677                }
678            }
679        }
680
681        for link in &links {
682            let full_name = join_path(prefix, &link.name);
683            // Every link is a listing entry whatever it points at; only a
684            // hard link names an object in this file to descend into.
685            self.catalog
686                .links
687                .insert(full_name.clone(), LinkClass::from_target(&link.target));
688            let LinkTarget::Hard { address } = &link.target else {
689                continue;
690            };
691            self.child(full_name, *address, depth, None)?;
692        }
693        Ok(())
694    }
695
696    /// Enumerate a group that stores its children in a symbol-table B-tree
697    /// plus local heap.
698    fn btree(
699        &mut self,
700        btree_addr: u64,
701        heap_addr: u64,
702        prefix: &str,
703        depth: usize,
704    ) -> IoResult<()> {
705        let sa = self.ctx().sizeof_addr as usize;
706        let ss = self.ctx().sizeof_size as usize;
707
708        // Read the local heap header + data for this group.
709        let heap_hdr_buf = self.handle.read_at_most(heap_addr, 64)?;
710        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
711        let heap_data = self
712            .handle
713            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;
714
715        // Collect all SNOD addresses by walking the B-tree.
716        let mut snod_tree_visited = std::collections::HashSet::new();
717        let snod_addrs = Hdf5Reader::collect_snod_addresses(
718            self.handle,
719            self.meta,
720            btree_addr,
721            0,
722            &mut snod_tree_visited,
723        )?;
724
725        // A symbol-table node is a fixed-size record sized by `sym_leaf_k`,
726        // which the superblock extension may have overridden.
727        let snod_size = self.meta.btree.symbol_table_node_size(sa, ss);
728        for snod_addr in snod_addrs {
729            let snod_buf = self.handle.read_at_most(snod_addr, snod_size)?;
730            let snod =
731                SymbolTableNode::decode(&snod_buf, sa, ss, self.meta.btree.sym_leaf_max_entries())?;
732
733            for entry in &snod.entries {
734                let name = local_heap_get_string(&heap_data, entry.name_offset)?;
735                // Skip empty names (root group self-reference).
736                if name.is_empty() {
737                    continue;
738                }
739                let full_name = join_path(prefix, &name);
740
741                // A `H5G_CACHED_SLINK` entry is a soft link: it names no
742                // object at all, and its value string lives in this group's
743                // local heap. Record the link and move on — reading its
744                // undefined object-header address is what used to drop it.
745                if let SymbolTableCache::SoftLink { value_offset } = entry.cache {
746                    let target = local_heap_get_string(&heap_data, value_offset as u64)?;
747                    self.catalog
748                        .links
749                        .insert(full_name, LinkClass::Soft { path: target });
750                    continue;
751                }
752                self.catalog
753                    .links
754                    .insert(full_name.clone(), LinkClass::Hard);
755                self.child(
756                    full_name,
757                    entry.obj_header_addr,
758                    depth,
759                    entry.cached_symbol_table(),
760                )?;
761            }
762        }
763
764        Ok(())
765    }
766
767    /// Record one child of a group and, when it is itself a group, descend.
768    ///
769    /// Both storages end here, so a child is classified, catalogued and
770    /// cycle-guarded the same way whichever way its name was found.
771    fn child(
772        &mut self,
773        full_name: String,
774        addr: u64,
775        depth: usize,
776        stab: Option<(u64, u64)>,
777    ) -> IoResult<()> {
778        // The entry names an object, so the object is in the listing whatever
779        // comes of reading it: a header that does not decode (a stale link
780        // left by a deletion, say) is reported against this name, never
781        // dropped from it.
782        let header = match Hdf5Reader::read_object_header_full(self.handle, self.meta, addr) {
783            Ok(h) => h,
784            Err(e) => {
785                self.catalog
786                    .unreadable
787                    .insert(full_name, format!("its object header does not decode: {e}"));
788                return Ok(());
789            }
790        };
791        match Hdf5Reader::classify_object(self.handle, &header, self.meta, &full_name, addr) {
792            ObjectKind::Dataset(info) => {
793                self.catalog.datasets.push(*info);
794                return Ok(());
795            }
796            ObjectKind::UnreadableDataset(why) => {
797                self.catalog.unreadable.insert(full_name, why);
798                return Ok(());
799            }
800            // A committed (named) datatype is neither a group nor a dataset,
801            // so it must not be recorded as either; the link record above
802            // already carries its name.
803            ObjectKind::CommittedDatatype(info) => {
804                self.catalog
805                    .datatype_object_paths
806                    .insert(addr, full_name.clone());
807                self.catalog.datatypes.insert(full_name, *info);
808                return Ok(());
809            }
810            ObjectKind::Group => {}
811        }
812
813        // It is a group. Record its path from the actual link record — before
814        // the cycle check, so a hard-link alias of an already-visited group
815        // still appears — whether or not it holds datasets or attributes.
816        self.catalog.group_paths.insert(full_name.clone());
817        // Capture group attributes (e.g. the NeXus `NX_class` marker), keyed
818        // by path. Through the shared collector, which carries a per-attribute
819        // failure as an entry naming it: a group whose attribute did not
820        // decode must not come back as a group with one fewer attribute.
821        // Recorded unconditionally, even for a group with none: the entry
822        // also carries the header's own creation-order and storage facts,
823        // which exist whether or not the group currently has any attributes.
824        let ctx = self.meta.ctx;
825        let attrs = collect_object_attributes(self.handle, &ctx, &header);
826        self.catalog
827            .group_attributes
828            .insert(full_name.clone(), attrs);
829        // Same unconditional recording for link storage and link
830        // creation-order: this group's own header answers both whether or
831        // not it descends any further.
832        self.catalog.group_link_storage.insert(
833            full_name.clone(),
834            describe_link_storage(Some(&header), &ctx, stab),
835        );
836
837        // Descend at most once per object header (cycle guard); a second path
838        // to it is a group hard link — record the alias for lookups instead.
839        if let Some(first) = self.visited.get(&addr) {
840            let first = first.clone();
841            self.catalog.group_aliases.insert(full_name, first);
842            return Ok(());
843        }
844        self.visited.insert(addr, full_name.clone());
845        self.group(Some(&header), &full_name, depth + 1, stab)
846    }
847}
848
849/// Join a group path prefix and a child name, with no leading `/` on a
850/// root-level name.
851fn join_path(prefix: &str, name: &str) -> String {
852    if prefix.is_empty() {
853        name.to_string()
854    } else {
855        format!("{}/{}", prefix, name)
856    }
857}
858
859/// Whether an object header declares link-message storage (compact or
860/// dense) rather than the legacy symbol-table format — `Walk::group`'s own
861/// dispatch predicate, factored out so [`describe_link_storage`] answers the
862/// same question by construction rather than by keeping two checks in sync.
863fn header_declares_link_storage(header: &ObjectHeader) -> bool {
864    header
865        .messages
866        .iter()
867        .any(|m| m.msg_type == MSG_LINK || m.msg_type == MSG_LINK_INFO)
868}
869
870/// A group's own link storage kind and link creation-order policy — h5py's
871/// `link_storage_str` and `get_link_creation_order()`, computed together
872/// because both read the same `Link Info` message.
873///
874/// `stab` is the symbol-table scratch-pad copy from the entry that named
875/// this group, exactly as [`CatalogWalk::group`] takes it; `header` is
876/// `None` only when the object header itself did not decode.
877fn describe_link_storage(
878    header: Option<&ObjectHeader>,
879    ctx: &FormatContext,
880    stab: Option<(u64, u64)>,
881) -> (LinkStorage, CreationOrder) {
882    if let Some(h) = header.filter(|h| header_declares_link_storage(h)) {
883        // Link-message storage: the `Link Info` message, when present, gives
884        // both facts at once. Its absence means compact and untracked — no
885        // message exists to carry a creation index in.
886        return h
887            .messages
888            .iter()
889            .find(|m| m.msg_type == MSG_LINK_INFO)
890            .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
891            .map(|(info, _)| {
892                let storage = if info.is_dense() {
893                    LinkStorage::Dense
894                } else {
895                    LinkStorage::Compact
896                };
897                (storage, info.creation_order())
898            })
899            .unwrap_or((LinkStorage::Compact, CreationOrder::Untracked));
900    }
901    // No link-message storage in the header (or no header to check at all):
902    // symbol-table format when the scratch-pad or the header's own `Symbol
903    // Table` message resolves an address pair. Creation order is always
904    // untracked here — the pre-1.8 format predates the feature, and 1.8+
905    // never tracks creation order without also converting to link storage.
906    let (btree_addr, heap_addr) = match stab {
907        Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
908        _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
909            Hdf5Reader::stab_from_header(h, ctx)
910        }),
911    };
912    let storage = if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
913        LinkStorage::SymbolTable
914    } else {
915        // Neither link-message nor symbol-table storage is declared — an
916        // object header this crate could not fully account for. Nothing in
917        // the 92-case oracle suite reaches this path; it exists so an
918        // unreadable root group still answers rather than panicking.
919        LinkStorage::Compact
920    };
921    (storage, CreationOrder::Untracked)
922}
923
924/// What an object header describes.
925///
926/// libhdf5 decides an object's class from *which* messages the header holds
927/// (`H5O_obj_class`) and only then reads their contents; the two questions are
928/// separate, and answering them with one `Option<DatasetReadInfo>` is what let
929/// a dataset whose datatype this crate cannot decode leave the catalog as if
930/// the file did not contain it. Each outcome now has its own name, so no
931/// caller can turn "unreadable" back into "absent".
932enum ObjectKind {
933    /// A dataset: it carries a datatype, a dataspace and a data layout, and
934    /// every message the payload depends on decoded.
935    Dataset(Box<DatasetReadInfo>),
936    /// A dataset whose payload depends on a message this crate cannot decode.
937    /// The string names what stopped it and reaches the caller of any typed
938    /// access to the name.
939    UnreadableDataset(String),
940    /// A group: it carries link, link-info, symbol-table or group-info
941    /// storage, or none of the messages that identify anything else.
942    Group,
943    /// A committed (named) datatype object: a datatype message with neither
944    /// group storage nor the dataspace/layout pair a dataset needs.
945    CommittedDatatype(Box<CommittedDatatypeInfo>),
946}
947
948/// Whether a header's messages say it is a committed (named) datatype: a
949/// datatype message, no group storage, and not the dataspace/layout pair a
950/// dataset needs.
951///
952/// The one authority for that question. [`Hdf5Reader::classify_object`] asks it
953/// of a file being read and `ReopenWalk::plan` of one being appended to, and a
954/// header that is a named datatype to one and an unclassifiable object to the
955/// other is how `named_datatype_names` came to answer differently in the two
956/// modes for the same file.
957pub(crate) fn header_is_committed_datatype(header: &ObjectHeader) -> bool {
958    let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
959    let is_group = present(MSG_LINK)
960        || present(MSG_LINK_INFO)
961        || present(MSG_SYMBOL_TABLE)
962        || present(MSG_GROUP_INFO);
963    !is_group && present(MSG_DATATYPE) && !(present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT))
964}
965
966/// A committed (named) datatype as read from its own object header.
967///
968/// `H5Tcommit` gives a type a name and a place in the file; every dataset and
969/// attribute built on it then stores a reference to this object rather than a
970/// copy of the type. It is a third kind of object beside groups and datasets,
971/// and classifying it as neither is what left its name in the file with
972/// nothing behind it.
973#[derive(Debug, Clone)]
974pub struct CommittedDatatypeInfo {
975    /// The type this object commits, or what stopped it from decoding. The
976    /// object is in the listing either way, exactly as an unreadable dataset
977    /// is: the name is in the file whether or not this crate can read what it
978    /// names.
979    datatype: Result<DatatypeMessage, String>,
980    /// Attributes attached to the committed datatype itself.
981    attributes: Vec<AttributeMessage>,
982}
983
984impl CommittedDatatypeInfo {
985    /// The committed type, or the reason it cannot be read.
986    pub fn datatype(&self) -> Result<&DatatypeMessage, &str> {
987        self.datatype.as_ref().map_err(String::as_str)
988    }
989
990    /// The attributes attached to the committed datatype.
991    pub fn attributes(&self) -> &[AttributeMessage] {
992        &self.attributes
993    }
994}
995
996/// Everything the superblock extension object header contributes to the
997/// file-level view.
998///
999/// `H5Fsuper.c::H5F__super_read` opens this header immediately after decoding
1000/// the superblock, before any user object is reachable, so every message here
1001/// is in force for the first metadata decode that follows.
1002#[derive(Debug, Clone, Default, PartialEq, Eq)]
1003pub struct SuperblockExtension {
1004    /// Shared Message Table message (0x000F): where the SOHM master table is.
1005    pub shared_message_table: Option<SharedMessageTableMessage>,
1006    /// v1 B-tree "K" values message (0x0013): non-default split ranks.
1007    pub btree_k: Option<BtreeKMessage>,
1008    /// Driver info message (0x0014).
1009    pub driver_info: Option<DriverInfoMessage>,
1010    /// File space info message (0x0017): allocation strategy, page size, and
1011    /// the persisted free-space manager addresses.
1012    pub file_space_info: Option<FileSpaceInfoMessage>,
1013}
1014
1015/// The datasets the discovery walk found, in the order it found them, with
1016/// an index from canonical path to position.
1017///
1018/// Every open and every read resolves a dataset by name, so a plain `Vec` a
1019/// lookup scans made each of them cost a pass over the whole catalog — a file
1020/// holding thousands of datasets paid that per access. The index is derived
1021/// in [`DatasetTable::new`], which is the only way to build one, so it cannot
1022/// fall out of step with the list it indexes.
1023/// One dataset's chunk index, as decoded from the file: `(chunk address,
1024/// on-disk byte count, filter mask)` per chunk the index records, in the
1025/// order the index walks them, alongside each chunk's chunk-grid coordinates
1026/// (`coords[i * rank .. (i + 1) * rank]`).
1027///
1028/// What the file decides is `entries` and `coords`; what a read then does with
1029/// an entry is that read's own business — a chunk outside the selection is
1030/// never fetched, a filtered chunk is read to its recorded size — so nothing
1031/// about one call is recorded here. `images` is the exception the rest of this
1032/// file is built around: decompressed chunk images, which say nothing about
1033/// one call either (an image is the decode of stored bytes this index names)
1034/// and which live here precisely so that they cannot outlive the index that
1035/// named them — see [`ChunkImageCache`].
1036struct DecodedChunkIndex {
1037    entries: Vec<(u64, u64, u32)>,
1038    coords: Vec<u64>,
1039    images: ChunkImageCache,
1040}
1041
1042impl DecodedChunkIndex {
1043    /// A freshly decoded index, with an empty image cache. The only way to
1044    /// build one, so no decode site can forget the cache or hand one index's
1045    /// images to another.
1046    fn new(entries: Vec<(u64, u64, u32)>, coords: Vec<u64>) -> Self {
1047        Self {
1048            entries,
1049            coords,
1050            images: ChunkImageCache::default(),
1051        }
1052    }
1053}
1054
1055/// The stored bytes one cached chunk image was decoded from: the chunk's file
1056/// address, the byte count the read asked for at it, and the per-chunk filter
1057/// mask the pipeline ran under.
1058///
1059/// An image is the decode of exactly these three, so an entry cannot answer
1060/// for a chunk stored somewhere else, read at another length, or filtered
1061/// through another mask — the key carries the whole of what the image depends
1062/// on apart from the pipeline, which belongs to the catalog entry this cache
1063/// hangs under.
1064#[derive(Clone, Copy, PartialEq, Eq, Hash)]
1065struct ChunkImageKey {
1066    addr: u64,
1067    len: usize,
1068    mask: u32,
1069}
1070
1071/// Bytes of decompressed chunk images one dataset keeps — libhdf5's
1072/// `H5D_CHUNK_CACHE_NBYTES_DEF`, the default `rdcc` size every dataset gets
1073/// there, so a comparison against it is a comparison of like for like.
1074const CHUNK_CACHE_BYTES: usize = 1024 * 1024;
1075
1076/// Images one dataset keeps at once — libhdf5's `H5D_CHUNK_CACHE_NSLOTS_DEF`.
1077/// The byte budget is the real bound; this one keeps a dataset of very small
1078/// chunks from filling the budget with thousands of entries, which is also
1079/// what bounds the least-recently-used scan below.
1080const CHUNK_CACHE_SLOTS: usize = 521;
1081
1082/// Decompressed chunk images, kept for the next read that wants the same
1083/// chunk.
1084///
1085/// Four consecutive 64 KiB slices of a 256 KiB deflated chunk are four reads
1086/// of one chunk; without a cache each inflates it whole and throws away three
1087/// quarters. libhdf5 gives every dataset a raw-data chunk cache for exactly
1088/// this (`H5D__chunk_lock`, `rdcc`), which is why a row-at-a-time walk of a
1089/// compressed dataset costs it one inflate per chunk rather than one per row.
1090///
1091/// MUST NOT outlive the [`DecodedChunkIndex`] the images were named by, and is
1092/// therefore a field of it: an image is reachable only through the index it
1093/// was decoded under, so everything that drops a decoded index —
1094/// `DatasetTable::entry_mut`, and the table rebuild a SWMR refresh does —
1095/// drops the images with it. There is no second lifetime to keep in step.
1096///
1097/// A read hands over an image only for a chunk it did not consume
1098/// ([`ChunkPlacement::leaves_chunk_unconsumed`]), so a whole-dataset read,
1099/// which takes every chunk entire and can never want one twice, never touches
1100/// the cache at all. [`ChunkImageCacheState::insert`] is the sole owner of
1101/// what the cache holds: what fits the budget, and what is evicted to keep it
1102/// fitting. Everything above it may offer an image; only it decides.
1103#[derive(Default)]
1104struct ChunkImageCache {
1105    /// Interior mutability because a read reaches the index through an `Arc`.
1106    /// A poisoned lock degrades to "no cache", never a failed read.
1107    state: std::sync::Mutex<ChunkImageCacheState>,
1108}
1109
1110#[derive(Default)]
1111struct ChunkImageCacheState {
1112    /// Key → (last-use stamp, image).
1113    images: std::collections::HashMap<ChunkImageKey, (u64, std::sync::Arc<Vec<u8>>)>,
1114    /// Summed `images` lengths, against [`CHUNK_CACHE_BYTES`].
1115    bytes: usize,
1116    /// Monotonic use counter; the smallest stamp is the eviction victim.
1117    clock: u64,
1118}
1119
1120impl ChunkImageCache {
1121    /// The image cached for `key`, if one is held.
1122    fn get(&self, key: &ChunkImageKey) -> Option<std::sync::Arc<Vec<u8>>> {
1123        let mut state = self.state.lock().ok()?;
1124        state.clock += 1;
1125        let clock = state.clock;
1126        let (stamp, image) = state.images.get_mut(key)?;
1127        *stamp = clock;
1128        Some(std::sync::Arc::clone(image))
1129    }
1130
1131    /// How many images are held. A poisoned lock holds none it could serve.
1132    #[cfg(test)]
1133    fn held(&self) -> usize {
1134        self.state.lock().map(|s| s.images.len()).unwrap_or(0)
1135    }
1136
1137    /// Keep `image` as the decode of `key`, and hand it back shared.
1138    fn keep(&self, key: ChunkImageKey, image: Vec<u8>) -> std::sync::Arc<Vec<u8>> {
1139        let image = std::sync::Arc::new(image);
1140        if let Ok(mut state) = self.state.lock() {
1141            state.insert(key, std::sync::Arc::clone(&image));
1142        }
1143        image
1144    }
1145}
1146
1147impl ChunkImageCacheState {
1148    /// Add one image, then evict least-recently-used entries until the cache
1149    /// is inside both bounds again.
1150    ///
1151    /// An image over the byte budget is refused outright rather than evicting
1152    /// everything to hold it, so the entry just inserted — which carries the
1153    /// newest stamp and is therefore the last one eviction would reach — is
1154    /// never the entry evicted.
1155    fn insert(&mut self, key: ChunkImageKey, image: std::sync::Arc<Vec<u8>>) {
1156        let len = image.len();
1157        if len > CHUNK_CACHE_BYTES {
1158            return;
1159        }
1160        self.clock += 1;
1161        if let Some((_, old)) = self.images.insert(key, (self.clock, image)) {
1162            self.bytes -= old.len();
1163        }
1164        self.bytes += len;
1165        while self.bytes > CHUNK_CACHE_BYTES || self.images.len() > CHUNK_CACHE_SLOTS {
1166            let Some(victim) = self
1167                .images
1168                .iter()
1169                .min_by_key(|(_, (stamp, _))| *stamp)
1170                .map(|(k, _)| *k)
1171            else {
1172                break;
1173            };
1174            if let Some((_, old)) = self.images.remove(&victim) {
1175                self.bytes -= old.len();
1176            }
1177        }
1178    }
1179}
1180
1181/// The reader's dataset catalog, and the one place a decoded chunk index
1182/// lives.
1183///
1184/// A cached [`DecodedChunkIndex`] MUST describe the file exactly as the
1185/// catalog entry beside it does, and MUST NOT be handed out for an index
1186/// address other than the one it was decoded from. The fields are private to
1187/// the module, so an entry is reachable read-only (`get`, `iter`, `Index`) or
1188/// through [`DatasetTable::entry_mut`], which drops that entry's cached index
1189/// before handing out the reference: changing what an entry says about the
1190/// file cannot leave a stale index behind it. Rebuilding the table — what a
1191/// SWMR refresh does — drops every cached index with it.
1192mod dataset_table {
1193    use super::{DatasetReadInfo, DecodedChunkIndex};
1194    use std::sync::Arc;
1195
1196    pub(super) struct DatasetTable {
1197        list: Vec<DatasetReadInfo>,
1198        /// Canonical path (no leading `/`) → position in `list`. A name the
1199        /// walk recorded twice keeps its first position, which is the entry a
1200        /// scan from the front would have found.
1201        by_name: std::collections::HashMap<String, usize>,
1202        /// Per entry, the index address a chunk index was decoded from and
1203        /// what it decoded to. Parallel to `list`.
1204        chunk_index: Vec<Option<(u64, Arc<DecodedChunkIndex>)>>,
1205    }
1206
1207    impl DatasetTable {
1208        pub(super) fn new(list: Vec<DatasetReadInfo>) -> Self {
1209            let mut by_name = std::collections::HashMap::with_capacity(list.len());
1210            for (i, ds) in list.iter().enumerate() {
1211                by_name.entry(ds.name.clone()).or_insert(i);
1212            }
1213            let chunk_index = list.iter().map(|_| None).collect();
1214            Self {
1215                list,
1216                by_name,
1217                chunk_index,
1218            }
1219        }
1220
1221        /// Where the dataset named `name` sits in the list, or `None` when the
1222        /// catalog holds no such name.
1223        pub(super) fn position(&self, name: &str) -> Option<usize> {
1224            self.by_name.get(name).copied()
1225        }
1226
1227        pub(super) fn get(&self, name: &str) -> Option<&DatasetReadInfo> {
1228            self.list.get(self.position(name)?)
1229        }
1230
1231        pub(super) fn iter(&self) -> std::slice::Iter<'_, DatasetReadInfo> {
1232            self.list.iter()
1233        }
1234
1235        /// The entry at `i`, mutably. Whatever the caller changes about it may
1236        /// be what a decoded chunk index was read against, so that index goes
1237        /// first: this is the only way to a `&mut DatasetReadInfo`, which is
1238        /// what keeps the cache from outliving the entry it describes.
1239        pub(super) fn entry_mut(&mut self, i: usize) -> &mut DatasetReadInfo {
1240            self.chunk_index[i] = None;
1241            &mut self.list[i]
1242        }
1243
1244        /// The chunk index cached for entry `i`, if one was decoded from
1245        /// `index_address`. A different address means the entry now points at
1246        /// another structure, and the cached one does not answer for it.
1247        pub(super) fn chunk_index(
1248            &self,
1249            i: usize,
1250            index_address: u64,
1251        ) -> Option<&Arc<DecodedChunkIndex>> {
1252            match self.chunk_index.get(i)? {
1253                Some((addr, index)) if *addr == index_address => Some(index),
1254                _ => None,
1255            }
1256        }
1257
1258        /// Keep `index` as entry `i`'s decoded chunk index for
1259        /// `index_address`, and hand it back.
1260        pub(super) fn cache_chunk_index(
1261            &mut self,
1262            i: usize,
1263            index_address: u64,
1264            index: DecodedChunkIndex,
1265        ) -> Arc<DecodedChunkIndex> {
1266            let index = Arc::new(index);
1267            self.chunk_index[i] = Some((index_address, Arc::clone(&index)));
1268            index
1269        }
1270    }
1271
1272    impl std::ops::Index<usize> for DatasetTable {
1273        type Output = DatasetReadInfo;
1274
1275        fn index(&self, i: usize) -> &DatasetReadInfo {
1276            &self.list[i]
1277        }
1278    }
1279}
1280
1281use dataset_table::DatasetTable;
1282
1283/// HDF5 file reader.
1284pub struct Hdf5Reader {
1285    handle: FileHandle,
1286    meta: FileMeta,
1287    /// Messages read from the superblock extension object header, empty when
1288    /// the file has no extension.
1289    ext: SuperblockExtension,
1290    /// End-of-file address from the superblock.
1291    _eof: u64,
1292    /// Superblock format version (0-3), decoded once at open time by
1293    /// `detect_superblock_version` and never re-derived: 0/1 is the legacy
1294    /// symbol-table root, 2/3 the link-message root.
1295    superblock_version: u8,
1296    datasets: DatasetTable,
1297    /// Dataset-shaped objects this crate cannot read, keyed by path (no
1298    /// leading `/`), the value naming what stopped it. They are listed with
1299    /// the readable datasets and refuse typed access with that reason: an
1300    /// object the file contains is never reported as one it does not.
1301    unreadable: std::collections::BTreeMap<String, String>,
1302    /// Attributes on the root group (file-level attributes).
1303    root_attributes: ObjectAttributes,
1304    /// Link storage kind and link creation-order policy of the root group.
1305    root_link_storage: (LinkStorage, CreationOrder),
1306    /// Attributes on non-root groups, keyed by group path (no leading `/`).
1307    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
1308    /// Link storage kind and link creation-order policy of non-root groups,
1309    /// keyed by group path (no leading `/`).
1310    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
1311    /// Every non-root group path the discovery walk traversed into (no
1312    /// leading `/`), regardless of whether the group has datasets or
1313    /// attributes. Built from actual link records, so empty groups,
1314    /// attribute-only groups, and subgroup-only groups are all included.
1315    group_paths: std::collections::BTreeSet<String>,
1316    /// Group hard links: alias path → the first-walked path of the same
1317    /// group object header (both without a leading `/`). The walk
1318    /// descends each header once, so objects under the alias are stored
1319    /// under the first path; lookups resolve alias prefixes through this
1320    /// map, as HDF5 path traversal does.
1321    group_aliases: std::collections::HashMap<String, String>,
1322    /// Every link record in the file, keyed by full path (no leading `/`).
1323    /// A listing is a listing of links, so this holds soft and external
1324    /// links as well as the hard links that name the objects above.
1325    links: std::collections::BTreeMap<String, LinkClass>,
1326    /// The path this file was opened with. An external link resolves its
1327    /// target relative to the directory holding it (libhdf5 keeps the same
1328    /// thing as `H5F_EXTPATH`), so the reader has to remember where it came
1329    /// from.
1330    path: PathBuf,
1331    /// The directory holding this HDF5 file, resolved once at open time.
1332    /// External raw-data files (H5O_EFL_ID) are named relative to it when
1333    /// `HDF5_EXTFILE_PREFIX` contains `${ORIGIN}` (`H5D__build_file_prefix`,
1334    /// H5Dint.c) — captured at open time rather than re-derived from the
1335    /// process's current directory at read time, matching libhdf5's own
1336    /// one-time capture in `H5F_t::extpath`.
1337    source_dir: PathBuf,
1338    /// The locking policy this file was opened under, reused verbatim for
1339    /// every external target: libhdf5 hands `H5F_prefix_open_file` the
1340    /// parent's file-access property list, so one `HDF5_USE_FILE_LOCKING`
1341    /// setting (or one `H5FileOptions::locking` call) governs every file a
1342    /// path touches, not just the first.
1343    locking: crate::io::locking::FileLocking,
1344    /// `H5Pset_elink_prefix` for every external link this reader crosses —
1345    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix).
1346    /// Propagated verbatim to every target this reader opens, the way a lapl
1347    /// reaches the next hop of a link chain upstream (measured against
1348    /// libhdf5 1.14.6: a two-hop chain resolves its second hop under the
1349    /// prefix given at the first).
1350    elink_prefix: Option<String>,
1351    /// Files opened on another file's behalf — external-link targets,
1352    /// external-reference targets and virtual-dataset sources — keyed by the
1353    /// resolved path that opened them, so N names for one file share one
1354    /// open handle. libhdf5 shares one open file the same way: `H5F_open`
1355    /// hands back the `H5F_shared_t` a path already open has rather than a
1356    /// second one (H5Fint.c:1906-1918).
1357    ///
1358    /// How long an entry lives is its [`CrossFileOwner`]'s business, and
1359    /// differs by what named it. A target's own external links are cached in
1360    /// that target's map, so the first reader in a chain transitively holds
1361    /// the whole chain open.
1362    external: std::collections::BTreeMap<PathBuf, CrossFileEntry>,
1363    /// What each virtual mapping's stored source file name resolved to,
1364    /// keyed by the canonical path of the virtual dataset that named it and
1365    /// that name exactly as the mapping holds it. Separate from
1366    /// [`external_resolved`](Self::external_resolved) because the two
1367    /// searches read different environment variables and property lists, so
1368    /// one name can resolve two ways depending on which named it — and keyed
1369    /// by the virtual dataset as well because
1370    /// [`DatasetAccess::virtual_prefix`] is a per-open property, so two
1371    /// virtual datasets naming one source file name can legitimately reach
1372    /// two different files.
1373    vds_resolved: std::collections::BTreeMap<(String, String), PathBuf>,
1374    /// What each external link's stored file name resolved to, keyed by that
1375    /// name exactly as the link holds it.
1376    ///
1377    /// The search runs once per distinct name per reader and its answer is
1378    /// then fixed: re-probing the filesystem on a later crossing would let one
1379    /// link answer differently mid-session, and would fail to find the handle
1380    /// it already holds once the target has been renamed or unlinked.
1381    external_resolved: std::collections::BTreeMap<String, PathBuf>,
1382    /// Committed (named) datatype objects, keyed by path (no leading `/`).
1383    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
1384    /// Object header address → absolute path, for every group and dataset the
1385    /// discovery walk reached plus the root group. This is what turns the
1386    /// address an object reference stores back into a name.
1387    object_paths: std::collections::HashMap<u64, String>,
1388    /// The dataset-access properties in force for each *open* dataset, keyed
1389    /// by canonical path (no leading `/`); an absent entry means
1390    /// [`DatasetAccess::default`].
1391    ///
1392    /// libhdf5 keeps this in the dataset's *shared* open-object info, which
1393    /// is why the first open of a dataset fixes it for every later one
1394    /// ([`apply_dataset_access`](Self::apply_dataset_access)):
1395    /// `H5D__virtual_init` puts the view and the printf gap into
1396    /// `dset->shared->layout.storage.u.virt` (H5Dvirtual.c:2178-2188), and
1397    /// `H5D__open_name` puts both file prefixes into `shared->extfile_prefix`
1398    /// and `shared->vds_prefix` (H5Dint.c:1488-1521) — for every dataset,
1399    /// not only a virtual one. It is the single owner of that answer: the
1400    /// extent resolution and the external-file read both read it and nothing
1401    /// else writes it, so a SWMR [`refresh`](Self::refresh) re-resolves under
1402    /// the same properties rather than reverting to the defaults.
1403    dataset_access: std::collections::BTreeMap<String, AccessInForce>,
1404}
1405
1406/// A live open on a dataset.
1407///
1408/// The reader holds only a [`Weak`](std::sync::Weak) to it and every handle
1409/// an open handed out holds the strong one, so "is this dataset still open"
1410/// is answered by the handles themselves — nothing has to tell the reader
1411/// when one is dropped, and a handle's drop takes no lock on the file.
1412pub(crate) type DatasetOpenToken = std::sync::Arc<()>;
1413
1414/// One file this reader opened on another file's behalf, and what keeps it
1415/// open.
1416struct CrossFileEntry {
1417    reader: Box<Hdf5Reader>,
1418    owner: CrossFileOwner,
1419}
1420
1421/// What holds a [`CrossFileEntry`] open, which is the same question as how
1422/// long it stays open.
1423///
1424/// libhdf5 asks it the same way and gets two different answers, because
1425/// nothing caches these files across the open that needed them: the default
1426/// external file cache is *disabled* — `H5F_ACS_EFC_SIZE_DEF` is 0
1427/// (H5Pfapl.c:191) and `H5F_open` builds no cache below that
1428/// (H5Fint.c:1217-1218) — so the `H5F_efc_close` both kinds of crossing end
1429/// with (H5Lexternal.c:241, H5Dvirtual.c:927) falls straight through to
1430/// `H5F_try_close` (H5Fefc.c:420-426). What is left holding the file is
1431/// whatever object the crossing opened inside it.
1432enum CrossFileOwner {
1433    /// This reader, until it is dropped.
1434    ///
1435    /// An external link's target is held by the object the traversal opened
1436    /// in it (`H5O_open_name`, H5Lexternal.c:225), and an external
1437    /// reference's by `H5R__reopen_file`'s file handle; this crate's
1438    /// equivalent of those objects is the reader itself, which holds the
1439    /// target's whole catalog and answers every later name from it.
1440    ///
1441    /// This is a **deliberate difference from libhdf5**, not a match.
1442    /// Measured against 1.14.6 by watching `/proc/self/fd`: a link target's
1443    /// descriptor appears at the traversal, survives while any object opened
1444    /// through the link is open, and goes at that object's close — and a
1445    /// traversal that keeps no object (`f["ext/data"][...]`, a listing, an
1446    /// `H5Lget_info`) leaves nothing open at all. It is *not* cached past
1447    /// that: see this enum's own note on the disabled default EFC.
1448    ///
1449    /// Matching it would mean this crate re-walked a target file's whole
1450    /// catalog on every name that crosses the link, because a reader — not
1451    /// a per-object handle — is what it opens a target *as*. The delta is a
1452    /// descriptor-and-lock window with no effect on any byte read or
1453    /// written, so the reader's lifetime stands and the window is documented
1454    /// here rather than paid for with that redesign.
1455    Reader,
1456    /// The virtual datasets that named this file as a source, by canonical
1457    /// path in *this* reader. Dropped once none of them has a live open.
1458    ///
1459    /// `H5D__virtual_open_source_dset` leaves the source *dataset* open in
1460    /// the virtual dataset's shared layout (H5Dvirtual.c:901-902), and that
1461    /// is what keeps the source file open; `H5D__virtual_reset_layout` closes
1462    /// it at the last `H5Dclose` of the virtual dataset (H5Dvirtual.c:709-710
1463    /// via `H5D__virtual_reset_source_dset`, :955). Measured against
1464    /// libhdf5 1.14.6 by watching `/proc/self/fd`: the source appears at the
1465    /// read that needs it, survives a second virtual dataset naming the same
1466    /// file until *both* are closed, and goes at the last `H5Dclose` — not at
1467    /// `H5Fclose`.
1468    VirtualOpens(std::collections::BTreeSet<String>),
1469}
1470
1471impl CrossFileOwner {
1472    /// Record that `also` now names this file too, keeping whichever
1473    /// ownership outlives the other. [`Reader`](Self::Reader) outlives every
1474    /// virtual open, so once a file is held that way it stays held.
1475    fn widen(&mut self, also: CrossFileOwner) {
1476        match (&mut *self, also) {
1477            (CrossFileOwner::Reader, _) => {}
1478            (slot, CrossFileOwner::Reader) => *slot = CrossFileOwner::Reader,
1479            (CrossFileOwner::VirtualOpens(have), CrossFileOwner::VirtualOpens(more)) => {
1480                have.extend(more)
1481            }
1482        }
1483    }
1484
1485    /// The owner of a source file one virtual dataset named.
1486    fn virtual_open(vds: &str) -> Self {
1487        CrossFileOwner::VirtualOpens(std::iter::once(vds.to_string()).collect())
1488    }
1489}
1490
1491/// The dataset-access properties one dataset was resolved under, and the
1492/// opens that fixed them.
1493struct AccessInForce {
1494    access: DatasetAccess,
1495    /// Live while at least one handle from the open that set `access` is
1496    /// alive. Once it is dead the properties are only a record of how the
1497    /// stamped extent was arrived at (what a SWMR refresh re-resolves
1498    /// under); the next open resolves afresh under its own.
1499    open: std::sync::Weak<()>,
1500}
1501
1502/// The absolute form of a discovery-walk path (which carries no leading `/`):
1503/// the root group's empty path becomes `/`, `entry/data` becomes
1504/// `/entry/data`.
1505fn absolute_path(path: &str) -> String {
1506    format!("/{}", path.trim_start_matches('/'))
1507}
1508
1509/// Total byte length of `dims.product() * element_size`, computed with
1510/// saturating arithmetic. `dims` and `element_size` are file-derived; a
1511/// crafted file with huge dimensions thus yields a saturated (too-large)
1512/// value — rejected downstream by the file-size/buffer checks — rather
1513/// than panicking in a debug build or wrapping in release.
1514fn saturating_byte_len(dims: &[u64], element_size: u64) -> u64 {
1515    dims.iter()
1516        .fold(1u64, |acc, &d| acc.saturating_mul(d))
1517        .saturating_mul(element_size)
1518}
1519
1520/// A fixed-length string attribute's value, under the padding rule its
1521/// datatype declares.
1522///
1523/// The single owner for the attribute side of that rule: both
1524/// [`H5Reader::attr_string_value`] and the writer-mode fallback in
1525/// `H5Attribute::read_string` end here, so a space-padded attribute does not
1526/// read back with its padding attached on one path and not the other. Bytes
1527/// that are not valid UTF-8 become U+FFFD, as they always have on this path.
1528///
1529/// A datatype that is not a string at all is read as null-terminated, which is
1530/// what asking for the string value of, say, an integer attribute has always
1531/// meant here.
1532pub(crate) fn fixed_string_attr_value(attr: &AttributeMessage) -> IoResult<String> {
1533    use crate::format::messages::datatype::{fixed_string_content, DatatypeMessage};
1534    let padding = match attr.datatype {
1535        DatatypeMessage::FixedString { padding, .. } => padding,
1536        _ => 0,
1537    };
1538    let content = fixed_string_content(&attr.data, padding).ok_or_else(|| {
1539        crate::io::IoError::InvalidState(format!(
1540            "attribute {:?} uses string padding rule {padding}, which the format reserves",
1541            attr.name
1542        ))
1543    })?;
1544    Ok(String::from_utf8_lossy(content).to_string())
1545}
1546
1547/// Materialize a `total`-byte fill buffer, mapping allocation failure to a
1548/// clean error. `total` on a read path comes from untrusted file fields, so
1549/// a crafted file declaring an absurd dataset size would otherwise abort the
1550/// process when `vec![0u8; total]` fails to allocate.
1551fn alloc_tiled_fill(total: usize, fill_value: Option<&[u8]>) -> IoResult<Vec<u8>> {
1552    try_tiled_fill(total, fill_value).map_err(|_| {
1553        crate::io::IoError::InvalidState(format!(
1554            "cannot allocate {total} bytes for dataset buffer (file may be corrupt)"
1555        ))
1556    })
1557}
1558
1559/// Build a `Vec<T>` of `count` elements out of the bytes `define` writes.
1560///
1561/// `define` receives the vector's whole byte image — `count *
1562/// size_of::<T>()` bytes — and **must define every one of them** before it
1563/// returns `Ok`; only then are the elements claimed. That contract is what
1564/// lets a full-image read land in its destination directly: the buffer a read
1565/// fills is already the buffer the caller keeps, so a 128 MiB image is
1566/// touched once by the read instead of once to zero it, once to read it and
1567/// once to copy it into a typed vector.
1568///
1569/// [`Hdf5Reader::read_dataset_raw_into_unconverted`] is the read side of that
1570/// contract — it is the single owner of read-destination semantics precisely
1571/// because it defines every byte of the buffer it is handed — and
1572/// [`read_dataset_raw_into`](Hdf5Reader::read_dataset_raw_into) inherits it.
1573///
1574/// `count` on a read path comes from untrusted file fields, so the
1575/// reservation is fallible: a crafted file declaring an absurd dataset size
1576/// gets a clean error rather than an allocator abort.
1577pub(crate) fn read_image_into_new<T, E, F>(count: usize, define: F) -> Result<Vec<T>, E>
1578where
1579    T: crate::types::H5Type,
1580    F: FnOnce(&mut [u8]) -> Result<(), E>,
1581    E: From<crate::io::IoError>,
1582{
1583    let too_big = || {
1584        E::from(crate::io::IoError::InvalidState(format!(
1585            "cannot allocate {count} elements of {} bytes for a dataset buffer \
1586             (file may be corrupt)",
1587            std::mem::size_of::<T>()
1588        )))
1589    };
1590    let bytes = count
1591        .checked_mul(std::mem::size_of::<T>())
1592        .ok_or_else(too_big)?;
1593    let mut out: Vec<T> = Vec::new();
1594    out.try_reserve_exact(count).map_err(|_| too_big())?;
1595
1596    // Safety: `try_reserve_exact` succeeded, so the allocation holds `bytes`
1597    // contiguous bytes aligned for `T`, and `as_mut_ptr` is non-null (a
1598    // dangling-but-aligned pointer with `bytes == 0`, which an empty slice
1599    // permits). The slice is the only live reference to that memory while
1600    // `define` runs. `define` writes every byte before returning `Ok` — its
1601    // documented contract — so the elements are initialized by the time
1602    // `set_len` claims them, and every byte pattern is a valid `T`, the same
1603    // property of `H5Type` implementors that a typed read reinterpreting the
1604    // stored image already rests on. A `define` that fails returns before
1605    // `set_len`, so the vector drops empty and nothing reads the bytes it
1606    // left undefined.
1607    let image = unsafe { std::slice::from_raw_parts_mut(out.as_mut_ptr().cast::<u8>(), bytes) };
1608    define(image)?;
1609    unsafe { out.set_len(count) };
1610    Ok(out)
1611}
1612
1613/// Fill an existing buffer in place with the dataset's tiled fill value (or
1614/// zero when no fill value is set), matching [`try_tiled_fill`]'s tiling.
1615///
1616/// Used to initialize a read destination — both an internally-allocated `Vec`
1617/// and a caller-provided read-into buffer (whose prior contents are arbitrary)
1618/// — so any region a chunked read leaves untouched reads back as the fill
1619/// value rather than stale bytes.
1620fn fill_tiled_into(out: &mut [u8], fill_value: Option<&[u8]>) {
1621    out.fill(0);
1622    if let Some(fv) = fill_value {
1623        if !fv.is_empty() && !out.is_empty() {
1624            for slot in out.chunks_mut(fv.len()) {
1625                let n = slot.len().min(fv.len());
1626                slot[..n].copy_from_slice(&fv[..n]);
1627            }
1628        }
1629    }
1630}
1631
1632/// Resolve the directory a raw-data file name is joined against, matching
1633/// libhdf5's `H5D__build_file_prefix` (H5Dint.c) for the given environment
1634/// variable — `HDF5_EXTFILE_PREFIX` for External Data Files
1635/// ([`resolve_extfile_prefix`]), `HDF5_VDS_PREFIX` for Virtual Dataset
1636/// sources ([`resolve_vdsfile_prefix`]); both features route through the
1637/// same C function, just keyed on a different variable. `prop` is the dapl
1638/// property the variable falls back to, `None` when the caller has none —
1639/// which behaves exactly as an unset or empty one would.
1640///
1641/// `${ORIGIN}` expands to `source_dir` (the directory holding the open
1642/// HDF5 file); any other value is used as a literal prefix; unset, empty,
1643/// or `"."` means "no prefix" (`H5_combine_path`'s own default), so a
1644/// relative name resolves against the process's current directory instead.
1645fn resolve_file_prefix(env_var: &str, prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1646    // `H5D__build_file_prefix` reads the environment variable first and only
1647    // falls back to the property list when it is unset or empty
1648    // (H5Dint.c:1077-1082, :1085-1090) — so an environment prefix *shadows*
1649    // the property rather than being tried before it. What
1650    // `H5F_prefix_open_file` then tries before this is the raw environment
1651    // string, split on `:` and unexpanded, which is a different candidate
1652    // from the expansion built here.
1653    let env = std::env::var(env_var).ok().filter(|v| !v.is_empty());
1654    let prefix = match env.as_deref() {
1655        Some(v) => v,
1656        None => prop.filter(|v| !v.is_empty())?,
1657    };
1658    if prefix.is_empty() || prefix == "." {
1659        return None;
1660    }
1661    Some(match prefix.strip_prefix("${ORIGIN}") {
1662        Some(rest) => {
1663            let rest = rest.trim_start_matches(['/', '\\']);
1664            if rest.is_empty() {
1665                source_dir.to_path_buf()
1666            } else {
1667                source_dir.join(rest)
1668            }
1669        }
1670        None => PathBuf::from(prefix),
1671    })
1672}
1673
1674/// Resolve the directory an external file list's stored names are joined
1675/// against — `HDF5_EXTFILE_PREFIX`, or `H5Pset_efile_prefix` when the
1676/// environment names none (see [`resolve_file_prefix`]).
1677///
1678/// The single owner of that rule for both directions of I/O: `H5D__efl_read`
1679/// and `H5D__efl_write` join against the same `shared->extfile_prefix`
1680/// (H5Defl.c:315-317, :429-431), which `H5D__build_file_prefix` built once
1681/// for the open that created the dataset's shared info.
1682pub(crate) fn resolve_extfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1683    resolve_file_prefix("HDF5_EXTFILE_PREFIX", prop, source_dir)
1684}
1685
1686/// Resolve the directory Virtual Dataset source file names are joined
1687/// against — `HDF5_VDS_PREFIX`, or `H5Pset_virtual_prefix` when the
1688/// environment names none (see [`resolve_file_prefix`]).
1689fn resolve_vdsfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1690    resolve_file_prefix("HDF5_VDS_PREFIX", prop, source_dir)
1691}
1692
1693/// Join a raw-data file `name` against a resolved prefix, matching
1694/// libhdf5's `H5_combine_path` (H5system.c): an absolute `name` is used
1695/// as-is regardless of the prefix, and no prefix means "relative to the
1696/// process's current directory" — both of which `Path::join` already
1697/// implements for an absolute joinee. Shared by External Data Files and
1698/// Virtual Dataset source resolution — both call the same C function.
1699pub(crate) fn combine_prefixed_path(prefix: Option<&Path>, name: &str) -> PathBuf {
1700    match prefix {
1701        Some(p) => p.join(name),
1702        None => PathBuf::from(name),
1703    }
1704}
1705
1706/// Read `len` bytes starting at *dataset-relative* offset `skip` from an
1707/// external file list into `out`, walking slots by cumulative declared
1708/// size exactly like libhdf5's `H5D__efl_read` (H5Defl.c). A read past an
1709/// individual slot's actual on-disk length reads back as zero — the file
1710/// backing a slot may be shorter than the space the layout reserved in
1711/// it — but a read past the *total* declared size of the file list is
1712/// still an error, matching `H5D__efl_read`'s own "read past logical end
1713/// of file" check.
1714///
1715/// An `H5O_EFL_UNLIMITED` last slot needs no special case: the walk below
1716/// never steps past it (`skip >= u64::MAX` is never true), which is upstream's
1717/// `H5O_EFL_UNLIMITED == size || addr < cur + size`, and the read it then
1718/// takes is the whole remainder, bounded by whatever the file physically
1719/// holds.
1720fn read_external_file_bytes(
1721    external_files: &[ExternalFileSegment],
1722    extfile_prefix: Option<&Path>,
1723    mut skip: u64,
1724    out: &mut [u8],
1725) -> IoResult<()> {
1726    let mut slot_idx = 0usize;
1727    while slot_idx < external_files.len() && skip >= external_files[slot_idx].size {
1728        skip -= external_files[slot_idx].size;
1729        slot_idx += 1;
1730    }
1731
1732    let mut written = 0usize;
1733    while written < out.len() {
1734        let Some(slot) = external_files.get(slot_idx) else {
1735            return Err(crate::io::IoError::InvalidState(
1736                "read past the logical end of the external file list".into(),
1737            ));
1738        };
1739        let full_path = combine_prefixed_path(extfile_prefix, &slot.name);
1740        let ext_handle = FileHandle::open_read_with_locking(&full_path, FileLocking::Disabled)
1741            .map_err(|e| {
1742                crate::io::IoError::InvalidState(format!(
1743                    "unable to open external raw data file {}: {e}",
1744                    full_path.display()
1745                ))
1746            })?;
1747        let avail_in_slot = slot.size.saturating_sub(skip);
1748        let want = (out.len() - written) as u64;
1749        let this_read = avail_in_slot.min(want) as usize;
1750        let dst = &mut out[written..written + this_read];
1751        // A short physical file — the reserved slot size exceeds what was
1752        // ever actually written to it — reads back as zero for the
1753        // remainder, exactly like `H5D__efl_read`.
1754        let at = slot.offset.checked_add(skip).ok_or_else(|| {
1755            crate::io::IoError::InvalidState(format!(
1756                "external file '{}' slot offset {} overflows {skip} bytes into the slot",
1757                slot.name, slot.offset
1758            ))
1759        })?;
1760        let got = ext_handle.read_at_most(at, this_read)?;
1761        dst[..got.len()].copy_from_slice(&got);
1762        dst[got.len()..].fill(0);
1763
1764        written += this_read;
1765        skip = 0;
1766        slot_idx += 1;
1767    }
1768    Ok(())
1769}
1770
1771/// Recursion ceiling for virtual dataset nesting (a VDS whose source is
1772/// itself a VDS — possibly in another file). Bounded so a crafted cyclic
1773/// mapping chain fails cleanly instead of recursing until the stack
1774/// overflows; real VDS chains do not nest anywhere near this deep.
1775const MAX_VIRTUAL_DEPTH: usize = 16;
1776
1777/// Replace every unlimited mapping with the concrete mapping its open-time
1778/// resolution makes it — the clipped selections `H5D__virtual_set_extent_unlim`
1779/// leaves in `clipped_virtual_select` / `clipped_source_select` for a read to
1780/// use (`H5D__virtual_read` never sees the unclipped ones).
1781///
1782/// A mapping with no resolution recorded is passed through unchanged, which is
1783/// what a bounded mapping needs and what a mapping list written before this
1784/// pass existed reduces to.
1785fn concrete_virtual_mappings(
1786    list: &VirtualMappingList,
1787    resolution: &[MappingResolution],
1788) -> IoResult<Vec<VirtualMapping>> {
1789    let mut out = Vec::with_capacity(list.mappings.len());
1790    for (i, m) in list.mappings.iter().enumerate() {
1791        match resolution.get(i) {
1792            Some(MappingResolution::Unlimited {
1793                virtual_clip,
1794                source_clip,
1795            }) => out.push(VirtualMapping {
1796                virtual_selection: m.virtual_selection.clip_unlimited(*virtual_clip)?,
1797                source_selection: m.source_selection.clip_unlimited(*source_clip)?,
1798                ..built_names(m, 0)?
1799            }),
1800            // One printf mapping is a whole family: block `j` of the virtual
1801            // selection (`H5S_hyper_get_unlim_block`) is filled by the source
1802            // dataset whose name substitutes `j`, taking the mapping's whole
1803            // (limited) source selection. Only the blocks that have a source
1804            // become mappings — a non-zero printf gap leaves the others
1805            // inside the extent, reading as the fill value.
1806            Some(MappingResolution::Printf { present, .. }) => {
1807                let Some(r) = regular_hyperslab(&m.virtual_selection) else {
1808                    continue;
1809                };
1810                let rank = r.start.len();
1811                for &j in present {
1812                    out.push(VirtualMapping {
1813                        virtual_selection: Selection::Hyperslab {
1814                            rank,
1815                            form: Hyperslab::Regular(r.unlim_block(j)),
1816                        },
1817                        ..built_names(m, j)?
1818                    });
1819                }
1820            }
1821            _ => out.push(built_names(m, 0)?),
1822        }
1823    }
1824    Ok(out)
1825}
1826
1827/// One mapping with both source names built for block `blockno` —
1828/// `H5D__virtual_build_source_name`. A mapping with no substitutions still
1829/// goes through this, because that is where `%%` is unescaped: upstream uses
1830/// the parsed name rather than the stored one for an ordinary mapping too
1831/// (`H5D__virtual_load_layout`).
1832fn built_names(m: &VirtualMapping, blockno: u64) -> IoResult<VirtualMapping> {
1833    Ok(VirtualMapping {
1834        source_file_name: parse_source_name(&m.source_file_name)?.build(blockno),
1835        source_dset_name: parse_source_name(&m.source_dset_name)?.build(blockno),
1836        ..m.clone()
1837    })
1838}
1839
1840/// Recursion bound for open-time virtual-extent resolution.
1841///
1842/// Resolving a virtual dataset's extent opens the source datasets its
1843/// unlimited mappings name, and a source may itself be a virtual dataset in
1844/// another file whose own open resolves its own extent. A crafted cyclic
1845/// chain would otherwise recurse until the stack overflows, so the nesting is
1846/// counted per thread and the resolution is skipped once it reaches
1847/// [`MAX_VIRTUAL_DEPTH`] — a dataset that deep keeps its stored extent
1848/// instead of taking one from a cycle.
1849struct VirtualResolveDepth;
1850
1851thread_local! {
1852    static VIRTUAL_RESOLVE_DEPTH: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
1853}
1854
1855impl VirtualResolveDepth {
1856    fn enter<F: FnOnce() -> IoResult<()>>(f: F) -> IoResult<()> {
1857        let depth = VIRTUAL_RESOLVE_DEPTH.with(std::cell::Cell::get);
1858        if depth >= MAX_VIRTUAL_DEPTH {
1859            return Ok(());
1860        }
1861        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth + 1));
1862        let out = f();
1863        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth));
1864        out
1865    }
1866}
1867
1868/// The regular (start, stride, count, block) form behind a selection, or
1869/// `None` — the only form `H5S_UNLIMITED` can appear in, so every unlimited
1870/// computation goes through it.
1871fn regular_hyperslab(sel: &Selection) -> Option<&RegularHyperslab> {
1872    match sel {
1873        Selection::Hyperslab {
1874            form: Hyperslab::Regular(r),
1875            ..
1876        } => Some(r),
1877        _ => None,
1878    }
1879}
1880
1881/// Read `source` (via `read_source_box`, one call per box) and scatter it
1882/// into `out` at the elements `target` selects.
1883///
1884/// The two selections are paired element by element in their own linear
1885/// order — `H5S_select_project_intersection` (H5Sselect.c:2402) walks a VDS
1886/// mapping's virtual and source selections with one iterator each and
1887/// matches the two streams off one against one. Its only precondition is the
1888/// one asserted there and enforced by `H5D_virtual_check_mapping_pre` when
1889/// the mapping is written (H5Dvirtual.c:254-257): the two hold the same
1890/// number of elements. Ranks may differ, and so may the boxes each side
1891/// decomposes into — a source `H5S_SEL_ALL` over a 2x4 dataset legitimately
1892/// fills two 1x4 blocks of a virtual dataset.
1893///
1894/// Each source box is read whole, once, and the runs the pairing places from
1895/// it are copied out of that one buffer.
1896fn copy_matched_selections(
1897    mut read_source_box: impl FnMut(&[u64], &[u64], &mut [u8]) -> IoResult<()>,
1898    source: &ResolvedSelection,
1899    target: &ResolvedSelection,
1900    element_size: u64,
1901    out: &mut [u8],
1902) -> IoResult<()> {
1903    let (n_source, n_target) = (source.n_elements(), target.n_elements());
1904    if n_source != n_target {
1905        return Err(crate::io::IoError::InvalidState(format!(
1906            "virtual dataset mapping's source selection holds {n_source} elements and its              virtual selection {n_target}, which H5D_virtual_check_mapping_pre refuses"
1907        )));
1908    }
1909    // Pair the two element streams, splitting a run of either side wherever
1910    // the other side's run ends first, and file each matched segment under
1911    // the source box it must be read out of.
1912    let mut per_box: Vec<Vec<(u64, u64, u64)>> = vec![Vec::new(); source.boxes.len()];
1913    let (mut si, mut ti) = (0usize, 0usize);
1914    let (mut s_done, mut t_done) = (0u64, 0u64);
1915    while si < source.runs.len() && ti < target.runs.len() {
1916        let (s, t) = (source.runs[si], target.runs[ti]);
1917        let len = (s.len - s_done).min(t.len - t_done);
1918        per_box[s.box_index].push((s.offset_in_box + s_done, t.offset_in_extent + t_done, len));
1919        s_done += len;
1920        t_done += len;
1921        if s_done == s.len {
1922            si += 1;
1923            s_done = 0;
1924        }
1925        if t_done == t.len {
1926            ti += 1;
1927            t_done = 0;
1928        }
1929    }
1930
1931    for (segments, (box_start, box_count)) in per_box.iter().zip(&source.boxes) {
1932        if segments.is_empty() {
1933            continue;
1934        }
1935        let nbytes = saturating_byte_len(box_count, element_size) as usize;
1936        let mut buf = alloc_tiled_fill(nbytes, None)?;
1937        read_source_box(box_start, box_count, &mut buf)?;
1938        for &(from, to, len) in segments {
1939            let (from, to, len) = (
1940                (from * element_size) as usize,
1941                (to * element_size) as usize,
1942                (len * element_size) as usize,
1943            );
1944            let src = buf.get(from..from + len).ok_or_else(|| {
1945                crate::io::IoError::InvalidState(
1946                    "virtual dataset mapping's source selection reaches past its source box".into(),
1947                )
1948            })?;
1949            let dst = out.get_mut(to..to + len).ok_or_else(|| {
1950                crate::io::IoError::InvalidState(
1951                    "virtual dataset mapping's virtual selection reaches past the dataset".into(),
1952                )
1953            })?;
1954            dst.copy_from_slice(src);
1955        }
1956    }
1957    Ok(())
1958}
1959
1960/// Read and decode the global-heap collection at `addr`, applying the
1961/// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
1962/// signature must be present and the declared size at least `H5HG_MINSIZE`
1963/// (4096 bytes). There is no upper size cap — libhdf5 has none, and this
1964/// crate's writers put a whole write call's strings into one collection,
1965/// which a cap would turn into silent data loss.
1966///
1967/// A free function (not a method) so both [`Hdf5Reader::read_heap_collection`]
1968/// and the static dataset-open path (which only has a `&mut FileHandle`, not
1969/// a full `&mut Hdf5Reader`) share the one implementation.
1970fn read_heap_collection_from(
1971    handle: &mut FileHandle,
1972    ctx: &FormatContext,
1973    addr: u64,
1974) -> IoResult<GlobalHeapCollection> {
1975    let ss = ctx.sizeof_size as usize;
1976    let header_len = 4 + 1 + 3 + ss;
1977    let header_buf = handle.read_at_most(addr, header_len)?;
1978    if header_buf.len() < header_len || header_buf[0..4] != *b"GCOL" {
1979        return Err(crate::io::IoError::InvalidState(format!(
1980            "bad global heap collection signature at address {addr:#x}"
1981        )));
1982    }
1983    let declared = read_uint(&header_buf[8..], ss) as usize;
1984    if declared < 4096 {
1985        return Err(crate::io::IoError::InvalidState(format!(
1986            "global heap collection at address {addr:#x} declares size {declared}, \
1987             below the 4096-byte minimum"
1988        )));
1989    }
1990    let heap_buf = handle.read_at(addr, declared)?;
1991    let (coll, _) = GlobalHeapCollection::decode(&heap_buf, ctx)?;
1992    Ok(coll)
1993}
1994
1995/// One chunk's on-disk read request, built by a read path before any I/O.
1996struct ChunkReadJob {
1997    /// Byte offset of the chunk's stored bytes.
1998    addr: u64,
1999    /// Number of bytes to read.
2000    len: usize,
2001    /// `true` → [`FileHandle::read_at_most`] (short reads near EOF are fine);
2002    /// `false` → [`FileHandle::read_at`] (exact, errors on a short read).
2003    at_most: bool,
2004    /// Per-chunk filter mask (ignored when the pipeline is `None`).
2005    mask: u32,
2006}
2007
2008/// Read one chunk's raw bytes according to its job.
2009fn read_chunk_raw(handle: &FileHandle, j: &ChunkReadJob) -> IoResult<Vec<u8>> {
2010    if j.at_most {
2011        Ok(handle.read_at_most(j.addr, j.len)?)
2012    } else {
2013        Ok(handle.read_at(j.addr, j.len)?)
2014    }
2015}
2016
2017/// Memory traffic one positioned read is worth: a `pread` costs about what
2018/// copying this many bytes costs. It is the exchange rate
2019/// [`read_chunk_runs_into`] weighs when a chunk's selected runs could each be
2020/// read on their own instead of reading the chunk whole and copying them out.
2021const PREAD_COST_BYTES: u64 = 16 * 1024;
2022
2023/// One chunk's intersection with the read box, in global dataset indices.
2024///
2025/// The single derivation of chunk ∩ selection ∩ extent:
2026/// [`for_each_chunk_run`] decomposes this box into contiguous runs and
2027/// [`ChunkOverlap::bytes`] measures it, so what a plan is measured to cover
2028/// and what the same plan places can never disagree.
2029struct ChunkOverlap {
2030    lo: Vec<u64>,
2031    hi: Vec<u64>,
2032}
2033
2034impl ChunkOverlap {
2035    /// `None` when the chunk and the read box do not meet. A corrupt index can
2036    /// place a chunk past the extent, so the arithmetic saturates and an empty
2037    /// span simply yields `None`.
2038    fn of(place: &ChunkPlacement, chunk_coords: &[u64]) -> Option<Self> {
2039        let ChunkPlacement {
2040            geo: ChunkOutputGeometry {
2041                dims, chunk_dims, ..
2042            },
2043            starts,
2044            counts,
2045        } = *place;
2046        let ndims = dims.len();
2047        if ndims == 0 {
2048            return None;
2049        }
2050        let mut lo = vec![0u64; ndims];
2051        let mut hi = vec![0u64; ndims];
2052        for d in 0..ndims {
2053            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
2054            let chunk_end = origin.saturating_add(chunk_dims[d]).min(dims[d]);
2055            let sel_end = starts[d].saturating_add(counts[d]);
2056            lo[d] = origin.max(starts[d]);
2057            hi[d] = chunk_end.min(sel_end);
2058            if lo[d] >= hi[d] {
2059                return None;
2060            }
2061        }
2062        Some(Self { lo, hi })
2063    }
2064
2065    /// Output bytes this overlap covers — exactly the summed length of the
2066    /// runs [`for_each_chunk_run`] yields for the same chunk.
2067    fn bytes(&self, element_size: u64) -> u64 {
2068        self.lo
2069            .iter()
2070            .zip(&self.hi)
2071            .map(|(&l, &h)| h - l)
2072            .product::<u64>()
2073            .saturating_mul(element_size)
2074    }
2075}
2076
2077/// Output bytes a chunk plan covers, counting each chunk-grid slot once.
2078///
2079/// Distinct slots own disjoint boxes of the index space and one chunk's runs
2080/// are disjoint from each other, so this sum is the exact volume of their
2081/// union — provided a slot a corrupt index names twice is counted once, which
2082/// is what `seen` is for. The caller compares it against the output length:
2083/// anything short means some output byte belongs to a chunk this plan will not
2084/// place, and only then does the fill value have to go down first.
2085fn planned_coverage(place: &ChunkPlacement, jobs: &[Option<ChunkReadJob>], coords: &[u64]) -> u64 {
2086    let rank = place.geo.dims.len();
2087    if rank == 0 {
2088        return 0;
2089    }
2090    let mut seen: std::collections::HashSet<&[u64]> = std::collections::HashSet::new();
2091    let mut covered = 0u64;
2092    for (i, job) in jobs.iter().enumerate() {
2093        if job.is_none() {
2094            continue;
2095        }
2096        let c = &coords[i * rank..(i + 1) * rank];
2097        if !seen.insert(c) {
2098            continue;
2099        }
2100        if let Some(overlap) = ChunkOverlap::of(place, c) {
2101            covered = covered.saturating_add(overlap.bytes(place.geo.element_size));
2102        }
2103    }
2104    covered
2105}
2106
2107/// Walk the contiguous byte runs of one chunk's intersection with the read box
2108/// `[starts, starts + counts)`, as `(offset in the chunk image, offset in the
2109/// output, run length in bytes)`.
2110///
2111/// The last axis is innermost in both the chunk and the output, so the
2112/// intersection box decomposes into one contiguous run per setting of the
2113/// outer axes — no per-element loop. The single owner of chunk placement
2114/// geometry: [`copy_chunk_runs`] copies these runs out of a decoded chunk
2115/// image and [`read_chunk_runs_into`] reads the same runs straight out of the
2116/// file, so the two cannot place a byte differently. A full read passes the
2117/// whole extent as the box, which is why it needs no placement path of its
2118/// own.
2119fn for_each_chunk_run(
2120    place: &ChunkPlacement,
2121    chunk_coords: &[u64],
2122    mut f: impl FnMut(u64, u64, usize),
2123) {
2124    let ChunkPlacement {
2125        geo:
2126            ChunkOutputGeometry {
2127                dims,
2128                chunk_dims,
2129                element_size,
2130            },
2131        starts,
2132        counts,
2133    } = *place;
2134    let ndims = dims.len();
2135    // Global intersection box [lo, hi) of chunk ∩ selection ∩ dataset; `None`
2136    // covers the rank-0 case the indexing below could not survive.
2137    let Some(ChunkOverlap { lo, hi }) = ChunkOverlap::of(place, chunk_coords) else {
2138        return;
2139    };
2140
2141    let chunk_strides = compute_strides(chunk_dims, element_size);
2142    let out_strides = compute_strides(counts, element_size);
2143    let last = ndims - 1;
2144    let run_bytes = ((hi[last] - lo[last]) * element_size) as usize;
2145
2146    // Iterate the outer box dimensions [0, last); each position places one
2147    // contiguous last-axis run.
2148    let outer_extent: Vec<u64> = (0..last).map(|d| hi[d] - lo[d]).collect();
2149    let n_outer: u64 = outer_extent.iter().product(); // empty product == 1
2150    let mut oc = vec![0u64; last];
2151    for _ in 0..n_outer {
2152        let mut src_off = 0u64;
2153        let mut dst_off = 0u64;
2154        for d in 0..ndims {
2155            let g = if d < last { lo[d] + oc[d] } else { lo[last] };
2156            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
2157            src_off += (g - origin) * chunk_strides[d];
2158            dst_off += (g - starts[d]) * out_strides[d];
2159        }
2160        f(src_off, dst_off, run_bytes);
2161        for d in (0..last).rev() {
2162            oc[d] += 1;
2163            if oc[d] < outer_extent[d] {
2164                break;
2165            }
2166            oc[d] = 0;
2167        }
2168    }
2169}
2170
2171/// Copy one decoded chunk image's intersection with the read box into
2172/// `output`.
2173///
2174/// A run reaching past the image — a chunk read short at the end of the file —
2175/// or past the output cannot be copied; it is appended to `skipped` as the
2176/// output range `(offset, length)` nothing wrote, which is what
2177/// [`place_chunk_jobs`] fills with the fill value.
2178fn copy_chunk_runs(
2179    chunk_data: &[u8],
2180    output: &mut [u8],
2181    place: &ChunkPlacement,
2182    chunk_coords: &[u64],
2183    skipped: &mut Vec<(usize, usize)>,
2184) {
2185    for_each_chunk_run(place, chunk_coords, |src, dst, len| {
2186        let (s, d) = (src as usize, dst as usize);
2187        if s + len <= chunk_data.len() && d + len <= output.len() {
2188            output[d..d + len].copy_from_slice(&chunk_data[s..s + len]);
2189        } else {
2190            push_skipped(skipped, output.len(), d, len);
2191        }
2192    });
2193}
2194
2195/// Record the output range `[dst, dst + len)` as one nothing wrote, clamped to
2196/// the output so a range a corrupt geometry pushed past the end never widens
2197/// into a slice the fill would panic on.
2198fn push_skipped(skipped: &mut Vec<(usize, usize)>, out_len: usize, dst: usize, len: usize) {
2199    let start = dst.min(out_len);
2200    let end = dst.saturating_add(len).min(out_len);
2201    if start < end {
2202        skipped.push((start, end - start));
2203    }
2204}
2205
2206/// Read one unfiltered chunk's intersection with the read box straight from
2207/// the file into `output`, never materializing the chunk.
2208///
2209/// An unfiltered chunk's stored bytes are the dataset's bytes in the dataset's
2210/// own order, so every run of the intersection is one positioned read — the
2211/// byte ranges libhdf5 reads for the same selection instead of the whole chunk
2212/// (`H5D__chunk_read`, H5Dchunk.c). Adjacent runs coalesce into one read.
2213/// `job.len` is how many bytes of the chunk the whole-chunk read would have
2214/// held, so the runs placed here are exactly the runs
2215/// [`copy_chunk_runs`] would have placed out of that image.
2216///
2217/// Returns whether the chunk's bytes landed in `output`; `false` means the
2218/// caller must still read the chunk whole, and the `skipped` ranges this call
2219/// appended are the caller's to discard (the whole-chunk path records its
2220/// own). That happens when the runs are too
2221/// small for a read each to beat one whole-chunk read plus the copy out of it,
2222/// and when a read fails — the whole-chunk path owns both the error and the
2223/// short read a truncated file gives, and re-places any run already placed
2224/// here from the same file offsets, so a fallback never leaves a half-written
2225/// output.
2226fn read_chunk_runs_into(
2227    handle: &FileHandle,
2228    job: &ChunkReadJob,
2229    place: &ChunkPlacement,
2230    chunk_coords: &[u64],
2231    output: &mut [u8],
2232    skipped: &mut Vec<(usize, usize)>,
2233    dst: ReadDst,
2234) -> bool {
2235    let (addr, image_len) = (job.addr, job.len);
2236    // (file offset, offset in output, length), coalesced as they are planned.
2237    let mut runs: Vec<(u64, usize, usize)> = Vec::new();
2238    let mut selected = 0u64;
2239    let out_len = output.len();
2240    for_each_chunk_run(place, chunk_coords, |src, out_off, len| {
2241        let (s, d) = (src as usize, out_off as usize);
2242        if s + len > image_len || d + len > out_len {
2243            push_skipped(skipped, out_len, d, len);
2244            return;
2245        }
2246        selected += len as u64;
2247        // `addr` is the index's claim: a run end that does not fit in a u64
2248        // simply does not coalesce, and the read at that address fails on
2249        // its own terms.
2250        let at = addr.saturating_add(src);
2251        if let Some(last) = runs.last_mut() {
2252            if last.0.checked_add(last.2 as u64) == Some(at) && last.1 + last.2 == d {
2253                last.2 += len;
2254                return;
2255            }
2256        }
2257        runs.push((at, d, len));
2258    });
2259    if runs.is_empty() {
2260        // Nothing to place, which only a chunk outside the extent or a short
2261        // image produces: leave it to the whole-chunk path so a chunk that
2262        // cannot be read at all still fails there.
2263        return false;
2264    }
2265    // One read per run pays off while the extra syscalls cost less than the
2266    // whole-chunk read and the copy out of it they replace.
2267    if (runs.len() as u64 - 1).saturating_mul(PREAD_COST_BYTES) > image_len as u64 + selected {
2268        return false;
2269    }
2270    for (offset, at, len) in runs {
2271        if handle
2272            .read_exact_at_into(offset, &mut output[at..at + len], dst)
2273            .is_err()
2274        {
2275            return false;
2276        }
2277    }
2278    true
2279}
2280
2281/// Place every chunk a chunked read planned into `output`.
2282///
2283/// The single owner of "planned chunk → output bytes" for all five chunk index
2284/// types: each read path walks its own index and builds the jobs, and this
2285/// decides, per chunk, whether the read box's byte runs come straight out of
2286/// the file ([`read_chunk_runs_into`]) or whether the chunk has to be read and
2287/// decoded whole first ([`copy_chunk_runs`]). A filtered chunk's stored bytes
2288/// have no byte-range correspondence to the dataset's, so it always reads
2289/// whole. `coords` is the packed chunk-grid position table
2290/// ([`crate::io::chunk_grid::coords_table`]): job `i` sits at rank-many values
2291/// from `i * rank`.
2292///
2293/// It is also the single owner of the fill: every byte of `output` this
2294/// returns `Ok` on is either a byte some chunk placed or a byte filled with
2295/// the tiled fill value, so callers hand it an arbitrary (even uninitialized)
2296/// buffer and a chunked read that reaches no chunk at all still routes through
2297/// here with an empty job list. The fill is derived from the plan rather than
2298/// laid down blanket-first: when [`planned_coverage`] proves the plan covers
2299/// the whole output, a 128 MiB read never pays a 128 MiB memset it is about to
2300/// overwrite, and only the runs a short chunk image left unwritten are filled
2301/// afterwards.
2302///
2303/// `cache` is the dataset's [`ChunkImageCache`] — present for every read that
2304/// reached its chunks through a decoded index. It is consulted only for a
2305/// filtered chunk this read leaves partly unconsumed: an unfiltered chunk's
2306/// selected runs come straight out of the file below and never materialize an
2307/// image there is anything to keep, and a chunk this read takes entire is one
2308/// no later read can want more of.
2309fn place_chunk_jobs(
2310    handle: &FileHandle,
2311    mut jobs: Vec<Option<ChunkReadJob>>,
2312    coords: &[u64],
2313    req: ChunkReadRequest,
2314    geo: &ChunkOutputGeometry,
2315    cache: Option<&ChunkImageCache>,
2316    output: &mut [u8],
2317) -> IoResult<()> {
2318    let ChunkReadRequest {
2319        pipeline,
2320        target,
2321        fill_value,
2322        dst,
2323    } = req;
2324    let rank = geo.dims.len();
2325    let zeros = vec![0u64; rank];
2326    let place = ChunkPlacement::resolve(geo, target, &zeros);
2327    let at = |i: usize| &coords[i * rank..(i + 1) * rank];
2328
2329    // Fill first unless the plan already accounts for every output byte; then
2330    // only the runs placement could not honour need filling afterwards.
2331    let prefilled = planned_coverage(&place, &jobs, coords) != output.len() as u64;
2332    if prefilled {
2333        fill_tiled_into(output, fill_value);
2334    }
2335    let mut skipped: Vec<(usize, usize)> = Vec::new();
2336
2337    if pipeline.is_none() {
2338        for (i, job) in jobs.iter_mut().enumerate() {
2339            let Some(j) = job.as_ref() else { continue };
2340            let mark = skipped.len();
2341            if read_chunk_runs_into(handle, j, &place, at(i), output, &mut skipped, dst) {
2342                *job = None;
2343            } else {
2344                // The whole-chunk path re-places this chunk and records what
2345                // it could not place; a half-planned attempt must not leave a
2346                // range behind that the fill would then write over good bytes.
2347                skipped.truncate(mark);
2348            }
2349        }
2350    }
2351
2352    // A filtered chunk this read only partly consumes is worth an image: the
2353    // next slice of it pays a copy instead of a second inflate. `hits[i]` is
2354    // an image the cache already held — its job is dropped, so nothing reads
2355    // or decodes it — and `keys[i]` marks a chunk whose image this read is to
2356    // hand over once it has one.
2357    let mut hits: Vec<Option<std::sync::Arc<Vec<u8>>>> = Vec::new();
2358    let mut keys: Vec<Option<ChunkImageKey>> = Vec::new();
2359    let cache = cache.filter(|_| pipeline.is_some());
2360    if let Some(cache) = cache {
2361        hits.resize_with(jobs.len(), || None);
2362        keys.resize(jobs.len(), None);
2363        for (i, job) in jobs.iter_mut().enumerate() {
2364            let Some(j) = job.as_ref() else { continue };
2365            if !place.leaves_chunk_unconsumed(at(i)) {
2366                continue;
2367            }
2368            let key = ChunkImageKey {
2369                addr: j.addr,
2370                len: j.len,
2371                mask: j.mask,
2372            };
2373            match cache.get(&key) {
2374                Some(image) => {
2375                    hits[i] = Some(image);
2376                    *job = None;
2377                }
2378                None => keys[i] = Some(key),
2379            }
2380        }
2381        for (i, image) in hits.iter().enumerate() {
2382            if let Some(image) = image {
2383                copy_chunk_runs(image, output, &place, at(i), &mut skipped);
2384            }
2385        }
2386    }
2387
2388    if jobs.iter().any(Option::is_some) {
2389        // Every chunk whose image is a contiguous stretch of `output` decodes
2390        // into that stretch; the rest come back as images to scatter. The
2391        // borrow of `output` the sinks hold ends with the call, which returns
2392        // nothing that points into it.
2393        let decoded = {
2394            let sinks = carve_sinks(output, &jobs, coords, &place);
2395            read_and_decompress_chunks(handle, pipeline, jobs, sinks, geo.image_bytes())?
2396        };
2397        for (i, chunk) in decoded.into_iter().enumerate() {
2398            match chunk {
2399                ChunkDecoded::Absent => {}
2400                // A chunk marked for the cache is by construction a staged
2401                // one: a chunk whose image is a contiguous stretch of the
2402                // output is a chunk this read consumes entire, which
2403                // `ChunkPlacement::leaves_chunk_unconsumed` refuses.
2404                ChunkDecoded::Image(data) => {
2405                    match keys.get_mut(i).and_then(Option::take).zip(cache) {
2406                        Some((key, cache)) => {
2407                            let image = cache.keep(key, data);
2408                            copy_chunk_runs(&image, output, &place, at(i), &mut skipped)
2409                        }
2410                        None => copy_chunk_runs(&data, output, &place, at(i), &mut skipped),
2411                    }
2412                }
2413                // A chunk that decoded short of its image placed no usable run
2414                // — the same verdict `copy_chunk_runs` passes on a run reaching
2415                // past a short image — so the whole stretch reverts to fill,
2416                // not just the part past the image: a decode writes its output
2417                // in blocks and may have reached past the byte it stopped on.
2418                ChunkDecoded::InPlace { dst, len, bytes } if bytes < len => {
2419                    fill_tiled_into(&mut output[dst..dst + len], fill_value)
2420                }
2421                ChunkDecoded::InPlace { .. } => {}
2422            }
2423        }
2424    }
2425    if !prefilled {
2426        for (dst, len) in skipped {
2427            fill_tiled_into(&mut output[dst..dst + len], fill_value);
2428        }
2429    }
2430    Ok(())
2431}
2432
2433/// Where one planned chunk's decoded image lands.
2434enum ChunkSink<'a> {
2435    /// The chunk's whole image is this stretch of the read's output, starting
2436    /// at output offset `dst`: the decoder writes it in place and nothing is
2437    /// copied afterwards.
2438    Direct { dst: usize, out: &'a mut [u8] },
2439    /// The image has no contiguous home in the output; it is materialized and
2440    /// then placed run by run.
2441    Staged,
2442}
2443
2444/// What one planned chunk left for the placement step.
2445enum ChunkDecoded {
2446    /// The slot planned no chunk (`jobs[i]` was `None`).
2447    Absent,
2448    /// The image is here and still has to be placed run by run.
2449    Image(Vec<u8>),
2450    /// The image went straight into `output[dst..dst + len]`. `bytes` is the
2451    /// length the pipeline produced there: short of `len` only for a stored
2452    /// chunk that decoded to less than its image.
2453    InPlace {
2454        dst: usize,
2455        len: usize,
2456        bytes: usize,
2457    },
2458}
2459
2460/// Hand every chunk whose image is one contiguous stretch of `output` that
2461/// stretch to decode into, and stage every other chunk.
2462///
2463/// A chunk's image *is* the output's own bytes exactly when its intersection
2464/// with the read box is a single run that starts at the image's first byte and
2465/// carries the image's whole length — the whole chunk, laid down contiguously.
2466/// The runs come from [`for_each_chunk_run`], the same walk that would have
2467/// copied the image out, so a stretch handed out here is byte-for-byte the
2468/// stretch the copy would have written.
2469///
2470/// Distinct chunk-grid slots own disjoint boxes of the output, so their
2471/// stretches never overlap; a corrupt index naming one slot twice would break
2472/// that, so a stretch overlapping one already handed out is staged instead.
2473fn carve_sinks<'a>(
2474    output: &'a mut [u8],
2475    jobs: &[Option<ChunkReadJob>],
2476    coords: &[u64],
2477    place: &ChunkPlacement,
2478) -> Vec<ChunkSink<'a>> {
2479    let mut sinks = Vec::with_capacity(jobs.len());
2480    sinks.resize_with(jobs.len(), || ChunkSink::Staged);
2481    let rank = place.geo.dims.len();
2482    let out_len = output.len();
2483    if rank == 0 {
2484        return sinks;
2485    }
2486    let Some(image_bytes) = place.geo.image_bytes() else {
2487        return sinks;
2488    };
2489    // A chunk is whole inside the read box only if the box is at least a chunk
2490    // wide in every dimension: a narrower selection has no direct chunk at all,
2491    // and testing it once here spares it the per-chunk walk.
2492    if place
2493        .counts
2494        .iter()
2495        .zip(place.geo.chunk_dims)
2496        .any(|(c, k)| c < k)
2497    {
2498        return sinks;
2499    }
2500
2501    let mut wanted: Vec<(usize, usize)> = Vec::new();
2502    for (i, job) in jobs.iter().enumerate() {
2503        if job.is_none() {
2504            continue;
2505        }
2506        let mut runs = 0usize;
2507        let mut first = (0u64, 0u64, 0usize);
2508        for_each_chunk_run(place, &coords[i * rank..(i + 1) * rank], |src, dst, len| {
2509            if runs == 0 {
2510                first = (src, dst, len);
2511            }
2512            runs += 1;
2513        });
2514        let (src, dst, len) = first;
2515        if runs == 1
2516            && src == 0
2517            && len as u64 == image_bytes
2518            && dst.saturating_add(len as u64) <= out_len as u64
2519        {
2520            wanted.push((dst as usize, i));
2521        }
2522    }
2523    wanted.sort_unstable();
2524
2525    let len = image_bytes as usize;
2526    let mut rest: &'a mut [u8] = output;
2527    let mut base = 0usize;
2528    for (dst, i) in wanted {
2529        if dst < base {
2530            continue;
2531        }
2532        let (_, tail) = std::mem::take(&mut rest).split_at_mut(dst - base);
2533        let (mine, tail) = tail.split_at_mut(len);
2534        sinks[i] = ChunkSink::Direct { dst, out: mine };
2535        rest = tail;
2536        base = dst + len;
2537    }
2538    sinks
2539}
2540
2541/// Run the reverse filter pipeline (if any) over one chunk's raw bytes,
2542/// straight into `out`, returning the length of the image it produced.
2543///
2544/// The counterpart of [`decompress_chunk`] for a chunk whose image is already
2545/// the output's own bytes. A length below `out.len()` means the stored chunk
2546/// decoded short; a length above it means the surplus was discarded, which is
2547/// what copying `out.len()` bytes out of a materialized image does too.
2548fn decompress_chunk_into(
2549    pipeline: Option<&FilterPipeline>,
2550    raw: &[u8],
2551    mask: u32,
2552    out: &mut [u8],
2553) -> IoResult<usize> {
2554    match pipeline {
2555        Some(pl) => Ok(filter::reverse_filters_masked_into(pl, raw, mask, out)?),
2556        None => {
2557            let n = raw.len().min(out.len());
2558            out[..n].copy_from_slice(&raw[..n]);
2559            Ok(raw.len())
2560        }
2561    }
2562}
2563
2564/// Run the reverse filter pipeline (if any) over one chunk's raw bytes into a
2565/// fresh image, for a chunk whose bytes have no contiguous home in the output.
2566///
2567/// `image_bytes` is what the layout says the chunk decodes to
2568/// ([`ChunkOutputGeometry::image_bytes`]), so the image is allocated once at
2569/// its real size instead of being grown to it — a filter that has to discover
2570/// the size re-enters its decoder once per doubling. Bytes past what the
2571/// pipeline produced are cut off, so a chunk that decoded short is as short
2572/// here as the growing spelling left it and [`copy_chunk_runs`] passes the
2573/// same verdict on the runs reaching past it. Bytes past `image_bytes` are cut
2574/// off too: no run of a chunk reaches past its own image.
2575fn decompress_chunk(
2576    pipeline: Option<&FilterPipeline>,
2577    raw: Vec<u8>,
2578    mask: u32,
2579    image_bytes: Option<u64>,
2580) -> IoResult<Vec<u8>> {
2581    let Some(pl) = pipeline else { return Ok(raw) };
2582    let Some(image_bytes) = image_bytes.and_then(|b| usize::try_from(b).ok()) else {
2583        // Geometry a corrupt file can carry says nothing about the size; the
2584        // pipeline discovers it.
2585        return Ok(filter::reverse_filters_masked(pl, &raw, mask)?);
2586    };
2587    let mut image = vec![0u8; image_bytes];
2588    let produced = filter::reverse_filters_masked_into(pl, &raw, mask, &mut image)?;
2589    image.truncate(produced.min(image_bytes));
2590    Ok(image)
2591}
2592
2593/// Planned chunk bytes a batch must carry before the rayon pool earns its
2594/// entry cost.
2595///
2596/// `ThreadPool::install` injects a job and blocks on a latch; with the workers
2597/// parked that costs tens of microseconds, which is more than a small
2598/// selection's entire read. One 64 KiB slice of a chunked dataset plans one or
2599/// two chunks, so it decodes on the thread that planned it and never pays the
2600/// dispatch.
2601#[cfg(feature = "parallel")]
2602const PARALLEL_MIN_JOB_BYTES: u64 = 256 * 1024;
2603
2604/// Whether a batch is worth handing to the pool: more than one chunk to read,
2605/// and enough bytes behind them to repay [`PARALLEL_MIN_JOB_BYTES`]. Skipped
2606/// chunks (`None`) count for nothing — a slice whose chunks were all placed by
2607/// [`read_chunk_runs_into`] leaves no work at all.
2608#[cfg(feature = "parallel")]
2609fn worth_parallel(jobs: &[Option<ChunkReadJob>]) -> bool {
2610    let mut n = 0usize;
2611    let mut bytes = 0u64;
2612    for j in jobs.iter().flatten() {
2613        n += 1;
2614        bytes = bytes.saturating_add(j.len as u64);
2615        if n > 1 && bytes >= PARALLEL_MIN_JOB_BYTES {
2616            return true;
2617        }
2618    }
2619    false
2620}
2621
2622/// Read and decompress a batch of chunk jobs, preserving job order.
2623///
2624/// `jobs[i] == None` yields `Ok(None)` — a chunk skipped as out-of-selection
2625/// or unallocated. Otherwise the chunk's raw bytes are read and, when
2626/// `pipeline` is `Some`, run through the reverse filter pipeline with the
2627/// job's mask. This is the single owner of the read-then-decompress step for
2628/// every chunk index type; each read path only builds the jobs and scatters
2629/// the results.
2630///
2631/// On Unix and Windows, positioned reads at distinct offsets on a shared
2632/// `&File` each carry their own explicit offset and never consult a shared file
2633/// cursor (on Windows the cursor may move as a side effect, but nothing reads
2634/// it), so read + decompress run fused in one parallel pass — overlapping chunk
2635/// I/O across cores, which the
2636/// C library's default (non-MPI) path does not do. On targets with neither
2637/// positioned API the seek-based fallback shares the file cursor, so reads
2638/// stay serial there while decompression still parallelizes.
2639fn read_and_decompress_chunks(
2640    handle: &FileHandle,
2641    pipeline: Option<&FilterPipeline>,
2642    jobs: Vec<Option<ChunkReadJob>>,
2643    sinks: Vec<ChunkSink<'_>>,
2644    image_bytes: Option<u64>,
2645) -> IoResult<Vec<ChunkDecoded>> {
2646    // Decompress one chunk's raw bytes into whatever its sink says.
2647    let deliver = |raw: Vec<u8>, mask: u32, sink: ChunkSink<'_>| -> IoResult<ChunkDecoded> {
2648        match sink {
2649            ChunkSink::Direct { dst, out } => {
2650                let len = out.len();
2651                let bytes = decompress_chunk_into(pipeline, &raw, mask, out)?;
2652                Ok(ChunkDecoded::InPlace { dst, len, bytes })
2653            }
2654            ChunkSink::Staged => Ok(ChunkDecoded::Image(decompress_chunk(
2655                pipeline,
2656                raw,
2657                mask,
2658                image_bytes,
2659            )?)),
2660        }
2661    };
2662    #[cfg(all(feature = "parallel", any(unix, windows)))]
2663    {
2664        use rayon::prelude::*;
2665        // Fused read + decompress for one job.
2666        let decode =
2667            |(job, sink): (Option<ChunkReadJob>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
2668                match job {
2669                    Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
2670                    None => Ok(ChunkDecoded::Absent),
2671                }
2672            };
2673        // Run on rust-hdf5's private half-cores pool, not rayon's global pool;
2674        // fall back to serial if the pool could not be built, and for a batch
2675        // too small to repay entering it.
2676        let pool = crate::parallel::io_pool().filter(|_| worth_parallel(&jobs));
2677        let work: Vec<_> = jobs.into_iter().zip(sinks).collect();
2678        match pool {
2679            Some(pool) => pool.install(|| {
2680                work.into_par_iter()
2681                    .map(&decode)
2682                    .collect::<IoResult<Vec<_>>>()
2683            }),
2684            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
2685        }
2686    }
2687    #[cfg(all(feature = "parallel", not(any(unix, windows))))]
2688    {
2689        use rayon::prelude::*;
2690        // No positioned read API here: concurrent reads would race the shared
2691        // file cursor, so read serially, then parallelize decompression on
2692        // rust-hdf5's private half-cores pool (not rayon's global pool).
2693        let parallel = worth_parallel(&jobs);
2694        let raws: Vec<Option<(Vec<u8>, u32)>> = jobs
2695            .into_iter()
2696            .map(|job| match job {
2697                Some(j) => Ok(Some((read_chunk_raw(handle, &j)?, j.mask))),
2698                None => Ok(None),
2699            })
2700            .collect::<IoResult<Vec<_>>>()?;
2701        let decode =
2702            |(r, sink): (Option<(Vec<u8>, u32)>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
2703                match r {
2704                    Some((raw, mask)) => deliver(raw, mask, sink),
2705                    None => Ok(ChunkDecoded::Absent),
2706                }
2707            };
2708        let work: Vec<_> = raws.into_iter().zip(sinks).collect();
2709        // Fall back to serial if the private pool could not be built, and for
2710        // a batch too small to repay entering it.
2711        match crate::parallel::io_pool().filter(|_| parallel) {
2712            Some(pool) => pool.install(|| {
2713                work.into_par_iter()
2714                    .map(&decode)
2715                    .collect::<IoResult<Vec<_>>>()
2716            }),
2717            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
2718        }
2719    }
2720    #[cfg(not(feature = "parallel"))]
2721    {
2722        jobs.into_iter()
2723            .zip(sinks)
2724            .map(|(job, sink)| match job {
2725                Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
2726                None => Ok(ChunkDecoded::Absent),
2727            })
2728            .collect()
2729    }
2730}
2731
2732impl Hdf5Reader {
2733    /// Open an existing HDF5 file in SWMR read mode using the env-var-derived
2734    /// locking policy.
2735    ///
2736    /// Currently identical to `open()`, but indicates intent to use
2737    /// `refresh()` for re-reading metadata written by a concurrent SWMR writer.
2738    pub fn open_swmr(path: &Path) -> IoResult<Self> {
2739        Self::open(path)
2740    }
2741
2742    /// Open an existing HDF5 file in SWMR read mode with an explicit locking
2743    /// policy.
2744    pub fn open_swmr_with_locking(
2745        path: &Path,
2746        locking: crate::io::locking::FileLocking,
2747    ) -> IoResult<Self> {
2748        Self::open_with_locking(path, locking)
2749    }
2750
2751    /// Open an existing HDF5 file for reading using the env-var-derived
2752    /// locking policy.
2753    ///
2754    /// Auto-detects the superblock version and uses the appropriate code path:
2755    /// - v0/v1: legacy format with symbol tables and B-tree v1
2756    /// - v2/v3: modern format with link messages
2757    pub fn open(path: &Path) -> IoResult<Self> {
2758        Self::open_with_locking(
2759            path,
2760            crate::io::locking::FileLocking::from_env_or(Default::default()),
2761        )
2762    }
2763
2764    /// Open an existing HDF5 file for reading with an explicit locking policy.
2765    pub fn open_with_locking(
2766        path: &Path,
2767        locking: crate::io::locking::FileLocking,
2768    ) -> IoResult<Self> {
2769        let mut handle = FileHandle::open_read_with_locking(path, locking)?;
2770
2771        // The superblock is not necessarily at the start of the file: a
2772        // userblock precedes it, and `H5FD_locate_signature` finds it by
2773        // probing offset 0 and then every power of two from 512 up. The offset
2774        // it is found at is where HDF5 addresses are measured from, so it
2775        // becomes the handle's base address and every later offset — including
2776        // the superblock read just below — is relative to it.
2777        let super_addr = handle
2778            .locate_signature()?
2779            .ok_or(crate::format::FormatError::InvalidSignature)?;
2780        handle.set_base(super_addr);
2781
2782        // Read enough bytes to detect the superblock version and parse it.
2783        let sb_buf = handle.read_at_most(0, 1024)?;
2784        let version = detect_superblock_version(&sb_buf)?;
2785
2786        let origin = Origin {
2787            path: path.to_path_buf(),
2788            locking,
2789        };
2790        let mut reader = match version {
2791            0 | 1 => Self::open_v0v1(handle, &sb_buf, origin)?,
2792            2 | 3 => Self::open_v2v3(handle, &sb_buf, origin)?,
2793            v => {
2794                return Err(crate::io::IoError::Format(
2795                    crate::format::FormatError::InvalidVersion(v),
2796                ))
2797            }
2798        };
2799        // Resolved from the path this file was opened with (not the
2800        // process's current directory at read time) — see `source_dir`.
2801        let canonical = std::fs::canonicalize(path)?;
2802        reader.source_dir = canonical
2803            .parent()
2804            .map(Path::to_path_buf)
2805            .unwrap_or_default();
2806        // Only now, with `source_dir` set, can a source name be resolved —
2807        // and a virtual dataset's extent is not final until they are.
2808        VirtualResolveDepth::enter(|| reader.resolve_virtual_extents())?;
2809        Ok(reader)
2810    }
2811
2812    /// Open a file with v2/v3 superblock (existing code path).
2813    fn open_v2v3(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
2814        let sb = SuperblockV2V3::decode(sb_buf)?;
2815
2816        let ctx = FormatContext {
2817            sizeof_addr: sb.sizeof_offsets,
2818            sizeof_size: sb.sizeof_lengths,
2819        };
2820
2821        // A v2/v3 superblock has no room for the B-tree K values, so they are
2822        // the library defaults unless the extension carries the K message.
2823        let (meta, ext) = Self::read_extension_and_meta(
2824            &mut handle,
2825            ctx,
2826            BTreeV1Config::default(),
2827            sb.superblock_extension_address,
2828        )?;
2829
2830        // Read root group object header, following continuation blocks.
2831        let root_header =
2832            Self::read_object_header_full(&mut handle, &meta, sb.root_group_object_header_address)?;
2833
2834        // Walk the root group to discover datasets, group attributes, and
2835        // every group path that exists, from whichever storage each group's
2836        // own header declares.
2837        let catalog = Self::build_catalog(
2838            &mut handle,
2839            &meta,
2840            Some(&root_header),
2841            sb.root_group_object_header_address,
2842            None,
2843        )?;
2844
2845        // Collect root group attributes
2846        let root_attributes = collect_object_attributes(&mut handle, &ctx, &root_header);
2847        // A v2/v3 root group is always addressed directly by the superblock,
2848        // never through a symbol-table scratch-pad.
2849        let root_link_storage = describe_link_storage(Some(&root_header), &ctx, None);
2850
2851        Ok(Self {
2852            handle,
2853            meta,
2854            ext,
2855            _eof: sb.end_of_file_address,
2856            superblock_version: sb.version,
2857            object_paths: catalog.object_paths(sb.root_group_object_header_address),
2858            datasets: DatasetTable::new(catalog.datasets),
2859            unreadable: catalog.unreadable,
2860            root_attributes,
2861            root_link_storage,
2862            group_attributes: catalog.group_attributes,
2863            group_link_storage: catalog.group_link_storage,
2864            group_paths: catalog.group_paths,
2865            group_aliases: catalog.group_aliases,
2866            links: catalog.links,
2867            datatypes: catalog.datatypes,
2868            path: origin.path,
2869            locking: origin.locking,
2870            elink_prefix: None,
2871            external: Default::default(),
2872            external_resolved: Default::default(),
2873            dataset_access: Default::default(),
2874            vds_resolved: Default::default(),
2875            // Overwritten by `open_with_locking` once this returns.
2876            source_dir: PathBuf::new(),
2877        })
2878    }
2879
2880    /// Open a file with v0/v1 superblock (legacy format).
2881    fn open_v0v1(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
2882        let sb = SuperblockV0V1::decode(sb_buf)?;
2883
2884        let ctx = FormatContext {
2885            sizeof_addr: sb.sizeof_offsets,
2886            sizeof_size: sb.sizeof_lengths,
2887        };
2888
2889        // A v0/v1 superblock carries the K values itself; a v0 superblock has
2890        // no chunk-tree field, so that one keeps the library default. The
2891        // extension's K message, when present, overrides all three.
2892        let sb_btree = BTreeV1Config {
2893            sym_leaf_k: sb.sym_leaf_k,
2894            snode_internal_k: sb.btree_internal_k,
2895            chunk_internal_k: sb
2896                .indexed_storage_k
2897                .unwrap_or(BTreeV1Config::default().chunk_internal_k),
2898        };
2899        let (meta, ext) = Self::read_extension_and_meta(
2900            &mut handle,
2901            ctx,
2902            sb_btree,
2903            sb.superblock_extension_address,
2904        )?;
2905
2906        let ste = &sb.root_symbol_table_entry;
2907        let root_obj_addr = ste.obj_header_addr;
2908        let ste_stab = ste.cached_symbol_table();
2909
2910        // Read the root group's object header (following continuations).
2911        let root_hdr = Self::read_object_header_full(&mut handle, &meta, root_obj_addr).ok();
2912
2913        // Collect the root group's own attributes.
2914        let root_attributes = match root_hdr {
2915            Some(ref h) => collect_object_attributes(&mut handle, &ctx, h),
2916            None => ObjectAttributes::default(),
2917        };
2918        // The root group's own link storage and link creation-order, from
2919        // the same header-first/scratch-pad-fallback rule the walk below
2920        // uses to choose how to enumerate it.
2921        let root_link_storage = describe_link_storage(root_hdr.as_ref(), &ctx, ste_stab);
2922
2923        // A v0/v1-superblock file whose root group has migrated to link
2924        // storage (more than ~8 objects, or one link the old format cannot
2925        // express) carries `Link` / `Link Info` messages in its object
2926        // header, and the superblock symbol-table scratch-pad is then stale.
2927        // The walk picks the storage from the header for that reason, taking
2928        // the scratch-pad only as the symbol-table addresses — and only for a
2929        // symbol-table root group (`H5G_CACHED_STAB`), which is the one cache
2930        // type those two addresses mean anything for.
2931        let catalog = Self::build_catalog(
2932            &mut handle,
2933            &meta,
2934            root_hdr.as_ref(),
2935            root_obj_addr,
2936            ste_stab,
2937        )?;
2938
2939        Ok(Self {
2940            handle,
2941            meta,
2942            ext,
2943            _eof: sb.end_of_file_address,
2944            superblock_version: sb.version,
2945            object_paths: catalog.object_paths(root_obj_addr),
2946            datasets: DatasetTable::new(catalog.datasets),
2947            unreadable: catalog.unreadable,
2948            root_attributes,
2949            root_link_storage,
2950            group_attributes: catalog.group_attributes,
2951            group_link_storage: catalog.group_link_storage,
2952            group_paths: catalog.group_paths,
2953            group_aliases: catalog.group_aliases,
2954            links: catalog.links,
2955            datatypes: catalog.datatypes,
2956            path: origin.path,
2957            locking: origin.locking,
2958            elink_prefix: None,
2959            external: Default::default(),
2960            external_resolved: Default::default(),
2961            dataset_access: Default::default(),
2962            vds_resolved: Default::default(),
2963            // Overwritten by `open_with_locking` once this returns.
2964            source_dir: PathBuf::new(),
2965        })
2966    }
2967
2968    /// Read the superblock extension object header at `ext_addr` (if any) and
2969    /// fold what it says into the file-level decode parameters.
2970    ///
2971    /// `sb_btree` is what the superblock alone implies; the extension's
2972    /// v1-B-tree-"K" message replaces all three ranks when present, exactly as
2973    /// `H5F__super_read` does after `H5O_msg_read(&ext_loc, H5O_BTREEK_ID)`.
2974    pub(crate) fn read_extension_and_meta(
2975        handle: &mut FileHandle,
2976        ctx: FormatContext,
2977        sb_btree: BTreeV1Config,
2978        ext_addr: u64,
2979    ) -> IoResult<(FileMeta, SuperblockExtension)> {
2980        let mut meta = FileMeta {
2981            ctx,
2982            btree: sb_btree,
2983            sohm: None,
2984        };
2985        let ext = Self::superblock_extension_at(handle, ctx, sb_btree, ext_addr)?;
2986        if let Some(k) = ext.btree_k {
2987            meta.btree = BTreeV1Config {
2988                sym_leaf_k: k.sym_leaf_k,
2989                snode_internal_k: k.snode_internal_k,
2990                chunk_internal_k: k.chunk_internal_k,
2991            };
2992        }
2993        // A zero rank would make every v1 B-tree node zero-sized and every
2994        // symbol-table node hold no entries; libhdf5 rejects it at creation
2995        // (`H5Pset_sym_k`, `H5Pset_istore_k`), so a file carrying one is
2996        // corrupt rather than merely unusual.
2997        let b = &meta.btree;
2998        if b.sym_leaf_k == 0 || b.snode_internal_k == 0 || b.chunk_internal_k == 0 {
2999            return Err(crate::io::IoError::Format(
3000                crate::format::FormatError::InvalidData(format!(
3001                    "v1 B-tree K values must be non-zero (sym_leaf={}, snode={}, chunk={})",
3002                    b.sym_leaf_k, b.snode_internal_k, b.chunk_internal_k
3003                )),
3004            ));
3005        }
3006        // `H5F__super_read` calls `H5SM_get_info` here, so the shared-message
3007        // table is in place before the root group — the first object header
3008        // that can hold a shared message — is opened.
3009        if let Some(smt) = &ext.shared_message_table {
3010            meta.sohm = Some(Self::read_sohm_table(handle, &meta.ctx, smt)?);
3011        }
3012        Ok((meta, ext))
3013    }
3014
3015    /// The superblock extension's messages, for the address the superblock
3016    /// names. Yields the default (every field `None`) when there is no
3017    /// extension.
3018    ///
3019    /// The extension header is read with the pre-extension parameters: its own
3020    /// messages are never shared and never in a v1 B-tree, so nothing it
3021    /// contains is needed to decode it. This is also how the writer's append
3022    /// path learns what the file declares before it rewrites anything.
3023    pub(crate) fn superblock_extension_at(
3024        handle: &mut FileHandle,
3025        ctx: FormatContext,
3026        btree: BTreeV1Config,
3027        ext_addr: u64,
3028    ) -> IoResult<SuperblockExtension> {
3029        if ext_addr == UNDEF_ADDR || ext_addr == 0 {
3030            return Ok(SuperblockExtension::default());
3031        }
3032        let meta = FileMeta {
3033            ctx,
3034            btree,
3035            sohm: None,
3036        };
3037        Self::read_superblock_extension(handle, &meta, ext_addr)
3038    }
3039
3040    /// Decode the messages of the superblock extension object header.
3041    ///
3042    /// Upstream reads each of these with `H5O_msg_exists` + `H5O_msg_read` and
3043    /// fails the open when one is present but undecodable; a message this
3044    /// crate does not model is skipped, as an unknown non-critical message is
3045    /// elsewhere.
3046    fn read_superblock_extension(
3047        handle: &mut FileHandle,
3048        meta: &FileMeta,
3049        addr: u64,
3050    ) -> IoResult<SuperblockExtension> {
3051        let header = Self::read_object_header_full(handle, meta, addr)?;
3052        let ctx = &meta.ctx;
3053        let mut ext = SuperblockExtension::default();
3054        for msg in &header.messages {
3055            match msg.msg_type {
3056                MSG_SHARED_MESSAGE_TABLE => {
3057                    ext.shared_message_table =
3058                        Some(SharedMessageTableMessage::decode(&msg.data, ctx)?);
3059                }
3060                MSG_BTREE_K => ext.btree_k = Some(BtreeKMessage::decode(&msg.data)?),
3061                MSG_DRIVER_INFO => ext.driver_info = Some(DriverInfoMessage::decode(&msg.data)?),
3062                MSG_FILE_SPACE_INFO => {
3063                    ext.file_space_info = Some(FileSpaceInfoMessage::decode(&msg.data, ctx)?);
3064                }
3065                _ => {}
3066            }
3067        }
3068        Ok(ext)
3069    }
3070
3071    /// Read the SOHM master table named by the extension's shared-message
3072    /// table message.
3073    ///
3074    /// The table's length is not stored with it: the index count comes from
3075    /// the message, exactly as `H5SM__cache_table_get_final_load_size` takes it
3076    /// from `H5F_SOHM_NINDEXES`.
3077    fn read_sohm_table(
3078        handle: &mut FileHandle,
3079        ctx: &FormatContext,
3080        smt: &SharedMessageTableMessage,
3081    ) -> IoResult<SohmMasterTable> {
3082        if smt.table_address == UNDEF_ADDR || smt.nindexes == 0 {
3083            return Ok(SohmMasterTable::default());
3084        }
3085        let size = SohmMasterTable::encoded_size(ctx, smt.nindexes);
3086        let buf = handle.read_at(smt.table_address, size)?;
3087        Ok(SohmMasterTable::decode(&buf, ctx, smt.nindexes)?)
3088    }
3089
3090    /// Extract the symbol-table message (btree_addr, heap_addr) from an
3091    /// already-decoded object header.
3092    fn stab_from_header(header: &ObjectHeader, ctx: &FormatContext) -> (u64, u64) {
3093        for msg in &header.messages {
3094            if msg.msg_type == MSG_SYMBOL_TABLE {
3095                let sa = ctx.sizeof_addr as usize;
3096                if msg.data.len() >= 2 * sa {
3097                    return (read_uint(&msg.data, sa), read_uint(&msg.data[sa..], sa));
3098                }
3099            }
3100        }
3101        (UNDEF_ADDR, UNDEF_ADDR)
3102    }
3103
3104    /// Build the file catalog for a whole file, starting at its root group.
3105    ///
3106    /// `root_header` is `None` only when the root object header did not
3107    /// decode; `root_stab` then carries the superblock symbol-table entry's
3108    /// cached B-tree and local heap, which is enough to list a legacy file.
3109    fn build_catalog(
3110        handle: &mut FileHandle,
3111        meta: &FileMeta,
3112        root_header: Option<&ObjectHeader>,
3113        root_addr: u64,
3114        root_stab: Option<(u64, u64)>,
3115    ) -> IoResult<Catalog> {
3116        let mut walk = CatalogWalk::new(handle, meta, root_addr);
3117        walk.group(root_header, "", 0, root_stab)?;
3118        Ok(walk.finish())
3119    }
3120
3121    /// Read every link stored in a group's dense (fractal-heap) link storage.
3122    ///
3123    /// The `Link Info` message gives the fractal-heap address; each managed
3124    /// object in the heap is an encoded `Link` message. Returns the decoded
3125    /// links (hard and soft).
3126    pub(crate) fn read_dense_links(
3127        handle: &mut FileHandle,
3128        ctx: &FormatContext,
3129        fractal_heap_addr: u64,
3130    ) -> IoResult<Vec<LinkMessage>> {
3131        // Read the fractal heap header. Its on-disk size depends only on the
3132        // address/length widths, so a generous prefix read covers it.
3133        let hdr_buf = handle.read_at_most(fractal_heap_addr, 512)?;
3134        let fh_header = FractalHeapHeader::decode(&hdr_buf, ctx)?;
3135
3136        // Walk the heap's managed blocks; each block hands back a payload
3137        // region holding one or more packed encoded `Link` messages.
3138        let mut br = HandleBlockReader { handle };
3139        let payloads = fractal_heap::collect_managed_objects(&fh_header, ctx, &mut br)?;
3140
3141        let mut links = Vec::new();
3142        for payload in payloads {
3143            // Decode packed `Link` messages sequentially. Each decode reports
3144            // its consumed length; stop at the first byte that is not a valid
3145            // link (trailing free space or an unrelated managed object).
3146            let mut pos = 0;
3147            while pos < payload.len() {
3148                // A v1 link message starts with version byte 1.
3149                if payload[pos] != 1 {
3150                    break;
3151                }
3152                match LinkMessage::decode(&payload[pos..], ctx) {
3153                    Ok((link, consumed)) if consumed > 0 => {
3154                        links.push(link);
3155                        pos += consumed;
3156                    }
3157                    _ => break,
3158                }
3159            }
3160        }
3161
3162        // That scan stops at the first byte that does not begin a link, which
3163        // is how trailing free space in a direct block ends it — and would
3164        // equally swallow a link the scan could not read. The heap header
3165        // counts its managed objects, so a short scan is detectable, and a
3166        // group listing that is short is the loss this guards against.
3167        if links.len() < fh_header.man_nobjs as usize {
3168            return Err(crate::io::IoError::InvalidState(format!(
3169                "dense link storage at address {fractal_heap_addr:#x} holds {} managed \
3170                 objects but only {} decoded as links",
3171                fh_header.man_nobjs,
3172                links.len()
3173            )));
3174        }
3175
3176        Ok(links)
3177    }
3178
3179    /// Recursively walk a B-tree v1 to collect leaf-level SNOD addresses.
3180    fn collect_snod_addresses(
3181        handle: &mut FileHandle,
3182        meta: &FileMeta,
3183        tree_addr: u64,
3184        depth: usize,
3185        visited: &mut std::collections::HashSet<u64>,
3186    ) -> IoResult<Vec<u64>> {
3187        let sizeof_addr = meta.ctx.sizeof_addr as usize;
3188        let sizeof_size = meta.ctx.sizeof_size as usize;
3189        // A well-formed v1 B-tree's level strictly decreases with depth;
3190        // bound the descent so a corrupt/cyclic tree cannot recurse forever.
3191        // The `visited` set additionally stops a corrupt tree whose child
3192        // points back at an ancestor node from fanning out exponentially.
3193        if depth > 256 || !visited.insert(tree_addr) {
3194            return Ok(Vec::new());
3195        }
3196        // A v1 B-tree node is a fixed-size record whose length follows from
3197        // the file's K values; reading exactly that much also bounds what a
3198        // corrupt address can pull in.
3199        let node_size = meta.btree.snode_btree_node_size(sizeof_addr, sizeof_size);
3200        let buf = handle.read_at_most(tree_addr, node_size)?;
3201        let node = BTreeV1Node::decode(
3202            &buf,
3203            sizeof_addr,
3204            sizeof_size,
3205            meta.btree.snode_max_entries(),
3206        )?;
3207
3208        if node.level == 0 {
3209            // Leaf level: children are SNOD addresses
3210            Ok(node.children.clone())
3211        } else {
3212            // Internal level: children are sub-TREE addresses
3213            let mut addrs = Vec::new();
3214            for &child_addr in &node.children {
3215                let child_addrs =
3216                    Self::collect_snod_addresses(handle, meta, child_addr, depth + 1, visited)?;
3217                addrs.extend(child_addrs);
3218            }
3219            Ok(addrs)
3220        }
3221    }
3222
3223    /// Read the object header at `addr` with every continuation block
3224    /// flattened in and every stored-shared message resolved to its literal
3225    /// body. One owner for both halves of the crate — see
3226    /// [`crate::io::object_header_io`].
3227    fn read_object_header_full(
3228        handle: &mut FileHandle,
3229        meta: &FileMeta,
3230        addr: u64,
3231    ) -> IoResult<ObjectHeader> {
3232        crate::io::object_header_io::read_object_header_full(handle, meta, addr)
3233    }
3234
3235    /// Read a committed datatype object's type and attributes from its header.
3236    ///
3237    /// A committed datatype's own message holds the type itself, but the
3238    /// format does not forbid it being a reference in turn, so it goes
3239    /// through the same resolver every other datatype message does.
3240    fn committed_datatype(
3241        handle: &mut FileHandle,
3242        header: &ObjectHeader,
3243        meta: &FileMeta,
3244    ) -> CommittedDatatypeInfo {
3245        let datatype = header
3246            .messages
3247            .iter()
3248            .find(|m| m.msg_type == MSG_DATATYPE)
3249            .cloned()
3250            .ok_or_else(|| "it holds no datatype message".to_string())
3251            .and_then(|m| {
3252                crate::io::object_header_io::read_datatype_message(handle, meta, &m).map_err(|e| {
3253                    match e {
3254                        crate::io::IoError::Unsupported(why) => why,
3255                        other => format!("its datatype message does not decode: {other}"),
3256                    }
3257                })
3258            });
3259        let attributes = header
3260            .messages
3261            .iter()
3262            .filter(|m| m.msg_type == MSG_ATTRIBUTE && m.flags & MSG_FLAG_SHARED == 0)
3263            .filter_map(|m| {
3264                AttributeMessage::decode(&m.data, &meta.ctx)
3265                    .ok()
3266                    .map(|(a, _)| a)
3267            })
3268            .collect();
3269        CommittedDatatypeInfo {
3270            datatype,
3271            attributes,
3272        }
3273    }
3274
3275    /// Classify one object from its (already read) header, and decode the
3276    /// dataset metadata while doing so.
3277    ///
3278    /// The class comes from which messages are present, never from whether
3279    /// they decode: an object holding a datatype, a dataspace and a data
3280    /// layout is a dataset even when this crate cannot decode one of them,
3281    /// and it says so as [`ObjectKind::UnreadableDataset`] rather than
3282    /// vanishing.
3283    ///
3284    /// Only the messages the payload depends on can make a dataset
3285    /// unreadable. A failed *attribute* decode leaves the dataset itself
3286    /// readable, so it does not.
3287    fn classify_object(
3288        handle: &mut FileHandle,
3289        header: &ObjectHeader,
3290        meta: &FileMeta,
3291        name: &str,
3292        addr: u64,
3293    ) -> ObjectKind {
3294        let ctx = &meta.ctx;
3295        let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
3296        let is_group = present(MSG_LINK)
3297            || present(MSG_LINK_INFO)
3298            || present(MSG_SYMBOL_TABLE)
3299            || present(MSG_GROUP_INFO);
3300        if is_group {
3301            return ObjectKind::Group;
3302        }
3303        if header_is_committed_datatype(header) {
3304            return ObjectKind::CommittedDatatype(Box::new(Self::committed_datatype(
3305                handle, header, meta,
3306            )));
3307        }
3308        let is_dataset =
3309            present(MSG_DATATYPE) && present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT);
3310        if !is_dataset {
3311            return ObjectKind::Group;
3312        }
3313
3314        let mut datatype = None;
3315        let mut dataspace = None;
3316        let mut layout = None;
3317        let mut filter_pipeline = None;
3318        let mut fill_value = None;
3319        // No message at all is the library default: a fresh dataset
3320        // creation property list starts fill_defined = 1
3321        // (`FillValueMessage::default`), so a dataset that never got one
3322        // written reads back exactly as if it had.
3323        let mut fill_defined: u8 = 1;
3324        let mut fill_write_time: u8 = FILL_TIME_IFSET;
3325        let mut alloc_time: u8 = ALLOC_TIME_LATE;
3326        // The first message that did not decode, kept verbatim: it is the
3327        // answer a caller gets when it asks for this dataset.
3328        let mut blocked: Option<String> = None;
3329        let mut block = |why: String| {
3330            if blocked.is_none() {
3331                blocked = Some(why);
3332            }
3333        };
3334        let mut external_file_list = None;
3335
3336        for msg in &header.messages {
3337            // A shared message holds a reference to where its body lives, not
3338            // the body. Decoding one as a body does not fail loudly — it
3339            // reads the reference's version byte as the body's — so anything
3340            // this crate does not follow is named here instead. A datatype
3341            // reference is followed; an attribute reference is skipped, since
3342            // an attribute never blocks the dataset it hangs on.
3343            let shared = msg.flags & MSG_FLAG_SHARED != 0;
3344            if shared && !matches!(msg.msg_type, MSG_DATATYPE | MSG_ATTRIBUTE) {
3345                block(format!(
3346                    "its message of type {:#04x} is a shared-message reference, which this \
3347                     crate follows only for datatypes",
3348                    msg.msg_type
3349                ));
3350                continue;
3351            }
3352            match msg.msg_type {
3353                // The resolver already says whether the type failed to decode
3354                // or sits somewhere this crate does not follow, so its wording
3355                // is the reason rather than something to wrap.
3356                MSG_DATATYPE => {
3357                    match crate::io::object_header_io::read_datatype_message(handle, meta, msg) {
3358                        Ok(dt) => datatype = Some(dt),
3359                        Err(crate::io::IoError::Unsupported(why)) => block(why),
3360                        Err(e) => block(format!("its datatype message does not decode: {e}")),
3361                    }
3362                }
3363                MSG_DATASPACE => match DataspaceMessage::decode(&msg.data, ctx) {
3364                    Ok((ds, _)) => dataspace = Some(ds),
3365                    Err(e) => block(format!("its dataspace message does not decode: {e}")),
3366                },
3367                MSG_DATA_LAYOUT => match DataLayoutMessage::decode(&msg.data, ctx) {
3368                    Ok((dl, _)) => layout = Some(dl),
3369                    Err(e) => block(format!("its data layout message does not decode: {e}")),
3370                },
3371                // A filter pipeline that does not decode would leave the raw
3372                // chunk bytes to be handed back as if they were never
3373                // filtered, and an undecodable fill value would leave
3374                // unwritten regions reading as zeros. Both change the data a
3375                // read returns, so both block the dataset.
3376                MSG_FILTER_PIPELINE => match FilterPipeline::decode(&msg.data) {
3377                    Ok((fp, _)) => {
3378                        if !fp.filters.is_empty() {
3379                            filter_pipeline = Some(fp);
3380                        }
3381                    }
3382                    Err(e) => block(format!("its filter pipeline message does not decode: {e}")),
3383                },
3384                MSG_FILL_VALUE => match FillValueMessage::decode(&msg.data) {
3385                    Ok((fv, _)) => {
3386                        fill_defined = fv.fill_defined;
3387                        fill_write_time = fv.fill_write_time;
3388                        alloc_time = fv.alloc_time;
3389                        if fv.fill_defined == 2 {
3390                            fill_value = fv.fill_value;
3391                        }
3392                    }
3393                    Err(e) => block(format!("its fill value message does not decode: {e}")),
3394                },
3395                MSG_EXTERNAL_FILE_LIST => {
3396                    // Unlike a layout message, this one *is* the storage: a
3397                    // dataset with an external file list has no data address
3398                    // of its own (H5Dlayout.c routes storage through this
3399                    // message instead), so a list that does not decode must
3400                    // block the dataset rather than read back as zero bytes.
3401                    match ExternalFileListMessage::decode(&msg.data, ctx) {
3402                        Ok((efl, _)) => external_file_list = Some(efl),
3403                        Err(e) => block(format!(
3404                            "its external file list message does not decode: {e}"
3405                        )),
3406                    }
3407                }
3408                _ => {}
3409            }
3410        }
3411
3412        // libhdf5 checks a layout against its sibling dataspace and datatype
3413        // as the dataset opens (`H5O__layout_decode` for the chunk rank,
3414        // `H5D__compact_init` for the compact size); the three decode side
3415        // by side here, so this is where those checks land.
3416        if let (Some(ds), Some(dt), Some(dl)) = (&dataspace, &datatype, &layout) {
3417            if let Err(e) = dl.check_against_dataset(ds, dt, ctx) {
3418                block(format!(
3419                    "its layout doesn't fit its dataspace and datatype: {e}"
3420                ));
3421            }
3422        }
3423
3424        if let Some(why) = blocked {
3425            return ObjectKind::UnreadableDataset(why);
3426        }
3427        // The storage a dataset names outside its layout message. Both are
3428        // resolved before the dataset is registered and both block it when
3429        // they do not resolve, for the same reason the decode above does: a
3430        // `Virtual` or external-file layout carries no address of its own, so
3431        // a dropped mapping reads back as fill with no error at all.
3432        let external_files = match external_file_list {
3433            Some(efl) => match Self::resolve_external_file_slots(handle, ctx, &efl) {
3434                Ok(slots) => slots,
3435                Err(e) => {
3436                    return ObjectKind::UnreadableDataset(format!(
3437                        "its external file list does not resolve: {e}"
3438                    ))
3439                }
3440            },
3441            None => Vec::new(),
3442        };
3443        let virtual_mappings = match &layout {
3444            Some(DataLayoutMessage::Virtual {
3445                heap_address,
3446                heap_index,
3447                ..
3448            }) if *heap_index != 0 => {
3449                match Self::resolve_virtual_mappings(handle, ctx, *heap_address, *heap_index, name)
3450                {
3451                    Ok(list) => Some(list),
3452                    Err(e) => {
3453                        return ObjectKind::UnreadableDataset(format!(
3454                            "its virtual dataset mapping list does not resolve: {e}"
3455                        ))
3456                    }
3457                }
3458            }
3459            _ => None,
3460        };
3461        // The attribute set is collected whole, or the object says it could
3462        // not be: a short list here would be a dataset reporting attributes
3463        // the file does not agree it has.
3464        let attributes = collect_object_attributes(handle, ctx, header);
3465        match (datatype, dataspace, layout) {
3466            (Some(dt), Some(ds), Some(dl)) => ObjectKind::Dataset(Box::new(DatasetReadInfo {
3467                name: name.to_string(),
3468                object_header_address: addr,
3469                datatype: dt,
3470                dataspace: ds,
3471                layout: dl,
3472                filter_pipeline,
3473                attributes,
3474                fill_value,
3475                fill_defined,
3476                fill_write_time,
3477                alloc_time,
3478                external_files,
3479                virtual_mappings,
3480                // Both filled in by `resolve_virtual_extents` once the
3481                // reader has the directory source names resolve against; a
3482                // catalog on its own cannot open another file.
3483                virtual_resolution: None,
3484                virtual_stored_dims: None,
3485            })),
3486            // The three messages are present and none of them reported an
3487            // error, so this is unreachable; report it as unreadable rather
3488            // than dropping the name on an invariant this function owns.
3489            _ => ObjectKind::UnreadableDataset(
3490                "its datatype, dataspace and data layout messages decoded but did not all \
3491                 produce a value"
3492                    .into(),
3493            ),
3494        }
3495    }
3496
3497    /// Every dataset in the file, by path (no leading `/`).
3498    ///
3499    /// A dataset this crate cannot read is still a dataset the file
3500    /// contains, so it is listed here alongside the readable ones and
3501    /// answers [`Self::unreadable_reason`]; opening it reports that reason.
3502    /// Resolve a virtual dataset's mapping list from the global heap object
3503    /// its layout message points at (`H5D__virtual_load_layout`,
3504    /// H5Dvirtual.c). Like the external-file-list decode above, a failure
3505    /// here must not fall back to silently treating the dataset as having
3506    /// no data: a `Virtual` layout carries no data address of its own, so a
3507    /// dropped mapping list would read back as all-fill with no error.
3508    fn resolve_virtual_mappings(
3509        handle: &mut FileHandle,
3510        ctx: &FormatContext,
3511        heap_address: u64,
3512        heap_index: u32,
3513        name: &str,
3514    ) -> IoResult<VirtualMappingList> {
3515        let coll = read_heap_collection_from(handle, ctx, heap_address)?;
3516        let idx = u16::try_from(heap_index).map_err(|_| {
3517            crate::io::IoError::InvalidState(format!(
3518                "dataset {name:?} virtual mapping heap index {heap_index} does not fit \
3519                 the 16-bit on-disk field"
3520            ))
3521        })?;
3522        let obj = coll.get_object(idx).ok_or_else(|| {
3523            crate::io::IoError::InvalidState(format!(
3524                "dataset {name:?} virtual mapping list object {idx} not found in the \
3525                 global heap collection at address {heap_address:#x}"
3526            ))
3527        })?;
3528        VirtualMappingList::decode(obj, ctx).map_err(|e| {
3529            crate::io::IoError::InvalidState(format!(
3530                "dataset {name:?} has a malformed virtual dataset mapping list: {e}"
3531            ))
3532        })
3533    }
3534
3535    /// Resolve every external-file slot's name through the local heap the
3536    /// EFL message points at (H5Oefl.c decodes only the byte offset; the
3537    /// string itself lives in a separate on-disk local heap, exactly like a
3538    /// v0/v1 group's link names — see [`local_heap_get_string`]).
3539    pub(crate) fn resolve_external_file_slots(
3540        handle: &mut FileHandle,
3541        ctx: &FormatContext,
3542        efl: &ExternalFileListMessage,
3543    ) -> IoResult<Vec<ExternalFileSegment>> {
3544        let sa = ctx.sizeof_addr as usize;
3545        let ss = ctx.sizeof_size as usize;
3546        let heap_hdr_buf = handle.read_at_most(efl.heap_addr, 64)?;
3547        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
3548        let heap_data = handle.read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;
3549
3550        efl.slots
3551            .iter()
3552            .map(|slot| {
3553                let name = local_heap_get_string(&heap_data, slot.name_offset)?;
3554                Ok(ExternalFileSegment {
3555                    name,
3556                    offset: slot.offset,
3557                    size: slot.size,
3558                })
3559            })
3560            .collect()
3561    }
3562
3563    /// Return the names of all datasets in the root group.
3564    pub fn dataset_names(&self) -> Vec<&str> {
3565        let mut names: Vec<&str> = self.datasets.iter().map(|d| d.name.as_str()).collect();
3566        names.extend(self.unreadable.keys().map(String::as_str));
3567        names
3568    }
3569
3570    /// Why the dataset at `path` (no leading `/`) cannot be read, or `None`
3571    /// when it can be — or does not exist.
3572    pub fn unreadable_reason(&mut self, path: &str) -> Option<&str> {
3573        if self.external_edge(path).is_some() {
3574            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
3575            let local = owner.canonical_path(&local);
3576            return owner.unreadable.get(&local).map(String::as_str);
3577        }
3578        let path = self.canonical_path(path);
3579        self.unreadable.get(&path).map(String::as_str)
3580    }
3581
3582    /// Every link record in the file, keyed by full path (no leading `/`).
3583    pub fn links(&self) -> &std::collections::BTreeMap<String, LinkClass> {
3584        &self.links
3585    }
3586
3587    /// The paths of every committed (named) datatype object in this file.
3588    ///
3589    /// A committed datatype is in neither [`dataset_names`](Self::dataset_names)
3590    /// nor the group listing — it is a third kind of object, and this is its
3591    /// listing.
3592    pub fn named_datatype_names(&self) -> Vec<&str> {
3593        self.datatypes.keys().map(String::as_str).collect()
3594    }
3595
3596    /// The committed datatype at `path` (no leading `/`), following group hard
3597    /// links, soft links and external links the way `H5Topen` does.
3598    ///
3599    /// `NotFound` means no committed datatype of that name; a name that *is*
3600    /// one but whose type this crate cannot decode answers `Unsupported` with
3601    /// the reason, never an absence.
3602    pub fn named_datatype(&mut self, path: &str) -> IoResult<&DatatypeMessage> {
3603        self.named_datatype_info(path)?
3604            .datatype()
3605            .map_err(|why| crate::io::IoError::Unsupported(why.to_string()))
3606    }
3607
3608    /// The attribute names of the committed datatype at `path`, in name
3609    /// order — matching h5py's default iteration for the (usual) case where
3610    /// the committed datatype does not track attribute creation order.
3611    /// Unlike [`Self::dataset_attr_names`] and its group/root counterparts,
3612    /// this path does not carry a per-attribute creation index to prefer
3613    /// when the object does track it: committed-datatype attributes are
3614    /// collected straight from compact header messages
3615    /// ([`Self::committed_datatype`]), without the envelope's creation index
3616    /// or dense-storage support the shared `AttributeEntry` collector has.
3617    pub fn named_datatype_attr_names(&mut self, path: &str) -> IoResult<Vec<String>> {
3618        let mut names: Vec<String> = self
3619            .named_datatype_info(path)?
3620            .attributes()
3621            .iter()
3622            .map(|a| a.name.clone())
3623            .collect();
3624        names.sort();
3625        Ok(names)
3626    }
3627
3628    /// The committed datatype at `path`'s own object-header attribute count.
3629    ///
3630    /// Committed-datatype attributes are collected only from compact header
3631    /// messages ([`Self::committed_datatype`]) — this crate does not model
3632    /// dense attribute storage on a named datatype — so unlike
3633    /// [`ObjectAttributes::header_count`] this is simply the count of what
3634    /// [`Self::named_datatype_attr_names`] already lists, with no separate
3635    /// dense-index path to fall back to.
3636    pub fn named_datatype_header_attr_count(&mut self, path: &str) -> IoResult<u64> {
3637        Ok(self.named_datatype_info(path)?.attributes().len() as u64)
3638    }
3639
3640    /// One attribute of the committed datatype at `path`, by name.
3641    pub fn named_datatype_attr(
3642        &mut self,
3643        path: &str,
3644        attr_name: &str,
3645    ) -> IoResult<&AttributeMessage> {
3646        let owned = attr_name.to_string();
3647        self.named_datatype_info(path)?
3648            .attributes()
3649            .iter()
3650            .find(|a| a.name == owned)
3651            .ok_or_else(|| crate::io::IoError::NotFound(format!("{path}:{attr_name}")))
3652    }
3653
3654    /// The committed datatype object at `path`, after link traversal.
3655    ///
3656    /// The object answers here whether or not its type decodes; the reason it
3657    /// does not is on [`CommittedDatatypeInfo::datatype`].
3658    pub fn named_datatype_info(&mut self, path: &str) -> IoResult<&CommittedDatatypeInfo> {
3659        if self.external_edge(path).is_some() {
3660            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS)?;
3661            let local = owner.canonical_path(&local);
3662            return owner
3663                .datatypes
3664                .get(&local)
3665                .ok_or(crate::io::IoError::NotFound(local));
3666        }
3667        let local = self.canonical_path(path);
3668        self.datatypes
3669            .get(&local)
3670            .ok_or(crate::io::IoError::NotFound(local))
3671    }
3672
3673    /// The class of the link at `path` (no leading `/`), or `None` when no
3674    /// link of that name exists. The path is traversed first, so a link
3675    /// reached through a group hard link, a soft link or an external link
3676    /// resolves — an external link's own record is found before the
3677    /// traversal crosses it, since a name matches its own link exactly.
3678    pub fn link_class(&mut self, path: &str) -> Option<&LinkClass> {
3679        let path = path.trim_start_matches('/');
3680        if self.links.contains_key(path) {
3681            return self.links.get(path);
3682        }
3683        if self.external_edge(path).is_some() {
3684            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
3685            let local = owner.canonical_path(&local);
3686            return owner.links.get(&local);
3687        }
3688        let path = self.canonical_path(path);
3689        self.links.get(&path)
3690    }
3691
3692    /// Follow a path (no leading `/`) the way `H5Dopen` / `H5Gopen` do:
3693    /// rewrite each component that is a group hard-link alias or a soft link
3694    /// until nothing changes, bounded so a link cycle cannot loop forever.
3695    ///
3696    /// This is the single owner of link traversal — every lookup that takes a
3697    /// caller-supplied path goes through it rather than comparing the path to
3698    /// a catalog key directly.
3699    fn traverse(&self, name: &str) -> Traversal {
3700        // libhdf5 bounds soft-link traversal at `H5L_NLINKS_DEF`; this covers
3701        // that and the hard-link alias rewrites interleaved with it.
3702        const MAX_TRAVERSALS: usize = 64;
3703        let mut name = name.trim_start_matches('/').to_string();
3704        let mut via = None;
3705        for _ in 0..MAX_TRAVERSALS {
3706            let Some((prefix, rewrite)) = self.longest_rewrite(&name) else {
3707                break;
3708            };
3709            let rest = name[prefix.len()..].to_string();
3710            match rewrite {
3711                Rewrite::Alias(first) => {
3712                    // `first` is empty for an alias of the root group;
3713                    // trimming keeps the no-leading-'/' form either way.
3714                    name = format!("{first}{rest}").trim_start_matches('/').to_string();
3715                }
3716                Rewrite::Soft(target) => {
3717                    let resolved = resolve_link_value(prefix, target);
3718                    via = Some(SoftLinkRef {
3719                        link: prefix.to_string(),
3720                        target: target.to_string(),
3721                    });
3722                    name = format!("{resolved}{rest}")
3723                        .trim_start_matches('/')
3724                        .to_string();
3725                }
3726                Rewrite::External { file, path } => {
3727                    return Traversal::External {
3728                        link: prefix.to_string(),
3729                        file: file.to_string(),
3730                        path: format!("{path}{rest}"),
3731                    };
3732                }
3733            }
3734        }
3735        Traversal::Path { path: name, via }
3736    }
3737
3738    /// The rewrite one traversal step applies to `path`, and the prefix it
3739    /// matched: the longest prefix of `path` that is a group hard-link alias
3740    /// or a soft/external link, so a nested alias wins over a shorter one
3741    /// that also covers the path, and an alias wins over a link naming the
3742    /// same prefix.
3743    ///
3744    /// A prefix covers `path` only when it *is* `path` or ends at one of its
3745    /// `/` boundaries, so the candidates are `path` and its own ancestors —
3746    /// walking those from the longest down asks the catalogs by key instead
3747    /// of comparing every alias and every link against the path, which is
3748    /// what made each traversal cost a pass over the file's whole link table.
3749    fn longest_rewrite<'a>(&'a self, path: &str) -> Option<(&'a str, Rewrite<'a>)> {
3750        let mut end = path.len();
3751        loop {
3752            let candidate = &path[..end];
3753            if let Some((alias, first)) = self.group_aliases.get_key_value(candidate) {
3754                return Some((alias.as_str(), Rewrite::Alias(first)));
3755            }
3756            match self.links.get_key_value(candidate) {
3757                Some((link, LinkClass::Soft { path })) => {
3758                    return Some((link.as_str(), Rewrite::Soft(path)))
3759                }
3760                Some((link, LinkClass::External { file, path })) => {
3761                    return Some((link.as_str(), Rewrite::External { file, path }))
3762                }
3763                // A hard or user-defined link rewrites nothing, and a shorter
3764                // prefix of the path may still rewrite it.
3765                _ => {}
3766            }
3767            end = candidate.rfind('/')?;
3768        }
3769    }
3770
3771    /// The messages read from the superblock extension object header. All
3772    /// fields are `None` for a file without an extension.
3773    pub fn superblock_extension(&self) -> &SuperblockExtension {
3774        &self.ext
3775    }
3776
3777    /// Bytes the file's on-disk free-space managers record as free —
3778    /// `H5Fget_freespace`, the number `h5stat -S` prints as "Amount of tracked
3779    /// free space".
3780    ///
3781    /// Zero for a file whose file-space info message names no manager, which
3782    /// includes every file written without `persist`. The strategy is not
3783    /// consulted: a manager's header and section-info blocks have one layout
3784    /// whichever strategy allocated the space they describe.
3785    pub fn tracked_free_space(&mut self) -> IoResult<u64> {
3786        let Some(info) = self.ext.file_space_info.clone() else {
3787            return Ok(0);
3788        };
3789        crate::io::free_space_io::tracked_free_space(&mut self.handle, &self.meta.ctx, &info)
3790    }
3791
3792    /// Size in bytes of the userblock preceding the superblock: the offset the
3793    /// signature was found at, which is also the file's base address. Zero for
3794    /// a file without a userblock.
3795    pub fn userblock_size(&self) -> u64 {
3796        self.handle.base()
3797    }
3798
3799    /// The superblock format version (0-3), decoded once at open time and
3800    /// immutable for the life of an open file — a live SWMR refresh rescans
3801    /// the file's contents but never its own format version.
3802    pub fn superblock_version(&self) -> u8 {
3803        self.superblock_version
3804    }
3805
3806    /// Rewrite a path (no leading `/`) into the path of the object it reaches
3807    /// after link traversal. A path that leaves the file through an external
3808    /// link comes back unchanged — the callers that must report that case use
3809    /// [`Self::traverse`] directly.
3810    pub fn canonical_path(&self, name: &str) -> String {
3811        match self.traverse(name) {
3812            Traversal::Path { path, .. } => path,
3813            Traversal::External { .. } => name.trim_start_matches('/').to_string(),
3814        }
3815    }
3816
3817    /// Where `name` leaves this file, or `None` when it resolves inside it.
3818    ///
3819    /// This is the one question every path-taking entry point asks before it
3820    /// looks anything up: a name that crosses an external link is not this
3821    /// file's to answer, and answering it from this file's catalog anyway is
3822    /// how such a name came back as a plain absence.
3823    pub(crate) fn external_edge(&self, name: &str) -> Option<ExternalEdge> {
3824        match self.traverse(name) {
3825            Traversal::Path { .. } => None,
3826            Traversal::External { link, file, path } => Some(ExternalEdge { link, file, path }),
3827        }
3828    }
3829
3830    /// Candidate filesystem paths for a file named from inside this one, in
3831    /// the order `H5F_prefix_open_file` tries them (H5Fint.c:826-1025):
3832    ///
3833    /// 1. an absolute name exactly as given (:854-887) — and if that misses,
3834    ///    every later step uses its last component instead, as the C does;
3835    /// 2. each `:`-separated component of `env_var`, joined with that name
3836    ///    (:889-937);
3837    /// 3. `prop_prefix`, the property-list prefix (:938-950);
3838    /// 4. the directory of the path this file was opened by — libhdf5's
3839    ///    `H5F_EXTPATH` (:952-969);
3840    /// 5. the bare relative name, against the process's working directory
3841    ///    (:971-977);
3842    /// 6. the directory of that path *resolved* — libhdf5's
3843    ///    `H5F_ACTUAL_NAME`, which differs from step 4 through a symlink
3844    ///    (:979-1004).
3845    ///
3846    /// Both kinds of cross-file name run this one order and differ only in
3847    /// the two parameters: an external link is `H5F_PREFIX_ELINK` with
3848    /// `HDF5_EXT_PREFIX` and `H5Pset_elink_prefix` (H5Lexternal.c:210-215),
3849    /// a virtual dataset's source is `H5F_PREFIX_VDS` with
3850    /// `HDF5_VDS_PREFIX` and `H5Pset_virtual_prefix` (H5Dvirtual.c:877-882).
3851    fn prefix_open_candidates(
3852        &self,
3853        env_var: &str,
3854        prop_prefix: Option<&Path>,
3855        file: &str,
3856    ) -> Vec<PathBuf> {
3857        let raw = Path::new(file);
3858        let mut candidates = Vec::new();
3859        if raw.is_absolute() {
3860            candidates.push(raw.to_path_buf());
3861        }
3862        // Every attempt after an absolute miss uses the bare file name.
3863        let base: &Path = if raw.is_absolute() {
3864            Path::new(raw.file_name().unwrap_or(raw.as_os_str()))
3865        } else {
3866            raw
3867        };
3868        if let Ok(prefixes) = std::env::var(env_var) {
3869            candidates.extend(
3870                prefixes
3871                    .split(':')
3872                    .filter(|p| !p.is_empty())
3873                    .map(|p| Path::new(p).join(base)),
3874            );
3875        }
3876        if let Some(prefix) = prop_prefix {
3877            candidates.push(prefix.join(base));
3878        }
3879        if let Some(dir) = self.path.parent().filter(|d| !d.as_os_str().is_empty()) {
3880            candidates.push(dir.join(base));
3881        }
3882        candidates.push(base.to_path_buf());
3883        if !self.source_dir.as_os_str().is_empty() {
3884            candidates.push(self.source_dir.join(base));
3885        }
3886        candidates
3887    }
3888
3889    /// Put an external-link prefix in force for this reader and every file
3890    /// it opens on another's behalf. Set once at open, before any name has
3891    /// been resolved, because an external link's answer is fixed the first
3892    /// time it is asked ([`external_resolved`](Self::external_resolved)).
3893    pub(crate) fn set_elink_prefix(&mut self, prefix: Option<String>) {
3894        self.elink_prefix = prefix;
3895    }
3896
3897    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for an
3898    /// external link. The property-list step is
3899    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix)
3900    /// exactly as given: `H5L__extern_traverse` peeks
3901    /// `H5L_ACS_ELINK_PREFIX_NAME` and hands it straight to the search
3902    /// (H5Lexternal.c:210-215), so unlike a virtual dataset's prefix it goes
3903    /// through no `H5D__build_file_prefix` — no `${ORIGIN}` expansion, and
3904    /// `HDF5_EXT_PREFIX` does not shadow it.
3905    fn external_candidates(&self, file: &str) -> Vec<PathBuf> {
3906        let prop = self.elink_prefix.as_deref().map(Path::new);
3907        self.prefix_open_candidates("HDF5_EXT_PREFIX", prop, file)
3908    }
3909
3910    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for a
3911    /// virtual dataset's source. The property-list step is whatever
3912    /// `H5D__build_file_prefix` puts in `dset->shared->vds_prefix`
3913    /// (H5Dint.c:1076-1119): `HDF5_VDS_PREFIX` if the environment names one,
3914    /// otherwise [`DatasetAccess::virtual_prefix`], either way with
3915    /// `${ORIGIN}` expanded.
3916    fn vds_candidates(&self, access: &DatasetAccess, file: &str) -> Vec<PathBuf> {
3917        let prop = resolve_vdsfile_prefix(access.virtual_prefix_value(), &self.source_dir);
3918        self.prefix_open_candidates("HDF5_VDS_PREFIX", prop.as_deref(), file)
3919    }
3920
3921    /// Open one external link's target file, or hand back the handle a
3922    /// previous link to the same resolved path already opened.
3923    fn external_target(&mut self, link: &str, file: &str) -> IoResult<&mut Hdf5Reader> {
3924        // The search runs once per link value; after that the answer is what
3925        // this reader resolved it to, whatever the filesystem does next.
3926        let resolved = match self.external_resolved.get(file) {
3927            Some(resolved) => resolved.clone(),
3928            None => {
3929                let candidates = self.external_candidates(file);
3930                let resolved = candidates
3931                    .iter()
3932                    .find(|p| p.is_file())
3933                    .cloned()
3934                    .ok_or_else(|| crate::io::IoError::ExternalFileNotFound {
3935                        link: link.to_string(),
3936                        file: file.to_string(),
3937                        searched: candidates.iter().map(|p| p.display().to_string()).collect(),
3938                    })?;
3939                self.external_resolved
3940                    .insert(file.to_string(), resolved.clone());
3941                resolved
3942            }
3943        };
3944        self.cross_file(resolved, CrossFileOwner::Reader)
3945    }
3946
3947    /// Open `resolved`, or hand back the handle a previous crossing to the
3948    /// same file already opened.
3949    ///
3950    /// The single owner of every file this reader opens on another file's
3951    /// behalf, so a path named by any number of external links, external
3952    /// references and virtual-dataset sources is opened once and read
3953    /// through one handle. What resolved the name to this path is the
3954    /// caller's business, and differs by kind: an external link and a
3955    /// virtual source each run `H5F_prefix_open_file`'s search order under
3956    /// their own prefix, a reference has no search order at all.
3957    ///
3958    /// `owner` says how long the handle stays open. One path can be reached
3959    /// by both kinds of crossing, and the wider ownership wins: a file an
3960    /// external link holds for this reader's life does not start expiring
3961    /// with a virtual dataset that also names it.
3962    fn cross_file(
3963        &mut self,
3964        resolved: PathBuf,
3965        owner: CrossFileOwner,
3966    ) -> IoResult<&mut Hdf5Reader> {
3967        let locking = self.locking;
3968        let elink_prefix = self.elink_prefix.clone();
3969        match self.external.entry(resolved) {
3970            std::collections::btree_map::Entry::Occupied(e) => {
3971                let e = e.into_mut();
3972                e.owner.widen(owner);
3973                Ok(&mut *e.reader)
3974            }
3975            std::collections::btree_map::Entry::Vacant(e) => {
3976                let mut reader = Hdf5Reader::open_with_locking(e.key(), locking)?;
3977                reader.elink_prefix.clone_from(&elink_prefix);
3978                Ok(&mut *e
3979                    .insert(CrossFileEntry {
3980                        reader: Box::new(reader),
3981                        owner,
3982                    })
3983                    .reader)
3984            }
3985        }
3986    }
3987
3988    /// Close every cross-file handle whose last owning virtual-dataset open
3989    /// has gone, which is where `H5D__virtual_reset_layout` closes the source
3990    /// datasets holding libhdf5's (H5Dvirtual.c:709-710).
3991    ///
3992    /// The single releaser of a [`CrossFileOwner::VirtualOpens`] entry —
3993    /// nothing else removes one, so a source cannot be closed while a handle
3994    /// on the virtual dataset that named it is still alive. Its two callers
3995    /// are the two moments the owning set can be empty: a handle's drop, and
3996    /// an extent resolution run with no handle open at all (this crate
3997    /// resolves at `H5Fopen`, where libhdf5 has nothing to resolve yet).
3998    pub(crate) fn release_closed_virtual_sources(&mut self) {
3999        let dead: Vec<PathBuf> = self
4000            .external
4001            .iter()
4002            .filter(|(_, e)| match &e.owner {
4003                CrossFileOwner::Reader => false,
4004                CrossFileOwner::VirtualOpens(vds) => !vds.iter().any(|v| self.is_open_dataset(v)),
4005            })
4006            .map(|(p, _)| p.clone())
4007            .collect();
4008        for path in dead {
4009            self.external.remove(&path);
4010        }
4011    }
4012
4013    /// Whether the dataset at canonical path `name` still has a live handle
4014    /// — libhdf5's "is this dataset in `H5FO_opened`".
4015    fn is_open_dataset(&self, name: &str) -> bool {
4016        self.dataset_access
4017            .get(name)
4018            .is_some_and(|e| e.open.strong_count() > 0)
4019    }
4020
4021    /// Open the file a virtual mapping's source name points at, or hand back
4022    /// the handle a previous mapping to the same file already opened.
4023    ///
4024    /// A virtual dataset's source is not opened by any path of its own:
4025    /// `H5D__virtual_open_source_dset` hands the name to
4026    /// `H5F_prefix_open_file` (H5Dvirtual.c:877-882) against the *primary*
4027    /// file's external file cache and with `source_fapl`, a copy of the
4028    /// primary file's own file-access property list
4029    /// (H5Dvirtual.c:2193-2194), which carries its `use_file_locking`
4030    /// verbatim (H5Fint.c:389). That is the same cache and the same call
4031    /// external links reach, keyed by the name the open used
4032    /// (H5Fefc.c:245), and it is released back to it with `H5F_efc_close`
4033    /// (H5Dvirtual.c:925-927). So a source file is this reader's
4034    /// [`cross_file`](Self::cross_file) like any other target: one handle
4035    /// per path, under this file's locking policy — measured against
4036    /// libhdf5 1.14.6, reading a cross-file VDS leaves the source flock'd
4037    /// under the default `HDF5_USE_FILE_LOCKING` and unlocked under
4038    /// `HDF5_USE_FILE_LOCKING=FALSE`.
4039    ///
4040    /// What it is *not* is a target this reader holds for its own life:
4041    /// `vds` — the canonical path of the virtual dataset naming the source —
4042    /// owns the handle, and it goes when that dataset's last handle does
4043    /// ([`CrossFileOwner::VirtualOpens`]).
4044    ///
4045    /// `None` where the file cannot be opened at all: `H5F_prefix_open_file`
4046    /// is asked to *try*, and a null source file is "no data there yet",
4047    /// not a failure.
4048    fn vds_source_file(&mut self, vds: &str, file_name: &str) -> Option<&mut Hdf5Reader> {
4049        // A name that has resolved once stays resolved, the same way an
4050        // external link's does — the handle this reader already holds is the
4051        // answer, whatever the filesystem does next. A name that has *not*
4052        // resolved is searched again on the next read, which is what the C
4053        // does too: it re-runs `H5D__virtual_open_source_dset` whenever the
4054        // source dataset is still unopened (H5Dvirtual.c:1421-1423,
4055        // :2558-2561).
4056        let key = (vds.to_string(), file_name.to_string());
4057        let resolved = match self.vds_resolved.get(&key) {
4058            Some(resolved) => resolved.clone(),
4059            None => {
4060                let access = self.access_in_force(vds);
4061                let resolved = self
4062                    .vds_candidates(&access, file_name)
4063                    .into_iter()
4064                    .find(|p| p.is_file())?;
4065                self.vds_resolved.insert(key, resolved.clone());
4066                resolved
4067            }
4068        };
4069        self.cross_file(resolved, CrossFileOwner::virtual_open(vds))
4070            .ok()
4071    }
4072
4073    /// The reader that owns `name`, the path of `name` inside it, and the last
4074    /// external link crossed to get there.
4075    ///
4076    /// This is the single owner of cross-file resolution: it follows external
4077    /// links until the remaining path resolves inside the reader it returns,
4078    /// so callers do exactly one delegation and never have to re-check.
4079    fn external_owner(
4080        &mut self,
4081        name: &str,
4082        hops: usize,
4083    ) -> IoResult<(&mut Self, String, Option<ExternalEdge>)> {
4084        let path = name.trim_start_matches('/').to_string();
4085        let Some(edge) = self.external_edge(&path) else {
4086            return Ok((self, path, None));
4087        };
4088        if hops == 0 {
4089            return Err(crate::io::IoError::InvalidState(format!(
4090                "resolving '{name}' crossed more than {MAX_EXTERNAL_HOPS} external links \
4091                 (libhdf5 stops at the same H5L_NUM_LINKS); the links may form a cycle"
4092            )));
4093        }
4094        let target = self.external_target(&edge.link, &edge.file)?;
4095        let (owner, path, deeper) = target.external_owner(&edge.path, hops - 1)?;
4096        Ok((owner, path, deeper.or(Some(edge))))
4097    }
4098
4099    /// Return metadata for a dataset by name. Like `H5Dopen`, the name may
4100    /// pass through group hard links, soft links and external links.
4101    pub fn dataset_info(&mut self, name: &str) -> Option<&DatasetReadInfo> {
4102        if self.external_edge(name).is_some() {
4103            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS).ok()?;
4104            return owner.dataset_info_local(&path);
4105        }
4106        self.dataset_info_local(name)
4107    }
4108
4109    /// [`dataset_info`](Self::dataset_info) restricted to this file: soft
4110    /// links and group hard links resolve, an external link does not. Every
4111    /// read path uses this, because by then the owning reader has already been
4112    /// selected and the path is local to it.
4113    fn dataset_info_local(&self, name: &str) -> Option<&DatasetReadInfo> {
4114        let name = self.canonical_path(name);
4115        self.datasets.get(&name)
4116    }
4117
4118    /// Where `name` sits in this file's catalog, resolving the same soft
4119    /// links and group hard links [`dataset_info_local`](Self::dataset_info_local)
4120    /// does. Read paths that need the entry more than once take the position
4121    /// once and index with it, rather than walking the path again per lookup.
4122    fn dataset_position(&self, name: &str) -> IoResult<usize> {
4123        let canonical = self.canonical_path(name);
4124        self.datasets
4125            .position(&canonical)
4126            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
4127    }
4128
4129    /// Open `name` as a dataset the way `H5Dopen2` does, reporting *why* it
4130    /// cannot be opened instead of collapsing every cause into absence: a
4131    /// soft link whose target does not exist is a dangling link, and a path
4132    /// through an external link is resolved in the file that link names.
4133    ///
4134    /// This is the gate every typed dataset access goes through. `access` is
4135    /// the dapl the open names; its properties are put in force for the
4136    /// dataset first, so the extent this returns is the one they resolve it
4137    /// to.
4138    ///
4139    /// Returns the token that holds the open alive alongside the extent: the
4140    /// caller must keep it for as long as its handle lives, because the
4141    /// properties this open put in force stay in force exactly that long
4142    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
4143    pub fn open_dataset_with(
4144        &mut self,
4145        name: &str,
4146        access: &DatasetAccess,
4147    ) -> IoResult<(Option<DatasetOpenToken>, &DatasetReadInfo)> {
4148        if self.external_edge(name).is_some() {
4149            let (owner, path, edge) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4150            let open = owner.apply_dataset_access(&path, access)?;
4151            return match owner.open_dataset_local(&path) {
4152                // The name is absent in the target file, which makes the link
4153                // that pointed there dangling — not the caller's path absent.
4154                Err(crate::io::IoError::NotFound(_)) => {
4155                    Err(edge.map_or_else(|| crate::io::IoError::NotFound(path), |e| e.dangling()))
4156                }
4157                other => other.map(|info| (open, info)),
4158            };
4159        }
4160        let open = self.apply_dataset_access(name, access)?;
4161        self.open_dataset_local(name).map(|info| (open, info))
4162    }
4163
4164    /// [`open_dataset`](Self::open_dataset) restricted to this file.
4165    fn open_dataset_local(&self, name: &str) -> IoResult<&DatasetReadInfo> {
4166        let Traversal::Path { path, via } = self.traverse(name) else {
4167            // `external_owner` only ever returns a reader in which the
4168            // remaining path resolves locally, so no caller can land here.
4169            return Err(crate::io::IoError::NotFound(name.to_string()));
4170        };
4171        if let Some(info) = self.datasets.get(&path) {
4172            return Ok(info);
4173        }
4174        if let Some(why) = self.unreadable.get(&path) {
4175            return Err(crate::io::IoError::Unsupported(format!(
4176                "'{name}' is a dataset this crate cannot read: {why}"
4177            )));
4178        }
4179        if let Some(SoftLinkRef { link, target }) = via {
4180            return Err(crate::io::IoError::DanglingLink { link, target });
4181        }
4182        Err(crate::io::IoError::NotFound(name.to_string()))
4183    }
4184
4185    /// Resolve one entry of an attribute list.
4186    ///
4187    /// The cases an attribute list can answer are kept apart here rather than
4188    /// at each call site: decoded, present but undecodable, absent from a set
4189    /// known to be whole, and absent from a set that was never read whole. A
4190    /// caller that collapsed any of the middle cases into the last would
4191    /// report an attribute the file contains as one it does not.
4192    fn resolve_attr<'a>(
4193        attrs: &'a ObjectAttributes,
4194        owner: &str,
4195        name: &str,
4196    ) -> IoResult<&'a AttributeMessage> {
4197        match attrs.entries.iter().find(|a| a.name() == name) {
4198            Some(entry) => entry.decoded().map_err(|reason| {
4199                crate::io::IoError::Unsupported(format!(
4200                    "attribute '{name}' on '{owner}' cannot be decoded: {reason}"
4201                ))
4202            }),
4203            // Not among what was read — but the part that was not read could
4204            // hold it, so an incomplete set cannot answer "absent".
4205            None => match attrs.unreadable_reason() {
4206                Some(reason) => Err(incomplete_error(owner, reason)),
4207                None => Err(crate::io::IoError::NotFound(format!("{owner}:{name}"))),
4208            },
4209        }
4210    }
4211
4212    /// Why the attribute `name` in `attrs` cannot be read, or `None` when it
4213    /// can be — or is not there at all, which the accessors above report as
4214    /// `NotFound`.
4215    fn attr_reason<'a>(attrs: &'a ObjectAttributes, name: &str) -> Option<&'a str> {
4216        attrs
4217            .entries
4218            .iter()
4219            .find(|a| a.name() == name)?
4220            .unreadable_reason()
4221    }
4222
4223    /// Why a dataset's attribute cannot be read, or `None` when it can be.
4224    pub fn dataset_attr_unreadable_reason(
4225        &mut self,
4226        ds_name: &str,
4227        attr_name: &str,
4228    ) -> Option<&str> {
4229        Self::attr_reason(&self.dataset_info(ds_name)?.attributes, attr_name)
4230    }
4231
4232    /// Why a root-level attribute cannot be read, or `None` when it can be.
4233    pub fn root_attr_unreadable_reason(&self, name: &str) -> Option<&str> {
4234        Self::attr_reason(&self.root_attributes, name)
4235    }
4236
4237    /// Why a non-root group's attribute cannot be read, or `None` when it can
4238    /// be.
4239    pub fn group_attr_unreadable_reason(&self, group_path: &str, name: &str) -> Option<&str> {
4240        Self::attr_reason(
4241            self.group_attributes
4242                .get(&self.canonical_path(group_path))?,
4243            name,
4244        )
4245    }
4246
4247    /// Why a dataset's attributes cannot be listed at all, or `None` when the
4248    /// set is whole. Object scope, unlike
4249    /// [`Self::dataset_attr_unreadable_reason`]: the failure belongs to no
4250    /// single name.
4251    pub fn dataset_attrs_unreadable_reason(&mut self, ds_name: &str) -> Option<&str> {
4252        self.dataset_info(ds_name)?.attributes.unreadable_reason()
4253    }
4254
4255    /// A dataset's own compact-vs-dense attribute storage.
4256    pub fn dataset_attr_storage(&mut self, ds_name: &str) -> IoResult<AttributeStorage> {
4257        Ok(self
4258            .dataset_info(ds_name)
4259            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?
4260            .attributes
4261            .storage())
4262    }
4263
4264    /// A dataset's own object-header attribute count.
4265    pub fn dataset_header_attr_count(&mut self, ds_name: &str) -> IoResult<u64> {
4266        let info = self
4267            .dataset_info(ds_name)
4268            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
4269        info.attributes.header_count(ds_name)
4270    }
4271
4272    /// Why the root group's attributes cannot be listed at all, or `None` when
4273    /// the set is whole.
4274    pub fn root_attrs_unreadable_reason(&self) -> Option<&str> {
4275        self.root_attributes.unreadable_reason()
4276    }
4277
4278    /// Why a non-root group's attributes cannot be listed at all, or `None`
4279    /// when the set is whole.
4280    pub fn group_attrs_unreadable_reason(&self, group_path: &str) -> Option<&str> {
4281        self.group_attributes
4282            .get(&self.canonical_path(group_path))?
4283            .unreadable_reason()
4284    }
4285
4286    /// Return the attribute names of a dataset.
4287    ///
4288    /// Includes attributes this crate cannot decode: the object header carries
4289    /// them, so the listing does too. [`Self::dataset_attr`] says why one of
4290    /// those cannot be read. An object whose attribute set could not be read
4291    /// whole has no listing to give and returns the reason instead — see
4292    /// [`Self::dataset_attrs_unreadable_reason`].
4293    pub fn dataset_attr_names(&mut self, name: &str) -> IoResult<Vec<String>> {
4294        let info = self
4295            .dataset_info(name)
4296            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4297        info.attributes.ordered_names(name)
4298    }
4299
4300    /// Return a specific attribute by dataset name and attribute name.
4301    pub fn dataset_attr(&mut self, ds_name: &str, attr_name: &str) -> IoResult<&AttributeMessage> {
4302        let info = self
4303            .dataset_info(ds_name)
4304            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
4305        Self::resolve_attr(&info.attributes, ds_name, attr_name)
4306    }
4307
4308    /// Return the names of root-level (file) attributes, undecodable ones
4309    /// included — see [`Self::dataset_attr_names`].
4310    pub fn root_attr_names(&self) -> IoResult<Vec<String>> {
4311        self.root_attributes.ordered_names("/")
4312    }
4313
4314    /// Return a root-level attribute by name.
4315    pub fn root_attr(&self, name: &str) -> IoResult<&AttributeMessage> {
4316        Self::resolve_attr(&self.root_attributes, "/", name)
4317    }
4318
4319    /// The root group's own attribute creation-order policy.
4320    pub fn root_attr_creation_order(&self) -> CreationOrder {
4321        self.root_attributes.creation_order()
4322    }
4323
4324    /// The root group's own compact-vs-dense attribute storage.
4325    pub fn root_attr_storage(&self) -> AttributeStorage {
4326        self.root_attributes.storage()
4327    }
4328
4329    /// The root group's own object-header attribute count.
4330    pub fn root_header_attr_count(&self) -> IoResult<u64> {
4331        self.root_attributes.header_count("/")
4332    }
4333
4334    /// The root group's own link creation-order policy.
4335    pub fn root_link_creation_order(&self) -> CreationOrder {
4336        self.root_link_storage.1
4337    }
4338
4339    /// The root group's own link storage kind: symbol-table (legacy),
4340    /// compact link messages, or dense (fractal heap plus name index).
4341    pub fn root_link_storage(&self) -> LinkStorage {
4342        self.root_link_storage.0
4343    }
4344
4345    /// Return the attribute names of a non-root group (path without a
4346    /// leading `/`, e.g. `"detector"` or `"entry/instrument"`; may pass
4347    /// through group hard links). Undecodable attributes included — see
4348    /// [`Self::dataset_attr_names`].
4349    pub fn group_attr_names(&mut self, group_path: &str) -> IoResult<Vec<String>> {
4350        if self.external_edge(group_path).is_some() {
4351            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
4352            // The empty remainder is the target file's root group, whose
4353            // attributes are not in the per-group map.
4354            if local.is_empty() {
4355                return owner.root_attr_names();
4356            }
4357            return owner.group_attr_names_local(&local);
4358        }
4359        self.group_attr_names_local(group_path)
4360    }
4361
4362    fn group_attr_names_local(&self, group_path: &str) -> IoResult<Vec<String>> {
4363        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
4364            return Ok(Vec::new());
4365        };
4366        attrs.ordered_names(group_path)
4367    }
4368
4369    /// A non-root group's own attribute creation-order policy. `Untracked`
4370    /// for a path the walk never reached, the same silent default
4371    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
4372    /// unknown group's attribute listing.
4373    pub fn group_attr_creation_order(&self, group_path: &str) -> CreationOrder {
4374        self.group_attributes
4375            .get(&self.canonical_path(group_path))
4376            .map(ObjectAttributes::creation_order)
4377            .unwrap_or_default()
4378    }
4379
4380    /// A non-root group's own compact-vs-dense attribute storage. `Compact`
4381    /// — the same silent default as an empty attribute set — for a path the
4382    /// walk never reached.
4383    pub fn group_attr_storage(&self, group_path: &str) -> AttributeStorage {
4384        self.group_attributes
4385            .get(&self.canonical_path(group_path))
4386            .map(ObjectAttributes::storage)
4387            .unwrap_or_default()
4388    }
4389
4390    /// A non-root group's own object-header attribute count. `0` for a path
4391    /// the walk never reached, the same silent default
4392    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
4393    /// unknown group's attribute listing.
4394    pub fn group_header_attr_count(&self, group_path: &str) -> IoResult<u64> {
4395        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
4396            return Ok(0);
4397        };
4398        attrs.header_count(group_path)
4399    }
4400
4401    /// A non-root group's own link creation-order policy. `Untracked` for a
4402    /// path the walk never reached, the same silent default
4403    /// [`group_attr_creation_order`](Self::group_attr_creation_order) gives.
4404    pub fn group_link_creation_order(&self, group_path: &str) -> CreationOrder {
4405        self.group_link_storage
4406            .get(&self.canonical_path(group_path))
4407            .map_or(CreationOrder::Untracked, |(_, order)| *order)
4408    }
4409
4410    /// A non-root group's own link storage kind. `Compact` — the same
4411    /// silent default as an empty link set — for a path the walk never
4412    /// reached.
4413    pub fn group_link_storage(&self, group_path: &str) -> LinkStorage {
4414        self.group_link_storage
4415            .get(&self.canonical_path(group_path))
4416            .map_or(LinkStorage::Compact, |(storage, _)| *storage)
4417    }
4418
4419    /// Return a non-root group's attribute by name.
4420    pub fn group_attr(&mut self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
4421        if self.external_edge(group_path).is_some() {
4422            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
4423            if local.is_empty() {
4424                return owner.root_attr(name);
4425            }
4426            return owner.group_attr_local(&local, name);
4427        }
4428        self.group_attr_local(group_path, name)
4429    }
4430
4431    fn group_attr_local(&self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
4432        match self.group_attributes.get(&self.canonical_path(group_path)) {
4433            Some(attrs) => Self::resolve_attr(attrs, group_path, name),
4434            // No entry at all: the walk found nothing to record on this group.
4435            None => Err(crate::io::IoError::NotFound(format!("{group_path}:{name}"))),
4436        }
4437    }
4438
4439    /// Return every non-root group path the discovery walk traversed into
4440    /// (no leading `/`). Built from actual link records, so empty groups,
4441    /// attribute-only groups, and subgroup-only groups are all included.
4442    pub fn group_paths(&self) -> &std::collections::BTreeSet<String> {
4443        &self.group_paths
4444    }
4445
4446    /// Report whether a group exists at `group_path` (no leading `/`;
4447    /// may pass through group hard links). The empty string denotes the
4448    /// root group, which always exists.
4449    pub fn has_group(&self, group_path: &str) -> bool {
4450        if group_path.is_empty() || self.group_paths.contains(group_path) {
4451            return true;
4452        }
4453        let canon = self.canonical_path(group_path);
4454        canon.is_empty() || self.group_paths.contains(&canon)
4455    }
4456
4457    /// Read and decode the global-heap collection at `addr`, applying the
4458    /// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
4459    /// signature must be present and the declared size at least
4460    /// `H5HG_MINSIZE` (4096 bytes). There is no upper size cap — libhdf5
4461    /// has none, and this crate's writers put a whole write call's strings
4462    /// into one collection, which a cap would turn into silent data loss.
4463    fn read_heap_collection(&mut self, addr: u64) -> IoResult<GlobalHeapCollection> {
4464        read_heap_collection_from(&mut self.handle, &self.meta.ctx, addr)
4465    }
4466
4467    /// Decode an attribute's value as a string, resolving a variable-length
4468    /// string attribute through the global heap (h5py writes string
4469    /// attributes as variable-length by default).
4470    pub fn attr_string_value(&mut self, attr: &AttributeMessage) -> IoResult<String> {
4471        use crate::format::messages::datatype::DatatypeMessage;
4472        if !matches!(attr.datatype, DatatypeMessage::VarLenString { .. }) {
4473            return fixed_string_attr_value(attr);
4474        }
4475        // Variable-length string: the attribute value is a global-heap
4476        // reference (sequence length + collection address + object index).
4477        if attr.data.len() < vlen_reference_size(&self.meta.ctx) {
4478            return Ok(String::new());
4479        }
4480        let (_seq, coll_addr, obj_index) = decode_vlen_reference(&attr.data, &self.meta.ctx)?;
4481        if coll_addr == UNDEF_ADDR || coll_addr == 0 {
4482            return Ok(String::new());
4483        }
4484        let coll = self.read_heap_collection(coll_addr)?;
4485        let idx = u16::try_from(obj_index).map_err(|_| {
4486            crate::io::IoError::InvalidState(format!(
4487                "global heap object index {obj_index} does not fit the 16-bit on-disk field"
4488            ))
4489        })?;
4490        let obj = coll.get_object(idx).ok_or_else(|| {
4491            crate::io::IoError::InvalidState(format!(
4492                "global heap object {idx} not found in the collection at address {coll_addr:#x}"
4493            ))
4494        })?;
4495        Ok(String::from_utf8_lossy(obj).to_string())
4496    }
4497
4498    /// The absolute path of the object whose header sits at `addr` — what an
4499    /// object reference to it names — or `None` when no group or dataset the
4500    /// discovery walk reached lives there (a reference into a file region the
4501    /// walk never traversed, or a stale one).
4502    pub fn path_for_object(&self, addr: u64) -> Option<&str> {
4503        self.object_paths.get(&addr).map(String::as_str)
4504    }
4505
4506    /// How the object header of `path` stores each message it does not hold
4507    /// privately, as `(message type, storage)` in header order.
4508    ///
4509    /// The observable is the message's flags byte, so this reads the raw
4510    /// header chain: every other read path resolves shared pointers into
4511    /// bodies and clears the flag on the way through, which is exactly the
4512    /// evidence wanted here.
4513    pub fn object_message_storage(&mut self, path: &str) -> IoResult<Vec<(u8, MessageStorage)>> {
4514        let addr = self.object_header_address(path)?;
4515        crate::io::object_header_io::read_header_message_storage(&mut self.handle, &self.meta, addr)
4516    }
4517
4518    /// The flags byte of every message the object header of `path` holds, as
4519    /// `(message type, flags)` in header order, null and continuation
4520    /// messages left out.
4521    ///
4522    /// The byte `h5debug` renders as `<C>`, `<DS>`, `<S>` and the rest
4523    /// (`H5O__debug_real`, H5Odbg.c:409-455). It says which messages the
4524    /// library may cache as never-changing and which it refuses to share, and
4525    /// nothing else in the file records either.
4526    pub fn object_message_flags(&mut self, path: &str) -> IoResult<Vec<(u8, u8)>> {
4527        let addr = self.object_header_address(path)?;
4528        crate::io::object_header_io::read_header_message_flags(&mut self.handle, &self.meta, addr)
4529    }
4530
4531    /// The class and version of every datatype message the object at `path`
4532    /// carries, outermost first; see
4533    /// [`DatatypeMessage::decode_versions`](crate::format::messages::datatype::DatatypeMessage::decode_versions).
4534    pub fn object_datatype_versions(
4535        &mut self,
4536        path: &str,
4537    ) -> IoResult<Vec<crate::format::messages::datatype::DatatypeNodeVersion>> {
4538        let addr = self.object_header_address(path)?;
4539        crate::io::object_header_io::read_header_datatype_versions(
4540            &mut self.handle,
4541            &self.meta,
4542            addr,
4543        )
4544    }
4545
4546    /// Whether the object at `path` records its times —
4547    /// `H5Pget_obj_track_times` on the property list it was created with, read
4548    /// back from the header that answers it; see
4549    /// [`ObjectHeader::recorded_times`](crate::format::object_header::ObjectHeader::recorded_times).
4550    pub fn object_records_times(&mut self, path: &str) -> IoResult<bool> {
4551        let addr = self.object_header_address(path)?;
4552        Ok(crate::io::object_header_io::read_header_recorded_times(
4553            &mut self.handle,
4554            &self.meta,
4555            addr,
4556        )?
4557        .is_some())
4558    }
4559
4560    /// The object header address `path` names.
4561    ///
4562    /// By name first, then by address: `object_paths` keeps one path per
4563    /// object header, so a hard link — two names, one header — is only ever
4564    /// found under whichever name the walk reached first.
4565    fn object_header_address(&mut self, path: &str) -> IoResult<u64> {
4566        if self.external_edge(path).is_some() {
4567            return Err(crate::io::IoError::NotFound(format!(
4568                "{path} is in another file; its header is not this file's to read"
4569            )));
4570        }
4571        match self.dataset_info(path) {
4572            Some(info) => Ok(info.object_header_address),
4573            None => {
4574                let want = absolute_path(&self.canonical_path(path));
4575                self.object_paths
4576                    .iter()
4577                    .find(|(_, p)| **p == want)
4578                    .map(|(addr, _)| *addr)
4579                    .ok_or_else(|| crate::io::IoError::NotFound(path.to_string()))
4580            }
4581        }
4582    }
4583
4584    /// Read a reference dataset's elements, resolved to the objects they name.
4585    pub fn read_references(&mut self, name: &str) -> IoResult<Vec<Reference>> {
4586        let datatype = self
4587            .dataset_info(name)
4588            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?
4589            .datatype
4590            .clone();
4591        let raw = self.read_dataset_raw(name)?;
4592        self.decode_references(&datatype, &raw)
4593    }
4594
4595    /// Read an attribute's value as reference elements.
4596    pub fn attr_references(&mut self, attr: &AttributeMessage) -> IoResult<Vec<Reference>> {
4597        self.decode_references(&attr.datatype, &attr.data)
4598    }
4599
4600    /// The single owner of reference decoding for both carriers of reference
4601    /// elements — dataset payloads and attribute values.
4602    fn decode_references(
4603        &mut self,
4604        datatype: &DatatypeMessage,
4605        bytes: &[u8],
4606    ) -> IoResult<Vec<Reference>> {
4607        let DatatypeMessage::Reference { size, kind } = datatype else {
4608            return Err(crate::io::IoError::InvalidState(format!(
4609                "datatype {datatype} is not a reference"
4610            )));
4611        };
4612        let (size, kind) = (*size as usize, *kind);
4613        if size == 0 {
4614            // A corrupt file can declare it; `chunks_exact(0)` panics.
4615            return Err(crate::io::IoError::InvalidState(
4616                "reference datatype has zero width".into(),
4617            ));
4618        }
4619        let encoding = kind.encoding();
4620
4621        // References written by one call share one heap collection, so read
4622        // each collection once rather than per element.
4623        let mut heaps = std::collections::HashMap::new();
4624        let mut out = Vec::with_capacity(bytes.len() / size);
4625        for elem in bytes.chunks_exact(size) {
4626            out.push(self.decode_reference_element(elem, encoding, &mut heaps)?);
4627        }
4628        Ok(out)
4629    }
4630
4631    /// One reference element, resolved against the file.
4632    ///
4633    /// `heaps` caches the global-heap collections region references point
4634    /// into, keyed by collection address.
4635    fn decode_reference_element(
4636        &mut self,
4637        elem: &[u8],
4638        encoding: ReferenceEncoding,
4639        heaps: &mut std::collections::HashMap<u64, GlobalHeapCollection>,
4640    ) -> IoResult<Reference> {
4641        match encoding {
4642            ReferenceEncoding::Old(OldReferenceKind::Object) => {
4643                match decode_object_element(elem, &self.meta.ctx)? {
4644                    None => Ok(Reference::Null),
4645                    Some(address) => Ok(self.resolve_reference(DecodedReference {
4646                        address,
4647                        file: None,
4648                        target: ReferenceTarget::Object,
4649                    })),
4650                }
4651            }
4652            ReferenceEncoding::Old(OldReferenceKind::DatasetRegion) => {
4653                let Some((coll_addr, obj_index)) = decode_region_element(elem, &self.meta.ctx)?
4654                else {
4655                    return Ok(Reference::Null);
4656                };
4657                let obj = self.heap_object(coll_addr, obj_index, heaps)?;
4658                let (address, selection) = decode_region_heap_object(obj, &self.meta.ctx)?;
4659                Ok(self.resolve_reference(DecodedReference {
4660                    address,
4661                    file: None,
4662                    target: ReferenceTarget::Region(selection),
4663                }))
4664            }
4665            ReferenceEncoding::Revised => {
4666                let (kind, external, body) = match decode_revised_element(elem, &self.meta.ctx)? {
4667                    RevisedElement::Null => return Ok(Reference::Null),
4668                    RevisedElement::Inline { kind, body } => (kind, false, body.to_vec()),
4669                    RevisedElement::Heap {
4670                        kind,
4671                        external,
4672                        collection,
4673                        index,
4674                    } => (
4675                        kind,
4676                        external,
4677                        self.heap_object(collection, index, heaps)?.to_vec(),
4678                    ),
4679                };
4680                match decode_revised_body(kind, external, &body, &self.meta.ctx)? {
4681                    None => Ok(Reference::Null),
4682                    Some(decoded) => Ok(self.resolve_reference(decoded)),
4683                }
4684            }
4685        }
4686    }
4687
4688    /// Attach the target's path to a decoded reference — the one place an
4689    /// address becomes a [`Reference`], so every kind resolves the same way.
4690    ///
4691    /// A reference naming another file is looked up in that file, which this
4692    /// opens by the name the reference carries and nothing else:
4693    /// `H5R__reopen_file` hands the name straight to `H5VL_file_open` with no
4694    /// prefix search, so it is read against the process working directory the
4695    /// way `H5Ropen_object` would read it (H5Rint.c:466, :487). A file that is
4696    /// not there leaves the path unresolved while the reference still names
4697    /// it, which is `H5Rget_file_name` answering from the reference alone
4698    /// while `H5Ropen_object` fails (H5R.c:1036-1039).
4699    fn resolve_reference(&mut self, decoded: DecodedReference) -> Reference {
4700        let DecodedReference {
4701            address,
4702            file,
4703            target,
4704        } = decoded;
4705        let path = match &file {
4706            None => self.path_for_object(address).map(str::to_string),
4707            Some(name) => self
4708                .cross_file(PathBuf::from(name), CrossFileOwner::Reader)
4709                .ok()
4710                .and_then(|target| target.path_for_object(address).map(str::to_string)),
4711        };
4712        match target {
4713            ReferenceTarget::Object => Reference::Object {
4714                address,
4715                file,
4716                path,
4717            },
4718            ReferenceTarget::Region(selection) => Reference::Region {
4719                address,
4720                file,
4721                path,
4722                selection,
4723            },
4724            ReferenceTarget::Attribute(name) => Reference::Attr {
4725                address,
4726                file,
4727                path,
4728                name,
4729            },
4730        }
4731    }
4732
4733    /// One global-heap object, reading its collection at most once.
4734    fn heap_object<'h>(
4735        &mut self,
4736        collection: u64,
4737        index: u32,
4738        heaps: &'h mut std::collections::HashMap<u64, GlobalHeapCollection>,
4739    ) -> IoResult<&'h [u8]> {
4740        if let std::collections::hash_map::Entry::Vacant(slot) = heaps.entry(collection) {
4741            slot.insert(self.read_heap_collection(collection)?);
4742        }
4743        let idx = u16::try_from(index).map_err(|_| {
4744            crate::io::IoError::InvalidState(format!(
4745                "global heap object index {index} does not fit the 16-bit on-disk field"
4746            ))
4747        })?;
4748        heaps[&collection].get_object(idx).ok_or_else(|| {
4749            crate::io::IoError::InvalidState(format!(
4750                "global heap object {idx} not found in the collection at address {collection:#x}"
4751            ))
4752        })
4753    }
4754
4755    /// Return the dimensions of a dataset.
4756    pub fn dataset_shape(&mut self, name: &str) -> IoResult<Vec<u64>> {
4757        let info = self
4758            .dataset_info(name)
4759            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4760        Ok(info.dataspace.dims.clone())
4761    }
4762
4763    /// Logical byte size of a dataset's full image (`product(dims) *
4764    /// element_size`), with the datatype needed for the post-filter conversion.
4765    fn raw_size_and_datatype(&self, name: &str) -> IoResult<(DatatypeMessage, u64)> {
4766        let info = self
4767            .dataset_info_local(name)
4768            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4769        Ok((info.datatype.clone(), Self::raw_size_of(info)))
4770    }
4771
4772    /// Logical byte size of `info`'s full image.
4773    ///
4774    /// The NULL dataspace (`dataspace.is_null()`) holds zero elements — not
4775    /// one, the way an empty `dims` would suggest by the same product-of-dims
4776    /// arithmetic a scalar dataspace uses (`dims` is empty for both).
4777    fn raw_size_of(info: &DatasetReadInfo) -> u64 {
4778        if info.dataspace.is_null() {
4779            0
4780        } else {
4781            saturating_byte_len(&info.dataspace.dims, info.datatype.element_size() as u64)
4782        }
4783    }
4784
4785    /// Logical byte size of a dataset's full image: how many bytes
4786    /// [`read_dataset_raw`](Self::read_dataset_raw) returns, and how large a
4787    /// buffer [`read_dataset_raw_into`](Self::read_dataset_raw_into) needs.
4788    ///
4789    /// Resolved in the file that owns the dataset, so a name crossing an
4790    /// external link answers with the target's size rather than an absence.
4791    pub fn dataset_raw_size(&mut self, name: &str) -> IoResult<u64> {
4792        if self.external_edge(name).is_some() {
4793            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4794            return owner.dataset_raw_size(&path);
4795        }
4796        let info = self
4797            .dataset_info_local(name)
4798            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4799        Ok(Self::raw_size_of(info))
4800    }
4801
4802    /// Everything a zero-copy view of `name` needs to know: the map of the
4803    /// file that owns the dataset, where in that file the dataset's image
4804    /// lies, and how its elements are stored.
4805    ///
4806    /// Facts only — whether they add up to a view `T` may be handed is
4807    /// [`crate::mapped::view`]'s decision, which is the single place that
4808    /// weighs them. Resolved in the file that owns the dataset, so a name
4809    /// crossing an external link answers with the target's map and the
4810    /// target's addresses rather than this file's.
4811    #[cfg(feature = "mmap")]
4812    pub(crate) fn dataset_view_source(&mut self, name: &str) -> IoResult<DatasetViewSource> {
4813        if self.external_edge(name).is_some() {
4814            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4815            return owner.dataset_view_source(&path);
4816        }
4817        let base = self.handle.base();
4818        let map = self.handle.map_snapshot();
4819        let info = self
4820            .dataset_info_local(name)
4821            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4822        let len = Self::raw_size_of(info);
4823        let storage = match &info.layout {
4824            // An external file list overrides contiguous storage: the layout
4825            // still says `Contiguous`, but the bytes are in other files
4826            // (H5Dlayout.c swaps the storage ops out whenever the message is
4827            // present), so nothing in this map holds them.
4828            DataLayoutMessage::Contiguous { .. } if !info.external_files.is_empty() => {
4829                ViewStorage::Elsewhere("its raw data is in external data files")
4830            }
4831            DataLayoutMessage::Contiguous { address, .. } if *address == UNDEF_ADDR => {
4832                ViewStorage::Unallocated
4833            }
4834            DataLayoutMessage::Contiguous { address, .. } => {
4835                let offset = address.checked_add(base).ok_or_else(|| {
4836                    crate::io::IoError::InvalidState(format!(
4837                        "dataset '{name}' claims raw data at {address}, which overflows \
4838                         past the userblock at {base}"
4839                    ))
4840                })?;
4841                ViewStorage::Contiguous { offset, len }
4842            }
4843            DataLayoutMessage::Compact { .. } => {
4844                ViewStorage::Elsewhere("its raw data is compact, stored inside the object header")
4845            }
4846            DataLayoutMessage::ChunkedV3 { .. } | DataLayoutMessage::ChunkedV4 { .. } => {
4847                ViewStorage::Elsewhere("it is chunked")
4848            }
4849            DataLayoutMessage::Virtual { .. } => {
4850                ViewStorage::Elsewhere("it is virtual, mapped from other datasets")
4851            }
4852        };
4853        Ok(DatasetViewSource {
4854            map,
4855            storage,
4856            datatype: info.datatype.clone(),
4857            dims: info.dataspace.dims.clone(),
4858        })
4859    }
4860
4861    /// Read the raw bytes of a dataset.
4862    pub fn read_dataset_raw(&mut self, name: &str) -> IoResult<Vec<u8>> {
4863        if self.external_edge(name).is_some() {
4864            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4865            return owner.read_dataset_raw(&path);
4866        }
4867        let (datatype, total) = self.raw_size_and_datatype(name)?;
4868        read_image_into_new(total as usize, |data| {
4869            self.read_dataset_raw_into_unconverted(name, data, ReadDst::Fresh)?;
4870            Self::apply_post_filter_conversion(data, &datatype)
4871        })
4872    }
4873
4874    /// Read the full raw dataset image into a caller-provided buffer.
4875    ///
4876    /// `out.len()` must equal the dataset's logical byte size
4877    /// (`product(dims) * element_size`); otherwise an error is returned. This
4878    /// is the no-allocation counterpart of [`read_dataset_raw`](Self::read_dataset_raw):
4879    /// the bytes are read straight into `out`, making it the zero-copy entry
4880    /// point for reading directly into a pinned/registered host buffer for an
4881    /// H2D transfer.
4882    pub fn read_dataset_raw_into(&mut self, name: &str, out: &mut [u8]) -> IoResult<()> {
4883        self.read_dataset_raw_into_dst(name, out, ReadDst::Reused)
4884    }
4885
4886    /// [`read_dataset_raw_into`](Self::read_dataset_raw_into) with the
4887    /// caller's destination fact made explicit, for internal callers whose
4888    /// buffer is a fresh allocation rather than a kept one (the allocating
4889    /// wrappers in `dataset.rs` / `swmr.rs`).
4890    pub(crate) fn read_dataset_raw_into_dst(
4891        &mut self,
4892        name: &str,
4893        out: &mut [u8],
4894        dst: ReadDst,
4895    ) -> IoResult<()> {
4896        if self.external_edge(name).is_some() {
4897            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4898            return owner.read_dataset_raw_into_dst(&path, out, dst);
4899        }
4900        let (datatype, total) = self.raw_size_and_datatype(name)?;
4901        if out.len() as u64 != total {
4902            return Err(crate::io::IoError::InvalidState(format!(
4903                "read_dataset_raw_into: buffer is {} bytes but dataset needs {}",
4904                out.len(),
4905                total
4906            )));
4907        }
4908        self.read_dataset_raw_into_unconverted(name, out, dst)?;
4909        Self::apply_post_filter_conversion(out, &datatype)?;
4910        Ok(())
4911    }
4912
4913    /// Fill `out` with the full raw dataset image, before the post-filter
4914    /// datatype conversion. The single owner of read-destination semantics for
4915    /// full reads: it fully defines every byte of `out` (reading allocated data
4916    /// straight in, pre-filling chunked or never-written regions with the tiled
4917    /// fill value), so callers supply only a correctly-sized buffer. Both the
4918    /// allocating `read_dataset_raw` and the zero-copy `read_dataset_raw_into`
4919    /// wrap it and apply the conversion exactly once.
4920    ///
4921    /// `out.len()` must equal `product(dims) * element_size`. `dst` is the
4922    /// caller's destination fact ([`ReadDst`]): whether `out` is a fresh
4923    /// allocation or a buffer the caller keeps across reads, which decides
4924    /// whether a mapped contiguous read is priced at the cold or the warm row.
4925    fn read_dataset_raw_into_unconverted(
4926        &mut self,
4927        name: &str,
4928        out: &mut [u8],
4929        dst: ReadDst,
4930    ) -> IoResult<()> {
4931        let info = self
4932            .dataset_info_local(name)
4933            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4934
4935        // Clone to avoid borrow conflict with &mut self in read methods.
4936        let layout = info.layout.clone();
4937        let pipeline = info.filter_pipeline.clone();
4938        let fill_value = info.fill_value.clone();
4939        let external_files = info.external_files.clone();
4940
4941        match &layout {
4942            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
4943                let prefix = self.extfile_prefix_in_force(name);
4944                read_external_file_bytes(&external_files, prefix.as_deref(), 0, out)?;
4945            }
4946            DataLayoutMessage::Contiguous { address, .. } => {
4947                if *address == UNDEF_ADDR {
4948                    // Never-written contiguous data reads back as the fill value.
4949                    fill_tiled_into(out, fill_value.as_deref());
4950                } else {
4951                    // Read exactly the logical image straight into `out`.
4952                    self.handle.read_exact_at_into(*address, out, dst)?;
4953                }
4954            }
4955            DataLayoutMessage::Compact { data } => {
4956                let n = out.len().min(data.len());
4957                out[..n].copy_from_slice(&data[..n]);
4958                if n < out.len() {
4959                    fill_tiled_into(&mut out[n..], fill_value.as_deref());
4960                }
4961            }
4962            DataLayoutMessage::ChunkedV3 {
4963                chunk_dims,
4964                b_tree_address,
4965            } => {
4966                // The layout's chunk_dims include the element size as the
4967                // trailing dimension. Strip it for chunk indexing. The chunk
4968                // read defines every byte of `out`, filling whatever no chunk
4969                // covers.
4970                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
4971                self.read_chunked_btree_v1(
4972                    name,
4973                    real_chunk_dims,
4974                    *b_tree_address,
4975                    ChunkReadRequest {
4976                        pipeline: pipeline.as_ref(),
4977                        target: ChunkTarget::Full,
4978                        fill_value: fill_value.as_deref(),
4979                        dst,
4980                    },
4981                    out,
4982                )?;
4983            }
4984            DataLayoutMessage::ChunkedV4 {
4985                chunk_dims,
4986                index_address,
4987                index_type,
4988                earray_params,
4989                single_chunk_filter,
4990                ..
4991            } => {
4992                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
4993                self.read_chunked_v4(
4994                    name,
4995                    real_chunk_dims,
4996                    ChunkIndexDesc {
4997                        index_type: *index_type,
4998                        index_address: *index_address,
4999                        earray_params: earray_params.as_ref(),
5000                        single_chunk_filter: *single_chunk_filter,
5001                    },
5002                    ChunkReadRequest {
5003                        pipeline: pipeline.as_ref(),
5004                        target: ChunkTarget::Full,
5005                        fill_value: fill_value.as_deref(),
5006                        dst,
5007                    },
5008                    out,
5009                )?;
5010            }
5011            DataLayoutMessage::Virtual { .. } => {
5012                fill_tiled_into(out, fill_value.as_deref());
5013                self.read_virtual_into(name, out, 0)?;
5014            }
5015        }
5016        Ok(())
5017    }
5018
5019    /// Set every virtual dataset's extent from the sources its unlimited
5020    /// mappings can reach — `H5D__virtual_set_extent_unlim` (H5Dvirtual.c),
5021    /// which libhdf5 runs when it opens such a dataset.
5022    ///
5023    /// INVARIANT: a virtual dataset's `dataspace.dims` are the extent its
5024    /// available sources give it. This is the single owner of that
5025    /// resolution: the dims a VDS with an unlimited mapping reports are not
5026    /// the ones its dataspace message stores, and every shape query, read and
5027    /// slice bound must see the same value, so the resolved extent is stamped
5028    /// in once — here, immediately after a catalog is built — rather than
5029    /// recomputed per call. The per-mapping clip sizes the extent came from
5030    /// are kept beside it in
5031    /// [`virtual_resolution`](DatasetReadInfo::virtual_resolution) so a read
5032    /// walks exactly the sources the extent was derived from.
5033    ///
5034    /// A source that cannot be opened contributes a clip size of 0, never an
5035    /// error: a virtual dataset whose sources are not written yet is legal,
5036    /// and reads back as the fill value (upstream's "clip_size = 0" arm when
5037    /// `H5D__virtual_open_source_dset` leaves the dataset closed).
5038    ///
5039    /// The default view is assumed throughout — `H5D_VDS_LAST_AVAILABLE` is
5040    /// `H5D_ACS_VDS_VIEW_DEF`, and `H5Pset_virtual_view` sets a *dataset
5041    /// access* property that is never stored in the file, so a reader opening
5042    /// a file it did not create always sees the default.
5043    fn resolve_virtual_extents(&mut self) -> IoResult<()> {
5044        let targets: Vec<usize> = self
5045            .datasets
5046            .iter()
5047            .enumerate()
5048            .filter(|(_, d)| d.virtual_mappings.is_some())
5049            .map(|(i, _)| i)
5050            .collect();
5051        for i in targets {
5052            // The catalog is freshly built, so `dataspace.dims` is still the
5053            // extent the dataspace message stores. Record it before the
5054            // resolution replaces it: that is what a later open under other
5055            // access properties has to resolve from.
5056            let stored = self.datasets[i].dataspace.dims.clone();
5057            self.datasets.entry_mut(i).virtual_stored_dims = Some(stored.clone());
5058            self.resolve_virtual_extent_of(i, &stored)?;
5059        }
5060        // This resolution belongs to no dataset open: libhdf5 runs its
5061        // equivalent from `H5D__virtual_init` at `H5Dopen` (H5Dvirtual.c:2178),
5062        // where the open that asked for it holds the source, while this one
5063        // runs at `H5Fopen` and at a SWMR refresh. Whatever it opened is
5064        // therefore unowned the moment it is done — and libhdf5 measured on
5065        // the same file has no source file open after `H5Fopen` either.
5066        self.release_closed_virtual_sources();
5067        Ok(())
5068    }
5069
5070    /// Resolve one virtual dataset's extent from its *stored* dims under the
5071    /// [`DatasetAccess`] in force for it, and stamp both the extent and the
5072    /// per-mapping resolutions in.
5073    fn resolve_virtual_extent_of(&mut self, i: usize, stored: &[u64]) -> IoResult<()> {
5074        let Some(mappings) = self.datasets[i].virtual_mappings.clone() else {
5075            return Ok(());
5076        };
5077        let vds = self.datasets[i].name.clone();
5078        let access = self.access_in_force(&vds);
5079        let (resolution, dims) =
5080            self.resolve_one_virtual_extent(&vds, &mappings, stored, &access)?;
5081        let entry = self.datasets.entry_mut(i);
5082        entry.dataspace.dims = dims;
5083        entry.virtual_resolution = Some(resolution);
5084        Ok(())
5085    }
5086
5087    /// The dataset-access properties in force for `name` (already canonical),
5088    /// libhdf5's defaults when no open has named others.
5089    fn access_in_force(&self, canonical: &str) -> DatasetAccess {
5090        self.dataset_access
5091            .get(canonical)
5092            .map(|e| e.access.clone())
5093            .unwrap_or_default()
5094    }
5095
5096    /// Open `name` under `access`: put those properties in force and
5097    /// re-resolve its extent under them, or — when the dataset already has a
5098    /// live handle — join that open and drop `access` on the floor.
5099    ///
5100    /// First open wins, which is `H5D_open`'s own rule. Only the open that
5101    /// finds no shared info for the dataset runs `H5D__open_oid(dataset,
5102    /// dapl_id)` and so reaches `H5D__virtual_init`, where the view and the
5103    /// printf gap are read out of the dapl into the *shared* layout storage
5104    /// (H5Dvirtual.c:2178-2188); an open that finds the dataset in
5105    /// `H5FO_opened` just points at that shared info and increments its
5106    /// count, never looking at its own dapl at all (H5Dint.c:1496-1500,
5107    /// :1523-1528). The shared info goes away with the last handle, so the
5108    /// next open after that resolves afresh. Measured against libhdf5 1.14.6
5109    /// and 2.0.0: a second open of a printf-gap VDS with a different gap
5110    /// reports the first open's extent and reads the first open's data —
5111    /// even through a second `H5Fopen` of the same file — and only once
5112    /// every handle is closed does a new open see its own gap.
5113    ///
5114    /// The one exception to "first open wins" is the external file prefix,
5115    /// which the joining open is not allowed to disagree about:
5116    /// `H5D__open_name` compares its own expanded prefix against the open
5117    /// dataset's and fails the open when they differ (H5Dint.c:1533-1545).
5118    /// Expanded, so two opens differing only in a property
5119    /// `HDF5_EXTFILE_PREFIX` shadows still agree — measured under libhdf5
5120    /// 1.14.6 and 2.0.0: with that variable set, an open naming no prefix
5121    /// joins one that named a directory, and without it the same pair is
5122    /// refused.
5123    ///
5124    /// Returns the token that keeps the open alive; the caller hands it to
5125    /// the dataset handle it builds. A name no dataset in this file answers
5126    /// to takes nothing and returns `None`.
5127    fn apply_dataset_access(
5128        &mut self,
5129        name: &str,
5130        access: &DatasetAccess,
5131    ) -> IoResult<Option<DatasetOpenToken>> {
5132        let canonical = self.canonical_path(name);
5133        let Some(i) = self.datasets.position(&canonical) else {
5134            return Ok(None);
5135        };
5136        if let Some(open) = self
5137            .dataset_access
5138            .get(&canonical)
5139            .and_then(|e| e.open.upgrade())
5140        {
5141            let in_force = self.extfile_prefix_of(&self.access_in_force(&canonical));
5142            if in_force != self.extfile_prefix_of(access) {
5143                return Err(crate::io::IoError::InvalidState(format!(
5144                    "dataset {canonical:?} is already open under a different external file \
5145                     prefix, and libhdf5 refuses to join an open that disagrees about one"
5146                )));
5147            }
5148            return Ok(Some(open));
5149        }
5150        let token: DatasetOpenToken = std::sync::Arc::new(());
5151        let unchanged = &self.access_in_force(&canonical) == access;
5152        let access = access.clone();
5153        self.dataset_access.insert(
5154            canonical,
5155            AccessInForce {
5156                access,
5157                open: std::sync::Arc::downgrade(&token),
5158            },
5159        );
5160        if !unchanged {
5161            if let Some(stored) = self.datasets[i].virtual_stored_dims.clone() {
5162                // A source may be a virtual dataset in this same file, and the
5163                // access propagates to it (H5Dvirtual.c:2224-2226), so this can
5164                // re-enter; the depth counter is the same cycle guard the
5165                // open-time resolution uses.
5166                VirtualResolveDepth::enter(|| self.resolve_virtual_extent_of(i, &stored))?;
5167            }
5168        }
5169        Ok(Some(token))
5170    }
5171
5172    /// The directory an external file list's stored names are joined against
5173    /// under `access` — `H5D__build_file_prefix(dset, H5F_PREFIX_EFILE)`
5174    /// (H5Dint.c:1084-1090), whose answer libhdf5 keeps in
5175    /// `dset->shared->extfile_prefix`.
5176    fn extfile_prefix_of(&self, access: &DatasetAccess) -> Option<PathBuf> {
5177        resolve_extfile_prefix(access.efile_prefix_value(), &self.source_dir)
5178    }
5179
5180    /// The external file prefix in force for the open dataset `name` — the
5181    /// one the open that is still holding it named, not whatever a later
5182    /// caller might have asked for
5183    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
5184    fn extfile_prefix_in_force(&self, name: &str) -> Option<PathBuf> {
5185        let canonical = self.canonical_path(name);
5186        self.extfile_prefix_of(&self.access_in_force(&canonical))
5187    }
5188
5189    /// [`resolve_virtual_extents`](Self::resolve_virtual_extents) for one
5190    /// dataset: the per-mapping resolutions and the extent they imply.
5191    fn resolve_one_virtual_extent(
5192        &mut self,
5193        vds: &str,
5194        mappings: &VirtualMappingList,
5195        curr_dims: &[u64],
5196        access: &DatasetAccess,
5197    ) -> IoResult<(Vec<MappingResolution>, Vec<u64>)> {
5198        let rank = curr_dims.len();
5199        let mut resolution = Vec::with_capacity(mappings.mappings.len());
5200        let mut new_dims: Vec<Option<u64>> = vec![None; rank];
5201        // `H5D_virtual_update_min_dims`: whatever the unlimited dimension
5202        // resolves to, the extent must still hold every bounded mapping.
5203        let mut min_dims = vec![0u64; rank];
5204        // `H5S_hyper_get_clip_extent_match`'s `incl_trail`: a
5205        // `H5D_VDS_FIRST_MISSING` view stops where the trailing partial
5206        // block would begin (H5Dvirtual.c:1447-1451).
5207        let incl_trail = access.view() == VirtualView::FirstMissing;
5208        // Where two mappings disagree about the unlimited dimension,
5209        // `H5D_VDS_FIRST_MISSING` takes the smallest clip and
5210        // `H5D_VDS_LAST_AVAILABLE` the largest (H5Dvirtual.c:1662-1667).
5211        let take_clip = |slot: &mut Option<u64>, clip: u64| {
5212            if slot.is_none_or(|d| if incl_trail { clip < d } else { clip > d }) {
5213                *slot = Some(clip);
5214            }
5215        };
5216
5217        for m in &mappings.mappings {
5218            let unlim_virtual = m.virtual_selection.unlim_dim();
5219            let res = match (unlim_virtual, m.source_selection.unlim_dim()) {
5220                (Some(vd), Some(sd)) => {
5221                    let source_clip = self
5222                        .virtual_source_dims(vds, m, access)
5223                        .ok()
5224                        .flatten()
5225                        .and_then(|d| d.get(sd).copied())
5226                        .unwrap_or(0);
5227                    let virtual_clip = match (
5228                        regular_hyperslab(&m.virtual_selection),
5229                        regular_hyperslab(&m.source_selection),
5230                    ) {
5231                        // `H5S_hyper_get_clip_extent_match`: how many slices
5232                        // the source supplies, then the virtual extent that
5233                        // covers exactly that many. Its `incl_trail`
5234                        // argument is `view == H5D_VDS_FIRST_MISSING`
5235                        // (H5Dvirtual.c:1447-1451).
5236                        (Some(v), Some(sr)) => {
5237                            v.clip_extent(sr.num_slices(source_clip), incl_trail)
5238                        }
5239                        _ => 0,
5240                    };
5241                    take_clip(&mut new_dims[vd], virtual_clip);
5242                    MappingResolution::Unlimited {
5243                        virtual_clip,
5244                        source_clip,
5245                    }
5246                }
5247                // Unlimited virtual selection, limited source selection:
5248                // the printf shape, where the successive blocks of the
5249                // virtual selection come from successively-named source
5250                // datasets.
5251                (Some(vd), None) => {
5252                    let (blocks, present) = self.printf_blocks_present(vds, m, access);
5253                    let virtual_clip = match (blocks, regular_hyperslab(&m.virtual_selection)) {
5254                        // `H5D__virtual_set_extent_unlim`'s "check for no
5255                        // datasets" arm, which is 0 under either view
5256                        // (H5Dvirtual.c:1623-1626).
5257                        (0, _) | (_, None) => 0,
5258                        // The extent ends just past the last block that has
5259                        // a source under `H5D_VDS_LAST_AVAILABLE`, and where
5260                        // the first missing block starts under
5261                        // `H5D_VDS_FIRST_MISSING` (H5Dvirtual.c:1630-1653).
5262                        (n, Some(r)) => match access.view() {
5263                            VirtualView::LastAvailable => {
5264                                let last = r.unlim_block(n - 1);
5265                                last.start[vd] + last.block[vd]
5266                            }
5267                            VirtualView::FirstMissing => r.unlim_block(n).start[vd],
5268                        },
5269                    };
5270                    take_clip(&mut new_dims[vd], virtual_clip);
5271                    MappingResolution::Printf { blocks, present }
5272                }
5273                _ => MappingResolution::Bounded,
5274            };
5275            if let Some((_, hi)) = m.virtual_selection.bounds() {
5276                for (d, &e) in hi.iter().enumerate().take(rank) {
5277                    if Some(d) != unlim_virtual && e + 1 > min_dims[d] {
5278                        min_dims[d] = e + 1;
5279                    }
5280                }
5281            }
5282            resolution.push(res);
5283        }
5284
5285        let dims = (0..rank)
5286            .map(|d| match new_dims[d] {
5287                Some(v) => v.max(min_dims[d]),
5288                None => curr_dims[d],
5289            })
5290            .collect();
5291        Ok((resolution, dims))
5292    }
5293
5294    /// The extent of the source dataset one mapping names, or `None` when it
5295    /// cannot be reached — `H5D__virtual_open_source_dset` leaving the source
5296    /// closed, which upstream reads as "no data there yet" rather than an
5297    /// error.
5298    fn virtual_source_dims(
5299        &mut self,
5300        vds: &str,
5301        m: &VirtualMapping,
5302        access: &DatasetAccess,
5303    ) -> IoResult<Option<Vec<u64>>> {
5304        let m = built_names(m, 0)?;
5305        Ok(self.source_dims(vds, &m.source_file_name, &m.source_dset_name, access))
5306    }
5307
5308    /// A printf mapping's `first_missing` and the blocks below it that
5309    /// actually have a source — upstream's search loop in
5310    /// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c:1519-1614), which stops
5311    /// at the first block whose source cannot be opened and looks
5312    /// [`DatasetAccess::virtual_printf_gap`] blocks past it before giving up.
5313    ///
5314    /// The loop bound is upstream's `j <= printf_gap + first_missing`
5315    /// rearranged so a large gap cannot overflow the sum: `first_missing` is
5316    /// never above `j` when the test runs, because it only ever becomes the
5317    /// *previous* `j` plus one.
5318    fn printf_blocks_present(
5319        &mut self,
5320        vds: &str,
5321        m: &VirtualMapping,
5322        access: &DatasetAccess,
5323    ) -> (u64, Vec<u64>) {
5324        let gap = access.effective_printf_gap();
5325        let mut first_missing = 0u64;
5326        let mut present = Vec::new();
5327        let mut j = 0u64;
5328        while j - first_missing <= gap {
5329            let Ok(built) = built_names(m, j) else {
5330                break;
5331            };
5332            if self
5333                .source_dims(
5334                    vds,
5335                    &built.source_file_name,
5336                    &built.source_dset_name,
5337                    access,
5338                )
5339                .is_some()
5340            {
5341                first_missing = j + 1;
5342                present.push(j);
5343            }
5344            j += 1;
5345        }
5346        (first_missing, present)
5347    }
5348
5349    /// The extent of one named source dataset, or `None` when the file or
5350    /// the dataset in it cannot be opened.
5351    ///
5352    /// `access` is the virtual dataset's own: `H5D__virtual_init` copies the
5353    /// dapl into the layout as `source_dapl` (H5Dvirtual.c:2224-2226) and
5354    /// every source is opened with it (H5Dvirtual.c:901-902), so a source
5355    /// that is itself a virtual dataset resolves under the same view and
5356    /// printf gap.
5357    fn source_dims(
5358        &mut self,
5359        vds: &str,
5360        file_name: &str,
5361        dset_name: &str,
5362        access: &DatasetAccess,
5363    ) -> Option<Vec<u64>> {
5364        let dset_name = dset_name.trim_start_matches('/');
5365        if file_name == "." {
5366            self.apply_dataset_access(dset_name, access).ok()?;
5367            return self
5368                .dataset_info_local(dset_name)
5369                .map(|i| i.dataspace.dims.clone());
5370        }
5371        let reader = self.vds_source_file(vds, file_name)?;
5372        reader.apply_dataset_access(dset_name, access).ok()?;
5373        reader
5374            .dataset_info(dset_name)
5375            .map(|i| i.dataspace.dims.clone())
5376    }
5377
5378    /// Fill `out` (shaped like the virtual dataset's own extent) by
5379    /// stitching each mapping's source bytes in order (`H5D__virtual_read`,
5380    /// H5Dvirtual.c). `out` must already be pre-filled with the tiled fill
5381    /// value — every element no mapping covers is left exactly as the
5382    /// caller filled it. Mappings apply in list order, so a later mapping's
5383    /// bytes win over an earlier one's on overlap, exactly like the C
5384    /// reader; an unlimited or printf mapping has already been replaced by
5385    /// the concrete mappings its open-time resolution made it
5386    /// (`H5D_VDS_LAST_AVAILABLE`, the default view — see
5387    /// [`Hdf5Reader::resolve_virtual_extents`]).
5388    ///
5389    /// A mapping whose source cannot be opened — the file is absent, or the
5390    /// dataset is not in it — contributes nothing and leaves its virtual
5391    /// region at the fill value, rather than failing the read.
5392    /// `H5D__virtual_open_source_dset` treats both as "no data there yet":
5393    /// it asks `H5F_prefix_open_file` to *try* the file and accepts a null
5394    /// one, and clears the error stack when the dataset is missing
5395    /// (H5Dvirtual.c:877-909); `H5D__virtual_read_one` then performs I/O
5396    /// "only ... if there is a projected memory space, otherwise there were
5397    /// no elements in the projection or the source dataset could not be
5398    /// opened" (H5Dvirtual.c:2661-2665).
5399    ///
5400    /// `depth` counts virtual-dataset nesting — a mapping whose source is
5401    /// itself a virtual dataset, possibly in another file — so a crafted
5402    /// cyclic mapping chain fails cleanly instead of recursing until the
5403    /// stack overflows.
5404    fn read_virtual_into(&mut self, name: &str, out: &mut [u8], depth: usize) -> IoResult<()> {
5405        if depth >= MAX_VIRTUAL_DEPTH {
5406            return Err(crate::io::IoError::InvalidState(format!(
5407                "dataset {name:?}: virtual dataset mapping nests {MAX_VIRTUAL_DEPTH} levels \
5408                 deep, aborting (possible cyclic mapping)"
5409            )));
5410        }
5411        let info = self
5412            .dataset_info(name)
5413            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
5414        let dims = info.dataspace.dims.clone();
5415        let element_size = info.datatype.element_size() as u64;
5416        let Some(mappings) = info.virtual_mappings.clone() else {
5417            // No mapping list written yet: every element is unmapped, and
5418            // `out` is already the fill value the caller pre-filled it with.
5419            return Ok(());
5420        };
5421        // Every unlimited mapping is replaced by the concrete one its
5422        // open-time resolution makes it, so the walk below only ever sees
5423        // bounded selections.
5424        let resolution = info.virtual_resolution.clone().unwrap_or_default();
5425        let mappings = concrete_virtual_mappings(&mappings, &resolution)?;
5426        // The same properties the extent resolved under reach every source
5427        // (H5Dvirtual.c:2224-2226, :901-902), so a source that is itself a
5428        // virtual dataset is read the same way this one is.
5429        let canonical = self.canonical_path(name);
5430        let access = self.access_in_force(&canonical);
5431
5432        for mapping in &mappings {
5433            let virtual_sel = mapping.virtual_selection.resolve(&dims).map_err(|e| {
5434                crate::io::IoError::InvalidState(format!(
5435                    "dataset {name:?}: virtual mapping's virtual selection is not \
5436                     supported: {e}"
5437                ))
5438            })?;
5439            if virtual_sel.runs.is_empty() {
5440                continue;
5441            }
5442
5443            let source_name = mapping.source_dset_name.trim_start_matches('/');
5444
5445            if mapping.source_file_name == "." {
5446                // `H5D__virtual_open_source_dset` opens the source for the
5447                // read and `H5D__virtual_reset_source_dset` closes it again,
5448                // so this open holds nothing past the mapping — dropping the
5449                // token is what that close does.
5450                self.apply_dataset_access(source_name, &access)?;
5451                let Some(src_dims) = self
5452                    .dataset_info(source_name)
5453                    .map(|i| i.dataspace.dims.clone())
5454                else {
5455                    continue;
5456                };
5457                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
5458                    crate::io::IoError::InvalidState(format!(
5459                        "dataset {name:?}: virtual mapping's source selection is not \
5460                         supported: {e}"
5461                    ))
5462                })?;
5463                copy_matched_selections(
5464                    |s, c, buf| {
5465                        // `buf` is `copy_matched_selections`' fresh per-box
5466                        // buffer, never the virtual dataset's own `out`.
5467                        self.read_slice_into_unconverted(
5468                            source_name,
5469                            s,
5470                            c,
5471                            buf,
5472                            depth + 1,
5473                            ReadDst::Fresh,
5474                        )
5475                    },
5476                    &source_sel,
5477                    &virtual_sel,
5478                    element_size,
5479                    out,
5480                )?;
5481            } else {
5482                let Some(src_reader) = self.vds_source_file(&canonical, &mapping.source_file_name)
5483                else {
5484                    continue;
5485                };
5486                src_reader.apply_dataset_access(source_name, &access)?;
5487                let Some(src_dims) = src_reader
5488                    .dataset_info(source_name)
5489                    .map(|i| i.dataspace.dims.clone())
5490                else {
5491                    continue;
5492                };
5493                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
5494                    crate::io::IoError::InvalidState(format!(
5495                        "dataset {name:?}: virtual mapping's source selection is not \
5496                         supported: {e}"
5497                    ))
5498                })?;
5499                copy_matched_selections(
5500                    |s, c, buf| {
5501                        // As above: `buf` is a fresh per-box buffer.
5502                        src_reader.read_slice_into_unconverted(
5503                            source_name,
5504                            s,
5505                            c,
5506                            buf,
5507                            depth + 1,
5508                            ReadDst::Fresh,
5509                        )
5510                    },
5511                    &source_sel,
5512                    &virtual_sel,
5513                    element_size,
5514                    out,
5515                )?;
5516            }
5517        }
5518        Ok(())
5519    }
5520
5521    /// Apply the post-filter datatype conversion (libhdf5's `H5T_convert`
5522    /// step) to a fully-decoded output buffer.
5523    ///
5524    /// For N-bit / reduced-precision `FixedPoint` datatypes the filter
5525    /// pipeline leaves the significant value occupying `bit_precision` bits
5526    /// at `bit_offset` within each element, zero-filled and not
5527    /// sign-extended. This rewrites every element so the value occupies the
5528    /// whole element at bit offset 0, sign-extended when signed. It is a
5529    /// no-op for ordinary full-width datatypes.
5530    fn apply_post_filter_conversion(buffer: &mut [u8], datatype: &DatatypeMessage) -> IoResult<()> {
5531        use crate::format::nbit_scaleoffset::{
5532            apply_datatype_conversion, datatype_needs_bit_conversion,
5533        };
5534        if datatype_needs_bit_conversion(datatype) {
5535            apply_datatype_conversion(buffer, datatype)?;
5536        }
5537        Ok(())
5538    }
5539
5540    /// Re-read the superblock and dataset metadata for SWMR.
5541    ///
5542    /// Call this periodically to pick up new data written by a concurrent
5543    /// SWMR writer. The superblock is re-read to get the latest EOF, then
5544    /// the root group is re-scanned for updated dataset headers (which may
5545    /// contain updated dataspace dimensions and chunk index addresses).
5546    pub fn refresh(&mut self) -> IoResult<()> {
5547        // Whatever the handle reads from must cover the file as the SWMR
5548        // writer has left it: a memory map taken at open ends where the file
5549        // ended then, so it is retaken before a byte of the new metadata is
5550        // decoded. Nothing happens for a handle reading through `pread`.
5551        self.handle.refresh_read_source();
5552
5553        // Re-read superblock to get latest EOF and root group address.
5554        let sb_buf = self.handle.read_at_most(0, 256)?;
5555
5556        // Only v2/v3 superblocks support SWMR refresh
5557        let sb = SuperblockV2V3::decode(&sb_buf)?;
5558
5559        let ctx = FormatContext {
5560            sizeof_addr: sb.sizeof_offsets,
5561            sizeof_size: sb.sizeof_lengths,
5562        };
5563
5564        // The superblock extension can also have changed under SWMR (a new
5565        // free-space or shared-message table), so re-read it before the walk.
5566        let (meta, ext) = Self::read_extension_and_meta(
5567            &mut self.handle,
5568            ctx,
5569            self.meta.btree,
5570            sb.superblock_extension_address,
5571        )?;
5572
5573        // Re-read root group object header, following continuation blocks.
5574        let root_header = Self::read_object_header_full(
5575            &mut self.handle,
5576            &meta,
5577            sb.root_group_object_header_address,
5578        )?;
5579
5580        // Re-scan datasets, group attributes, group paths, and link records.
5581        let catalog = Self::build_catalog(
5582            &mut self.handle,
5583            &meta,
5584            Some(&root_header),
5585            sb.root_group_object_header_address,
5586            None,
5587        )?;
5588
5589        // Root link storage from the freshly re-read header, the same way
5590        // `open_v2v3` derives it at open time — SWMR refresh is v2/v3-only,
5591        // so there is no symbol-table scratch-pad to fall back to here either.
5592        let root_link_storage = describe_link_storage(Some(&root_header), &meta.ctx, None);
5593
5594        self._eof = sb.end_of_file_address;
5595        self.meta = meta;
5596        self.ext = ext;
5597        self.object_paths = catalog.object_paths(sb.root_group_object_header_address);
5598        self.datasets = DatasetTable::new(catalog.datasets);
5599        self.unreadable = catalog.unreadable;
5600        self.root_link_storage = root_link_storage;
5601        self.group_attributes = catalog.group_attributes;
5602        self.group_link_storage = catalog.group_link_storage;
5603        self.group_paths = catalog.group_paths;
5604        self.group_aliases = catalog.group_aliases;
5605        self.links = catalog.links;
5606        self.datatypes = catalog.datatypes;
5607        // The catalog is freshly built, so every virtual dataset's resolved
5608        // extent went with the old one — a SWMR refresh is exactly when a
5609        // source may have grown.
5610        self.resolve_virtual_extents()?;
5611
5612        Ok(())
5613    }
5614
5615    /// The dataset at `pos`'s chunk index, from the cache when it already
5616    /// holds one decoded from `index_address`, otherwise by running `decode`
5617    /// once and keeping what it returns.
5618    ///
5619    /// The single owner of the chunk-index cache: nothing else reads or
5620    /// writes it. Every chunked read of a dataset asks the same question of
5621    /// the same on-disk structure — a thousand small slice reads re-walked
5622    /// the fixed array a thousand times — and only a change to the catalog
5623    /// entry can change the answer, which drops the entry's cache
5624    /// (`DatasetTable::entry_mut`, and a SWMR refresh rebuilding the table).
5625    fn decoded_chunk_index<F>(
5626        &mut self,
5627        pos: usize,
5628        index_address: u64,
5629        decode: F,
5630    ) -> IoResult<std::sync::Arc<DecodedChunkIndex>>
5631    where
5632        F: FnOnce(&mut Self) -> IoResult<DecodedChunkIndex>,
5633    {
5634        if let Some(hit) = self.datasets.chunk_index(pos, index_address) {
5635            return Ok(std::sync::Arc::clone(hit));
5636        }
5637        let decoded = decode(self)?;
5638        Ok(self.datasets.cache_chunk_index(pos, index_address, decoded))
5639    }
5640
5641    /// Read chunked dataset data by walking the chunk index.
5642    ///
5643    /// `desc` bundles the version-4 chunk-index descriptor extracted from the
5644    /// data-layout message (kind, address, and per-kind parameters), so this
5645    /// entry point takes one descriptor rather than a long parameter list.
5646    ///
5647    /// Scatters only; `output` must already be sized to the target extent.
5648    /// Every byte of it is defined before this returns `Ok`: what no chunk
5649    /// covers is filled with the tiled fill value by
5650    /// [`place_chunk_jobs`], the one exit every branch here takes.
5651    fn read_chunked_v4(
5652        &mut self,
5653        name: &str,
5654        chunk_dims: &[u64],
5655        desc: ChunkIndexDesc<'_>,
5656        req: ChunkReadRequest,
5657        output: &mut [u8],
5658    ) -> IoResult<()> {
5659        let ChunkReadRequest {
5660            pipeline, target, ..
5661        } = req;
5662        let ChunkIndexDesc {
5663            index_type,
5664            index_address,
5665            earray_params,
5666            single_chunk_filter,
5667        } = desc;
5668        let pos = self.dataset_position(name)?;
5669        let info = &self.datasets[pos];
5670        let dims = info.dataspace.dims.clone();
5671        let element_size = info.datatype.element_size() as u64;
5672
5673        match index_type {
5674            data_layout::ChunkIndexType::SingleChunk => {
5675                // Single chunk: the index_address IS the chunk address
5676                let total_size: u64 = saturating_byte_len(&dims, element_size);
5677                let geo = ChunkOutputGeometry {
5678                    dims: &dims,
5679                    chunk_dims,
5680                    element_size,
5681                };
5682                if index_address == UNDEF_ADDR || total_size == 0 {
5683                    // Unallocated single chunk: nothing to read, so the empty
5684                    // plan makes the whole output fill.
5685                    return place_chunk_jobs(
5686                        &self.handle,
5687                        Vec::new(),
5688                        &[],
5689                        req,
5690                        &geo,
5691                        None,
5692                        output,
5693                    );
5694                }
5695                // A filtered single chunk records its exact on-disk size and
5696                // per-chunk filter mask in the layout message. Use them to
5697                // read precisely the stored bytes and to skip the filters the
5698                // mask marks as not applied. Without those params
5699                // (older/edge layouts) fall back to the read-extra-and-inflate
5700                // heuristic with the full pipeline.
5701                let job = match (pipeline, single_chunk_filter) {
5702                    (Some(_), Some(scf)) => ChunkReadJob {
5703                        addr: index_address,
5704                        len: scf.nbytes as usize,
5705                        at_most: false,
5706                        mask: scf.filter_mask,
5707                    },
5708                    (Some(_), None) => ChunkReadJob {
5709                        addr: index_address,
5710                        len: total_size.saturating_mul(2) as usize,
5711                        at_most: true,
5712                        mask: 0,
5713                    },
5714                    (None, _) => ChunkReadJob {
5715                        addr: index_address,
5716                        len: total_size as usize,
5717                        at_most: false,
5718                        mask: 0,
5719                    },
5720                };
5721                // The lone chunk spans the whole dataset; place it respecting
5722                // the dataset extent (Full) or the selection (Slice). This
5723                // index type has no on-disk structure to decode — the layout
5724                // message is the index — but it is still recorded through the
5725                // one owner, so that its chunk's image reaches the cache by
5726                // the same route every other index type's chunks do.
5727                let index = self.decoded_chunk_index(pos, index_address, |_| {
5728                    Ok(DecodedChunkIndex::new(
5729                        vec![(job.addr, job.len as u64, job.mask)],
5730                        vec![0u64; dims.len()],
5731                    ))
5732                })?;
5733                place_chunk_jobs(
5734                    &self.handle,
5735                    vec![Some(job)],
5736                    &index.coords,
5737                    req,
5738                    &geo,
5739                    Some(&index.images),
5740                    output,
5741                )
5742            }
5743            data_layout::ChunkIndexType::Implicit => {
5744                self.read_chunked_implicit(name, chunk_dims, index_address, req, output)
5745            }
5746            data_layout::ChunkIndexType::FixedArray => {
5747                self.read_chunked_fixed_array(name, chunk_dims, index_address, req, output)
5748            }
5749            data_layout::ChunkIndexType::BTreeV2 => {
5750                self.read_chunked_btree_v2(name, chunk_dims, index_address, req, output)
5751            }
5752            data_layout::ChunkIndexType::ExtensibleArray => {
5753                let params = earray_params.ok_or_else(|| {
5754                    crate::io::IoError::InvalidState("missing earray params".into())
5755                })?;
5756
5757                if index_address == UNDEF_ADDR {
5758                    // Unallocated: the empty plan makes the whole output fill.
5759                    let geo = ChunkOutputGeometry {
5760                        dims: &dims,
5761                        chunk_dims,
5762                        element_size,
5763                    };
5764                    return place_chunk_jobs(
5765                        &self.handle,
5766                        Vec::new(),
5767                        &[],
5768                        req,
5769                        &geo,
5770                        None,
5771                        output,
5772                    );
5773                }
5774
5775                // Total slot count of the index grid. The maximum extent
5776                // decides the multipliers (libhdf5 max_down_chunks); an
5777                // unlimited dimension 0 is bounded by the current extent for
5778                // this read — a slot beyond it (written before a shrink) is
5779                // not visible.
5780                let max_dims = self.datasets[pos].dataspace.max_dims.clone();
5781
5782                // Chunks are placed N-dimensionally: each slot decodes
5783                // (row-major, against the index grid) to chunk-grid
5784                // coordinates, so sub-frame chunks (a chunk smaller than a
5785                // full frame) land correctly.
5786                let rank = dims.len();
5787                let index = self.decoded_chunk_index(pos, index_address, |reader| {
5788                    let grid =
5789                        crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
5790                    let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
5791                    let mut entries = reader.collect_ea_chunk_entries(
5792                        index_address,
5793                        params,
5794                        &dims,
5795                        max_dims.as_deref(),
5796                        chunk_dims,
5797                        element_size,
5798                    )?;
5799                    entries.truncate(std::cmp::min(chunks_total as usize, entries.len()));
5800                    let coords = crate::io::chunk_grid::coords_table(
5801                        &dims,
5802                        max_dims.as_deref(),
5803                        chunk_dims,
5804                        entries.len(),
5805                    )?;
5806                    Ok(DecodedChunkIndex::new(entries, coords))
5807                })?;
5808                let chunk_entries = &index.entries;
5809                let slot_coords = &index.coords;
5810                let chunk_coords = |i: usize| -> &[u64] { &slot_coords[i * rank..(i + 1) * rank] };
5811
5812                // Build one read job per chunk (no I/O yet), then read +
5813                // decompress them together (in parallel where positioned reads
5814                // are race-free), then scatter serially. Filtered chunks record
5815                // their exact on-disk size, so read exactly that; unfiltered
5816                // chunks read at-most since the entry size can exceed the file
5817                // tail. Skip conditions differ between the two, so build jobs
5818                // per branch.
5819                let jobs: Vec<Option<ChunkReadJob>> = if pipeline.is_some() {
5820                    let file_size = self.handle.file_size()?;
5821                    chunk_entries
5822                        .iter()
5823                        .enumerate()
5824                        .map(|(i, &(addr, nbytes, mask))| {
5825                            if addr == UNDEF_ADDR
5826                                || nbytes == 0
5827                                || addr >= file_size
5828                                || nbytes > file_size
5829                                || !target.overlaps(chunk_coords(i), chunk_dims)
5830                            {
5831                                None
5832                            } else {
5833                                Some(ChunkReadJob {
5834                                    addr,
5835                                    len: nbytes as usize,
5836                                    at_most: false,
5837                                    mask,
5838                                })
5839                            }
5840                        })
5841                        .collect()
5842                } else {
5843                    chunk_entries
5844                        .iter()
5845                        .enumerate()
5846                        .map(|(i, &(addr, nbytes, _))| {
5847                            if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(i), chunk_dims) {
5848                                None
5849                            } else {
5850                                Some(ChunkReadJob {
5851                                    addr,
5852                                    len: nbytes as usize,
5853                                    at_most: true,
5854                                    mask: 0,
5855                                })
5856                            }
5857                        })
5858                        .collect()
5859                };
5860
5861                let geo = ChunkOutputGeometry {
5862                    dims: &dims,
5863                    chunk_dims,
5864                    element_size,
5865                };
5866                place_chunk_jobs(
5867                    &self.handle,
5868                    jobs,
5869                    slot_coords,
5870                    req,
5871                    &geo,
5872                    Some(&index.images),
5873                    output,
5874                )
5875            }
5876        }
5877    }
5878
5879    /// Collect a fixed-array dataset's per-chunk `(address, on-disk byte
5880    /// count, filter mask)` entries, indexed by index-grid linear slot
5881    /// ([`crate::io::chunk_grid`]). Empty when the index or its data block is
5882    /// unallocated.
5883    ///
5884    /// Shared by the full/slice chunked reader
5885    /// ([`read_chunked_fixed_array`](Self::read_chunked_fixed_array)) and the
5886    /// direct single-chunk read
5887    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the fixed-array
5888    /// wire format has one decoder.
5889    fn collect_fa_chunk_entries(
5890        &mut self,
5891        chunk_dims: &[u64],
5892        ndims: usize,
5893        element_size: u64,
5894        index_address: u64,
5895    ) -> IoResult<Vec<(u64, u64, u32)>> {
5896        use crate::format::chunk_index::fixed_array::*;
5897
5898        if index_address == UNDEF_ADDR {
5899            // Unallocated: no chunks recorded.
5900            return Ok(Vec::new());
5901        }
5902
5903        // Read FA header
5904        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
5905        let fa_hdr = FixedArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;
5906
5907        if fa_hdr.data_blk_addr == UNDEF_ADDR {
5908            // Unallocated data block: no chunks recorded.
5909            return Ok(Vec::new());
5910        }
5911
5912        // The chunk shape (from the layout message) must match the
5913        // dataspace rank; otherwise the chunk-grid indexing panics.
5914        if chunk_dims.len() != ndims {
5915            return Err(crate::io::IoError::InvalidState(format!(
5916                "fixed-array dataset rank {} does not match chunk rank {}",
5917                ndims,
5918                chunk_dims.len()
5919            )));
5920        }
5921
5922        let is_filtered = fa_hdr.client_id == FA_CLIENT_FILT_CHUNK;
5923        let sizeof_addr = self.meta.ctx.sizeof_addr as usize;
5924        // chunk_size_len = element_size - sizeof_addr - filter_mask(4)
5925        let chunk_size_len = if is_filtered {
5926            (fa_hdr.element_size as usize)
5927                .checked_sub(sizeof_addr + 4)
5928                .ok_or_else(|| {
5929                    crate::io::IoError::InvalidState(
5930                        "fixed array filtered element_size too small".into(),
5931                    )
5932                })?
5933        } else {
5934            0
5935        };
5936        // The compressed-size field is read into a u64; reject a width that
5937        // would overflow the read_size helper.
5938        if chunk_size_len > 8 {
5939            return Err(crate::io::IoError::InvalidState(format!(
5940                "fixed array filtered chunk-size width {chunk_size_len} exceeds 8 bytes"
5941            )));
5942        }
5943
5944        // Compute chunk byte size
5945        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
5946
5947        // Bytes one element takes in the data block (or a page of it).
5948        let elem_size = if is_filtered {
5949            sizeof_addr + chunk_size_len + 4
5950        } else {
5951            sizeof_addr
5952        };
5953        // The element count is the header's claim, and everything below is
5954        // sized from it: refuse one whose elements could not all be in the
5955        // file before it sizes a reservation, a read, or a page count.
5956        let file_size = self.handle.file_size()?;
5957        let num_elmts = usize::try_from(fa_hdr.num_elmts)
5958            .ok()
5959            .filter(|&n| {
5960                (n as u64)
5961                    .checked_mul(elem_size as u64)
5962                    .is_some_and(|bytes| bytes <= file_size)
5963            })
5964            .ok_or_else(|| {
5965                crate::io::IoError::InvalidState(format!(
5966                    "fixed array declares {} elements of {elem_size} bytes, more than the \
5967                     {file_size}-byte file holds",
5968                    fa_hdr.num_elmts
5969                ))
5970            })?;
5971
5972        // Collect per-chunk (address, compressed_size). compressed_size is the
5973        // exact on-disk byte count for filtered chunks, or chunk_bytes when
5974        // unfiltered.
5975        // (chunk address, on-disk byte count, filter mask). The mask is the
5976        // per-chunk filter mask for filtered chunks, 0 when unfiltered.
5977        let mut chunk_entries: Vec<(u64, u64, u32)> = Vec::with_capacity(num_elmts);
5978
5979        if fa_hdr.is_paged() {
5980            // Paged data block: prefix (with page-init bitmap) followed by pages.
5981            let npages = fa_hdr.npages();
5982            let dblk_page_nelmts = fa_hdr.dblk_page_nelmts();
5983            let prefix_len = 4 + 1 + 1 + sizeof_addr + (npages as usize).div_ceil(8) + 4;
5984            let prefix_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, prefix_len)?;
5985            let prefix = FixedArrayPagedPrefix::decode(&prefix_buf, &self.meta.ctx, npages)?;
5986
5987            // All pages have the same on-disk stride; only the last page holds
5988            // fewer elements (libhdf5: dblk_page_size is constant).
5989            let page_stride = dblk_page_nelmts as usize * elem_size + 4;
5990            let pages_base = fa_hdr.data_blk_addr + prefix.prefix_size as u64;
5991
5992            for p in 0..npages as usize {
5993                // Elements on this page (last page may be short).
5994                let page_nelmts = if p + 1 == npages as usize {
5995                    let rem = fa_hdr.num_elmts % dblk_page_nelmts;
5996                    if rem == 0 {
5997                        dblk_page_nelmts
5998                    } else {
5999                        rem
6000                    }
6001                } else {
6002                    dblk_page_nelmts
6003                } as usize;
6004
6005                if !prefix.page_initialized(p) {
6006                    // Uninitialized page: all chunk entries are undefined.
6007                    chunk_entries
6008                        .extend(std::iter::repeat_n((UNDEF_ADDR, 0u64, 0u32), page_nelmts));
6009                    continue;
6010                }
6011
6012                let page_addr = pages_base + (p as u64) * page_stride as u64;
6013                let page_size = page_nelmts * elem_size + 4;
6014                let page_buf = self.handle.read_at_most(page_addr, page_size)?;
6015
6016                if is_filtered {
6017                    let elems = decode_filtered_page(
6018                        &page_buf,
6019                        &self.meta.ctx,
6020                        page_nelmts,
6021                        chunk_size_len,
6022                    )?;
6023                    for e in elems {
6024                        chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
6025                    }
6026                } else {
6027                    let addrs = decode_unfiltered_page(&page_buf, &self.meta.ctx, page_nelmts)?;
6028                    for addr in addrs {
6029                        chunk_entries.push((addr, chunk_bytes, 0));
6030                    }
6031                }
6032            }
6033        } else {
6034            // Non-paged data block: all elements live inline in the data block.
6035            let dblk_size = 4 + 1 + 1 + sizeof_addr + num_elmts * elem_size + 4;
6036            let dblk_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, dblk_size)?;
6037
6038            if is_filtered {
6039                let fa_dblk = FixedArrayDataBlock::decode_filtered(
6040                    &dblk_buf,
6041                    &self.meta.ctx,
6042                    num_elmts,
6043                    chunk_size_len,
6044                )?;
6045                for e in &fa_dblk.filtered_elements {
6046                    chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
6047                }
6048            } else {
6049                let fa_dblk =
6050                    FixedArrayDataBlock::decode_unfiltered(&dblk_buf, &self.meta.ctx, num_elmts)?;
6051                for &addr in &fa_dblk.elements {
6052                    chunk_entries.push((addr, chunk_bytes, 0));
6053                }
6054            }
6055        }
6056
6057        Ok(chunk_entries)
6058    }
6059
6060    /// Read a dataset indexed by a fixed array.
6061    ///
6062    /// Scatters only; `output` must already be sized to the target extent.
6063    /// Every byte of it is defined before this returns `Ok`: what no chunk
6064    /// covers is filled with the tiled fill value by
6065    /// [`place_chunk_jobs`], the one exit every branch here takes.
6066    fn read_chunked_fixed_array(
6067        &mut self,
6068        name: &str,
6069        chunk_dims: &[u64],
6070        index_address: u64,
6071        req: ChunkReadRequest,
6072        output: &mut [u8],
6073    ) -> IoResult<()> {
6074        let ChunkReadRequest {
6075            pipeline, target, ..
6076        } = req;
6077        let pos = self.dataset_position(name)?;
6078        let info = &self.datasets[pos];
6079        let dims = info.dataspace.dims.clone();
6080        let element_size = info.datatype.element_size() as u64;
6081        let max_dims = info.dataspace.max_dims.clone();
6082        let ndims = dims.len();
6083        let geo = ChunkOutputGeometry {
6084            dims: &dims,
6085            chunk_dims,
6086            element_size,
6087        };
6088
6089        // Index-grid slot -> chunk-grid coordinates (row-major, against the
6090        // maximum extent — the array was sized from its chunk grid, so a slot
6091        // beyond the current extent still decodes to its true position and
6092        // then simply falls outside the read target). A zero chunk dimension
6093        // from a malformed layout message is rejected inside.
6094        let index = self.decoded_chunk_index(pos, index_address, |reader| {
6095            let entries =
6096                reader.collect_fa_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
6097            let coords = crate::io::chunk_grid::coords_table(
6098                &dims,
6099                max_dims.as_deref(),
6100                chunk_dims,
6101                entries.len(),
6102            )?;
6103            Ok(DecodedChunkIndex::new(entries, coords))
6104        })?;
6105        if index.entries.is_empty() {
6106            // Unallocated index/data block: the empty plan makes the whole
6107            // output fill.
6108            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6109        }
6110        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6111        let chunk_coords = |i: usize| -> &[u64] { &index.coords[i * ndims..(i + 1) * ndims] };
6112
6113        // Build one read job per chunk (no I/O yet). Filtered chunks carry
6114        // their exact compressed size (read at-most, since a zero size means
6115        // "unknown" and falls back to a generous estimate); unfiltered chunks
6116        // read the exact chunk byte count. For a slice, chunks outside the
6117        // selection become None and are never read.
6118        let jobs: Vec<Option<ChunkReadJob>> = index
6119            .entries
6120            .iter()
6121            .enumerate()
6122            .map(|(linear_idx, &(addr, comp_size, mask))| {
6123                if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(linear_idx), chunk_dims) {
6124                    None
6125                } else if pipeline.is_some() {
6126                    let read_len = if comp_size > 0 {
6127                        comp_size as usize
6128                    } else {
6129                        chunk_bytes as usize * 2
6130                    };
6131                    Some(ChunkReadJob {
6132                        addr,
6133                        len: read_len,
6134                        at_most: true,
6135                        mask,
6136                    })
6137                } else {
6138                    Some(ChunkReadJob {
6139                        addr,
6140                        len: chunk_bytes as usize,
6141                        at_most: false,
6142                        mask,
6143                    })
6144                }
6145            })
6146            .collect();
6147
6148        // Read and place each chunk: the selected byte runs straight out of
6149        // the file where that beats reading the chunk whole.
6150        place_chunk_jobs(
6151            &self.handle,
6152            jobs,
6153            &index.coords,
6154            req,
6155            &geo,
6156            Some(&index.images),
6157            output,
6158        )
6159    }
6160
6161    /// Read a dataset indexed by the implicit ("none") chunk index.
6162    ///
6163    /// There is no on-disk index structure at all (`H5Dnone.c`): every chunk
6164    /// slot in the maximum-extent grid is allocated in one block at dataset
6165    /// creation, so a chunk's address is purely arithmetic — `index_address +
6166    /// slot * chunk_bytes`, where `slot` is its row-major position in the
6167    /// same maximum-extent grid the fixed/extensible-array/v2-B-tree indexes
6168    /// use (`H5D__chunk_set_info_real`'s `max_down_chunks`). This index type
6169    /// is only ever selected for a fixed (non-unlimited) chunked dataset with
6170    /// early allocation and no filters, so there is no per-chunk allocation
6171    /// flag, compressed size, or filter mask to track.
6172    ///
6173    /// Scatters only; `output` must already be sized to the target extent.
6174    /// Every byte of it is defined before this returns `Ok`: what no chunk
6175    /// covers is filled with the tiled fill value by
6176    /// [`place_chunk_jobs`], the one exit every branch here takes.
6177    fn read_chunked_implicit(
6178        &mut self,
6179        name: &str,
6180        chunk_dims: &[u64],
6181        index_address: u64,
6182        req: ChunkReadRequest,
6183        output: &mut [u8],
6184    ) -> IoResult<()> {
6185        let target = req.target;
6186        let pos = self.dataset_position(name)?;
6187        let info = &self.datasets[pos];
6188        let dims = info.dataspace.dims.clone();
6189        let element_size = info.datatype.element_size() as u64;
6190        let ndims = dims.len();
6191
6192        if index_address == UNDEF_ADDR {
6193            // Unallocated: the empty plan makes the whole output fill.
6194            let geo = ChunkOutputGeometry {
6195                dims: &dims,
6196                chunk_dims,
6197                element_size,
6198            };
6199            return place_chunk_jobs(
6200                &self.handle,
6201                Vec::new(),
6202                &[],
6203                ChunkReadRequest {
6204                    pipeline: None,
6205                    ..req
6206                },
6207                &geo,
6208                None,
6209                output,
6210            );
6211        }
6212
6213        if chunk_dims.len() != ndims {
6214            return Err(crate::io::IoError::InvalidState(format!(
6215                "implicit-index dataset rank {} does not match chunk rank {}",
6216                ndims,
6217                chunk_dims.len()
6218            )));
6219        }
6220
6221        let max_dims = self.datasets[pos].dataspace.max_dims.clone();
6222        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6223
6224        // Every slot's coordinates and address are computed directly, not
6225        // read from disk, so there is no "unallocated chunk" case here the
6226        // way a sparse index has one: `at_most: false` because
6227        // `H5D__none_idx_create` guarantees the whole block is present.
6228        let index = self.decoded_chunk_index(pos, index_address, |_| {
6229            let grid = crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
6230            let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
6231            let coords = crate::io::chunk_grid::coords_table(
6232                &dims,
6233                max_dims.as_deref(),
6234                chunk_dims,
6235                chunks_total as usize,
6236            )?;
6237            let entries = (0..chunks_total)
6238                .map(|i| (index_address + i * chunk_bytes, chunk_bytes, 0))
6239                .collect();
6240            Ok(DecodedChunkIndex::new(entries, coords))
6241        })?;
6242        let slot_coords = &index.coords;
6243
6244        let jobs: Vec<Option<ChunkReadJob>> = index
6245            .entries
6246            .iter()
6247            .enumerate()
6248            .map(|(i, &(addr, len, _))| {
6249                let coords = &slot_coords[i * ndims..(i + 1) * ndims];
6250                if !target.overlaps(coords, chunk_dims) {
6251                    None
6252                } else {
6253                    Some(ChunkReadJob {
6254                        addr,
6255                        len: len as usize,
6256                        at_most: false,
6257                        mask: 0,
6258                    })
6259                }
6260            })
6261            .collect();
6262
6263        let geo = ChunkOutputGeometry {
6264            dims: &dims,
6265            chunk_dims,
6266            element_size,
6267        };
6268        place_chunk_jobs(
6269            &self.handle,
6270            jobs,
6271            slot_coords,
6272            ChunkReadRequest {
6273                pipeline: None,
6274                ..req
6275            },
6276            &geo,
6277            Some(&index.images),
6278            output,
6279        )
6280    }
6281
6282    /// Collect a v2-B-tree-indexed dataset's per-chunk `(address, read
6283    /// size, scaled chunk-grid offsets, filter mask)` entries, walking the
6284    /// tree once. `read_size` is the compressed size for a filtered chunk,
6285    /// the full chunk size otherwise. Empty when the index has no records.
6286    ///
6287    /// Shared by the full/slice chunked reader
6288    /// ([`read_chunked_btree_v2`](Self::read_chunked_btree_v2)) and the
6289    /// direct single-chunk read
6290    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the v2 B-tree
6291    /// record format has one decoder.
6292    fn collect_bt2_chunk_entries(
6293        &mut self,
6294        chunk_dims: &[u64],
6295        ndims: usize,
6296        element_size: u64,
6297        index_address: u64,
6298    ) -> IoResult<Vec<Bt2ChunkEntry>> {
6299        use crate::format::chunk_index::btree_v2::*;
6300
6301        if index_address == UNDEF_ADDR {
6302            // Unallocated: no chunks recorded.
6303            return Ok(Vec::new());
6304        }
6305
6306        // Read BT2 header
6307        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
6308        let bt2_hdr = Bt2Header::decode(&hdr_buf, &self.meta.ctx)?;
6309
6310        if bt2_hdr.root_node_addr == UNDEF_ADDR || bt2_hdr.total_num_records == 0 {
6311            // No records.
6312            return Ok(Vec::new());
6313        }
6314
6315        // Walk the B-tree to any depth, collecting every record's raw bytes
6316        // from the internal nodes and leaves.
6317        let ctx = self.meta.ctx;
6318        let record_bytes = collect_btree_v2_records(
6319            &bt2_hdr,
6320            &ctx,
6321            &mut HandleBlockReader {
6322                handle: &mut self.handle,
6323            },
6324        )?;
6325        let total_records = if bt2_hdr.record_size > 0 {
6326            record_bytes.len() / bt2_hdr.record_size as usize
6327        } else {
6328            0
6329        };
6330
6331        // Decode records
6332        // Compute chunk byte size
6333        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6334
6335        // Unify filtered and unfiltered records into (address, read_size,
6336        // scaled offsets, filter mask). read_size is the compressed size for
6337        // filtered chunks, the full chunk size otherwise; the mask is 0 for
6338        // unfiltered records.
6339        let entries: Vec<Bt2ChunkEntry> = if bt2_hdr.record_type == BT2_TYPE_CHUNK_UNFILT {
6340            Bt2ChunkIndex::decode_unfiltered_records(
6341                &record_bytes,
6342                total_records,
6343                ndims,
6344                &self.meta.ctx,
6345            )?
6346            .into_iter()
6347            .map(|r| (r.chunk_address, chunk_bytes as usize, r.scaled_offsets, 0))
6348            .collect()
6349        } else {
6350            Bt2ChunkIndex::decode_filtered_records(
6351                &record_bytes,
6352                total_records,
6353                ndims,
6354                bt2_hdr.record_size,
6355                &self.meta.ctx,
6356            )?
6357            .into_iter()
6358            .map(|r| {
6359                (
6360                    r.chunk_address,
6361                    r.chunk_size as usize,
6362                    r.scaled_offsets,
6363                    r.filter_mask,
6364                )
6365            })
6366            .collect()
6367        };
6368
6369        Ok(entries)
6370    }
6371
6372    /// Read a dataset indexed by a B-tree v2.
6373    ///
6374    /// Scatters only; `output` must already be sized to the target extent.
6375    /// Every byte of it is defined before this returns `Ok`: what no chunk
6376    /// covers is filled with the tiled fill value by
6377    /// [`place_chunk_jobs`], the one exit every branch here takes.
6378    fn read_chunked_btree_v2(
6379        &mut self,
6380        name: &str,
6381        chunk_dims: &[u64],
6382        index_address: u64,
6383        req: ChunkReadRequest,
6384        output: &mut [u8],
6385    ) -> IoResult<()> {
6386        let target = req.target;
6387        let pos = self.dataset_position(name)?;
6388        let info = &self.datasets[pos];
6389        let dims = info.dataspace.dims.clone();
6390        let element_size = info.datatype.element_size() as u64;
6391        let ndims = dims.len();
6392        let geo = ChunkOutputGeometry {
6393            dims: &dims,
6394            chunk_dims,
6395            element_size,
6396        };
6397
6398        // A record carries its own scaled (chunk-grid) offsets, always `ndims`
6399        // of them, so flattening them into the coordinate table loses nothing.
6400        let index = self.decoded_chunk_index(pos, index_address, |reader| {
6401            let records =
6402                reader.collect_bt2_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
6403            let mut entries = Vec::with_capacity(records.len());
6404            let mut coords = Vec::with_capacity(records.len() * ndims);
6405            for (addr, read_size, scaled, mask) in records {
6406                entries.push((addr, read_size as u64, mask));
6407                coords.extend_from_slice(&scaled);
6408            }
6409            Ok(DecodedChunkIndex::new(entries, coords))
6410        })?;
6411        if index.entries.is_empty() {
6412            // Unallocated index or no records: the empty plan makes the whole
6413            // output fill.
6414            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6415        }
6416
6417        // Build one read job per chunk (no I/O yet), placing each by its
6418        // scaled offsets. For a slice, chunks outside the selection become
6419        // None and are never read.
6420        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
6421        for (i, &(addr, read_size, mask)) in index.entries.iter().enumerate() {
6422            let scaled = &index.coords[i * ndims..(i + 1) * ndims];
6423            if addr == UNDEF_ADDR || read_size == 0 || !target.overlaps(scaled, chunk_dims) {
6424                jobs.push(None);
6425            } else {
6426                jobs.push(Some(ChunkReadJob {
6427                    addr,
6428                    len: read_size as usize,
6429                    at_most: false,
6430                    mask,
6431                }));
6432            }
6433        }
6434
6435        // Read and place each chunk N-dimensionally by its scaled offsets.
6436        place_chunk_jobs(
6437            &self.handle,
6438            jobs,
6439            &index.coords,
6440            req,
6441            &geo,
6442            Some(&index.images),
6443            output,
6444        )
6445    }
6446
6447    /// Read a chunked dataset indexed by a version-1 B-tree (layout
6448    /// message version 3, class 2 "chunked").
6449    ///
6450    /// `chunk_dims` excludes the trailing element-size dimension.
6451    ///
6452    /// Scatters only; `output` must already be sized to the target extent.
6453    /// Every byte of it is defined before this returns `Ok`: what no chunk
6454    /// covers is filled with the tiled fill value by
6455    /// [`place_chunk_jobs`], the one exit every branch here takes.
6456    fn read_chunked_btree_v1(
6457        &mut self,
6458        name: &str,
6459        chunk_dims: &[u64],
6460        b_tree_address: u64,
6461        req: ChunkReadRequest,
6462        output: &mut [u8],
6463    ) -> IoResult<()> {
6464        let ChunkReadRequest {
6465            pipeline, target, ..
6466        } = req;
6467        let pos = self.dataset_position(name)?;
6468        let info = &self.datasets[pos];
6469        let dims = info.dataspace.dims.clone();
6470        let element_size = info.datatype.element_size() as u64;
6471        let ndims = dims.len();
6472
6473        // The chunk shape must match the dataspace rank or the chunk-grid
6474        // indexing below panics.
6475        if chunk_dims.len() != ndims {
6476            return Err(crate::io::IoError::InvalidState(format!(
6477                "B-tree-v1 dataset rank {} does not match chunk rank {}",
6478                ndims,
6479                chunk_dims.len()
6480            )));
6481        }
6482
6483        let total_size: u64 = saturating_byte_len(&dims, element_size);
6484        if b_tree_address == UNDEF_ADDR || total_size == 0 {
6485            // Unallocated: the empty plan makes the whole output fill.
6486            let geo = ChunkOutputGeometry {
6487                dims: &dims,
6488                chunk_dims,
6489                element_size,
6490            };
6491            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6492        }
6493
6494        // Walk the B-tree, collecting every leaf entry as
6495        // (element_offsets, chunk_address, chunk_size, filter_mask), and turn
6496        // each key's element offsets into chunk-grid coordinates. The keys
6497        // carry rank + 1 offsets; the trailing element-size dimension offset
6498        // is always 0 and is dropped.
6499        let file_size = self.handle.file_size()?;
6500        let index = self.decoded_chunk_index(pos, b_tree_address, |reader| {
6501            let mut leaves: Vec<(Vec<u64>, u64, u32, u32)> = Vec::new();
6502            reader.collect_btree_v1_chunks(b_tree_address, ndims, file_size, 0, &mut leaves)?;
6503            let mut entries = Vec::with_capacity(leaves.len());
6504            let mut coords = Vec::with_capacity(leaves.len() * ndims);
6505            for (offsets, addr, chunk_size, mask) in leaves {
6506                for (d, &cd) in chunk_dims.iter().enumerate().take(ndims) {
6507                    coords.push(offsets[d].checked_div(cd).unwrap_or(0));
6508                }
6509                entries.push((addr, chunk_size as u64, mask));
6510            }
6511            Ok(DecodedChunkIndex::new(entries, coords))
6512        })?;
6513
6514        // The uncompressed byte size of a full chunk.
6515        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6516
6517        // Build one read job per chunk (no I/O yet), placing each by its
6518        // scaled coordinates.
6519        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
6520        for (i, &(addr, chunk_size, mask)) in index.entries.iter().enumerate() {
6521            let skip = addr == UNDEF_ADDR
6522                || chunk_size == 0
6523                || addr >= file_size
6524                || chunk_size > file_size
6525                || !target.overlaps(&index.coords[i * ndims..(i + 1) * ndims], chunk_dims);
6526            jobs.push(if skip {
6527                None
6528            } else {
6529                Some(ChunkReadJob {
6530                    addr,
6531                    len: chunk_size as usize,
6532                    at_most: false,
6533                    mask,
6534                })
6535            });
6536        }
6537
6538        // Read and place each chunk N-dimensionally by its scaled offsets.
6539        let geo = ChunkOutputGeometry {
6540            dims: &dims,
6541            chunk_dims,
6542            element_size,
6543        };
6544        place_chunk_jobs(
6545            &self.handle,
6546            jobs,
6547            &index.coords,
6548            req,
6549            &geo,
6550            Some(&index.images),
6551            output,
6552        )?;
6553
6554        // libhdf5 stores raw byte sizes; verify the uncompressed chunk
6555        // size is consistent for unfiltered datasets so a corrupt index
6556        // surfaces instead of silently producing garbage.
6557        if pipeline.is_none() {
6558            for &(addr, chunk_size, _) in &index.entries {
6559                if addr != UNDEF_ADDR && chunk_size != chunk_bytes && chunk_size != 0 {
6560                    return Err(crate::io::IoError::InvalidState(format!(
6561                        "chunk B-tree v1: unfiltered chunk size {} != expected {}",
6562                        chunk_size, chunk_bytes
6563                    )));
6564                }
6565            }
6566        }
6567
6568        Ok(())
6569    }
6570
6571    /// Recursively walk a version-1 raw-data-chunk B-tree, collecting every
6572    /// leaf entry as `(element_offsets, chunk_address, chunk_size,
6573    /// filter_mask)`.
6574    ///
6575    /// `rank` is the chunk rank excluding the trailing element-size
6576    /// dimension. Recursion is bounded by the node level read from disk:
6577    /// each recursive step descends to a strictly lower level, and the
6578    /// `depth` counter caps the descent at the 1-byte level field's range.
6579    fn collect_btree_v1_chunks(
6580        &mut self,
6581        addr: u64,
6582        rank: usize,
6583        file_size: u64,
6584        depth: u32,
6585        out: &mut Vec<(Vec<u64>, u64, u32, u32)>,
6586    ) -> IoResult<()> {
6587        // A v1 B-tree node level fits in one byte, so the tree can be at
6588        // most 256 levels deep; this also stops cyclic/corrupt indices.
6589        if depth > 256 {
6590            return Err(crate::io::IoError::InvalidState(
6591                "chunk B-tree v1 exceeds maximum depth".into(),
6592            ));
6593        }
6594        if addr == UNDEF_ADDR || addr >= file_size {
6595            return Ok(());
6596        }
6597
6598        // A node is a fixed-size record: the header (8 + 2*sizeof_addr) plus
6599        // `2 * chunk_internal_k` interleaved keys/children.
6600        let sa = self.meta.ctx.sizeof_addr as usize;
6601        let node_size = self.meta.btree.chunk_btree_node_size(sa, rank);
6602        let buf = self.handle.read_at_most(addr, node_size)?;
6603        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.meta.btree.chunk_max_entries())?;
6604
6605        if node.level == 0 {
6606            // Leaf node: each child points at chunk data.
6607            for (i, &child_addr) in node.children.iter().enumerate() {
6608                let key = &node.keys[i];
6609                out.push((
6610                    key.offsets[..rank].to_vec(),
6611                    child_addr,
6612                    key.chunk_size,
6613                    key.filter_mask,
6614                ));
6615            }
6616        } else {
6617            // Internal node: each child points at a sub-TREE node one
6618            // level below. `node.level` is read from disk and decreases on
6619            // every descent, so it also bounds the recursion.
6620            let children: Vec<u64> = node.children.clone();
6621            for child_addr in children {
6622                if child_addr == UNDEF_ADDR || child_addr >= file_size {
6623                    continue;
6624                }
6625                self.collect_btree_v1_chunks(child_addr, rank, file_size, depth + 1, out)?;
6626            }
6627        }
6628        Ok(())
6629    }
6630
6631    /// Read variable-length string data from a dataset.
6632    ///
6633    /// h5py stores vlen strings as global heap references. Each element
6634    /// in the raw data is a (collection_address, object_index) pair that
6635    /// points to a string blob in a global heap collection.
6636    ///
6637    /// Returns a Vec<String> with one entry per element.
6638    pub fn read_vlen_strings(&mut self, name: &str) -> IoResult<Vec<String>> {
6639        // A vlen string is a vlen sequence of `u8` reinterpreted as UTF-8.
6640        // The global-heap walk is identical; decode the raw object bytes.
6641        Ok(self
6642            .read_vlen_objects(name)?
6643            .into_iter()
6644            .map(|bytes| String::from_utf8_lossy(&bytes).to_string())
6645            .collect())
6646    }
6647
6648    /// Read a 1-D variable-length byte-array dataset (vlen sequence of `u8`).
6649    ///
6650    /// Returns a `Vec<Vec<u8>>` with one byte array per element. Missing or
6651    /// undefined references yield an empty `Vec`.
6652    pub fn read_vlen_bytes(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
6653        self.read_vlen_objects(name)
6654    }
6655
6656    /// Shared owner of the global-heap walk for variable-length datasets.
6657    ///
6658    /// Each element of the raw data is a vlen reference (collection address +
6659    /// object index) into a global heap collection. Returns the raw object
6660    /// bytes for each element, with an empty `Vec` for undefined/missing
6661    /// references. Both `read_vlen_strings` (UTF-8 view) and `read_vlen_bytes`
6662    /// (raw view) layer on top of this.
6663    fn read_vlen_objects(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
6664        if self.external_edge(name).is_some() {
6665            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
6666            return owner.read_vlen_objects(&path);
6667        }
6668        let info = self
6669            .dataset_info_local(name)
6670            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
6671        let dims = info.dataspace.dims.clone();
6672        let layout = info.layout.clone();
6673        let external_files = info.external_files.clone();
6674        let total_elements: u64 = dims.iter().fold(1u64, |acc, &d| acc.saturating_mul(d));
6675
6676        let raw = match &layout {
6677            DataLayoutMessage::Contiguous { size, .. } if !external_files.is_empty() => {
6678                let prefix = self.extfile_prefix_in_force(name);
6679                let mut buf = vec![0u8; *size as usize];
6680                read_external_file_bytes(&external_files, prefix.as_deref(), 0, &mut buf)?;
6681                buf
6682            }
6683            DataLayoutMessage::Contiguous { address, size } => {
6684                if *address == UNDEF_ADDR {
6685                    return Ok(vec![]);
6686                }
6687                self.handle.read_at(*address, *size as usize)?
6688            }
6689            DataLayoutMessage::Compact { data } => data.clone(),
6690            _ => {
6691                // For chunked, read the full dataset first
6692                self.read_dataset_raw(name)?
6693            }
6694        };
6695
6696        let ref_size = vlen_reference_size(&self.meta.ctx);
6697        // The extent is the file's claim; the loop below stops at the image
6698        // it actually read, so the reservation is bounded by that image.
6699        let mut items: Vec<Vec<u8>> =
6700            Vec::with_capacity((total_elements as usize).min(raw.len() / ref_size));
6701
6702        // Cache global heap collections to avoid re-reading.
6703        // Store as (collection, index→offset lookup) for O(1) object access.
6704        let mut heap_cache: std::collections::HashMap<
6705            u64,
6706            (GlobalHeapCollection, std::collections::HashMap<u16, usize>),
6707        > = std::collections::HashMap::new();
6708
6709        for i in 0..total_elements as usize {
6710            let offset = i * ref_size;
6711            if offset + ref_size > raw.len() {
6712                break;
6713            }
6714
6715            let (_seq_len, collection_addr, obj_index) =
6716                decode_vlen_reference(&raw[offset..], &self.meta.ctx)?;
6717
6718            if collection_addr == UNDEF_ADDR || collection_addr == 0 {
6719                items.push(Vec::new());
6720                continue;
6721            }
6722
6723            // Read or get cached global heap collection
6724            if let std::collections::hash_map::Entry::Vacant(e) = heap_cache.entry(collection_addr)
6725            {
6726                let coll = self.read_heap_collection(collection_addr)?;
6727                let lookup: std::collections::HashMap<u16, usize> = coll
6728                    .objects
6729                    .iter()
6730                    .enumerate()
6731                    .map(|(i, o)| (o.index, i))
6732                    .collect();
6733                e.insert((coll, lookup));
6734            }
6735
6736            let idx = u16::try_from(obj_index).map_err(|_| {
6737                crate::io::IoError::InvalidState(format!(
6738                    "global heap object index {obj_index} does not fit the 16-bit on-disk field \
6739                     (element {i} of \"{name}\")"
6740                ))
6741            })?;
6742            let (coll, lookup) = &heap_cache[&collection_addr];
6743            let &oi = lookup.get(&idx).ok_or_else(|| {
6744                crate::io::IoError::InvalidState(format!(
6745                    "global heap object {idx} not found in the collection at address \
6746                     {collection_addr:#x} (element {i} of \"{name}\")"
6747                ))
6748            })?;
6749            items.push(coll.objects[oi].data.clone());
6750        }
6751
6752        Ok(items)
6753    }
6754
6755    /// Collect chunk (address, size) entries from an EA index.
6756    /// Returns a vector indexed by chunk linear index.
6757    fn collect_ea_chunk_entries(
6758        &mut self,
6759        index_address: u64,
6760        params: &data_layout::EarrayParams,
6761        dims: &[u64],
6762        max_dims: Option<&[u64]>,
6763        chunk_dims: &[u64],
6764        element_size: u64,
6765    ) -> IoResult<Vec<(u64, u64, u32)>> {
6766        use crate::format::chunk_index::extensible_array::{self as ea, *};
6767
6768        if index_address == UNDEF_ADDR {
6769            return Ok(vec![]);
6770        }
6771        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
6772        let ea_hdr = ExtensibleArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;
6773        if ea_hdr.idx_blk_addr == UNDEF_ADDR {
6774            return Ok(vec![]);
6775        }
6776
6777        // Slot count of the index grid, bounding the collection walk. The
6778        // maximum extent decides the multipliers (sub-frame chunks make this
6779        // larger than the dim-0 chunk count alone); the unlimited dimension 0
6780        // is bounded by the current extent.
6781        let chunks_dim0: usize = crate::io::chunk_grid::index_grid(dims, max_dims, chunk_dims)?
6782            .iter()
6783            .fold(1usize, |acc, &n| acc.saturating_mul(n as usize));
6784        let geo = EaGeometry::new(
6785            params.idx_blk_elmts,
6786            params.data_blk_min_elmts,
6787            params.sup_blk_min_data_ptrs,
6788            params.max_nelmts_bits,
6789            params.max_dblk_page_nelmts_bits,
6790        )?;
6791        let chunk_bytes = saturating_byte_len(chunk_dims, element_size);
6792        let is_filtered = ea_hdr.class_id == ea::EA_CLS_FILT_CHUNK;
6793        let chunk_size_len = if is_filtered {
6794            ea_hdr.raw_elmt_size - self.meta.ctx.sizeof_addr - 4
6795        } else {
6796            0
6797        };
6798        let max_nelmts_bits = params.max_nelmts_bits;
6799        // Each entry is (chunk address, on-disk byte count, filter mask). The
6800        // mask is the per-chunk filter mask for filtered datasets (0 for an
6801        // unfiltered index, where it is meaningless).
6802        let mut entries: Vec<(u64, u64, u32)> = Vec::new();
6803
6804        // Read the index block: direct elements + the data-block / super-block
6805        // address arrays (the address arrays are filter-agnostic).
6806        let (dblk_addrs, sblk_addrs): (Vec<u64>, Vec<u64>) = if is_filtered {
6807            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
6808            let fiblk = ea::FilteredIndexBlock::decode(
6809                &buf,
6810                &self.meta.ctx,
6811                params.idx_blk_elmts as usize,
6812                geo.ndblk_addrs,
6813                geo.nsblk_addrs,
6814                chunk_size_len,
6815            )?;
6816            for e in &fiblk.elements {
6817                entries.push((e.addr, e.nbytes, e.filter_mask));
6818            }
6819            (fiblk.dblk_addrs, fiblk.sblk_addrs)
6820        } else {
6821            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
6822            let iblk = ExtensibleArrayIndexBlock::decode(
6823                &buf,
6824                &self.meta.ctx,
6825                params.idx_blk_elmts as usize,
6826                geo.ndblk_addrs,
6827                geo.nsblk_addrs,
6828            )?;
6829            for &addr in &iblk.elements {
6830                entries.push((addr, chunk_bytes, 0));
6831            }
6832            (iblk.dblk_addrs, iblk.sblk_addrs)
6833        };
6834
6835        // Walk super blocks in order, collecting each data block's entries.
6836        let sa = self.meta.ctx.sizeof_addr as usize;
6837        let raw_elmt_size = if is_filtered {
6838            ea::FilteredChunkEntry::raw_size(self.meta.ctx.sizeof_addr, chunk_size_len) as usize
6839        } else {
6840            sa
6841        };
6842        'outer: for (u, s) in geo.sblk.iter().enumerate() {
6843            if entries.len() >= chunks_dim0 {
6844                break;
6845            }
6846            let dblk_nelmts = s.dblk_nelmts as usize;
6847            let paged = geo.is_sblk_paged(u);
6848
6849            // This super block's data-block addresses, plus its page-init
6850            // bitmap region (empty unless the super block is paged).
6851            let (this_dblk_addrs, page_init): (Vec<u64>, Vec<u8>) = if u < geo.iblock_nsblks {
6852                let start = s.start_dblk as usize;
6853                (
6854                    (0..s.ndblks as usize)
6855                        .map(|d| dblk_addrs.get(start + d).copied().unwrap_or(UNDEF_ADDR))
6856                        .collect(),
6857                    Vec::new(),
6858                )
6859            } else {
6860                let sblk_addr = sblk_addrs
6861                    .get(u - geo.iblock_nsblks)
6862                    .copied()
6863                    .unwrap_or(UNDEF_ADDR);
6864                if sblk_addr == UNDEF_ADDR {
6865                    (vec![UNDEF_ADDR; s.ndblks as usize], Vec::new())
6866                } else {
6867                    let page_init_total = if paged {
6868                        s.ndblks as usize * geo.dblk_page_init_size(u)
6869                    } else {
6870                        0
6871                    };
6872                    // Size the read from the super block's geometry rather
6873                    // than a fixed cap: signature+version+class+header_addr
6874                    // + block_offset(<=8) + page-init bitmaps
6875                    // + ndblks data-block addresses + checksum.
6876                    let sblk_size =
6877                        4 + 1 + 1 + sa + 8 + page_init_total + s.ndblks as usize * sa + 4;
6878                    let buf = self.handle.read_at_most(sblk_addr, sblk_size)?;
6879                    let sb = ExtensibleArraySuperBlock::decode(
6880                        &buf,
6881                        &self.meta.ctx,
6882                        max_nelmts_bits,
6883                        s.ndblks as usize,
6884                        page_init_total,
6885                    )?;
6886                    (sb.dblk_addrs, sb.page_init)
6887                }
6888            };
6889
6890            let npages = geo.npages(u) as usize;
6891            let page_size = geo.dblk_page_size(raw_elmt_size);
6892            let prefix = geo.dblk_prefix_size(self.meta.ctx.sizeof_addr, max_nelmts_bits);
6893
6894            for (d, &dblk_addr) in this_dblk_addrs.iter().enumerate() {
6895                if dblk_addr == UNDEF_ADDR {
6896                    entries.extend(std::iter::repeat_n((UNDEF_ADDR, 0, 0), dblk_nelmts));
6897                } else if paged {
6898                    // Paged data block: only a prefix lives at `dblk_addr`;
6899                    // the elements live in `npages` page structures that
6900                    // follow it on disk. The super block's page-init bitmap
6901                    // is one flat MSB-first bitmap (H5VM bit ops) indexed by
6902                    // `dblk_idx * npages + page_idx` (H5EA.c), not a series
6903                    // of per-data-block sub-bitmaps.
6904                    for p in 0..npages {
6905                        let bit = d * npages + p;
6906                        let initialized = page_init[bit / 8] & (0x80u8 >> (bit % 8)) != 0;
6907                        if !initialized {
6908                            entries.extend(std::iter::repeat_n(
6909                                (UNDEF_ADDR, 0, 0),
6910                                geo.dblk_page_nelmts as usize,
6911                            ));
6912                            continue;
6913                        }
6914                        let page_addr = dblk_addr + prefix as u64 + (p as u64) * page_size as u64;
6915                        let page = self.handle.read_at(page_addr, page_size)?;
6916                        for k in 0..geo.dblk_page_nelmts as usize {
6917                            let off = k * raw_elmt_size;
6918                            if is_filtered {
6919                                let e = ea::FilteredChunkEntry::decode(
6920                                    &page[off..],
6921                                    sa,
6922                                    chunk_size_len as usize,
6923                                );
6924                                entries.push((e.addr, e.nbytes, e.filter_mask));
6925                            } else {
6926                                entries.push((read_addr(&page[off..], sa), chunk_bytes, 0));
6927                            }
6928                        }
6929                    }
6930                } else if is_filtered {
6931                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
6932                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
6933                    let dblk = ea::FilteredDataBlock::decode(
6934                        &buf,
6935                        &self.meta.ctx,
6936                        max_nelmts_bits,
6937                        dblk_nelmts,
6938                        chunk_size_len,
6939                    )?;
6940                    for e in &dblk.elements {
6941                        entries.push((e.addr, e.nbytes, e.filter_mask));
6942                    }
6943                } else {
6944                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
6945                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
6946                    let dblk = ExtensibleArrayDataBlock::decode(
6947                        &buf,
6948                        &self.meta.ctx,
6949                        max_nelmts_bits,
6950                        dblk_nelmts,
6951                    )?;
6952                    for &addr in &dblk.elements {
6953                        entries.push((addr, chunk_bytes, 0));
6954                    }
6955                }
6956                if entries.len() >= chunks_dim0 {
6957                    break 'outer;
6958                }
6959            }
6960        }
6961        Ok(entries)
6962    }
6963
6964    /// Read a slice (hyperslab) of a dataset, whatever its layout.
6965    ///
6966    /// Contiguous and compact datasets are read run by run; a chunked one
6967    /// reads only the chunks the selection overlaps, and any gap the writer
6968    /// never filled comes back as the fill value.
6969    ///
6970    /// `starts` and `counts` define the N-dimensional selection:
6971    /// starts[d] is the first index along dim d, counts[d] is how many.
6972    /// Returns the selected data in row-major order.
6973    pub fn read_slice(&mut self, name: &str, starts: &[u64], counts: &[u64]) -> IoResult<Vec<u8>> {
6974        if self.external_edge(name).is_some() {
6975            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
6976            return owner.read_slice(&path, starts, counts);
6977        }
6978        let (datatype, out_bytes) = self.slice_size_and_datatype(name, starts, counts)?;
6979        // The selection lands in the vector this returns:
6980        // `read_slice_into_unconverted` defines every byte of the buffer it is
6981        // handed, so there is nothing to zero first and nothing to copy after.
6982        read_image_into_new::<u8, _, _>(out_bytes as usize, |image| {
6983            self.read_slice_into_unconverted(name, starts, counts, image, 0, ReadDst::Fresh)?;
6984            Self::apply_post_filter_conversion(image, &datatype)
6985        })
6986    }
6987
6988    /// Read a hyperslab straight into a caller-provided buffer (no allocation).
6989    ///
6990    /// `out.len()` must equal `product(counts) * element_size`; otherwise an
6991    /// error is returned. The no-allocation counterpart of
6992    /// [`read_slice`](Self::read_slice) and the zero-copy entry point for
6993    /// reading a selection directly into a pinned/registered host buffer for an
6994    /// H2D transfer.
6995    pub fn read_slice_into(
6996        &mut self,
6997        name: &str,
6998        starts: &[u64],
6999        counts: &[u64],
7000        out: &mut [u8],
7001    ) -> IoResult<()> {
7002        self.read_slice_into_dst(name, starts, counts, out, ReadDst::Reused)
7003    }
7004
7005    /// [`read_slice_into`](Self::read_slice_into) with the caller's
7006    /// destination fact made explicit, for internal callers whose buffer is a
7007    /// fresh allocation rather than a kept one (the allocating wrappers in
7008    /// `dataset.rs`).
7009    pub(crate) fn read_slice_into_dst(
7010        &mut self,
7011        name: &str,
7012        starts: &[u64],
7013        counts: &[u64],
7014        out: &mut [u8],
7015        dst: ReadDst,
7016    ) -> IoResult<()> {
7017        if self.external_edge(name).is_some() {
7018            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7019            return owner.read_slice_into_dst(&path, starts, counts, out, dst);
7020        }
7021        let (datatype, out_bytes) = self.slice_size_and_datatype(name, starts, counts)?;
7022        if out.len() as u64 != out_bytes {
7023            return Err(crate::io::IoError::InvalidState(format!(
7024                "read_slice_into: buffer is {} bytes but selection needs {}",
7025                out.len(),
7026                out_bytes
7027            )));
7028        }
7029        self.read_slice_into_unconverted(name, starts, counts, out, 0, dst)?;
7030        Self::apply_post_filter_conversion(out, &datatype)?;
7031        Ok(())
7032    }
7033
7034    /// Logical byte size of a hyperslab (`product(counts) * element_size`) with
7035    /// the datatype needed for the post-filter conversion.
7036    fn slice_size_and_datatype(
7037        &self,
7038        name: &str,
7039        starts: &[u64],
7040        counts: &[u64],
7041    ) -> IoResult<(DatatypeMessage, u64)> {
7042        let info = self
7043            .dataset_info_local(name)
7044            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7045        // Bounds first: a selection the extent does not admit is refused
7046        // before its byte size is computed, so an oversized count is reported
7047        // as such rather than as a failed allocation.
7048        check_hyperslab(&info.dataspace.dims, starts, counts)?;
7049        let out_bytes = saturating_byte_len(counts, info.datatype.element_size() as u64);
7050        Ok((info.datatype.clone(), out_bytes))
7051    }
7052
7053    /// Fill `out` with a hyperslab selection, before the post-filter datatype
7054    /// conversion. The single owner of read-destination semantics for slice
7055    /// reads (mirrors [`read_dataset_raw_into_unconverted`](Self::read_dataset_raw_into_unconverted)):
7056    /// it validates the selection, then fully defines every byte of `out`
7057    /// (contiguous/compact runs cover the whole selection; chunked layouts
7058    /// pre-fill with the tiled fill value and scatter only overlapping chunks).
7059    /// Both [`read_slice`](Self::read_slice) and [`read_slice_into`](Self::read_slice_into)
7060    /// wrap it and apply the conversion exactly once.
7061    ///
7062    /// `out.len()` must equal `product(counts) * element_size`. `depth`
7063    /// counts virtual-dataset nesting for a caller reached through
7064    /// [`read_virtual_into`](Self::read_virtual_into); pass `0` for a
7065    /// top-level call. `dst` is the caller's destination fact ([`ReadDst`]):
7066    /// whether `out` is a fresh allocation or a buffer the caller keeps
7067    /// across reads, which decides whether a mapped contiguous read is priced
7068    /// at the cold or the warm row.
7069    fn read_slice_into_unconverted(
7070        &mut self,
7071        name: &str,
7072        starts: &[u64],
7073        counts: &[u64],
7074        out: &mut [u8],
7075        depth: usize,
7076        dst: ReadDst,
7077    ) -> IoResult<()> {
7078        let info = self
7079            .dataset_info_local(name)
7080            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7081        let dims = info.dataspace.dims.clone();
7082        let element_size = info.datatype.element_size() as u64;
7083        let layout = info.layout.clone();
7084        let pipeline = info.filter_pipeline.clone();
7085        let fill_value = info.fill_value.clone();
7086        let external_files = info.external_files.clone();
7087        let ndims = dims.len();
7088
7089        check_hyperslab(&dims, starts, counts)?;
7090        if ndims == 0 {
7091            return Err(crate::io::IoError::InvalidState(
7092                "read_slice does not support scalar datasets; use read_dataset_raw".into(),
7093            ));
7094        }
7095
7096        match &layout {
7097            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
7098                // Same coalesced run geometry as the normal contiguous case
7099                // below, but each run is read through the external file
7100                // list instead of straight from this file (H5D__efl_read,
7101                // H5Defl.c) — `src_off` is already dataset-relative, which
7102                // is exactly what `read_external_file_bytes` walks slots by.
7103                let prefix = self.extfile_prefix_in_force(name);
7104                for_each_contiguous_run(
7105                    &dims,
7106                    starts,
7107                    counts,
7108                    element_size,
7109                    |src_off, out_off, len| {
7110                        read_external_file_bytes(
7111                            &external_files,
7112                            prefix.as_deref(),
7113                            src_off,
7114                            &mut out[out_off..out_off + len],
7115                        )
7116                    },
7117                )?;
7118            }
7119            DataLayoutMessage::Contiguous { address, .. } => {
7120                if *address == UNDEF_ADDR {
7121                    // Never-written: the selection reads back as the fill value.
7122                    fill_tiled_into(out, fill_value.as_deref());
7123                } else {
7124                    // Read each maximal contiguous run straight into `out`.
7125                    // Trailing full-selected dimensions coalesce, so a slice
7126                    // like `[:, r0:r1, :]` of `[nproj, nz, nx]` becomes `nproj`
7127                    // reads of `(r1-r0)*nx` elements instead of `nproj*(r1-r0)`
7128                    // per-`nx`-row reads. The 1-D case folds to a single run.
7129                    // The runs cover the whole selection, so every byte of `out`
7130                    // is written.
7131                    let base = *address;
7132                    for_each_contiguous_run(
7133                        &dims,
7134                        starts,
7135                        counts,
7136                        element_size,
7137                        |src_off, out_off, len| {
7138                            // `base` is the file's claim; a sum that wraps
7139                            // would land on unrelated bytes, so it is an
7140                            // error, not an offset.
7141                            let at = base.checked_add(src_off).ok_or_else(|| {
7142                                crate::io::IoError::InvalidState(format!(
7143                                    "dataset '{name}' claims raw data at {base}, which \
7144                                     overflows {src_off} bytes into the selection"
7145                                ))
7146                            })?;
7147                            self.handle
7148                                .read_exact_at_into(at, &mut out[out_off..out_off + len], dst)
7149                                .map_err(Into::into)
7150                        },
7151                    )?;
7152                }
7153            }
7154            DataLayoutMessage::Compact { data } => {
7155                // Same coalesced geometry, copying from the in-memory full
7156                // dataset instead of reading from the file.
7157                for_each_contiguous_run(
7158                    &dims,
7159                    starts,
7160                    counts,
7161                    element_size,
7162                    |src_off, out_off, len| {
7163                        let src = src_off as usize;
7164                        out[out_off..out_off + len].copy_from_slice(&data[src..src + len]);
7165                        Ok(())
7166                    },
7167                )?;
7168            }
7169            DataLayoutMessage::ChunkedV3 {
7170                chunk_dims,
7171                b_tree_address,
7172            } => {
7173                // Walk the v1 B-tree index reading only chunks that overlap the
7174                // selection, scattering each chunk∩selection into the slice
7175                // buffer; whatever no chunk covers is filled there. The
7176                // unconverted read keeps the post-filter conversion to exactly
7177                // once, in the wrappers.
7178                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7179                self.read_chunked_btree_v1(
7180                    name,
7181                    real_chunk_dims,
7182                    *b_tree_address,
7183                    ChunkReadRequest {
7184                        pipeline: pipeline.as_ref(),
7185                        target: ChunkTarget::Slice { starts, counts },
7186                        fill_value: fill_value.as_deref(),
7187                        dst,
7188                    },
7189                    out,
7190                )?;
7191            }
7192            DataLayoutMessage::ChunkedV4 {
7193                chunk_dims,
7194                index_address,
7195                index_type,
7196                earray_params,
7197                single_chunk_filter,
7198                ..
7199            } => {
7200                // Same selection-aware chunk read for every v4 index kind
7201                // (single chunk, fixed/extensible array, B-tree v2): only
7202                // overlapping chunks are read and only their intersection with
7203                // the selection is scattered into the slice output.
7204                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7205                self.read_chunked_v4(
7206                    name,
7207                    real_chunk_dims,
7208                    ChunkIndexDesc {
7209                        index_type: *index_type,
7210                        index_address: *index_address,
7211                        earray_params: earray_params.as_ref(),
7212                        single_chunk_filter: *single_chunk_filter,
7213                    },
7214                    ChunkReadRequest {
7215                        pipeline: pipeline.as_ref(),
7216                        target: ChunkTarget::Slice { starts, counts },
7217                        fill_value: fill_value.as_deref(),
7218                        dst,
7219                    },
7220                    out,
7221                )?;
7222            }
7223            DataLayoutMessage::Virtual { .. } => {
7224                // Stitch the full virtual image, then extract the
7225                // requested region from it. This reads more than the
7226                // selection strictly needs (no per-mapping intersection
7227                // against the caller's box), but a virtual dataset's data
7228                // is composed from other datasets rather than stored
7229                // contiguously, so there is no cheaper selective path
7230                // without duplicating `read_virtual_into`'s mapping walk
7231                // for a bounded region — correctness, not I/O pruning, is
7232                // what a VDS slice read needs here. The extraction below
7233                // writes every byte of `out`, so no pre-fill is needed on top
7234                // of the full image's own.
7235                let total = saturating_byte_len(&dims, element_size) as usize;
7236                let mut full = alloc_tiled_fill(total, fill_value.as_deref())?;
7237                self.read_virtual_into(name, &mut full, depth)?;
7238                for_each_contiguous_run(
7239                    &dims,
7240                    starts,
7241                    counts,
7242                    element_size,
7243                    |src_off, out_off, len| {
7244                        let src_off = src_off as usize;
7245                        out[out_off..out_off + len].copy_from_slice(&full[src_off..src_off + len]);
7246                        Ok(())
7247                    },
7248                )?;
7249            }
7250        }
7251        Ok(())
7252    }
7253
7254    /// Read a strided hyperslab — h5py's stepped slicing (`ds[a:b:s]`) or
7255    /// the general `start`/`stride`/`count`/`block` form of
7256    /// `H5Sselect_hyperslab` — into a caller-provided buffer.
7257    ///
7258    /// One tuple per dimension: `start[d]` is the first index, `stride[d]`
7259    /// the spacing between selected blocks (`1` = the classic contiguous
7260    /// selection [`read_slice`](Self::read_slice) reads), `count[d]` how
7261    /// many blocks, and `block[d]` how many contiguous elements each block
7262    /// covers. `out` is row-major over `count[d] * block[d]` per dimension —
7263    /// exactly the shape h5py's stepped slicing produces — and `out.len()`
7264    /// must equal that times the element size.
7265    ///
7266    /// Built on the same selection-decomposition primitives the virtual
7267    /// dataset reader uses for its per-mapping scatter
7268    /// ([`Selection::resolve`], [`copy_matched_selections`]) rather than a
7269    /// second box walker: the requested selection and a densely-packed
7270    /// "output" selection sharing the same `count` hold the same elements in
7271    /// the same order, so pairing the two element streams places each one
7272    /// where h5py's stepped slicing puts it, and each source box is read with
7273    /// the ordinary per-layout selective read
7274    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
7275    ///
7276    /// The single owner of read-destination semantics for stepped selections:
7277    /// it defines every byte of `out` before returning `Ok`, so a typed caller
7278    /// reads straight into the vector it keeps rather than into a byte buffer
7279    /// it then copies.
7280    pub fn read_hyperslab_into(
7281        &mut self,
7282        name: &str,
7283        start: &[u64],
7284        stride: &[u64],
7285        count: &[u64],
7286        block: &[u64],
7287        out: &mut [u8],
7288    ) -> IoResult<()> {
7289        if self.external_edge(name).is_some() {
7290            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7291            return owner.read_hyperslab_into(&path, start, stride, count, block, out);
7292        }
7293        let info = self
7294            .dataset_info_local(name)
7295            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7296        let dims = info.dataspace.dims.clone();
7297        let datatype = info.datatype.clone();
7298        let element_size = datatype.element_size() as u64;
7299        let rank = dims.len();
7300
7301        if start.len() != rank || stride.len() != rank || count.len() != rank || block.len() != rank
7302        {
7303            return Err(crate::io::IoError::InvalidState(
7304                "start/stride/count/block length must match dataset rank".into(),
7305            ));
7306        }
7307        if stride.contains(&0) {
7308            return Err(crate::io::IoError::InvalidState(
7309                "hyperslab stride must be nonzero in every dimension".into(),
7310            ));
7311        }
7312        let out_bytes = saturating_byte_len(
7313            &(0..rank)
7314                .map(|d| count[d].saturating_mul(block[d]))
7315                .collect::<Vec<u64>>(),
7316            element_size,
7317        );
7318        if out.len() as u64 != out_bytes {
7319            return Err(crate::io::IoError::InvalidState(format!(
7320                "read_hyperslab_into: buffer is {} bytes but selection needs {out_bytes}",
7321                out.len(),
7322            )));
7323        }
7324
7325        let src_sel = Selection::Hyperslab {
7326            rank,
7327            form: Hyperslab::Regular(RegularHyperslab {
7328                start: start.to_vec(),
7329                stride: stride.to_vec(),
7330                count: count.to_vec(),
7331                block: block.to_vec(),
7332            }),
7333        };
7334        let out_dims: Vec<u64> = (0..rank)
7335            .map(|d| count[d].saturating_mul(block[d]))
7336            .collect();
7337        // A densely-packed selection sharing the same `count`: it holds the
7338        // same elements as `src_sel` in the same order, and its extent is the
7339        // output buffer, so pairing the two element streams lands every
7340        // source element at its h5py stepped-slicing position.
7341        let dst_sel = Selection::Hyperslab {
7342            rank,
7343            form: Hyperslab::Regular(RegularHyperslab {
7344                start: vec![0u64; rank],
7345                stride: block.to_vec(),
7346                count: count.to_vec(),
7347                block: block.to_vec(),
7348            }),
7349        };
7350
7351        // `copy_matched_selections` is shared with the virtual-dataset
7352        // scatter, where a target selection may legitimately leave elements
7353        // untouched, so it carries no whole-output guarantee of its own: zero
7354        // first, exactly as the allocating form always did.
7355        fill_tiled_into(out, None);
7356        copy_matched_selections(
7357            |bstart, bcount, buf| {
7358                // `buf` is `copy_matched_selections`' fresh per-box buffer,
7359                // not the caller's `out`.
7360                self.read_slice_into_unconverted(name, bstart, bcount, buf, 0, ReadDst::Fresh)
7361            },
7362            &src_sel.resolve(&dims)?,
7363            &dst_sel.resolve(&out_dims)?,
7364            element_size,
7365            out,
7366        )?;
7367        Self::apply_post_filter_conversion(out, &datatype)
7368    }
7369
7370    /// Read a list of coordinates in one call — h5py fancy indexing with a
7371    /// coordinate list (`H5S_SEL_POINTS`) — into a caller-provided buffer.
7372    ///
7373    /// `points[i]` is a `rank`-length coordinate; `out` holds one element per
7374    /// point, `element_size` bytes each, in the same order as `points` (point
7375    /// selection order is significant, see [`PointSelection`]), so `out.len()`
7376    /// must equal `points.len() * element_size`. Backed by
7377    /// [`Selection::Points`] and [`Selection::to_boxes`] — the same
7378    /// decomposition [`read_hyperslab_into`](Self::read_hyperslab_into) and
7379    /// the virtual dataset reader use — each point's 1-element box is read
7380    /// with the ordinary per-layout selective read
7381    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
7382    /// A 1-element box is already a flat `element_size`-byte run, so
7383    /// placing it needs no further run-decomposition
7384    /// ([`for_each_dual_run`] would degenerate to exactly this copy).
7385    ///
7386    /// One element-sized box per point covers the buffer exactly, so every
7387    /// byte of `out` is defined before it returns `Ok` and a typed caller
7388    /// reads straight into the vector it keeps.
7389    /// `dst` is the caller's destination fact ([`ReadDst`]) for `out` — a
7390    /// required argument rather than a defaulted wrapper, since unlike the
7391    /// raw/slice pair this read has no kept-buffer caller to assert it for.
7392    pub fn read_points_into(
7393        &mut self,
7394        name: &str,
7395        points: &[Vec<u64>],
7396        out: &mut [u8],
7397        dst: ReadDst,
7398    ) -> IoResult<()> {
7399        if self.external_edge(name).is_some() {
7400            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7401            return owner.read_points_into(&path, points, out, dst);
7402        }
7403        let info = self
7404            .dataset_info_local(name)
7405            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7406        let dims = info.dataspace.dims.clone();
7407        let datatype = info.datatype.clone();
7408        let element_size = datatype.element_size() as u64;
7409        let rank = dims.len();
7410
7411        for p in points {
7412            if p.len() != rank {
7413                return Err(crate::io::IoError::InvalidState(format!(
7414                    "point coordinate has {} entries but the dataset has {} dimensions",
7415                    p.len(),
7416                    rank
7417                )));
7418            }
7419        }
7420
7421        let sel = Selection::Points(PointSelection {
7422            rank,
7423            points: points.to_vec(),
7424        });
7425        let boxes = sel.to_boxes(&dims)?;
7426
7427        let es = element_size as usize;
7428        if out.len() != points.len() * es {
7429            return Err(crate::io::IoError::InvalidState(format!(
7430                "read_points_into: buffer is {} bytes but {} points need {}",
7431                out.len(),
7432                points.len(),
7433                points.len() * es
7434            )));
7435        }
7436        for (i, (bstart, bcount)) in boxes.iter().enumerate() {
7437            self.read_slice_into_unconverted(
7438                name,
7439                bstart,
7440                bcount,
7441                &mut out[i * es..(i + 1) * es],
7442                0,
7443                dst,
7444            )?;
7445        }
7446        Self::apply_post_filter_conversion(out, &datatype)
7447    }
7448
7449    /// Read one chunk's raw (still-filtered) bytes and its filter mask —
7450    /// the read half of `H5Dread_chunk` (h5py:
7451    /// `Dataset.id.read_direct_chunk`).
7452    ///
7453    /// `chunk_coords` is the chunk's position in the chunk grid, one
7454    /// coordinate per dimension counted in chunks (not elements) — the same
7455    /// addressing [`write_chunk_raw_at`](crate::Dataset::write_chunk_raw_at)
7456    /// uses on the write side. The bytes returned are exactly what is
7457    /// stored on disk: filtered/compressed if the dataset has a filter
7458    /// pipeline, with no decompression applied — the caller runs the
7459    /// pipeline itself (honoring the returned mask, which marks any filter
7460    /// this particular chunk skipped) if it wants decoded data.
7461    ///
7462    /// Resolved through whichever chunk index the dataset uses, reusing the
7463    /// same per-index decoders the full/slice chunked reader walks
7464    /// ([`collect_fa_chunk_entries`](Self::collect_fa_chunk_entries),
7465    /// [`collect_ea_chunk_entries`](Self::collect_ea_chunk_entries),
7466    /// [`collect_bt2_chunk_entries`](Self::collect_bt2_chunk_entries),
7467    /// [`collect_btree_v1_chunks`](Self::collect_btree_v1_chunks)) rather
7468    /// than a new index walker.
7469    ///
7470    /// `Err` when the dataset is not chunked, `chunk_coords` has the wrong
7471    /// rank, or the chunk at those coordinates has never been written.
7472    pub fn read_chunk_raw_at(
7473        &mut self,
7474        name: &str,
7475        chunk_coords: &[u64],
7476    ) -> IoResult<(Vec<u8>, u32)> {
7477        if self.external_edge(name).is_some() {
7478            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7479            return owner.read_chunk_raw_at(&path, chunk_coords);
7480        }
7481        let info = self
7482            .dataset_info_local(name)
7483            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7484        let dims = info.dataspace.dims.clone();
7485        let max_dims = info.dataspace.max_dims.clone();
7486        let element_size = info.datatype.element_size() as u64;
7487        let layout = info.layout.clone();
7488        let ndims = dims.len();
7489
7490        if chunk_coords.len() != ndims {
7491            return Err(crate::io::IoError::InvalidState(format!(
7492                "chunk_coords has {} entries but the dataset has {} dimensions",
7493                chunk_coords.len(),
7494                ndims
7495            )));
7496        }
7497
7498        let not_written = || {
7499            crate::io::IoError::InvalidState(format!(
7500                "chunk at coordinates {chunk_coords:?} has not been written"
7501            ))
7502        };
7503
7504        match &layout {
7505            DataLayoutMessage::ChunkedV3 {
7506                chunk_dims,
7507                b_tree_address,
7508            } => {
7509                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7510                if *b_tree_address == UNDEF_ADDR {
7511                    return Err(not_written());
7512                }
7513                let file_size = self.handle.file_size()?;
7514                let mut entries = Vec::new();
7515                self.collect_btree_v1_chunks(*b_tree_address, ndims, file_size, 0, &mut entries)?;
7516                for (offsets, addr, chunk_size, mask) in &entries {
7517                    if *addr == UNDEF_ADDR {
7518                        continue;
7519                    }
7520                    let mut scaled = Vec::with_capacity(ndims);
7521                    for d in 0..ndims {
7522                        scaled.push(offsets[d].checked_div(real_chunk_dims[d]).unwrap_or(0));
7523                    }
7524                    if scaled.as_slice() == chunk_coords {
7525                        return Ok((self.handle.read_at(*addr, *chunk_size as usize)?, *mask));
7526                    }
7527                }
7528                Err(not_written())
7529            }
7530            DataLayoutMessage::ChunkedV4 {
7531                chunk_dims,
7532                index_type,
7533                index_address,
7534                earray_params,
7535                single_chunk_filter,
7536                ..
7537            } => {
7538                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7539                match index_type {
7540                    data_layout::ChunkIndexType::SingleChunk => {
7541                        if chunk_coords.iter().any(|&c| c != 0) {
7542                            return Err(crate::io::IoError::InvalidState(format!(
7543                                "chunk coordinates {chunk_coords:?} are outside the chunk \
7544                                 grid (0..1): this dataset has a single-chunk index"
7545                            )));
7546                        }
7547                        if *index_address == UNDEF_ADDR {
7548                            return Err(not_written());
7549                        }
7550                        match single_chunk_filter {
7551                            Some(scf) => Ok((
7552                                self.handle.read_at(*index_address, scf.nbytes as usize)?,
7553                                scf.filter_mask,
7554                            )),
7555                            None => {
7556                                let total = saturating_byte_len(&dims, element_size);
7557                                Ok((self.handle.read_at(*index_address, total as usize)?, 0))
7558                            }
7559                        }
7560                    }
7561                    data_layout::ChunkIndexType::Implicit => {
7562                        if *index_address == UNDEF_ADDR {
7563                            return Err(not_written());
7564                        }
7565                        let linear = crate::io::chunk_grid::linear_index(
7566                            &dims,
7567                            max_dims.as_deref(),
7568                            real_chunk_dims,
7569                            chunk_coords,
7570                        )?;
7571                        let chunk_bytes = saturating_byte_len(real_chunk_dims, element_size);
7572                        let addr = index_address.saturating_add(linear.saturating_mul(chunk_bytes));
7573                        Ok((self.handle.read_at(addr, chunk_bytes as usize)?, 0))
7574                    }
7575                    data_layout::ChunkIndexType::FixedArray => {
7576                        let linear = crate::io::chunk_grid::linear_index(
7577                            &dims,
7578                            max_dims.as_deref(),
7579                            real_chunk_dims,
7580                            chunk_coords,
7581                        )?;
7582                        let entries = self.collect_fa_chunk_entries(
7583                            real_chunk_dims,
7584                            ndims,
7585                            element_size,
7586                            *index_address,
7587                        )?;
7588                        match entries.get(linear as usize) {
7589                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
7590                                Ok((self.handle.read_at(addr, size as usize)?, mask))
7591                            }
7592                            _ => Err(not_written()),
7593                        }
7594                    }
7595                    data_layout::ChunkIndexType::ExtensibleArray => {
7596                        let params = earray_params.as_ref().ok_or_else(|| {
7597                            crate::io::IoError::InvalidState("missing earray params".into())
7598                        })?;
7599                        let linear = crate::io::chunk_grid::linear_index(
7600                            &dims,
7601                            max_dims.as_deref(),
7602                            real_chunk_dims,
7603                            chunk_coords,
7604                        )?;
7605                        let entries = self.collect_ea_chunk_entries(
7606                            *index_address,
7607                            params,
7608                            &dims,
7609                            max_dims.as_deref(),
7610                            real_chunk_dims,
7611                            element_size,
7612                        )?;
7613                        match entries.get(linear as usize) {
7614                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
7615                                Ok((self.handle.read_at(addr, size as usize)?, mask))
7616                            }
7617                            _ => Err(not_written()),
7618                        }
7619                    }
7620                    data_layout::ChunkIndexType::BTreeV2 => {
7621                        let entries = self.collect_bt2_chunk_entries(
7622                            real_chunk_dims,
7623                            ndims,
7624                            element_size,
7625                            *index_address,
7626                        )?;
7627                        match entries
7628                            .iter()
7629                            .find(|(_, _, scaled, _)| scaled.as_slice() == chunk_coords)
7630                        {
7631                            Some(&(addr, size, _, mask)) if addr != UNDEF_ADDR => {
7632                                Ok((self.handle.read_at(addr, size)?, mask))
7633                            }
7634                            _ => Err(not_written()),
7635                        }
7636                    }
7637                }
7638            }
7639            _ => Err(crate::io::IoError::InvalidState(
7640                "read_chunk_raw_at is only for chunked datasets".into(),
7641            )),
7642        }
7643    }
7644}
7645
7646/// Adapts a `FileHandle` to the `BlockReader` trait used by the fractal-heap
7647/// walker, so heap blocks can be fetched from the open file.
7648///
7649/// The handle is shared, not exclusive: every read goes through
7650/// `FileHandle::read_at_most`, which takes `&self`, so the writer can walk a
7651/// structure it is about to free while holding only `&self` itself.
7652pub(crate) struct HandleBlockReader<'a> {
7653    pub(crate) handle: &'a FileHandle,
7654}
7655
7656/// One object's attribute set, or the reason it could not be read whole.
7657///
7658/// [`AttributeEntry`] carries a per-attribute failure, which needs a name to
7659/// hang on. Two failures have none. A dense set is indexed by name *hash*, so
7660/// a heap or index that will not read yields no names at all; and an attribute
7661/// message too damaged to yield its own name cannot be listed under one
7662/// either. The only listing that can report those honestly is the object's, so
7663/// this type carries the object-scope reason beside the entries, and every
7664/// accessor that would present the set as whole returns the reason instead.
7665///
7666/// The entries are deliberately unreachable while the set is incomplete: the
7667/// only way out is [`Self::into_complete`], which refuses. That is what keeps
7668/// the writer from rebuilding an object header out of a partial set and
7669/// deleting the attributes it never saw.
7670#[derive(Debug, Clone, Default, PartialEq)]
7671pub struct ObjectAttributes {
7672    entries: Vec<AttributeEntry>,
7673    incomplete: Option<String>,
7674    /// This object's own attribute creation-order policy, from the header
7675    /// flag bits `H5Pget_attr_creation_order` reads back
7676    /// (`attribute_creation_order`) — a structural fact about the header,
7677    /// known even when the entries above are `incomplete`.
7678    creation_order: CreationOrder,
7679    /// This object's own compact-vs-dense attribute storage, from the
7680    /// attribute info message's heap address (or its absence) — likewise a
7681    /// structural fact known even when the entries are `incomplete`.
7682    storage: AttributeStorage,
7683}
7684
7685impl ObjectAttributes {
7686    /// Record an attribute this collector could name.
7687    fn push(&mut self, entry: AttributeEntry) {
7688        self.entries.push(entry);
7689    }
7690
7691    /// Record that part of the set could not be read. The first reason stands:
7692    /// it is the one that explains the earliest missing attributes.
7693    fn mark_incomplete(&mut self, reason: String) {
7694        if self.incomplete.is_none() {
7695            self.incomplete = Some(reason);
7696        }
7697    }
7698
7699    /// Why this object's attributes cannot be listed, or `None` when the set
7700    /// is whole. Individual entries in a whole set may still be undecodable —
7701    /// [`AttributeEntry::unreadable_reason`] answers for those.
7702    pub fn unreadable_reason(&self) -> Option<&str> {
7703        self.incomplete.as_deref()
7704    }
7705
7706    /// This object's own attribute creation-order policy —
7707    /// `H5Pget_attr_creation_order`'s answer, read off the object header's
7708    /// own flag bits rather than derived from the entries. Available even
7709    /// when the set is [`incomplete`](Self::unreadable_reason): it names
7710    /// nothing that failed to decode.
7711    pub fn creation_order(&self) -> CreationOrder {
7712        self.creation_order
7713    }
7714
7715    /// This object's own compact-vs-dense attribute storage — h5py's
7716    /// `h5o.get_info(...).meta_size.attr.index_size` check, read off the
7717    /// attribute info message's heap address rather than derived from the
7718    /// entries. Available even when the set is
7719    /// [`incomplete`](Self::unreadable_reason).
7720    pub fn storage(&self) -> AttributeStorage {
7721        self.storage
7722    }
7723
7724    /// This object's own attribute count as `H5Oget_info().num_attrs`
7725    /// reports it — the object header's count, not necessarily the same
7726    /// enumeration path as [`ordered_names`](Self::ordered_names).
7727    ///
7728    /// `H5O__attr_count_real` derives this from the attribute info message
7729    /// when the header carries one (the dense name-index record count, or
7730    /// the compact message count `H5O__attr_open_by_idx` already counted
7731    /// while building it) and from the raw attribute-message envelope count
7732    /// otherwise. Both reduce to the number of attributes this collector
7733    /// successfully names: a conformant writer creates the info message
7734    /// exactly when it has attributes to report through it, so a whole set's
7735    /// length already equals what libhdf5's header-count algorithm answers,
7736    /// without replaying its v1-header/v2-header branch here.
7737    pub fn header_count(&self, owner: &str) -> IoResult<u64> {
7738        Ok(self.complete(owner)?.len() as u64)
7739    }
7740
7741    /// The entries, once the set is known to be whole.
7742    ///
7743    /// The sole route from an `ObjectAttributes` to an owned entry list. A
7744    /// caller that rewrites the object header — the append path — must take
7745    /// this route, so an unread set stops the rewrite instead of erasing the
7746    /// attributes behind it.
7747    pub(crate) fn into_complete(self, owner: &str) -> IoResult<Vec<AttributeEntry>> {
7748        match self.incomplete {
7749            Some(reason) => Err(incomplete_error(owner, &reason)),
7750            None => Ok(self.entries),
7751        }
7752    }
7753
7754    /// The entries, once the set is known to be whole, borrowed.
7755    fn complete(&self, owner: &str) -> IoResult<&[AttributeEntry]> {
7756        match &self.incomplete {
7757            Some(reason) => Err(incomplete_error(owner, reason)),
7758            None => Ok(&self.entries),
7759        }
7760    }
7761
7762    /// This object's attribute names, once the set is known to be whole, in
7763    /// the order h5py's default iteration produces them: creation order when
7764    /// the object tracks it, name order otherwise
7765    /// (`H5A__compact_cmp_corder`/`H5A__compact_cmp_name` for compact
7766    /// storage, the matching v2 B-tree index for dense) — never the physical
7767    /// order the entries happen to sit in, which `entries` otherwise
7768    /// preserves for the writer's rewrite path.
7769    pub(crate) fn ordered_names(&self, owner: &str) -> IoResult<Vec<String>> {
7770        let mut ordered: Vec<&AttributeEntry> = self.complete(owner)?.iter().collect();
7771        if !ordered.is_empty() && ordered.iter().all(|e| e.creation_index().is_some()) {
7772            ordered.sort_by_key(|e| e.creation_index());
7773        } else {
7774            ordered.sort_by(|a, b| a.name().cmp(b.name()));
7775        }
7776        Ok(ordered.into_iter().map(|e| e.name().to_string()).collect())
7777    }
7778}
7779
7780/// The one wording for "this object's attributes are not all here".
7781///
7782/// `Unsupported`, the same variant an undecodable dataset message raises: the
7783/// name is in the listing and the content is out of reach, which is what the
7784/// variant is for. Wrapping a `FormatError` here instead would put the same
7785/// condition behind two different public variants.
7786fn incomplete_error(owner: &str, reason: &str) -> crate::io::IoError {
7787    crate::io::IoError::Unsupported(format!(
7788        "attributes of '{owner}' cannot be read whole: {reason}"
7789    ))
7790}
7791
7792/// Every attribute attached to an object, whichever storage it uses.
7793///
7794/// This is the only place attributes are pulled off an object header — reader
7795/// and writer alike. Compact storage keeps them as `Attribute` messages in the
7796/// header itself; once an object crosses the phase-change threshold libhdf5
7797/// moves *all* of them into a fractal heap named by the `Attribute Info`
7798/// message and leaves no attribute message behind
7799/// (`H5Oattribute.c::H5O__attr_create`). Scanning only the messages therefore
7800/// reports zero attributes for a dense object: a silent loss on read, and a
7801/// silent deletion when the writer rebuilds that object's header from what it
7802/// collected.
7803///
7804/// An attribute this crate cannot decode is kept, named, with the reason
7805/// attached: a listing that omitted it would report a file that does not
7806/// contain it. What cannot be named at all — a damaged attribute message, an
7807/// attribute info message that will not decode, a dense set whose heap or name
7808/// index will not read — marks the whole set incomplete, so the object reports
7809/// the failure rather than a short list.
7810pub(crate) fn collect_object_attributes(
7811    handle: &mut FileHandle,
7812    ctx: &FormatContext,
7813    header: &ObjectHeader,
7814) -> ObjectAttributes {
7815    let mut attrs = ObjectAttributes {
7816        creation_order: header.attribute_creation_order(),
7817        ..ObjectAttributes::default()
7818    };
7819    // The message envelope carries a creation index only when the header says
7820    // the object tracks one; the field is not even encoded otherwise
7821    // (`H5O_SIZEOF_MSGHDR_OH`), so reading it as an index would report zero
7822    // for every attribute of an untracked object.
7823    let tracked = header.has_creation_order();
7824    for msg in &header.messages {
7825        match msg.msg_type {
7826            MSG_ATTRIBUTE => match AttributeEntry::parse(&msg.data, ctx) {
7827                Ok(entry) => {
7828                    attrs.push(entry.with_creation_index(tracked.then_some(msg.creation_index)))
7829                }
7830                Err(e) => attrs.mark_incomplete(format!("an attribute message is unreadable: {e}")),
7831            },
7832            MSG_ATTR_INFO => match AttributeInfoMessage::decode(&msg.data, ctx) {
7833                Ok((info, _)) => {
7834                    attrs.storage = if info.is_dense() {
7835                        AttributeStorage::Dense
7836                    } else {
7837                        AttributeStorage::Compact
7838                    };
7839                    let mut br = HandleBlockReader { handle };
7840                    match crate::format::dense_attr::read_dense_attributes(&info, ctx, &mut br) {
7841                        Ok(dense) => attrs.entries.extend(dense),
7842                        Err(e) => attrs
7843                            .mark_incomplete(format!("dense attribute storage is unreadable: {e}")),
7844                    }
7845                }
7846                Err(e) => {
7847                    attrs.mark_incomplete(format!("the attribute info message is unreadable: {e}"))
7848                }
7849            },
7850            _ => {}
7851        }
7852    }
7853    attrs
7854}
7855
7856impl BlockReader for HandleBlockReader<'_> {
7857    fn read_block(&mut self, offset: u64, len: usize) -> crate::format::FormatResult<Vec<u8>> {
7858        // `read_at_most`, not `read_at`: a metadata block allocated at the end
7859        // of the file can be shorter on disk than its nominal size, and every
7860        // decoder re-checks the length it needs.
7861        self.handle.read_at_most(offset, len).map_err(|e| {
7862            crate::format::FormatError::InvalidData(format!(
7863                "metadata block read failed at {:#x}: {}",
7864                offset, e
7865            ))
7866        })
7867    }
7868}
7869
7870#[cfg(test)]
7871mod tests {
7872    use super::*;
7873    use std::io::Write;
7874
7875    /// Per-call unique temp path. PID + atomic counter avoids
7876    /// path collisions across concurrent cargo invocations and
7877    /// kernel-side flock release races.
7878    fn temp_path(name: &str) -> std::path::PathBuf {
7879        use std::sync::atomic::{AtomicU64, Ordering};
7880        static COUNTER: AtomicU64 = AtomicU64::new(0);
7881        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
7882        std::env::temp_dir().join(format!(
7883            "rust_hdf5_reader_test_{}_{}_{}.h5",
7884            name,
7885            std::process::id(),
7886            n
7887        ))
7888    }
7889
7890    /// Write `bytes` to a fresh temp file and open a read handle on it.
7891    fn handle_over(name: &str, bytes: &[u8]) -> (std::path::PathBuf, FileHandle) {
7892        let path = temp_path(name);
7893        std::fs::File::create(&path)
7894            .unwrap()
7895            .write_all(bytes)
7896            .unwrap();
7897        let handle =
7898            FileHandle::open_read_with_locking(&path, crate::io::locking::FileLocking::Disabled)
7899                .unwrap();
7900        (path, handle)
7901    }
7902
7903    /// The chunk-index cache. A decoded index MUST describe the file exactly
7904    /// as the catalog entry beside it does, and MUST NOT be served for an
7905    /// index address other than the one it was decoded from.
7906    /// [`Hdf5Reader::decoded_chunk_index`] is the single owner — the only
7907    /// reader and writer of the cache — and `DatasetTable` is what keeps a
7908    /// stale index unreachable: `entry_mut` is the one way to a mutable
7909    /// entry and it drops that entry's index first, where the `IndexMut` it
7910    /// replaced would have left it standing.
7911    mod chunk_index_cache {
7912        use super::*;
7913
7914        /// `d` = `[0, 1, .., 7]` in two chunks of four, fixed extent, so the
7915        /// layout points at a fixed array a read has to walk.
7916        fn chunked_file(name: &str) -> (std::path::PathBuf, Hdf5Reader) {
7917            let path = temp_path(name);
7918            {
7919                let file = crate::H5File::create(&path).unwrap();
7920                let ds = file
7921                    .new_dataset::<u8>()
7922                    .shape([8usize])
7923                    .chunk(&[4])
7924                    .create("d")
7925                    .unwrap();
7926                ds.write_raw(&(0u8..8).collect::<Vec<u8>>()).unwrap();
7927                file.close().unwrap();
7928            }
7929            let reader = Hdf5Reader::open(&path).unwrap();
7930            (path, reader)
7931        }
7932
7933        /// The address of the chunk index `d`'s layout points at.
7934        fn index_address(reader: &Hdf5Reader, pos: usize) -> u64 {
7935            match &reader.datasets[pos].layout {
7936                DataLayoutMessage::ChunkedV4 { index_address, .. } => *index_address,
7937                _ => panic!("expected a version-4 chunked layout"),
7938            }
7939        }
7940
7941        #[test]
7942        fn a_read_leaves_its_decoded_index_for_the_next_read() {
7943            let (path, mut r) = chunked_file("index_cache_hit");
7944            let pos = r.dataset_position("d").unwrap();
7945            let addr = index_address(&r, pos);
7946            assert!(
7947                r.datasets.chunk_index(pos, addr).is_none(),
7948                "a reader that has read nothing holds a decoded index"
7949            );
7950
7951            assert_eq!(r.read_slice("d", &[0], &[4]).unwrap(), vec![0u8, 1, 2, 3]);
7952            let cached = r
7953                .datasets
7954                .chunk_index(pos, addr)
7955                .expect("the read did not keep the index it decoded");
7956            assert_eq!(cached.entries.len(), 2, "one entry per chunk");
7957            assert_eq!(cached.coords, vec![0, 1], "slot 0 and slot 1 of the grid");
7958
7959            // The second read answers from it and reaches the same bytes.
7960            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
7961            let _ = std::fs::remove_file(&path);
7962        }
7963
7964        #[test]
7965        fn an_index_is_never_served_for_another_address() {
7966            let (path, mut r) = chunked_file("index_cache_addr");
7967            let pos = r.dataset_position("d").unwrap();
7968            let addr = index_address(&r, pos);
7969            r.read_slice("d", &[0], &[4]).unwrap();
7970
7971            assert!(
7972                r.datasets.chunk_index(pos, addr.wrapping_add(1)).is_none(),
7973                "an index decoded from {addr:#x} answered for another address"
7974            );
7975            let _ = std::fs::remove_file(&path);
7976        }
7977
7978        #[test]
7979        fn changing_an_entry_drops_the_index_decoded_against_it() {
7980            let (path, mut r) = chunked_file("index_cache_entry_mut");
7981            let pos = r.dataset_position("d").unwrap();
7982            let addr = index_address(&r, pos);
7983            r.read_slice("d", &[0], &[4]).unwrap();
7984            assert!(r.datasets.chunk_index(pos, addr).is_some());
7985
7986            // What a caller reaches an entry mutably for — the virtual extent
7987            // resolution rewrites `dataspace.dims` — is what the coordinates
7988            // in a decoded index were built against.
7989            let _ = r.datasets.entry_mut(pos);
7990            assert!(
7991                r.datasets.chunk_index(pos, addr).is_none(),
7992                "a mutable entry left its decoded index behind"
7993            );
7994            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
7995            let _ = std::fs::remove_file(&path);
7996        }
7997    }
7998
7999    /// The decompressed-chunk cache. An image MUST be served only for the
8000    /// stored bytes it was decoded from, MUST be kept only for a chunk the
8001    /// read left partly unconsumed, and MUST NOT outlive the decoded index
8002    /// that named its chunk — which it cannot, being a field of it, so the
8003    /// invalidation these tests exercise is the index's own (`entry_mut`, and
8004    /// the table rebuild a SWMR refresh does).
8005    mod chunk_image_cache {
8006        use super::*;
8007
8008        /// `d` = a ramp of `n` f64 in chunks of `chunk`, deflated or not.
8009        #[cfg(feature = "deflate")]
8010        fn dataset(name: &str, n: usize, chunk: usize, deflate: bool) -> std::path::PathBuf {
8011            let path = temp_path(name);
8012            let data: Vec<f64> = (0..n).map(|i| (i % 251) as f64).collect();
8013            let file = crate::H5File::create(&path).unwrap();
8014            let mut b = file.new_dataset::<f64>().shape([n]).chunk(&[chunk]);
8015            if deflate {
8016                b = b.deflate(6);
8017            }
8018            b.create("d").unwrap().write_raw(&data).unwrap();
8019            file.close().unwrap();
8020            path
8021        }
8022
8023        /// `d` = a ramp of `rows * cols` f64 in `chunk`-shaped chunks. A
8024        /// two-dimensional chunk is never one contiguous stretch of a read's
8025        /// output, so every chunk of it is staged and materialized — which is
8026        /// what leaves the cache free to keep it, where a one-dimensional
8027        /// whole-chunk read would have decoded straight into the output.
8028        fn dataset_2d(
8029            name: &str,
8030            rows: usize,
8031            cols: usize,
8032            chunk: [usize; 2],
8033            deflate: bool,
8034        ) -> std::path::PathBuf {
8035            let path = temp_path(name);
8036            let data: Vec<f64> = (0..rows * cols).map(|i| (i % 251) as f64).collect();
8037            let file = crate::H5File::create(&path).unwrap();
8038            let mut b = file
8039                .new_dataset::<f64>()
8040                .shape([rows, cols])
8041                .chunk(&chunk[..]);
8042            if deflate {
8043                b = b.deflate(6);
8044            }
8045            b.create("d").unwrap().write_raw(&data).unwrap();
8046            file.close().unwrap();
8047            path
8048        }
8049
8050        /// How many images the reader holds for `d`.
8051        fn held(reader: &Hdf5Reader) -> usize {
8052            let pos = reader.dataset_position("d").unwrap();
8053            let DataLayoutMessage::ChunkedV4 { index_address, .. } = &reader.datasets[pos].layout
8054            else {
8055                panic!("expected a version-4 chunked layout");
8056            };
8057            match reader.datasets.chunk_index(pos, *index_address) {
8058                Some(index) => index.images.held(),
8059                None => 0,
8060            }
8061        }
8062
8063        /// Four consecutive quarter-chunk slices are one chunk read four
8064        /// times: the first decodes it, the other three place it out of the
8065        /// image it left, and all four read what the file holds.
8066        #[test]
8067        #[cfg(feature = "deflate")]
8068        fn a_partial_read_keeps_the_chunk_it_did_not_finish() {
8069            let path = dataset("images_partial", 4096, 1024, true);
8070            let mut r = Hdf5Reader::open(&path).unwrap();
8071            let whole = {
8072                let mut fresh = Hdf5Reader::open(&path).unwrap();
8073                fresh.read_dataset_raw("d").unwrap()
8074            };
8075
8076            for q in 0..4u64 {
8077                let got = r.read_slice("d", &[q * 256], &[256]).unwrap();
8078                let from = (q as usize) * 256 * 8;
8079                assert_eq!(got, whole[from..from + 256 * 8], "quarter {q}");
8080                assert_eq!(held(&r), 1, "quarter {q} left the wrong image count");
8081            }
8082            let _ = std::fs::remove_file(&path);
8083        }
8084
8085        /// A whole-dataset read takes every chunk entire, so there is nothing
8086        /// a later read could want more of: it keeps nothing, and never pays
8087        /// to copy an image it is about to throw away.
8088        ///
8089        /// Two-dimensional on purpose. A one-dimensional full read decodes
8090        /// each chunk straight into the output and so never has an image to
8091        /// offer in the first place; here every chunk really is materialized,
8092        /// and only "the read consumed it" keeps it out of the cache.
8093        #[test]
8094        #[cfg(feature = "deflate")]
8095        fn a_whole_dataset_read_keeps_nothing() {
8096            let path = dataset_2d("images_full", 256, 256, [64, 64], true);
8097            let mut r = Hdf5Reader::open(&path).unwrap();
8098            r.read_dataset_raw("d").unwrap();
8099            assert_eq!(held(&r), 0, "a full read kept a chunk image");
8100            let _ = std::fs::remove_file(&path);
8101        }
8102
8103        /// The extent cutting the last chunk short does not make a full read
8104        /// look partial: what that chunk has to give is what the extent
8105        /// reaches, and the read took all of it.
8106        #[test]
8107        #[cfg(feature = "deflate")]
8108        fn a_whole_read_of_a_ragged_extent_keeps_nothing() {
8109            let path = dataset("images_ragged", 3000, 1024, true);
8110            let mut r = Hdf5Reader::open(&path).unwrap();
8111            r.read_dataset_raw("d").unwrap();
8112            assert_eq!(held(&r), 0, "the ragged edge chunk was kept");
8113            let _ = std::fs::remove_file(&path);
8114        }
8115
8116        /// An unfiltered chunk never reaches the cache, whatever a read does
8117        /// with it: its stored bytes are the dataset's bytes, so a partial
8118        /// read has nothing to inflate and nothing to save by inflating once.
8119        ///
8120        /// A single-column selection is the case that bites. Its runs are one
8121        /// element each, far too many for a positioned read apiece, so
8122        /// `read_chunk_runs_into` declines and every chunk really is read and
8123        /// materialized whole — leaving an image the cache would take if it
8124        /// were offered one.
8125        #[test]
8126        fn an_unfiltered_partial_read_keeps_nothing() {
8127            let path = dataset_2d("images_plain", 256, 256, [64, 64], false);
8128            let mut r = Hdf5Reader::open(&path).unwrap();
8129            let column = r.read_slice("d", &[0, 0], &[256, 1]).unwrap();
8130            assert_eq!(column.len(), 256 * 8);
8131            assert_eq!(held(&r), 0, "an unfiltered chunk reached the cache");
8132            let _ = std::fs::remove_file(&path);
8133        }
8134
8135        /// The same single-column walk over a *filtered* dataset is the
8136        /// region-of-interest case the cache exists for: each column of chunks
8137        /// inflates once however many columns are read out of it.
8138        #[test]
8139        #[cfg(feature = "deflate")]
8140        fn a_filtered_column_walk_reuses_each_chunk() {
8141            let path = dataset_2d("images_column", 256, 256, [64, 64], true);
8142            let mut warm = Hdf5Reader::open(&path).unwrap();
8143            for c in 0..4u64 {
8144                let mut cold = Hdf5Reader::open(&path).unwrap();
8145                assert_eq!(
8146                    warm.read_slice("d", &[0, c], &[256, 1]).unwrap(),
8147                    cold.read_slice("d", &[0, c], &[256, 1]).unwrap(),
8148                    "column {c} differed once served from the cache"
8149                );
8150            }
8151            // The four chunks of the first chunk-column, still held after the
8152            // fourth read of them.
8153            assert_eq!(held(&warm), 4, "the column walk re-inflated its chunks");
8154            let _ = std::fs::remove_file(&path);
8155        }
8156
8157        /// A chunk whose image is larger than the whole budget would evict
8158        /// everything to hold one entry, so it is never kept.
8159        #[test]
8160        #[cfg(feature = "deflate")]
8161        fn a_chunk_over_the_budget_is_never_kept() {
8162            let over = CHUNK_CACHE_BYTES / 8 + 1;
8163            let path = dataset("images_oversize", over * 2, over, true);
8164            let mut r = Hdf5Reader::open(&path).unwrap();
8165            r.read_slice("d", &[0], &[1024]).unwrap();
8166            assert_eq!(held(&r), 0, "a chunk over the budget was kept");
8167            let _ = std::fs::remove_file(&path);
8168        }
8169
8170        /// The budget's boundary, on the state machine itself: an image of
8171        /// exactly the budget is kept, and one byte more is refused — refused
8172        /// rather than admitted and then evicted, so that what the cache
8173        /// already holds survives an image that was never going to fit.
8174        #[test]
8175        fn an_image_over_the_budget_never_evicts_the_ones_under_it() {
8176            let cache = ChunkImageCache::default();
8177            let key = |addr| ChunkImageKey {
8178                addr,
8179                len: 1,
8180                mask: 0,
8181            };
8182            for i in 0..3 {
8183                cache.keep(key(i), vec![0u8; CHUNK_CACHE_BYTES / 4]);
8184            }
8185            assert_eq!(cache.held(), 3, "three quarter-budget images did not fit");
8186
8187            cache.keep(key(9), vec![0u8; CHUNK_CACHE_BYTES + 1]);
8188            assert_eq!(
8189                cache.held(),
8190                3,
8191                "an image that cannot fit evicted the ones that did"
8192            );
8193
8194            cache.keep(key(8), vec![0u8; CHUNK_CACHE_BYTES]);
8195            assert_eq!(
8196                cache.held(),
8197                1,
8198                "an image of exactly the budget was refused"
8199            );
8200        }
8201
8202        /// The budget is bytes: eight partial reads of eight 256 KiB chunks
8203        /// leave the four the budget pays for, not eight.
8204        #[test]
8205        #[cfg(feature = "deflate")]
8206        fn the_cache_holds_no_more_than_its_byte_budget() {
8207            let chunk = 32 * 1024; // f64 -> 256 KiB
8208            let path = dataset("images_budget", chunk * 8, chunk, true);
8209            let mut r = Hdf5Reader::open(&path).unwrap();
8210            for c in 0..8u64 {
8211                r.read_slice("d", &[c * chunk as u64], &[1024]).unwrap();
8212            }
8213            assert_eq!(
8214                held(&r),
8215                CHUNK_CACHE_BYTES / (chunk * 8),
8216                "the cache outgrew its byte budget"
8217            );
8218            let _ = std::fs::remove_file(&path);
8219        }
8220
8221        /// Reaching an entry mutably drops the index decoded against it, and
8222        /// the images live in that index — so what a read cached against the
8223        /// old entry cannot be served against the new one.
8224        #[test]
8225        #[cfg(feature = "deflate")]
8226        fn changing_an_entry_drops_the_chunk_images() {
8227            let path = dataset("images_entry_mut", 4096, 1024, true);
8228            let mut r = Hdf5Reader::open(&path).unwrap();
8229            r.read_slice("d", &[0], &[256]).unwrap();
8230            assert_eq!(held(&r), 1, "the read kept no image to invalidate");
8231
8232            let pos = r.dataset_position("d").unwrap();
8233            let _ = r.datasets.entry_mut(pos);
8234            assert_eq!(held(&r), 0, "a mutable entry left its chunk images behind");
8235            let _ = std::fs::remove_file(&path);
8236        }
8237
8238        /// Every slice a cache-holding reader returns is the slice a reader
8239        /// that never cached anything returns, over slices that start inside
8240        /// a chunk, end inside one, and span several.
8241        #[test]
8242        #[cfg(feature = "deflate")]
8243        fn a_cached_chunk_places_what_a_cold_read_places() {
8244            let path = dataset("images_differential", 4096, 1024, true);
8245            let mut warm = Hdf5Reader::open(&path).unwrap();
8246            for &(start, count) in &[
8247                (0u64, 100u64),
8248                (100, 100),
8249                (900, 300),
8250                (1024, 1),
8251                (1500, 2000),
8252                (3000, 1096),
8253                (4095, 1),
8254                (0, 4096),
8255            ] {
8256                let mut cold = Hdf5Reader::open(&path).unwrap();
8257                assert_eq!(
8258                    warm.read_slice("d", &[start], &[count]).unwrap(),
8259                    cold.read_slice("d", &[start], &[count]).unwrap(),
8260                    "slice {start}+{count} differed once served from the cache"
8261                );
8262            }
8263            let _ = std::fs::remove_file(&path);
8264        }
8265    }
8266
8267    /// `place_chunk_jobs` is the single owner of "every byte of a chunked
8268    /// read's output is defined": it skips the blanket fill only when
8269    /// [`planned_coverage`] proves the plan reaches every output byte, and
8270    /// fills whatever a plan or a short image leaves behind. The buffer these
8271    /// hand it is poisoned, so a byte no one defines shows up as `0xAA`.
8272    mod chunk_fill_plan {
8273        use super::*;
8274
8275        /// dims [8] of 1-byte elements in chunks of 4: slot 0 holds `ABCD` at
8276        /// file offset 0, slot 1 holds `EFGH` at offset 4.
8277        const DIMS: [u64; 1] = [8];
8278        const CHUNKS: [u64; 1] = [4];
8279        const FILL: [u8; 1] = [0x7E];
8280
8281        fn place(
8282            name: &str,
8283            jobs: Vec<Option<ChunkReadJob>>,
8284            coords: &[u64],
8285        ) -> (std::path::PathBuf, Vec<u8>) {
8286            let (path, handle) = handle_over(name, b"ABCDEFGH");
8287            let geo = ChunkOutputGeometry {
8288                dims: &DIMS,
8289                chunk_dims: &CHUNKS,
8290                element_size: 1,
8291            };
8292            let mut out = vec![0xAAu8; 8];
8293            place_chunk_jobs(
8294                &handle,
8295                jobs,
8296                coords,
8297                ChunkReadRequest {
8298                    pipeline: None,
8299                    target: ChunkTarget::Full,
8300                    fill_value: Some(&FILL),
8301                    dst: ReadDst::Fresh,
8302                },
8303                &geo,
8304                None,
8305                &mut out,
8306            )
8307            .unwrap();
8308            (path, out)
8309        }
8310
8311        fn job(addr: u64, len: usize) -> Option<ChunkReadJob> {
8312            Some(ChunkReadJob {
8313                addr,
8314                len,
8315                at_most: true,
8316                mask: 0,
8317            })
8318        }
8319
8320        /// The owner path: a plan that tiles the output places every byte, so
8321        /// nothing is filled and nothing is left poisoned.
8322        #[test]
8323        fn a_plan_that_covers_the_output_leaves_no_byte_to_fill() {
8324            let geo = ChunkOutputGeometry {
8325                dims: &DIMS,
8326                chunk_dims: &CHUNKS,
8327                element_size: 1,
8328            };
8329            let zeros = [0u64];
8330            let placement = ChunkPlacement::resolve(&geo, ChunkTarget::Full, &zeros);
8331            let jobs = vec![job(0, 4), job(4, 4)];
8332            assert_eq!(
8333                planned_coverage(&placement, &jobs, &[0, 1]),
8334                8,
8335                "a tiling plan must measure as covering the whole output"
8336            );
8337            let (path, out) = place("cover_full", jobs, &[0, 1]);
8338            assert_eq!(out, b"ABCDEFGH");
8339            let _ = std::fs::remove_file(path);
8340        }
8341
8342        /// A slot the plan never reaches reads back as the tiled fill value,
8343        /// which is what the blanket pre-fill used to guarantee.
8344        #[test]
8345        fn a_slot_the_plan_skips_reads_back_as_fill() {
8346            let (path, out) = place("cover_gap", vec![job(0, 4), None], &[0, 1]);
8347            assert_eq!(out, b"ABCD~~~~");
8348            let _ = std::fs::remove_file(path);
8349        }
8350
8351        /// A corrupt index naming one chunk-grid slot twice must not be
8352        /// mistaken for a plan that covers two slots: counting the duplicate
8353        /// once leaves the coverage short, so the fill still goes down and the
8354        /// slot no entry named reads as fill rather than as poison.
8355        #[test]
8356        fn duplicate_chunk_coordinates_still_leave_defined_output() {
8357            let (path, out) = place("cover_dup", vec![job(0, 4), job(0, 4)], &[0, 0]);
8358            assert_eq!(out, b"ABCD~~~~");
8359            let _ = std::fs::remove_file(path);
8360        }
8361
8362        /// A chunk whose stored image is shorter than the read box needs it to
8363        /// be places no run at all; the plan counted it, so the shortfall is
8364        /// filled after placement instead of before it.
8365        #[test]
8366        fn a_chunk_read_short_has_its_runs_filled_after_placement() {
8367            let (path, out) = place("cover_short", vec![job(0, 4), job(4, 2)], &[0, 1]);
8368            assert_eq!(out, b"ABCD~~~~");
8369            let _ = std::fs::remove_file(path);
8370        }
8371
8372        /// A chunk the index places a few bytes short of `u64::MAX` has runs
8373        /// whose ends do not fit in a `u64`. Two rows of one chunk are the
8374        /// second run that asks whether it continues the first; the answer
8375        /// comes from the read at that address (fill for a read that may
8376        /// come up short, an error for one that may not), not from the
8377        /// arithmetic.
8378        #[test]
8379        fn a_chunk_address_near_u64_max_fails_the_read_not_the_arithmetic() {
8380            let (path, handle) = handle_over("cover_wrap", b"ABEFCDGH");
8381            let dims = [2u64, 4];
8382            let chunks = [2u64, 2];
8383            let geo = ChunkOutputGeometry {
8384                dims: &dims,
8385                chunk_dims: &chunks,
8386                element_size: 1,
8387            };
8388            let run = |at_most: bool, out: &mut [u8]| {
8389                place_chunk_jobs(
8390                    &handle,
8391                    vec![
8392                        job(0, 4),
8393                        Some(ChunkReadJob {
8394                            addr: u64::MAX - 1,
8395                            len: 4,
8396                            at_most,
8397                            mask: 0,
8398                        }),
8399                    ],
8400                    &[0, 0, 0, 1],
8401                    ChunkReadRequest {
8402                        pipeline: None,
8403                        target: ChunkTarget::Full,
8404                        fill_value: Some(&FILL),
8405                        dst: ReadDst::Fresh,
8406                    },
8407                    &geo,
8408                    None,
8409                    out,
8410                )
8411            };
8412            let mut out = vec![0xAAu8; 8];
8413            run(true, &mut out).unwrap();
8414            assert_eq!(out, b"AB~~EF~~");
8415            let mut out = vec![0xAAu8; 8];
8416            run(false, &mut out).expect_err("an exact read at u64::MAX - 1 cannot succeed");
8417            let _ = std::fs::remove_file(path);
8418        }
8419    }
8420
8421    /// Helper: write a little-endian u64 truncated to `n` bytes.
8422    fn write_le(buf: &mut Vec<u8>, value: u64, n: usize) {
8423        buf.extend_from_slice(&value.to_le_bytes()[..n]);
8424    }
8425
8426    /// Build a minimal v0 HDF5 file in memory with one dataset containing
8427    /// `dataset_data`. Returns the complete file bytes.
8428    ///
8429    /// The file structure is:
8430    /// - Superblock v0 with root group STE
8431    /// - Root group object header (v1) with symbol table message
8432    /// - Local heap (header + data) with dataset name
8433    /// - B-tree v1 (group, leaf) pointing to one SNOD
8434    /// - SNOD with one entry for the dataset
8435    /// - Dataset object header (v1) with dataspace, datatype, layout messages
8436    /// - Raw dataset data (contiguous)
8437    fn build_v0_file(dataset_name: &str, dims: &[u64], data: &[u8]) -> Vec<u8> {
8438        let sa: usize = 8; // sizeof_addr
8439        let ss: usize = 8; // sizeof_size
8440        let ndims = dims.len();
8441        let element_size = data.len() as u64 / dims.iter().product::<u64>();
8442
8443        // We'll lay out the file regions in order, computing offsets as we go.
8444        let mut file = Vec::new();
8445
8446        // ---- Plan layout offsets ----
8447        // We need to know the addresses before writing, so let's compute them.
8448        // Superblock: starts at 0
8449        let sb_size = 8 + 8 + 4 + 4 * sa + (ss + sa + 4 + 4 + 16); // sig + header + flags + 4 addrs + STE
8450                                                                   // Pad to 8-byte alignment
8451        let sb_size_aligned = (sb_size + 7) & !7;
8452
8453        // Root group object header (v1): after superblock
8454        let root_ohdr_addr = sb_size_aligned as u64;
8455        // The root ohdr contains a symbol table message (type 0x11):
8456        //   btree_addr(8) + heap_addr(8) = 16 bytes
8457        // v1 message wire format: type(2) + size(2) + flags(1) + reserved(3) + data
8458        let stab_msg_data_size = 2 * sa; // btree + heap addr
8459        let stab_msg_wire = 8 + stab_msg_data_size;
8460        let stab_msg_wire_aligned = (stab_msg_wire + 7) & !7;
8461        let root_ohdr_data_size = stab_msg_wire_aligned;
8462        let root_ohdr_total = 16 + root_ohdr_data_size; // v1 16-byte prefix + messages
8463        let root_ohdr_total_aligned = (root_ohdr_total + 7) & !7;
8464
8465        // Local heap header: after root ohdr
8466        let heap_hdr_addr = root_ohdr_addr + root_ohdr_total_aligned as u64;
8467        let heap_hdr_size = 4 + 1 + 3 + ss + ss + sa;
8468        let heap_hdr_size_aligned = (heap_hdr_size + 7) & !7;
8469
8470        // Local heap data: after heap header
8471        let heap_data_addr = heap_hdr_addr + heap_hdr_size_aligned as u64;
8472        // Data: empty string at offset 0 (for root), then dataset_name at offset 1
8473        let name_bytes = dataset_name.as_bytes();
8474        let heap_data_content_size = 1 + name_bytes.len() + 1; // \0 + name + \0
8475        let heap_data_size = (heap_data_content_size + 7) & !7;
8476
8477        // B-tree v1 node: after heap data
8478        let btree_addr = heap_data_addr + heap_data_size as u64;
8479        // B-tree header: TREE(4) + type(1) + level(1) + entries_used(2) + left(sa) + right(sa)
8480        // Plus interleaved keys/children: key[0](ss), child[0](sa), key[1](ss)
8481        let btree_size = 4 + 1 + 1 + 2 + 2 * sa + 2 * ss + sa;
8482        let btree_size_aligned = (btree_size + 7) & !7;
8483
8484        // SNOD: after B-tree
8485        let snod_addr = btree_addr + btree_size_aligned as u64;
8486        // SNOD header: SNOD(4) + version(1) + reserved(1) + num_symbols(2)
8487        // + 1 entry: name_offset(ss) + obj_header_addr(sa) + cache_type(4) + reserved(4) + scratch(16)
8488        let entry_size = ss + sa + 4 + 4 + 16;
8489        let snod_size = 8 + entry_size;
8490        let snod_size_aligned = (snod_size + 7) & !7;
8491
8492        // Dataset object header (v1): after SNOD
8493        let ds_ohdr_addr = snod_addr + snod_size_aligned as u64;
8494        // Messages: dataspace(0x01), datatype(0x03), data_layout(0x08)
8495
8496        // Dataspace v1: version(1) + ndims(1) + flags(1) + reserved(1) + reserved(4) + ndims*ss
8497        let ds_msg_data_size = 8 + ndims * ss;
8498        let ds_msg_wire = 8 + ds_msg_data_size;
8499        let ds_msg_wire_aligned = (ds_msg_wire + 7) & !7;
8500
8501        // Datatype: for integer types, 12 bytes
8502        // Use i32: class=0, version=1, size=4, bit_offset=0, bit_precision=32, signed
8503        let dt_msg_data_size = 12;
8504        let dt_msg_wire = 8 + dt_msg_data_size;
8505        let dt_msg_wire_aligned = (dt_msg_wire + 7) & !7;
8506
8507        // Data layout v3 contiguous: version(1) + class(1) + addr(sa) + size(ss)
8508        let dl_msg_data_size = 2 + sa + ss;
8509        let dl_msg_wire = 8 + dl_msg_data_size;
8510        let dl_msg_wire_aligned = (dl_msg_wire + 7) & !7;
8511
8512        let ds_ohdr_data_size = ds_msg_wire_aligned + dt_msg_wire_aligned + dl_msg_wire_aligned;
8513        let ds_ohdr_total = 16 + ds_ohdr_data_size; // v1 16-byte prefix
8514        let ds_ohdr_total_aligned = (ds_ohdr_total + 7) & !7;
8515
8516        // Raw data: after dataset object header
8517        let raw_data_addr = ds_ohdr_addr + ds_ohdr_total_aligned as u64;
8518        let raw_data_size = data.len();
8519
8520        let eof = raw_data_addr + raw_data_size as u64;
8521
8522        // ---- Write the file ----
8523
8524        // 1. Superblock v0
8525        let sig: [u8; 8] = [0x89, 0x48, 0x44, 0x46, 0x0d, 0x0a, 0x1a, 0x0a];
8526        file.extend_from_slice(&sig);
8527        file.push(0); // version 0
8528        file.push(0); // free-space version
8529        file.push(0); // root group STE version
8530        file.push(0); // reserved
8531        file.push(0); // shared header version
8532        file.push(sa as u8); // sizeof_addr
8533        file.push(ss as u8); // sizeof_size
8534        file.push(0); // reserved
8535        file.extend_from_slice(&4u16.to_le_bytes()); // sym_leaf_k
8536        file.extend_from_slice(&32u16.to_le_bytes()); // btree_internal_k
8537        file.extend_from_slice(&0u32.to_le_bytes()); // file_consistency_flags
8538                                                     // base_addr
8539        write_le(&mut file, 0, sa);
8540        // extension_addr = UNDEF
8541        write_le(&mut file, UNDEF_ADDR, sa);
8542        // eof_addr
8543        write_le(&mut file, eof, sa);
8544        // driver_info_addr = UNDEF
8545        write_le(&mut file, UNDEF_ADDR, sa);
8546        // Root group STE:
8547        write_le(&mut file, 0, ss); // name_offset
8548        write_le(&mut file, root_ohdr_addr, sa); // obj_header_addr
8549        file.extend_from_slice(&1u32.to_le_bytes()); // cache_type = 1 (stab)
8550        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
8551                                                     // scratch pad: btree_addr + heap_addr
8552        write_le(&mut file, btree_addr, sa);
8553        write_le(&mut file, heap_hdr_addr, sa);
8554        // Pad superblock
8555        while file.len() < sb_size_aligned {
8556            file.push(0);
8557        }
8558
8559        // 2. Root group object header (v1, 16-byte prefix)
8560        assert_eq!(file.len(), root_ohdr_addr as usize);
8561        file.push(1); // version
8562        file.push(0); // reserved
8563        file.extend_from_slice(&1u16.to_le_bytes()); // num_messages = 1
8564        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
8565        file.extend_from_slice(&(root_ohdr_data_size as u32).to_le_bytes());
8566        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)
8567                                           // Symbol table message (type 0x0011)
8568        file.extend_from_slice(&0x0011u16.to_le_bytes()); // type
8569        file.extend_from_slice(&(stab_msg_data_size as u16).to_le_bytes()); // size
8570        file.push(0); // flags
8571        file.extend_from_slice(&[0u8; 3]); // reserved
8572        write_le(&mut file, btree_addr, sa);
8573        write_le(&mut file, heap_hdr_addr, sa);
8574        // Pad
8575        while file.len() < (root_ohdr_addr as usize + root_ohdr_total_aligned) {
8576            file.push(0);
8577        }
8578
8579        // 3. Local heap header
8580        assert_eq!(file.len(), heap_hdr_addr as usize);
8581        file.extend_from_slice(b"HEAP");
8582        file.push(0); // version
8583        file.extend_from_slice(&[0u8; 3]); // reserved
8584        write_le(&mut file, heap_data_size as u64, ss); // data_size
8585        write_le(&mut file, u64::MAX, ss); // free_list_offset (none)
8586        write_le(&mut file, heap_data_addr, sa); // data_addr
8587        while file.len() < (heap_hdr_addr as usize + heap_hdr_size_aligned) {
8588            file.push(0);
8589        }
8590
8591        // 4. Local heap data
8592        assert_eq!(file.len(), heap_data_addr as usize);
8593        file.push(0); // offset 0: empty string (root self-reference)
8594        file.extend_from_slice(name_bytes); // offset 1: dataset name
8595        file.push(0); // null terminator
8596        while file.len() < (heap_data_addr as usize + heap_data_size) {
8597            file.push(0);
8598        }
8599
8600        // 5. B-tree v1 (leaf, 1 entry)
8601        assert_eq!(file.len(), btree_addr as usize);
8602        file.extend_from_slice(b"TREE");
8603        file.push(0); // type = group
8604        file.push(0); // level = leaf
8605        file.extend_from_slice(&1u16.to_le_bytes()); // entries_used = 1
8606        write_le(&mut file, UNDEF_ADDR, sa); // left sibling
8607        write_le(&mut file, UNDEF_ADDR, sa); // right sibling
8608                                             // key[0] = 0 (first name offset)
8609        write_le(&mut file, 0, ss);
8610        // child[0] = snod_addr
8611        write_le(&mut file, snod_addr, sa);
8612        // key[1] = dataset name offset (after root)
8613        write_le(&mut file, 1, ss);
8614        while file.len() < (btree_addr as usize + btree_size_aligned) {
8615            file.push(0);
8616        }
8617
8618        // 6. SNOD with 1 entry
8619        assert_eq!(file.len(), snod_addr as usize);
8620        file.extend_from_slice(b"SNOD");
8621        file.push(1); // version
8622        file.push(0); // reserved
8623        file.extend_from_slice(&1u16.to_le_bytes()); // num_symbols = 1
8624                                                     // Entry: dataset
8625        write_le(&mut file, 1, ss); // name_offset = 1 (index into local heap)
8626        write_le(&mut file, ds_ohdr_addr, sa); // obj_header_addr
8627        file.extend_from_slice(&0u32.to_le_bytes()); // cache_type = 0 (not a group)
8628        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
8629        file.extend_from_slice(&[0u8; 16]); // scratch pad (unused)
8630        while file.len() < (snod_addr as usize + snod_size_aligned) {
8631            file.push(0);
8632        }
8633
8634        // 7. Dataset object header (v1, 16-byte prefix)
8635        assert_eq!(file.len(), ds_ohdr_addr as usize);
8636        file.push(1); // version
8637        file.push(0); // reserved
8638        file.extend_from_slice(&3u16.to_le_bytes()); // num_messages = 3
8639        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
8640        file.extend_from_slice(&(ds_ohdr_data_size as u32).to_le_bytes());
8641        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)
8642
8643        // Message 1: Dataspace (type 0x01) - version 1
8644        file.extend_from_slice(&0x0001u16.to_le_bytes());
8645        file.extend_from_slice(&(ds_msg_data_size as u16).to_le_bytes());
8646        file.push(0); // flags
8647        file.extend_from_slice(&[0u8; 3]); // reserved
8648                                           // Dataspace v1 payload:
8649        file.push(1); // version = 1
8650        file.push(ndims as u8);
8651        file.push(0); // flags (no max dims)
8652        file.push(0); // reserved
8653        file.extend_from_slice(&[0u8; 4]); // reserved (4 bytes)
8654        for &d in dims {
8655            write_le(&mut file, d, ss);
8656        }
8657        // Pad message
8658        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned;
8659        while file.len() < target {
8660            file.push(0);
8661        }
8662
8663        // Message 2: Datatype (type 0x03) - i32
8664        file.extend_from_slice(&0x0003u16.to_le_bytes());
8665        file.extend_from_slice(&(dt_msg_data_size as u16).to_le_bytes());
8666        file.push(0); // flags
8667        file.extend_from_slice(&[0u8; 3]); // reserved
8668                                           // Datatype payload: class=0 (fixed point), version=1
8669        file.push(0x10); // class(0) | version(1)<<4
8670        file.push(0x08); // byte_order=LE, signed=true (bit 3)
8671        file.push(0); // flags byte 1
8672        file.push(0); // flags byte 2
8673        file.extend_from_slice(&(element_size as u32).to_le_bytes()); // element size
8674        file.extend_from_slice(&0u16.to_le_bytes()); // bit_offset
8675        file.extend_from_slice(&((element_size * 8) as u16).to_le_bytes()); // bit_precision
8676        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned + dt_msg_wire_aligned;
8677        while file.len() < target {
8678            file.push(0);
8679        }
8680
8681        // Message 3: Data Layout (type 0x08) - contiguous v3
8682        file.extend_from_slice(&0x0008u16.to_le_bytes());
8683        file.extend_from_slice(&(dl_msg_data_size as u16).to_le_bytes());
8684        file.push(0); // flags
8685        file.extend_from_slice(&[0u8; 3]); // reserved
8686                                           // Data layout payload:
8687        file.push(3); // version = 3
8688        file.push(1); // class = contiguous
8689        write_le(&mut file, raw_data_addr, sa); // address
8690        write_le(&mut file, raw_data_size as u64, ss); // size
8691        let target = ds_ohdr_addr as usize + ds_ohdr_total_aligned;
8692        while file.len() < target {
8693            file.push(0);
8694        }
8695
8696        // 8. Raw data
8697        assert_eq!(file.len(), raw_data_addr as usize);
8698        file.extend_from_slice(data);
8699
8700        assert_eq!(file.len(), eof as usize);
8701        file
8702    }
8703
8704    #[test]
8705    fn test_read_v0_file_with_one_dataset() {
8706        let dims = [3u64, 4];
8707        let values: Vec<i32> = (0..12).collect();
8708        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8709
8710        let file_bytes = build_v0_file("my_dataset", &dims, &raw_data);
8711
8712        // Write to a temp file
8713        let path = temp_path("v0_reader");
8714        {
8715            let mut f = std::fs::File::create(&path).unwrap();
8716            f.write_all(&file_bytes).unwrap();
8717            f.sync_all().unwrap();
8718        }
8719
8720        // Read it back
8721        let mut reader = Hdf5Reader::open(&path).unwrap();
8722        let names = reader.dataset_names();
8723        assert_eq!(names, vec!["my_dataset"]);
8724
8725        let shape = reader.dataset_shape("my_dataset").unwrap();
8726        assert_eq!(shape, vec![3, 4]);
8727
8728        let data = reader.read_dataset_raw("my_dataset").unwrap();
8729        assert_eq!(data, raw_data);
8730
8731        // Verify the values
8732        let read_values: Vec<i32> = data
8733            .as_chunks::<4>()
8734            .0
8735            .iter()
8736            .map(|c| i32::from_le_bytes(*c))
8737            .collect();
8738        assert_eq!(read_values, values);
8739
8740        std::fs::remove_file(&path).ok();
8741    }
8742
8743    #[test]
8744    fn test_read_v0_file_1d_dataset() {
8745        let dims = [5u64];
8746        let values: Vec<i32> = vec![100, 200, 300, 400, 500];
8747        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8748
8749        let file_bytes = build_v0_file("data_1d", &dims, &raw_data);
8750
8751        let path = temp_path("v0_1d");
8752        {
8753            let mut f = std::fs::File::create(&path).unwrap();
8754            f.write_all(&file_bytes).unwrap();
8755        }
8756
8757        let mut reader = Hdf5Reader::open(&path).unwrap();
8758        assert_eq!(reader.dataset_names(), vec!["data_1d"]);
8759        assert_eq!(reader.dataset_shape("data_1d").unwrap(), vec![5]);
8760
8761        let data = reader.read_dataset_raw("data_1d").unwrap();
8762        let read_values: Vec<i32> = data
8763            .as_chunks::<4>()
8764            .0
8765            .iter()
8766            .map(|c| i32::from_le_bytes(*c))
8767            .collect();
8768        assert_eq!(read_values, values);
8769
8770        std::fs::remove_file(&path).ok();
8771    }
8772
8773    #[test]
8774    fn test_detect_v2v3_still_works() {
8775        // Verify that opening a v3 file written by our writer still works
8776        let path = temp_path("detect_v3");
8777        {
8778            use crate::io::writer::Hdf5Writer;
8779            let writer = Hdf5Writer::create(&path).unwrap();
8780            let datatype = crate::format::messages::datatype::DatatypeMessage::i32_type();
8781            let idx = writer.create_dataset("test", datatype, &[4]).unwrap();
8782            let data = [1i32, 2, 3, 4];
8783            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
8784            writer.write_dataset_raw(idx, &raw).unwrap();
8785            writer.close().unwrap();
8786        }
8787
8788        let mut reader = Hdf5Reader::open(&path).unwrap();
8789        assert_eq!(reader.dataset_names(), vec!["test"]);
8790        let shape = reader.dataset_shape("test").unwrap();
8791        assert_eq!(shape, vec![4]);
8792
8793        let data = reader.read_dataset_raw("test").unwrap();
8794        let vals: Vec<i32> = data
8795            .as_chunks::<4>()
8796            .0
8797            .iter()
8798            .map(|c| i32::from_le_bytes(*c))
8799            .collect();
8800        assert_eq!(vals, vec![1, 2, 3, 4]);
8801
8802        std::fs::remove_file(&path).ok();
8803    }
8804
8805    /// Collect the (src_off, out_off, len) runs the coalescer emits.
8806    fn collect_runs(
8807        dims: &[u64],
8808        starts: &[u64],
8809        counts: &[u64],
8810        es: u64,
8811    ) -> Vec<(u64, usize, usize)> {
8812        let mut v = Vec::new();
8813        for_each_contiguous_run(dims, starts, counts, es, |s, o, l| {
8814            v.push((s, o, l));
8815            Ok(())
8816        })
8817        .unwrap();
8818        v
8819    }
8820
8821    #[test]
8822    fn coalesce_1d_is_single_run() {
8823        // 1-D selection is always one contiguous run.
8824        assert_eq!(collect_runs(&[10], &[2], &[3], 1), vec![(2, 0, 3)]);
8825        // element_size scales offsets and length.
8826        assert_eq!(collect_runs(&[10], &[2], &[3], 4), vec![(8, 0, 12)]);
8827    }
8828
8829    #[test]
8830    fn coalesce_full_last_dim_merges_into_one_run() {
8831        // 2-D, full last dim => the whole [r0:r1, :] block is one run.
8832        // dims=[4,5], select rows 1..3, all 5 columns.
8833        assert_eq!(collect_runs(&[4, 5], &[1, 0], &[2, 5], 1), vec![(5, 0, 10)]);
8834    }
8835
8836    #[test]
8837    fn coalesce_partial_last_dim_keeps_one_run_per_row() {
8838        // 2-D, partial last dim => no merge; one run per selected row.
8839        // dims=[4,5] strides=[5,1]; select rows 1..3, cols 1..4.
8840        assert_eq!(
8841            collect_runs(&[4, 5], &[1, 1], &[2, 3], 1),
8842            vec![(6, 0, 3), (11, 3, 3)]
8843        );
8844        // Same shape, element_size=4: strides=[20,4], inner_base=4.
8845        assert_eq!(
8846            collect_runs(&[4, 5], &[1, 1], &[2, 3], 4),
8847            vec![(24, 0, 12), (44, 12, 12)]
8848        );
8849    }
8850
8851    #[test]
8852    fn coalesce_3d_reported_case_one_run_per_outer_index() {
8853        // The reported workload: [:, r0:r1, :] of [nproj, nz, nx].
8854        // dims=[3,4,5] strides=[20,5,1]; select all of dim0, rows 1..3 of
8855        // dim1, all of dim2. Last dim full => merge dim1+dim2; dim1 partial
8856        // => one run per dim0 index (3 runs, not 3*2=6 rows).
8857        assert_eq!(
8858            collect_runs(&[3, 4, 5], &[0, 1, 0], &[3, 2, 5], 1),
8859            vec![(5, 0, 10), (25, 10, 10), (45, 20, 10)]
8860        );
8861    }
8862
8863    #[test]
8864    fn coalesce_3d_full_inner_dims_is_single_run() {
8865        // [r0:r1, :, :] => both inner dims full => one contiguous run.
8866        // dims=[3,4,5] strides=[20,5,1]; select rows 1..3 of dim0.
8867        assert_eq!(
8868            collect_runs(&[3, 4, 5], &[1, 0, 0], &[2, 4, 5], 1),
8869            vec![(20, 0, 40)]
8870        );
8871    }
8872
8873    /// Build a contiguous i32 dataset and verify `read_slice` returns the
8874    /// correct bytes for both coalesced and non-coalesced selections.
8875    #[test]
8876    fn read_slice_contiguous_3d_matches_naive_extraction() {
8877        let dims = [3u64, 4, 5];
8878        let total: usize = (dims[0] * dims[1] * dims[2]) as usize;
8879        let values: Vec<i32> = (0..total as i32).collect();
8880        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8881        let file_bytes = build_v0_file("vol", &dims, &raw_data);
8882
8883        let path = temp_path("slice_3d_contig");
8884        {
8885            let mut f = std::fs::File::create(&path).unwrap();
8886            f.write_all(&file_bytes).unwrap();
8887            f.sync_all().unwrap();
8888        }
8889        let mut reader = Hdf5Reader::open(&path).unwrap();
8890
8891        // Naive row-major extraction for an arbitrary [starts, counts).
8892        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
8893            let mut out = Vec::new();
8894            for i in 0..counts[0] {
8895                for j in 0..counts[1] {
8896                    for k in 0..counts[2] {
8897                        let gi = starts[0] + i;
8898                        let gj = starts[1] + j;
8899                        let gk = starts[2] + k;
8900                        out.push(values[(gi * dims[1] * dims[2] + gj * dims[2] + gk) as usize]);
8901                    }
8902                }
8903            }
8904            out
8905        };
8906        let decode = |raw: Vec<u8>| -> Vec<i32> {
8907            raw.as_chunks::<4>()
8908                .0
8909                .iter()
8910                .map(|c| i32::from_le_bytes(*c))
8911                .collect()
8912        };
8913
8914        // Mix of coalesced and non-coalesced selections.
8915        let cases: &[([u64; 3], [u64; 3])] = &[
8916            ([0, 1, 0], [3, 2, 5]), // [:, 1:3, :]  -> coalesced (3 runs)
8917            ([1, 0, 0], [2, 4, 5]), // [1:3, :, :]  -> single run
8918            ([0, 0, 1], [3, 4, 3]), // [:, :, 1:4]  -> partial last dim, no merge
8919            ([1, 2, 1], [2, 2, 4]), // interior block, partial all dims
8920            ([0, 0, 0], [3, 4, 5]), // whole dataset -> single run
8921            ([2, 3, 4], [1, 1, 1]), // single element
8922        ];
8923        for &(starts, counts) in cases {
8924            let got = decode(reader.read_slice("vol", &starts, &counts).unwrap());
8925            assert_eq!(
8926                got,
8927                expect(starts, counts),
8928                "slice starts={starts:?} counts={counts:?}"
8929            );
8930        }
8931
8932        let _ = std::fs::remove_file(&path);
8933    }
8934
8935    /// Run the collector against a header built by hand. The handle is only
8936    /// touched when a message sends the collector to the heap, which none of
8937    /// these do, so an empty file is enough of a file.
8938    fn collect_from(
8939        messages: Vec<crate::format::object_header::ObjectHeaderMessage>,
8940    ) -> Result<Vec<String>, String> {
8941        let path = temp_path("collect");
8942        std::fs::File::create(&path).unwrap();
8943        let mut handle = FileHandle::open_read(&path).unwrap();
8944        let ctx = FormatContext {
8945            sizeof_addr: 8,
8946            sizeof_size: 8,
8947        };
8948        let header = ObjectHeader {
8949            flags: 0x02,
8950            times: None,
8951            messages,
8952        };
8953        let attrs = collect_object_attributes(&mut handle, &ctx, &header);
8954        drop(handle);
8955        let _ = std::fs::remove_file(&path);
8956        attrs
8957            .complete("obj")
8958            .map(|e| e.iter().map(|a| a.name().to_string()).collect())
8959            .map_err(|e| e.to_string())
8960    }
8961
8962    fn msg(msg_type: u8, data: Vec<u8>) -> crate::format::object_header::ObjectHeaderMessage {
8963        crate::format::object_header::ObjectHeaderMessage {
8964            msg_type,
8965            flags: 0,
8966            data,
8967            creation_index: 0,
8968        }
8969    }
8970
8971    /// An attribute info message that will not decode takes the object's whole
8972    /// attribute set with it — the dense storage it names is where the
8973    /// attributes are. The listing must say so rather than come back short.
8974    #[test]
8975    fn an_undecodable_attribute_info_message_fails_the_listing() {
8976        // Version 9: `H5O_AINFO_VERSION_0` is the only one that exists.
8977        let err = collect_from(vec![msg(MSG_ATTR_INFO, vec![9, 0])]).unwrap_err();
8978        assert!(
8979            err.contains("attributes of 'obj' cannot be read whole")
8980                && err.contains("attribute info message"),
8981            "{err}"
8982        );
8983    }
8984
8985    /// An attribute message damaged past its own name has no name to be listed
8986    /// under, so it too is an object-level failure — the one case
8987    /// [`AttributeEntry::parse`] cannot name.
8988    #[test]
8989    fn an_unnameable_attribute_message_fails_the_listing() {
8990        let err = collect_from(vec![msg(MSG_ATTRIBUTE, vec![1, 0, 0])]).unwrap_err();
8991        assert!(
8992            err.contains("attributes of 'obj' cannot be read whole")
8993                && err.contains("attribute message"),
8994            "{err}"
8995        );
8996    }
8997
8998    /// The failure is the object's, not one name's: a set that reads whole
8999    /// still lists, and a compact attribute info message is not dense storage.
9000    #[test]
9001    fn a_whole_attribute_set_still_lists() {
9002        let ctx = FormatContext {
9003            sizeof_addr: 8,
9004            sizeof_size: 8,
9005        };
9006        let ainfo = crate::format::messages::attr_info::AttributeInfoMessage::compact();
9007        let names = collect_from(vec![msg(MSG_ATTR_INFO, ainfo.encode(&ctx))]).unwrap();
9008        assert!(names.is_empty(), "{names:?}");
9009    }
9010
9011    /// A space-padded fixed-length string attribute reads back without its
9012    /// padding: `H5T__conv_s_s` ends the value after the last non-space byte,
9013    /// and nothing else in the element marks where it stops.
9014    #[test]
9015    fn fixed_string_attr_value_honors_the_declared_pad() {
9016        use crate::format::messages::dataspace::DataspaceMessage;
9017        use crate::format::messages::datatype::DatatypeMessage;
9018
9019        let attr = |padding: u8, data: &[u8]| AttributeMessage {
9020            name: "units".to_string(),
9021            datatype: DatatypeMessage::FixedString {
9022                size: data.len() as u32,
9023                padding,
9024                charset: 0,
9025            },
9026            dataspace: DataspaceMessage::scalar(),
9027            data: data.to_vec(),
9028        };
9029
9030        // Space padded: no NUL anywhere, so truncating at the first NUL kept
9031        // the padding.
9032        assert_eq!(
9033            fixed_string_attr_value(&attr(2, b"volt    ")).unwrap(),
9034            "volt"
9035        );
9036        // Null terminated and null padded end at the first NUL, so trailing
9037        // spaces before it are content.
9038        assert_eq!(
9039            fixed_string_attr_value(&attr(0, b"volt  \0\0")).unwrap(),
9040            "volt  "
9041        );
9042        assert_eq!(
9043            fixed_string_attr_value(&attr(1, b"volt\0\0\0\0")).unwrap(),
9044            "volt"
9045        );
9046        // A reserved rule is named rather than guessed at.
9047        let err = fixed_string_attr_value(&attr(7, b"volt    ")).unwrap_err();
9048        assert!(
9049            err.to_string().contains("padding rule 7"),
9050            "unexpected error: {err}"
9051        );
9052    }
9053}
9054
9055#[cfg(test)]
9056mod h5py_debug_tests {
9057    use super::*;
9058
9059    #[test]
9060    fn debug_read_h5py() {
9061        let path = std::path::Path::new("/tmp/test_h5py_default.h5");
9062        if !path.exists() {
9063            return;
9064        }
9065
9066        let handle = FileHandle::open_read(path).unwrap();
9067        let sb_buf = handle.read_at_most(0, 1024).unwrap();
9068        let version = detect_superblock_version(&sb_buf).unwrap();
9069        eprintln!("Superblock version: {}", version);
9070
9071        let sb = SuperblockV0V1::decode(&sb_buf).unwrap();
9072        eprintln!(
9073            "sizeof_addr={}, sizeof_size={}",
9074            sb.sizeof_offsets, sb.sizeof_lengths
9075        );
9076        let (ste_btree, ste_heap) = sb
9077            .root_symbol_table_entry
9078            .cached_symbol_table()
9079            .unwrap_or((UNDEF_ADDR, UNDEF_ADDR));
9080        eprintln!(
9081            "STE: obj_header={}, cache={:?}, btree={}, heap={}",
9082            sb.root_symbol_table_entry.obj_header_addr,
9083            sb.root_symbol_table_entry.cache,
9084            ste_btree,
9085            ste_heap
9086        );
9087
9088        let ctx = FormatContext {
9089            sizeof_addr: sb.sizeof_offsets,
9090            sizeof_size: sb.sizeof_lengths,
9091        };
9092
9093        // Read local heap
9094        let heap_buf = handle.read_at_most(ste_heap, 128).unwrap();
9095        let heap_hdr = LocalHeapHeader::decode(
9096            &heap_buf,
9097            ctx.sizeof_addr as usize,
9098            ctx.sizeof_size as usize,
9099        )
9100        .unwrap();
9101        eprintln!(
9102            "Heap data_addr={}, data_size={}",
9103            heap_hdr.data_addr, heap_hdr.data_size
9104        );
9105
9106        let heap_data = handle
9107            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)
9108            .unwrap();
9109        eprintln!(
9110            "Heap data bytes: {:?}",
9111            &heap_data[..std::cmp::min(64, heap_data.len())]
9112        );
9113
9114        // Read btree
9115        let btree_buf = handle.read_at_most(ste_btree, 8192).unwrap();
9116        let btree = BTreeV1Node::decode(
9117            &btree_buf,
9118            ctx.sizeof_addr as usize,
9119            ctx.sizeof_size as usize,
9120            BTreeV1Config::default().snode_max_entries(),
9121        )
9122        .unwrap();
9123        eprintln!(
9124            "BTree: type={}, level={}, entries={}, children={:?}",
9125            btree.node_type, btree.level, btree.entries_used, btree.children
9126        );
9127
9128        // Read SNOD
9129        for &child in &btree.children {
9130            let snod_buf = handle.read_at_most(child, 8192).unwrap();
9131            let snod = SymbolTableNode::decode(
9132                &snod_buf,
9133                ctx.sizeof_addr as usize,
9134                ctx.sizeof_size as usize,
9135                BTreeV1Config::default().sym_leaf_max_entries(),
9136            )
9137            .unwrap();
9138            eprintln!("SNOD at {}: {} entries", child, snod.entries.len());
9139            for entry in &snod.entries {
9140                let name = local_heap_get_string(&heap_data, entry.name_offset).unwrap();
9141                eprintln!(
9142                    "  entry: name='{}' (offset={}), obj_header={}, cache={:?}",
9143                    name, entry.name_offset, entry.obj_header_addr, entry.cache
9144                );
9145            }
9146        }
9147
9148        // Try full open
9149        let reader = Hdf5Reader::open(path).unwrap();
9150        eprintln!("Datasets found: {:?}", reader.dataset_names());
9151    }
9152
9153    // ====================================================================
9154    // Group/link discovery: continuation blocks, dense links, v0/v1 groups.
9155    //
9156    // These tests generate HDF5 fixtures with h5py (HDF5 2.0.0). If the
9157    // pinned Python interpreter is not present, the test skips so the suite
9158    // still runs in environments without it.
9159    // ====================================================================
9160
9161    const TEST_PYTHON: &str = "/Users/stevek/mamba/envs/bs2026.1/bin/python";
9162
9163    /// Per-call unique temp path (PID + atomic counter) to avoid collisions
9164    /// across concurrent test runs.
9165    fn temp_path(name: &str) -> std::path::PathBuf {
9166        use std::sync::atomic::{AtomicU64, Ordering};
9167        static COUNTER: AtomicU64 = AtomicU64::new(0);
9168        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
9169        std::env::temp_dir().join(format!(
9170            "rust_hdf5_gap_test_{}_{}_{}.h5",
9171            name,
9172            std::process::id(),
9173            n
9174        ))
9175    }
9176
9177    /// Run a Python snippet to generate a fixture; returns false if Python
9178    /// is unavailable so the caller can skip the test.
9179    fn gen_fixture(script: &str) -> bool {
9180        if !std::path::Path::new(TEST_PYTHON).exists() {
9181            return false;
9182        }
9183        let status = std::process::Command::new(TEST_PYTHON)
9184            .arg("-c")
9185            .arg(script)
9186            .status();
9187        matches!(status, Ok(s) if s.success())
9188    }
9189
9190    #[test]
9191    fn gap1_v2_root_continuation_block() {
9192        let path = temp_path("gap1_cont");
9193        let p = path.display().to_string();
9194        // ~6 datasets in a v2 root group forces an object-header
9195        // continuation block.
9196        let script = format!(
9197            "import h5py,numpy as np\n\
9198             f=h5py.File(r'{p}','w',libver='latest')\n\
9199             [f.create_dataset('ds_%d'%i,data=np.arange(i*10,i*10+10,dtype='int32')) for i in range(6)]\n\
9200             f.close()"
9201        );
9202        if !gen_fixture(&script) {
9203            eprintln!("skipping gap1: python unavailable");
9204            return;
9205        }
9206
9207        let mut reader = Hdf5Reader::open(&path).unwrap();
9208        let mut names = reader.dataset_names();
9209        names.sort();
9210        assert_eq!(
9211            names,
9212            vec!["ds_0", "ds_1", "ds_2", "ds_3", "ds_4", "ds_5"],
9213            "all 6 datasets must be found across the continuation block"
9214        );
9215        // Element-exact read of one dataset.
9216        let raw = reader.read_dataset_raw("ds_3").unwrap();
9217        let vals: Vec<i32> = raw
9218            .as_chunks::<4>()
9219            .0
9220            .iter()
9221            .map(|c| i32::from_le_bytes(*c))
9222            .collect();
9223        assert_eq!(vals, (30..40).collect::<Vec<i32>>());
9224        let _ = std::fs::remove_file(&path);
9225    }
9226
9227    #[test]
9228    fn gap2_v2_dense_fractal_heap_links() {
9229        let path = temp_path("gap2_dense");
9230        let p = path.display().to_string();
9231        // 14 datasets in one v2 group forces dense (fractal-heap) link
9232        // storage.
9233        let script = format!(
9234            "import h5py,numpy as np\n\
9235             f=h5py.File(r'{p}','w',libver='latest')\n\
9236             g=f.create_group('dense')\n\
9237             [g.create_dataset('d%02d'%i,data=np.full(4,i,dtype='float64')) for i in range(14)]\n\
9238             f.close()"
9239        );
9240        if !gen_fixture(&script) {
9241            eprintln!("skipping gap2: python unavailable");
9242            return;
9243        }
9244
9245        let mut reader = Hdf5Reader::open(&path).unwrap();
9246        let mut names = reader.dataset_names();
9247        names.sort();
9248        let expected: Vec<String> = (0..14).map(|i| format!("dense/d{:02}", i)).collect();
9249        assert_eq!(
9250            names, expected,
9251            "all 14 dense-stored links must be recovered from the fractal heap"
9252        );
9253        // Element-exact read of one dense-stored dataset.
9254        let raw = reader.read_dataset_raw("dense/d07").unwrap();
9255        let vals: Vec<f64> = raw
9256            .as_chunks::<8>()
9257            .0
9258            .iter()
9259            .map(|c| f64::from_le_bytes(*c))
9260            .collect();
9261        assert_eq!(vals, vec![7.0; 4]);
9262        let _ = std::fs::remove_file(&path);
9263    }
9264
9265    #[test]
9266    fn gap3_v0v1_legacy_subgroups() {
9267        let path = temp_path("gap3_legacy");
9268        let p = path.display().to_string();
9269        // libver='earliest' => v0 superblock, symbol-table groups; datasets
9270        // nested inside subgroups.
9271        let script = format!(
9272            "import h5py,numpy as np\n\
9273             f=h5py.File(r'{p}','w',libver='earliest')\n\
9274             g1=f.create_group('grp1')\n\
9275             g1.create_dataset('a',data=np.arange(5,dtype='int16'))\n\
9276             g2=g1.create_group('sub')\n\
9277             g2.create_dataset('b',data=np.arange(7,dtype='int64'))\n\
9278             f.create_dataset('top',data=np.arange(3,dtype='int32'))\n\
9279             f.close()"
9280        );
9281        if !gen_fixture(&script) {
9282            eprintln!("skipping gap3: python unavailable");
9283            return;
9284        }
9285
9286        let mut reader = Hdf5Reader::open(&path).unwrap();
9287        let mut names = reader.dataset_names();
9288        names.sort();
9289        assert_eq!(
9290            names,
9291            vec!["grp1/a", "grp1/sub/b", "top"],
9292            "datasets nested in legacy symbol-table subgroups must be found"
9293        );
9294        // Element-exact read of a doubly-nested dataset.
9295        let raw = reader.read_dataset_raw("grp1/sub/b").unwrap();
9296        let vals: Vec<i64> = raw
9297            .as_chunks::<8>()
9298            .0
9299            .iter()
9300            .map(|c| i64::from_le_bytes(*c))
9301            .collect();
9302        assert_eq!(vals, (0..7).collect::<Vec<i64>>());
9303        let _ = std::fs::remove_file(&path);
9304    }
9305
9306    /// N-bit chunked datasets with non-zero bit offset and signed types with
9307    /// negative values must read back element-exact through the crate's
9308    /// chunked readers. The post-filter datatype conversion shifts/masks/
9309    /// sign-extends each element after the filter pipeline.
9310    #[test]
9311    fn nbit_chunked_post_filter_conversion() {
9312        let path = temp_path("nbit_conv");
9313        let p = path.display().to_string();
9314        // Build N-bit datasets with reduced precision + non-zero offset via
9315        // h5py's low-level filter API (h5py has no high-level N-bit knob).
9316        let script = format!(
9317            "import h5py,numpy as np\n\
9318             from h5py import h5t,h5p,h5s,h5d,h5f,h5z\n\
9319             fid=h5f.create(r'{p}'.encode())\n\
9320             def mk(name,bt,prec,off,npd,vals,chunk):\n\
9321            \x20dt=bt.copy();dt.set_precision(prec);dt.set_offset(off)\n\
9322            \x20arr=np.ascontiguousarray(np.asarray(vals,dtype=npd))\n\
9323            \x20sp=h5s.create_simple(arr.shape)\n\
9324            \x20dc=h5p.create(h5p.DATASET_CREATE);dc.set_chunk(chunk)\n\
9325            \x20dc.set_filter(h5z.FILTER_NBIT,h5z.FLAG_OPTIONAL,())\n\
9326            \x20ds=h5d.create(fid,name.encode(),dt,sp,dc)\n\
9327            \x20ds.write(h5s.ALL,h5s.ALL,arr);ds.close()\n\
9328             mk('u4_p17_o3',h5t.STD_U32LE,17,3,'u4',[0,1,1000,65535,131071,70000,42,99999],(4,))\n\
9329             mk('i4_p13_o5',h5t.STD_I32LE,13,5,'i4',[-5,-1,0,1,7,-4096,4095,-77,42,100,-100,3],(4,))\n\
9330             mk('i2_p9_o4',h5t.STD_I16LE,9,4,'i2',[-256,-1,0,1,255,-7,7,-200],(3,))\n\
9331             mk('i4_2d_p11_o6',h5t.STD_I32LE,11,6,'i4',np.array([[-1024,-1,0,5],[1023,-77,88,-3]],dtype='i4'),(1,4))\n\
9332             fid.close()"
9333        );
9334        if !gen_fixture(&script) {
9335            eprintln!("skipping nbit_chunked_post_filter_conversion: python unavailable");
9336            return;
9337        }
9338
9339        let mut reader = Hdf5Reader::open(&path).unwrap();
9340
9341        // Unsigned u4, precision 17, bit offset 3.
9342        let raw = reader.read_dataset_raw("u4_p17_o3").unwrap();
9343        let got: Vec<u32> = raw
9344            .as_chunks::<4>()
9345            .0
9346            .iter()
9347            .map(|c| u32::from_le_bytes(*c))
9348            .collect();
9349        assert_eq!(
9350            got,
9351            vec![0u32, 1, 1000, 65535, 131071, 70000, 42, 99999],
9352            "u4 N-bit dataset must decode to exact unsigned values"
9353        );
9354
9355        // Signed i4 with negatives, precision 13, bit offset 5.
9356        let raw = reader.read_dataset_raw("i4_p13_o5").unwrap();
9357        let got: Vec<i32> = raw
9358            .as_chunks::<4>()
9359            .0
9360            .iter()
9361            .map(|c| i32::from_le_bytes(*c))
9362            .collect();
9363        assert_eq!(
9364            got,
9365            vec![-5i32, -1, 0, 1, 7, -4096, 4095, -77, 42, 100, -100, 3],
9366            "i4 N-bit dataset must sign-extend negative values"
9367        );
9368
9369        // Signed i2 with negatives, precision 9, bit offset 4.
9370        let raw = reader.read_dataset_raw("i2_p9_o4").unwrap();
9371        let got: Vec<i16> = raw
9372            .as_chunks::<2>()
9373            .0
9374            .iter()
9375            .map(|c| i16::from_le_bytes(*c))
9376            .collect();
9377        assert_eq!(
9378            got,
9379            vec![-256i16, -1, 0, 1, 255, -7, 7, -200],
9380            "i2 N-bit dataset must sign-extend negative values"
9381        );
9382
9383        // 2D signed i4, precision 11, bit offset 6 (1-row chunks).
9384        let raw = reader.read_dataset_raw("i4_2d_p11_o6").unwrap();
9385        let got: Vec<i32> = raw
9386            .as_chunks::<4>()
9387            .0
9388            .iter()
9389            .map(|c| i32::from_le_bytes(*c))
9390            .collect();
9391        assert_eq!(
9392            got,
9393            vec![-1024i32, -1, 0, 5, 1023, -77, 88, -3],
9394            "2D i4 N-bit dataset must decode element-exact"
9395        );
9396
9397        // read_slice path must also apply the conversion exactly once.
9398        let raw = reader.read_slice("i4_p13_o5", &[4], &[3]).unwrap();
9399        let got: Vec<i32> = raw
9400            .as_chunks::<4>()
9401            .0
9402            .iter()
9403            .map(|c| i32::from_le_bytes(*c))
9404            .collect();
9405        assert_eq!(got, vec![7i32, -4096, 4095], "read_slice must convert too");
9406
9407        // 2D slice: second row, all columns.
9408        let raw = reader.read_slice("i4_2d_p11_o6", &[1, 0], &[1, 4]).unwrap();
9409        let got: Vec<i32> = raw
9410            .as_chunks::<4>()
9411            .0
9412            .iter()
9413            .map(|c| i32::from_le_bytes(*c))
9414            .collect();
9415        assert_eq!(
9416            got,
9417            vec![1023i32, -77, 88, -3],
9418            "2D read_slice must convert"
9419        );
9420
9421        let _ = std::fs::remove_file(&path);
9422    }
9423
9424    /// Partial-slice reads of chunked datasets must skip non-overlapping
9425    /// chunks yet return exactly the selected region, for every chunk index
9426    /// type: v1 B-tree (libver=earliest), single chunk, fixed array,
9427    /// extensible array (one unlimited dim), and v2 B-tree (>1 unlimited dim).
9428    ///
9429    /// The dataset is a 5×4×6 int32 `arange`, chunked 2×2×2 so the chunk grid
9430    /// is ragged (edge chunks) and most selections touch a strict subset of
9431    /// chunks. Each slice is checked against the row-major `arange` value so a
9432    /// dropped/misplaced chunk or a mis-sized output buffer is caught.
9433    #[test]
9434    fn read_slice_chunked_all_index_types() {
9435        let latest = temp_path("slice_chunk_latest");
9436        let earliest = temp_path("slice_chunk_earliest");
9437        let pl = latest.display().to_string();
9438        let pe = earliest.display().to_string();
9439        // libver=latest selects modern indices by maxshape: fixed -> Fixed
9440        // Array, one unlimited dim -> Extensible Array, >1 unlimited -> v2
9441        // B-tree, single chunk -> Single Chunk index. libver=earliest always
9442        // uses the v1 B-tree chunk index.
9443        let script = format!(
9444            "import h5py,numpy as np\n\
9445             a=np.arange(5*4*6,dtype='int32').reshape(5,4,6)\n\
9446             f=h5py.File(r'{pl}','w',libver='latest')\n\
9447             f.create_dataset('single',data=a,chunks=(5,4,6))\n\
9448             f.create_dataset('fa',data=a,chunks=(2,2,2))\n\
9449             f.create_dataset('ea',data=a,chunks=(2,2,2),maxshape=(None,4,6))\n\
9450             f.create_dataset('btv2',data=a,chunks=(2,2,2),maxshape=(None,None,6))\n\
9451             f.close()\n\
9452             g=h5py.File(r'{pe}','w',libver='earliest')\n\
9453             g.create_dataset('btv1',data=a,chunks=(2,2,2))\n\
9454             g.close()"
9455        );
9456        if !gen_fixture(&script) {
9457            eprintln!("skipping read_slice_chunked_all_index_types: python unavailable");
9458            return;
9459        }
9460
9461        let dims = [5u64, 4, 6];
9462        // Row-major value of element (i,j,k) in the arange dataset.
9463        let val = |i: u64, j: u64, k: u64| (i * dims[1] * dims[2] + j * dims[2] + k) as i32;
9464        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
9465            let mut out = Vec::new();
9466            for i in 0..counts[0] {
9467                for j in 0..counts[1] {
9468                    for k in 0..counts[2] {
9469                        out.push(val(starts[0] + i, starts[1] + j, starts[2] + k));
9470                    }
9471                }
9472            }
9473            out
9474        };
9475        let decode = |raw: Vec<u8>| -> Vec<i32> {
9476            raw.as_chunks::<4>()
9477                .0
9478                .iter()
9479                .map(|c| i32::from_le_bytes(*c))
9480                .collect()
9481        };
9482        let cases: &[([u64; 3], [u64; 3])] = &[
9483            ([0, 1, 0], [5, 2, 6]), // full last dim, partial mid -> coalesced runs
9484            ([1, 0, 0], [3, 4, 6]), // partial dim0, inner dims full -> one run
9485            ([0, 0, 2], [5, 4, 3]), // partial last dim -> one run per (i,j) row
9486            ([2, 1, 3], [1, 2, 2]), // interior block spanning few chunks
9487            ([0, 0, 0], [5, 4, 6]), // whole dataset via read_slice
9488            ([4, 3, 5], [1, 1, 1]), // single element at the far edge chunk
9489            ([1, 1, 1], [3, 3, 4]), // straddles chunk boundaries on all axes
9490        ];
9491
9492        let mut reader_l = Hdf5Reader::open(&latest).unwrap();
9493        for name in ["single", "fa", "ea", "btv2"] {
9494            // Full read sanity first, then every slice.
9495            let full = decode(reader_l.read_dataset_raw(name).unwrap());
9496            assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "{name} full read");
9497            for &(starts, counts) in cases {
9498                let got = decode(reader_l.read_slice(name, &starts, &counts).unwrap());
9499                assert_eq!(
9500                    got,
9501                    expect(starts, counts),
9502                    "{name} slice starts={starts:?} counts={counts:?}"
9503                );
9504            }
9505        }
9506
9507        let mut reader_e = Hdf5Reader::open(&earliest).unwrap();
9508        let full = decode(reader_e.read_dataset_raw("btv1").unwrap());
9509        assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "btv1 full read");
9510        for &(starts, counts) in cases {
9511            let got = decode(reader_e.read_slice("btv1", &starts, &counts).unwrap());
9512            assert_eq!(
9513                got,
9514                expect(starts, counts),
9515                "btv1 slice starts={starts:?} counts={counts:?}"
9516            );
9517        }
9518
9519        let _ = std::fs::remove_file(&latest);
9520        let _ = std::fs::remove_file(&earliest);
9521    }
9522
9523    /// A slice read must place the same bytes whichever way a chunk reaches
9524    /// the output: read run by run straight out of the file — what unfiltered
9525    /// data allows, its stored bytes being the dataset's bytes — or copied out
9526    /// of a decoded whole-chunk image, which filtered data always needs and
9527    /// which runs too small to be worth a positioned read each fall back to.
9528    /// Both sinks are driven off one array here, so naive extraction is the
9529    /// shared oracle for the two of them and for every run shape in between.
9530    #[test]
9531    #[cfg(feature = "deflate")]
9532    fn slice_reads_agree_however_the_chunk_reaches_the_output() {
9533        let path = temp_path("slice_run_placement");
9534        let (rows, cols) = (8usize, 4096usize); // a chunk row is 32 KiB
9535        let data: Vec<f64> = (0..rows * cols).map(|i| i as f64).collect();
9536        let (rows, cols) = (rows as u64, cols as u64);
9537        {
9538            let file = crate::H5File::create(&path).unwrap();
9539            let ds = file
9540                .new_dataset::<f64>()
9541                .shape([rows as usize, cols as usize])
9542                .chunk(&[2, cols as usize])
9543                .create("plain")
9544                .unwrap();
9545            ds.write_raw(&data).unwrap();
9546            let ds = file
9547                .new_dataset::<f64>()
9548                .shape([rows as usize, cols as usize])
9549                .chunk(&[2, cols as usize])
9550                .deflate(1)
9551                .create("zipped")
9552                .unwrap();
9553            ds.write_raw(&data).unwrap();
9554            file.close().unwrap();
9555        }
9556
9557        let expect = |starts: [u64; 2], counts: [u64; 2]| -> Vec<f64> {
9558            let mut out = Vec::new();
9559            for i in 0..counts[0] {
9560                for j in 0..counts[1] {
9561                    out.push(data[((starts[0] + i) * cols + starts[1] + j) as usize]);
9562                }
9563            }
9564            out
9565        };
9566        let decode = |raw: Vec<u8>| -> Vec<f64> {
9567            raw.as_chunks::<8>()
9568                .0
9569                .iter()
9570                .map(|c| f64::from_le_bytes(*c))
9571                .collect()
9572        };
9573        let cases: &[([u64; 2], [u64; 2])] = &[
9574            ([0, 0], [rows, cols]), // whole dataset: one run per chunk
9575            ([3, 0], [4, cols]),    // straddles chunk rows, full width
9576            ([1, 7], [5, 3]),       // 24-byte runs: not worth a read each
9577            ([2, 1000], [2, 2048]), // 16 KiB runs, off the chunk row origin
9578            ([7, 4095], [1, 1]),    // one element in the last chunk
9579        ];
9580
9581        let mut reader = Hdf5Reader::open(&path).unwrap();
9582        for name in ["plain", "zipped"] {
9583            let full = decode(reader.read_dataset_raw(name).unwrap());
9584            assert_eq!(full, data, "{name} full read");
9585            for &(starts, counts) in cases {
9586                let got = decode(reader.read_slice(name, &starts, &counts).unwrap());
9587                assert_eq!(
9588                    got,
9589                    expect(starts, counts),
9590                    "{name} slice starts={starts:?} counts={counts:?}"
9591                );
9592            }
9593        }
9594
9595        std::fs::remove_file(&path).ok();
9596    }
9597}