Skip to main content

rust_hdf5/io/
reader.rs

1//! HDF5 file reader.
2//!
3//! Opens an HDF5 file, parses the superblock and root group, and provides
4//! access to dataset metadata and raw data.
5//!
6//! Supports both legacy (v0/v1 superblock, v1 object headers, symbol tables)
7//! and modern (v2/v3 superblock, v2 object headers, link messages) formats.
8
9use std::path::{Path, PathBuf};
10
11use crate::dataset::{DatasetAccess, VirtualView};
12use crate::format::btree_v1::{BTreeV1Config, BTreeV1Node, ChunkBTreeV1Node};
13use crate::format::bytes::read_le_uint as read_uint;
14use crate::format::creation_order::CreationOrder;
15use crate::format::fractal_heap::{self, FractalHeapHeader};
16use crate::format::global_heap::{
17    decode_vlen_reference, vlen_reference_size, GlobalHeapCollection,
18};
19use crate::format::local_heap::{local_heap_get_string, LocalHeapHeader};
20use crate::format::messages::attr_info::AttributeInfoMessage;
21use crate::format::messages::attribute::{AttributeEntry, AttributeMessage};
22use crate::format::messages::data_layout::{self, DataLayoutMessage};
23use crate::format::messages::dataspace::DataspaceMessage;
24use crate::format::messages::datatype::{DatatypeMessage, OldReferenceKind, ReferenceEncoding};
25use crate::format::messages::external_file_list::ExternalFileListMessage;
26use crate::format::messages::fill_value::{
27    try_tiled_fill, FillValueMessage, ALLOC_TIME_LATE, FILL_TIME_IFSET,
28};
29use crate::format::messages::filter::{self, FilterPipeline};
30use crate::format::messages::link::LinkMessage;
31use crate::format::messages::link::LinkTarget;
32use crate::format::messages::link_info::LinkInfoMessage;
33use crate::format::messages::shared::{MessageStorage, MSG_FLAG_SHARED};
34use crate::format::messages::superblock_ext::{
35    BtreeKMessage, DriverInfoMessage, FileSpaceInfoMessage, SharedMessageTableMessage,
36};
37use crate::format::messages::virtual_mapping::{
38    parse_source_name, VirtualMapping, VirtualMappingList,
39};
40use crate::format::messages::*;
41use crate::format::object_header::ObjectHeader;
42use crate::format::reference::{
43    decode_object_element, decode_region_element, decode_region_heap_object, decode_revised_body,
44    decode_revised_element, DecodedReference, Reference, ReferenceTarget, RevisedElement,
45};
46use crate::format::selection::{
47    Hyperslab, PointSelection, RegularHyperslab, ResolvedSelection, Selection,
48};
49use crate::format::sohm::SohmMasterTable;
50use crate::format::storage_kind::{AttributeStorage, LinkStorage};
51use crate::format::superblock::{
52    detect_superblock_version, SuperblockV0V1, SuperblockV2V3, SymbolTableCache,
53};
54use crate::format::symbol_table::SymbolTableNode;
55use crate::format::{BlockReader, FormatContext, UNDEF_ADDR};
56
57use crate::io::file_handle::FileHandle;
58use crate::io::hyperslab::{compute_strides, for_each_contiguous_run};
59use crate::io::locking::FileLocking;
60use crate::io::{FileMeta, IoResult};
61
62/// The version-4 chunk-index descriptor pulled from a data-layout message:
63/// the index kind, its address, and the per-kind parameters the reader needs
64/// to walk it. Bundled so the chunked read entry point takes one descriptor
65/// instead of a long parameter list.
66struct ChunkIndexDesc<'a> {
67    /// The kind of chunk index (single chunk, fixed/extensible array, …).
68    index_type: data_layout::ChunkIndexType,
69    /// Address of the chunk index structure (or the chunk itself, for a
70    /// single chunk). `UNDEF_ADDR` when unallocated.
71    index_address: u64,
72    /// Extensible-array parameters (present iff `index_type == ExtensibleArray`).
73    earray_params: Option<&'a data_layout::EarrayParams>,
74    /// Filtered single-chunk parameters (present iff `index_type ==
75    /// SingleChunk` and the layout's filtered flag is set): the chunk's exact
76    /// on-disk size and per-chunk filter mask.
77    single_chunk_filter: Option<data_layout::SingleChunkFilter>,
78}
79
80/// One v2-B-tree chunk-index record, resolved to `(chunk address, on-disk
81/// read size, scaled chunk-grid offsets, filter mask)` — see
82/// [`Hdf5Reader::collect_bt2_chunk_entries`].
83type Bt2ChunkEntry = (u64, usize, Vec<u64>, u32);
84
85/// What a chunked read should produce: the whole dataset, or one hyperslab.
86///
87/// Threaded through every chunked reader so the index walk, raw read, and
88/// filter pipeline are shared between full reads and slice reads. For a
89/// `Slice`, the reader allocates a `counts`-shaped buffer, skips reading any
90/// chunk that does not overlap the selection (the I/O win), and scatters only
91/// the chunk∩selection intersection. `Full` reads and places every chunk.
92#[derive(Clone, Copy)]
93enum ChunkTarget<'a> {
94    Full,
95    Slice {
96        starts: &'a [u64],
97        counts: &'a [u64],
98    },
99}
100
101impl<'a> ChunkTarget<'a> {
102    /// Whether a chunk at chunk-grid `coords` (extent `chunk_dims`) intersects
103    /// the target. `Full` always intersects; a `Slice` intersects iff every
104    /// dimension's chunk span `[origin, origin+chunk_dims)` overlaps the
105    /// selection span `[start, start+count)`.
106    fn overlaps(&self, coords: &[u64], chunk_dims: &[u64]) -> bool {
107        match self {
108            ChunkTarget::Full => true,
109            ChunkTarget::Slice { starts, counts } => coords.iter().enumerate().all(|(d, &c)| {
110                let origin = c.saturating_mul(chunk_dims[d]);
111                let chunk_end = origin.saturating_add(chunk_dims[d]);
112                let sel_end = starts[d].saturating_add(counts[d]);
113                origin < sel_end && starts[d] < chunk_end
114            }),
115        }
116    }
117}
118
119/// What one chunked read wants of every chunk, apart from which dataset it is
120/// reading: the filter pipeline the chunks were stored through, the box of the
121/// dataset it fills, and the fill value every byte no chunk covers takes.
122///
123/// Carried as one value because all three are constant across a read and are
124/// threaded unchanged from the entry point through each index type down to
125/// [`place_chunk_jobs`] — so a new index reader cannot pick up the target and
126/// forget the fill.
127#[derive(Clone, Copy)]
128struct ChunkReadRequest<'a> {
129    pipeline: Option<&'a FilterPipeline>,
130    target: ChunkTarget<'a>,
131    fill_value: Option<&'a [u8]>,
132}
133
134/// The dataset extent, chunk shape and element size a chunk-placement call
135/// reads through — constant across every chunk of one read, so
136/// [`for_each_chunk_run`] and the two sinks built on it take this once instead
137/// of the three fields separately, leaving only what actually varies per chunk
138/// (its data and grid coordinates) as their own parameters.
139#[derive(Clone, Copy)]
140struct ChunkOutputGeometry<'a> {
141    dims: &'a [u64],
142    chunk_dims: &'a [u64],
143    element_size: u64,
144}
145
146/// One chunked read's placement constants: the geometry above plus the box in
147/// dataset coordinates the read fills.
148///
149/// A full read fills the whole extent, which is the box `starts = 0`,
150/// `counts = dims` describes, so resolving the target once
151/// ([`ChunkPlacement::resolve`]) leaves the placement code below with one
152/// shape instead of a full-read case and a slice case.
153#[derive(Clone, Copy)]
154struct ChunkPlacement<'a> {
155    geo: ChunkOutputGeometry<'a>,
156    starts: &'a [u64],
157    counts: &'a [u64],
158}
159
160impl ChunkOutputGeometry<'_> {
161    /// Bytes one whole chunk's image holds — the product of the chunk shape,
162    /// element-wide. `None` only for geometry a corrupt file can carry: a
163    /// shape whose product overflows, or an empty image.
164    ///
165    /// The size the layout says a decoded chunk is. Both the destination a
166    /// chunk decodes into and the buffer a staged chunk decodes into come from
167    /// here, so neither has to discover it by growing.
168    fn image_bytes(&self) -> Option<u64> {
169        self.chunk_dims
170            .iter()
171            .copied()
172            .try_fold(1u64, |a, d| a.checked_mul(d))
173            .and_then(|elems| elems.checked_mul(self.element_size))
174            .filter(|b| *b > 0)
175    }
176
177    /// Bytes the chunk at `coords` holds that the dataset extent reaches:
178    /// [`image_bytes`](Self::image_bytes), except at an edge where the extent
179    /// cuts the chunk short. A read covering all of them has taken everything
180    /// that chunk has to give, which is the same clip
181    /// [`ChunkOverlap::of`] applies.
182    fn resident_bytes(&self, coords: &[u64]) -> u64 {
183        let mut elems = 1u64;
184        for (d, &c) in coords.iter().enumerate().take(self.dims.len()) {
185            let origin = c.saturating_mul(self.chunk_dims[d]);
186            let end = origin.saturating_add(self.chunk_dims[d]).min(self.dims[d]);
187            elems = elems.saturating_mul(end.saturating_sub(origin));
188        }
189        elems.saturating_mul(self.element_size)
190    }
191}
192
193impl<'a> ChunkPlacement<'a> {
194    /// Resolve a read target against the geometry it reads through. `zeros`
195    /// lends a full read the origin it fills from and does not carry itself.
196    fn resolve(geo: &ChunkOutputGeometry<'a>, target: ChunkTarget<'a>, zeros: &'a [u64]) -> Self {
197        let (starts, counts) = match target {
198            ChunkTarget::Full => (zeros, geo.dims),
199            ChunkTarget::Slice { starts, counts } => (starts, counts),
200        };
201        ChunkPlacement {
202            geo: *geo,
203            starts,
204            counts,
205        }
206    }
207
208    /// Whether this read leaves part of the chunk at `coords` untouched, so a
209    /// later read can still want the rest — which is what makes that chunk's
210    /// decoded image worth keeping.
211    ///
212    /// What the chunk has to give is the chunk clipped to the dataset extent,
213    /// the same clip [`ChunkOverlap::of`] applies and not the chunk image: an
214    /// edge chunk the extent cuts short is fully consumed by a read that takes
215    /// what is inside the extent. So a whole-dataset read leaves nothing,
216    /// whatever the extent does at the edges.
217    fn leaves_chunk_unconsumed(&self, coords: &[u64]) -> bool {
218        match ChunkOverlap::of(self, coords) {
219            Some(overlap) => overlap.bytes(self.geo.element_size) < self.geo.resident_bytes(coords),
220            None => false,
221        }
222    }
223}
224
225/// One resolved external-file slot (H5O_EFL_ID): the on-disk message
226/// stores each slot's name as an offset into a local heap, so this is that
227/// slot after the heap lookup, in the order the dataset's logical byte
228/// range concatenates them.
229#[derive(Debug, Clone, PartialEq, Eq)]
230pub struct ExternalFileSegment {
231    /// The external file's name, exactly as stored — relative names are
232    /// resolved against `HDF5_EXTFILE_PREFIX` at read time, not here.
233    pub name: String,
234    /// Byte offset within the named file where this slot's reserved
235    /// region begins.
236    pub offset: u64,
237    /// Bytes reserved for this slot. `u64::MAX` (`H5O_EFL_UNLIMITED`)
238    /// marks the last slot as unlimited/growable.
239    pub size: u64,
240}
241
242/// Where a dataset's stored image lies, as far as a zero-copy view of it is
243/// concerned.
244///
245/// [`Hdf5Reader::dataset_view_source`] is the only producer; it reports what
246/// the layout message says and judges nothing.
247#[cfg(feature = "mmap")]
248pub(crate) enum ViewStorage {
249    /// One stretch of this file: `len` bytes at absolute file offset
250    /// `offset`, with the userblock already added, so it indexes the map
251    /// directly.
252    Contiguous { offset: u64, len: u64 },
253    /// No storage is allocated. Every element reads as the fill value, which
254    /// is a property of the header, not bytes anywhere in the file.
255    Unallocated,
256    /// The image is not one stretch of this file. The phrase says what it is
257    /// instead, and lands verbatim in the refusal.
258    Elsewhere(&'static str),
259}
260
261/// Everything a zero-copy view of one dataset rests on, gathered from the
262/// file that owns it.
263#[cfg(feature = "mmap")]
264pub(crate) struct DatasetViewSource {
265    /// The owning file's whole-file map, or `None` when that file is read
266    /// through `pread`.
267    pub map: Option<std::sync::Arc<memmap2::Mmap>>,
268    /// Where the dataset's image lies in that file.
269    pub storage: ViewStorage,
270    /// How one element is stored, which decides whether the stored bytes are
271    /// already the host image of a `T`.
272    pub datatype: DatatypeMessage,
273    /// The dataset's extent, for turning a requested range into a byte run.
274    pub dims: Vec<u64>,
275}
276
277/// Read-side metadata for a single dataset.
278pub struct DatasetReadInfo {
279    /// Dataset name (the link name in the root group).
280    pub name: String,
281    /// Address of the dataset's object header — what an object reference to
282    /// this dataset stores.
283    pub object_header_address: u64,
284    /// Element datatype.
285    pub datatype: DatatypeMessage,
286    /// Dataspace (dimensionality).
287    pub dataspace: DataspaceMessage,
288    /// Data layout (contiguous, compact, or chunked).
289    pub layout: DataLayoutMessage,
290    /// Filter pipeline for compressed chunks (None = uncompressed).
291    pub filter_pipeline: Option<FilterPipeline>,
292    /// Attributes attached to this dataset.
293    pub attributes: ObjectAttributes,
294    /// User-defined fill value bytes (one element wide), decoded from the
295    /// fill-value message when `fill_defined == 2`. `None` => default
296    /// zero-fill. Applied to unallocated chunks and unwritten regions.
297    pub fill_value: Option<Vec<u8>>,
298    /// The fill-value message's own definedness byte
299    /// (`H5D_fill_value_t`/`H5Pfill_value_defined`): 0 = explicitly
300    /// undefined (no fill is ever performed), 1 = default (zero-fill, no
301    /// value stored), 2 = user-defined (`fill_value` carries the bytes). A
302    /// dataset with no fill-value message at all reads as 1, matching a
303    /// fresh dataset creation property list (`FillValueMessage::default`).
304    pub fill_defined: u8,
305    /// The fill-value message's write-time byte (`H5D_fill_time_t`): 0 =
306    /// `H5D_FILL_TIME_ALLOC`, 1 = `H5D_FILL_TIME_NEVER`, 2 =
307    /// `H5D_FILL_TIME_IFSET`. A dataset with no fill-value message at all
308    /// reads as 2, `H5D_CRT_FILL_TIME_DEF` — the same "no message" default
309    /// [`fill_defined`](Self::fill_defined) uses.
310    pub fill_write_time: u8,
311    /// The fill-value message's space-allocation-time byte
312    /// (`H5D_alloc_time_t`): 1 = `H5D_ALLOC_TIME_EARLY`, 2 =
313    /// `H5D_ALLOC_TIME_LATE`, 3 = `H5D_ALLOC_TIME_INCR`. A dataset with no
314    /// fill-value message at all reads as `ALLOC_TIME_LATE`, matching
315    /// [`FillValueMessage::default`]'s "no message" convention.
316    pub alloc_time: u8,
317    /// External raw-data segments (H5O_EFL_ID). Non-empty only when this
318    /// dataset's storage is an External Data Files list instead of a
319    /// normal contiguous block — `layout` still reports `Contiguous` with
320    /// an undefined address in that case (H5Dlayout.c overrides the
321    /// layout's storage ops whenever this message is present).
322    pub external_files: Vec<ExternalFileSegment>,
323    /// Virtual dataset source/virtual mappings (H5D_VIRTUAL), resolved from
324    /// the global heap object `layout`'s `Virtual` variant points at.
325    /// `Some` only when `layout` is `DataLayoutMessage::Virtual` and it
326    /// names a mapping list (`heap_index != 0`); `None` for every other
327    /// layout, and for a virtual dataset that has no mappings yet.
328    pub virtual_mappings: Option<VirtualMappingList>,
329    /// What each mapping in [`virtual_mappings`](Self::virtual_mappings)
330    /// resolved to when the file was opened, in the same order — see
331    /// [`MappingResolution`] and `Hdf5Reader::resolve_virtual_extents`.
332    /// `None` for every non-virtual dataset and for a virtual one with no
333    /// mapping list.
334    pub virtual_resolution: Option<Vec<MappingResolution>>,
335    /// The extent this dataset's dataspace message stores, kept because
336    /// `dataspace.dims` holds the extent the *sources* gave it once
337    /// `resolve_virtual_extents` has run. `H5D__virtual_set_extent_unlim`
338    /// resolves from the space freshly loaded from the object header on
339    /// every `H5Dopen` (H5Dvirtual.c:1386), so a later open under different
340    /// [`DatasetAccess`] must resolve from this, not from its own last
341    /// answer.
342    ///
343    /// `Some` exactly for a virtual dataset whose extent has been resolved;
344    /// `None` for every dataset whose extent is simply its stored one.
345    pub virtual_stored_dims: Option<Vec<u64>>,
346}
347
348/// What one virtual-dataset mapping resolved to at open time —
349/// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c), which libhdf5 runs when
350/// the dataset is opened and which both the dataset's reported extent and
351/// every read of it depend on.
352#[derive(Debug, Clone, PartialEq, Eq)]
353pub enum MappingResolution {
354    /// Neither selection grows: the mapping is already concrete, and the
355    /// dataset's stored extent is its extent.
356    Bounded,
357    /// Both selections are unlimited (`unlim_dim_virtual >= 0` and
358    /// `unlim_dim_source >= 0`). `virtual_clip` is how far this mapping
359    /// reaches in its unlimited virtual dimension, `source_clip` the source
360    /// dataset's own extent in its unlimited source dimension — the two
361    /// values `H5S_hyper_clip_unlim` clips the mapping's selections to.
362    Unlimited { virtual_clip: u64, source_clip: u64 },
363    /// A printf mapping (unlimited virtual selection, limited source
364    /// selection, `%b` in a source name).
365    ///
366    /// `blocks` is upstream's `first_missing`: the scan stops at the first
367    /// block whose source is absent and then looks
368    /// [`DatasetAccess::virtual_printf_gap`] blocks further, so blocks
369    /// `0..blocks` are the ones the extent covers. `present` lists which of
370    /// them actually have a source — with a non-zero gap the others are
371    /// inside the extent but read as the fill value.
372    Printf { blocks: u64, present: Vec<u64> },
373}
374
375/// The class of one link record in a group: what `H5Lget_info` reports,
376/// carrying the value `H5Lget_val` returns for the classes that have one.
377///
378/// Every link a group holds gets one of these, whether or not the object it
379/// names can be opened — a listing is a listing of links, not of objects.
380#[derive(Debug, Clone, PartialEq, Eq)]
381pub enum LinkClass {
382    /// Another name for an object in this file.
383    Hard,
384    /// A path inside this file, resolved when the link is traversed.
385    Soft { path: String },
386    /// A path inside another file. Listed and reported, but not followed.
387    External { file: String, path: String },
388    /// A user-defined link class this reader has no interpreter for; libhdf5
389    /// needs a registered link class for these too.
390    UserDefined { link_type: u8 },
391}
392
393impl LinkClass {
394    pub(crate) fn from_target(target: &LinkTarget) -> Self {
395        match target {
396            LinkTarget::Hard { .. } => Self::Hard,
397            LinkTarget::Soft { target } => Self::Soft {
398                path: target.clone(),
399            },
400            LinkTarget::External { file, path } => Self::External {
401                file: file.clone(),
402                path: path.clone(),
403            },
404            LinkTarget::UserDefined { link_type, .. } => Self::UserDefined {
405                link_type: *link_type,
406            },
407        }
408    }
409}
410
411/// The soft link a traversal crossed, kept so a lookup that finds nothing can
412/// say the link dangles instead of reporting a bare absence.
413struct SoftLinkRef {
414    link: String,
415    target: String,
416}
417
418/// Where a path leaves this file: the external link it crosses, the file that
419/// link names, and the remainder of the path inside that file.
420pub(crate) struct ExternalEdge {
421    pub link: String,
422    pub file: String,
423    pub path: String,
424}
425
426impl ExternalEdge {
427    /// The error for "the link resolved to a file, but the object it names is
428    /// not in it" — the external counterpart of a dangling soft link.
429    fn dangling(&self) -> crate::io::IoError {
430        crate::io::IoError::DanglingLink {
431            link: self.link.clone(),
432            target: format!("{}::{}", self.file, self.path),
433        }
434    }
435}
436
437/// How a reader was opened, carried from the entry point to the per-superblock
438/// constructors: both facts an external link needs later (where to resolve a
439/// relative target from, and under which locking policy to open it) are fixed
440/// at open time and belong together.
441struct Origin {
442    path: PathBuf,
443    locking: crate::io::locking::FileLocking,
444}
445
446/// Bound on how many external links one path resolution may cross, matching
447/// libhdf5's `H5L_NUM_LINKS` (the `H5Pset_nlinks` default). Two files that
448/// link to each other form a cycle whose every hop opens a fresh target, so it
449/// is this count, not target identity, that terminates the walk.
450const MAX_EXTERNAL_HOPS: usize = 16;
451
452/// What path traversal produced.
453enum Traversal {
454    /// A path in this file, after every group hard-link alias and soft link
455    /// on it was followed. `via` names the last soft link crossed, if any.
456    Path {
457        path: String,
458        via: Option<SoftLinkRef>,
459    },
460    /// A component of the path is an external link, so the path leaves this
461    /// file. `path` is the remainder inside `file`.
462    External {
463        link: String,
464        file: String,
465        path: String,
466    },
467}
468
469/// One rewrite a traversal step can apply to the path prefix it matched.
470#[derive(Clone, Copy)]
471enum Rewrite<'a> {
472    /// A group hard link: continue from the group's first-walked path.
473    Alias(&'a str),
474    /// A soft link: continue from its value.
475    Soft(&'a str),
476    /// An external link: stop, the rest of the path is in another file.
477    External { file: &'a str, path: &'a str },
478}
479
480/// Resolve a soft link's value against the group the link lives in, the way
481/// `H5G_traverse` does: a value starting with `/` is absolute, anything else
482/// is relative to that group. `.` and `..` components fold. The result has no
483/// leading `/`.
484fn resolve_link_value(link_path: &str, value: &str) -> String {
485    let mut components: Vec<&str> = Vec::new();
486    if !value.starts_with('/') {
487        // The link's own parent group is everything before its last component.
488        if let Some(parent) = link_path.rsplit_once('/').map(|(p, _)| p) {
489            components.extend(parent.split('/').filter(|c| !c.is_empty()));
490        }
491    }
492    for component in value.split('/') {
493        match component {
494            "" | "." => {}
495            ".." => {
496                components.pop();
497            }
498            c => components.push(c),
499        }
500    }
501    components.join("/")
502}
503
504/// Everything one discovery walk found: the objects, the link records that
505/// name them, and the group metadata the lookup paths need. Carried as one
506/// value so the walk has a single owner rather than a widening tuple, and so
507/// every walk (link-message and symbol-table alike) fills the same fields.
508#[derive(Default)]
509struct Catalog {
510    datasets: Vec<DatasetReadInfo>,
511    /// Dataset-shaped objects this crate cannot read, keyed by path; the
512    /// value names what stopped it. They are listed exactly like readable
513    /// datasets — the name is in the file either way — and refuse typed
514    /// access with that reason.
515    unreadable: std::collections::BTreeMap<String, String>,
516    /// Attributes on non-root groups, keyed by group path.
517    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
518    /// Link storage kind and link creation-order policy of non-root groups,
519    /// keyed by group path.
520    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
521    /// Every non-root group path the walk traversed into.
522    group_paths: std::collections::BTreeSet<String>,
523    /// Group object header address → the first path that reached it (no
524    /// leading `/`), taken from the walk's cycle guard. What turns the
525    /// address an object reference stores back into a name.
526    group_object_paths: std::collections::HashMap<u64, String>,
527    /// Group hard-link aliases: alias path → first-walked path.
528    group_aliases: std::collections::HashMap<String, String>,
529    /// Every link record seen, keyed by its full path (no leading `/`).
530    links: std::collections::BTreeMap<String, LinkClass>,
531    /// Committed (named) datatype objects, keyed by path.
532    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
533    /// Committed datatype object header address → its path (no leading `/`).
534    /// The third object kind needs the same address→name entry groups and
535    /// datasets have, or a path that names one resolves to nothing.
536    datatype_object_paths: std::collections::HashMap<u64, String>,
537}
538
539impl Catalog {
540    /// The address→absolute-path catalog this walk implies, with `root_addr`
541    /// named `/` whether or not the walk itself reached it.
542    ///
543    /// The single owner of the catalog: file open and the SWMR
544    /// [`Hdf5Reader::refresh`] rescan both build it here, so a dataset that
545    /// appears after open is resolvable exactly as one present at open is.
546    fn object_paths(&self, root_addr: u64) -> std::collections::HashMap<u64, String> {
547        let mut paths = std::collections::HashMap::new();
548        paths.insert(root_addr, "/".to_string());
549        for (addr, path) in &self.group_object_paths {
550            paths.insert(*addr, absolute_path(path));
551        }
552        for ds in &self.datasets {
553            paths.insert(ds.object_header_address, absolute_path(&ds.name));
554        }
555        for (addr, path) in &self.datatype_object_paths {
556            paths.insert(*addr, absolute_path(path));
557        }
558        paths
559    }
560}
561
562/// The state one catalog walk threads through every group it visits.
563///
564/// A group stores its children either as `Link` messages in its object header
565/// (with dense overflow in a fractal heap) or in the legacy symbol-table
566/// B-tree plus local heap — and *which* it uses is a property of that group
567/// alone. One file mixes the two freely: writing a single link that the old
568/// format cannot express (an external link, a creation-order-tracked group)
569/// migrates just that group, leaving its parent and its children where they
570/// were. Walking a group in the format its *parent* used therefore finds no
571/// children at all and reports an empty group, which is why the two storages
572/// share one walker here: [`CatalogWalk::group`] asks each object header what
573/// it declares, and [`CatalogWalk::child`] is the single place a child is
574/// classified, recorded and descended into.
575struct CatalogWalk<'a> {
576    handle: &'a mut FileHandle,
577    meta: &'a FileMeta,
578    catalog: Catalog,
579    /// Object headers already descended into, keyed to the first path that
580    /// reached them: a later path to the same header is a group hard link,
581    /// recorded in `group_aliases` so lookups resolve through it instead of
582    /// walking (and cycling) a second time.
583    visited: std::collections::HashMap<u64, String>,
584}
585
586impl<'a> CatalogWalk<'a> {
587    /// Bound group nesting on a hostile or corrupt file.
588    const MAX_DEPTH: usize = 256;
589
590    /// Start a walk at the root group's object header address, seeded so a
591    /// hard link cycling back to the root is not descended into again.
592    fn new(handle: &'a mut FileHandle, meta: &'a FileMeta, root_addr: u64) -> Self {
593        let mut visited = std::collections::HashMap::new();
594        visited.insert(root_addr, String::new());
595        Self {
596            handle,
597            meta,
598            catalog: Catalog::default(),
599            visited,
600        }
601    }
602
603    /// The address/length widths, which most of the walk needs and `meta`
604    /// carries alongside the file's B-tree ranks and shared-message table.
605    fn ctx(&self) -> &FormatContext {
606        &self.meta.ctx
607    }
608
609    fn finish(mut self) -> Catalog {
610        self.catalog.group_object_paths = self.visited;
611        self.catalog
612    }
613
614    /// Enumerate one group's children, choosing the storage from what this
615    /// group's own header declares.
616    ///
617    /// `stab` is the symbol-table scratch-pad copy from the entry that named
618    /// this group (the superblock's root entry, or the parent's symbol-table
619    /// entry), which is the only source of those addresses when the header
620    /// itself did not decode. `header` is `None` in exactly that case.
621    fn group(
622        &mut self,
623        header: Option<&ObjectHeader>,
624        prefix: &str,
625        depth: usize,
626        stab: Option<(u64, u64)>,
627    ) -> IoResult<()> {
628        if depth > Self::MAX_DEPTH {
629            return Ok(());
630        }
631        let link_storage = header.filter(|h| header_declares_link_storage(h));
632        if let Some(h) = link_storage {
633            return self.links(h, prefix, depth);
634        }
635        // Symbol-table storage: the scratch-pad copy wins when it is set,
636        // otherwise the addresses come from the group's own `stab` message.
637        let (btree_addr, heap_addr) = match stab {
638            Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
639            _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
640                Hdf5Reader::stab_from_header(h, self.ctx())
641            }),
642        };
643        if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
644            self.btree(btree_addr, heap_addr, prefix, depth)?;
645        }
646        Ok(())
647    }
648
649    /// Enumerate a group that stores its children as `Link` messages.
650    fn links(&mut self, header: &ObjectHeader, prefix: &str, depth: usize) -> IoResult<()> {
651        // Collect every link in this group: inline `Link` messages plus, for
652        // groups using dense storage, links held in a fractal heap referenced
653        // by the `Link Info` message.
654        //
655        // A link message that does not decode has no name to report the
656        // failure against, and a `Link Info` message that does not decode
657        // hides a whole group's dense storage. Either way the listing would
658        // come back silently short, so both are errors: a listing this
659        // reader cannot complete must not present itself as complete.
660        let mut links: Vec<LinkMessage> = Vec::new();
661        for msg in &header.messages {
662            if msg.msg_type == MSG_LINK {
663                let (link, _) = LinkMessage::decode(&msg.data, self.ctx())?;
664                links.push(link);
665            } else if msg.msg_type == MSG_LINK_INFO {
666                let (info, _) = LinkInfoMessage::decode(&msg.data, self.ctx())?;
667                if info.fractal_heap_address != UNDEF_ADDR {
668                    let ctx = self.meta.ctx;
669                    let dense =
670                        Hdf5Reader::read_dense_links(self.handle, &ctx, info.fractal_heap_address)?;
671                    links.extend(dense);
672                }
673            }
674        }
675
676        for link in &links {
677            let full_name = join_path(prefix, &link.name);
678            // Every link is a listing entry whatever it points at; only a
679            // hard link names an object in this file to descend into.
680            self.catalog
681                .links
682                .insert(full_name.clone(), LinkClass::from_target(&link.target));
683            let LinkTarget::Hard { address } = &link.target else {
684                continue;
685            };
686            self.child(full_name, *address, depth, None)?;
687        }
688        Ok(())
689    }
690
691    /// Enumerate a group that stores its children in a symbol-table B-tree
692    /// plus local heap.
693    fn btree(
694        &mut self,
695        btree_addr: u64,
696        heap_addr: u64,
697        prefix: &str,
698        depth: usize,
699    ) -> IoResult<()> {
700        let sa = self.ctx().sizeof_addr as usize;
701        let ss = self.ctx().sizeof_size as usize;
702
703        // Read the local heap header + data for this group.
704        let heap_hdr_buf = self.handle.read_at_most(heap_addr, 64)?;
705        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
706        let heap_data = self
707            .handle
708            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;
709
710        // Collect all SNOD addresses by walking the B-tree.
711        let mut snod_tree_visited = std::collections::HashSet::new();
712        let snod_addrs = Hdf5Reader::collect_snod_addresses(
713            self.handle,
714            self.meta,
715            btree_addr,
716            0,
717            &mut snod_tree_visited,
718        )?;
719
720        // A symbol-table node is a fixed-size record sized by `sym_leaf_k`,
721        // which the superblock extension may have overridden.
722        let snod_size = self.meta.btree.symbol_table_node_size(sa, ss);
723        for snod_addr in snod_addrs {
724            let snod_buf = self.handle.read_at_most(snod_addr, snod_size)?;
725            let snod =
726                SymbolTableNode::decode(&snod_buf, sa, ss, self.meta.btree.sym_leaf_max_entries())?;
727
728            for entry in &snod.entries {
729                let name = local_heap_get_string(&heap_data, entry.name_offset)?;
730                // Skip empty names (root group self-reference).
731                if name.is_empty() {
732                    continue;
733                }
734                let full_name = join_path(prefix, &name);
735
736                // A `H5G_CACHED_SLINK` entry is a soft link: it names no
737                // object at all, and its value string lives in this group's
738                // local heap. Record the link and move on — reading its
739                // undefined object-header address is what used to drop it.
740                if let SymbolTableCache::SoftLink { value_offset } = entry.cache {
741                    let target = local_heap_get_string(&heap_data, value_offset as u64)?;
742                    self.catalog
743                        .links
744                        .insert(full_name, LinkClass::Soft { path: target });
745                    continue;
746                }
747                self.catalog
748                    .links
749                    .insert(full_name.clone(), LinkClass::Hard);
750                self.child(
751                    full_name,
752                    entry.obj_header_addr,
753                    depth,
754                    entry.cached_symbol_table(),
755                )?;
756            }
757        }
758
759        Ok(())
760    }
761
762    /// Record one child of a group and, when it is itself a group, descend.
763    ///
764    /// Both storages end here, so a child is classified, catalogued and
765    /// cycle-guarded the same way whichever way its name was found.
766    fn child(
767        &mut self,
768        full_name: String,
769        addr: u64,
770        depth: usize,
771        stab: Option<(u64, u64)>,
772    ) -> IoResult<()> {
773        // The entry names an object, so the object is in the listing whatever
774        // comes of reading it: a header that does not decode (a stale link
775        // left by a deletion, say) is reported against this name, never
776        // dropped from it.
777        let header = match Hdf5Reader::read_object_header_full(self.handle, self.meta, addr) {
778            Ok(h) => h,
779            Err(e) => {
780                self.catalog
781                    .unreadable
782                    .insert(full_name, format!("its object header does not decode: {e}"));
783                return Ok(());
784            }
785        };
786        match Hdf5Reader::classify_object(self.handle, &header, self.meta, &full_name, addr) {
787            ObjectKind::Dataset(info) => {
788                self.catalog.datasets.push(*info);
789                return Ok(());
790            }
791            ObjectKind::UnreadableDataset(why) => {
792                self.catalog.unreadable.insert(full_name, why);
793                return Ok(());
794            }
795            // A committed (named) datatype is neither a group nor a dataset,
796            // so it must not be recorded as either; the link record above
797            // already carries its name.
798            ObjectKind::CommittedDatatype(info) => {
799                self.catalog
800                    .datatype_object_paths
801                    .insert(addr, full_name.clone());
802                self.catalog.datatypes.insert(full_name, *info);
803                return Ok(());
804            }
805            ObjectKind::Group => {}
806        }
807
808        // It is a group. Record its path from the actual link record — before
809        // the cycle check, so a hard-link alias of an already-visited group
810        // still appears — whether or not it holds datasets or attributes.
811        self.catalog.group_paths.insert(full_name.clone());
812        // Capture group attributes (e.g. the NeXus `NX_class` marker), keyed
813        // by path. Through the shared collector, which carries a per-attribute
814        // failure as an entry naming it: a group whose attribute did not
815        // decode must not come back as a group with one fewer attribute.
816        // Recorded unconditionally, even for a group with none: the entry
817        // also carries the header's own creation-order and storage facts,
818        // which exist whether or not the group currently has any attributes.
819        let ctx = self.meta.ctx;
820        let attrs = collect_object_attributes(self.handle, &ctx, &header);
821        self.catalog
822            .group_attributes
823            .insert(full_name.clone(), attrs);
824        // Same unconditional recording for link storage and link
825        // creation-order: this group's own header answers both whether or
826        // not it descends any further.
827        self.catalog.group_link_storage.insert(
828            full_name.clone(),
829            describe_link_storage(Some(&header), &ctx, stab),
830        );
831
832        // Descend at most once per object header (cycle guard); a second path
833        // to it is a group hard link — record the alias for lookups instead.
834        if let Some(first) = self.visited.get(&addr) {
835            let first = first.clone();
836            self.catalog.group_aliases.insert(full_name, first);
837            return Ok(());
838        }
839        self.visited.insert(addr, full_name.clone());
840        self.group(Some(&header), &full_name, depth + 1, stab)
841    }
842}
843
844/// Join a group path prefix and a child name, with no leading `/` on a
845/// root-level name.
846fn join_path(prefix: &str, name: &str) -> String {
847    if prefix.is_empty() {
848        name.to_string()
849    } else {
850        format!("{}/{}", prefix, name)
851    }
852}
853
854/// Whether an object header declares link-message storage (compact or
855/// dense) rather than the legacy symbol-table format — `Walk::group`'s own
856/// dispatch predicate, factored out so [`describe_link_storage`] answers the
857/// same question by construction rather than by keeping two checks in sync.
858fn header_declares_link_storage(header: &ObjectHeader) -> bool {
859    header
860        .messages
861        .iter()
862        .any(|m| m.msg_type == MSG_LINK || m.msg_type == MSG_LINK_INFO)
863}
864
865/// A group's own link storage kind and link creation-order policy — h5py's
866/// `link_storage_str` and `get_link_creation_order()`, computed together
867/// because both read the same `Link Info` message.
868///
869/// `stab` is the symbol-table scratch-pad copy from the entry that named
870/// this group, exactly as [`CatalogWalk::group`] takes it; `header` is
871/// `None` only when the object header itself did not decode.
872fn describe_link_storage(
873    header: Option<&ObjectHeader>,
874    ctx: &FormatContext,
875    stab: Option<(u64, u64)>,
876) -> (LinkStorage, CreationOrder) {
877    if let Some(h) = header.filter(|h| header_declares_link_storage(h)) {
878        // Link-message storage: the `Link Info` message, when present, gives
879        // both facts at once. Its absence means compact and untracked — no
880        // message exists to carry a creation index in.
881        return h
882            .messages
883            .iter()
884            .find(|m| m.msg_type == MSG_LINK_INFO)
885            .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
886            .map(|(info, _)| {
887                let storage = if info.is_dense() {
888                    LinkStorage::Dense
889                } else {
890                    LinkStorage::Compact
891                };
892                (storage, info.creation_order())
893            })
894            .unwrap_or((LinkStorage::Compact, CreationOrder::Untracked));
895    }
896    // No link-message storage in the header (or no header to check at all):
897    // symbol-table format when the scratch-pad or the header's own `Symbol
898    // Table` message resolves an address pair. Creation order is always
899    // untracked here — the pre-1.8 format predates the feature, and 1.8+
900    // never tracks creation order without also converting to link storage.
901    let (btree_addr, heap_addr) = match stab {
902        Some(pair) if pair.0 != UNDEF_ADDR && pair.1 != UNDEF_ADDR => pair,
903        _ => header.map_or((UNDEF_ADDR, UNDEF_ADDR), |h| {
904            Hdf5Reader::stab_from_header(h, ctx)
905        }),
906    };
907    let storage = if btree_addr != UNDEF_ADDR && heap_addr != UNDEF_ADDR {
908        LinkStorage::SymbolTable
909    } else {
910        // Neither link-message nor symbol-table storage is declared — an
911        // object header this crate could not fully account for. Nothing in
912        // the 92-case oracle suite reaches this path; it exists so an
913        // unreadable root group still answers rather than panicking.
914        LinkStorage::Compact
915    };
916    (storage, CreationOrder::Untracked)
917}
918
919/// What an object header describes.
920///
921/// libhdf5 decides an object's class from *which* messages the header holds
922/// (`H5O_obj_class`) and only then reads their contents; the two questions are
923/// separate, and answering them with one `Option<DatasetReadInfo>` is what let
924/// a dataset whose datatype this crate cannot decode leave the catalog as if
925/// the file did not contain it. Each outcome now has its own name, so no
926/// caller can turn "unreadable" back into "absent".
927enum ObjectKind {
928    /// A dataset: it carries a datatype, a dataspace and a data layout, and
929    /// every message the payload depends on decoded.
930    Dataset(Box<DatasetReadInfo>),
931    /// A dataset whose payload depends on a message this crate cannot decode.
932    /// The string names what stopped it and reaches the caller of any typed
933    /// access to the name.
934    UnreadableDataset(String),
935    /// A group: it carries link, link-info, symbol-table or group-info
936    /// storage, or none of the messages that identify anything else.
937    Group,
938    /// A committed (named) datatype object: a datatype message with neither
939    /// group storage nor the dataspace/layout pair a dataset needs.
940    CommittedDatatype(Box<CommittedDatatypeInfo>),
941}
942
943/// Whether a header's messages say it is a committed (named) datatype: a
944/// datatype message, no group storage, and not the dataspace/layout pair a
945/// dataset needs.
946///
947/// The one authority for that question. [`Hdf5Reader::classify_object`] asks it
948/// of a file being read and `ReopenWalk::plan` of one being appended to, and a
949/// header that is a named datatype to one and an unclassifiable object to the
950/// other is how `named_datatype_names` came to answer differently in the two
951/// modes for the same file.
952pub(crate) fn header_is_committed_datatype(header: &ObjectHeader) -> bool {
953    let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
954    let is_group = present(MSG_LINK)
955        || present(MSG_LINK_INFO)
956        || present(MSG_SYMBOL_TABLE)
957        || present(MSG_GROUP_INFO);
958    !is_group && present(MSG_DATATYPE) && !(present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT))
959}
960
961/// A committed (named) datatype as read from its own object header.
962///
963/// `H5Tcommit` gives a type a name and a place in the file; every dataset and
964/// attribute built on it then stores a reference to this object rather than a
965/// copy of the type. It is a third kind of object beside groups and datasets,
966/// and classifying it as neither is what left its name in the file with
967/// nothing behind it.
968#[derive(Debug, Clone)]
969pub struct CommittedDatatypeInfo {
970    /// The type this object commits, or what stopped it from decoding. The
971    /// object is in the listing either way, exactly as an unreadable dataset
972    /// is: the name is in the file whether or not this crate can read what it
973    /// names.
974    datatype: Result<DatatypeMessage, String>,
975    /// Attributes attached to the committed datatype itself.
976    attributes: Vec<AttributeMessage>,
977}
978
979impl CommittedDatatypeInfo {
980    /// The committed type, or the reason it cannot be read.
981    pub fn datatype(&self) -> Result<&DatatypeMessage, &str> {
982        self.datatype.as_ref().map_err(String::as_str)
983    }
984
985    /// The attributes attached to the committed datatype.
986    pub fn attributes(&self) -> &[AttributeMessage] {
987        &self.attributes
988    }
989}
990
991/// Everything the superblock extension object header contributes to the
992/// file-level view.
993///
994/// `H5Fsuper.c::H5F__super_read` opens this header immediately after decoding
995/// the superblock, before any user object is reachable, so every message here
996/// is in force for the first metadata decode that follows.
997#[derive(Debug, Clone, Default, PartialEq, Eq)]
998pub struct SuperblockExtension {
999    /// Shared Message Table message (0x000F): where the SOHM master table is.
1000    pub shared_message_table: Option<SharedMessageTableMessage>,
1001    /// v1 B-tree "K" values message (0x0013): non-default split ranks.
1002    pub btree_k: Option<BtreeKMessage>,
1003    /// Driver info message (0x0014).
1004    pub driver_info: Option<DriverInfoMessage>,
1005    /// File space info message (0x0017): allocation strategy, page size, and
1006    /// the persisted free-space manager addresses.
1007    pub file_space_info: Option<FileSpaceInfoMessage>,
1008}
1009
1010/// The datasets the discovery walk found, in the order it found them, with
1011/// an index from canonical path to position.
1012///
1013/// Every open and every read resolves a dataset by name, so a plain `Vec` a
1014/// lookup scans made each of them cost a pass over the whole catalog — a file
1015/// holding thousands of datasets paid that per access. The index is derived
1016/// in [`DatasetTable::new`], which is the only way to build one, so it cannot
1017/// fall out of step with the list it indexes.
1018/// One dataset's chunk index, as decoded from the file: `(chunk address,
1019/// on-disk byte count, filter mask)` per chunk the index records, in the
1020/// order the index walks them, alongside each chunk's chunk-grid coordinates
1021/// (`coords[i * rank .. (i + 1) * rank]`).
1022///
1023/// What the file decides is `entries` and `coords`; what a read then does with
1024/// an entry is that read's own business — a chunk outside the selection is
1025/// never fetched, a filtered chunk is read to its recorded size — so nothing
1026/// about one call is recorded here. `images` is the exception the rest of this
1027/// file is built around: decompressed chunk images, which say nothing about
1028/// one call either (an image is the decode of stored bytes this index names)
1029/// and which live here precisely so that they cannot outlive the index that
1030/// named them — see [`ChunkImageCache`].
1031struct DecodedChunkIndex {
1032    entries: Vec<(u64, u64, u32)>,
1033    coords: Vec<u64>,
1034    images: ChunkImageCache,
1035}
1036
1037impl DecodedChunkIndex {
1038    /// A freshly decoded index, with an empty image cache. The only way to
1039    /// build one, so no decode site can forget the cache or hand one index's
1040    /// images to another.
1041    fn new(entries: Vec<(u64, u64, u32)>, coords: Vec<u64>) -> Self {
1042        Self {
1043            entries,
1044            coords,
1045            images: ChunkImageCache::default(),
1046        }
1047    }
1048}
1049
1050/// The stored bytes one cached chunk image was decoded from: the chunk's file
1051/// address, the byte count the read asked for at it, and the per-chunk filter
1052/// mask the pipeline ran under.
1053///
1054/// An image is the decode of exactly these three, so an entry cannot answer
1055/// for a chunk stored somewhere else, read at another length, or filtered
1056/// through another mask — the key carries the whole of what the image depends
1057/// on apart from the pipeline, which belongs to the catalog entry this cache
1058/// hangs under.
1059#[derive(Clone, Copy, PartialEq, Eq, Hash)]
1060struct ChunkImageKey {
1061    addr: u64,
1062    len: usize,
1063    mask: u32,
1064}
1065
1066/// Bytes of decompressed chunk images one dataset keeps — libhdf5's
1067/// `H5D_CHUNK_CACHE_NBYTES_DEF`, the default `rdcc` size every dataset gets
1068/// there, so a comparison against it is a comparison of like for like.
1069const CHUNK_CACHE_BYTES: usize = 1024 * 1024;
1070
1071/// Images one dataset keeps at once — libhdf5's `H5D_CHUNK_CACHE_NSLOTS_DEF`.
1072/// The byte budget is the real bound; this one keeps a dataset of very small
1073/// chunks from filling the budget with thousands of entries, which is also
1074/// what bounds the least-recently-used scan below.
1075const CHUNK_CACHE_SLOTS: usize = 521;
1076
1077/// Decompressed chunk images, kept for the next read that wants the same
1078/// chunk.
1079///
1080/// Four consecutive 64 KiB slices of a 256 KiB deflated chunk are four reads
1081/// of one chunk; without a cache each inflates it whole and throws away three
1082/// quarters. libhdf5 gives every dataset a raw-data chunk cache for exactly
1083/// this (`H5D__chunk_lock`, `rdcc`), which is why a row-at-a-time walk of a
1084/// compressed dataset costs it one inflate per chunk rather than one per row.
1085///
1086/// MUST NOT outlive the [`DecodedChunkIndex`] the images were named by, and is
1087/// therefore a field of it: an image is reachable only through the index it
1088/// was decoded under, so everything that drops a decoded index —
1089/// `DatasetTable::entry_mut`, and the table rebuild a SWMR refresh does —
1090/// drops the images with it. There is no second lifetime to keep in step.
1091///
1092/// A read hands over an image only for a chunk it did not consume
1093/// ([`ChunkPlacement::leaves_chunk_unconsumed`]), so a whole-dataset read,
1094/// which takes every chunk entire and can never want one twice, never touches
1095/// the cache at all. [`ChunkImageCacheState::insert`] is the sole owner of
1096/// what the cache holds: what fits the budget, and what is evicted to keep it
1097/// fitting. Everything above it may offer an image; only it decides.
1098#[derive(Default)]
1099struct ChunkImageCache {
1100    /// Interior mutability because a read reaches the index through an `Arc`.
1101    /// A poisoned lock degrades to "no cache", never a failed read.
1102    state: std::sync::Mutex<ChunkImageCacheState>,
1103}
1104
1105#[derive(Default)]
1106struct ChunkImageCacheState {
1107    /// Key → (last-use stamp, image).
1108    images: std::collections::HashMap<ChunkImageKey, (u64, std::sync::Arc<Vec<u8>>)>,
1109    /// Summed `images` lengths, against [`CHUNK_CACHE_BYTES`].
1110    bytes: usize,
1111    /// Monotonic use counter; the smallest stamp is the eviction victim.
1112    clock: u64,
1113}
1114
1115impl ChunkImageCache {
1116    /// The image cached for `key`, if one is held.
1117    fn get(&self, key: &ChunkImageKey) -> Option<std::sync::Arc<Vec<u8>>> {
1118        let mut state = self.state.lock().ok()?;
1119        state.clock += 1;
1120        let clock = state.clock;
1121        let (stamp, image) = state.images.get_mut(key)?;
1122        *stamp = clock;
1123        Some(std::sync::Arc::clone(image))
1124    }
1125
1126    /// How many images are held. A poisoned lock holds none it could serve.
1127    #[cfg(test)]
1128    fn held(&self) -> usize {
1129        self.state.lock().map(|s| s.images.len()).unwrap_or(0)
1130    }
1131
1132    /// Keep `image` as the decode of `key`, and hand it back shared.
1133    fn keep(&self, key: ChunkImageKey, image: Vec<u8>) -> std::sync::Arc<Vec<u8>> {
1134        let image = std::sync::Arc::new(image);
1135        if let Ok(mut state) = self.state.lock() {
1136            state.insert(key, std::sync::Arc::clone(&image));
1137        }
1138        image
1139    }
1140}
1141
1142impl ChunkImageCacheState {
1143    /// Add one image, then evict least-recently-used entries until the cache
1144    /// is inside both bounds again.
1145    ///
1146    /// An image over the byte budget is refused outright rather than evicting
1147    /// everything to hold it, so the entry just inserted — which carries the
1148    /// newest stamp and is therefore the last one eviction would reach — is
1149    /// never the entry evicted.
1150    fn insert(&mut self, key: ChunkImageKey, image: std::sync::Arc<Vec<u8>>) {
1151        let len = image.len();
1152        if len > CHUNK_CACHE_BYTES {
1153            return;
1154        }
1155        self.clock += 1;
1156        if let Some((_, old)) = self.images.insert(key, (self.clock, image)) {
1157            self.bytes -= old.len();
1158        }
1159        self.bytes += len;
1160        while self.bytes > CHUNK_CACHE_BYTES || self.images.len() > CHUNK_CACHE_SLOTS {
1161            let Some(victim) = self
1162                .images
1163                .iter()
1164                .min_by_key(|(_, (stamp, _))| *stamp)
1165                .map(|(k, _)| *k)
1166            else {
1167                break;
1168            };
1169            if let Some((_, old)) = self.images.remove(&victim) {
1170                self.bytes -= old.len();
1171            }
1172        }
1173    }
1174}
1175
1176/// The reader's dataset catalog, and the one place a decoded chunk index
1177/// lives.
1178///
1179/// A cached [`DecodedChunkIndex`] MUST describe the file exactly as the
1180/// catalog entry beside it does, and MUST NOT be handed out for an index
1181/// address other than the one it was decoded from. The fields are private to
1182/// the module, so an entry is reachable read-only (`get`, `iter`, `Index`) or
1183/// through [`DatasetTable::entry_mut`], which drops that entry's cached index
1184/// before handing out the reference: changing what an entry says about the
1185/// file cannot leave a stale index behind it. Rebuilding the table — what a
1186/// SWMR refresh does — drops every cached index with it.
1187mod dataset_table {
1188    use super::{DatasetReadInfo, DecodedChunkIndex};
1189    use std::sync::Arc;
1190
1191    pub(super) struct DatasetTable {
1192        list: Vec<DatasetReadInfo>,
1193        /// Canonical path (no leading `/`) → position in `list`. A name the
1194        /// walk recorded twice keeps its first position, which is the entry a
1195        /// scan from the front would have found.
1196        by_name: std::collections::HashMap<String, usize>,
1197        /// Per entry, the index address a chunk index was decoded from and
1198        /// what it decoded to. Parallel to `list`.
1199        chunk_index: Vec<Option<(u64, Arc<DecodedChunkIndex>)>>,
1200    }
1201
1202    impl DatasetTable {
1203        pub(super) fn new(list: Vec<DatasetReadInfo>) -> Self {
1204            let mut by_name = std::collections::HashMap::with_capacity(list.len());
1205            for (i, ds) in list.iter().enumerate() {
1206                by_name.entry(ds.name.clone()).or_insert(i);
1207            }
1208            let chunk_index = list.iter().map(|_| None).collect();
1209            Self {
1210                list,
1211                by_name,
1212                chunk_index,
1213            }
1214        }
1215
1216        /// Where the dataset named `name` sits in the list, or `None` when the
1217        /// catalog holds no such name.
1218        pub(super) fn position(&self, name: &str) -> Option<usize> {
1219            self.by_name.get(name).copied()
1220        }
1221
1222        pub(super) fn get(&self, name: &str) -> Option<&DatasetReadInfo> {
1223            self.list.get(self.position(name)?)
1224        }
1225
1226        pub(super) fn iter(&self) -> std::slice::Iter<'_, DatasetReadInfo> {
1227            self.list.iter()
1228        }
1229
1230        /// The entry at `i`, mutably. Whatever the caller changes about it may
1231        /// be what a decoded chunk index was read against, so that index goes
1232        /// first: this is the only way to a `&mut DatasetReadInfo`, which is
1233        /// what keeps the cache from outliving the entry it describes.
1234        pub(super) fn entry_mut(&mut self, i: usize) -> &mut DatasetReadInfo {
1235            self.chunk_index[i] = None;
1236            &mut self.list[i]
1237        }
1238
1239        /// The chunk index cached for entry `i`, if one was decoded from
1240        /// `index_address`. A different address means the entry now points at
1241        /// another structure, and the cached one does not answer for it.
1242        pub(super) fn chunk_index(
1243            &self,
1244            i: usize,
1245            index_address: u64,
1246        ) -> Option<&Arc<DecodedChunkIndex>> {
1247            match self.chunk_index.get(i)? {
1248                Some((addr, index)) if *addr == index_address => Some(index),
1249                _ => None,
1250            }
1251        }
1252
1253        /// Keep `index` as entry `i`'s decoded chunk index for
1254        /// `index_address`, and hand it back.
1255        pub(super) fn cache_chunk_index(
1256            &mut self,
1257            i: usize,
1258            index_address: u64,
1259            index: DecodedChunkIndex,
1260        ) -> Arc<DecodedChunkIndex> {
1261            let index = Arc::new(index);
1262            self.chunk_index[i] = Some((index_address, Arc::clone(&index)));
1263            index
1264        }
1265    }
1266
1267    impl std::ops::Index<usize> for DatasetTable {
1268        type Output = DatasetReadInfo;
1269
1270        fn index(&self, i: usize) -> &DatasetReadInfo {
1271            &self.list[i]
1272        }
1273    }
1274}
1275
1276use dataset_table::DatasetTable;
1277
1278/// HDF5 file reader.
1279pub struct Hdf5Reader {
1280    handle: FileHandle,
1281    meta: FileMeta,
1282    /// Messages read from the superblock extension object header, empty when
1283    /// the file has no extension.
1284    ext: SuperblockExtension,
1285    /// End-of-file address from the superblock.
1286    _eof: u64,
1287    /// Superblock format version (0-3), decoded once at open time by
1288    /// `detect_superblock_version` and never re-derived: 0/1 is the legacy
1289    /// symbol-table root, 2/3 the link-message root.
1290    superblock_version: u8,
1291    datasets: DatasetTable,
1292    /// Dataset-shaped objects this crate cannot read, keyed by path (no
1293    /// leading `/`), the value naming what stopped it. They are listed with
1294    /// the readable datasets and refuse typed access with that reason: an
1295    /// object the file contains is never reported as one it does not.
1296    unreadable: std::collections::BTreeMap<String, String>,
1297    /// Attributes on the root group (file-level attributes).
1298    root_attributes: ObjectAttributes,
1299    /// Link storage kind and link creation-order policy of the root group.
1300    root_link_storage: (LinkStorage, CreationOrder),
1301    /// Attributes on non-root groups, keyed by group path (no leading `/`).
1302    group_attributes: std::collections::HashMap<String, ObjectAttributes>,
1303    /// Link storage kind and link creation-order policy of non-root groups,
1304    /// keyed by group path (no leading `/`).
1305    group_link_storage: std::collections::HashMap<String, (LinkStorage, CreationOrder)>,
1306    /// Every non-root group path the discovery walk traversed into (no
1307    /// leading `/`), regardless of whether the group has datasets or
1308    /// attributes. Built from actual link records, so empty groups,
1309    /// attribute-only groups, and subgroup-only groups are all included.
1310    group_paths: std::collections::BTreeSet<String>,
1311    /// Group hard links: alias path → the first-walked path of the same
1312    /// group object header (both without a leading `/`). The walk
1313    /// descends each header once, so objects under the alias are stored
1314    /// under the first path; lookups resolve alias prefixes through this
1315    /// map, as HDF5 path traversal does.
1316    group_aliases: std::collections::HashMap<String, String>,
1317    /// Every link record in the file, keyed by full path (no leading `/`).
1318    /// A listing is a listing of links, so this holds soft and external
1319    /// links as well as the hard links that name the objects above.
1320    links: std::collections::BTreeMap<String, LinkClass>,
1321    /// The path this file was opened with. An external link resolves its
1322    /// target relative to the directory holding it (libhdf5 keeps the same
1323    /// thing as `H5F_EXTPATH`), so the reader has to remember where it came
1324    /// from.
1325    path: PathBuf,
1326    /// The directory holding this HDF5 file, resolved once at open time.
1327    /// External raw-data files (H5O_EFL_ID) are named relative to it when
1328    /// `HDF5_EXTFILE_PREFIX` contains `${ORIGIN}` (`H5D__build_file_prefix`,
1329    /// H5Dint.c) — captured at open time rather than re-derived from the
1330    /// process's current directory at read time, matching libhdf5's own
1331    /// one-time capture in `H5F_t::extpath`.
1332    source_dir: PathBuf,
1333    /// The locking policy this file was opened under, reused verbatim for
1334    /// every external target: libhdf5 hands `H5F_prefix_open_file` the
1335    /// parent's file-access property list, so one `HDF5_USE_FILE_LOCKING`
1336    /// setting (or one `H5FileOptions::locking` call) governs every file a
1337    /// path touches, not just the first.
1338    locking: crate::io::locking::FileLocking,
1339    /// `H5Pset_elink_prefix` for every external link this reader crosses —
1340    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix).
1341    /// Propagated verbatim to every target this reader opens, the way a lapl
1342    /// reaches the next hop of a link chain upstream (measured against
1343    /// libhdf5 1.14.6: a two-hop chain resolves its second hop under the
1344    /// prefix given at the first).
1345    elink_prefix: Option<String>,
1346    /// Files opened on another file's behalf — external-link targets,
1347    /// external-reference targets and virtual-dataset sources — keyed by the
1348    /// resolved path that opened them, so N names for one file share one
1349    /// open handle. libhdf5 shares one open file the same way: `H5F_open`
1350    /// hands back the `H5F_shared_t` a path already open has rather than a
1351    /// second one (H5Fint.c:1906-1918).
1352    ///
1353    /// How long an entry lives is its [`CrossFileOwner`]'s business, and
1354    /// differs by what named it. A target's own external links are cached in
1355    /// that target's map, so the first reader in a chain transitively holds
1356    /// the whole chain open.
1357    external: std::collections::BTreeMap<PathBuf, CrossFileEntry>,
1358    /// What each virtual mapping's stored source file name resolved to,
1359    /// keyed by the canonical path of the virtual dataset that named it and
1360    /// that name exactly as the mapping holds it. Separate from
1361    /// [`external_resolved`](Self::external_resolved) because the two
1362    /// searches read different environment variables and property lists, so
1363    /// one name can resolve two ways depending on which named it — and keyed
1364    /// by the virtual dataset as well because
1365    /// [`DatasetAccess::virtual_prefix`] is a per-open property, so two
1366    /// virtual datasets naming one source file name can legitimately reach
1367    /// two different files.
1368    vds_resolved: std::collections::BTreeMap<(String, String), PathBuf>,
1369    /// What each external link's stored file name resolved to, keyed by that
1370    /// name exactly as the link holds it.
1371    ///
1372    /// The search runs once per distinct name per reader and its answer is
1373    /// then fixed: re-probing the filesystem on a later crossing would let one
1374    /// link answer differently mid-session, and would fail to find the handle
1375    /// it already holds once the target has been renamed or unlinked.
1376    external_resolved: std::collections::BTreeMap<String, PathBuf>,
1377    /// Committed (named) datatype objects, keyed by path (no leading `/`).
1378    datatypes: std::collections::BTreeMap<String, CommittedDatatypeInfo>,
1379    /// Object header address → absolute path, for every group and dataset the
1380    /// discovery walk reached plus the root group. This is what turns the
1381    /// address an object reference stores back into a name.
1382    object_paths: std::collections::HashMap<u64, String>,
1383    /// The dataset-access properties in force for each *open* dataset, keyed
1384    /// by canonical path (no leading `/`); an absent entry means
1385    /// [`DatasetAccess::default`].
1386    ///
1387    /// libhdf5 keeps this in the dataset's *shared* open-object info, which
1388    /// is why the first open of a dataset fixes it for every later one
1389    /// ([`apply_dataset_access`](Self::apply_dataset_access)):
1390    /// `H5D__virtual_init` puts the view and the printf gap into
1391    /// `dset->shared->layout.storage.u.virt` (H5Dvirtual.c:2178-2188), and
1392    /// `H5D__open_name` puts both file prefixes into `shared->extfile_prefix`
1393    /// and `shared->vds_prefix` (H5Dint.c:1488-1521) — for every dataset,
1394    /// not only a virtual one. It is the single owner of that answer: the
1395    /// extent resolution and the external-file read both read it and nothing
1396    /// else writes it, so a SWMR [`refresh`](Self::refresh) re-resolves under
1397    /// the same properties rather than reverting to the defaults.
1398    dataset_access: std::collections::BTreeMap<String, AccessInForce>,
1399}
1400
1401/// A live open on a dataset.
1402///
1403/// The reader holds only a [`Weak`](std::sync::Weak) to it and every handle
1404/// an open handed out holds the strong one, so "is this dataset still open"
1405/// is answered by the handles themselves — nothing has to tell the reader
1406/// when one is dropped, and a handle's drop takes no lock on the file.
1407pub(crate) type DatasetOpenToken = std::sync::Arc<()>;
1408
1409/// One file this reader opened on another file's behalf, and what keeps it
1410/// open.
1411struct CrossFileEntry {
1412    reader: Box<Hdf5Reader>,
1413    owner: CrossFileOwner,
1414}
1415
1416/// What holds a [`CrossFileEntry`] open, which is the same question as how
1417/// long it stays open.
1418///
1419/// libhdf5 asks it the same way and gets two different answers, because
1420/// nothing caches these files across the open that needed them: the default
1421/// external file cache is *disabled* — `H5F_ACS_EFC_SIZE_DEF` is 0
1422/// (H5Pfapl.c:191) and `H5F_open` builds no cache below that
1423/// (H5Fint.c:1217-1218) — so the `H5F_efc_close` both kinds of crossing end
1424/// with (H5Lexternal.c:241, H5Dvirtual.c:927) falls straight through to
1425/// `H5F_try_close` (H5Fefc.c:420-426). What is left holding the file is
1426/// whatever object the crossing opened inside it.
1427enum CrossFileOwner {
1428    /// This reader, until it is dropped.
1429    ///
1430    /// An external link's target is held by the object the traversal opened
1431    /// in it (`H5O_open_name`, H5Lexternal.c:225), and an external
1432    /// reference's by `H5R__reopen_file`'s file handle; this crate's
1433    /// equivalent of those objects is the reader itself, which holds the
1434    /// target's whole catalog and answers every later name from it.
1435    ///
1436    /// This is a **deliberate difference from libhdf5**, not a match.
1437    /// Measured against 1.14.6 by watching `/proc/self/fd`: a link target's
1438    /// descriptor appears at the traversal, survives while any object opened
1439    /// through the link is open, and goes at that object's close — and a
1440    /// traversal that keeps no object (`f["ext/data"][...]`, a listing, an
1441    /// `H5Lget_info`) leaves nothing open at all. It is *not* cached past
1442    /// that: see this enum's own note on the disabled default EFC.
1443    ///
1444    /// Matching it would mean this crate re-walked a target file's whole
1445    /// catalog on every name that crosses the link, because a reader — not
1446    /// a per-object handle — is what it opens a target *as*. The delta is a
1447    /// descriptor-and-lock window with no effect on any byte read or
1448    /// written, so the reader's lifetime stands and the window is documented
1449    /// here rather than paid for with that redesign.
1450    Reader,
1451    /// The virtual datasets that named this file as a source, by canonical
1452    /// path in *this* reader. Dropped once none of them has a live open.
1453    ///
1454    /// `H5D__virtual_open_source_dset` leaves the source *dataset* open in
1455    /// the virtual dataset's shared layout (H5Dvirtual.c:901-902), and that
1456    /// is what keeps the source file open; `H5D__virtual_reset_layout` closes
1457    /// it at the last `H5Dclose` of the virtual dataset (H5Dvirtual.c:709-710
1458    /// via `H5D__virtual_reset_source_dset`, :955). Measured against
1459    /// libhdf5 1.14.6 by watching `/proc/self/fd`: the source appears at the
1460    /// read that needs it, survives a second virtual dataset naming the same
1461    /// file until *both* are closed, and goes at the last `H5Dclose` — not at
1462    /// `H5Fclose`.
1463    VirtualOpens(std::collections::BTreeSet<String>),
1464}
1465
1466impl CrossFileOwner {
1467    /// Record that `also` now names this file too, keeping whichever
1468    /// ownership outlives the other. [`Reader`](Self::Reader) outlives every
1469    /// virtual open, so once a file is held that way it stays held.
1470    fn widen(&mut self, also: CrossFileOwner) {
1471        match (&mut *self, also) {
1472            (CrossFileOwner::Reader, _) => {}
1473            (slot, CrossFileOwner::Reader) => *slot = CrossFileOwner::Reader,
1474            (CrossFileOwner::VirtualOpens(have), CrossFileOwner::VirtualOpens(more)) => {
1475                have.extend(more)
1476            }
1477        }
1478    }
1479
1480    /// The owner of a source file one virtual dataset named.
1481    fn virtual_open(vds: &str) -> Self {
1482        CrossFileOwner::VirtualOpens(std::iter::once(vds.to_string()).collect())
1483    }
1484}
1485
1486/// The dataset-access properties one dataset was resolved under, and the
1487/// opens that fixed them.
1488struct AccessInForce {
1489    access: DatasetAccess,
1490    /// Live while at least one handle from the open that set `access` is
1491    /// alive. Once it is dead the properties are only a record of how the
1492    /// stamped extent was arrived at (what a SWMR refresh re-resolves
1493    /// under); the next open resolves afresh under its own.
1494    open: std::sync::Weak<()>,
1495}
1496
1497/// The absolute form of a discovery-walk path (which carries no leading `/`):
1498/// the root group's empty path becomes `/`, `entry/data` becomes
1499/// `/entry/data`.
1500fn absolute_path(path: &str) -> String {
1501    format!("/{}", path.trim_start_matches('/'))
1502}
1503
1504/// Total byte length of `dims.product() * element_size`, computed with
1505/// saturating arithmetic. `dims` and `element_size` are file-derived; a
1506/// crafted file with huge dimensions thus yields a saturated (too-large)
1507/// value — rejected downstream by the file-size/buffer checks — rather
1508/// than panicking in a debug build or wrapping in release.
1509fn saturating_byte_len(dims: &[u64], element_size: u64) -> u64 {
1510    dims.iter()
1511        .fold(1u64, |acc, &d| acc.saturating_mul(d))
1512        .saturating_mul(element_size)
1513}
1514
1515/// A fixed-length string attribute's value, under the padding rule its
1516/// datatype declares.
1517///
1518/// The single owner for the attribute side of that rule: both
1519/// [`H5Reader::attr_string_value`] and the writer-mode fallback in
1520/// `H5Attribute::read_string` end here, so a space-padded attribute does not
1521/// read back with its padding attached on one path and not the other. Bytes
1522/// that are not valid UTF-8 become U+FFFD, as they always have on this path.
1523///
1524/// A datatype that is not a string at all is read as null-terminated, which is
1525/// what asking for the string value of, say, an integer attribute has always
1526/// meant here.
1527pub(crate) fn fixed_string_attr_value(attr: &AttributeMessage) -> IoResult<String> {
1528    use crate::format::messages::datatype::{fixed_string_content, DatatypeMessage};
1529    let padding = match attr.datatype {
1530        DatatypeMessage::FixedString { padding, .. } => padding,
1531        _ => 0,
1532    };
1533    let content = fixed_string_content(&attr.data, padding).ok_or_else(|| {
1534        crate::io::IoError::InvalidState(format!(
1535            "attribute {:?} uses string padding rule {padding}, which the format reserves",
1536            attr.name
1537        ))
1538    })?;
1539    Ok(String::from_utf8_lossy(content).to_string())
1540}
1541
1542/// Materialize a `total`-byte fill buffer, mapping allocation failure to a
1543/// clean error. `total` on a read path comes from untrusted file fields, so
1544/// a crafted file declaring an absurd dataset size would otherwise abort the
1545/// process when `vec![0u8; total]` fails to allocate.
1546fn alloc_tiled_fill(total: usize, fill_value: Option<&[u8]>) -> IoResult<Vec<u8>> {
1547    try_tiled_fill(total, fill_value).map_err(|_| {
1548        crate::io::IoError::InvalidState(format!(
1549            "cannot allocate {total} bytes for dataset buffer (file may be corrupt)"
1550        ))
1551    })
1552}
1553
1554/// Build a `Vec<T>` of `count` elements out of the bytes `define` writes.
1555///
1556/// `define` receives the vector's whole byte image — `count *
1557/// size_of::<T>()` bytes — and **must define every one of them** before it
1558/// returns `Ok`; only then are the elements claimed. That contract is what
1559/// lets a full-image read land in its destination directly: the buffer a read
1560/// fills is already the buffer the caller keeps, so a 128 MiB image is
1561/// touched once by the read instead of once to zero it, once to read it and
1562/// once to copy it into a typed vector.
1563///
1564/// [`Hdf5Reader::read_dataset_raw_into_unconverted`] is the read side of that
1565/// contract — it is the single owner of read-destination semantics precisely
1566/// because it defines every byte of the buffer it is handed — and
1567/// [`read_dataset_raw_into`](Hdf5Reader::read_dataset_raw_into) inherits it.
1568///
1569/// `count` on a read path comes from untrusted file fields, so the
1570/// reservation is fallible: a crafted file declaring an absurd dataset size
1571/// gets a clean error rather than an allocator abort.
1572pub(crate) fn read_image_into_new<T, E, F>(count: usize, define: F) -> Result<Vec<T>, E>
1573where
1574    T: crate::types::H5Type,
1575    F: FnOnce(&mut [u8]) -> Result<(), E>,
1576    E: From<crate::io::IoError>,
1577{
1578    let too_big = || {
1579        E::from(crate::io::IoError::InvalidState(format!(
1580            "cannot allocate {count} elements of {} bytes for a dataset buffer \
1581             (file may be corrupt)",
1582            std::mem::size_of::<T>()
1583        )))
1584    };
1585    let bytes = count
1586        .checked_mul(std::mem::size_of::<T>())
1587        .ok_or_else(too_big)?;
1588    let mut out: Vec<T> = Vec::new();
1589    out.try_reserve_exact(count).map_err(|_| too_big())?;
1590
1591    // Safety: `try_reserve_exact` succeeded, so the allocation holds `bytes`
1592    // contiguous bytes aligned for `T`, and `as_mut_ptr` is non-null (a
1593    // dangling-but-aligned pointer with `bytes == 0`, which an empty slice
1594    // permits). The slice is the only live reference to that memory while
1595    // `define` runs. `define` writes every byte before returning `Ok` — its
1596    // documented contract — so the elements are initialized by the time
1597    // `set_len` claims them, and every byte pattern is a valid `T`, the same
1598    // property of `H5Type` implementors that a typed read reinterpreting the
1599    // stored image already rests on. A `define` that fails returns before
1600    // `set_len`, so the vector drops empty and nothing reads the bytes it
1601    // left undefined.
1602    let image = unsafe { std::slice::from_raw_parts_mut(out.as_mut_ptr().cast::<u8>(), bytes) };
1603    define(image)?;
1604    unsafe { out.set_len(count) };
1605    Ok(out)
1606}
1607
1608/// Fill an existing buffer in place with the dataset's tiled fill value (or
1609/// zero when no fill value is set), matching [`try_tiled_fill`]'s tiling.
1610///
1611/// Used to initialize a read destination — both an internally-allocated `Vec`
1612/// and a caller-provided read-into buffer (whose prior contents are arbitrary)
1613/// — so any region a chunked read leaves untouched reads back as the fill
1614/// value rather than stale bytes.
1615fn fill_tiled_into(out: &mut [u8], fill_value: Option<&[u8]>) {
1616    out.fill(0);
1617    if let Some(fv) = fill_value {
1618        if !fv.is_empty() && !out.is_empty() {
1619            for slot in out.chunks_mut(fv.len()) {
1620                let n = slot.len().min(fv.len());
1621                slot[..n].copy_from_slice(&fv[..n]);
1622            }
1623        }
1624    }
1625}
1626
1627/// Resolve the directory a raw-data file name is joined against, matching
1628/// libhdf5's `H5D__build_file_prefix` (H5Dint.c) for the given environment
1629/// variable — `HDF5_EXTFILE_PREFIX` for External Data Files
1630/// ([`resolve_extfile_prefix`]), `HDF5_VDS_PREFIX` for Virtual Dataset
1631/// sources ([`resolve_vdsfile_prefix`]); both features route through the
1632/// same C function, just keyed on a different variable. `prop` is the dapl
1633/// property the variable falls back to, `None` when the caller has none —
1634/// which behaves exactly as an unset or empty one would.
1635///
1636/// `${ORIGIN}` expands to `source_dir` (the directory holding the open
1637/// HDF5 file); any other value is used as a literal prefix; unset, empty,
1638/// or `"."` means "no prefix" (`H5_combine_path`'s own default), so a
1639/// relative name resolves against the process's current directory instead.
1640fn resolve_file_prefix(env_var: &str, prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1641    // `H5D__build_file_prefix` reads the environment variable first and only
1642    // falls back to the property list when it is unset or empty
1643    // (H5Dint.c:1077-1082, :1085-1090) — so an environment prefix *shadows*
1644    // the property rather than being tried before it. What
1645    // `H5F_prefix_open_file` then tries before this is the raw environment
1646    // string, split on `:` and unexpanded, which is a different candidate
1647    // from the expansion built here.
1648    let env = std::env::var(env_var).ok().filter(|v| !v.is_empty());
1649    let prefix = match env.as_deref() {
1650        Some(v) => v,
1651        None => prop.filter(|v| !v.is_empty())?,
1652    };
1653    if prefix.is_empty() || prefix == "." {
1654        return None;
1655    }
1656    Some(match prefix.strip_prefix("${ORIGIN}") {
1657        Some(rest) => {
1658            let rest = rest.trim_start_matches(['/', '\\']);
1659            if rest.is_empty() {
1660                source_dir.to_path_buf()
1661            } else {
1662                source_dir.join(rest)
1663            }
1664        }
1665        None => PathBuf::from(prefix),
1666    })
1667}
1668
1669/// Resolve the directory an external file list's stored names are joined
1670/// against — `HDF5_EXTFILE_PREFIX`, or `H5Pset_efile_prefix` when the
1671/// environment names none (see [`resolve_file_prefix`]).
1672///
1673/// The single owner of that rule for both directions of I/O: `H5D__efl_read`
1674/// and `H5D__efl_write` join against the same `shared->extfile_prefix`
1675/// (H5Defl.c:315-317, :429-431), which `H5D__build_file_prefix` built once
1676/// for the open that created the dataset's shared info.
1677pub(crate) fn resolve_extfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1678    resolve_file_prefix("HDF5_EXTFILE_PREFIX", prop, source_dir)
1679}
1680
1681/// Resolve the directory Virtual Dataset source file names are joined
1682/// against — `HDF5_VDS_PREFIX`, or `H5Pset_virtual_prefix` when the
1683/// environment names none (see [`resolve_file_prefix`]).
1684fn resolve_vdsfile_prefix(prop: Option<&str>, source_dir: &Path) -> Option<PathBuf> {
1685    resolve_file_prefix("HDF5_VDS_PREFIX", prop, source_dir)
1686}
1687
1688/// Join a raw-data file `name` against a resolved prefix, matching
1689/// libhdf5's `H5_combine_path` (H5system.c): an absolute `name` is used
1690/// as-is regardless of the prefix, and no prefix means "relative to the
1691/// process's current directory" — both of which `Path::join` already
1692/// implements for an absolute joinee. Shared by External Data Files and
1693/// Virtual Dataset source resolution — both call the same C function.
1694pub(crate) fn combine_prefixed_path(prefix: Option<&Path>, name: &str) -> PathBuf {
1695    match prefix {
1696        Some(p) => p.join(name),
1697        None => PathBuf::from(name),
1698    }
1699}
1700
1701/// Read `len` bytes starting at *dataset-relative* offset `skip` from an
1702/// external file list into `out`, walking slots by cumulative declared
1703/// size exactly like libhdf5's `H5D__efl_read` (H5Defl.c). A read past an
1704/// individual slot's actual on-disk length reads back as zero — the file
1705/// backing a slot may be shorter than the space the layout reserved in
1706/// it — but a read past the *total* declared size of the file list is
1707/// still an error, matching `H5D__efl_read`'s own "read past logical end
1708/// of file" check.
1709///
1710/// An `H5O_EFL_UNLIMITED` last slot needs no special case: the walk below
1711/// never steps past it (`skip >= u64::MAX` is never true), which is upstream's
1712/// `H5O_EFL_UNLIMITED == size || addr < cur + size`, and the read it then
1713/// takes is the whole remainder, bounded by whatever the file physically
1714/// holds.
1715fn read_external_file_bytes(
1716    external_files: &[ExternalFileSegment],
1717    extfile_prefix: Option<&Path>,
1718    mut skip: u64,
1719    out: &mut [u8],
1720) -> IoResult<()> {
1721    let mut slot_idx = 0usize;
1722    while slot_idx < external_files.len() && skip >= external_files[slot_idx].size {
1723        skip -= external_files[slot_idx].size;
1724        slot_idx += 1;
1725    }
1726
1727    let mut written = 0usize;
1728    while written < out.len() {
1729        let Some(slot) = external_files.get(slot_idx) else {
1730            return Err(crate::io::IoError::InvalidState(
1731                "read past the logical end of the external file list".into(),
1732            ));
1733        };
1734        let full_path = combine_prefixed_path(extfile_prefix, &slot.name);
1735        let ext_handle = FileHandle::open_read_with_locking(&full_path, FileLocking::Disabled)
1736            .map_err(|e| {
1737                crate::io::IoError::InvalidState(format!(
1738                    "unable to open external raw data file {}: {e}",
1739                    full_path.display()
1740                ))
1741            })?;
1742        let avail_in_slot = slot.size.saturating_sub(skip);
1743        let want = (out.len() - written) as u64;
1744        let this_read = avail_in_slot.min(want) as usize;
1745        let dst = &mut out[written..written + this_read];
1746        // A short physical file — the reserved slot size exceeds what was
1747        // ever actually written to it — reads back as zero for the
1748        // remainder, exactly like `H5D__efl_read`.
1749        let got = ext_handle.read_at_most(slot.offset + skip, this_read)?;
1750        dst[..got.len()].copy_from_slice(&got);
1751        dst[got.len()..].fill(0);
1752
1753        written += this_read;
1754        skip = 0;
1755        slot_idx += 1;
1756    }
1757    Ok(())
1758}
1759
1760/// Recursion ceiling for virtual dataset nesting (a VDS whose source is
1761/// itself a VDS — possibly in another file). Bounded so a crafted cyclic
1762/// mapping chain fails cleanly instead of recursing until the stack
1763/// overflows; real VDS chains do not nest anywhere near this deep.
1764const MAX_VIRTUAL_DEPTH: usize = 16;
1765
1766/// Replace every unlimited mapping with the concrete mapping its open-time
1767/// resolution makes it — the clipped selections `H5D__virtual_set_extent_unlim`
1768/// leaves in `clipped_virtual_select` / `clipped_source_select` for a read to
1769/// use (`H5D__virtual_read` never sees the unclipped ones).
1770///
1771/// A mapping with no resolution recorded is passed through unchanged, which is
1772/// what a bounded mapping needs and what a mapping list written before this
1773/// pass existed reduces to.
1774fn concrete_virtual_mappings(
1775    list: &VirtualMappingList,
1776    resolution: &[MappingResolution],
1777) -> IoResult<Vec<VirtualMapping>> {
1778    let mut out = Vec::with_capacity(list.mappings.len());
1779    for (i, m) in list.mappings.iter().enumerate() {
1780        match resolution.get(i) {
1781            Some(MappingResolution::Unlimited {
1782                virtual_clip,
1783                source_clip,
1784            }) => out.push(VirtualMapping {
1785                virtual_selection: m.virtual_selection.clip_unlimited(*virtual_clip)?,
1786                source_selection: m.source_selection.clip_unlimited(*source_clip)?,
1787                ..built_names(m, 0)?
1788            }),
1789            // One printf mapping is a whole family: block `j` of the virtual
1790            // selection (`H5S_hyper_get_unlim_block`) is filled by the source
1791            // dataset whose name substitutes `j`, taking the mapping's whole
1792            // (limited) source selection. Only the blocks that have a source
1793            // become mappings — a non-zero printf gap leaves the others
1794            // inside the extent, reading as the fill value.
1795            Some(MappingResolution::Printf { present, .. }) => {
1796                let Some(r) = regular_hyperslab(&m.virtual_selection) else {
1797                    continue;
1798                };
1799                let rank = r.start.len();
1800                for &j in present {
1801                    out.push(VirtualMapping {
1802                        virtual_selection: Selection::Hyperslab {
1803                            rank,
1804                            form: Hyperslab::Regular(r.unlim_block(j)),
1805                        },
1806                        ..built_names(m, j)?
1807                    });
1808                }
1809            }
1810            _ => out.push(built_names(m, 0)?),
1811        }
1812    }
1813    Ok(out)
1814}
1815
1816/// One mapping with both source names built for block `blockno` —
1817/// `H5D__virtual_build_source_name`. A mapping with no substitutions still
1818/// goes through this, because that is where `%%` is unescaped: upstream uses
1819/// the parsed name rather than the stored one for an ordinary mapping too
1820/// (`H5D__virtual_load_layout`).
1821fn built_names(m: &VirtualMapping, blockno: u64) -> IoResult<VirtualMapping> {
1822    Ok(VirtualMapping {
1823        source_file_name: parse_source_name(&m.source_file_name)?.build(blockno),
1824        source_dset_name: parse_source_name(&m.source_dset_name)?.build(blockno),
1825        ..m.clone()
1826    })
1827}
1828
1829/// Recursion bound for open-time virtual-extent resolution.
1830///
1831/// Resolving a virtual dataset's extent opens the source datasets its
1832/// unlimited mappings name, and a source may itself be a virtual dataset in
1833/// another file whose own open resolves its own extent. A crafted cyclic
1834/// chain would otherwise recurse until the stack overflows, so the nesting is
1835/// counted per thread and the resolution is skipped once it reaches
1836/// [`MAX_VIRTUAL_DEPTH`] — a dataset that deep keeps its stored extent
1837/// instead of taking one from a cycle.
1838struct VirtualResolveDepth;
1839
1840thread_local! {
1841    static VIRTUAL_RESOLVE_DEPTH: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
1842}
1843
1844impl VirtualResolveDepth {
1845    fn enter<F: FnOnce() -> IoResult<()>>(f: F) -> IoResult<()> {
1846        let depth = VIRTUAL_RESOLVE_DEPTH.with(std::cell::Cell::get);
1847        if depth >= MAX_VIRTUAL_DEPTH {
1848            return Ok(());
1849        }
1850        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth + 1));
1851        let out = f();
1852        VIRTUAL_RESOLVE_DEPTH.with(|d| d.set(depth));
1853        out
1854    }
1855}
1856
1857/// The regular (start, stride, count, block) form behind a selection, or
1858/// `None` — the only form `H5S_UNLIMITED` can appear in, so every unlimited
1859/// computation goes through it.
1860fn regular_hyperslab(sel: &Selection) -> Option<&RegularHyperslab> {
1861    match sel {
1862        Selection::Hyperslab {
1863            form: Hyperslab::Regular(r),
1864            ..
1865        } => Some(r),
1866        _ => None,
1867    }
1868}
1869
1870/// Read `source` (via `read_source_box`, one call per box) and scatter it
1871/// into `out` at the elements `target` selects.
1872///
1873/// The two selections are paired element by element in their own linear
1874/// order — `H5S_select_project_intersection` (H5Sselect.c:2402) walks a VDS
1875/// mapping's virtual and source selections with one iterator each and
1876/// matches the two streams off one against one. Its only precondition is the
1877/// one asserted there and enforced by `H5D_virtual_check_mapping_pre` when
1878/// the mapping is written (H5Dvirtual.c:254-257): the two hold the same
1879/// number of elements. Ranks may differ, and so may the boxes each side
1880/// decomposes into — a source `H5S_SEL_ALL` over a 2x4 dataset legitimately
1881/// fills two 1x4 blocks of a virtual dataset.
1882///
1883/// Each source box is read whole, once, and the runs the pairing places from
1884/// it are copied out of that one buffer.
1885fn copy_matched_selections(
1886    mut read_source_box: impl FnMut(&[u64], &[u64], &mut [u8]) -> IoResult<()>,
1887    source: &ResolvedSelection,
1888    target: &ResolvedSelection,
1889    element_size: u64,
1890    out: &mut [u8],
1891) -> IoResult<()> {
1892    let (n_source, n_target) = (source.n_elements(), target.n_elements());
1893    if n_source != n_target {
1894        return Err(crate::io::IoError::InvalidState(format!(
1895            "virtual dataset mapping's source selection holds {n_source} elements and its              virtual selection {n_target}, which H5D_virtual_check_mapping_pre refuses"
1896        )));
1897    }
1898    // Pair the two element streams, splitting a run of either side wherever
1899    // the other side's run ends first, and file each matched segment under
1900    // the source box it must be read out of.
1901    let mut per_box: Vec<Vec<(u64, u64, u64)>> = vec![Vec::new(); source.boxes.len()];
1902    let (mut si, mut ti) = (0usize, 0usize);
1903    let (mut s_done, mut t_done) = (0u64, 0u64);
1904    while si < source.runs.len() && ti < target.runs.len() {
1905        let (s, t) = (source.runs[si], target.runs[ti]);
1906        let len = (s.len - s_done).min(t.len - t_done);
1907        per_box[s.box_index].push((s.offset_in_box + s_done, t.offset_in_extent + t_done, len));
1908        s_done += len;
1909        t_done += len;
1910        if s_done == s.len {
1911            si += 1;
1912            s_done = 0;
1913        }
1914        if t_done == t.len {
1915            ti += 1;
1916            t_done = 0;
1917        }
1918    }
1919
1920    for (segments, (box_start, box_count)) in per_box.iter().zip(&source.boxes) {
1921        if segments.is_empty() {
1922            continue;
1923        }
1924        let nbytes = saturating_byte_len(box_count, element_size) as usize;
1925        let mut buf = alloc_tiled_fill(nbytes, None)?;
1926        read_source_box(box_start, box_count, &mut buf)?;
1927        for &(from, to, len) in segments {
1928            let (from, to, len) = (
1929                (from * element_size) as usize,
1930                (to * element_size) as usize,
1931                (len * element_size) as usize,
1932            );
1933            let src = buf.get(from..from + len).ok_or_else(|| {
1934                crate::io::IoError::InvalidState(
1935                    "virtual dataset mapping's source selection reaches past its source box".into(),
1936                )
1937            })?;
1938            let dst = out.get_mut(to..to + len).ok_or_else(|| {
1939                crate::io::IoError::InvalidState(
1940                    "virtual dataset mapping's virtual selection reaches past the dataset".into(),
1941                )
1942            })?;
1943            dst.copy_from_slice(src);
1944        }
1945    }
1946    Ok(())
1947}
1948
1949/// Read and decode the global-heap collection at `addr`, applying the
1950/// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
1951/// signature must be present and the declared size at least `H5HG_MINSIZE`
1952/// (4096 bytes). There is no upper size cap — libhdf5 has none, and this
1953/// crate's writers put a whole write call's strings into one collection,
1954/// which a cap would turn into silent data loss.
1955///
1956/// A free function (not a method) so both [`Hdf5Reader::read_heap_collection`]
1957/// and the static dataset-open path (which only has a `&mut FileHandle`, not
1958/// a full `&mut Hdf5Reader`) share the one implementation.
1959fn read_heap_collection_from(
1960    handle: &mut FileHandle,
1961    ctx: &FormatContext,
1962    addr: u64,
1963) -> IoResult<GlobalHeapCollection> {
1964    let ss = ctx.sizeof_size as usize;
1965    let header_len = 4 + 1 + 3 + ss;
1966    let header_buf = handle.read_at_most(addr, header_len)?;
1967    if header_buf.len() < header_len || header_buf[0..4] != *b"GCOL" {
1968        return Err(crate::io::IoError::InvalidState(format!(
1969            "bad global heap collection signature at address {addr:#x}"
1970        )));
1971    }
1972    let declared = read_uint(&header_buf[8..], ss) as usize;
1973    if declared < 4096 {
1974        return Err(crate::io::IoError::InvalidState(format!(
1975            "global heap collection at address {addr:#x} declares size {declared}, \
1976             below the 4096-byte minimum"
1977        )));
1978    }
1979    let heap_buf = handle.read_at(addr, declared)?;
1980    let (coll, _) = GlobalHeapCollection::decode(&heap_buf, ctx)?;
1981    Ok(coll)
1982}
1983
1984/// One chunk's on-disk read request, built by a read path before any I/O.
1985struct ChunkReadJob {
1986    /// Byte offset of the chunk's stored bytes.
1987    addr: u64,
1988    /// Number of bytes to read.
1989    len: usize,
1990    /// `true` → [`FileHandle::read_at_most`] (short reads near EOF are fine);
1991    /// `false` → [`FileHandle::read_at`] (exact, errors on a short read).
1992    at_most: bool,
1993    /// Per-chunk filter mask (ignored when the pipeline is `None`).
1994    mask: u32,
1995}
1996
1997/// Read one chunk's raw bytes according to its job.
1998fn read_chunk_raw(handle: &FileHandle, j: &ChunkReadJob) -> IoResult<Vec<u8>> {
1999    if j.at_most {
2000        Ok(handle.read_at_most(j.addr, j.len)?)
2001    } else {
2002        Ok(handle.read_at(j.addr, j.len)?)
2003    }
2004}
2005
2006/// Memory traffic one positioned read is worth: a `pread` costs about what
2007/// copying this many bytes costs. It is the exchange rate
2008/// [`read_chunk_runs_into`] weighs when a chunk's selected runs could each be
2009/// read on their own instead of reading the chunk whole and copying them out.
2010const PREAD_COST_BYTES: u64 = 16 * 1024;
2011
2012/// One chunk's intersection with the read box, in global dataset indices.
2013///
2014/// The single derivation of chunk ∩ selection ∩ extent:
2015/// [`for_each_chunk_run`] decomposes this box into contiguous runs and
2016/// [`ChunkOverlap::bytes`] measures it, so what a plan is measured to cover
2017/// and what the same plan places can never disagree.
2018struct ChunkOverlap {
2019    lo: Vec<u64>,
2020    hi: Vec<u64>,
2021}
2022
2023impl ChunkOverlap {
2024    /// `None` when the chunk and the read box do not meet. A corrupt index can
2025    /// place a chunk past the extent, so the arithmetic saturates and an empty
2026    /// span simply yields `None`.
2027    fn of(place: &ChunkPlacement, chunk_coords: &[u64]) -> Option<Self> {
2028        let ChunkPlacement {
2029            geo: ChunkOutputGeometry {
2030                dims, chunk_dims, ..
2031            },
2032            starts,
2033            counts,
2034        } = *place;
2035        let ndims = dims.len();
2036        if ndims == 0 {
2037            return None;
2038        }
2039        let mut lo = vec![0u64; ndims];
2040        let mut hi = vec![0u64; ndims];
2041        for d in 0..ndims {
2042            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
2043            let chunk_end = origin.saturating_add(chunk_dims[d]).min(dims[d]);
2044            let sel_end = starts[d].saturating_add(counts[d]);
2045            lo[d] = origin.max(starts[d]);
2046            hi[d] = chunk_end.min(sel_end);
2047            if lo[d] >= hi[d] {
2048                return None;
2049            }
2050        }
2051        Some(Self { lo, hi })
2052    }
2053
2054    /// Output bytes this overlap covers — exactly the summed length of the
2055    /// runs [`for_each_chunk_run`] yields for the same chunk.
2056    fn bytes(&self, element_size: u64) -> u64 {
2057        self.lo
2058            .iter()
2059            .zip(&self.hi)
2060            .map(|(&l, &h)| h - l)
2061            .product::<u64>()
2062            .saturating_mul(element_size)
2063    }
2064}
2065
2066/// Output bytes a chunk plan covers, counting each chunk-grid slot once.
2067///
2068/// Distinct slots own disjoint boxes of the index space and one chunk's runs
2069/// are disjoint from each other, so this sum is the exact volume of their
2070/// union — provided a slot a corrupt index names twice is counted once, which
2071/// is what `seen` is for. The caller compares it against the output length:
2072/// anything short means some output byte belongs to a chunk this plan will not
2073/// place, and only then does the fill value have to go down first.
2074fn planned_coverage(place: &ChunkPlacement, jobs: &[Option<ChunkReadJob>], coords: &[u64]) -> u64 {
2075    let rank = place.geo.dims.len();
2076    if rank == 0 {
2077        return 0;
2078    }
2079    let mut seen: std::collections::HashSet<&[u64]> = std::collections::HashSet::new();
2080    let mut covered = 0u64;
2081    for (i, job) in jobs.iter().enumerate() {
2082        if job.is_none() {
2083            continue;
2084        }
2085        let c = &coords[i * rank..(i + 1) * rank];
2086        if !seen.insert(c) {
2087            continue;
2088        }
2089        if let Some(overlap) = ChunkOverlap::of(place, c) {
2090            covered = covered.saturating_add(overlap.bytes(place.geo.element_size));
2091        }
2092    }
2093    covered
2094}
2095
2096/// Walk the contiguous byte runs of one chunk's intersection with the read box
2097/// `[starts, starts + counts)`, as `(offset in the chunk image, offset in the
2098/// output, run length in bytes)`.
2099///
2100/// The last axis is innermost in both the chunk and the output, so the
2101/// intersection box decomposes into one contiguous run per setting of the
2102/// outer axes — no per-element loop. The single owner of chunk placement
2103/// geometry: [`copy_chunk_runs`] copies these runs out of a decoded chunk
2104/// image and [`read_chunk_runs_into`] reads the same runs straight out of the
2105/// file, so the two cannot place a byte differently. A full read passes the
2106/// whole extent as the box, which is why it needs no placement path of its
2107/// own.
2108fn for_each_chunk_run(
2109    place: &ChunkPlacement,
2110    chunk_coords: &[u64],
2111    mut f: impl FnMut(u64, u64, usize),
2112) {
2113    let ChunkPlacement {
2114        geo:
2115            ChunkOutputGeometry {
2116                dims,
2117                chunk_dims,
2118                element_size,
2119            },
2120        starts,
2121        counts,
2122    } = *place;
2123    let ndims = dims.len();
2124    // Global intersection box [lo, hi) of chunk ∩ selection ∩ dataset; `None`
2125    // covers the rank-0 case the indexing below could not survive.
2126    let Some(ChunkOverlap { lo, hi }) = ChunkOverlap::of(place, chunk_coords) else {
2127        return;
2128    };
2129
2130    let chunk_strides = compute_strides(chunk_dims, element_size);
2131    let out_strides = compute_strides(counts, element_size);
2132    let last = ndims - 1;
2133    let run_bytes = ((hi[last] - lo[last]) * element_size) as usize;
2134
2135    // Iterate the outer box dimensions [0, last); each position places one
2136    // contiguous last-axis run.
2137    let outer_extent: Vec<u64> = (0..last).map(|d| hi[d] - lo[d]).collect();
2138    let n_outer: u64 = outer_extent.iter().product(); // empty product == 1
2139    let mut oc = vec![0u64; last];
2140    for _ in 0..n_outer {
2141        let mut src_off = 0u64;
2142        let mut dst_off = 0u64;
2143        for d in 0..ndims {
2144            let g = if d < last { lo[d] + oc[d] } else { lo[last] };
2145            let origin = chunk_coords[d].saturating_mul(chunk_dims[d]);
2146            src_off += (g - origin) * chunk_strides[d];
2147            dst_off += (g - starts[d]) * out_strides[d];
2148        }
2149        f(src_off, dst_off, run_bytes);
2150        for d in (0..last).rev() {
2151            oc[d] += 1;
2152            if oc[d] < outer_extent[d] {
2153                break;
2154            }
2155            oc[d] = 0;
2156        }
2157    }
2158}
2159
2160/// Copy one decoded chunk image's intersection with the read box into
2161/// `output`.
2162///
2163/// A run reaching past the image — a chunk read short at the end of the file —
2164/// or past the output cannot be copied; it is appended to `skipped` as the
2165/// output range `(offset, length)` nothing wrote, which is what
2166/// [`place_chunk_jobs`] fills with the fill value.
2167fn copy_chunk_runs(
2168    chunk_data: &[u8],
2169    output: &mut [u8],
2170    place: &ChunkPlacement,
2171    chunk_coords: &[u64],
2172    skipped: &mut Vec<(usize, usize)>,
2173) {
2174    for_each_chunk_run(place, chunk_coords, |src, dst, len| {
2175        let (s, d) = (src as usize, dst as usize);
2176        if s + len <= chunk_data.len() && d + len <= output.len() {
2177            output[d..d + len].copy_from_slice(&chunk_data[s..s + len]);
2178        } else {
2179            push_skipped(skipped, output.len(), d, len);
2180        }
2181    });
2182}
2183
2184/// Record the output range `[dst, dst + len)` as one nothing wrote, clamped to
2185/// the output so a range a corrupt geometry pushed past the end never widens
2186/// into a slice the fill would panic on.
2187fn push_skipped(skipped: &mut Vec<(usize, usize)>, out_len: usize, dst: usize, len: usize) {
2188    let start = dst.min(out_len);
2189    let end = dst.saturating_add(len).min(out_len);
2190    if start < end {
2191        skipped.push((start, end - start));
2192    }
2193}
2194
2195/// Read one unfiltered chunk's intersection with the read box straight from
2196/// the file into `output`, never materializing the chunk.
2197///
2198/// An unfiltered chunk's stored bytes are the dataset's bytes in the dataset's
2199/// own order, so every run of the intersection is one positioned read — the
2200/// byte ranges libhdf5 reads for the same selection instead of the whole chunk
2201/// (`H5D__chunk_read`, H5Dchunk.c). Adjacent runs coalesce into one read.
2202/// `image_len` is how many bytes of the chunk the whole-chunk read would have
2203/// held, so the runs placed here are exactly the runs
2204/// [`copy_chunk_runs`] would have placed out of that image.
2205///
2206/// Returns whether the chunk's bytes landed in `output`; `false` means the
2207/// caller must still read the chunk whole, and the `skipped` ranges this call
2208/// appended are the caller's to discard (the whole-chunk path records its
2209/// own). That happens when the runs are too
2210/// small for a read each to beat one whole-chunk read plus the copy out of it,
2211/// and when a read fails — the whole-chunk path owns both the error and the
2212/// short read a truncated file gives, and re-places any run already placed
2213/// here from the same file offsets, so a fallback never leaves a half-written
2214/// output.
2215fn read_chunk_runs_into(
2216    handle: &FileHandle,
2217    addr: u64,
2218    image_len: usize,
2219    place: &ChunkPlacement,
2220    chunk_coords: &[u64],
2221    output: &mut [u8],
2222    skipped: &mut Vec<(usize, usize)>,
2223) -> bool {
2224    // (file offset, offset in output, length), coalesced as they are planned.
2225    let mut runs: Vec<(u64, usize, usize)> = Vec::new();
2226    let mut selected = 0u64;
2227    let out_len = output.len();
2228    for_each_chunk_run(place, chunk_coords, |src, dst, len| {
2229        let (s, d) = (src as usize, dst as usize);
2230        if s + len > image_len || d + len > out_len {
2231            push_skipped(skipped, out_len, d, len);
2232            return;
2233        }
2234        selected += len as u64;
2235        if let Some(last) = runs.last_mut() {
2236            if last.0 + last.2 as u64 == addr.saturating_add(src) && last.1 + last.2 == d {
2237                last.2 += len;
2238                return;
2239            }
2240        }
2241        runs.push((addr.saturating_add(src), d, len));
2242    });
2243    if runs.is_empty() {
2244        // Nothing to place, which only a chunk outside the extent or a short
2245        // image produces: leave it to the whole-chunk path so a chunk that
2246        // cannot be read at all still fails there.
2247        return false;
2248    }
2249    // One read per run pays off while the extra syscalls cost less than the
2250    // whole-chunk read and the copy out of it they replace.
2251    if (runs.len() as u64 - 1).saturating_mul(PREAD_COST_BYTES) > image_len as u64 + selected {
2252        return false;
2253    }
2254    for (offset, dst, len) in runs {
2255        if handle
2256            .read_exact_at_into(offset, &mut output[dst..dst + len])
2257            .is_err()
2258        {
2259            return false;
2260        }
2261    }
2262    true
2263}
2264
2265/// Place every chunk a chunked read planned into `output`.
2266///
2267/// The single owner of "planned chunk → output bytes" for all five chunk index
2268/// types: each read path walks its own index and builds the jobs, and this
2269/// decides, per chunk, whether the read box's byte runs come straight out of
2270/// the file ([`read_chunk_runs_into`]) or whether the chunk has to be read and
2271/// decoded whole first ([`copy_chunk_runs`]). A filtered chunk's stored bytes
2272/// have no byte-range correspondence to the dataset's, so it always reads
2273/// whole. `coords` is the packed chunk-grid position table
2274/// ([`crate::io::chunk_grid::coords_table`]): job `i` sits at rank-many values
2275/// from `i * rank`.
2276///
2277/// It is also the single owner of the fill: every byte of `output` this
2278/// returns `Ok` on is either a byte some chunk placed or a byte filled with
2279/// the tiled fill value, so callers hand it an arbitrary (even uninitialized)
2280/// buffer and a chunked read that reaches no chunk at all still routes through
2281/// here with an empty job list. The fill is derived from the plan rather than
2282/// laid down blanket-first: when [`planned_coverage`] proves the plan covers
2283/// the whole output, a 128 MiB read never pays a 128 MiB memset it is about to
2284/// overwrite, and only the runs a short chunk image left unwritten are filled
2285/// afterwards.
2286///
2287/// `cache` is the dataset's [`ChunkImageCache`] — present for every read that
2288/// reached its chunks through a decoded index. It is consulted only for a
2289/// filtered chunk this read leaves partly unconsumed: an unfiltered chunk's
2290/// selected runs come straight out of the file below and never materialize an
2291/// image there is anything to keep, and a chunk this read takes entire is one
2292/// no later read can want more of.
2293fn place_chunk_jobs(
2294    handle: &FileHandle,
2295    mut jobs: Vec<Option<ChunkReadJob>>,
2296    coords: &[u64],
2297    req: ChunkReadRequest,
2298    geo: &ChunkOutputGeometry,
2299    cache: Option<&ChunkImageCache>,
2300    output: &mut [u8],
2301) -> IoResult<()> {
2302    let ChunkReadRequest {
2303        pipeline,
2304        target,
2305        fill_value,
2306    } = req;
2307    let rank = geo.dims.len();
2308    let zeros = vec![0u64; rank];
2309    let place = ChunkPlacement::resolve(geo, target, &zeros);
2310    let at = |i: usize| &coords[i * rank..(i + 1) * rank];
2311
2312    // Fill first unless the plan already accounts for every output byte; then
2313    // only the runs placement could not honour need filling afterwards.
2314    let prefilled = planned_coverage(&place, &jobs, coords) != output.len() as u64;
2315    if prefilled {
2316        fill_tiled_into(output, fill_value);
2317    }
2318    let mut skipped: Vec<(usize, usize)> = Vec::new();
2319
2320    if pipeline.is_none() {
2321        for (i, job) in jobs.iter_mut().enumerate() {
2322            let Some(j) = job.as_ref() else { continue };
2323            let mark = skipped.len();
2324            if read_chunk_runs_into(handle, j.addr, j.len, &place, at(i), output, &mut skipped) {
2325                *job = None;
2326            } else {
2327                // The whole-chunk path re-places this chunk and records what
2328                // it could not place; a half-planned attempt must not leave a
2329                // range behind that the fill would then write over good bytes.
2330                skipped.truncate(mark);
2331            }
2332        }
2333    }
2334
2335    // A filtered chunk this read only partly consumes is worth an image: the
2336    // next slice of it pays a copy instead of a second inflate. `hits[i]` is
2337    // an image the cache already held — its job is dropped, so nothing reads
2338    // or decodes it — and `keys[i]` marks a chunk whose image this read is to
2339    // hand over once it has one.
2340    let mut hits: Vec<Option<std::sync::Arc<Vec<u8>>>> = Vec::new();
2341    let mut keys: Vec<Option<ChunkImageKey>> = Vec::new();
2342    let cache = cache.filter(|_| pipeline.is_some());
2343    if let Some(cache) = cache {
2344        hits.resize_with(jobs.len(), || None);
2345        keys.resize(jobs.len(), None);
2346        for (i, job) in jobs.iter_mut().enumerate() {
2347            let Some(j) = job.as_ref() else { continue };
2348            if !place.leaves_chunk_unconsumed(at(i)) {
2349                continue;
2350            }
2351            let key = ChunkImageKey {
2352                addr: j.addr,
2353                len: j.len,
2354                mask: j.mask,
2355            };
2356            match cache.get(&key) {
2357                Some(image) => {
2358                    hits[i] = Some(image);
2359                    *job = None;
2360                }
2361                None => keys[i] = Some(key),
2362            }
2363        }
2364        for (i, image) in hits.iter().enumerate() {
2365            if let Some(image) = image {
2366                copy_chunk_runs(image, output, &place, at(i), &mut skipped);
2367            }
2368        }
2369    }
2370
2371    if jobs.iter().any(Option::is_some) {
2372        // Every chunk whose image is a contiguous stretch of `output` decodes
2373        // into that stretch; the rest come back as images to scatter. The
2374        // borrow of `output` the sinks hold ends with the call, which returns
2375        // nothing that points into it.
2376        let decoded = {
2377            let sinks = carve_sinks(output, &jobs, coords, &place);
2378            read_and_decompress_chunks(handle, pipeline, jobs, sinks, geo.image_bytes())?
2379        };
2380        for (i, chunk) in decoded.into_iter().enumerate() {
2381            match chunk {
2382                ChunkDecoded::Absent => {}
2383                // A chunk marked for the cache is by construction a staged
2384                // one: a chunk whose image is a contiguous stretch of the
2385                // output is a chunk this read consumes entire, which
2386                // `ChunkPlacement::leaves_chunk_unconsumed` refuses.
2387                ChunkDecoded::Image(data) => {
2388                    match keys.get_mut(i).and_then(Option::take).zip(cache) {
2389                        Some((key, cache)) => {
2390                            let image = cache.keep(key, data);
2391                            copy_chunk_runs(&image, output, &place, at(i), &mut skipped)
2392                        }
2393                        None => copy_chunk_runs(&data, output, &place, at(i), &mut skipped),
2394                    }
2395                }
2396                // A chunk that decoded short of its image placed no usable run
2397                // — the same verdict `copy_chunk_runs` passes on a run reaching
2398                // past a short image — so the whole stretch reverts to fill,
2399                // not just the part past the image: a decode writes its output
2400                // in blocks and may have reached past the byte it stopped on.
2401                ChunkDecoded::InPlace { dst, len, bytes } if bytes < len => {
2402                    fill_tiled_into(&mut output[dst..dst + len], fill_value)
2403                }
2404                ChunkDecoded::InPlace { .. } => {}
2405            }
2406        }
2407    }
2408    if !prefilled {
2409        for (dst, len) in skipped {
2410            fill_tiled_into(&mut output[dst..dst + len], fill_value);
2411        }
2412    }
2413    Ok(())
2414}
2415
2416/// Where one planned chunk's decoded image lands.
2417enum ChunkSink<'a> {
2418    /// The chunk's whole image is this stretch of the read's output, starting
2419    /// at output offset `dst`: the decoder writes it in place and nothing is
2420    /// copied afterwards.
2421    Direct { dst: usize, out: &'a mut [u8] },
2422    /// The image has no contiguous home in the output; it is materialized and
2423    /// then placed run by run.
2424    Staged,
2425}
2426
2427/// What one planned chunk left for the placement step.
2428enum ChunkDecoded {
2429    /// The slot planned no chunk (`jobs[i]` was `None`).
2430    Absent,
2431    /// The image is here and still has to be placed run by run.
2432    Image(Vec<u8>),
2433    /// The image went straight into `output[dst..dst + len]`. `bytes` is the
2434    /// length the pipeline produced there: short of `len` only for a stored
2435    /// chunk that decoded to less than its image.
2436    InPlace {
2437        dst: usize,
2438        len: usize,
2439        bytes: usize,
2440    },
2441}
2442
2443/// Hand every chunk whose image is one contiguous stretch of `output` that
2444/// stretch to decode into, and stage every other chunk.
2445///
2446/// A chunk's image *is* the output's own bytes exactly when its intersection
2447/// with the read box is a single run that starts at the image's first byte and
2448/// carries the image's whole length — the whole chunk, laid down contiguously.
2449/// The runs come from [`for_each_chunk_run`], the same walk that would have
2450/// copied the image out, so a stretch handed out here is byte-for-byte the
2451/// stretch the copy would have written.
2452///
2453/// Distinct chunk-grid slots own disjoint boxes of the output, so their
2454/// stretches never overlap; a corrupt index naming one slot twice would break
2455/// that, so a stretch overlapping one already handed out is staged instead.
2456fn carve_sinks<'a>(
2457    output: &'a mut [u8],
2458    jobs: &[Option<ChunkReadJob>],
2459    coords: &[u64],
2460    place: &ChunkPlacement,
2461) -> Vec<ChunkSink<'a>> {
2462    let mut sinks = Vec::with_capacity(jobs.len());
2463    sinks.resize_with(jobs.len(), || ChunkSink::Staged);
2464    let rank = place.geo.dims.len();
2465    let out_len = output.len();
2466    if rank == 0 {
2467        return sinks;
2468    }
2469    let Some(image_bytes) = place.geo.image_bytes() else {
2470        return sinks;
2471    };
2472    // A chunk is whole inside the read box only if the box is at least a chunk
2473    // wide in every dimension: a narrower selection has no direct chunk at all,
2474    // and testing it once here spares it the per-chunk walk.
2475    if place
2476        .counts
2477        .iter()
2478        .zip(place.geo.chunk_dims)
2479        .any(|(c, k)| c < k)
2480    {
2481        return sinks;
2482    }
2483
2484    let mut wanted: Vec<(usize, usize)> = Vec::new();
2485    for (i, job) in jobs.iter().enumerate() {
2486        if job.is_none() {
2487            continue;
2488        }
2489        let mut runs = 0usize;
2490        let mut first = (0u64, 0u64, 0usize);
2491        for_each_chunk_run(place, &coords[i * rank..(i + 1) * rank], |src, dst, len| {
2492            if runs == 0 {
2493                first = (src, dst, len);
2494            }
2495            runs += 1;
2496        });
2497        let (src, dst, len) = first;
2498        if runs == 1
2499            && src == 0
2500            && len as u64 == image_bytes
2501            && dst.saturating_add(len as u64) <= out_len as u64
2502        {
2503            wanted.push((dst as usize, i));
2504        }
2505    }
2506    wanted.sort_unstable();
2507
2508    let len = image_bytes as usize;
2509    let mut rest: &'a mut [u8] = output;
2510    let mut base = 0usize;
2511    for (dst, i) in wanted {
2512        if dst < base {
2513            continue;
2514        }
2515        let (_, tail) = std::mem::take(&mut rest).split_at_mut(dst - base);
2516        let (mine, tail) = tail.split_at_mut(len);
2517        sinks[i] = ChunkSink::Direct { dst, out: mine };
2518        rest = tail;
2519        base = dst + len;
2520    }
2521    sinks
2522}
2523
2524/// Run the reverse filter pipeline (if any) over one chunk's raw bytes,
2525/// straight into `out`, returning the length of the image it produced.
2526///
2527/// The counterpart of [`decompress_chunk`] for a chunk whose image is already
2528/// the output's own bytes. A length below `out.len()` means the stored chunk
2529/// decoded short; a length above it means the surplus was discarded, which is
2530/// what copying `out.len()` bytes out of a materialized image does too.
2531fn decompress_chunk_into(
2532    pipeline: Option<&FilterPipeline>,
2533    raw: &[u8],
2534    mask: u32,
2535    out: &mut [u8],
2536) -> IoResult<usize> {
2537    match pipeline {
2538        Some(pl) => Ok(filter::reverse_filters_masked_into(pl, raw, mask, out)?),
2539        None => {
2540            let n = raw.len().min(out.len());
2541            out[..n].copy_from_slice(&raw[..n]);
2542            Ok(raw.len())
2543        }
2544    }
2545}
2546
2547/// Run the reverse filter pipeline (if any) over one chunk's raw bytes into a
2548/// fresh image, for a chunk whose bytes have no contiguous home in the output.
2549///
2550/// `image_bytes` is what the layout says the chunk decodes to
2551/// ([`ChunkOutputGeometry::image_bytes`]), so the image is allocated once at
2552/// its real size instead of being grown to it — a filter that has to discover
2553/// the size re-enters its decoder once per doubling. Bytes past what the
2554/// pipeline produced are cut off, so a chunk that decoded short is as short
2555/// here as the growing spelling left it and [`copy_chunk_runs`] passes the
2556/// same verdict on the runs reaching past it. Bytes past `image_bytes` are cut
2557/// off too: no run of a chunk reaches past its own image.
2558fn decompress_chunk(
2559    pipeline: Option<&FilterPipeline>,
2560    raw: Vec<u8>,
2561    mask: u32,
2562    image_bytes: Option<u64>,
2563) -> IoResult<Vec<u8>> {
2564    let Some(pl) = pipeline else { return Ok(raw) };
2565    let Some(image_bytes) = image_bytes.and_then(|b| usize::try_from(b).ok()) else {
2566        // Geometry a corrupt file can carry says nothing about the size; the
2567        // pipeline discovers it.
2568        return Ok(filter::reverse_filters_masked(pl, &raw, mask)?);
2569    };
2570    let mut image = vec![0u8; image_bytes];
2571    let produced = filter::reverse_filters_masked_into(pl, &raw, mask, &mut image)?;
2572    image.truncate(produced.min(image_bytes));
2573    Ok(image)
2574}
2575
2576/// Planned chunk bytes a batch must carry before the rayon pool earns its
2577/// entry cost.
2578///
2579/// `ThreadPool::install` injects a job and blocks on a latch; with the workers
2580/// parked that costs tens of microseconds, which is more than a small
2581/// selection's entire read. One 64 KiB slice of a chunked dataset plans one or
2582/// two chunks, so it decodes on the thread that planned it and never pays the
2583/// dispatch.
2584#[cfg(feature = "parallel")]
2585const PARALLEL_MIN_JOB_BYTES: u64 = 256 * 1024;
2586
2587/// Whether a batch is worth handing to the pool: more than one chunk to read,
2588/// and enough bytes behind them to repay [`PARALLEL_MIN_JOB_BYTES`]. Skipped
2589/// chunks (`None`) count for nothing — a slice whose chunks were all placed by
2590/// [`read_chunk_runs_into`] leaves no work at all.
2591#[cfg(feature = "parallel")]
2592fn worth_parallel(jobs: &[Option<ChunkReadJob>]) -> bool {
2593    let mut n = 0usize;
2594    let mut bytes = 0u64;
2595    for j in jobs.iter().flatten() {
2596        n += 1;
2597        bytes = bytes.saturating_add(j.len as u64);
2598        if n > 1 && bytes >= PARALLEL_MIN_JOB_BYTES {
2599            return true;
2600        }
2601    }
2602    false
2603}
2604
2605/// Read and decompress a batch of chunk jobs, preserving job order.
2606///
2607/// `jobs[i] == None` yields `Ok(None)` — a chunk skipped as out-of-selection
2608/// or unallocated. Otherwise the chunk's raw bytes are read and, when
2609/// `pipeline` is `Some`, run through the reverse filter pipeline with the
2610/// job's mask. This is the single owner of the read-then-decompress step for
2611/// every chunk index type; each read path only builds the jobs and scatters
2612/// the results.
2613///
2614/// On Unix and Windows, positioned reads at distinct offsets on a shared
2615/// `&File` each carry their own explicit offset and never consult a shared file
2616/// cursor (on Windows the cursor may move as a side effect, but nothing reads
2617/// it), so read + decompress run fused in one parallel pass — overlapping chunk
2618/// I/O across cores, which the
2619/// C library's default (non-MPI) path does not do. On targets with neither
2620/// positioned API the seek-based fallback shares the file cursor, so reads
2621/// stay serial there while decompression still parallelizes.
2622fn read_and_decompress_chunks(
2623    handle: &FileHandle,
2624    pipeline: Option<&FilterPipeline>,
2625    jobs: Vec<Option<ChunkReadJob>>,
2626    sinks: Vec<ChunkSink<'_>>,
2627    image_bytes: Option<u64>,
2628) -> IoResult<Vec<ChunkDecoded>> {
2629    // Decompress one chunk's raw bytes into whatever its sink says.
2630    let deliver = |raw: Vec<u8>, mask: u32, sink: ChunkSink<'_>| -> IoResult<ChunkDecoded> {
2631        match sink {
2632            ChunkSink::Direct { dst, out } => {
2633                let len = out.len();
2634                let bytes = decompress_chunk_into(pipeline, &raw, mask, out)?;
2635                Ok(ChunkDecoded::InPlace { dst, len, bytes })
2636            }
2637            ChunkSink::Staged => Ok(ChunkDecoded::Image(decompress_chunk(
2638                pipeline,
2639                raw,
2640                mask,
2641                image_bytes,
2642            )?)),
2643        }
2644    };
2645    #[cfg(all(feature = "parallel", any(unix, windows)))]
2646    {
2647        use rayon::prelude::*;
2648        // Fused read + decompress for one job.
2649        let decode =
2650            |(job, sink): (Option<ChunkReadJob>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
2651                match job {
2652                    Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
2653                    None => Ok(ChunkDecoded::Absent),
2654                }
2655            };
2656        // Run on rust-hdf5's private half-cores pool, not rayon's global pool;
2657        // fall back to serial if the pool could not be built, and for a batch
2658        // too small to repay entering it.
2659        let pool = crate::parallel::io_pool().filter(|_| worth_parallel(&jobs));
2660        let work: Vec<_> = jobs.into_iter().zip(sinks).collect();
2661        match pool {
2662            Some(pool) => pool.install(|| {
2663                work.into_par_iter()
2664                    .map(&decode)
2665                    .collect::<IoResult<Vec<_>>>()
2666            }),
2667            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
2668        }
2669    }
2670    #[cfg(all(feature = "parallel", not(any(unix, windows))))]
2671    {
2672        use rayon::prelude::*;
2673        // No positioned read API here: concurrent reads would race the shared
2674        // file cursor, so read serially, then parallelize decompression on
2675        // rust-hdf5's private half-cores pool (not rayon's global pool).
2676        let parallel = worth_parallel(&jobs);
2677        let raws: Vec<Option<(Vec<u8>, u32)>> = jobs
2678            .into_iter()
2679            .map(|job| match job {
2680                Some(j) => Ok(Some((read_chunk_raw(handle, &j)?, j.mask))),
2681                None => Ok(None),
2682            })
2683            .collect::<IoResult<Vec<_>>>()?;
2684        let decode =
2685            |(r, sink): (Option<(Vec<u8>, u32)>, ChunkSink<'_>)| -> IoResult<ChunkDecoded> {
2686                match r {
2687                    Some((raw, mask)) => deliver(raw, mask, sink),
2688                    None => Ok(ChunkDecoded::Absent),
2689                }
2690            };
2691        let work: Vec<_> = raws.into_iter().zip(sinks).collect();
2692        // Fall back to serial if the private pool could not be built, and for
2693        // a batch too small to repay entering it.
2694        match crate::parallel::io_pool().filter(|_| parallel) {
2695            Some(pool) => pool.install(|| {
2696                work.into_par_iter()
2697                    .map(&decode)
2698                    .collect::<IoResult<Vec<_>>>()
2699            }),
2700            None => work.into_iter().map(decode).collect::<IoResult<Vec<_>>>(),
2701        }
2702    }
2703    #[cfg(not(feature = "parallel"))]
2704    {
2705        jobs.into_iter()
2706            .zip(sinks)
2707            .map(|(job, sink)| match job {
2708                Some(j) => deliver(read_chunk_raw(handle, &j)?, j.mask, sink),
2709                None => Ok(ChunkDecoded::Absent),
2710            })
2711            .collect()
2712    }
2713}
2714
2715impl Hdf5Reader {
2716    /// Open an existing HDF5 file in SWMR read mode using the env-var-derived
2717    /// locking policy.
2718    ///
2719    /// Currently identical to `open()`, but indicates intent to use
2720    /// `refresh()` for re-reading metadata written by a concurrent SWMR writer.
2721    pub fn open_swmr(path: &Path) -> IoResult<Self> {
2722        Self::open(path)
2723    }
2724
2725    /// Open an existing HDF5 file in SWMR read mode with an explicit locking
2726    /// policy.
2727    pub fn open_swmr_with_locking(
2728        path: &Path,
2729        locking: crate::io::locking::FileLocking,
2730    ) -> IoResult<Self> {
2731        Self::open_with_locking(path, locking)
2732    }
2733
2734    /// Open an existing HDF5 file for reading using the env-var-derived
2735    /// locking policy.
2736    ///
2737    /// Auto-detects the superblock version and uses the appropriate code path:
2738    /// - v0/v1: legacy format with symbol tables and B-tree v1
2739    /// - v2/v3: modern format with link messages
2740    pub fn open(path: &Path) -> IoResult<Self> {
2741        Self::open_with_locking(
2742            path,
2743            crate::io::locking::FileLocking::from_env_or(Default::default()),
2744        )
2745    }
2746
2747    /// Open an existing HDF5 file for reading with an explicit locking policy.
2748    pub fn open_with_locking(
2749        path: &Path,
2750        locking: crate::io::locking::FileLocking,
2751    ) -> IoResult<Self> {
2752        let mut handle = FileHandle::open_read_with_locking(path, locking)?;
2753
2754        // The superblock is not necessarily at the start of the file: a
2755        // userblock precedes it, and `H5FD_locate_signature` finds it by
2756        // probing offset 0 and then every power of two from 512 up. The offset
2757        // it is found at is where HDF5 addresses are measured from, so it
2758        // becomes the handle's base address and every later offset — including
2759        // the superblock read just below — is relative to it.
2760        let super_addr = handle
2761            .locate_signature()?
2762            .ok_or(crate::format::FormatError::InvalidSignature)?;
2763        handle.set_base(super_addr);
2764
2765        // Read enough bytes to detect the superblock version and parse it.
2766        let sb_buf = handle.read_at_most(0, 1024)?;
2767        let version = detect_superblock_version(&sb_buf)?;
2768
2769        let origin = Origin {
2770            path: path.to_path_buf(),
2771            locking,
2772        };
2773        let mut reader = match version {
2774            0 | 1 => Self::open_v0v1(handle, &sb_buf, origin)?,
2775            2 | 3 => Self::open_v2v3(handle, &sb_buf, origin)?,
2776            v => {
2777                return Err(crate::io::IoError::Format(
2778                    crate::format::FormatError::InvalidVersion(v),
2779                ))
2780            }
2781        };
2782        // Resolved from the path this file was opened with (not the
2783        // process's current directory at read time) — see `source_dir`.
2784        let canonical = std::fs::canonicalize(path)?;
2785        reader.source_dir = canonical
2786            .parent()
2787            .map(Path::to_path_buf)
2788            .unwrap_or_default();
2789        // Only now, with `source_dir` set, can a source name be resolved —
2790        // and a virtual dataset's extent is not final until they are.
2791        VirtualResolveDepth::enter(|| reader.resolve_virtual_extents())?;
2792        Ok(reader)
2793    }
2794
2795    /// Open a file with v2/v3 superblock (existing code path).
2796    fn open_v2v3(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
2797        let sb = SuperblockV2V3::decode(sb_buf)?;
2798
2799        let ctx = FormatContext {
2800            sizeof_addr: sb.sizeof_offsets,
2801            sizeof_size: sb.sizeof_lengths,
2802        };
2803
2804        // A v2/v3 superblock has no room for the B-tree K values, so they are
2805        // the library defaults unless the extension carries the K message.
2806        let (meta, ext) = Self::read_extension_and_meta(
2807            &mut handle,
2808            ctx,
2809            BTreeV1Config::default(),
2810            sb.superblock_extension_address,
2811        )?;
2812
2813        // Read root group object header, following continuation blocks.
2814        let root_header =
2815            Self::read_object_header_full(&mut handle, &meta, sb.root_group_object_header_address)?;
2816
2817        // Walk the root group to discover datasets, group attributes, and
2818        // every group path that exists, from whichever storage each group's
2819        // own header declares.
2820        let catalog = Self::build_catalog(
2821            &mut handle,
2822            &meta,
2823            Some(&root_header),
2824            sb.root_group_object_header_address,
2825            None,
2826        )?;
2827
2828        // Collect root group attributes
2829        let root_attributes = collect_object_attributes(&mut handle, &ctx, &root_header);
2830        // A v2/v3 root group is always addressed directly by the superblock,
2831        // never through a symbol-table scratch-pad.
2832        let root_link_storage = describe_link_storage(Some(&root_header), &ctx, None);
2833
2834        Ok(Self {
2835            handle,
2836            meta,
2837            ext,
2838            _eof: sb.end_of_file_address,
2839            superblock_version: sb.version,
2840            object_paths: catalog.object_paths(sb.root_group_object_header_address),
2841            datasets: DatasetTable::new(catalog.datasets),
2842            unreadable: catalog.unreadable,
2843            root_attributes,
2844            root_link_storage,
2845            group_attributes: catalog.group_attributes,
2846            group_link_storage: catalog.group_link_storage,
2847            group_paths: catalog.group_paths,
2848            group_aliases: catalog.group_aliases,
2849            links: catalog.links,
2850            datatypes: catalog.datatypes,
2851            path: origin.path,
2852            locking: origin.locking,
2853            elink_prefix: None,
2854            external: Default::default(),
2855            external_resolved: Default::default(),
2856            dataset_access: Default::default(),
2857            vds_resolved: Default::default(),
2858            // Overwritten by `open_with_locking` once this returns.
2859            source_dir: PathBuf::new(),
2860        })
2861    }
2862
2863    /// Open a file with v0/v1 superblock (legacy format).
2864    fn open_v0v1(mut handle: FileHandle, sb_buf: &[u8], origin: Origin) -> IoResult<Self> {
2865        let sb = SuperblockV0V1::decode(sb_buf)?;
2866
2867        let ctx = FormatContext {
2868            sizeof_addr: sb.sizeof_offsets,
2869            sizeof_size: sb.sizeof_lengths,
2870        };
2871
2872        // A v0/v1 superblock carries the K values itself; a v0 superblock has
2873        // no chunk-tree field, so that one keeps the library default. The
2874        // extension's K message, when present, overrides all three.
2875        let sb_btree = BTreeV1Config {
2876            sym_leaf_k: sb.sym_leaf_k,
2877            snode_internal_k: sb.btree_internal_k,
2878            chunk_internal_k: sb
2879                .indexed_storage_k
2880                .unwrap_or(BTreeV1Config::default().chunk_internal_k),
2881        };
2882        let (meta, ext) = Self::read_extension_and_meta(
2883            &mut handle,
2884            ctx,
2885            sb_btree,
2886            sb.superblock_extension_address,
2887        )?;
2888
2889        let ste = &sb.root_symbol_table_entry;
2890        let root_obj_addr = ste.obj_header_addr;
2891        let ste_stab = ste.cached_symbol_table();
2892
2893        // Read the root group's object header (following continuations).
2894        let root_hdr = Self::read_object_header_full(&mut handle, &meta, root_obj_addr).ok();
2895
2896        // Collect the root group's own attributes.
2897        let root_attributes = match root_hdr {
2898            Some(ref h) => collect_object_attributes(&mut handle, &ctx, h),
2899            None => ObjectAttributes::default(),
2900        };
2901        // The root group's own link storage and link creation-order, from
2902        // the same header-first/scratch-pad-fallback rule the walk below
2903        // uses to choose how to enumerate it.
2904        let root_link_storage = describe_link_storage(root_hdr.as_ref(), &ctx, ste_stab);
2905
2906        // A v0/v1-superblock file whose root group has migrated to link
2907        // storage (more than ~8 objects, or one link the old format cannot
2908        // express) carries `Link` / `Link Info` messages in its object
2909        // header, and the superblock symbol-table scratch-pad is then stale.
2910        // The walk picks the storage from the header for that reason, taking
2911        // the scratch-pad only as the symbol-table addresses — and only for a
2912        // symbol-table root group (`H5G_CACHED_STAB`), which is the one cache
2913        // type those two addresses mean anything for.
2914        let catalog = Self::build_catalog(
2915            &mut handle,
2916            &meta,
2917            root_hdr.as_ref(),
2918            root_obj_addr,
2919            ste_stab,
2920        )?;
2921
2922        Ok(Self {
2923            handle,
2924            meta,
2925            ext,
2926            _eof: sb.end_of_file_address,
2927            superblock_version: sb.version,
2928            object_paths: catalog.object_paths(root_obj_addr),
2929            datasets: DatasetTable::new(catalog.datasets),
2930            unreadable: catalog.unreadable,
2931            root_attributes,
2932            root_link_storage,
2933            group_attributes: catalog.group_attributes,
2934            group_link_storage: catalog.group_link_storage,
2935            group_paths: catalog.group_paths,
2936            group_aliases: catalog.group_aliases,
2937            links: catalog.links,
2938            datatypes: catalog.datatypes,
2939            path: origin.path,
2940            locking: origin.locking,
2941            elink_prefix: None,
2942            external: Default::default(),
2943            external_resolved: Default::default(),
2944            dataset_access: Default::default(),
2945            vds_resolved: Default::default(),
2946            // Overwritten by `open_with_locking` once this returns.
2947            source_dir: PathBuf::new(),
2948        })
2949    }
2950
2951    /// Read the superblock extension object header at `ext_addr` (if any) and
2952    /// fold what it says into the file-level decode parameters.
2953    ///
2954    /// `sb_btree` is what the superblock alone implies; the extension's
2955    /// v1-B-tree-"K" message replaces all three ranks when present, exactly as
2956    /// `H5F__super_read` does after `H5O_msg_read(&ext_loc, H5O_BTREEK_ID)`.
2957    pub(crate) fn read_extension_and_meta(
2958        handle: &mut FileHandle,
2959        ctx: FormatContext,
2960        sb_btree: BTreeV1Config,
2961        ext_addr: u64,
2962    ) -> IoResult<(FileMeta, SuperblockExtension)> {
2963        let mut meta = FileMeta {
2964            ctx,
2965            btree: sb_btree,
2966            sohm: None,
2967        };
2968        let ext = Self::superblock_extension_at(handle, ctx, sb_btree, ext_addr)?;
2969        if let Some(k) = ext.btree_k {
2970            meta.btree = BTreeV1Config {
2971                sym_leaf_k: k.sym_leaf_k,
2972                snode_internal_k: k.snode_internal_k,
2973                chunk_internal_k: k.chunk_internal_k,
2974            };
2975        }
2976        // A zero rank would make every v1 B-tree node zero-sized and every
2977        // symbol-table node hold no entries; libhdf5 rejects it at creation
2978        // (`H5Pset_sym_k`, `H5Pset_istore_k`), so a file carrying one is
2979        // corrupt rather than merely unusual.
2980        let b = &meta.btree;
2981        if b.sym_leaf_k == 0 || b.snode_internal_k == 0 || b.chunk_internal_k == 0 {
2982            return Err(crate::io::IoError::Format(
2983                crate::format::FormatError::InvalidData(format!(
2984                    "v1 B-tree K values must be non-zero (sym_leaf={}, snode={}, chunk={})",
2985                    b.sym_leaf_k, b.snode_internal_k, b.chunk_internal_k
2986                )),
2987            ));
2988        }
2989        // `H5F__super_read` calls `H5SM_get_info` here, so the shared-message
2990        // table is in place before the root group — the first object header
2991        // that can hold a shared message — is opened.
2992        if let Some(smt) = &ext.shared_message_table {
2993            meta.sohm = Some(Self::read_sohm_table(handle, &meta.ctx, smt)?);
2994        }
2995        Ok((meta, ext))
2996    }
2997
2998    /// The superblock extension's messages, for the address the superblock
2999    /// names. Yields the default (every field `None`) when there is no
3000    /// extension.
3001    ///
3002    /// The extension header is read with the pre-extension parameters: its own
3003    /// messages are never shared and never in a v1 B-tree, so nothing it
3004    /// contains is needed to decode it. This is also how the writer's append
3005    /// path learns what the file declares before it rewrites anything.
3006    pub(crate) fn superblock_extension_at(
3007        handle: &mut FileHandle,
3008        ctx: FormatContext,
3009        btree: BTreeV1Config,
3010        ext_addr: u64,
3011    ) -> IoResult<SuperblockExtension> {
3012        if ext_addr == UNDEF_ADDR || ext_addr == 0 {
3013            return Ok(SuperblockExtension::default());
3014        }
3015        let meta = FileMeta {
3016            ctx,
3017            btree,
3018            sohm: None,
3019        };
3020        Self::read_superblock_extension(handle, &meta, ext_addr)
3021    }
3022
3023    /// Decode the messages of the superblock extension object header.
3024    ///
3025    /// Upstream reads each of these with `H5O_msg_exists` + `H5O_msg_read` and
3026    /// fails the open when one is present but undecodable; a message this
3027    /// crate does not model is skipped, as an unknown non-critical message is
3028    /// elsewhere.
3029    fn read_superblock_extension(
3030        handle: &mut FileHandle,
3031        meta: &FileMeta,
3032        addr: u64,
3033    ) -> IoResult<SuperblockExtension> {
3034        let header = Self::read_object_header_full(handle, meta, addr)?;
3035        let ctx = &meta.ctx;
3036        let mut ext = SuperblockExtension::default();
3037        for msg in &header.messages {
3038            match msg.msg_type {
3039                MSG_SHARED_MESSAGE_TABLE => {
3040                    ext.shared_message_table =
3041                        Some(SharedMessageTableMessage::decode(&msg.data, ctx)?);
3042                }
3043                MSG_BTREE_K => ext.btree_k = Some(BtreeKMessage::decode(&msg.data)?),
3044                MSG_DRIVER_INFO => ext.driver_info = Some(DriverInfoMessage::decode(&msg.data)?),
3045                MSG_FILE_SPACE_INFO => {
3046                    ext.file_space_info = Some(FileSpaceInfoMessage::decode(&msg.data, ctx)?);
3047                }
3048                _ => {}
3049            }
3050        }
3051        Ok(ext)
3052    }
3053
3054    /// Read the SOHM master table named by the extension's shared-message
3055    /// table message.
3056    ///
3057    /// The table's length is not stored with it: the index count comes from
3058    /// the message, exactly as `H5SM__cache_table_get_final_load_size` takes it
3059    /// from `H5F_SOHM_NINDEXES`.
3060    fn read_sohm_table(
3061        handle: &mut FileHandle,
3062        ctx: &FormatContext,
3063        smt: &SharedMessageTableMessage,
3064    ) -> IoResult<SohmMasterTable> {
3065        if smt.table_address == UNDEF_ADDR || smt.nindexes == 0 {
3066            return Ok(SohmMasterTable::default());
3067        }
3068        let size = SohmMasterTable::encoded_size(ctx, smt.nindexes);
3069        let buf = handle.read_at(smt.table_address, size)?;
3070        Ok(SohmMasterTable::decode(&buf, ctx, smt.nindexes)?)
3071    }
3072
3073    /// Extract the symbol-table message (btree_addr, heap_addr) from an
3074    /// already-decoded object header.
3075    fn stab_from_header(header: &ObjectHeader, ctx: &FormatContext) -> (u64, u64) {
3076        for msg in &header.messages {
3077            if msg.msg_type == MSG_SYMBOL_TABLE {
3078                let sa = ctx.sizeof_addr as usize;
3079                if msg.data.len() >= 2 * sa {
3080                    return (read_uint(&msg.data, sa), read_uint(&msg.data[sa..], sa));
3081                }
3082            }
3083        }
3084        (UNDEF_ADDR, UNDEF_ADDR)
3085    }
3086
3087    /// Build the file catalog for a whole file, starting at its root group.
3088    ///
3089    /// `root_header` is `None` only when the root object header did not
3090    /// decode; `root_stab` then carries the superblock symbol-table entry's
3091    /// cached B-tree and local heap, which is enough to list a legacy file.
3092    fn build_catalog(
3093        handle: &mut FileHandle,
3094        meta: &FileMeta,
3095        root_header: Option<&ObjectHeader>,
3096        root_addr: u64,
3097        root_stab: Option<(u64, u64)>,
3098    ) -> IoResult<Catalog> {
3099        let mut walk = CatalogWalk::new(handle, meta, root_addr);
3100        walk.group(root_header, "", 0, root_stab)?;
3101        Ok(walk.finish())
3102    }
3103
3104    /// Read every link stored in a group's dense (fractal-heap) link storage.
3105    ///
3106    /// The `Link Info` message gives the fractal-heap address; each managed
3107    /// object in the heap is an encoded `Link` message. Returns the decoded
3108    /// links (hard and soft).
3109    pub(crate) fn read_dense_links(
3110        handle: &mut FileHandle,
3111        ctx: &FormatContext,
3112        fractal_heap_addr: u64,
3113    ) -> IoResult<Vec<LinkMessage>> {
3114        // Read the fractal heap header. Its on-disk size depends only on the
3115        // address/length widths, so a generous prefix read covers it.
3116        let hdr_buf = handle.read_at_most(fractal_heap_addr, 512)?;
3117        let fh_header = FractalHeapHeader::decode(&hdr_buf, ctx)?;
3118
3119        // Walk the heap's managed blocks; each block hands back a payload
3120        // region holding one or more packed encoded `Link` messages.
3121        let mut br = HandleBlockReader { handle };
3122        let payloads = fractal_heap::collect_managed_objects(&fh_header, ctx, &mut br)?;
3123
3124        let mut links = Vec::new();
3125        for payload in payloads {
3126            // Decode packed `Link` messages sequentially. Each decode reports
3127            // its consumed length; stop at the first byte that is not a valid
3128            // link (trailing free space or an unrelated managed object).
3129            let mut pos = 0;
3130            while pos < payload.len() {
3131                // A v1 link message starts with version byte 1.
3132                if payload[pos] != 1 {
3133                    break;
3134                }
3135                match LinkMessage::decode(&payload[pos..], ctx) {
3136                    Ok((link, consumed)) if consumed > 0 => {
3137                        links.push(link);
3138                        pos += consumed;
3139                    }
3140                    _ => break,
3141                }
3142            }
3143        }
3144
3145        // That scan stops at the first byte that does not begin a link, which
3146        // is how trailing free space in a direct block ends it — and would
3147        // equally swallow a link the scan could not read. The heap header
3148        // counts its managed objects, so a short scan is detectable, and a
3149        // group listing that is short is the loss this guards against.
3150        if links.len() < fh_header.man_nobjs as usize {
3151            return Err(crate::io::IoError::InvalidState(format!(
3152                "dense link storage at address {fractal_heap_addr:#x} holds {} managed \
3153                 objects but only {} decoded as links",
3154                fh_header.man_nobjs,
3155                links.len()
3156            )));
3157        }
3158
3159        Ok(links)
3160    }
3161
3162    /// Recursively walk a B-tree v1 to collect leaf-level SNOD addresses.
3163    fn collect_snod_addresses(
3164        handle: &mut FileHandle,
3165        meta: &FileMeta,
3166        tree_addr: u64,
3167        depth: usize,
3168        visited: &mut std::collections::HashSet<u64>,
3169    ) -> IoResult<Vec<u64>> {
3170        let sizeof_addr = meta.ctx.sizeof_addr as usize;
3171        let sizeof_size = meta.ctx.sizeof_size as usize;
3172        // A well-formed v1 B-tree's level strictly decreases with depth;
3173        // bound the descent so a corrupt/cyclic tree cannot recurse forever.
3174        // The `visited` set additionally stops a corrupt tree whose child
3175        // points back at an ancestor node from fanning out exponentially.
3176        if depth > 256 || !visited.insert(tree_addr) {
3177            return Ok(Vec::new());
3178        }
3179        // A v1 B-tree node is a fixed-size record whose length follows from
3180        // the file's K values; reading exactly that much also bounds what a
3181        // corrupt address can pull in.
3182        let node_size = meta.btree.snode_btree_node_size(sizeof_addr, sizeof_size);
3183        let buf = handle.read_at_most(tree_addr, node_size)?;
3184        let node = BTreeV1Node::decode(
3185            &buf,
3186            sizeof_addr,
3187            sizeof_size,
3188            meta.btree.snode_max_entries(),
3189        )?;
3190
3191        if node.level == 0 {
3192            // Leaf level: children are SNOD addresses
3193            Ok(node.children.clone())
3194        } else {
3195            // Internal level: children are sub-TREE addresses
3196            let mut addrs = Vec::new();
3197            for &child_addr in &node.children {
3198                let child_addrs =
3199                    Self::collect_snod_addresses(handle, meta, child_addr, depth + 1, visited)?;
3200                addrs.extend(child_addrs);
3201            }
3202            Ok(addrs)
3203        }
3204    }
3205
3206    /// Read the object header at `addr` with every continuation block
3207    /// flattened in and every stored-shared message resolved to its literal
3208    /// body. One owner for both halves of the crate — see
3209    /// [`crate::io::object_header_io`].
3210    fn read_object_header_full(
3211        handle: &mut FileHandle,
3212        meta: &FileMeta,
3213        addr: u64,
3214    ) -> IoResult<ObjectHeader> {
3215        crate::io::object_header_io::read_object_header_full(handle, meta, addr)
3216    }
3217
3218    /// Read a committed datatype object's type and attributes from its header.
3219    ///
3220    /// A committed datatype's own message holds the type itself, but the
3221    /// format does not forbid it being a reference in turn, so it goes
3222    /// through the same resolver every other datatype message does.
3223    fn committed_datatype(
3224        handle: &mut FileHandle,
3225        header: &ObjectHeader,
3226        meta: &FileMeta,
3227    ) -> CommittedDatatypeInfo {
3228        let datatype = header
3229            .messages
3230            .iter()
3231            .find(|m| m.msg_type == MSG_DATATYPE)
3232            .cloned()
3233            .ok_or_else(|| "it holds no datatype message".to_string())
3234            .and_then(|m| {
3235                crate::io::object_header_io::read_datatype_message(handle, meta, &m).map_err(|e| {
3236                    match e {
3237                        crate::io::IoError::Unsupported(why) => why,
3238                        other => format!("its datatype message does not decode: {other}"),
3239                    }
3240                })
3241            });
3242        let attributes = header
3243            .messages
3244            .iter()
3245            .filter(|m| m.msg_type == MSG_ATTRIBUTE && m.flags & MSG_FLAG_SHARED == 0)
3246            .filter_map(|m| {
3247                AttributeMessage::decode(&m.data, &meta.ctx)
3248                    .ok()
3249                    .map(|(a, _)| a)
3250            })
3251            .collect();
3252        CommittedDatatypeInfo {
3253            datatype,
3254            attributes,
3255        }
3256    }
3257
3258    /// Classify one object from its (already read) header, and decode the
3259    /// dataset metadata while doing so.
3260    ///
3261    /// The class comes from which messages are present, never from whether
3262    /// they decode: an object holding a datatype, a dataspace and a data
3263    /// layout is a dataset even when this crate cannot decode one of them,
3264    /// and it says so as [`ObjectKind::UnreadableDataset`] rather than
3265    /// vanishing.
3266    ///
3267    /// Only the messages the payload depends on can make a dataset
3268    /// unreadable. A failed *attribute* decode leaves the dataset itself
3269    /// readable, so it does not.
3270    fn classify_object(
3271        handle: &mut FileHandle,
3272        header: &ObjectHeader,
3273        meta: &FileMeta,
3274        name: &str,
3275        addr: u64,
3276    ) -> ObjectKind {
3277        let ctx = &meta.ctx;
3278        let present = |t: u8| header.messages.iter().any(|m| m.msg_type == t);
3279        let is_group = present(MSG_LINK)
3280            || present(MSG_LINK_INFO)
3281            || present(MSG_SYMBOL_TABLE)
3282            || present(MSG_GROUP_INFO);
3283        if is_group {
3284            return ObjectKind::Group;
3285        }
3286        if header_is_committed_datatype(header) {
3287            return ObjectKind::CommittedDatatype(Box::new(Self::committed_datatype(
3288                handle, header, meta,
3289            )));
3290        }
3291        let is_dataset =
3292            present(MSG_DATATYPE) && present(MSG_DATASPACE) && present(MSG_DATA_LAYOUT);
3293        if !is_dataset {
3294            return ObjectKind::Group;
3295        }
3296
3297        let mut datatype = None;
3298        let mut dataspace = None;
3299        let mut layout = None;
3300        let mut filter_pipeline = None;
3301        let mut fill_value = None;
3302        // No message at all is the library default: a fresh dataset
3303        // creation property list starts fill_defined = 1
3304        // (`FillValueMessage::default`), so a dataset that never got one
3305        // written reads back exactly as if it had.
3306        let mut fill_defined: u8 = 1;
3307        let mut fill_write_time: u8 = FILL_TIME_IFSET;
3308        let mut alloc_time: u8 = ALLOC_TIME_LATE;
3309        // The first message that did not decode, kept verbatim: it is the
3310        // answer a caller gets when it asks for this dataset.
3311        let mut blocked: Option<String> = None;
3312        let mut block = |why: String| {
3313            if blocked.is_none() {
3314                blocked = Some(why);
3315            }
3316        };
3317        let mut external_file_list = None;
3318
3319        for msg in &header.messages {
3320            // A shared message holds a reference to where its body lives, not
3321            // the body. Decoding one as a body does not fail loudly — it
3322            // reads the reference's version byte as the body's — so anything
3323            // this crate does not follow is named here instead. A datatype
3324            // reference is followed; an attribute reference is skipped, since
3325            // an attribute never blocks the dataset it hangs on.
3326            let shared = msg.flags & MSG_FLAG_SHARED != 0;
3327            if shared && !matches!(msg.msg_type, MSG_DATATYPE | MSG_ATTRIBUTE) {
3328                block(format!(
3329                    "its message of type {:#04x} is a shared-message reference, which this \
3330                     crate follows only for datatypes",
3331                    msg.msg_type
3332                ));
3333                continue;
3334            }
3335            match msg.msg_type {
3336                // The resolver already says whether the type failed to decode
3337                // or sits somewhere this crate does not follow, so its wording
3338                // is the reason rather than something to wrap.
3339                MSG_DATATYPE => {
3340                    match crate::io::object_header_io::read_datatype_message(handle, meta, msg) {
3341                        Ok(dt) => datatype = Some(dt),
3342                        Err(crate::io::IoError::Unsupported(why)) => block(why),
3343                        Err(e) => block(format!("its datatype message does not decode: {e}")),
3344                    }
3345                }
3346                MSG_DATASPACE => match DataspaceMessage::decode(&msg.data, ctx) {
3347                    Ok((ds, _)) => dataspace = Some(ds),
3348                    Err(e) => block(format!("its dataspace message does not decode: {e}")),
3349                },
3350                MSG_DATA_LAYOUT => match DataLayoutMessage::decode(&msg.data, ctx) {
3351                    Ok((dl, _)) => layout = Some(dl),
3352                    Err(e) => block(format!("its data layout message does not decode: {e}")),
3353                },
3354                // A filter pipeline that does not decode would leave the raw
3355                // chunk bytes to be handed back as if they were never
3356                // filtered, and an undecodable fill value would leave
3357                // unwritten regions reading as zeros. Both change the data a
3358                // read returns, so both block the dataset.
3359                MSG_FILTER_PIPELINE => match FilterPipeline::decode(&msg.data) {
3360                    Ok((fp, _)) => {
3361                        if !fp.filters.is_empty() {
3362                            filter_pipeline = Some(fp);
3363                        }
3364                    }
3365                    Err(e) => block(format!("its filter pipeline message does not decode: {e}")),
3366                },
3367                MSG_FILL_VALUE => match FillValueMessage::decode(&msg.data) {
3368                    Ok((fv, _)) => {
3369                        fill_defined = fv.fill_defined;
3370                        fill_write_time = fv.fill_write_time;
3371                        alloc_time = fv.alloc_time;
3372                        if fv.fill_defined == 2 {
3373                            fill_value = fv.fill_value;
3374                        }
3375                    }
3376                    Err(e) => block(format!("its fill value message does not decode: {e}")),
3377                },
3378                MSG_EXTERNAL_FILE_LIST => {
3379                    // Unlike a layout message, this one *is* the storage: a
3380                    // dataset with an external file list has no data address
3381                    // of its own (H5Dlayout.c routes storage through this
3382                    // message instead), so a list that does not decode must
3383                    // block the dataset rather than read back as zero bytes.
3384                    match ExternalFileListMessage::decode(&msg.data, ctx) {
3385                        Ok((efl, _)) => external_file_list = Some(efl),
3386                        Err(e) => block(format!(
3387                            "its external file list message does not decode: {e}"
3388                        )),
3389                    }
3390                }
3391                _ => {}
3392            }
3393        }
3394
3395        if let Some(why) = blocked {
3396            return ObjectKind::UnreadableDataset(why);
3397        }
3398        // The storage a dataset names outside its layout message. Both are
3399        // resolved before the dataset is registered and both block it when
3400        // they do not resolve, for the same reason the decode above does: a
3401        // `Virtual` or external-file layout carries no address of its own, so
3402        // a dropped mapping reads back as fill with no error at all.
3403        let external_files = match external_file_list {
3404            Some(efl) => match Self::resolve_external_file_slots(handle, ctx, &efl) {
3405                Ok(slots) => slots,
3406                Err(e) => {
3407                    return ObjectKind::UnreadableDataset(format!(
3408                        "its external file list does not resolve: {e}"
3409                    ))
3410                }
3411            },
3412            None => Vec::new(),
3413        };
3414        let virtual_mappings = match &layout {
3415            Some(DataLayoutMessage::Virtual {
3416                heap_address,
3417                heap_index,
3418                ..
3419            }) if *heap_index != 0 => {
3420                match Self::resolve_virtual_mappings(handle, ctx, *heap_address, *heap_index, name)
3421                {
3422                    Ok(list) => Some(list),
3423                    Err(e) => {
3424                        return ObjectKind::UnreadableDataset(format!(
3425                            "its virtual dataset mapping list does not resolve: {e}"
3426                        ))
3427                    }
3428                }
3429            }
3430            _ => None,
3431        };
3432        // The attribute set is collected whole, or the object says it could
3433        // not be: a short list here would be a dataset reporting attributes
3434        // the file does not agree it has.
3435        let attributes = collect_object_attributes(handle, ctx, header);
3436        match (datatype, dataspace, layout) {
3437            (Some(dt), Some(ds), Some(dl)) => ObjectKind::Dataset(Box::new(DatasetReadInfo {
3438                name: name.to_string(),
3439                object_header_address: addr,
3440                datatype: dt,
3441                dataspace: ds,
3442                layout: dl,
3443                filter_pipeline,
3444                attributes,
3445                fill_value,
3446                fill_defined,
3447                fill_write_time,
3448                alloc_time,
3449                external_files,
3450                virtual_mappings,
3451                // Both filled in by `resolve_virtual_extents` once the
3452                // reader has the directory source names resolve against; a
3453                // catalog on its own cannot open another file.
3454                virtual_resolution: None,
3455                virtual_stored_dims: None,
3456            })),
3457            // The three messages are present and none of them reported an
3458            // error, so this is unreachable; report it as unreadable rather
3459            // than dropping the name on an invariant this function owns.
3460            _ => ObjectKind::UnreadableDataset(
3461                "its datatype, dataspace and data layout messages decoded but did not all \
3462                 produce a value"
3463                    .into(),
3464            ),
3465        }
3466    }
3467
3468    /// Every dataset in the file, by path (no leading `/`).
3469    ///
3470    /// A dataset this crate cannot read is still a dataset the file
3471    /// contains, so it is listed here alongside the readable ones and
3472    /// answers [`Self::unreadable_reason`]; opening it reports that reason.
3473    /// Resolve a virtual dataset's mapping list from the global heap object
3474    /// its layout message points at (`H5D__virtual_load_layout`,
3475    /// H5Dvirtual.c). Like the external-file-list decode above, a failure
3476    /// here must not fall back to silently treating the dataset as having
3477    /// no data: a `Virtual` layout carries no data address of its own, so a
3478    /// dropped mapping list would read back as all-fill with no error.
3479    fn resolve_virtual_mappings(
3480        handle: &mut FileHandle,
3481        ctx: &FormatContext,
3482        heap_address: u64,
3483        heap_index: u32,
3484        name: &str,
3485    ) -> IoResult<VirtualMappingList> {
3486        let coll = read_heap_collection_from(handle, ctx, heap_address)?;
3487        let idx = u16::try_from(heap_index).map_err(|_| {
3488            crate::io::IoError::InvalidState(format!(
3489                "dataset {name:?} virtual mapping heap index {heap_index} does not fit \
3490                 the 16-bit on-disk field"
3491            ))
3492        })?;
3493        let obj = coll.get_object(idx).ok_or_else(|| {
3494            crate::io::IoError::InvalidState(format!(
3495                "dataset {name:?} virtual mapping list object {idx} not found in the \
3496                 global heap collection at address {heap_address:#x}"
3497            ))
3498        })?;
3499        VirtualMappingList::decode(obj, ctx).map_err(|e| {
3500            crate::io::IoError::InvalidState(format!(
3501                "dataset {name:?} has a malformed virtual dataset mapping list: {e}"
3502            ))
3503        })
3504    }
3505
3506    /// Resolve every external-file slot's name through the local heap the
3507    /// EFL message points at (H5Oefl.c decodes only the byte offset; the
3508    /// string itself lives in a separate on-disk local heap, exactly like a
3509    /// v0/v1 group's link names — see [`local_heap_get_string`]).
3510    pub(crate) fn resolve_external_file_slots(
3511        handle: &mut FileHandle,
3512        ctx: &FormatContext,
3513        efl: &ExternalFileListMessage,
3514    ) -> IoResult<Vec<ExternalFileSegment>> {
3515        let sa = ctx.sizeof_addr as usize;
3516        let ss = ctx.sizeof_size as usize;
3517        let heap_hdr_buf = handle.read_at_most(efl.heap_addr, 64)?;
3518        let heap_hdr = LocalHeapHeader::decode(&heap_hdr_buf, sa, ss)?;
3519        let heap_data = handle.read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)?;
3520
3521        efl.slots
3522            .iter()
3523            .map(|slot| {
3524                let name = local_heap_get_string(&heap_data, slot.name_offset)?;
3525                Ok(ExternalFileSegment {
3526                    name,
3527                    offset: slot.offset,
3528                    size: slot.size,
3529                })
3530            })
3531            .collect()
3532    }
3533
3534    /// Return the names of all datasets in the root group.
3535    pub fn dataset_names(&self) -> Vec<&str> {
3536        let mut names: Vec<&str> = self.datasets.iter().map(|d| d.name.as_str()).collect();
3537        names.extend(self.unreadable.keys().map(String::as_str));
3538        names
3539    }
3540
3541    /// Why the dataset at `path` (no leading `/`) cannot be read, or `None`
3542    /// when it can be — or does not exist.
3543    pub fn unreadable_reason(&mut self, path: &str) -> Option<&str> {
3544        if self.external_edge(path).is_some() {
3545            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
3546            let local = owner.canonical_path(&local);
3547            return owner.unreadable.get(&local).map(String::as_str);
3548        }
3549        let path = self.canonical_path(path);
3550        self.unreadable.get(&path).map(String::as_str)
3551    }
3552
3553    /// Every link record in the file, keyed by full path (no leading `/`).
3554    pub fn links(&self) -> &std::collections::BTreeMap<String, LinkClass> {
3555        &self.links
3556    }
3557
3558    /// The paths of every committed (named) datatype object in this file.
3559    ///
3560    /// A committed datatype is in neither [`dataset_names`](Self::dataset_names)
3561    /// nor the group listing — it is a third kind of object, and this is its
3562    /// listing.
3563    pub fn named_datatype_names(&self) -> Vec<&str> {
3564        self.datatypes.keys().map(String::as_str).collect()
3565    }
3566
3567    /// The committed datatype at `path` (no leading `/`), following group hard
3568    /// links, soft links and external links the way `H5Topen` does.
3569    ///
3570    /// `NotFound` means no committed datatype of that name; a name that *is*
3571    /// one but whose type this crate cannot decode answers `Unsupported` with
3572    /// the reason, never an absence.
3573    pub fn named_datatype(&mut self, path: &str) -> IoResult<&DatatypeMessage> {
3574        self.named_datatype_info(path)?
3575            .datatype()
3576            .map_err(|why| crate::io::IoError::Unsupported(why.to_string()))
3577    }
3578
3579    /// The attribute names of the committed datatype at `path`, in name
3580    /// order — matching h5py's default iteration for the (usual) case where
3581    /// the committed datatype does not track attribute creation order.
3582    /// Unlike [`Self::dataset_attr_names`] and its group/root counterparts,
3583    /// this path does not carry a per-attribute creation index to prefer
3584    /// when the object does track it: committed-datatype attributes are
3585    /// collected straight from compact header messages
3586    /// ([`Self::committed_datatype`]), without the envelope's creation index
3587    /// or dense-storage support the shared `AttributeEntry` collector has.
3588    pub fn named_datatype_attr_names(&mut self, path: &str) -> IoResult<Vec<String>> {
3589        let mut names: Vec<String> = self
3590            .named_datatype_info(path)?
3591            .attributes()
3592            .iter()
3593            .map(|a| a.name.clone())
3594            .collect();
3595        names.sort();
3596        Ok(names)
3597    }
3598
3599    /// The committed datatype at `path`'s own object-header attribute count.
3600    ///
3601    /// Committed-datatype attributes are collected only from compact header
3602    /// messages ([`Self::committed_datatype`]) — this crate does not model
3603    /// dense attribute storage on a named datatype — so unlike
3604    /// [`ObjectAttributes::header_count`] this is simply the count of what
3605    /// [`Self::named_datatype_attr_names`] already lists, with no separate
3606    /// dense-index path to fall back to.
3607    pub fn named_datatype_header_attr_count(&mut self, path: &str) -> IoResult<u64> {
3608        Ok(self.named_datatype_info(path)?.attributes().len() as u64)
3609    }
3610
3611    /// One attribute of the committed datatype at `path`, by name.
3612    pub fn named_datatype_attr(
3613        &mut self,
3614        path: &str,
3615        attr_name: &str,
3616    ) -> IoResult<&AttributeMessage> {
3617        let owned = attr_name.to_string();
3618        self.named_datatype_info(path)?
3619            .attributes()
3620            .iter()
3621            .find(|a| a.name == owned)
3622            .ok_or_else(|| crate::io::IoError::NotFound(format!("{path}:{attr_name}")))
3623    }
3624
3625    /// The committed datatype object at `path`, after link traversal.
3626    ///
3627    /// The object answers here whether or not its type decodes; the reason it
3628    /// does not is on [`CommittedDatatypeInfo::datatype`].
3629    pub fn named_datatype_info(&mut self, path: &str) -> IoResult<&CommittedDatatypeInfo> {
3630        if self.external_edge(path).is_some() {
3631            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS)?;
3632            let local = owner.canonical_path(&local);
3633            return owner
3634                .datatypes
3635                .get(&local)
3636                .ok_or(crate::io::IoError::NotFound(local));
3637        }
3638        let local = self.canonical_path(path);
3639        self.datatypes
3640            .get(&local)
3641            .ok_or(crate::io::IoError::NotFound(local))
3642    }
3643
3644    /// The class of the link at `path` (no leading `/`), or `None` when no
3645    /// link of that name exists. The path is traversed first, so a link
3646    /// reached through a group hard link, a soft link or an external link
3647    /// resolves — an external link's own record is found before the
3648    /// traversal crosses it, since a name matches its own link exactly.
3649    pub fn link_class(&mut self, path: &str) -> Option<&LinkClass> {
3650        let path = path.trim_start_matches('/');
3651        if self.links.contains_key(path) {
3652            return self.links.get(path);
3653        }
3654        if self.external_edge(path).is_some() {
3655            let (owner, local, _) = self.external_owner(path, MAX_EXTERNAL_HOPS).ok()?;
3656            let local = owner.canonical_path(&local);
3657            return owner.links.get(&local);
3658        }
3659        let path = self.canonical_path(path);
3660        self.links.get(&path)
3661    }
3662
3663    /// Follow a path (no leading `/`) the way `H5Dopen` / `H5Gopen` do:
3664    /// rewrite each component that is a group hard-link alias or a soft link
3665    /// until nothing changes, bounded so a link cycle cannot loop forever.
3666    ///
3667    /// This is the single owner of link traversal — every lookup that takes a
3668    /// caller-supplied path goes through it rather than comparing the path to
3669    /// a catalog key directly.
3670    fn traverse(&self, name: &str) -> Traversal {
3671        // libhdf5 bounds soft-link traversal at `H5L_NLINKS_DEF`; this covers
3672        // that and the hard-link alias rewrites interleaved with it.
3673        const MAX_TRAVERSALS: usize = 64;
3674        let mut name = name.trim_start_matches('/').to_string();
3675        let mut via = None;
3676        for _ in 0..MAX_TRAVERSALS {
3677            let Some((prefix, rewrite)) = self.longest_rewrite(&name) else {
3678                break;
3679            };
3680            let rest = name[prefix.len()..].to_string();
3681            match rewrite {
3682                Rewrite::Alias(first) => {
3683                    // `first` is empty for an alias of the root group;
3684                    // trimming keeps the no-leading-'/' form either way.
3685                    name = format!("{first}{rest}").trim_start_matches('/').to_string();
3686                }
3687                Rewrite::Soft(target) => {
3688                    let resolved = resolve_link_value(prefix, target);
3689                    via = Some(SoftLinkRef {
3690                        link: prefix.to_string(),
3691                        target: target.to_string(),
3692                    });
3693                    name = format!("{resolved}{rest}")
3694                        .trim_start_matches('/')
3695                        .to_string();
3696                }
3697                Rewrite::External { file, path } => {
3698                    return Traversal::External {
3699                        link: prefix.to_string(),
3700                        file: file.to_string(),
3701                        path: format!("{path}{rest}"),
3702                    };
3703                }
3704            }
3705        }
3706        Traversal::Path { path: name, via }
3707    }
3708
3709    /// The rewrite one traversal step applies to `path`, and the prefix it
3710    /// matched: the longest prefix of `path` that is a group hard-link alias
3711    /// or a soft/external link, so a nested alias wins over a shorter one
3712    /// that also covers the path, and an alias wins over a link naming the
3713    /// same prefix.
3714    ///
3715    /// A prefix covers `path` only when it *is* `path` or ends at one of its
3716    /// `/` boundaries, so the candidates are `path` and its own ancestors —
3717    /// walking those from the longest down asks the catalogs by key instead
3718    /// of comparing every alias and every link against the path, which is
3719    /// what made each traversal cost a pass over the file's whole link table.
3720    fn longest_rewrite<'a>(&'a self, path: &str) -> Option<(&'a str, Rewrite<'a>)> {
3721        let mut end = path.len();
3722        loop {
3723            let candidate = &path[..end];
3724            if let Some((alias, first)) = self.group_aliases.get_key_value(candidate) {
3725                return Some((alias.as_str(), Rewrite::Alias(first)));
3726            }
3727            match self.links.get_key_value(candidate) {
3728                Some((link, LinkClass::Soft { path })) => {
3729                    return Some((link.as_str(), Rewrite::Soft(path)))
3730                }
3731                Some((link, LinkClass::External { file, path })) => {
3732                    return Some((link.as_str(), Rewrite::External { file, path }))
3733                }
3734                // A hard or user-defined link rewrites nothing, and a shorter
3735                // prefix of the path may still rewrite it.
3736                _ => {}
3737            }
3738            end = candidate.rfind('/')?;
3739        }
3740    }
3741
3742    /// The messages read from the superblock extension object header. All
3743    /// fields are `None` for a file without an extension.
3744    pub fn superblock_extension(&self) -> &SuperblockExtension {
3745        &self.ext
3746    }
3747
3748    /// Bytes the file's on-disk free-space managers record as free —
3749    /// `H5Fget_freespace`, the number `h5stat -S` prints as "Amount of tracked
3750    /// free space".
3751    ///
3752    /// Zero for a file whose file-space info message names no manager, which
3753    /// includes every file written without `persist`. The strategy is not
3754    /// consulted: a manager's header and section-info blocks have one layout
3755    /// whichever strategy allocated the space they describe.
3756    pub fn tracked_free_space(&mut self) -> IoResult<u64> {
3757        let Some(info) = self.ext.file_space_info.clone() else {
3758            return Ok(0);
3759        };
3760        crate::io::free_space_io::tracked_free_space(&mut self.handle, &self.meta.ctx, &info)
3761    }
3762
3763    /// Size in bytes of the userblock preceding the superblock: the offset the
3764    /// signature was found at, which is also the file's base address. Zero for
3765    /// a file without a userblock.
3766    pub fn userblock_size(&self) -> u64 {
3767        self.handle.base()
3768    }
3769
3770    /// The superblock format version (0-3), decoded once at open time and
3771    /// immutable for the life of an open file — a live SWMR refresh rescans
3772    /// the file's contents but never its own format version.
3773    pub fn superblock_version(&self) -> u8 {
3774        self.superblock_version
3775    }
3776
3777    /// Rewrite a path (no leading `/`) into the path of the object it reaches
3778    /// after link traversal. A path that leaves the file through an external
3779    /// link comes back unchanged — the callers that must report that case use
3780    /// [`Self::traverse`] directly.
3781    pub fn canonical_path(&self, name: &str) -> String {
3782        match self.traverse(name) {
3783            Traversal::Path { path, .. } => path,
3784            Traversal::External { .. } => name.trim_start_matches('/').to_string(),
3785        }
3786    }
3787
3788    /// Where `name` leaves this file, or `None` when it resolves inside it.
3789    ///
3790    /// This is the one question every path-taking entry point asks before it
3791    /// looks anything up: a name that crosses an external link is not this
3792    /// file's to answer, and answering it from this file's catalog anyway is
3793    /// how such a name came back as a plain absence.
3794    pub(crate) fn external_edge(&self, name: &str) -> Option<ExternalEdge> {
3795        match self.traverse(name) {
3796            Traversal::Path { .. } => None,
3797            Traversal::External { link, file, path } => Some(ExternalEdge { link, file, path }),
3798        }
3799    }
3800
3801    /// Candidate filesystem paths for a file named from inside this one, in
3802    /// the order `H5F_prefix_open_file` tries them (H5Fint.c:826-1025):
3803    ///
3804    /// 1. an absolute name exactly as given (:854-887) — and if that misses,
3805    ///    every later step uses its last component instead, as the C does;
3806    /// 2. each `:`-separated component of `env_var`, joined with that name
3807    ///    (:889-937);
3808    /// 3. `prop_prefix`, the property-list prefix (:938-950);
3809    /// 4. the directory of the path this file was opened by — libhdf5's
3810    ///    `H5F_EXTPATH` (:952-969);
3811    /// 5. the bare relative name, against the process's working directory
3812    ///    (:971-977);
3813    /// 6. the directory of that path *resolved* — libhdf5's
3814    ///    `H5F_ACTUAL_NAME`, which differs from step 4 through a symlink
3815    ///    (:979-1004).
3816    ///
3817    /// Both kinds of cross-file name run this one order and differ only in
3818    /// the two parameters: an external link is `H5F_PREFIX_ELINK` with
3819    /// `HDF5_EXT_PREFIX` and `H5Pset_elink_prefix` (H5Lexternal.c:210-215),
3820    /// a virtual dataset's source is `H5F_PREFIX_VDS` with
3821    /// `HDF5_VDS_PREFIX` and `H5Pset_virtual_prefix` (H5Dvirtual.c:877-882).
3822    fn prefix_open_candidates(
3823        &self,
3824        env_var: &str,
3825        prop_prefix: Option<&Path>,
3826        file: &str,
3827    ) -> Vec<PathBuf> {
3828        let raw = Path::new(file);
3829        let mut candidates = Vec::new();
3830        if raw.is_absolute() {
3831            candidates.push(raw.to_path_buf());
3832        }
3833        // Every attempt after an absolute miss uses the bare file name.
3834        let base: &Path = if raw.is_absolute() {
3835            Path::new(raw.file_name().unwrap_or(raw.as_os_str()))
3836        } else {
3837            raw
3838        };
3839        if let Ok(prefixes) = std::env::var(env_var) {
3840            candidates.extend(
3841                prefixes
3842                    .split(':')
3843                    .filter(|p| !p.is_empty())
3844                    .map(|p| Path::new(p).join(base)),
3845            );
3846        }
3847        if let Some(prefix) = prop_prefix {
3848            candidates.push(prefix.join(base));
3849        }
3850        if let Some(dir) = self.path.parent().filter(|d| !d.as_os_str().is_empty()) {
3851            candidates.push(dir.join(base));
3852        }
3853        candidates.push(base.to_path_buf());
3854        if !self.source_dir.as_os_str().is_empty() {
3855            candidates.push(self.source_dir.join(base));
3856        }
3857        candidates
3858    }
3859
3860    /// Put an external-link prefix in force for this reader and every file
3861    /// it opens on another's behalf. Set once at open, before any name has
3862    /// been resolved, because an external link's answer is fixed the first
3863    /// time it is asked ([`external_resolved`](Self::external_resolved)).
3864    pub(crate) fn set_elink_prefix(&mut self, prefix: Option<String>) {
3865        self.elink_prefix = prefix;
3866    }
3867
3868    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for an
3869    /// external link. The property-list step is
3870    /// [`H5FileOptions::elink_prefix`](crate::H5FileOptions::elink_prefix)
3871    /// exactly as given: `H5L__extern_traverse` peeks
3872    /// `H5L_ACS_ELINK_PREFIX_NAME` and hands it straight to the search
3873    /// (H5Lexternal.c:210-215), so unlike a virtual dataset's prefix it goes
3874    /// through no `H5D__build_file_prefix` — no `${ORIGIN}` expansion, and
3875    /// `HDF5_EXT_PREFIX` does not shadow it.
3876    fn external_candidates(&self, file: &str) -> Vec<PathBuf> {
3877        let prop = self.elink_prefix.as_deref().map(Path::new);
3878        self.prefix_open_candidates("HDF5_EXT_PREFIX", prop, file)
3879    }
3880
3881    /// [`prefix_open_candidates`](Self::prefix_open_candidates) for a
3882    /// virtual dataset's source. The property-list step is whatever
3883    /// `H5D__build_file_prefix` puts in `dset->shared->vds_prefix`
3884    /// (H5Dint.c:1076-1119): `HDF5_VDS_PREFIX` if the environment names one,
3885    /// otherwise [`DatasetAccess::virtual_prefix`], either way with
3886    /// `${ORIGIN}` expanded.
3887    fn vds_candidates(&self, access: &DatasetAccess, file: &str) -> Vec<PathBuf> {
3888        let prop = resolve_vdsfile_prefix(access.virtual_prefix_value(), &self.source_dir);
3889        self.prefix_open_candidates("HDF5_VDS_PREFIX", prop.as_deref(), file)
3890    }
3891
3892    /// Open one external link's target file, or hand back the handle a
3893    /// previous link to the same resolved path already opened.
3894    fn external_target(&mut self, link: &str, file: &str) -> IoResult<&mut Hdf5Reader> {
3895        // The search runs once per link value; after that the answer is what
3896        // this reader resolved it to, whatever the filesystem does next.
3897        let resolved = match self.external_resolved.get(file) {
3898            Some(resolved) => resolved.clone(),
3899            None => {
3900                let candidates = self.external_candidates(file);
3901                let resolved = candidates
3902                    .iter()
3903                    .find(|p| p.is_file())
3904                    .cloned()
3905                    .ok_or_else(|| crate::io::IoError::ExternalFileNotFound {
3906                        link: link.to_string(),
3907                        file: file.to_string(),
3908                        searched: candidates.iter().map(|p| p.display().to_string()).collect(),
3909                    })?;
3910                self.external_resolved
3911                    .insert(file.to_string(), resolved.clone());
3912                resolved
3913            }
3914        };
3915        self.cross_file(resolved, CrossFileOwner::Reader)
3916    }
3917
3918    /// Open `resolved`, or hand back the handle a previous crossing to the
3919    /// same file already opened.
3920    ///
3921    /// The single owner of every file this reader opens on another file's
3922    /// behalf, so a path named by any number of external links, external
3923    /// references and virtual-dataset sources is opened once and read
3924    /// through one handle. What resolved the name to this path is the
3925    /// caller's business, and differs by kind: an external link and a
3926    /// virtual source each run `H5F_prefix_open_file`'s search order under
3927    /// their own prefix, a reference has no search order at all.
3928    ///
3929    /// `owner` says how long the handle stays open. One path can be reached
3930    /// by both kinds of crossing, and the wider ownership wins: a file an
3931    /// external link holds for this reader's life does not start expiring
3932    /// with a virtual dataset that also names it.
3933    fn cross_file(
3934        &mut self,
3935        resolved: PathBuf,
3936        owner: CrossFileOwner,
3937    ) -> IoResult<&mut Hdf5Reader> {
3938        let locking = self.locking;
3939        let elink_prefix = self.elink_prefix.clone();
3940        match self.external.entry(resolved) {
3941            std::collections::btree_map::Entry::Occupied(e) => {
3942                let e = e.into_mut();
3943                e.owner.widen(owner);
3944                Ok(&mut *e.reader)
3945            }
3946            std::collections::btree_map::Entry::Vacant(e) => {
3947                let mut reader = Hdf5Reader::open_with_locking(e.key(), locking)?;
3948                reader.elink_prefix.clone_from(&elink_prefix);
3949                Ok(&mut *e
3950                    .insert(CrossFileEntry {
3951                        reader: Box::new(reader),
3952                        owner,
3953                    })
3954                    .reader)
3955            }
3956        }
3957    }
3958
3959    /// Close every cross-file handle whose last owning virtual-dataset open
3960    /// has gone, which is where `H5D__virtual_reset_layout` closes the source
3961    /// datasets holding libhdf5's (H5Dvirtual.c:709-710).
3962    ///
3963    /// The single releaser of a [`CrossFileOwner::VirtualOpens`] entry —
3964    /// nothing else removes one, so a source cannot be closed while a handle
3965    /// on the virtual dataset that named it is still alive. Its two callers
3966    /// are the two moments the owning set can be empty: a handle's drop, and
3967    /// an extent resolution run with no handle open at all (this crate
3968    /// resolves at `H5Fopen`, where libhdf5 has nothing to resolve yet).
3969    pub(crate) fn release_closed_virtual_sources(&mut self) {
3970        let dead: Vec<PathBuf> = self
3971            .external
3972            .iter()
3973            .filter(|(_, e)| match &e.owner {
3974                CrossFileOwner::Reader => false,
3975                CrossFileOwner::VirtualOpens(vds) => !vds.iter().any(|v| self.is_open_dataset(v)),
3976            })
3977            .map(|(p, _)| p.clone())
3978            .collect();
3979        for path in dead {
3980            self.external.remove(&path);
3981        }
3982    }
3983
3984    /// Whether the dataset at canonical path `name` still has a live handle
3985    /// — libhdf5's "is this dataset in `H5FO_opened`".
3986    fn is_open_dataset(&self, name: &str) -> bool {
3987        self.dataset_access
3988            .get(name)
3989            .is_some_and(|e| e.open.strong_count() > 0)
3990    }
3991
3992    /// Open the file a virtual mapping's source name points at, or hand back
3993    /// the handle a previous mapping to the same file already opened.
3994    ///
3995    /// A virtual dataset's source is not opened by any path of its own:
3996    /// `H5D__virtual_open_source_dset` hands the name to
3997    /// `H5F_prefix_open_file` (H5Dvirtual.c:877-882) against the *primary*
3998    /// file's external file cache and with `source_fapl`, a copy of the
3999    /// primary file's own file-access property list
4000    /// (H5Dvirtual.c:2193-2194), which carries its `use_file_locking`
4001    /// verbatim (H5Fint.c:389). That is the same cache and the same call
4002    /// external links reach, keyed by the name the open used
4003    /// (H5Fefc.c:245), and it is released back to it with `H5F_efc_close`
4004    /// (H5Dvirtual.c:925-927). So a source file is this reader's
4005    /// [`cross_file`](Self::cross_file) like any other target: one handle
4006    /// per path, under this file's locking policy — measured against
4007    /// libhdf5 1.14.6, reading a cross-file VDS leaves the source flock'd
4008    /// under the default `HDF5_USE_FILE_LOCKING` and unlocked under
4009    /// `HDF5_USE_FILE_LOCKING=FALSE`.
4010    ///
4011    /// What it is *not* is a target this reader holds for its own life:
4012    /// `vds` — the canonical path of the virtual dataset naming the source —
4013    /// owns the handle, and it goes when that dataset's last handle does
4014    /// ([`CrossFileOwner::VirtualOpens`]).
4015    ///
4016    /// `None` where the file cannot be opened at all: `H5F_prefix_open_file`
4017    /// is asked to *try*, and a null source file is "no data there yet",
4018    /// not a failure.
4019    fn vds_source_file(&mut self, vds: &str, file_name: &str) -> Option<&mut Hdf5Reader> {
4020        // A name that has resolved once stays resolved, the same way an
4021        // external link's does — the handle this reader already holds is the
4022        // answer, whatever the filesystem does next. A name that has *not*
4023        // resolved is searched again on the next read, which is what the C
4024        // does too: it re-runs `H5D__virtual_open_source_dset` whenever the
4025        // source dataset is still unopened (H5Dvirtual.c:1421-1423,
4026        // :2558-2561).
4027        let key = (vds.to_string(), file_name.to_string());
4028        let resolved = match self.vds_resolved.get(&key) {
4029            Some(resolved) => resolved.clone(),
4030            None => {
4031                let access = self.access_in_force(vds);
4032                let resolved = self
4033                    .vds_candidates(&access, file_name)
4034                    .into_iter()
4035                    .find(|p| p.is_file())?;
4036                self.vds_resolved.insert(key, resolved.clone());
4037                resolved
4038            }
4039        };
4040        self.cross_file(resolved, CrossFileOwner::virtual_open(vds))
4041            .ok()
4042    }
4043
4044    /// The reader that owns `name`, the path of `name` inside it, and the last
4045    /// external link crossed to get there.
4046    ///
4047    /// This is the single owner of cross-file resolution: it follows external
4048    /// links until the remaining path resolves inside the reader it returns,
4049    /// so callers do exactly one delegation and never have to re-check.
4050    fn external_owner(
4051        &mut self,
4052        name: &str,
4053        hops: usize,
4054    ) -> IoResult<(&mut Self, String, Option<ExternalEdge>)> {
4055        let path = name.trim_start_matches('/').to_string();
4056        let Some(edge) = self.external_edge(&path) else {
4057            return Ok((self, path, None));
4058        };
4059        if hops == 0 {
4060            return Err(crate::io::IoError::InvalidState(format!(
4061                "resolving '{name}' crossed more than {MAX_EXTERNAL_HOPS} external links \
4062                 (libhdf5 stops at the same H5L_NUM_LINKS); the links may form a cycle"
4063            )));
4064        }
4065        let target = self.external_target(&edge.link, &edge.file)?;
4066        let (owner, path, deeper) = target.external_owner(&edge.path, hops - 1)?;
4067        Ok((owner, path, deeper.or(Some(edge))))
4068    }
4069
4070    /// Return metadata for a dataset by name. Like `H5Dopen`, the name may
4071    /// pass through group hard links, soft links and external links.
4072    pub fn dataset_info(&mut self, name: &str) -> Option<&DatasetReadInfo> {
4073        if self.external_edge(name).is_some() {
4074            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS).ok()?;
4075            return owner.dataset_info_local(&path);
4076        }
4077        self.dataset_info_local(name)
4078    }
4079
4080    /// [`dataset_info`](Self::dataset_info) restricted to this file: soft
4081    /// links and group hard links resolve, an external link does not. Every
4082    /// read path uses this, because by then the owning reader has already been
4083    /// selected and the path is local to it.
4084    fn dataset_info_local(&self, name: &str) -> Option<&DatasetReadInfo> {
4085        let name = self.canonical_path(name);
4086        self.datasets.get(&name)
4087    }
4088
4089    /// Where `name` sits in this file's catalog, resolving the same soft
4090    /// links and group hard links [`dataset_info_local`](Self::dataset_info_local)
4091    /// does. Read paths that need the entry more than once take the position
4092    /// once and index with it, rather than walking the path again per lookup.
4093    fn dataset_position(&self, name: &str) -> IoResult<usize> {
4094        let canonical = self.canonical_path(name);
4095        self.datasets
4096            .position(&canonical)
4097            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
4098    }
4099
4100    /// Open `name` as a dataset the way `H5Dopen2` does, reporting *why* it
4101    /// cannot be opened instead of collapsing every cause into absence: a
4102    /// soft link whose target does not exist is a dangling link, and a path
4103    /// through an external link is resolved in the file that link names.
4104    ///
4105    /// This is the gate every typed dataset access goes through. `access` is
4106    /// the dapl the open names; its properties are put in force for the
4107    /// dataset first, so the extent this returns is the one they resolve it
4108    /// to.
4109    ///
4110    /// Returns the token that holds the open alive alongside the extent: the
4111    /// caller must keep it for as long as its handle lives, because the
4112    /// properties this open put in force stay in force exactly that long
4113    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
4114    pub fn open_dataset_with(
4115        &mut self,
4116        name: &str,
4117        access: &DatasetAccess,
4118    ) -> IoResult<(Option<DatasetOpenToken>, &DatasetReadInfo)> {
4119        if self.external_edge(name).is_some() {
4120            let (owner, path, edge) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4121            let open = owner.apply_dataset_access(&path, access)?;
4122            return match owner.open_dataset_local(&path) {
4123                // The name is absent in the target file, which makes the link
4124                // that pointed there dangling — not the caller's path absent.
4125                Err(crate::io::IoError::NotFound(_)) => {
4126                    Err(edge.map_or_else(|| crate::io::IoError::NotFound(path), |e| e.dangling()))
4127                }
4128                other => other.map(|info| (open, info)),
4129            };
4130        }
4131        let open = self.apply_dataset_access(name, access)?;
4132        self.open_dataset_local(name).map(|info| (open, info))
4133    }
4134
4135    /// [`open_dataset`](Self::open_dataset) restricted to this file.
4136    fn open_dataset_local(&self, name: &str) -> IoResult<&DatasetReadInfo> {
4137        let Traversal::Path { path, via } = self.traverse(name) else {
4138            // `external_owner` only ever returns a reader in which the
4139            // remaining path resolves locally, so no caller can land here.
4140            return Err(crate::io::IoError::NotFound(name.to_string()));
4141        };
4142        if let Some(info) = self.datasets.get(&path) {
4143            return Ok(info);
4144        }
4145        if let Some(why) = self.unreadable.get(&path) {
4146            return Err(crate::io::IoError::Unsupported(format!(
4147                "'{name}' is a dataset this crate cannot read: {why}"
4148            )));
4149        }
4150        if let Some(SoftLinkRef { link, target }) = via {
4151            return Err(crate::io::IoError::DanglingLink { link, target });
4152        }
4153        Err(crate::io::IoError::NotFound(name.to_string()))
4154    }
4155
4156    /// Resolve one entry of an attribute list.
4157    ///
4158    /// The cases an attribute list can answer are kept apart here rather than
4159    /// at each call site: decoded, present but undecodable, absent from a set
4160    /// known to be whole, and absent from a set that was never read whole. A
4161    /// caller that collapsed any of the middle cases into the last would
4162    /// report an attribute the file contains as one it does not.
4163    fn resolve_attr<'a>(
4164        attrs: &'a ObjectAttributes,
4165        owner: &str,
4166        name: &str,
4167    ) -> IoResult<&'a AttributeMessage> {
4168        match attrs.entries.iter().find(|a| a.name() == name) {
4169            Some(entry) => entry.decoded().map_err(|reason| {
4170                crate::io::IoError::Unsupported(format!(
4171                    "attribute '{name}' on '{owner}' cannot be decoded: {reason}"
4172                ))
4173            }),
4174            // Not among what was read — but the part that was not read could
4175            // hold it, so an incomplete set cannot answer "absent".
4176            None => match attrs.unreadable_reason() {
4177                Some(reason) => Err(incomplete_error(owner, reason)),
4178                None => Err(crate::io::IoError::NotFound(format!("{owner}:{name}"))),
4179            },
4180        }
4181    }
4182
4183    /// Why the attribute `name` in `attrs` cannot be read, or `None` when it
4184    /// can be — or is not there at all, which the accessors above report as
4185    /// `NotFound`.
4186    fn attr_reason<'a>(attrs: &'a ObjectAttributes, name: &str) -> Option<&'a str> {
4187        attrs
4188            .entries
4189            .iter()
4190            .find(|a| a.name() == name)?
4191            .unreadable_reason()
4192    }
4193
4194    /// Why a dataset's attribute cannot be read, or `None` when it can be.
4195    pub fn dataset_attr_unreadable_reason(
4196        &mut self,
4197        ds_name: &str,
4198        attr_name: &str,
4199    ) -> Option<&str> {
4200        Self::attr_reason(&self.dataset_info(ds_name)?.attributes, attr_name)
4201    }
4202
4203    /// Why a root-level attribute cannot be read, or `None` when it can be.
4204    pub fn root_attr_unreadable_reason(&self, name: &str) -> Option<&str> {
4205        Self::attr_reason(&self.root_attributes, name)
4206    }
4207
4208    /// Why a non-root group's attribute cannot be read, or `None` when it can
4209    /// be.
4210    pub fn group_attr_unreadable_reason(&self, group_path: &str, name: &str) -> Option<&str> {
4211        Self::attr_reason(
4212            self.group_attributes
4213                .get(&self.canonical_path(group_path))?,
4214            name,
4215        )
4216    }
4217
4218    /// Why a dataset's attributes cannot be listed at all, or `None` when the
4219    /// set is whole. Object scope, unlike
4220    /// [`Self::dataset_attr_unreadable_reason`]: the failure belongs to no
4221    /// single name.
4222    pub fn dataset_attrs_unreadable_reason(&mut self, ds_name: &str) -> Option<&str> {
4223        self.dataset_info(ds_name)?.attributes.unreadable_reason()
4224    }
4225
4226    /// A dataset's own compact-vs-dense attribute storage.
4227    pub fn dataset_attr_storage(&mut self, ds_name: &str) -> IoResult<AttributeStorage> {
4228        Ok(self
4229            .dataset_info(ds_name)
4230            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?
4231            .attributes
4232            .storage())
4233    }
4234
4235    /// A dataset's own object-header attribute count.
4236    pub fn dataset_header_attr_count(&mut self, ds_name: &str) -> IoResult<u64> {
4237        let info = self
4238            .dataset_info(ds_name)
4239            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
4240        info.attributes.header_count(ds_name)
4241    }
4242
4243    /// Why the root group's attributes cannot be listed at all, or `None` when
4244    /// the set is whole.
4245    pub fn root_attrs_unreadable_reason(&self) -> Option<&str> {
4246        self.root_attributes.unreadable_reason()
4247    }
4248
4249    /// Why a non-root group's attributes cannot be listed at all, or `None`
4250    /// when the set is whole.
4251    pub fn group_attrs_unreadable_reason(&self, group_path: &str) -> Option<&str> {
4252        self.group_attributes
4253            .get(&self.canonical_path(group_path))?
4254            .unreadable_reason()
4255    }
4256
4257    /// Return the attribute names of a dataset.
4258    ///
4259    /// Includes attributes this crate cannot decode: the object header carries
4260    /// them, so the listing does too. [`Self::dataset_attr`] says why one of
4261    /// those cannot be read. An object whose attribute set could not be read
4262    /// whole has no listing to give and returns the reason instead — see
4263    /// [`Self::dataset_attrs_unreadable_reason`].
4264    pub fn dataset_attr_names(&mut self, name: &str) -> IoResult<Vec<String>> {
4265        let info = self
4266            .dataset_info(name)
4267            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4268        info.attributes.ordered_names(name)
4269    }
4270
4271    /// Return a specific attribute by dataset name and attribute name.
4272    pub fn dataset_attr(&mut self, ds_name: &str, attr_name: &str) -> IoResult<&AttributeMessage> {
4273        let info = self
4274            .dataset_info(ds_name)
4275            .ok_or_else(|| crate::io::IoError::NotFound(ds_name.to_string()))?;
4276        Self::resolve_attr(&info.attributes, ds_name, attr_name)
4277    }
4278
4279    /// Return the names of root-level (file) attributes, undecodable ones
4280    /// included — see [`Self::dataset_attr_names`].
4281    pub fn root_attr_names(&self) -> IoResult<Vec<String>> {
4282        self.root_attributes.ordered_names("/")
4283    }
4284
4285    /// Return a root-level attribute by name.
4286    pub fn root_attr(&self, name: &str) -> IoResult<&AttributeMessage> {
4287        Self::resolve_attr(&self.root_attributes, "/", name)
4288    }
4289
4290    /// The root group's own attribute creation-order policy.
4291    pub fn root_attr_creation_order(&self) -> CreationOrder {
4292        self.root_attributes.creation_order()
4293    }
4294
4295    /// The root group's own compact-vs-dense attribute storage.
4296    pub fn root_attr_storage(&self) -> AttributeStorage {
4297        self.root_attributes.storage()
4298    }
4299
4300    /// The root group's own object-header attribute count.
4301    pub fn root_header_attr_count(&self) -> IoResult<u64> {
4302        self.root_attributes.header_count("/")
4303    }
4304
4305    /// The root group's own link creation-order policy.
4306    pub fn root_link_creation_order(&self) -> CreationOrder {
4307        self.root_link_storage.1
4308    }
4309
4310    /// The root group's own link storage kind: symbol-table (legacy),
4311    /// compact link messages, or dense (fractal heap plus name index).
4312    pub fn root_link_storage(&self) -> LinkStorage {
4313        self.root_link_storage.0
4314    }
4315
4316    /// Return the attribute names of a non-root group (path without a
4317    /// leading `/`, e.g. `"detector"` or `"entry/instrument"`; may pass
4318    /// through group hard links). Undecodable attributes included — see
4319    /// [`Self::dataset_attr_names`].
4320    pub fn group_attr_names(&mut self, group_path: &str) -> IoResult<Vec<String>> {
4321        if self.external_edge(group_path).is_some() {
4322            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
4323            // The empty remainder is the target file's root group, whose
4324            // attributes are not in the per-group map.
4325            if local.is_empty() {
4326                return owner.root_attr_names();
4327            }
4328            return owner.group_attr_names_local(&local);
4329        }
4330        self.group_attr_names_local(group_path)
4331    }
4332
4333    fn group_attr_names_local(&self, group_path: &str) -> IoResult<Vec<String>> {
4334        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
4335            return Ok(Vec::new());
4336        };
4337        attrs.ordered_names(group_path)
4338    }
4339
4340    /// A non-root group's own attribute creation-order policy. `Untracked`
4341    /// for a path the walk never reached, the same silent default
4342    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
4343    /// unknown group's attribute listing.
4344    pub fn group_attr_creation_order(&self, group_path: &str) -> CreationOrder {
4345        self.group_attributes
4346            .get(&self.canonical_path(group_path))
4347            .map(ObjectAttributes::creation_order)
4348            .unwrap_or_default()
4349    }
4350
4351    /// A non-root group's own compact-vs-dense attribute storage. `Compact`
4352    /// — the same silent default as an empty attribute set — for a path the
4353    /// walk never reached.
4354    pub fn group_attr_storage(&self, group_path: &str) -> AttributeStorage {
4355        self.group_attributes
4356            .get(&self.canonical_path(group_path))
4357            .map(ObjectAttributes::storage)
4358            .unwrap_or_default()
4359    }
4360
4361    /// A non-root group's own object-header attribute count. `0` for a path
4362    /// the walk never reached, the same silent default
4363    /// [`group_attr_names_local`](Self::group_attr_names_local) gives an
4364    /// unknown group's attribute listing.
4365    pub fn group_header_attr_count(&self, group_path: &str) -> IoResult<u64> {
4366        let Some(attrs) = self.group_attributes.get(&self.canonical_path(group_path)) else {
4367            return Ok(0);
4368        };
4369        attrs.header_count(group_path)
4370    }
4371
4372    /// A non-root group's own link creation-order policy. `Untracked` for a
4373    /// path the walk never reached, the same silent default
4374    /// [`group_attr_creation_order`](Self::group_attr_creation_order) gives.
4375    pub fn group_link_creation_order(&self, group_path: &str) -> CreationOrder {
4376        self.group_link_storage
4377            .get(&self.canonical_path(group_path))
4378            .map_or(CreationOrder::Untracked, |(_, order)| *order)
4379    }
4380
4381    /// A non-root group's own link storage kind. `Compact` — the same
4382    /// silent default as an empty link set — for a path the walk never
4383    /// reached.
4384    pub fn group_link_storage(&self, group_path: &str) -> LinkStorage {
4385        self.group_link_storage
4386            .get(&self.canonical_path(group_path))
4387            .map_or(LinkStorage::Compact, |(storage, _)| *storage)
4388    }
4389
4390    /// Return a non-root group's attribute by name.
4391    pub fn group_attr(&mut self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
4392        if self.external_edge(group_path).is_some() {
4393            let (owner, local, _) = self.external_owner(group_path, MAX_EXTERNAL_HOPS)?;
4394            if local.is_empty() {
4395                return owner.root_attr(name);
4396            }
4397            return owner.group_attr_local(&local, name);
4398        }
4399        self.group_attr_local(group_path, name)
4400    }
4401
4402    fn group_attr_local(&self, group_path: &str, name: &str) -> IoResult<&AttributeMessage> {
4403        match self.group_attributes.get(&self.canonical_path(group_path)) {
4404            Some(attrs) => Self::resolve_attr(attrs, group_path, name),
4405            // No entry at all: the walk found nothing to record on this group.
4406            None => Err(crate::io::IoError::NotFound(format!("{group_path}:{name}"))),
4407        }
4408    }
4409
4410    /// Return every non-root group path the discovery walk traversed into
4411    /// (no leading `/`). Built from actual link records, so empty groups,
4412    /// attribute-only groups, and subgroup-only groups are all included.
4413    pub fn group_paths(&self) -> &std::collections::BTreeSet<String> {
4414        &self.group_paths
4415    }
4416
4417    /// Report whether a group exists at `group_path` (no leading `/`;
4418    /// may pass through group hard links). The empty string denotes the
4419    /// root group, which always exists.
4420    pub fn has_group(&self, group_path: &str) -> bool {
4421        if group_path.is_empty() || self.group_paths.contains(group_path) {
4422            return true;
4423        }
4424        let canon = self.canonical_path(group_path);
4425        canon.is_empty() || self.group_paths.contains(&canon)
4426    }
4427
4428    /// Read and decode the global-heap collection at `addr`, applying the
4429    /// validation of libhdf5's `H5HG__cache_heap_deserialize`: the `GCOL`
4430    /// signature must be present and the declared size at least
4431    /// `H5HG_MINSIZE` (4096 bytes). There is no upper size cap — libhdf5
4432    /// has none, and this crate's writers put a whole write call's strings
4433    /// into one collection, which a cap would turn into silent data loss.
4434    fn read_heap_collection(&mut self, addr: u64) -> IoResult<GlobalHeapCollection> {
4435        read_heap_collection_from(&mut self.handle, &self.meta.ctx, addr)
4436    }
4437
4438    /// Decode an attribute's value as a string, resolving a variable-length
4439    /// string attribute through the global heap (h5py writes string
4440    /// attributes as variable-length by default).
4441    pub fn attr_string_value(&mut self, attr: &AttributeMessage) -> IoResult<String> {
4442        use crate::format::messages::datatype::DatatypeMessage;
4443        if !matches!(attr.datatype, DatatypeMessage::VarLenString { .. }) {
4444            return fixed_string_attr_value(attr);
4445        }
4446        // Variable-length string: the attribute value is a global-heap
4447        // reference (sequence length + collection address + object index).
4448        if attr.data.len() < vlen_reference_size(&self.meta.ctx) {
4449            return Ok(String::new());
4450        }
4451        let (_seq, coll_addr, obj_index) = decode_vlen_reference(&attr.data, &self.meta.ctx)?;
4452        if coll_addr == UNDEF_ADDR || coll_addr == 0 {
4453            return Ok(String::new());
4454        }
4455        let coll = self.read_heap_collection(coll_addr)?;
4456        let idx = u16::try_from(obj_index).map_err(|_| {
4457            crate::io::IoError::InvalidState(format!(
4458                "global heap object index {obj_index} does not fit the 16-bit on-disk field"
4459            ))
4460        })?;
4461        let obj = coll.get_object(idx).ok_or_else(|| {
4462            crate::io::IoError::InvalidState(format!(
4463                "global heap object {idx} not found in the collection at address {coll_addr:#x}"
4464            ))
4465        })?;
4466        Ok(String::from_utf8_lossy(obj).to_string())
4467    }
4468
4469    /// The absolute path of the object whose header sits at `addr` — what an
4470    /// object reference to it names — or `None` when no group or dataset the
4471    /// discovery walk reached lives there (a reference into a file region the
4472    /// walk never traversed, or a stale one).
4473    pub fn path_for_object(&self, addr: u64) -> Option<&str> {
4474        self.object_paths.get(&addr).map(String::as_str)
4475    }
4476
4477    /// How the object header of `path` stores each message it does not hold
4478    /// privately, as `(message type, storage)` in header order.
4479    ///
4480    /// The observable is the message's flags byte, so this reads the raw
4481    /// header chain: every other read path resolves shared pointers into
4482    /// bodies and clears the flag on the way through, which is exactly the
4483    /// evidence wanted here.
4484    pub fn object_message_storage(&mut self, path: &str) -> IoResult<Vec<(u8, MessageStorage)>> {
4485        let addr = self.object_header_address(path)?;
4486        crate::io::object_header_io::read_header_message_storage(&mut self.handle, &self.meta, addr)
4487    }
4488
4489    /// The flags byte of every message the object header of `path` holds, as
4490    /// `(message type, flags)` in header order, null and continuation
4491    /// messages left out.
4492    ///
4493    /// The byte `h5debug` renders as `<C>`, `<DS>`, `<S>` and the rest
4494    /// (`H5O__debug_real`, H5Odbg.c:409-455). It says which messages the
4495    /// library may cache as never-changing and which it refuses to share, and
4496    /// nothing else in the file records either.
4497    pub fn object_message_flags(&mut self, path: &str) -> IoResult<Vec<(u8, u8)>> {
4498        let addr = self.object_header_address(path)?;
4499        crate::io::object_header_io::read_header_message_flags(&mut self.handle, &self.meta, addr)
4500    }
4501
4502    /// The class and version of every datatype message the object at `path`
4503    /// carries, outermost first; see
4504    /// [`DatatypeMessage::decode_versions`](crate::format::messages::datatype::DatatypeMessage::decode_versions).
4505    pub fn object_datatype_versions(
4506        &mut self,
4507        path: &str,
4508    ) -> IoResult<Vec<crate::format::messages::datatype::DatatypeNodeVersion>> {
4509        let addr = self.object_header_address(path)?;
4510        crate::io::object_header_io::read_header_datatype_versions(
4511            &mut self.handle,
4512            &self.meta,
4513            addr,
4514        )
4515    }
4516
4517    /// Whether the object at `path` records its times —
4518    /// `H5Pget_obj_track_times` on the property list it was created with, read
4519    /// back from the header that answers it; see
4520    /// [`ObjectHeader::recorded_times`](crate::format::object_header::ObjectHeader::recorded_times).
4521    pub fn object_records_times(&mut self, path: &str) -> IoResult<bool> {
4522        let addr = self.object_header_address(path)?;
4523        Ok(crate::io::object_header_io::read_header_recorded_times(
4524            &mut self.handle,
4525            &self.meta,
4526            addr,
4527        )?
4528        .is_some())
4529    }
4530
4531    /// The object header address `path` names.
4532    ///
4533    /// By name first, then by address: `object_paths` keeps one path per
4534    /// object header, so a hard link — two names, one header — is only ever
4535    /// found under whichever name the walk reached first.
4536    fn object_header_address(&mut self, path: &str) -> IoResult<u64> {
4537        if self.external_edge(path).is_some() {
4538            return Err(crate::io::IoError::NotFound(format!(
4539                "{path} is in another file; its header is not this file's to read"
4540            )));
4541        }
4542        match self.dataset_info(path) {
4543            Some(info) => Ok(info.object_header_address),
4544            None => {
4545                let want = absolute_path(&self.canonical_path(path));
4546                self.object_paths
4547                    .iter()
4548                    .find(|(_, p)| **p == want)
4549                    .map(|(addr, _)| *addr)
4550                    .ok_or_else(|| crate::io::IoError::NotFound(path.to_string()))
4551            }
4552        }
4553    }
4554
4555    /// Read a reference dataset's elements, resolved to the objects they name.
4556    pub fn read_references(&mut self, name: &str) -> IoResult<Vec<Reference>> {
4557        let datatype = self
4558            .dataset_info(name)
4559            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?
4560            .datatype
4561            .clone();
4562        let raw = self.read_dataset_raw(name)?;
4563        self.decode_references(&datatype, &raw)
4564    }
4565
4566    /// Read an attribute's value as reference elements.
4567    pub fn attr_references(&mut self, attr: &AttributeMessage) -> IoResult<Vec<Reference>> {
4568        self.decode_references(&attr.datatype, &attr.data)
4569    }
4570
4571    /// The single owner of reference decoding for both carriers of reference
4572    /// elements — dataset payloads and attribute values.
4573    fn decode_references(
4574        &mut self,
4575        datatype: &DatatypeMessage,
4576        bytes: &[u8],
4577    ) -> IoResult<Vec<Reference>> {
4578        let DatatypeMessage::Reference { size, kind } = datatype else {
4579            return Err(crate::io::IoError::InvalidState(format!(
4580                "datatype {datatype} is not a reference"
4581            )));
4582        };
4583        let (size, kind) = (*size as usize, *kind);
4584        if size == 0 {
4585            // A corrupt file can declare it; `chunks_exact(0)` panics.
4586            return Err(crate::io::IoError::InvalidState(
4587                "reference datatype has zero width".into(),
4588            ));
4589        }
4590        let encoding = kind.encoding();
4591
4592        // References written by one call share one heap collection, so read
4593        // each collection once rather than per element.
4594        let mut heaps = std::collections::HashMap::new();
4595        let mut out = Vec::with_capacity(bytes.len() / size);
4596        for elem in bytes.chunks_exact(size) {
4597            out.push(self.decode_reference_element(elem, encoding, &mut heaps)?);
4598        }
4599        Ok(out)
4600    }
4601
4602    /// One reference element, resolved against the file.
4603    ///
4604    /// `heaps` caches the global-heap collections region references point
4605    /// into, keyed by collection address.
4606    fn decode_reference_element(
4607        &mut self,
4608        elem: &[u8],
4609        encoding: ReferenceEncoding,
4610        heaps: &mut std::collections::HashMap<u64, GlobalHeapCollection>,
4611    ) -> IoResult<Reference> {
4612        match encoding {
4613            ReferenceEncoding::Old(OldReferenceKind::Object) => {
4614                match decode_object_element(elem, &self.meta.ctx)? {
4615                    None => Ok(Reference::Null),
4616                    Some(address) => Ok(self.resolve_reference(DecodedReference {
4617                        address,
4618                        file: None,
4619                        target: ReferenceTarget::Object,
4620                    })),
4621                }
4622            }
4623            ReferenceEncoding::Old(OldReferenceKind::DatasetRegion) => {
4624                let Some((coll_addr, obj_index)) = decode_region_element(elem, &self.meta.ctx)?
4625                else {
4626                    return Ok(Reference::Null);
4627                };
4628                let obj = self.heap_object(coll_addr, obj_index, heaps)?;
4629                let (address, selection) = decode_region_heap_object(obj, &self.meta.ctx)?;
4630                Ok(self.resolve_reference(DecodedReference {
4631                    address,
4632                    file: None,
4633                    target: ReferenceTarget::Region(selection),
4634                }))
4635            }
4636            ReferenceEncoding::Revised => {
4637                let (kind, external, body) = match decode_revised_element(elem, &self.meta.ctx)? {
4638                    RevisedElement::Null => return Ok(Reference::Null),
4639                    RevisedElement::Inline { kind, body } => (kind, false, body.to_vec()),
4640                    RevisedElement::Heap {
4641                        kind,
4642                        external,
4643                        collection,
4644                        index,
4645                    } => (
4646                        kind,
4647                        external,
4648                        self.heap_object(collection, index, heaps)?.to_vec(),
4649                    ),
4650                };
4651                match decode_revised_body(kind, external, &body, &self.meta.ctx)? {
4652                    None => Ok(Reference::Null),
4653                    Some(decoded) => Ok(self.resolve_reference(decoded)),
4654                }
4655            }
4656        }
4657    }
4658
4659    /// Attach the target's path to a decoded reference — the one place an
4660    /// address becomes a [`Reference`], so every kind resolves the same way.
4661    ///
4662    /// A reference naming another file is looked up in that file, which this
4663    /// opens by the name the reference carries and nothing else:
4664    /// `H5R__reopen_file` hands the name straight to `H5VL_file_open` with no
4665    /// prefix search, so it is read against the process working directory the
4666    /// way `H5Ropen_object` would read it (H5Rint.c:466, :487). A file that is
4667    /// not there leaves the path unresolved while the reference still names
4668    /// it, which is `H5Rget_file_name` answering from the reference alone
4669    /// while `H5Ropen_object` fails (H5R.c:1036-1039).
4670    fn resolve_reference(&mut self, decoded: DecodedReference) -> Reference {
4671        let DecodedReference {
4672            address,
4673            file,
4674            target,
4675        } = decoded;
4676        let path = match &file {
4677            None => self.path_for_object(address).map(str::to_string),
4678            Some(name) => self
4679                .cross_file(PathBuf::from(name), CrossFileOwner::Reader)
4680                .ok()
4681                .and_then(|target| target.path_for_object(address).map(str::to_string)),
4682        };
4683        match target {
4684            ReferenceTarget::Object => Reference::Object {
4685                address,
4686                file,
4687                path,
4688            },
4689            ReferenceTarget::Region(selection) => Reference::Region {
4690                address,
4691                file,
4692                path,
4693                selection,
4694            },
4695            ReferenceTarget::Attribute(name) => Reference::Attr {
4696                address,
4697                file,
4698                path,
4699                name,
4700            },
4701        }
4702    }
4703
4704    /// One global-heap object, reading its collection at most once.
4705    fn heap_object<'h>(
4706        &mut self,
4707        collection: u64,
4708        index: u32,
4709        heaps: &'h mut std::collections::HashMap<u64, GlobalHeapCollection>,
4710    ) -> IoResult<&'h [u8]> {
4711        if let std::collections::hash_map::Entry::Vacant(slot) = heaps.entry(collection) {
4712            slot.insert(self.read_heap_collection(collection)?);
4713        }
4714        let idx = u16::try_from(index).map_err(|_| {
4715            crate::io::IoError::InvalidState(format!(
4716                "global heap object index {index} does not fit the 16-bit on-disk field"
4717            ))
4718        })?;
4719        heaps[&collection].get_object(idx).ok_or_else(|| {
4720            crate::io::IoError::InvalidState(format!(
4721                "global heap object {idx} not found in the collection at address {collection:#x}"
4722            ))
4723        })
4724    }
4725
4726    /// Return the dimensions of a dataset.
4727    pub fn dataset_shape(&mut self, name: &str) -> IoResult<Vec<u64>> {
4728        let info = self
4729            .dataset_info(name)
4730            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4731        Ok(info.dataspace.dims.clone())
4732    }
4733
4734    /// Logical byte size of a dataset's full image (`product(dims) *
4735    /// element_size`), with the datatype needed for the post-filter conversion.
4736    fn raw_size_and_datatype(&self, name: &str) -> IoResult<(DatatypeMessage, u64)> {
4737        let info = self
4738            .dataset_info_local(name)
4739            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4740        Ok((info.datatype.clone(), Self::raw_size_of(info)))
4741    }
4742
4743    /// Logical byte size of `info`'s full image.
4744    ///
4745    /// The NULL dataspace (`dataspace.is_null()`) holds zero elements — not
4746    /// one, the way an empty `dims` would suggest by the same product-of-dims
4747    /// arithmetic a scalar dataspace uses (`dims` is empty for both).
4748    fn raw_size_of(info: &DatasetReadInfo) -> u64 {
4749        if info.dataspace.is_null() {
4750            0
4751        } else {
4752            saturating_byte_len(&info.dataspace.dims, info.datatype.element_size() as u64)
4753        }
4754    }
4755
4756    /// Logical byte size of a dataset's full image: how many bytes
4757    /// [`read_dataset_raw`](Self::read_dataset_raw) returns, and how large a
4758    /// buffer [`read_dataset_raw_into`](Self::read_dataset_raw_into) needs.
4759    ///
4760    /// Resolved in the file that owns the dataset, so a name crossing an
4761    /// external link answers with the target's size rather than an absence.
4762    pub fn dataset_raw_size(&mut self, name: &str) -> IoResult<u64> {
4763        if self.external_edge(name).is_some() {
4764            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4765            return owner.dataset_raw_size(&path);
4766        }
4767        let info = self
4768            .dataset_info_local(name)
4769            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4770        Ok(Self::raw_size_of(info))
4771    }
4772
4773    /// Everything a zero-copy view of `name` needs to know: the map of the
4774    /// file that owns the dataset, where in that file the dataset's image
4775    /// lies, and how its elements are stored.
4776    ///
4777    /// Facts only — whether they add up to a view `T` may be handed is
4778    /// [`crate::mapped::view`]'s decision, which is the single place that
4779    /// weighs them. Resolved in the file that owns the dataset, so a name
4780    /// crossing an external link answers with the target's map and the
4781    /// target's addresses rather than this file's.
4782    #[cfg(feature = "mmap")]
4783    pub(crate) fn dataset_view_source(&mut self, name: &str) -> IoResult<DatasetViewSource> {
4784        if self.external_edge(name).is_some() {
4785            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4786            return owner.dataset_view_source(&path);
4787        }
4788        let base = self.handle.base();
4789        let map = self.handle.map_snapshot();
4790        let info = self
4791            .dataset_info_local(name)
4792            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4793        let len = Self::raw_size_of(info);
4794        let storage = match &info.layout {
4795            // An external file list overrides contiguous storage: the layout
4796            // still says `Contiguous`, but the bytes are in other files
4797            // (H5Dlayout.c swaps the storage ops out whenever the message is
4798            // present), so nothing in this map holds them.
4799            DataLayoutMessage::Contiguous { .. } if !info.external_files.is_empty() => {
4800                ViewStorage::Elsewhere("its raw data is in external data files")
4801            }
4802            DataLayoutMessage::Contiguous { address, .. } if *address == UNDEF_ADDR => {
4803                ViewStorage::Unallocated
4804            }
4805            DataLayoutMessage::Contiguous { address, .. } => {
4806                let offset = address.checked_add(base).ok_or_else(|| {
4807                    crate::io::IoError::InvalidState(format!(
4808                        "dataset '{name}' claims raw data at {address}, which overflows \
4809                         past the userblock at {base}"
4810                    ))
4811                })?;
4812                ViewStorage::Contiguous { offset, len }
4813            }
4814            DataLayoutMessage::Compact { .. } => {
4815                ViewStorage::Elsewhere("its raw data is compact, stored inside the object header")
4816            }
4817            DataLayoutMessage::ChunkedV3 { .. } | DataLayoutMessage::ChunkedV4 { .. } => {
4818                ViewStorage::Elsewhere("it is chunked")
4819            }
4820            DataLayoutMessage::Virtual { .. } => {
4821                ViewStorage::Elsewhere("it is virtual, mapped from other datasets")
4822            }
4823        };
4824        Ok(DatasetViewSource {
4825            map,
4826            storage,
4827            datatype: info.datatype.clone(),
4828            dims: info.dataspace.dims.clone(),
4829        })
4830    }
4831
4832    /// Read the raw bytes of a dataset.
4833    pub fn read_dataset_raw(&mut self, name: &str) -> IoResult<Vec<u8>> {
4834        if self.external_edge(name).is_some() {
4835            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4836            return owner.read_dataset_raw(&path);
4837        }
4838        let (datatype, total) = self.raw_size_and_datatype(name)?;
4839        read_image_into_new(total as usize, |data| {
4840            self.read_dataset_raw_into_unconverted(name, data)?;
4841            Self::apply_post_filter_conversion(data, &datatype)
4842        })
4843    }
4844
4845    /// Read the full raw dataset image into a caller-provided buffer.
4846    ///
4847    /// `out.len()` must equal the dataset's logical byte size
4848    /// (`product(dims) * element_size`); otherwise an error is returned. This
4849    /// is the no-allocation counterpart of [`read_dataset_raw`](Self::read_dataset_raw):
4850    /// the bytes are read straight into `out`, making it the zero-copy entry
4851    /// point for reading directly into a pinned/registered host buffer for an
4852    /// H2D transfer.
4853    pub fn read_dataset_raw_into(&mut self, name: &str, out: &mut [u8]) -> IoResult<()> {
4854        if self.external_edge(name).is_some() {
4855            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
4856            return owner.read_dataset_raw_into(&path, out);
4857        }
4858        let (datatype, total) = self.raw_size_and_datatype(name)?;
4859        if out.len() as u64 != total {
4860            return Err(crate::io::IoError::InvalidState(format!(
4861                "read_dataset_raw_into: buffer is {} bytes but dataset needs {}",
4862                out.len(),
4863                total
4864            )));
4865        }
4866        self.read_dataset_raw_into_unconverted(name, out)?;
4867        Self::apply_post_filter_conversion(out, &datatype)?;
4868        Ok(())
4869    }
4870
4871    /// Fill `out` with the full raw dataset image, before the post-filter
4872    /// datatype conversion. The single owner of read-destination semantics for
4873    /// full reads: it fully defines every byte of `out` (reading allocated data
4874    /// straight in, pre-filling chunked or never-written regions with the tiled
4875    /// fill value), so callers supply only a correctly-sized buffer. Both the
4876    /// allocating `read_dataset_raw` and the zero-copy `read_dataset_raw_into`
4877    /// wrap it and apply the conversion exactly once.
4878    ///
4879    /// `out.len()` must equal `product(dims) * element_size`.
4880    fn read_dataset_raw_into_unconverted(&mut self, name: &str, out: &mut [u8]) -> IoResult<()> {
4881        let info = self
4882            .dataset_info_local(name)
4883            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
4884
4885        // Clone to avoid borrow conflict with &mut self in read methods.
4886        let layout = info.layout.clone();
4887        let pipeline = info.filter_pipeline.clone();
4888        let fill_value = info.fill_value.clone();
4889        let external_files = info.external_files.clone();
4890
4891        match &layout {
4892            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
4893                let prefix = self.extfile_prefix_in_force(name);
4894                read_external_file_bytes(&external_files, prefix.as_deref(), 0, out)?;
4895            }
4896            DataLayoutMessage::Contiguous { address, .. } => {
4897                if *address == UNDEF_ADDR {
4898                    // Never-written contiguous data reads back as the fill value.
4899                    fill_tiled_into(out, fill_value.as_deref());
4900                } else {
4901                    // Read exactly the logical image straight into `out`.
4902                    self.handle.read_exact_at_into(*address, out)?;
4903                }
4904            }
4905            DataLayoutMessage::Compact { data } => {
4906                let n = out.len().min(data.len());
4907                out[..n].copy_from_slice(&data[..n]);
4908                if n < out.len() {
4909                    fill_tiled_into(&mut out[n..], fill_value.as_deref());
4910                }
4911            }
4912            DataLayoutMessage::ChunkedV3 {
4913                chunk_dims,
4914                b_tree_address,
4915            } => {
4916                // The layout's chunk_dims include the element size as the
4917                // trailing dimension. Strip it for chunk indexing. The chunk
4918                // read defines every byte of `out`, filling whatever no chunk
4919                // covers.
4920                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
4921                self.read_chunked_btree_v1(
4922                    name,
4923                    real_chunk_dims,
4924                    *b_tree_address,
4925                    ChunkReadRequest {
4926                        pipeline: pipeline.as_ref(),
4927                        target: ChunkTarget::Full,
4928                        fill_value: fill_value.as_deref(),
4929                    },
4930                    out,
4931                )?;
4932            }
4933            DataLayoutMessage::ChunkedV4 {
4934                chunk_dims,
4935                index_address,
4936                index_type,
4937                earray_params,
4938                single_chunk_filter,
4939                ..
4940            } => {
4941                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
4942                self.read_chunked_v4(
4943                    name,
4944                    real_chunk_dims,
4945                    ChunkIndexDesc {
4946                        index_type: *index_type,
4947                        index_address: *index_address,
4948                        earray_params: earray_params.as_ref(),
4949                        single_chunk_filter: *single_chunk_filter,
4950                    },
4951                    ChunkReadRequest {
4952                        pipeline: pipeline.as_ref(),
4953                        target: ChunkTarget::Full,
4954                        fill_value: fill_value.as_deref(),
4955                    },
4956                    out,
4957                )?;
4958            }
4959            DataLayoutMessage::Virtual { .. } => {
4960                fill_tiled_into(out, fill_value.as_deref());
4961                self.read_virtual_into(name, out, 0)?;
4962            }
4963        }
4964        Ok(())
4965    }
4966
4967    /// Set every virtual dataset's extent from the sources its unlimited
4968    /// mappings can reach — `H5D__virtual_set_extent_unlim` (H5Dvirtual.c),
4969    /// which libhdf5 runs when it opens such a dataset.
4970    ///
4971    /// INVARIANT: a virtual dataset's `dataspace.dims` are the extent its
4972    /// available sources give it. This is the single owner of that
4973    /// resolution: the dims a VDS with an unlimited mapping reports are not
4974    /// the ones its dataspace message stores, and every shape query, read and
4975    /// slice bound must see the same value, so the resolved extent is stamped
4976    /// in once — here, immediately after a catalog is built — rather than
4977    /// recomputed per call. The per-mapping clip sizes the extent came from
4978    /// are kept beside it in
4979    /// [`virtual_resolution`](DatasetReadInfo::virtual_resolution) so a read
4980    /// walks exactly the sources the extent was derived from.
4981    ///
4982    /// A source that cannot be opened contributes a clip size of 0, never an
4983    /// error: a virtual dataset whose sources are not written yet is legal,
4984    /// and reads back as the fill value (upstream's "clip_size = 0" arm when
4985    /// `H5D__virtual_open_source_dset` leaves the dataset closed).
4986    ///
4987    /// The default view is assumed throughout — `H5D_VDS_LAST_AVAILABLE` is
4988    /// `H5D_ACS_VDS_VIEW_DEF`, and `H5Pset_virtual_view` sets a *dataset
4989    /// access* property that is never stored in the file, so a reader opening
4990    /// a file it did not create always sees the default.
4991    fn resolve_virtual_extents(&mut self) -> IoResult<()> {
4992        let targets: Vec<usize> = self
4993            .datasets
4994            .iter()
4995            .enumerate()
4996            .filter(|(_, d)| d.virtual_mappings.is_some())
4997            .map(|(i, _)| i)
4998            .collect();
4999        for i in targets {
5000            // The catalog is freshly built, so `dataspace.dims` is still the
5001            // extent the dataspace message stores. Record it before the
5002            // resolution replaces it: that is what a later open under other
5003            // access properties has to resolve from.
5004            let stored = self.datasets[i].dataspace.dims.clone();
5005            self.datasets.entry_mut(i).virtual_stored_dims = Some(stored.clone());
5006            self.resolve_virtual_extent_of(i, &stored)?;
5007        }
5008        // This resolution belongs to no dataset open: libhdf5 runs its
5009        // equivalent from `H5D__virtual_init` at `H5Dopen` (H5Dvirtual.c:2178),
5010        // where the open that asked for it holds the source, while this one
5011        // runs at `H5Fopen` and at a SWMR refresh. Whatever it opened is
5012        // therefore unowned the moment it is done — and libhdf5 measured on
5013        // the same file has no source file open after `H5Fopen` either.
5014        self.release_closed_virtual_sources();
5015        Ok(())
5016    }
5017
5018    /// Resolve one virtual dataset's extent from its *stored* dims under the
5019    /// [`DatasetAccess`] in force for it, and stamp both the extent and the
5020    /// per-mapping resolutions in.
5021    fn resolve_virtual_extent_of(&mut self, i: usize, stored: &[u64]) -> IoResult<()> {
5022        let Some(mappings) = self.datasets[i].virtual_mappings.clone() else {
5023            return Ok(());
5024        };
5025        let vds = self.datasets[i].name.clone();
5026        let access = self.access_in_force(&vds);
5027        let (resolution, dims) =
5028            self.resolve_one_virtual_extent(&vds, &mappings, stored, &access)?;
5029        let entry = self.datasets.entry_mut(i);
5030        entry.dataspace.dims = dims;
5031        entry.virtual_resolution = Some(resolution);
5032        Ok(())
5033    }
5034
5035    /// The dataset-access properties in force for `name` (already canonical),
5036    /// libhdf5's defaults when no open has named others.
5037    fn access_in_force(&self, canonical: &str) -> DatasetAccess {
5038        self.dataset_access
5039            .get(canonical)
5040            .map(|e| e.access.clone())
5041            .unwrap_or_default()
5042    }
5043
5044    /// Open `name` under `access`: put those properties in force and
5045    /// re-resolve its extent under them, or — when the dataset already has a
5046    /// live handle — join that open and drop `access` on the floor.
5047    ///
5048    /// First open wins, which is `H5D_open`'s own rule. Only the open that
5049    /// finds no shared info for the dataset runs `H5D__open_oid(dataset,
5050    /// dapl_id)` and so reaches `H5D__virtual_init`, where the view and the
5051    /// printf gap are read out of the dapl into the *shared* layout storage
5052    /// (H5Dvirtual.c:2178-2188); an open that finds the dataset in
5053    /// `H5FO_opened` just points at that shared info and increments its
5054    /// count, never looking at its own dapl at all (H5Dint.c:1496-1500,
5055    /// :1523-1528). The shared info goes away with the last handle, so the
5056    /// next open after that resolves afresh. Measured against libhdf5 1.14.6
5057    /// and 2.0.0: a second open of a printf-gap VDS with a different gap
5058    /// reports the first open's extent and reads the first open's data —
5059    /// even through a second `H5Fopen` of the same file — and only once
5060    /// every handle is closed does a new open see its own gap.
5061    ///
5062    /// The one exception to "first open wins" is the external file prefix,
5063    /// which the joining open is not allowed to disagree about:
5064    /// `H5D__open_name` compares its own expanded prefix against the open
5065    /// dataset's and fails the open when they differ (H5Dint.c:1533-1545).
5066    /// Expanded, so two opens differing only in a property
5067    /// `HDF5_EXTFILE_PREFIX` shadows still agree — measured under libhdf5
5068    /// 1.14.6 and 2.0.0: with that variable set, an open naming no prefix
5069    /// joins one that named a directory, and without it the same pair is
5070    /// refused.
5071    ///
5072    /// Returns the token that keeps the open alive; the caller hands it to
5073    /// the dataset handle it builds. A name no dataset in this file answers
5074    /// to takes nothing and returns `None`.
5075    fn apply_dataset_access(
5076        &mut self,
5077        name: &str,
5078        access: &DatasetAccess,
5079    ) -> IoResult<Option<DatasetOpenToken>> {
5080        let canonical = self.canonical_path(name);
5081        let Some(i) = self.datasets.position(&canonical) else {
5082            return Ok(None);
5083        };
5084        if let Some(open) = self
5085            .dataset_access
5086            .get(&canonical)
5087            .and_then(|e| e.open.upgrade())
5088        {
5089            let in_force = self.extfile_prefix_of(&self.access_in_force(&canonical));
5090            if in_force != self.extfile_prefix_of(access) {
5091                return Err(crate::io::IoError::InvalidState(format!(
5092                    "dataset {canonical:?} is already open under a different external file \
5093                     prefix, and libhdf5 refuses to join an open that disagrees about one"
5094                )));
5095            }
5096            return Ok(Some(open));
5097        }
5098        let token: DatasetOpenToken = std::sync::Arc::new(());
5099        let unchanged = &self.access_in_force(&canonical) == access;
5100        let access = access.clone();
5101        self.dataset_access.insert(
5102            canonical,
5103            AccessInForce {
5104                access,
5105                open: std::sync::Arc::downgrade(&token),
5106            },
5107        );
5108        if !unchanged {
5109            if let Some(stored) = self.datasets[i].virtual_stored_dims.clone() {
5110                // A source may be a virtual dataset in this same file, and the
5111                // access propagates to it (H5Dvirtual.c:2224-2226), so this can
5112                // re-enter; the depth counter is the same cycle guard the
5113                // open-time resolution uses.
5114                VirtualResolveDepth::enter(|| self.resolve_virtual_extent_of(i, &stored))?;
5115            }
5116        }
5117        Ok(Some(token))
5118    }
5119
5120    /// The directory an external file list's stored names are joined against
5121    /// under `access` — `H5D__build_file_prefix(dset, H5F_PREFIX_EFILE)`
5122    /// (H5Dint.c:1084-1090), whose answer libhdf5 keeps in
5123    /// `dset->shared->extfile_prefix`.
5124    fn extfile_prefix_of(&self, access: &DatasetAccess) -> Option<PathBuf> {
5125        resolve_extfile_prefix(access.efile_prefix_value(), &self.source_dir)
5126    }
5127
5128    /// The external file prefix in force for the open dataset `name` — the
5129    /// one the open that is still holding it named, not whatever a later
5130    /// caller might have asked for
5131    /// ([`apply_dataset_access`](Self::apply_dataset_access)).
5132    fn extfile_prefix_in_force(&self, name: &str) -> Option<PathBuf> {
5133        let canonical = self.canonical_path(name);
5134        self.extfile_prefix_of(&self.access_in_force(&canonical))
5135    }
5136
5137    /// [`resolve_virtual_extents`](Self::resolve_virtual_extents) for one
5138    /// dataset: the per-mapping resolutions and the extent they imply.
5139    fn resolve_one_virtual_extent(
5140        &mut self,
5141        vds: &str,
5142        mappings: &VirtualMappingList,
5143        curr_dims: &[u64],
5144        access: &DatasetAccess,
5145    ) -> IoResult<(Vec<MappingResolution>, Vec<u64>)> {
5146        let rank = curr_dims.len();
5147        let mut resolution = Vec::with_capacity(mappings.mappings.len());
5148        let mut new_dims: Vec<Option<u64>> = vec![None; rank];
5149        // `H5D_virtual_update_min_dims`: whatever the unlimited dimension
5150        // resolves to, the extent must still hold every bounded mapping.
5151        let mut min_dims = vec![0u64; rank];
5152        // `H5S_hyper_get_clip_extent_match`'s `incl_trail`: a
5153        // `H5D_VDS_FIRST_MISSING` view stops where the trailing partial
5154        // block would begin (H5Dvirtual.c:1447-1451).
5155        let incl_trail = access.view() == VirtualView::FirstMissing;
5156        // Where two mappings disagree about the unlimited dimension,
5157        // `H5D_VDS_FIRST_MISSING` takes the smallest clip and
5158        // `H5D_VDS_LAST_AVAILABLE` the largest (H5Dvirtual.c:1662-1667).
5159        let take_clip = |slot: &mut Option<u64>, clip: u64| {
5160            if slot.is_none_or(|d| if incl_trail { clip < d } else { clip > d }) {
5161                *slot = Some(clip);
5162            }
5163        };
5164
5165        for m in &mappings.mappings {
5166            let unlim_virtual = m.virtual_selection.unlim_dim();
5167            let res = match (unlim_virtual, m.source_selection.unlim_dim()) {
5168                (Some(vd), Some(sd)) => {
5169                    let source_clip = self
5170                        .virtual_source_dims(vds, m, access)
5171                        .ok()
5172                        .flatten()
5173                        .and_then(|d| d.get(sd).copied())
5174                        .unwrap_or(0);
5175                    let virtual_clip = match (
5176                        regular_hyperslab(&m.virtual_selection),
5177                        regular_hyperslab(&m.source_selection),
5178                    ) {
5179                        // `H5S_hyper_get_clip_extent_match`: how many slices
5180                        // the source supplies, then the virtual extent that
5181                        // covers exactly that many. Its `incl_trail`
5182                        // argument is `view == H5D_VDS_FIRST_MISSING`
5183                        // (H5Dvirtual.c:1447-1451).
5184                        (Some(v), Some(sr)) => {
5185                            v.clip_extent(sr.num_slices(source_clip), incl_trail)
5186                        }
5187                        _ => 0,
5188                    };
5189                    take_clip(&mut new_dims[vd], virtual_clip);
5190                    MappingResolution::Unlimited {
5191                        virtual_clip,
5192                        source_clip,
5193                    }
5194                }
5195                // Unlimited virtual selection, limited source selection:
5196                // the printf shape, where the successive blocks of the
5197                // virtual selection come from successively-named source
5198                // datasets.
5199                (Some(vd), None) => {
5200                    let (blocks, present) = self.printf_blocks_present(vds, m, access);
5201                    let virtual_clip = match (blocks, regular_hyperslab(&m.virtual_selection)) {
5202                        // `H5D__virtual_set_extent_unlim`'s "check for no
5203                        // datasets" arm, which is 0 under either view
5204                        // (H5Dvirtual.c:1623-1626).
5205                        (0, _) | (_, None) => 0,
5206                        // The extent ends just past the last block that has
5207                        // a source under `H5D_VDS_LAST_AVAILABLE`, and where
5208                        // the first missing block starts under
5209                        // `H5D_VDS_FIRST_MISSING` (H5Dvirtual.c:1630-1653).
5210                        (n, Some(r)) => match access.view() {
5211                            VirtualView::LastAvailable => {
5212                                let last = r.unlim_block(n - 1);
5213                                last.start[vd] + last.block[vd]
5214                            }
5215                            VirtualView::FirstMissing => r.unlim_block(n).start[vd],
5216                        },
5217                    };
5218                    take_clip(&mut new_dims[vd], virtual_clip);
5219                    MappingResolution::Printf { blocks, present }
5220                }
5221                _ => MappingResolution::Bounded,
5222            };
5223            if let Some((_, hi)) = m.virtual_selection.bounds() {
5224                for (d, &e) in hi.iter().enumerate().take(rank) {
5225                    if Some(d) != unlim_virtual && e + 1 > min_dims[d] {
5226                        min_dims[d] = e + 1;
5227                    }
5228                }
5229            }
5230            resolution.push(res);
5231        }
5232
5233        let dims = (0..rank)
5234            .map(|d| match new_dims[d] {
5235                Some(v) => v.max(min_dims[d]),
5236                None => curr_dims[d],
5237            })
5238            .collect();
5239        Ok((resolution, dims))
5240    }
5241
5242    /// The extent of the source dataset one mapping names, or `None` when it
5243    /// cannot be reached — `H5D__virtual_open_source_dset` leaving the source
5244    /// closed, which upstream reads as "no data there yet" rather than an
5245    /// error.
5246    fn virtual_source_dims(
5247        &mut self,
5248        vds: &str,
5249        m: &VirtualMapping,
5250        access: &DatasetAccess,
5251    ) -> IoResult<Option<Vec<u64>>> {
5252        let m = built_names(m, 0)?;
5253        Ok(self.source_dims(vds, &m.source_file_name, &m.source_dset_name, access))
5254    }
5255
5256    /// A printf mapping's `first_missing` and the blocks below it that
5257    /// actually have a source — upstream's search loop in
5258    /// `H5D__virtual_set_extent_unlim` (H5Dvirtual.c:1519-1614), which stops
5259    /// at the first block whose source cannot be opened and looks
5260    /// [`DatasetAccess::virtual_printf_gap`] blocks past it before giving up.
5261    ///
5262    /// The loop bound is upstream's `j <= printf_gap + first_missing`
5263    /// rearranged so a large gap cannot overflow the sum: `first_missing` is
5264    /// never above `j` when the test runs, because it only ever becomes the
5265    /// *previous* `j` plus one.
5266    fn printf_blocks_present(
5267        &mut self,
5268        vds: &str,
5269        m: &VirtualMapping,
5270        access: &DatasetAccess,
5271    ) -> (u64, Vec<u64>) {
5272        let gap = access.effective_printf_gap();
5273        let mut first_missing = 0u64;
5274        let mut present = Vec::new();
5275        let mut j = 0u64;
5276        while j - first_missing <= gap {
5277            let Ok(built) = built_names(m, j) else {
5278                break;
5279            };
5280            if self
5281                .source_dims(
5282                    vds,
5283                    &built.source_file_name,
5284                    &built.source_dset_name,
5285                    access,
5286                )
5287                .is_some()
5288            {
5289                first_missing = j + 1;
5290                present.push(j);
5291            }
5292            j += 1;
5293        }
5294        (first_missing, present)
5295    }
5296
5297    /// The extent of one named source dataset, or `None` when the file or
5298    /// the dataset in it cannot be opened.
5299    ///
5300    /// `access` is the virtual dataset's own: `H5D__virtual_init` copies the
5301    /// dapl into the layout as `source_dapl` (H5Dvirtual.c:2224-2226) and
5302    /// every source is opened with it (H5Dvirtual.c:901-902), so a source
5303    /// that is itself a virtual dataset resolves under the same view and
5304    /// printf gap.
5305    fn source_dims(
5306        &mut self,
5307        vds: &str,
5308        file_name: &str,
5309        dset_name: &str,
5310        access: &DatasetAccess,
5311    ) -> Option<Vec<u64>> {
5312        let dset_name = dset_name.trim_start_matches('/');
5313        if file_name == "." {
5314            self.apply_dataset_access(dset_name, access).ok()?;
5315            return self
5316                .dataset_info_local(dset_name)
5317                .map(|i| i.dataspace.dims.clone());
5318        }
5319        let reader = self.vds_source_file(vds, file_name)?;
5320        reader.apply_dataset_access(dset_name, access).ok()?;
5321        reader
5322            .dataset_info(dset_name)
5323            .map(|i| i.dataspace.dims.clone())
5324    }
5325
5326    /// Fill `out` (shaped like the virtual dataset's own extent) by
5327    /// stitching each mapping's source bytes in order (`H5D__virtual_read`,
5328    /// H5Dvirtual.c). `out` must already be pre-filled with the tiled fill
5329    /// value — every element no mapping covers is left exactly as the
5330    /// caller filled it. Mappings apply in list order, so a later mapping's
5331    /// bytes win over an earlier one's on overlap, exactly like the C
5332    /// reader; an unlimited or printf mapping has already been replaced by
5333    /// the concrete mappings its open-time resolution made it
5334    /// (`H5D_VDS_LAST_AVAILABLE`, the default view — see
5335    /// [`Hdf5Reader::resolve_virtual_extents`]).
5336    ///
5337    /// A mapping whose source cannot be opened — the file is absent, or the
5338    /// dataset is not in it — contributes nothing and leaves its virtual
5339    /// region at the fill value, rather than failing the read.
5340    /// `H5D__virtual_open_source_dset` treats both as "no data there yet":
5341    /// it asks `H5F_prefix_open_file` to *try* the file and accepts a null
5342    /// one, and clears the error stack when the dataset is missing
5343    /// (H5Dvirtual.c:877-909); `H5D__virtual_read_one` then performs I/O
5344    /// "only ... if there is a projected memory space, otherwise there were
5345    /// no elements in the projection or the source dataset could not be
5346    /// opened" (H5Dvirtual.c:2661-2665).
5347    ///
5348    /// `depth` counts virtual-dataset nesting — a mapping whose source is
5349    /// itself a virtual dataset, possibly in another file — so a crafted
5350    /// cyclic mapping chain fails cleanly instead of recursing until the
5351    /// stack overflows.
5352    fn read_virtual_into(&mut self, name: &str, out: &mut [u8], depth: usize) -> IoResult<()> {
5353        if depth >= MAX_VIRTUAL_DEPTH {
5354            return Err(crate::io::IoError::InvalidState(format!(
5355                "dataset {name:?}: virtual dataset mapping nests {MAX_VIRTUAL_DEPTH} levels \
5356                 deep, aborting (possible cyclic mapping)"
5357            )));
5358        }
5359        let info = self
5360            .dataset_info(name)
5361            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
5362        let dims = info.dataspace.dims.clone();
5363        let element_size = info.datatype.element_size() as u64;
5364        let Some(mappings) = info.virtual_mappings.clone() else {
5365            // No mapping list written yet: every element is unmapped, and
5366            // `out` is already the fill value the caller pre-filled it with.
5367            return Ok(());
5368        };
5369        // Every unlimited mapping is replaced by the concrete one its
5370        // open-time resolution makes it, so the walk below only ever sees
5371        // bounded selections.
5372        let resolution = info.virtual_resolution.clone().unwrap_or_default();
5373        let mappings = concrete_virtual_mappings(&mappings, &resolution)?;
5374        // The same properties the extent resolved under reach every source
5375        // (H5Dvirtual.c:2224-2226, :901-902), so a source that is itself a
5376        // virtual dataset is read the same way this one is.
5377        let canonical = self.canonical_path(name);
5378        let access = self.access_in_force(&canonical);
5379
5380        for mapping in &mappings {
5381            let virtual_sel = mapping.virtual_selection.resolve(&dims).map_err(|e| {
5382                crate::io::IoError::InvalidState(format!(
5383                    "dataset {name:?}: virtual mapping's virtual selection is not \
5384                     supported: {e}"
5385                ))
5386            })?;
5387            if virtual_sel.runs.is_empty() {
5388                continue;
5389            }
5390
5391            let source_name = mapping.source_dset_name.trim_start_matches('/');
5392
5393            if mapping.source_file_name == "." {
5394                // `H5D__virtual_open_source_dset` opens the source for the
5395                // read and `H5D__virtual_reset_source_dset` closes it again,
5396                // so this open holds nothing past the mapping — dropping the
5397                // token is what that close does.
5398                self.apply_dataset_access(source_name, &access)?;
5399                let Some(src_dims) = self
5400                    .dataset_info(source_name)
5401                    .map(|i| i.dataspace.dims.clone())
5402                else {
5403                    continue;
5404                };
5405                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
5406                    crate::io::IoError::InvalidState(format!(
5407                        "dataset {name:?}: virtual mapping's source selection is not \
5408                         supported: {e}"
5409                    ))
5410                })?;
5411                copy_matched_selections(
5412                    |s, c, buf| self.read_slice_into_unconverted(source_name, s, c, buf, depth + 1),
5413                    &source_sel,
5414                    &virtual_sel,
5415                    element_size,
5416                    out,
5417                )?;
5418            } else {
5419                let Some(src_reader) = self.vds_source_file(&canonical, &mapping.source_file_name)
5420                else {
5421                    continue;
5422                };
5423                src_reader.apply_dataset_access(source_name, &access)?;
5424                let Some(src_dims) = src_reader
5425                    .dataset_info(source_name)
5426                    .map(|i| i.dataspace.dims.clone())
5427                else {
5428                    continue;
5429                };
5430                let source_sel = mapping.source_selection.resolve(&src_dims).map_err(|e| {
5431                    crate::io::IoError::InvalidState(format!(
5432                        "dataset {name:?}: virtual mapping's source selection is not \
5433                         supported: {e}"
5434                    ))
5435                })?;
5436                copy_matched_selections(
5437                    |s, c, buf| {
5438                        src_reader.read_slice_into_unconverted(source_name, s, c, buf, depth + 1)
5439                    },
5440                    &source_sel,
5441                    &virtual_sel,
5442                    element_size,
5443                    out,
5444                )?;
5445            }
5446        }
5447        Ok(())
5448    }
5449
5450    /// Apply the post-filter datatype conversion (libhdf5's `H5T_convert`
5451    /// step) to a fully-decoded output buffer.
5452    ///
5453    /// For N-bit / reduced-precision `FixedPoint` datatypes the filter
5454    /// pipeline leaves the significant value occupying `bit_precision` bits
5455    /// at `bit_offset` within each element, zero-filled and not
5456    /// sign-extended. This rewrites every element so the value occupies the
5457    /// whole element at bit offset 0, sign-extended when signed. It is a
5458    /// no-op for ordinary full-width datatypes.
5459    fn apply_post_filter_conversion(buffer: &mut [u8], datatype: &DatatypeMessage) -> IoResult<()> {
5460        use crate::format::nbit_scaleoffset::{
5461            apply_datatype_conversion, datatype_needs_bit_conversion,
5462        };
5463        if datatype_needs_bit_conversion(datatype) {
5464            apply_datatype_conversion(buffer, datatype)?;
5465        }
5466        Ok(())
5467    }
5468
5469    /// Re-read the superblock and dataset metadata for SWMR.
5470    ///
5471    /// Call this periodically to pick up new data written by a concurrent
5472    /// SWMR writer. The superblock is re-read to get the latest EOF, then
5473    /// the root group is re-scanned for updated dataset headers (which may
5474    /// contain updated dataspace dimensions and chunk index addresses).
5475    pub fn refresh(&mut self) -> IoResult<()> {
5476        // Whatever the handle reads from must cover the file as the SWMR
5477        // writer has left it: a memory map taken at open ends where the file
5478        // ended then, so it is retaken before a byte of the new metadata is
5479        // decoded. Nothing happens for a handle reading through `pread`.
5480        self.handle.refresh_read_source();
5481
5482        // Re-read superblock to get latest EOF and root group address.
5483        let sb_buf = self.handle.read_at_most(0, 256)?;
5484
5485        // Only v2/v3 superblocks support SWMR refresh
5486        let sb = SuperblockV2V3::decode(&sb_buf)?;
5487
5488        let ctx = FormatContext {
5489            sizeof_addr: sb.sizeof_offsets,
5490            sizeof_size: sb.sizeof_lengths,
5491        };
5492
5493        // The superblock extension can also have changed under SWMR (a new
5494        // free-space or shared-message table), so re-read it before the walk.
5495        let (meta, ext) = Self::read_extension_and_meta(
5496            &mut self.handle,
5497            ctx,
5498            self.meta.btree,
5499            sb.superblock_extension_address,
5500        )?;
5501
5502        // Re-read root group object header, following continuation blocks.
5503        let root_header = Self::read_object_header_full(
5504            &mut self.handle,
5505            &meta,
5506            sb.root_group_object_header_address,
5507        )?;
5508
5509        // Re-scan datasets, group attributes, group paths, and link records.
5510        let catalog = Self::build_catalog(
5511            &mut self.handle,
5512            &meta,
5513            Some(&root_header),
5514            sb.root_group_object_header_address,
5515            None,
5516        )?;
5517
5518        // Root link storage from the freshly re-read header, the same way
5519        // `open_v2v3` derives it at open time — SWMR refresh is v2/v3-only,
5520        // so there is no symbol-table scratch-pad to fall back to here either.
5521        let root_link_storage = describe_link_storage(Some(&root_header), &meta.ctx, None);
5522
5523        self._eof = sb.end_of_file_address;
5524        self.meta = meta;
5525        self.ext = ext;
5526        self.object_paths = catalog.object_paths(sb.root_group_object_header_address);
5527        self.datasets = DatasetTable::new(catalog.datasets);
5528        self.unreadable = catalog.unreadable;
5529        self.root_link_storage = root_link_storage;
5530        self.group_attributes = catalog.group_attributes;
5531        self.group_link_storage = catalog.group_link_storage;
5532        self.group_paths = catalog.group_paths;
5533        self.group_aliases = catalog.group_aliases;
5534        self.links = catalog.links;
5535        self.datatypes = catalog.datatypes;
5536        // The catalog is freshly built, so every virtual dataset's resolved
5537        // extent went with the old one — a SWMR refresh is exactly when a
5538        // source may have grown.
5539        self.resolve_virtual_extents()?;
5540
5541        Ok(())
5542    }
5543
5544    /// The dataset at `pos`'s chunk index, from the cache when it already
5545    /// holds one decoded from `index_address`, otherwise by running `decode`
5546    /// once and keeping what it returns.
5547    ///
5548    /// The single owner of the chunk-index cache: nothing else reads or
5549    /// writes it. Every chunked read of a dataset asks the same question of
5550    /// the same on-disk structure — a thousand small slice reads re-walked
5551    /// the fixed array a thousand times — and only a change to the catalog
5552    /// entry can change the answer, which drops the entry's cache
5553    /// (`DatasetTable::entry_mut`, and a SWMR refresh rebuilding the table).
5554    fn decoded_chunk_index<F>(
5555        &mut self,
5556        pos: usize,
5557        index_address: u64,
5558        decode: F,
5559    ) -> IoResult<std::sync::Arc<DecodedChunkIndex>>
5560    where
5561        F: FnOnce(&mut Self) -> IoResult<DecodedChunkIndex>,
5562    {
5563        if let Some(hit) = self.datasets.chunk_index(pos, index_address) {
5564            return Ok(std::sync::Arc::clone(hit));
5565        }
5566        let decoded = decode(self)?;
5567        Ok(self.datasets.cache_chunk_index(pos, index_address, decoded))
5568    }
5569
5570    /// Read chunked dataset data by walking the chunk index.
5571    ///
5572    /// `desc` bundles the version-4 chunk-index descriptor extracted from the
5573    /// data-layout message (kind, address, and per-kind parameters), so this
5574    /// entry point takes one descriptor rather than a long parameter list.
5575    ///
5576    /// Scatters only; `output` must already be sized to the target extent.
5577    /// Every byte of it is defined before this returns `Ok`: what no chunk
5578    /// covers is filled with the tiled fill value by
5579    /// [`place_chunk_jobs`], the one exit every branch here takes.
5580    fn read_chunked_v4(
5581        &mut self,
5582        name: &str,
5583        chunk_dims: &[u64],
5584        desc: ChunkIndexDesc<'_>,
5585        req: ChunkReadRequest,
5586        output: &mut [u8],
5587    ) -> IoResult<()> {
5588        let ChunkReadRequest {
5589            pipeline,
5590            target,
5591            fill_value: _,
5592        } = req;
5593        let ChunkIndexDesc {
5594            index_type,
5595            index_address,
5596            earray_params,
5597            single_chunk_filter,
5598        } = desc;
5599        let pos = self.dataset_position(name)?;
5600        let info = &self.datasets[pos];
5601        let dims = info.dataspace.dims.clone();
5602        let element_size = info.datatype.element_size() as u64;
5603
5604        match index_type {
5605            data_layout::ChunkIndexType::SingleChunk => {
5606                // Single chunk: the index_address IS the chunk address
5607                let total_size: u64 = saturating_byte_len(&dims, element_size);
5608                let geo = ChunkOutputGeometry {
5609                    dims: &dims,
5610                    chunk_dims,
5611                    element_size,
5612                };
5613                if index_address == UNDEF_ADDR || total_size == 0 {
5614                    // Unallocated single chunk: nothing to read, so the empty
5615                    // plan makes the whole output fill.
5616                    return place_chunk_jobs(
5617                        &self.handle,
5618                        Vec::new(),
5619                        &[],
5620                        req,
5621                        &geo,
5622                        None,
5623                        output,
5624                    );
5625                }
5626                // A filtered single chunk records its exact on-disk size and
5627                // per-chunk filter mask in the layout message. Use them to
5628                // read precisely the stored bytes and to skip the filters the
5629                // mask marks as not applied. Without those params
5630                // (older/edge layouts) fall back to the read-extra-and-inflate
5631                // heuristic with the full pipeline.
5632                let job = match (pipeline, single_chunk_filter) {
5633                    (Some(_), Some(scf)) => ChunkReadJob {
5634                        addr: index_address,
5635                        len: scf.nbytes as usize,
5636                        at_most: false,
5637                        mask: scf.filter_mask,
5638                    },
5639                    (Some(_), None) => ChunkReadJob {
5640                        addr: index_address,
5641                        len: total_size.saturating_mul(2) as usize,
5642                        at_most: true,
5643                        mask: 0,
5644                    },
5645                    (None, _) => ChunkReadJob {
5646                        addr: index_address,
5647                        len: total_size as usize,
5648                        at_most: false,
5649                        mask: 0,
5650                    },
5651                };
5652                // The lone chunk spans the whole dataset; place it respecting
5653                // the dataset extent (Full) or the selection (Slice). This
5654                // index type has no on-disk structure to decode — the layout
5655                // message is the index — but it is still recorded through the
5656                // one owner, so that its chunk's image reaches the cache by
5657                // the same route every other index type's chunks do.
5658                let index = self.decoded_chunk_index(pos, index_address, |_| {
5659                    Ok(DecodedChunkIndex::new(
5660                        vec![(job.addr, job.len as u64, job.mask)],
5661                        vec![0u64; dims.len()],
5662                    ))
5663                })?;
5664                place_chunk_jobs(
5665                    &self.handle,
5666                    vec![Some(job)],
5667                    &index.coords,
5668                    req,
5669                    &geo,
5670                    Some(&index.images),
5671                    output,
5672                )
5673            }
5674            data_layout::ChunkIndexType::Implicit => {
5675                self.read_chunked_implicit(name, chunk_dims, index_address, req, output)
5676            }
5677            data_layout::ChunkIndexType::FixedArray => {
5678                self.read_chunked_fixed_array(name, chunk_dims, index_address, req, output)
5679            }
5680            data_layout::ChunkIndexType::BTreeV2 => {
5681                self.read_chunked_btree_v2(name, chunk_dims, index_address, req, output)
5682            }
5683            data_layout::ChunkIndexType::ExtensibleArray => {
5684                let params = earray_params.ok_or_else(|| {
5685                    crate::io::IoError::InvalidState("missing earray params".into())
5686                })?;
5687
5688                if index_address == UNDEF_ADDR {
5689                    // Unallocated: the empty plan makes the whole output fill.
5690                    let geo = ChunkOutputGeometry {
5691                        dims: &dims,
5692                        chunk_dims,
5693                        element_size,
5694                    };
5695                    return place_chunk_jobs(
5696                        &self.handle,
5697                        Vec::new(),
5698                        &[],
5699                        req,
5700                        &geo,
5701                        None,
5702                        output,
5703                    );
5704                }
5705
5706                // Total slot count of the index grid. The maximum extent
5707                // decides the multipliers (libhdf5 max_down_chunks); an
5708                // unlimited dimension 0 is bounded by the current extent for
5709                // this read — a slot beyond it (written before a shrink) is
5710                // not visible.
5711                let max_dims = self.datasets[pos].dataspace.max_dims.clone();
5712
5713                // Chunks are placed N-dimensionally: each slot decodes
5714                // (row-major, against the index grid) to chunk-grid
5715                // coordinates, so sub-frame chunks (a chunk smaller than a
5716                // full frame) land correctly.
5717                let rank = dims.len();
5718                let index = self.decoded_chunk_index(pos, index_address, |reader| {
5719                    let grid =
5720                        crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
5721                    let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
5722                    let mut entries = reader.collect_ea_chunk_entries(
5723                        index_address,
5724                        params,
5725                        &dims,
5726                        max_dims.as_deref(),
5727                        chunk_dims,
5728                        element_size,
5729                    )?;
5730                    entries.truncate(std::cmp::min(chunks_total as usize, entries.len()));
5731                    let coords = crate::io::chunk_grid::coords_table(
5732                        &dims,
5733                        max_dims.as_deref(),
5734                        chunk_dims,
5735                        entries.len(),
5736                    )?;
5737                    Ok(DecodedChunkIndex::new(entries, coords))
5738                })?;
5739                let chunk_entries = &index.entries;
5740                let slot_coords = &index.coords;
5741                let chunk_coords = |i: usize| -> &[u64] { &slot_coords[i * rank..(i + 1) * rank] };
5742
5743                // Build one read job per chunk (no I/O yet), then read +
5744                // decompress them together (in parallel where positioned reads
5745                // are race-free), then scatter serially. Filtered chunks record
5746                // their exact on-disk size, so read exactly that; unfiltered
5747                // chunks read at-most since the entry size can exceed the file
5748                // tail. Skip conditions differ between the two, so build jobs
5749                // per branch.
5750                let jobs: Vec<Option<ChunkReadJob>> = if pipeline.is_some() {
5751                    let file_size = self.handle.file_size()?;
5752                    chunk_entries
5753                        .iter()
5754                        .enumerate()
5755                        .map(|(i, &(addr, nbytes, mask))| {
5756                            if addr == UNDEF_ADDR
5757                                || nbytes == 0
5758                                || addr >= file_size
5759                                || nbytes > file_size
5760                                || !target.overlaps(chunk_coords(i), chunk_dims)
5761                            {
5762                                None
5763                            } else {
5764                                Some(ChunkReadJob {
5765                                    addr,
5766                                    len: nbytes as usize,
5767                                    at_most: false,
5768                                    mask,
5769                                })
5770                            }
5771                        })
5772                        .collect()
5773                } else {
5774                    chunk_entries
5775                        .iter()
5776                        .enumerate()
5777                        .map(|(i, &(addr, nbytes, _))| {
5778                            if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(i), chunk_dims) {
5779                                None
5780                            } else {
5781                                Some(ChunkReadJob {
5782                                    addr,
5783                                    len: nbytes as usize,
5784                                    at_most: true,
5785                                    mask: 0,
5786                                })
5787                            }
5788                        })
5789                        .collect()
5790                };
5791
5792                let geo = ChunkOutputGeometry {
5793                    dims: &dims,
5794                    chunk_dims,
5795                    element_size,
5796                };
5797                place_chunk_jobs(
5798                    &self.handle,
5799                    jobs,
5800                    slot_coords,
5801                    req,
5802                    &geo,
5803                    Some(&index.images),
5804                    output,
5805                )
5806            }
5807        }
5808    }
5809
5810    /// Collect a fixed-array dataset's per-chunk `(address, on-disk byte
5811    /// count, filter mask)` entries, indexed by index-grid linear slot
5812    /// ([`crate::io::chunk_grid`]). Empty when the index or its data block is
5813    /// unallocated.
5814    ///
5815    /// Shared by the full/slice chunked reader
5816    /// ([`read_chunked_fixed_array`](Self::read_chunked_fixed_array)) and the
5817    /// direct single-chunk read
5818    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the fixed-array
5819    /// wire format has one decoder.
5820    fn collect_fa_chunk_entries(
5821        &mut self,
5822        chunk_dims: &[u64],
5823        ndims: usize,
5824        element_size: u64,
5825        index_address: u64,
5826    ) -> IoResult<Vec<(u64, u64, u32)>> {
5827        use crate::format::chunk_index::fixed_array::*;
5828
5829        if index_address == UNDEF_ADDR {
5830            // Unallocated: no chunks recorded.
5831            return Ok(Vec::new());
5832        }
5833
5834        // Read FA header
5835        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
5836        let fa_hdr = FixedArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;
5837
5838        if fa_hdr.data_blk_addr == UNDEF_ADDR {
5839            // Unallocated data block: no chunks recorded.
5840            return Ok(Vec::new());
5841        }
5842
5843        // The chunk shape (from the layout message) must match the
5844        // dataspace rank; otherwise the chunk-grid indexing panics.
5845        if chunk_dims.len() != ndims {
5846            return Err(crate::io::IoError::InvalidState(format!(
5847                "fixed-array dataset rank {} does not match chunk rank {}",
5848                ndims,
5849                chunk_dims.len()
5850            )));
5851        }
5852
5853        let is_filtered = fa_hdr.client_id == FA_CLIENT_FILT_CHUNK;
5854        let sizeof_addr = self.meta.ctx.sizeof_addr as usize;
5855        // chunk_size_len = element_size - sizeof_addr - filter_mask(4)
5856        let chunk_size_len = if is_filtered {
5857            (fa_hdr.element_size as usize)
5858                .checked_sub(sizeof_addr + 4)
5859                .ok_or_else(|| {
5860                    crate::io::IoError::InvalidState(
5861                        "fixed array filtered element_size too small".into(),
5862                    )
5863                })?
5864        } else {
5865            0
5866        };
5867        // The compressed-size field is read into a u64; reject a width that
5868        // would overflow the read_size helper.
5869        if chunk_size_len > 8 {
5870            return Err(crate::io::IoError::InvalidState(format!(
5871                "fixed array filtered chunk-size width {chunk_size_len} exceeds 8 bytes"
5872            )));
5873        }
5874
5875        // Compute chunk byte size
5876        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
5877
5878        // Collect per-chunk (address, compressed_size). compressed_size is the
5879        // exact on-disk byte count for filtered chunks, or chunk_bytes when
5880        // unfiltered.
5881        let num_elmts = fa_hdr.num_elmts as usize;
5882        // (chunk address, on-disk byte count, filter mask). The mask is the
5883        // per-chunk filter mask for filtered chunks, 0 when unfiltered.
5884        let mut chunk_entries: Vec<(u64, u64, u32)> = Vec::with_capacity(num_elmts);
5885
5886        if fa_hdr.is_paged() {
5887            // Paged data block: prefix (with page-init bitmap) followed by pages.
5888            let npages = fa_hdr.npages();
5889            let dblk_page_nelmts = fa_hdr.dblk_page_nelmts();
5890            let prefix_len = 4 + 1 + 1 + sizeof_addr + (npages as usize).div_ceil(8) + 4;
5891            let prefix_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, prefix_len)?;
5892            let prefix = FixedArrayPagedPrefix::decode(&prefix_buf, &self.meta.ctx, npages)?;
5893
5894            let elem_size = if is_filtered {
5895                sizeof_addr + chunk_size_len + 4
5896            } else {
5897                sizeof_addr
5898            };
5899            // All pages have the same on-disk stride; only the last page holds
5900            // fewer elements (libhdf5: dblk_page_size is constant).
5901            let page_stride = dblk_page_nelmts as usize * elem_size + 4;
5902            let pages_base = fa_hdr.data_blk_addr + prefix.prefix_size as u64;
5903
5904            for p in 0..npages as usize {
5905                // Elements on this page (last page may be short).
5906                let page_nelmts = if p + 1 == npages as usize {
5907                    let rem = fa_hdr.num_elmts % dblk_page_nelmts;
5908                    if rem == 0 {
5909                        dblk_page_nelmts
5910                    } else {
5911                        rem
5912                    }
5913                } else {
5914                    dblk_page_nelmts
5915                } as usize;
5916
5917                if !prefix.page_initialized(p) {
5918                    // Uninitialized page: all chunk entries are undefined.
5919                    chunk_entries
5920                        .extend(std::iter::repeat_n((UNDEF_ADDR, 0u64, 0u32), page_nelmts));
5921                    continue;
5922                }
5923
5924                let page_addr = pages_base + (p as u64) * page_stride as u64;
5925                let page_size = page_nelmts * elem_size + 4;
5926                let page_buf = self.handle.read_at_most(page_addr, page_size)?;
5927
5928                if is_filtered {
5929                    let elems = decode_filtered_page(
5930                        &page_buf,
5931                        &self.meta.ctx,
5932                        page_nelmts,
5933                        chunk_size_len,
5934                    )?;
5935                    for e in elems {
5936                        chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
5937                    }
5938                } else {
5939                    let addrs = decode_unfiltered_page(&page_buf, &self.meta.ctx, page_nelmts)?;
5940                    for addr in addrs {
5941                        chunk_entries.push((addr, chunk_bytes, 0));
5942                    }
5943                }
5944            }
5945        } else {
5946            // Non-paged data block: all elements live inline in the data block.
5947            let elem_size = if is_filtered {
5948                sizeof_addr + chunk_size_len + 4
5949            } else {
5950                sizeof_addr
5951            };
5952            let dblk_size = 4 + 1 + 1 + sizeof_addr + num_elmts * elem_size + 4;
5953            let dblk_buf = self.handle.read_at_most(fa_hdr.data_blk_addr, dblk_size)?;
5954
5955            if is_filtered {
5956                let fa_dblk = FixedArrayDataBlock::decode_filtered(
5957                    &dblk_buf,
5958                    &self.meta.ctx,
5959                    num_elmts,
5960                    chunk_size_len,
5961                )?;
5962                for e in &fa_dblk.filtered_elements {
5963                    chunk_entries.push((e.address, e.chunk_size, e.filter_mask));
5964                }
5965            } else {
5966                let fa_dblk =
5967                    FixedArrayDataBlock::decode_unfiltered(&dblk_buf, &self.meta.ctx, num_elmts)?;
5968                for &addr in &fa_dblk.elements {
5969                    chunk_entries.push((addr, chunk_bytes, 0));
5970                }
5971            }
5972        }
5973
5974        Ok(chunk_entries)
5975    }
5976
5977    /// Read a dataset indexed by a fixed array.
5978    ///
5979    /// Scatters only; `output` must already be sized to the target extent.
5980    /// Every byte of it is defined before this returns `Ok`: what no chunk
5981    /// covers is filled with the tiled fill value by
5982    /// [`place_chunk_jobs`], the one exit every branch here takes.
5983    fn read_chunked_fixed_array(
5984        &mut self,
5985        name: &str,
5986        chunk_dims: &[u64],
5987        index_address: u64,
5988        req: ChunkReadRequest,
5989        output: &mut [u8],
5990    ) -> IoResult<()> {
5991        let ChunkReadRequest {
5992            pipeline, target, ..
5993        } = req;
5994        let pos = self.dataset_position(name)?;
5995        let info = &self.datasets[pos];
5996        let dims = info.dataspace.dims.clone();
5997        let element_size = info.datatype.element_size() as u64;
5998        let max_dims = info.dataspace.max_dims.clone();
5999        let ndims = dims.len();
6000        let geo = ChunkOutputGeometry {
6001            dims: &dims,
6002            chunk_dims,
6003            element_size,
6004        };
6005
6006        // Index-grid slot -> chunk-grid coordinates (row-major, against the
6007        // maximum extent — the array was sized from its chunk grid, so a slot
6008        // beyond the current extent still decodes to its true position and
6009        // then simply falls outside the read target). A zero chunk dimension
6010        // from a malformed layout message is rejected inside.
6011        let index = self.decoded_chunk_index(pos, index_address, |reader| {
6012            let entries =
6013                reader.collect_fa_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
6014            let coords = crate::io::chunk_grid::coords_table(
6015                &dims,
6016                max_dims.as_deref(),
6017                chunk_dims,
6018                entries.len(),
6019            )?;
6020            Ok(DecodedChunkIndex::new(entries, coords))
6021        })?;
6022        if index.entries.is_empty() {
6023            // Unallocated index/data block: the empty plan makes the whole
6024            // output fill.
6025            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6026        }
6027        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6028        let chunk_coords = |i: usize| -> &[u64] { &index.coords[i * ndims..(i + 1) * ndims] };
6029
6030        // Build one read job per chunk (no I/O yet). Filtered chunks carry
6031        // their exact compressed size (read at-most, since a zero size means
6032        // "unknown" and falls back to a generous estimate); unfiltered chunks
6033        // read the exact chunk byte count. For a slice, chunks outside the
6034        // selection become None and are never read.
6035        let jobs: Vec<Option<ChunkReadJob>> = index
6036            .entries
6037            .iter()
6038            .enumerate()
6039            .map(|(linear_idx, &(addr, comp_size, mask))| {
6040                if addr == UNDEF_ADDR || !target.overlaps(chunk_coords(linear_idx), chunk_dims) {
6041                    None
6042                } else if pipeline.is_some() {
6043                    let read_len = if comp_size > 0 {
6044                        comp_size as usize
6045                    } else {
6046                        chunk_bytes as usize * 2
6047                    };
6048                    Some(ChunkReadJob {
6049                        addr,
6050                        len: read_len,
6051                        at_most: true,
6052                        mask,
6053                    })
6054                } else {
6055                    Some(ChunkReadJob {
6056                        addr,
6057                        len: chunk_bytes as usize,
6058                        at_most: false,
6059                        mask,
6060                    })
6061                }
6062            })
6063            .collect();
6064
6065        // Read and place each chunk: the selected byte runs straight out of
6066        // the file where that beats reading the chunk whole.
6067        place_chunk_jobs(
6068            &self.handle,
6069            jobs,
6070            &index.coords,
6071            req,
6072            &geo,
6073            Some(&index.images),
6074            output,
6075        )
6076    }
6077
6078    /// Read a dataset indexed by the implicit ("none") chunk index.
6079    ///
6080    /// There is no on-disk index structure at all (`H5Dnone.c`): every chunk
6081    /// slot in the maximum-extent grid is allocated in one block at dataset
6082    /// creation, so a chunk's address is purely arithmetic — `index_address +
6083    /// slot * chunk_bytes`, where `slot` is its row-major position in the
6084    /// same maximum-extent grid the fixed/extensible-array/v2-B-tree indexes
6085    /// use (`H5D__chunk_set_info_real`'s `max_down_chunks`). This index type
6086    /// is only ever selected for a fixed (non-unlimited) chunked dataset with
6087    /// early allocation and no filters, so there is no per-chunk allocation
6088    /// flag, compressed size, or filter mask to track.
6089    ///
6090    /// Scatters only; `output` must already be sized to the target extent.
6091    /// Every byte of it is defined before this returns `Ok`: what no chunk
6092    /// covers is filled with the tiled fill value by
6093    /// [`place_chunk_jobs`], the one exit every branch here takes.
6094    fn read_chunked_implicit(
6095        &mut self,
6096        name: &str,
6097        chunk_dims: &[u64],
6098        index_address: u64,
6099        req: ChunkReadRequest,
6100        output: &mut [u8],
6101    ) -> IoResult<()> {
6102        let target = req.target;
6103        let pos = self.dataset_position(name)?;
6104        let info = &self.datasets[pos];
6105        let dims = info.dataspace.dims.clone();
6106        let element_size = info.datatype.element_size() as u64;
6107        let ndims = dims.len();
6108
6109        if index_address == UNDEF_ADDR {
6110            // Unallocated: the empty plan makes the whole output fill.
6111            let geo = ChunkOutputGeometry {
6112                dims: &dims,
6113                chunk_dims,
6114                element_size,
6115            };
6116            return place_chunk_jobs(
6117                &self.handle,
6118                Vec::new(),
6119                &[],
6120                ChunkReadRequest {
6121                    pipeline: None,
6122                    ..req
6123                },
6124                &geo,
6125                None,
6126                output,
6127            );
6128        }
6129
6130        if chunk_dims.len() != ndims {
6131            return Err(crate::io::IoError::InvalidState(format!(
6132                "implicit-index dataset rank {} does not match chunk rank {}",
6133                ndims,
6134                chunk_dims.len()
6135            )));
6136        }
6137
6138        let max_dims = self.datasets[pos].dataspace.max_dims.clone();
6139        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6140
6141        // Every slot's coordinates and address are computed directly, not
6142        // read from disk, so there is no "unallocated chunk" case here the
6143        // way a sparse index has one: `at_most: false` because
6144        // `H5D__none_idx_create` guarantees the whole block is present.
6145        let index = self.decoded_chunk_index(pos, index_address, |_| {
6146            let grid = crate::io::chunk_grid::index_grid(&dims, max_dims.as_deref(), chunk_dims)?;
6147            let chunks_total: u64 = grid.iter().fold(1u64, |acc, &n| acc.saturating_mul(n));
6148            let coords = crate::io::chunk_grid::coords_table(
6149                &dims,
6150                max_dims.as_deref(),
6151                chunk_dims,
6152                chunks_total as usize,
6153            )?;
6154            let entries = (0..chunks_total)
6155                .map(|i| (index_address + i * chunk_bytes, chunk_bytes, 0))
6156                .collect();
6157            Ok(DecodedChunkIndex::new(entries, coords))
6158        })?;
6159        let slot_coords = &index.coords;
6160
6161        let jobs: Vec<Option<ChunkReadJob>> = index
6162            .entries
6163            .iter()
6164            .enumerate()
6165            .map(|(i, &(addr, len, _))| {
6166                let coords = &slot_coords[i * ndims..(i + 1) * ndims];
6167                if !target.overlaps(coords, chunk_dims) {
6168                    None
6169                } else {
6170                    Some(ChunkReadJob {
6171                        addr,
6172                        len: len as usize,
6173                        at_most: false,
6174                        mask: 0,
6175                    })
6176                }
6177            })
6178            .collect();
6179
6180        let geo = ChunkOutputGeometry {
6181            dims: &dims,
6182            chunk_dims,
6183            element_size,
6184        };
6185        place_chunk_jobs(
6186            &self.handle,
6187            jobs,
6188            slot_coords,
6189            ChunkReadRequest {
6190                pipeline: None,
6191                ..req
6192            },
6193            &geo,
6194            Some(&index.images),
6195            output,
6196        )
6197    }
6198
6199    /// Collect a v2-B-tree-indexed dataset's per-chunk `(address, read
6200    /// size, scaled chunk-grid offsets, filter mask)` entries, walking the
6201    /// tree once. `read_size` is the compressed size for a filtered chunk,
6202    /// the full chunk size otherwise. Empty when the index has no records.
6203    ///
6204    /// Shared by the full/slice chunked reader
6205    /// ([`read_chunked_btree_v2`](Self::read_chunked_btree_v2)) and the
6206    /// direct single-chunk read
6207    /// ([`read_chunk_raw_at`](Self::read_chunk_raw_at)), so the v2 B-tree
6208    /// record format has one decoder.
6209    fn collect_bt2_chunk_entries(
6210        &mut self,
6211        chunk_dims: &[u64],
6212        ndims: usize,
6213        element_size: u64,
6214        index_address: u64,
6215    ) -> IoResult<Vec<Bt2ChunkEntry>> {
6216        use crate::format::chunk_index::btree_v2::*;
6217
6218        if index_address == UNDEF_ADDR {
6219            // Unallocated: no chunks recorded.
6220            return Ok(Vec::new());
6221        }
6222
6223        // Read BT2 header
6224        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
6225        let bt2_hdr = Bt2Header::decode(&hdr_buf, &self.meta.ctx)?;
6226
6227        if bt2_hdr.root_node_addr == UNDEF_ADDR || bt2_hdr.total_num_records == 0 {
6228            // No records.
6229            return Ok(Vec::new());
6230        }
6231
6232        // Walk the B-tree to any depth, collecting every record's raw bytes
6233        // from the internal nodes and leaves.
6234        let ctx = self.meta.ctx;
6235        let record_bytes = collect_btree_v2_records(
6236            &bt2_hdr,
6237            &ctx,
6238            &mut HandleBlockReader {
6239                handle: &mut self.handle,
6240            },
6241        )?;
6242        let total_records = if bt2_hdr.record_size > 0 {
6243            record_bytes.len() / bt2_hdr.record_size as usize
6244        } else {
6245            0
6246        };
6247
6248        // Decode records
6249        // Compute chunk byte size
6250        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6251
6252        // Unify filtered and unfiltered records into (address, read_size,
6253        // scaled offsets, filter mask). read_size is the compressed size for
6254        // filtered chunks, the full chunk size otherwise; the mask is 0 for
6255        // unfiltered records.
6256        let entries: Vec<Bt2ChunkEntry> = if bt2_hdr.record_type == BT2_TYPE_CHUNK_UNFILT {
6257            Bt2ChunkIndex::decode_unfiltered_records(
6258                &record_bytes,
6259                total_records,
6260                ndims,
6261                &self.meta.ctx,
6262            )?
6263            .into_iter()
6264            .map(|r| (r.chunk_address, chunk_bytes as usize, r.scaled_offsets, 0))
6265            .collect()
6266        } else {
6267            Bt2ChunkIndex::decode_filtered_records(
6268                &record_bytes,
6269                total_records,
6270                ndims,
6271                bt2_hdr.record_size,
6272                &self.meta.ctx,
6273            )?
6274            .into_iter()
6275            .map(|r| {
6276                (
6277                    r.chunk_address,
6278                    r.chunk_size as usize,
6279                    r.scaled_offsets,
6280                    r.filter_mask,
6281                )
6282            })
6283            .collect()
6284        };
6285
6286        Ok(entries)
6287    }
6288
6289    /// Read a dataset indexed by a B-tree v2.
6290    ///
6291    /// Scatters only; `output` must already be sized to the target extent.
6292    /// Every byte of it is defined before this returns `Ok`: what no chunk
6293    /// covers is filled with the tiled fill value by
6294    /// [`place_chunk_jobs`], the one exit every branch here takes.
6295    fn read_chunked_btree_v2(
6296        &mut self,
6297        name: &str,
6298        chunk_dims: &[u64],
6299        index_address: u64,
6300        req: ChunkReadRequest,
6301        output: &mut [u8],
6302    ) -> IoResult<()> {
6303        let target = req.target;
6304        let pos = self.dataset_position(name)?;
6305        let info = &self.datasets[pos];
6306        let dims = info.dataspace.dims.clone();
6307        let element_size = info.datatype.element_size() as u64;
6308        let ndims = dims.len();
6309        let geo = ChunkOutputGeometry {
6310            dims: &dims,
6311            chunk_dims,
6312            element_size,
6313        };
6314
6315        // A record carries its own scaled (chunk-grid) offsets, always `ndims`
6316        // of them, so flattening them into the coordinate table loses nothing.
6317        let index = self.decoded_chunk_index(pos, index_address, |reader| {
6318            let records =
6319                reader.collect_bt2_chunk_entries(chunk_dims, ndims, element_size, index_address)?;
6320            let mut entries = Vec::with_capacity(records.len());
6321            let mut coords = Vec::with_capacity(records.len() * ndims);
6322            for (addr, read_size, scaled, mask) in records {
6323                entries.push((addr, read_size as u64, mask));
6324                coords.extend_from_slice(&scaled);
6325            }
6326            Ok(DecodedChunkIndex::new(entries, coords))
6327        })?;
6328        if index.entries.is_empty() {
6329            // Unallocated index or no records: the empty plan makes the whole
6330            // output fill.
6331            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6332        }
6333
6334        // Build one read job per chunk (no I/O yet), placing each by its
6335        // scaled offsets. For a slice, chunks outside the selection become
6336        // None and are never read.
6337        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
6338        for (i, &(addr, read_size, mask)) in index.entries.iter().enumerate() {
6339            let scaled = &index.coords[i * ndims..(i + 1) * ndims];
6340            if addr == UNDEF_ADDR || read_size == 0 || !target.overlaps(scaled, chunk_dims) {
6341                jobs.push(None);
6342            } else {
6343                jobs.push(Some(ChunkReadJob {
6344                    addr,
6345                    len: read_size as usize,
6346                    at_most: false,
6347                    mask,
6348                }));
6349            }
6350        }
6351
6352        // Read and place each chunk N-dimensionally by its scaled offsets.
6353        place_chunk_jobs(
6354            &self.handle,
6355            jobs,
6356            &index.coords,
6357            req,
6358            &geo,
6359            Some(&index.images),
6360            output,
6361        )
6362    }
6363
6364    /// Read a chunked dataset indexed by a version-1 B-tree (layout
6365    /// message version 3, class 2 "chunked").
6366    ///
6367    /// `chunk_dims` excludes the trailing element-size dimension.
6368    ///
6369    /// Scatters only; `output` must already be sized to the target extent.
6370    /// Every byte of it is defined before this returns `Ok`: what no chunk
6371    /// covers is filled with the tiled fill value by
6372    /// [`place_chunk_jobs`], the one exit every branch here takes.
6373    fn read_chunked_btree_v1(
6374        &mut self,
6375        name: &str,
6376        chunk_dims: &[u64],
6377        b_tree_address: u64,
6378        req: ChunkReadRequest,
6379        output: &mut [u8],
6380    ) -> IoResult<()> {
6381        let ChunkReadRequest {
6382            pipeline, target, ..
6383        } = req;
6384        let pos = self.dataset_position(name)?;
6385        let info = &self.datasets[pos];
6386        let dims = info.dataspace.dims.clone();
6387        let element_size = info.datatype.element_size() as u64;
6388        let ndims = dims.len();
6389
6390        // The chunk shape must match the dataspace rank or the chunk-grid
6391        // indexing below panics.
6392        if chunk_dims.len() != ndims {
6393            return Err(crate::io::IoError::InvalidState(format!(
6394                "B-tree-v1 dataset rank {} does not match chunk rank {}",
6395                ndims,
6396                chunk_dims.len()
6397            )));
6398        }
6399
6400        let total_size: u64 = saturating_byte_len(&dims, element_size);
6401        if b_tree_address == UNDEF_ADDR || total_size == 0 {
6402            // Unallocated: the empty plan makes the whole output fill.
6403            let geo = ChunkOutputGeometry {
6404                dims: &dims,
6405                chunk_dims,
6406                element_size,
6407            };
6408            return place_chunk_jobs(&self.handle, Vec::new(), &[], req, &geo, None, output);
6409        }
6410
6411        // Walk the B-tree, collecting every leaf entry as
6412        // (element_offsets, chunk_address, chunk_size, filter_mask), and turn
6413        // each key's element offsets into chunk-grid coordinates. The keys
6414        // carry rank + 1 offsets; the trailing element-size dimension offset
6415        // is always 0 and is dropped.
6416        let file_size = self.handle.file_size()?;
6417        let index = self.decoded_chunk_index(pos, b_tree_address, |reader| {
6418            let mut leaves: Vec<(Vec<u64>, u64, u32, u32)> = Vec::new();
6419            reader.collect_btree_v1_chunks(b_tree_address, ndims, file_size, 0, &mut leaves)?;
6420            let mut entries = Vec::with_capacity(leaves.len());
6421            let mut coords = Vec::with_capacity(leaves.len() * ndims);
6422            for (offsets, addr, chunk_size, mask) in leaves {
6423                for (d, &cd) in chunk_dims.iter().enumerate().take(ndims) {
6424                    coords.push(offsets[d].checked_div(cd).unwrap_or(0));
6425                }
6426                entries.push((addr, chunk_size as u64, mask));
6427            }
6428            Ok(DecodedChunkIndex::new(entries, coords))
6429        })?;
6430
6431        // The uncompressed byte size of a full chunk.
6432        let chunk_bytes: u64 = saturating_byte_len(chunk_dims, element_size);
6433
6434        // Build one read job per chunk (no I/O yet), placing each by its
6435        // scaled coordinates.
6436        let mut jobs: Vec<Option<ChunkReadJob>> = Vec::with_capacity(index.entries.len());
6437        for (i, &(addr, chunk_size, mask)) in index.entries.iter().enumerate() {
6438            let skip = addr == UNDEF_ADDR
6439                || chunk_size == 0
6440                || addr >= file_size
6441                || chunk_size > file_size
6442                || !target.overlaps(&index.coords[i * ndims..(i + 1) * ndims], chunk_dims);
6443            jobs.push(if skip {
6444                None
6445            } else {
6446                Some(ChunkReadJob {
6447                    addr,
6448                    len: chunk_size as usize,
6449                    at_most: false,
6450                    mask,
6451                })
6452            });
6453        }
6454
6455        // Read and place each chunk N-dimensionally by its scaled offsets.
6456        let geo = ChunkOutputGeometry {
6457            dims: &dims,
6458            chunk_dims,
6459            element_size,
6460        };
6461        place_chunk_jobs(
6462            &self.handle,
6463            jobs,
6464            &index.coords,
6465            req,
6466            &geo,
6467            Some(&index.images),
6468            output,
6469        )?;
6470
6471        // libhdf5 stores raw byte sizes; verify the uncompressed chunk
6472        // size is consistent for unfiltered datasets so a corrupt index
6473        // surfaces instead of silently producing garbage.
6474        if pipeline.is_none() {
6475            for &(addr, chunk_size, _) in &index.entries {
6476                if addr != UNDEF_ADDR && chunk_size != chunk_bytes && chunk_size != 0 {
6477                    return Err(crate::io::IoError::InvalidState(format!(
6478                        "chunk B-tree v1: unfiltered chunk size {} != expected {}",
6479                        chunk_size, chunk_bytes
6480                    )));
6481                }
6482            }
6483        }
6484
6485        Ok(())
6486    }
6487
6488    /// Recursively walk a version-1 raw-data-chunk B-tree, collecting every
6489    /// leaf entry as `(element_offsets, chunk_address, chunk_size,
6490    /// filter_mask)`.
6491    ///
6492    /// `rank` is the chunk rank excluding the trailing element-size
6493    /// dimension. Recursion is bounded by the node level read from disk:
6494    /// each recursive step descends to a strictly lower level, and the
6495    /// `depth` counter caps the descent at the 1-byte level field's range.
6496    fn collect_btree_v1_chunks(
6497        &mut self,
6498        addr: u64,
6499        rank: usize,
6500        file_size: u64,
6501        depth: u32,
6502        out: &mut Vec<(Vec<u64>, u64, u32, u32)>,
6503    ) -> IoResult<()> {
6504        // A v1 B-tree node level fits in one byte, so the tree can be at
6505        // most 256 levels deep; this also stops cyclic/corrupt indices.
6506        if depth > 256 {
6507            return Err(crate::io::IoError::InvalidState(
6508                "chunk B-tree v1 exceeds maximum depth".into(),
6509            ));
6510        }
6511        if addr == UNDEF_ADDR || addr >= file_size {
6512            return Ok(());
6513        }
6514
6515        // A node is a fixed-size record: the header (8 + 2*sizeof_addr) plus
6516        // `2 * chunk_internal_k` interleaved keys/children.
6517        let sa = self.meta.ctx.sizeof_addr as usize;
6518        let node_size = self.meta.btree.chunk_btree_node_size(sa, rank);
6519        let buf = self.handle.read_at_most(addr, node_size)?;
6520        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.meta.btree.chunk_max_entries())?;
6521
6522        if node.level == 0 {
6523            // Leaf node: each child points at chunk data.
6524            for (i, &child_addr) in node.children.iter().enumerate() {
6525                let key = &node.keys[i];
6526                out.push((
6527                    key.offsets[..rank].to_vec(),
6528                    child_addr,
6529                    key.chunk_size,
6530                    key.filter_mask,
6531                ));
6532            }
6533        } else {
6534            // Internal node: each child points at a sub-TREE node one
6535            // level below. `node.level` is read from disk and decreases on
6536            // every descent, so it also bounds the recursion.
6537            let children: Vec<u64> = node.children.clone();
6538            for child_addr in children {
6539                if child_addr == UNDEF_ADDR || child_addr >= file_size {
6540                    continue;
6541                }
6542                self.collect_btree_v1_chunks(child_addr, rank, file_size, depth + 1, out)?;
6543            }
6544        }
6545        Ok(())
6546    }
6547
6548    /// Read variable-length string data from a dataset.
6549    ///
6550    /// h5py stores vlen strings as global heap references. Each element
6551    /// in the raw data is a (collection_address, object_index) pair that
6552    /// points to a string blob in a global heap collection.
6553    ///
6554    /// Returns a Vec<String> with one entry per element.
6555    pub fn read_vlen_strings(&mut self, name: &str) -> IoResult<Vec<String>> {
6556        // A vlen string is a vlen sequence of `u8` reinterpreted as UTF-8.
6557        // The global-heap walk is identical; decode the raw object bytes.
6558        Ok(self
6559            .read_vlen_objects(name)?
6560            .into_iter()
6561            .map(|bytes| String::from_utf8_lossy(&bytes).to_string())
6562            .collect())
6563    }
6564
6565    /// Read a 1-D variable-length byte-array dataset (vlen sequence of `u8`).
6566    ///
6567    /// Returns a `Vec<Vec<u8>>` with one byte array per element. Missing or
6568    /// undefined references yield an empty `Vec`.
6569    pub fn read_vlen_bytes(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
6570        self.read_vlen_objects(name)
6571    }
6572
6573    /// Shared owner of the global-heap walk for variable-length datasets.
6574    ///
6575    /// Each element of the raw data is a vlen reference (collection address +
6576    /// object index) into a global heap collection. Returns the raw object
6577    /// bytes for each element, with an empty `Vec` for undefined/missing
6578    /// references. Both `read_vlen_strings` (UTF-8 view) and `read_vlen_bytes`
6579    /// (raw view) layer on top of this.
6580    fn read_vlen_objects(&mut self, name: &str) -> IoResult<Vec<Vec<u8>>> {
6581        if self.external_edge(name).is_some() {
6582            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
6583            return owner.read_vlen_objects(&path);
6584        }
6585        let info = self
6586            .dataset_info_local(name)
6587            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
6588        let dims = info.dataspace.dims.clone();
6589        let layout = info.layout.clone();
6590        let external_files = info.external_files.clone();
6591        let total_elements: u64 = dims.iter().fold(1u64, |acc, &d| acc.saturating_mul(d));
6592
6593        let raw = match &layout {
6594            DataLayoutMessage::Contiguous { size, .. } if !external_files.is_empty() => {
6595                let prefix = self.extfile_prefix_in_force(name);
6596                let mut buf = vec![0u8; *size as usize];
6597                read_external_file_bytes(&external_files, prefix.as_deref(), 0, &mut buf)?;
6598                buf
6599            }
6600            DataLayoutMessage::Contiguous { address, size } => {
6601                if *address == UNDEF_ADDR {
6602                    return Ok(vec![]);
6603                }
6604                self.handle.read_at(*address, *size as usize)?
6605            }
6606            DataLayoutMessage::Compact { data } => data.clone(),
6607            _ => {
6608                // For chunked, read the full dataset first
6609                self.read_dataset_raw(name)?
6610            }
6611        };
6612
6613        let ref_size = vlen_reference_size(&self.meta.ctx);
6614        let mut items: Vec<Vec<u8>> = Vec::with_capacity(total_elements as usize);
6615
6616        // Cache global heap collections to avoid re-reading.
6617        // Store as (collection, index→offset lookup) for O(1) object access.
6618        let mut heap_cache: std::collections::HashMap<
6619            u64,
6620            (GlobalHeapCollection, std::collections::HashMap<u16, usize>),
6621        > = std::collections::HashMap::new();
6622
6623        for i in 0..total_elements as usize {
6624            let offset = i * ref_size;
6625            if offset + ref_size > raw.len() {
6626                break;
6627            }
6628
6629            let (_seq_len, collection_addr, obj_index) =
6630                decode_vlen_reference(&raw[offset..], &self.meta.ctx)?;
6631
6632            if collection_addr == UNDEF_ADDR || collection_addr == 0 {
6633                items.push(Vec::new());
6634                continue;
6635            }
6636
6637            // Read or get cached global heap collection
6638            if let std::collections::hash_map::Entry::Vacant(e) = heap_cache.entry(collection_addr)
6639            {
6640                let coll = self.read_heap_collection(collection_addr)?;
6641                let lookup: std::collections::HashMap<u16, usize> = coll
6642                    .objects
6643                    .iter()
6644                    .enumerate()
6645                    .map(|(i, o)| (o.index, i))
6646                    .collect();
6647                e.insert((coll, lookup));
6648            }
6649
6650            let idx = u16::try_from(obj_index).map_err(|_| {
6651                crate::io::IoError::InvalidState(format!(
6652                    "global heap object index {obj_index} does not fit the 16-bit on-disk field \
6653                     (element {i} of \"{name}\")"
6654                ))
6655            })?;
6656            let (coll, lookup) = &heap_cache[&collection_addr];
6657            let &oi = lookup.get(&idx).ok_or_else(|| {
6658                crate::io::IoError::InvalidState(format!(
6659                    "global heap object {idx} not found in the collection at address \
6660                     {collection_addr:#x} (element {i} of \"{name}\")"
6661                ))
6662            })?;
6663            items.push(coll.objects[oi].data.clone());
6664        }
6665
6666        Ok(items)
6667    }
6668
6669    /// Collect chunk (address, size) entries from an EA index.
6670    /// Returns a vector indexed by chunk linear index.
6671    fn collect_ea_chunk_entries(
6672        &mut self,
6673        index_address: u64,
6674        params: &data_layout::EarrayParams,
6675        dims: &[u64],
6676        max_dims: Option<&[u64]>,
6677        chunk_dims: &[u64],
6678        element_size: u64,
6679    ) -> IoResult<Vec<(u64, u64, u32)>> {
6680        use crate::format::chunk_index::extensible_array::{self as ea, *};
6681
6682        if index_address == UNDEF_ADDR {
6683            return Ok(vec![]);
6684        }
6685        let hdr_buf = self.handle.read_at_most(index_address, 256)?;
6686        let ea_hdr = ExtensibleArrayHeader::decode(&hdr_buf, &self.meta.ctx)?;
6687        if ea_hdr.idx_blk_addr == UNDEF_ADDR {
6688            return Ok(vec![]);
6689        }
6690
6691        // Slot count of the index grid, bounding the collection walk. The
6692        // maximum extent decides the multipliers (sub-frame chunks make this
6693        // larger than the dim-0 chunk count alone); the unlimited dimension 0
6694        // is bounded by the current extent.
6695        let chunks_dim0: usize = crate::io::chunk_grid::index_grid(dims, max_dims, chunk_dims)?
6696            .iter()
6697            .fold(1usize, |acc, &n| acc.saturating_mul(n as usize));
6698        let geo = EaGeometry::new(
6699            params.idx_blk_elmts,
6700            params.data_blk_min_elmts,
6701            params.sup_blk_min_data_ptrs,
6702            params.max_nelmts_bits,
6703            params.max_dblk_page_nelmts_bits,
6704        )?;
6705        let chunk_bytes = saturating_byte_len(chunk_dims, element_size);
6706        let is_filtered = ea_hdr.class_id == ea::EA_CLS_FILT_CHUNK;
6707        let chunk_size_len = if is_filtered {
6708            ea_hdr.raw_elmt_size - self.meta.ctx.sizeof_addr - 4
6709        } else {
6710            0
6711        };
6712        let max_nelmts_bits = params.max_nelmts_bits;
6713        // Each entry is (chunk address, on-disk byte count, filter mask). The
6714        // mask is the per-chunk filter mask for filtered datasets (0 for an
6715        // unfiltered index, where it is meaningless).
6716        let mut entries: Vec<(u64, u64, u32)> = Vec::new();
6717
6718        // Read the index block: direct elements + the data-block / super-block
6719        // address arrays (the address arrays are filter-agnostic).
6720        let (dblk_addrs, sblk_addrs): (Vec<u64>, Vec<u64>) = if is_filtered {
6721            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
6722            let fiblk = ea::FilteredIndexBlock::decode(
6723                &buf,
6724                &self.meta.ctx,
6725                params.idx_blk_elmts as usize,
6726                geo.ndblk_addrs,
6727                geo.nsblk_addrs,
6728                chunk_size_len,
6729            )?;
6730            for e in &fiblk.elements {
6731                entries.push((e.addr, e.nbytes, e.filter_mask));
6732            }
6733            (fiblk.dblk_addrs, fiblk.sblk_addrs)
6734        } else {
6735            let buf = self.handle.read_at_most(ea_hdr.idx_blk_addr, 65536)?;
6736            let iblk = ExtensibleArrayIndexBlock::decode(
6737                &buf,
6738                &self.meta.ctx,
6739                params.idx_blk_elmts as usize,
6740                geo.ndblk_addrs,
6741                geo.nsblk_addrs,
6742            )?;
6743            for &addr in &iblk.elements {
6744                entries.push((addr, chunk_bytes, 0));
6745            }
6746            (iblk.dblk_addrs, iblk.sblk_addrs)
6747        };
6748
6749        // Walk super blocks in order, collecting each data block's entries.
6750        let sa = self.meta.ctx.sizeof_addr as usize;
6751        let raw_elmt_size = if is_filtered {
6752            ea::FilteredChunkEntry::raw_size(self.meta.ctx.sizeof_addr, chunk_size_len) as usize
6753        } else {
6754            sa
6755        };
6756        'outer: for (u, s) in geo.sblk.iter().enumerate() {
6757            if entries.len() >= chunks_dim0 {
6758                break;
6759            }
6760            let dblk_nelmts = s.dblk_nelmts as usize;
6761            let paged = geo.is_sblk_paged(u);
6762
6763            // This super block's data-block addresses, plus its page-init
6764            // bitmap region (empty unless the super block is paged).
6765            let (this_dblk_addrs, page_init): (Vec<u64>, Vec<u8>) = if u < geo.iblock_nsblks {
6766                let start = s.start_dblk as usize;
6767                (
6768                    (0..s.ndblks as usize)
6769                        .map(|d| dblk_addrs.get(start + d).copied().unwrap_or(UNDEF_ADDR))
6770                        .collect(),
6771                    Vec::new(),
6772                )
6773            } else {
6774                let sblk_addr = sblk_addrs
6775                    .get(u - geo.iblock_nsblks)
6776                    .copied()
6777                    .unwrap_or(UNDEF_ADDR);
6778                if sblk_addr == UNDEF_ADDR {
6779                    (vec![UNDEF_ADDR; s.ndblks as usize], Vec::new())
6780                } else {
6781                    let page_init_total = if paged {
6782                        s.ndblks as usize * geo.dblk_page_init_size(u)
6783                    } else {
6784                        0
6785                    };
6786                    // Size the read from the super block's geometry rather
6787                    // than a fixed cap: signature+version+class+header_addr
6788                    // + block_offset(<=8) + page-init bitmaps
6789                    // + ndblks data-block addresses + checksum.
6790                    let sblk_size =
6791                        4 + 1 + 1 + sa + 8 + page_init_total + s.ndblks as usize * sa + 4;
6792                    let buf = self.handle.read_at_most(sblk_addr, sblk_size)?;
6793                    let sb = ExtensibleArraySuperBlock::decode(
6794                        &buf,
6795                        &self.meta.ctx,
6796                        max_nelmts_bits,
6797                        s.ndblks as usize,
6798                        page_init_total,
6799                    )?;
6800                    (sb.dblk_addrs, sb.page_init)
6801                }
6802            };
6803
6804            let npages = geo.npages(u) as usize;
6805            let page_size = geo.dblk_page_size(raw_elmt_size);
6806            let prefix = geo.dblk_prefix_size(self.meta.ctx.sizeof_addr, max_nelmts_bits);
6807
6808            for (d, &dblk_addr) in this_dblk_addrs.iter().enumerate() {
6809                if dblk_addr == UNDEF_ADDR {
6810                    entries.extend(std::iter::repeat_n((UNDEF_ADDR, 0, 0), dblk_nelmts));
6811                } else if paged {
6812                    // Paged data block: only a prefix lives at `dblk_addr`;
6813                    // the elements live in `npages` page structures that
6814                    // follow it on disk. The super block's page-init bitmap
6815                    // is one flat MSB-first bitmap (H5VM bit ops) indexed by
6816                    // `dblk_idx * npages + page_idx` (H5EA.c), not a series
6817                    // of per-data-block sub-bitmaps.
6818                    for p in 0..npages {
6819                        let bit = d * npages + p;
6820                        let initialized = page_init[bit / 8] & (0x80u8 >> (bit % 8)) != 0;
6821                        if !initialized {
6822                            entries.extend(std::iter::repeat_n(
6823                                (UNDEF_ADDR, 0, 0),
6824                                geo.dblk_page_nelmts as usize,
6825                            ));
6826                            continue;
6827                        }
6828                        let page_addr = dblk_addr + prefix as u64 + (p as u64) * page_size as u64;
6829                        let page = self.handle.read_at(page_addr, page_size)?;
6830                        for k in 0..geo.dblk_page_nelmts as usize {
6831                            let off = k * raw_elmt_size;
6832                            if is_filtered {
6833                                let e = ea::FilteredChunkEntry::decode(
6834                                    &page[off..],
6835                                    sa,
6836                                    chunk_size_len as usize,
6837                                );
6838                                entries.push((e.addr, e.nbytes, e.filter_mask));
6839                            } else {
6840                                entries.push((read_addr(&page[off..], sa), chunk_bytes, 0));
6841                            }
6842                        }
6843                    }
6844                } else if is_filtered {
6845                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
6846                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
6847                    let dblk = ea::FilteredDataBlock::decode(
6848                        &buf,
6849                        &self.meta.ctx,
6850                        max_nelmts_bits,
6851                        dblk_nelmts,
6852                        chunk_size_len,
6853                    )?;
6854                    for e in &dblk.elements {
6855                        entries.push((e.addr, e.nbytes, e.filter_mask));
6856                    }
6857                } else {
6858                    let dblk_size = prefix + dblk_nelmts * raw_elmt_size;
6859                    let buf = self.handle.read_at_most(dblk_addr, dblk_size)?;
6860                    let dblk = ExtensibleArrayDataBlock::decode(
6861                        &buf,
6862                        &self.meta.ctx,
6863                        max_nelmts_bits,
6864                        dblk_nelmts,
6865                    )?;
6866                    for &addr in &dblk.elements {
6867                        entries.push((addr, chunk_bytes, 0));
6868                    }
6869                }
6870                if entries.len() >= chunks_dim0 {
6871                    break 'outer;
6872                }
6873            }
6874        }
6875        Ok(entries)
6876    }
6877
6878    /// Read a slice (hyperslab) of a dataset, whatever its layout.
6879    ///
6880    /// Contiguous and compact datasets are read run by run; a chunked one
6881    /// reads only the chunks the selection overlaps, and any gap the writer
6882    /// never filled comes back as the fill value.
6883    ///
6884    /// `starts` and `counts` define the N-dimensional selection:
6885    /// starts[d] is the first index along dim d, counts[d] is how many.
6886    /// Returns the selected data in row-major order.
6887    pub fn read_slice(&mut self, name: &str, starts: &[u64], counts: &[u64]) -> IoResult<Vec<u8>> {
6888        if self.external_edge(name).is_some() {
6889            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
6890            return owner.read_slice(&path, starts, counts);
6891        }
6892        let (datatype, out_bytes) = self.slice_size_and_datatype(name, counts)?;
6893        // The selection lands in the vector this returns:
6894        // `read_slice_into_unconverted` defines every byte of the buffer it is
6895        // handed, so there is nothing to zero first and nothing to copy after.
6896        read_image_into_new::<u8, _, _>(out_bytes as usize, |image| {
6897            self.read_slice_into_unconverted(name, starts, counts, image, 0)?;
6898            Self::apply_post_filter_conversion(image, &datatype)
6899        })
6900    }
6901
6902    /// Read a hyperslab straight into a caller-provided buffer (no allocation).
6903    ///
6904    /// `out.len()` must equal `product(counts) * element_size`; otherwise an
6905    /// error is returned. The no-allocation counterpart of
6906    /// [`read_slice`](Self::read_slice) and the zero-copy entry point for
6907    /// reading a selection directly into a pinned/registered host buffer for an
6908    /// H2D transfer.
6909    pub fn read_slice_into(
6910        &mut self,
6911        name: &str,
6912        starts: &[u64],
6913        counts: &[u64],
6914        out: &mut [u8],
6915    ) -> IoResult<()> {
6916        if self.external_edge(name).is_some() {
6917            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
6918            return owner.read_slice_into(&path, starts, counts, out);
6919        }
6920        let (datatype, out_bytes) = self.slice_size_and_datatype(name, counts)?;
6921        if out.len() as u64 != out_bytes {
6922            return Err(crate::io::IoError::InvalidState(format!(
6923                "read_slice_into: buffer is {} bytes but selection needs {}",
6924                out.len(),
6925                out_bytes
6926            )));
6927        }
6928        self.read_slice_into_unconverted(name, starts, counts, out, 0)?;
6929        Self::apply_post_filter_conversion(out, &datatype)?;
6930        Ok(())
6931    }
6932
6933    /// Logical byte size of a hyperslab (`product(counts) * element_size`) with
6934    /// the datatype needed for the post-filter conversion.
6935    fn slice_size_and_datatype(
6936        &self,
6937        name: &str,
6938        counts: &[u64],
6939    ) -> IoResult<(DatatypeMessage, u64)> {
6940        let info = self
6941            .dataset_info_local(name)
6942            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
6943        let out_bytes = saturating_byte_len(counts, info.datatype.element_size() as u64);
6944        Ok((info.datatype.clone(), out_bytes))
6945    }
6946
6947    /// Fill `out` with a hyperslab selection, before the post-filter datatype
6948    /// conversion. The single owner of read-destination semantics for slice
6949    /// reads (mirrors [`read_dataset_raw_into_unconverted`](Self::read_dataset_raw_into_unconverted)):
6950    /// it validates the selection, then fully defines every byte of `out`
6951    /// (contiguous/compact runs cover the whole selection; chunked layouts
6952    /// pre-fill with the tiled fill value and scatter only overlapping chunks).
6953    /// Both [`read_slice`](Self::read_slice) and [`read_slice_into`](Self::read_slice_into)
6954    /// wrap it and apply the conversion exactly once.
6955    ///
6956    /// `out.len()` must equal `product(counts) * element_size`. `depth`
6957    /// counts virtual-dataset nesting for a caller reached through
6958    /// [`read_virtual_into`](Self::read_virtual_into); pass `0` for a
6959    /// top-level call.
6960    fn read_slice_into_unconverted(
6961        &mut self,
6962        name: &str,
6963        starts: &[u64],
6964        counts: &[u64],
6965        out: &mut [u8],
6966        depth: usize,
6967    ) -> IoResult<()> {
6968        let info = self
6969            .dataset_info_local(name)
6970            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
6971        let dims = info.dataspace.dims.clone();
6972        let element_size = info.datatype.element_size() as u64;
6973        let layout = info.layout.clone();
6974        let pipeline = info.filter_pipeline.clone();
6975        let fill_value = info.fill_value.clone();
6976        let external_files = info.external_files.clone();
6977        let ndims = dims.len();
6978
6979        if starts.len() != ndims || counts.len() != ndims {
6980            return Err(crate::io::IoError::InvalidState(
6981                "starts/counts length must match dataset rank".into(),
6982            ));
6983        }
6984        if ndims == 0 {
6985            return Err(crate::io::IoError::InvalidState(
6986                "read_slice does not support scalar datasets; use read_dataset_raw".into(),
6987            ));
6988        }
6989        for d in 0..ndims {
6990            if starts[d] + counts[d] > dims[d] {
6991                return Err(crate::io::IoError::InvalidState(format!(
6992                    "slice out of bounds: dim {} start {} + count {} > {}",
6993                    d, starts[d], counts[d], dims[d]
6994                )));
6995            }
6996        }
6997
6998        match &layout {
6999            DataLayoutMessage::Contiguous { .. } if !external_files.is_empty() => {
7000                // Same coalesced run geometry as the normal contiguous case
7001                // below, but each run is read through the external file
7002                // list instead of straight from this file (H5D__efl_read,
7003                // H5Defl.c) — `src_off` is already dataset-relative, which
7004                // is exactly what `read_external_file_bytes` walks slots by.
7005                let prefix = self.extfile_prefix_in_force(name);
7006                for_each_contiguous_run(
7007                    &dims,
7008                    starts,
7009                    counts,
7010                    element_size,
7011                    |src_off, out_off, len| {
7012                        read_external_file_bytes(
7013                            &external_files,
7014                            prefix.as_deref(),
7015                            src_off,
7016                            &mut out[out_off..out_off + len],
7017                        )
7018                    },
7019                )?;
7020            }
7021            DataLayoutMessage::Contiguous { address, .. } => {
7022                if *address == UNDEF_ADDR {
7023                    // Never-written: the selection reads back as the fill value.
7024                    fill_tiled_into(out, fill_value.as_deref());
7025                } else {
7026                    // Read each maximal contiguous run straight into `out`.
7027                    // Trailing full-selected dimensions coalesce, so a slice
7028                    // like `[:, r0:r1, :]` of `[nproj, nz, nx]` becomes `nproj`
7029                    // reads of `(r1-r0)*nx` elements instead of `nproj*(r1-r0)`
7030                    // per-`nx`-row reads. The 1-D case folds to a single run.
7031                    // The runs cover the whole selection, so every byte of `out`
7032                    // is written.
7033                    let base = *address;
7034                    for_each_contiguous_run(
7035                        &dims,
7036                        starts,
7037                        counts,
7038                        element_size,
7039                        |src_off, out_off, len| {
7040                            self.handle
7041                                .read_exact_at_into(
7042                                    base + src_off,
7043                                    &mut out[out_off..out_off + len],
7044                                )
7045                                .map_err(Into::into)
7046                        },
7047                    )?;
7048                }
7049            }
7050            DataLayoutMessage::Compact { data } => {
7051                // Same coalesced geometry, copying from the in-memory full
7052                // dataset instead of reading from the file.
7053                for_each_contiguous_run(
7054                    &dims,
7055                    starts,
7056                    counts,
7057                    element_size,
7058                    |src_off, out_off, len| {
7059                        let src = src_off as usize;
7060                        out[out_off..out_off + len].copy_from_slice(&data[src..src + len]);
7061                        Ok(())
7062                    },
7063                )?;
7064            }
7065            DataLayoutMessage::ChunkedV3 {
7066                chunk_dims,
7067                b_tree_address,
7068            } => {
7069                // Walk the v1 B-tree index reading only chunks that overlap the
7070                // selection, scattering each chunk∩selection into the slice
7071                // buffer; whatever no chunk covers is filled there. The
7072                // unconverted read keeps the post-filter conversion to exactly
7073                // once, in the wrappers.
7074                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7075                self.read_chunked_btree_v1(
7076                    name,
7077                    real_chunk_dims,
7078                    *b_tree_address,
7079                    ChunkReadRequest {
7080                        pipeline: pipeline.as_ref(),
7081                        target: ChunkTarget::Slice { starts, counts },
7082                        fill_value: fill_value.as_deref(),
7083                    },
7084                    out,
7085                )?;
7086            }
7087            DataLayoutMessage::ChunkedV4 {
7088                chunk_dims,
7089                index_address,
7090                index_type,
7091                earray_params,
7092                single_chunk_filter,
7093                ..
7094            } => {
7095                // Same selection-aware chunk read for every v4 index kind
7096                // (single chunk, fixed/extensible array, B-tree v2): only
7097                // overlapping chunks are read and only their intersection with
7098                // the selection is scattered into the slice output.
7099                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7100                self.read_chunked_v4(
7101                    name,
7102                    real_chunk_dims,
7103                    ChunkIndexDesc {
7104                        index_type: *index_type,
7105                        index_address: *index_address,
7106                        earray_params: earray_params.as_ref(),
7107                        single_chunk_filter: *single_chunk_filter,
7108                    },
7109                    ChunkReadRequest {
7110                        pipeline: pipeline.as_ref(),
7111                        target: ChunkTarget::Slice { starts, counts },
7112                        fill_value: fill_value.as_deref(),
7113                    },
7114                    out,
7115                )?;
7116            }
7117            DataLayoutMessage::Virtual { .. } => {
7118                // Stitch the full virtual image, then extract the
7119                // requested region from it. This reads more than the
7120                // selection strictly needs (no per-mapping intersection
7121                // against the caller's box), but a virtual dataset's data
7122                // is composed from other datasets rather than stored
7123                // contiguously, so there is no cheaper selective path
7124                // without duplicating `read_virtual_into`'s mapping walk
7125                // for a bounded region — correctness, not I/O pruning, is
7126                // what a VDS slice read needs here. The extraction below
7127                // writes every byte of `out`, so no pre-fill is needed on top
7128                // of the full image's own.
7129                let total = saturating_byte_len(&dims, element_size) as usize;
7130                let mut full = alloc_tiled_fill(total, fill_value.as_deref())?;
7131                self.read_virtual_into(name, &mut full, depth)?;
7132                for_each_contiguous_run(
7133                    &dims,
7134                    starts,
7135                    counts,
7136                    element_size,
7137                    |src_off, out_off, len| {
7138                        let src_off = src_off as usize;
7139                        out[out_off..out_off + len].copy_from_slice(&full[src_off..src_off + len]);
7140                        Ok(())
7141                    },
7142                )?;
7143            }
7144        }
7145        Ok(())
7146    }
7147
7148    /// Read a strided hyperslab — h5py's stepped slicing (`ds[a:b:s]`) or
7149    /// the general `start`/`stride`/`count`/`block` form of
7150    /// `H5Sselect_hyperslab` — into a caller-provided buffer.
7151    ///
7152    /// One tuple per dimension: `start[d]` is the first index, `stride[d]`
7153    /// the spacing between selected blocks (`1` = the classic contiguous
7154    /// selection [`read_slice`](Self::read_slice) reads), `count[d]` how
7155    /// many blocks, and `block[d]` how many contiguous elements each block
7156    /// covers. `out` is row-major over `count[d] * block[d]` per dimension —
7157    /// exactly the shape h5py's stepped slicing produces — and `out.len()`
7158    /// must equal that times the element size.
7159    ///
7160    /// Built on the same selection-decomposition primitives the virtual
7161    /// dataset reader uses for its per-mapping scatter
7162    /// ([`Selection::resolve`], [`copy_matched_selections`]) rather than a
7163    /// second box walker: the requested selection and a densely-packed
7164    /// "output" selection sharing the same `count` hold the same elements in
7165    /// the same order, so pairing the two element streams places each one
7166    /// where h5py's stepped slicing puts it, and each source box is read with
7167    /// the ordinary per-layout selective read
7168    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
7169    ///
7170    /// The single owner of read-destination semantics for stepped selections:
7171    /// it defines every byte of `out` before returning `Ok`, so a typed caller
7172    /// reads straight into the vector it keeps rather than into a byte buffer
7173    /// it then copies.
7174    pub fn read_hyperslab_into(
7175        &mut self,
7176        name: &str,
7177        start: &[u64],
7178        stride: &[u64],
7179        count: &[u64],
7180        block: &[u64],
7181        out: &mut [u8],
7182    ) -> IoResult<()> {
7183        if self.external_edge(name).is_some() {
7184            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7185            return owner.read_hyperslab_into(&path, start, stride, count, block, out);
7186        }
7187        let info = self
7188            .dataset_info_local(name)
7189            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7190        let dims = info.dataspace.dims.clone();
7191        let datatype = info.datatype.clone();
7192        let element_size = datatype.element_size() as u64;
7193        let rank = dims.len();
7194
7195        if start.len() != rank || stride.len() != rank || count.len() != rank || block.len() != rank
7196        {
7197            return Err(crate::io::IoError::InvalidState(
7198                "start/stride/count/block length must match dataset rank".into(),
7199            ));
7200        }
7201        if stride.contains(&0) {
7202            return Err(crate::io::IoError::InvalidState(
7203                "hyperslab stride must be nonzero in every dimension".into(),
7204            ));
7205        }
7206        let out_bytes = saturating_byte_len(
7207            &(0..rank)
7208                .map(|d| count[d].saturating_mul(block[d]))
7209                .collect::<Vec<u64>>(),
7210            element_size,
7211        );
7212        if out.len() as u64 != out_bytes {
7213            return Err(crate::io::IoError::InvalidState(format!(
7214                "read_hyperslab_into: buffer is {} bytes but selection needs {out_bytes}",
7215                out.len(),
7216            )));
7217        }
7218
7219        let src_sel = Selection::Hyperslab {
7220            rank,
7221            form: Hyperslab::Regular(RegularHyperslab {
7222                start: start.to_vec(),
7223                stride: stride.to_vec(),
7224                count: count.to_vec(),
7225                block: block.to_vec(),
7226            }),
7227        };
7228        let out_dims: Vec<u64> = (0..rank)
7229            .map(|d| count[d].saturating_mul(block[d]))
7230            .collect();
7231        // A densely-packed selection sharing the same `count`: it holds the
7232        // same elements as `src_sel` in the same order, and its extent is the
7233        // output buffer, so pairing the two element streams lands every
7234        // source element at its h5py stepped-slicing position.
7235        let dst_sel = Selection::Hyperslab {
7236            rank,
7237            form: Hyperslab::Regular(RegularHyperslab {
7238                start: vec![0u64; rank],
7239                stride: block.to_vec(),
7240                count: count.to_vec(),
7241                block: block.to_vec(),
7242            }),
7243        };
7244
7245        // `copy_matched_selections` is shared with the virtual-dataset
7246        // scatter, where a target selection may legitimately leave elements
7247        // untouched, so it carries no whole-output guarantee of its own: zero
7248        // first, exactly as the allocating form always did.
7249        fill_tiled_into(out, None);
7250        copy_matched_selections(
7251            |bstart, bcount, buf| self.read_slice_into_unconverted(name, bstart, bcount, buf, 0),
7252            &src_sel.resolve(&dims)?,
7253            &dst_sel.resolve(&out_dims)?,
7254            element_size,
7255            out,
7256        )?;
7257        Self::apply_post_filter_conversion(out, &datatype)
7258    }
7259
7260    /// Read a list of coordinates in one call — h5py fancy indexing with a
7261    /// coordinate list (`H5S_SEL_POINTS`) — into a caller-provided buffer.
7262    ///
7263    /// `points[i]` is a `rank`-length coordinate; `out` holds one element per
7264    /// point, `element_size` bytes each, in the same order as `points` (point
7265    /// selection order is significant, see [`PointSelection`]), so `out.len()`
7266    /// must equal `points.len() * element_size`. Backed by
7267    /// [`Selection::Points`] and [`Selection::to_boxes`] — the same
7268    /// decomposition [`read_hyperslab_into`](Self::read_hyperslab_into) and
7269    /// the virtual dataset reader use — each point's 1-element box is read
7270    /// with the ordinary per-layout selective read
7271    /// ([`read_slice_into_unconverted`](Self::read_slice_into_unconverted)).
7272    /// A 1-element box is already a flat `element_size`-byte run, so
7273    /// placing it needs no further run-decomposition
7274    /// ([`for_each_dual_run`] would degenerate to exactly this copy).
7275    ///
7276    /// One element-sized box per point covers the buffer exactly, so every
7277    /// byte of `out` is defined before it returns `Ok` and a typed caller
7278    /// reads straight into the vector it keeps.
7279    pub fn read_points_into(
7280        &mut self,
7281        name: &str,
7282        points: &[Vec<u64>],
7283        out: &mut [u8],
7284    ) -> IoResult<()> {
7285        if self.external_edge(name).is_some() {
7286            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7287            return owner.read_points_into(&path, points, out);
7288        }
7289        let info = self
7290            .dataset_info_local(name)
7291            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7292        let dims = info.dataspace.dims.clone();
7293        let datatype = info.datatype.clone();
7294        let element_size = datatype.element_size() as u64;
7295        let rank = dims.len();
7296
7297        for p in points {
7298            if p.len() != rank {
7299                return Err(crate::io::IoError::InvalidState(format!(
7300                    "point coordinate has {} entries but the dataset has {} dimensions",
7301                    p.len(),
7302                    rank
7303                )));
7304            }
7305        }
7306
7307        let sel = Selection::Points(PointSelection {
7308            rank,
7309            points: points.to_vec(),
7310        });
7311        let boxes = sel.to_boxes(&dims)?;
7312
7313        let es = element_size as usize;
7314        if out.len() != points.len() * es {
7315            return Err(crate::io::IoError::InvalidState(format!(
7316                "read_points_into: buffer is {} bytes but {} points need {}",
7317                out.len(),
7318                points.len(),
7319                points.len() * es
7320            )));
7321        }
7322        for (i, (bstart, bcount)) in boxes.iter().enumerate() {
7323            self.read_slice_into_unconverted(
7324                name,
7325                bstart,
7326                bcount,
7327                &mut out[i * es..(i + 1) * es],
7328                0,
7329            )?;
7330        }
7331        Self::apply_post_filter_conversion(out, &datatype)
7332    }
7333
7334    /// Read one chunk's raw (still-filtered) bytes and its filter mask —
7335    /// the read half of `H5Dread_chunk` (h5py:
7336    /// `Dataset.id.read_direct_chunk`).
7337    ///
7338    /// `chunk_coords` is the chunk's position in the chunk grid, one
7339    /// coordinate per dimension counted in chunks (not elements) — the same
7340    /// addressing [`write_chunk_raw_at`](crate::Dataset::write_chunk_raw_at)
7341    /// uses on the write side. The bytes returned are exactly what is
7342    /// stored on disk: filtered/compressed if the dataset has a filter
7343    /// pipeline, with no decompression applied — the caller runs the
7344    /// pipeline itself (honoring the returned mask, which marks any filter
7345    /// this particular chunk skipped) if it wants decoded data.
7346    ///
7347    /// Resolved through whichever chunk index the dataset uses, reusing the
7348    /// same per-index decoders the full/slice chunked reader walks
7349    /// ([`collect_fa_chunk_entries`](Self::collect_fa_chunk_entries),
7350    /// [`collect_ea_chunk_entries`](Self::collect_ea_chunk_entries),
7351    /// [`collect_bt2_chunk_entries`](Self::collect_bt2_chunk_entries),
7352    /// [`collect_btree_v1_chunks`](Self::collect_btree_v1_chunks)) rather
7353    /// than a new index walker.
7354    ///
7355    /// `Err` when the dataset is not chunked, `chunk_coords` has the wrong
7356    /// rank, or the chunk at those coordinates has never been written.
7357    pub fn read_chunk_raw_at(
7358        &mut self,
7359        name: &str,
7360        chunk_coords: &[u64],
7361    ) -> IoResult<(Vec<u8>, u32)> {
7362        if self.external_edge(name).is_some() {
7363            let (owner, path, _) = self.external_owner(name, MAX_EXTERNAL_HOPS)?;
7364            return owner.read_chunk_raw_at(&path, chunk_coords);
7365        }
7366        let info = self
7367            .dataset_info_local(name)
7368            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))?;
7369        let dims = info.dataspace.dims.clone();
7370        let max_dims = info.dataspace.max_dims.clone();
7371        let element_size = info.datatype.element_size() as u64;
7372        let layout = info.layout.clone();
7373        let ndims = dims.len();
7374
7375        if chunk_coords.len() != ndims {
7376            return Err(crate::io::IoError::InvalidState(format!(
7377                "chunk_coords has {} entries but the dataset has {} dimensions",
7378                chunk_coords.len(),
7379                ndims
7380            )));
7381        }
7382
7383        let not_written = || {
7384            crate::io::IoError::InvalidState(format!(
7385                "chunk at coordinates {chunk_coords:?} has not been written"
7386            ))
7387        };
7388
7389        match &layout {
7390            DataLayoutMessage::ChunkedV3 {
7391                chunk_dims,
7392                b_tree_address,
7393            } => {
7394                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7395                if *b_tree_address == UNDEF_ADDR {
7396                    return Err(not_written());
7397                }
7398                let file_size = self.handle.file_size()?;
7399                let mut entries = Vec::new();
7400                self.collect_btree_v1_chunks(*b_tree_address, ndims, file_size, 0, &mut entries)?;
7401                for (offsets, addr, chunk_size, mask) in &entries {
7402                    if *addr == UNDEF_ADDR {
7403                        continue;
7404                    }
7405                    let mut scaled = Vec::with_capacity(ndims);
7406                    for d in 0..ndims {
7407                        scaled.push(offsets[d].checked_div(real_chunk_dims[d]).unwrap_or(0));
7408                    }
7409                    if scaled.as_slice() == chunk_coords {
7410                        return Ok((self.handle.read_at(*addr, *chunk_size as usize)?, *mask));
7411                    }
7412                }
7413                Err(not_written())
7414            }
7415            DataLayoutMessage::ChunkedV4 {
7416                chunk_dims,
7417                index_type,
7418                index_address,
7419                earray_params,
7420                single_chunk_filter,
7421                ..
7422            } => {
7423                let real_chunk_dims = &chunk_dims[..chunk_dims.len() - 1];
7424                match index_type {
7425                    data_layout::ChunkIndexType::SingleChunk => {
7426                        if chunk_coords.iter().any(|&c| c != 0) {
7427                            return Err(crate::io::IoError::InvalidState(format!(
7428                                "chunk coordinates {chunk_coords:?} are outside the chunk \
7429                                 grid (0..1): this dataset has a single-chunk index"
7430                            )));
7431                        }
7432                        if *index_address == UNDEF_ADDR {
7433                            return Err(not_written());
7434                        }
7435                        match single_chunk_filter {
7436                            Some(scf) => Ok((
7437                                self.handle.read_at(*index_address, scf.nbytes as usize)?,
7438                                scf.filter_mask,
7439                            )),
7440                            None => {
7441                                let total = saturating_byte_len(&dims, element_size);
7442                                Ok((self.handle.read_at(*index_address, total as usize)?, 0))
7443                            }
7444                        }
7445                    }
7446                    data_layout::ChunkIndexType::Implicit => {
7447                        if *index_address == UNDEF_ADDR {
7448                            return Err(not_written());
7449                        }
7450                        let linear = crate::io::chunk_grid::linear_index(
7451                            &dims,
7452                            max_dims.as_deref(),
7453                            real_chunk_dims,
7454                            chunk_coords,
7455                        )?;
7456                        let chunk_bytes = saturating_byte_len(real_chunk_dims, element_size);
7457                        let addr = index_address.saturating_add(linear.saturating_mul(chunk_bytes));
7458                        Ok((self.handle.read_at(addr, chunk_bytes as usize)?, 0))
7459                    }
7460                    data_layout::ChunkIndexType::FixedArray => {
7461                        let linear = crate::io::chunk_grid::linear_index(
7462                            &dims,
7463                            max_dims.as_deref(),
7464                            real_chunk_dims,
7465                            chunk_coords,
7466                        )?;
7467                        let entries = self.collect_fa_chunk_entries(
7468                            real_chunk_dims,
7469                            ndims,
7470                            element_size,
7471                            *index_address,
7472                        )?;
7473                        match entries.get(linear as usize) {
7474                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
7475                                Ok((self.handle.read_at(addr, size as usize)?, mask))
7476                            }
7477                            _ => Err(not_written()),
7478                        }
7479                    }
7480                    data_layout::ChunkIndexType::ExtensibleArray => {
7481                        let params = earray_params.as_ref().ok_or_else(|| {
7482                            crate::io::IoError::InvalidState("missing earray params".into())
7483                        })?;
7484                        let linear = crate::io::chunk_grid::linear_index(
7485                            &dims,
7486                            max_dims.as_deref(),
7487                            real_chunk_dims,
7488                            chunk_coords,
7489                        )?;
7490                        let entries = self.collect_ea_chunk_entries(
7491                            *index_address,
7492                            params,
7493                            &dims,
7494                            max_dims.as_deref(),
7495                            real_chunk_dims,
7496                            element_size,
7497                        )?;
7498                        match entries.get(linear as usize) {
7499                            Some(&(addr, size, mask)) if addr != UNDEF_ADDR => {
7500                                Ok((self.handle.read_at(addr, size as usize)?, mask))
7501                            }
7502                            _ => Err(not_written()),
7503                        }
7504                    }
7505                    data_layout::ChunkIndexType::BTreeV2 => {
7506                        let entries = self.collect_bt2_chunk_entries(
7507                            real_chunk_dims,
7508                            ndims,
7509                            element_size,
7510                            *index_address,
7511                        )?;
7512                        match entries
7513                            .iter()
7514                            .find(|(_, _, scaled, _)| scaled.as_slice() == chunk_coords)
7515                        {
7516                            Some(&(addr, size, _, mask)) if addr != UNDEF_ADDR => {
7517                                Ok((self.handle.read_at(addr, size)?, mask))
7518                            }
7519                            _ => Err(not_written()),
7520                        }
7521                    }
7522                }
7523            }
7524            _ => Err(crate::io::IoError::InvalidState(
7525                "read_chunk_raw_at is only for chunked datasets".into(),
7526            )),
7527        }
7528    }
7529}
7530
7531/// Adapts a `FileHandle` to the `BlockReader` trait used by the fractal-heap
7532/// walker, so heap blocks can be fetched from the open file.
7533///
7534/// The handle is shared, not exclusive: every read goes through
7535/// `FileHandle::read_at_most`, which takes `&self`, so the writer can walk a
7536/// structure it is about to free while holding only `&self` itself.
7537pub(crate) struct HandleBlockReader<'a> {
7538    pub(crate) handle: &'a FileHandle,
7539}
7540
7541/// One object's attribute set, or the reason it could not be read whole.
7542///
7543/// [`AttributeEntry`] carries a per-attribute failure, which needs a name to
7544/// hang on. Two failures have none. A dense set is indexed by name *hash*, so
7545/// a heap or index that will not read yields no names at all; and an attribute
7546/// message too damaged to yield its own name cannot be listed under one
7547/// either. The only listing that can report those honestly is the object's, so
7548/// this type carries the object-scope reason beside the entries, and every
7549/// accessor that would present the set as whole returns the reason instead.
7550///
7551/// The entries are deliberately unreachable while the set is incomplete: the
7552/// only way out is [`Self::into_complete`], which refuses. That is what keeps
7553/// the writer from rebuilding an object header out of a partial set and
7554/// deleting the attributes it never saw.
7555#[derive(Debug, Clone, Default, PartialEq)]
7556pub struct ObjectAttributes {
7557    entries: Vec<AttributeEntry>,
7558    incomplete: Option<String>,
7559    /// This object's own attribute creation-order policy, from the header
7560    /// flag bits `H5Pget_attr_creation_order` reads back
7561    /// (`attribute_creation_order`) — a structural fact about the header,
7562    /// known even when the entries above are `incomplete`.
7563    creation_order: CreationOrder,
7564    /// This object's own compact-vs-dense attribute storage, from the
7565    /// attribute info message's heap address (or its absence) — likewise a
7566    /// structural fact known even when the entries are `incomplete`.
7567    storage: AttributeStorage,
7568}
7569
7570impl ObjectAttributes {
7571    /// Record an attribute this collector could name.
7572    fn push(&mut self, entry: AttributeEntry) {
7573        self.entries.push(entry);
7574    }
7575
7576    /// Record that part of the set could not be read. The first reason stands:
7577    /// it is the one that explains the earliest missing attributes.
7578    fn mark_incomplete(&mut self, reason: String) {
7579        if self.incomplete.is_none() {
7580            self.incomplete = Some(reason);
7581        }
7582    }
7583
7584    /// Why this object's attributes cannot be listed, or `None` when the set
7585    /// is whole. Individual entries in a whole set may still be undecodable —
7586    /// [`AttributeEntry::unreadable_reason`] answers for those.
7587    pub fn unreadable_reason(&self) -> Option<&str> {
7588        self.incomplete.as_deref()
7589    }
7590
7591    /// This object's own attribute creation-order policy —
7592    /// `H5Pget_attr_creation_order`'s answer, read off the object header's
7593    /// own flag bits rather than derived from the entries. Available even
7594    /// when the set is [`incomplete`](Self::unreadable_reason): it names
7595    /// nothing that failed to decode.
7596    pub fn creation_order(&self) -> CreationOrder {
7597        self.creation_order
7598    }
7599
7600    /// This object's own compact-vs-dense attribute storage — h5py's
7601    /// `h5o.get_info(...).meta_size.attr.index_size` check, read off the
7602    /// attribute info message's heap address rather than derived from the
7603    /// entries. Available even when the set is
7604    /// [`incomplete`](Self::unreadable_reason).
7605    pub fn storage(&self) -> AttributeStorage {
7606        self.storage
7607    }
7608
7609    /// This object's own attribute count as `H5Oget_info().num_attrs`
7610    /// reports it — the object header's count, not necessarily the same
7611    /// enumeration path as [`ordered_names`](Self::ordered_names).
7612    ///
7613    /// `H5O__attr_count_real` derives this from the attribute info message
7614    /// when the header carries one (the dense name-index record count, or
7615    /// the compact message count `H5O__attr_open_by_idx` already counted
7616    /// while building it) and from the raw attribute-message envelope count
7617    /// otherwise. Both reduce to the number of attributes this collector
7618    /// successfully names: a conformant writer creates the info message
7619    /// exactly when it has attributes to report through it, so a whole set's
7620    /// length already equals what libhdf5's header-count algorithm answers,
7621    /// without replaying its v1-header/v2-header branch here.
7622    pub fn header_count(&self, owner: &str) -> IoResult<u64> {
7623        Ok(self.complete(owner)?.len() as u64)
7624    }
7625
7626    /// The entries, once the set is known to be whole.
7627    ///
7628    /// The sole route from an `ObjectAttributes` to an owned entry list. A
7629    /// caller that rewrites the object header — the append path — must take
7630    /// this route, so an unread set stops the rewrite instead of erasing the
7631    /// attributes behind it.
7632    pub(crate) fn into_complete(self, owner: &str) -> IoResult<Vec<AttributeEntry>> {
7633        match self.incomplete {
7634            Some(reason) => Err(incomplete_error(owner, &reason)),
7635            None => Ok(self.entries),
7636        }
7637    }
7638
7639    /// The entries, once the set is known to be whole, borrowed.
7640    fn complete(&self, owner: &str) -> IoResult<&[AttributeEntry]> {
7641        match &self.incomplete {
7642            Some(reason) => Err(incomplete_error(owner, reason)),
7643            None => Ok(&self.entries),
7644        }
7645    }
7646
7647    /// This object's attribute names, once the set is known to be whole, in
7648    /// the order h5py's default iteration produces them: creation order when
7649    /// the object tracks it, name order otherwise
7650    /// (`H5A__compact_cmp_corder`/`H5A__compact_cmp_name` for compact
7651    /// storage, the matching v2 B-tree index for dense) — never the physical
7652    /// order the entries happen to sit in, which `entries` otherwise
7653    /// preserves for the writer's rewrite path.
7654    pub(crate) fn ordered_names(&self, owner: &str) -> IoResult<Vec<String>> {
7655        let mut ordered: Vec<&AttributeEntry> = self.complete(owner)?.iter().collect();
7656        if !ordered.is_empty() && ordered.iter().all(|e| e.creation_index().is_some()) {
7657            ordered.sort_by_key(|e| e.creation_index());
7658        } else {
7659            ordered.sort_by(|a, b| a.name().cmp(b.name()));
7660        }
7661        Ok(ordered.into_iter().map(|e| e.name().to_string()).collect())
7662    }
7663}
7664
7665/// The one wording for "this object's attributes are not all here".
7666///
7667/// `Unsupported`, the same variant an undecodable dataset message raises: the
7668/// name is in the listing and the content is out of reach, which is what the
7669/// variant is for. Wrapping a `FormatError` here instead would put the same
7670/// condition behind two different public variants.
7671fn incomplete_error(owner: &str, reason: &str) -> crate::io::IoError {
7672    crate::io::IoError::Unsupported(format!(
7673        "attributes of '{owner}' cannot be read whole: {reason}"
7674    ))
7675}
7676
7677/// Every attribute attached to an object, whichever storage it uses.
7678///
7679/// This is the only place attributes are pulled off an object header — reader
7680/// and writer alike. Compact storage keeps them as `Attribute` messages in the
7681/// header itself; once an object crosses the phase-change threshold libhdf5
7682/// moves *all* of them into a fractal heap named by the `Attribute Info`
7683/// message and leaves no attribute message behind
7684/// (`H5Oattribute.c::H5O__attr_create`). Scanning only the messages therefore
7685/// reports zero attributes for a dense object: a silent loss on read, and a
7686/// silent deletion when the writer rebuilds that object's header from what it
7687/// collected.
7688///
7689/// An attribute this crate cannot decode is kept, named, with the reason
7690/// attached: a listing that omitted it would report a file that does not
7691/// contain it. What cannot be named at all — a damaged attribute message, an
7692/// attribute info message that will not decode, a dense set whose heap or name
7693/// index will not read — marks the whole set incomplete, so the object reports
7694/// the failure rather than a short list.
7695pub(crate) fn collect_object_attributes(
7696    handle: &mut FileHandle,
7697    ctx: &FormatContext,
7698    header: &ObjectHeader,
7699) -> ObjectAttributes {
7700    let mut attrs = ObjectAttributes {
7701        creation_order: header.attribute_creation_order(),
7702        ..ObjectAttributes::default()
7703    };
7704    // The message envelope carries a creation index only when the header says
7705    // the object tracks one; the field is not even encoded otherwise
7706    // (`H5O_SIZEOF_MSGHDR_OH`), so reading it as an index would report zero
7707    // for every attribute of an untracked object.
7708    let tracked = header.has_creation_order();
7709    for msg in &header.messages {
7710        match msg.msg_type {
7711            MSG_ATTRIBUTE => match AttributeEntry::parse(&msg.data, ctx) {
7712                Ok(entry) => {
7713                    attrs.push(entry.with_creation_index(tracked.then_some(msg.creation_index)))
7714                }
7715                Err(e) => attrs.mark_incomplete(format!("an attribute message is unreadable: {e}")),
7716            },
7717            MSG_ATTR_INFO => match AttributeInfoMessage::decode(&msg.data, ctx) {
7718                Ok((info, _)) => {
7719                    attrs.storage = if info.is_dense() {
7720                        AttributeStorage::Dense
7721                    } else {
7722                        AttributeStorage::Compact
7723                    };
7724                    let mut br = HandleBlockReader { handle };
7725                    match crate::format::dense_attr::read_dense_attributes(&info, ctx, &mut br) {
7726                        Ok(dense) => attrs.entries.extend(dense),
7727                        Err(e) => attrs
7728                            .mark_incomplete(format!("dense attribute storage is unreadable: {e}")),
7729                    }
7730                }
7731                Err(e) => {
7732                    attrs.mark_incomplete(format!("the attribute info message is unreadable: {e}"))
7733                }
7734            },
7735            _ => {}
7736        }
7737    }
7738    attrs
7739}
7740
7741impl BlockReader for HandleBlockReader<'_> {
7742    fn read_block(&mut self, offset: u64, len: usize) -> crate::format::FormatResult<Vec<u8>> {
7743        // `read_at_most`, not `read_at`: a metadata block allocated at the end
7744        // of the file can be shorter on disk than its nominal size, and every
7745        // decoder re-checks the length it needs.
7746        self.handle.read_at_most(offset, len).map_err(|e| {
7747            crate::format::FormatError::InvalidData(format!(
7748                "metadata block read failed at {:#x}: {}",
7749                offset, e
7750            ))
7751        })
7752    }
7753}
7754
7755#[cfg(test)]
7756mod tests {
7757    use super::*;
7758    use std::io::Write;
7759
7760    /// Per-call unique temp path. PID + atomic counter avoids
7761    /// path collisions across concurrent cargo invocations and
7762    /// kernel-side flock release races.
7763    fn temp_path(name: &str) -> std::path::PathBuf {
7764        use std::sync::atomic::{AtomicU64, Ordering};
7765        static COUNTER: AtomicU64 = AtomicU64::new(0);
7766        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
7767        std::env::temp_dir().join(format!(
7768            "rust_hdf5_reader_test_{}_{}_{}.h5",
7769            name,
7770            std::process::id(),
7771            n
7772        ))
7773    }
7774
7775    /// Write `bytes` to a fresh temp file and open a read handle on it.
7776    fn handle_over(name: &str, bytes: &[u8]) -> (std::path::PathBuf, FileHandle) {
7777        let path = temp_path(name);
7778        std::fs::File::create(&path)
7779            .unwrap()
7780            .write_all(bytes)
7781            .unwrap();
7782        let handle =
7783            FileHandle::open_read_with_locking(&path, crate::io::locking::FileLocking::Disabled)
7784                .unwrap();
7785        (path, handle)
7786    }
7787
7788    /// The chunk-index cache. A decoded index MUST describe the file exactly
7789    /// as the catalog entry beside it does, and MUST NOT be served for an
7790    /// index address other than the one it was decoded from.
7791    /// [`Hdf5Reader::decoded_chunk_index`] is the single owner — the only
7792    /// reader and writer of the cache — and `DatasetTable` is what keeps a
7793    /// stale index unreachable: `entry_mut` is the one way to a mutable
7794    /// entry and it drops that entry's index first, where the `IndexMut` it
7795    /// replaced would have left it standing.
7796    mod chunk_index_cache {
7797        use super::*;
7798
7799        /// `d` = `[0, 1, .., 7]` in two chunks of four, fixed extent, so the
7800        /// layout points at a fixed array a read has to walk.
7801        fn chunked_file(name: &str) -> (std::path::PathBuf, Hdf5Reader) {
7802            let path = temp_path(name);
7803            {
7804                let file = crate::H5File::create(&path).unwrap();
7805                let ds = file
7806                    .new_dataset::<u8>()
7807                    .shape([8usize])
7808                    .chunk(&[4])
7809                    .create("d")
7810                    .unwrap();
7811                ds.write_raw(&(0u8..8).collect::<Vec<u8>>()).unwrap();
7812                file.close().unwrap();
7813            }
7814            let reader = Hdf5Reader::open(&path).unwrap();
7815            (path, reader)
7816        }
7817
7818        /// The address of the chunk index `d`'s layout points at.
7819        fn index_address(reader: &Hdf5Reader, pos: usize) -> u64 {
7820            match &reader.datasets[pos].layout {
7821                DataLayoutMessage::ChunkedV4 { index_address, .. } => *index_address,
7822                _ => panic!("expected a version-4 chunked layout"),
7823            }
7824        }
7825
7826        #[test]
7827        fn a_read_leaves_its_decoded_index_for_the_next_read() {
7828            let (path, mut r) = chunked_file("index_cache_hit");
7829            let pos = r.dataset_position("d").unwrap();
7830            let addr = index_address(&r, pos);
7831            assert!(
7832                r.datasets.chunk_index(pos, addr).is_none(),
7833                "a reader that has read nothing holds a decoded index"
7834            );
7835
7836            assert_eq!(r.read_slice("d", &[0], &[4]).unwrap(), vec![0u8, 1, 2, 3]);
7837            let cached = r
7838                .datasets
7839                .chunk_index(pos, addr)
7840                .expect("the read did not keep the index it decoded");
7841            assert_eq!(cached.entries.len(), 2, "one entry per chunk");
7842            assert_eq!(cached.coords, vec![0, 1], "slot 0 and slot 1 of the grid");
7843
7844            // The second read answers from it and reaches the same bytes.
7845            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
7846            let _ = std::fs::remove_file(&path);
7847        }
7848
7849        #[test]
7850        fn an_index_is_never_served_for_another_address() {
7851            let (path, mut r) = chunked_file("index_cache_addr");
7852            let pos = r.dataset_position("d").unwrap();
7853            let addr = index_address(&r, pos);
7854            r.read_slice("d", &[0], &[4]).unwrap();
7855
7856            assert!(
7857                r.datasets.chunk_index(pos, addr.wrapping_add(1)).is_none(),
7858                "an index decoded from {addr:#x} answered for another address"
7859            );
7860            let _ = std::fs::remove_file(&path);
7861        }
7862
7863        #[test]
7864        fn changing_an_entry_drops_the_index_decoded_against_it() {
7865            let (path, mut r) = chunked_file("index_cache_entry_mut");
7866            let pos = r.dataset_position("d").unwrap();
7867            let addr = index_address(&r, pos);
7868            r.read_slice("d", &[0], &[4]).unwrap();
7869            assert!(r.datasets.chunk_index(pos, addr).is_some());
7870
7871            // What a caller reaches an entry mutably for — the virtual extent
7872            // resolution rewrites `dataspace.dims` — is what the coordinates
7873            // in a decoded index were built against.
7874            let _ = r.datasets.entry_mut(pos);
7875            assert!(
7876                r.datasets.chunk_index(pos, addr).is_none(),
7877                "a mutable entry left its decoded index behind"
7878            );
7879            assert_eq!(r.read_slice("d", &[4], &[4]).unwrap(), vec![4u8, 5, 6, 7]);
7880            let _ = std::fs::remove_file(&path);
7881        }
7882    }
7883
7884    /// The decompressed-chunk cache. An image MUST be served only for the
7885    /// stored bytes it was decoded from, MUST be kept only for a chunk the
7886    /// read left partly unconsumed, and MUST NOT outlive the decoded index
7887    /// that named its chunk — which it cannot, being a field of it, so the
7888    /// invalidation these tests exercise is the index's own (`entry_mut`, and
7889    /// the table rebuild a SWMR refresh does).
7890    mod chunk_image_cache {
7891        use super::*;
7892
7893        /// `d` = a ramp of `n` f64 in chunks of `chunk`, deflated or not.
7894        #[cfg(feature = "deflate")]
7895        fn dataset(name: &str, n: usize, chunk: usize, deflate: bool) -> std::path::PathBuf {
7896            let path = temp_path(name);
7897            let data: Vec<f64> = (0..n).map(|i| (i % 251) as f64).collect();
7898            let file = crate::H5File::create(&path).unwrap();
7899            let mut b = file.new_dataset::<f64>().shape([n]).chunk(&[chunk]);
7900            if deflate {
7901                b = b.deflate(6);
7902            }
7903            b.create("d").unwrap().write_raw(&data).unwrap();
7904            file.close().unwrap();
7905            path
7906        }
7907
7908        /// `d` = a ramp of `rows * cols` f64 in `chunk`-shaped chunks. A
7909        /// two-dimensional chunk is never one contiguous stretch of a read's
7910        /// output, so every chunk of it is staged and materialized — which is
7911        /// what leaves the cache free to keep it, where a one-dimensional
7912        /// whole-chunk read would have decoded straight into the output.
7913        fn dataset_2d(
7914            name: &str,
7915            rows: usize,
7916            cols: usize,
7917            chunk: [usize; 2],
7918            deflate: bool,
7919        ) -> std::path::PathBuf {
7920            let path = temp_path(name);
7921            let data: Vec<f64> = (0..rows * cols).map(|i| (i % 251) as f64).collect();
7922            let file = crate::H5File::create(&path).unwrap();
7923            let mut b = file
7924                .new_dataset::<f64>()
7925                .shape([rows, cols])
7926                .chunk(&chunk[..]);
7927            if deflate {
7928                b = b.deflate(6);
7929            }
7930            b.create("d").unwrap().write_raw(&data).unwrap();
7931            file.close().unwrap();
7932            path
7933        }
7934
7935        /// How many images the reader holds for `d`.
7936        fn held(reader: &Hdf5Reader) -> usize {
7937            let pos = reader.dataset_position("d").unwrap();
7938            let DataLayoutMessage::ChunkedV4 { index_address, .. } = &reader.datasets[pos].layout
7939            else {
7940                panic!("expected a version-4 chunked layout");
7941            };
7942            match reader.datasets.chunk_index(pos, *index_address) {
7943                Some(index) => index.images.held(),
7944                None => 0,
7945            }
7946        }
7947
7948        /// Four consecutive quarter-chunk slices are one chunk read four
7949        /// times: the first decodes it, the other three place it out of the
7950        /// image it left, and all four read what the file holds.
7951        #[test]
7952        #[cfg(feature = "deflate")]
7953        fn a_partial_read_keeps_the_chunk_it_did_not_finish() {
7954            let path = dataset("images_partial", 4096, 1024, true);
7955            let mut r = Hdf5Reader::open(&path).unwrap();
7956            let whole = {
7957                let mut fresh = Hdf5Reader::open(&path).unwrap();
7958                fresh.read_dataset_raw("d").unwrap()
7959            };
7960
7961            for q in 0..4u64 {
7962                let got = r.read_slice("d", &[q * 256], &[256]).unwrap();
7963                let from = (q as usize) * 256 * 8;
7964                assert_eq!(got, whole[from..from + 256 * 8], "quarter {q}");
7965                assert_eq!(held(&r), 1, "quarter {q} left the wrong image count");
7966            }
7967            let _ = std::fs::remove_file(&path);
7968        }
7969
7970        /// A whole-dataset read takes every chunk entire, so there is nothing
7971        /// a later read could want more of: it keeps nothing, and never pays
7972        /// to copy an image it is about to throw away.
7973        ///
7974        /// Two-dimensional on purpose. A one-dimensional full read decodes
7975        /// each chunk straight into the output and so never has an image to
7976        /// offer in the first place; here every chunk really is materialized,
7977        /// and only "the read consumed it" keeps it out of the cache.
7978        #[test]
7979        #[cfg(feature = "deflate")]
7980        fn a_whole_dataset_read_keeps_nothing() {
7981            let path = dataset_2d("images_full", 256, 256, [64, 64], true);
7982            let mut r = Hdf5Reader::open(&path).unwrap();
7983            r.read_dataset_raw("d").unwrap();
7984            assert_eq!(held(&r), 0, "a full read kept a chunk image");
7985            let _ = std::fs::remove_file(&path);
7986        }
7987
7988        /// The extent cutting the last chunk short does not make a full read
7989        /// look partial: what that chunk has to give is what the extent
7990        /// reaches, and the read took all of it.
7991        #[test]
7992        #[cfg(feature = "deflate")]
7993        fn a_whole_read_of_a_ragged_extent_keeps_nothing() {
7994            let path = dataset("images_ragged", 3000, 1024, true);
7995            let mut r = Hdf5Reader::open(&path).unwrap();
7996            r.read_dataset_raw("d").unwrap();
7997            assert_eq!(held(&r), 0, "the ragged edge chunk was kept");
7998            let _ = std::fs::remove_file(&path);
7999        }
8000
8001        /// An unfiltered chunk never reaches the cache, whatever a read does
8002        /// with it: its stored bytes are the dataset's bytes, so a partial
8003        /// read has nothing to inflate and nothing to save by inflating once.
8004        ///
8005        /// A single-column selection is the case that bites. Its runs are one
8006        /// element each, far too many for a positioned read apiece, so
8007        /// `read_chunk_runs_into` declines and every chunk really is read and
8008        /// materialized whole — leaving an image the cache would take if it
8009        /// were offered one.
8010        #[test]
8011        fn an_unfiltered_partial_read_keeps_nothing() {
8012            let path = dataset_2d("images_plain", 256, 256, [64, 64], false);
8013            let mut r = Hdf5Reader::open(&path).unwrap();
8014            let column = r.read_slice("d", &[0, 0], &[256, 1]).unwrap();
8015            assert_eq!(column.len(), 256 * 8);
8016            assert_eq!(held(&r), 0, "an unfiltered chunk reached the cache");
8017            let _ = std::fs::remove_file(&path);
8018        }
8019
8020        /// The same single-column walk over a *filtered* dataset is the
8021        /// region-of-interest case the cache exists for: each column of chunks
8022        /// inflates once however many columns are read out of it.
8023        #[test]
8024        #[cfg(feature = "deflate")]
8025        fn a_filtered_column_walk_reuses_each_chunk() {
8026            let path = dataset_2d("images_column", 256, 256, [64, 64], true);
8027            let mut warm = Hdf5Reader::open(&path).unwrap();
8028            for c in 0..4u64 {
8029                let mut cold = Hdf5Reader::open(&path).unwrap();
8030                assert_eq!(
8031                    warm.read_slice("d", &[0, c], &[256, 1]).unwrap(),
8032                    cold.read_slice("d", &[0, c], &[256, 1]).unwrap(),
8033                    "column {c} differed once served from the cache"
8034                );
8035            }
8036            // The four chunks of the first chunk-column, still held after the
8037            // fourth read of them.
8038            assert_eq!(held(&warm), 4, "the column walk re-inflated its chunks");
8039            let _ = std::fs::remove_file(&path);
8040        }
8041
8042        /// A chunk whose image is larger than the whole budget would evict
8043        /// everything to hold one entry, so it is never kept.
8044        #[test]
8045        #[cfg(feature = "deflate")]
8046        fn a_chunk_over_the_budget_is_never_kept() {
8047            let over = CHUNK_CACHE_BYTES / 8 + 1;
8048            let path = dataset("images_oversize", over * 2, over, true);
8049            let mut r = Hdf5Reader::open(&path).unwrap();
8050            r.read_slice("d", &[0], &[1024]).unwrap();
8051            assert_eq!(held(&r), 0, "a chunk over the budget was kept");
8052            let _ = std::fs::remove_file(&path);
8053        }
8054
8055        /// The budget's boundary, on the state machine itself: an image of
8056        /// exactly the budget is kept, and one byte more is refused — refused
8057        /// rather than admitted and then evicted, so that what the cache
8058        /// already holds survives an image that was never going to fit.
8059        #[test]
8060        fn an_image_over_the_budget_never_evicts_the_ones_under_it() {
8061            let cache = ChunkImageCache::default();
8062            let key = |addr| ChunkImageKey {
8063                addr,
8064                len: 1,
8065                mask: 0,
8066            };
8067            for i in 0..3 {
8068                cache.keep(key(i), vec![0u8; CHUNK_CACHE_BYTES / 4]);
8069            }
8070            assert_eq!(cache.held(), 3, "three quarter-budget images did not fit");
8071
8072            cache.keep(key(9), vec![0u8; CHUNK_CACHE_BYTES + 1]);
8073            assert_eq!(
8074                cache.held(),
8075                3,
8076                "an image that cannot fit evicted the ones that did"
8077            );
8078
8079            cache.keep(key(8), vec![0u8; CHUNK_CACHE_BYTES]);
8080            assert_eq!(
8081                cache.held(),
8082                1,
8083                "an image of exactly the budget was refused"
8084            );
8085        }
8086
8087        /// The budget is bytes: eight partial reads of eight 256 KiB chunks
8088        /// leave the four the budget pays for, not eight.
8089        #[test]
8090        #[cfg(feature = "deflate")]
8091        fn the_cache_holds_no_more_than_its_byte_budget() {
8092            let chunk = 32 * 1024; // f64 -> 256 KiB
8093            let path = dataset("images_budget", chunk * 8, chunk, true);
8094            let mut r = Hdf5Reader::open(&path).unwrap();
8095            for c in 0..8u64 {
8096                r.read_slice("d", &[c * chunk as u64], &[1024]).unwrap();
8097            }
8098            assert_eq!(
8099                held(&r),
8100                CHUNK_CACHE_BYTES / (chunk * 8),
8101                "the cache outgrew its byte budget"
8102            );
8103            let _ = std::fs::remove_file(&path);
8104        }
8105
8106        /// Reaching an entry mutably drops the index decoded against it, and
8107        /// the images live in that index — so what a read cached against the
8108        /// old entry cannot be served against the new one.
8109        #[test]
8110        #[cfg(feature = "deflate")]
8111        fn changing_an_entry_drops_the_chunk_images() {
8112            let path = dataset("images_entry_mut", 4096, 1024, true);
8113            let mut r = Hdf5Reader::open(&path).unwrap();
8114            r.read_slice("d", &[0], &[256]).unwrap();
8115            assert_eq!(held(&r), 1, "the read kept no image to invalidate");
8116
8117            let pos = r.dataset_position("d").unwrap();
8118            let _ = r.datasets.entry_mut(pos);
8119            assert_eq!(held(&r), 0, "a mutable entry left its chunk images behind");
8120            let _ = std::fs::remove_file(&path);
8121        }
8122
8123        /// Every slice a cache-holding reader returns is the slice a reader
8124        /// that never cached anything returns, over slices that start inside
8125        /// a chunk, end inside one, and span several.
8126        #[test]
8127        #[cfg(feature = "deflate")]
8128        fn a_cached_chunk_places_what_a_cold_read_places() {
8129            let path = dataset("images_differential", 4096, 1024, true);
8130            let mut warm = Hdf5Reader::open(&path).unwrap();
8131            for &(start, count) in &[
8132                (0u64, 100u64),
8133                (100, 100),
8134                (900, 300),
8135                (1024, 1),
8136                (1500, 2000),
8137                (3000, 1096),
8138                (4095, 1),
8139                (0, 4096),
8140            ] {
8141                let mut cold = Hdf5Reader::open(&path).unwrap();
8142                assert_eq!(
8143                    warm.read_slice("d", &[start], &[count]).unwrap(),
8144                    cold.read_slice("d", &[start], &[count]).unwrap(),
8145                    "slice {start}+{count} differed once served from the cache"
8146                );
8147            }
8148            let _ = std::fs::remove_file(&path);
8149        }
8150    }
8151
8152    /// `place_chunk_jobs` is the single owner of "every byte of a chunked
8153    /// read's output is defined": it skips the blanket fill only when
8154    /// [`planned_coverage`] proves the plan reaches every output byte, and
8155    /// fills whatever a plan or a short image leaves behind. The buffer these
8156    /// hand it is poisoned, so a byte no one defines shows up as `0xAA`.
8157    mod chunk_fill_plan {
8158        use super::*;
8159
8160        /// dims [8] of 1-byte elements in chunks of 4: slot 0 holds `ABCD` at
8161        /// file offset 0, slot 1 holds `EFGH` at offset 4.
8162        const DIMS: [u64; 1] = [8];
8163        const CHUNKS: [u64; 1] = [4];
8164        const FILL: [u8; 1] = [0x7E];
8165
8166        fn place(
8167            name: &str,
8168            jobs: Vec<Option<ChunkReadJob>>,
8169            coords: &[u64],
8170        ) -> (std::path::PathBuf, Vec<u8>) {
8171            let (path, handle) = handle_over(name, b"ABCDEFGH");
8172            let geo = ChunkOutputGeometry {
8173                dims: &DIMS,
8174                chunk_dims: &CHUNKS,
8175                element_size: 1,
8176            };
8177            let mut out = vec![0xAAu8; 8];
8178            place_chunk_jobs(
8179                &handle,
8180                jobs,
8181                coords,
8182                ChunkReadRequest {
8183                    pipeline: None,
8184                    target: ChunkTarget::Full,
8185                    fill_value: Some(&FILL),
8186                },
8187                &geo,
8188                None,
8189                &mut out,
8190            )
8191            .unwrap();
8192            (path, out)
8193        }
8194
8195        fn job(addr: u64, len: usize) -> Option<ChunkReadJob> {
8196            Some(ChunkReadJob {
8197                addr,
8198                len,
8199                at_most: true,
8200                mask: 0,
8201            })
8202        }
8203
8204        /// The owner path: a plan that tiles the output places every byte, so
8205        /// nothing is filled and nothing is left poisoned.
8206        #[test]
8207        fn a_plan_that_covers_the_output_leaves_no_byte_to_fill() {
8208            let geo = ChunkOutputGeometry {
8209                dims: &DIMS,
8210                chunk_dims: &CHUNKS,
8211                element_size: 1,
8212            };
8213            let zeros = [0u64];
8214            let placement = ChunkPlacement::resolve(&geo, ChunkTarget::Full, &zeros);
8215            let jobs = vec![job(0, 4), job(4, 4)];
8216            assert_eq!(
8217                planned_coverage(&placement, &jobs, &[0, 1]),
8218                8,
8219                "a tiling plan must measure as covering the whole output"
8220            );
8221            let (path, out) = place("cover_full", jobs, &[0, 1]);
8222            assert_eq!(out, b"ABCDEFGH");
8223            let _ = std::fs::remove_file(path);
8224        }
8225
8226        /// A slot the plan never reaches reads back as the tiled fill value,
8227        /// which is what the blanket pre-fill used to guarantee.
8228        #[test]
8229        fn a_slot_the_plan_skips_reads_back_as_fill() {
8230            let (path, out) = place("cover_gap", vec![job(0, 4), None], &[0, 1]);
8231            assert_eq!(out, b"ABCD~~~~");
8232            let _ = std::fs::remove_file(path);
8233        }
8234
8235        /// A corrupt index naming one chunk-grid slot twice must not be
8236        /// mistaken for a plan that covers two slots: counting the duplicate
8237        /// once leaves the coverage short, so the fill still goes down and the
8238        /// slot no entry named reads as fill rather than as poison.
8239        #[test]
8240        fn duplicate_chunk_coordinates_still_leave_defined_output() {
8241            let (path, out) = place("cover_dup", vec![job(0, 4), job(0, 4)], &[0, 0]);
8242            assert_eq!(out, b"ABCD~~~~");
8243            let _ = std::fs::remove_file(path);
8244        }
8245
8246        /// A chunk whose stored image is shorter than the read box needs it to
8247        /// be places no run at all; the plan counted it, so the shortfall is
8248        /// filled after placement instead of before it.
8249        #[test]
8250        fn a_chunk_read_short_has_its_runs_filled_after_placement() {
8251            let (path, out) = place("cover_short", vec![job(0, 4), job(4, 2)], &[0, 1]);
8252            assert_eq!(out, b"ABCD~~~~");
8253            let _ = std::fs::remove_file(path);
8254        }
8255    }
8256
8257    /// Helper: write a little-endian u64 truncated to `n` bytes.
8258    fn write_le(buf: &mut Vec<u8>, value: u64, n: usize) {
8259        buf.extend_from_slice(&value.to_le_bytes()[..n]);
8260    }
8261
8262    /// Build a minimal v0 HDF5 file in memory with one dataset containing
8263    /// `dataset_data`. Returns the complete file bytes.
8264    ///
8265    /// The file structure is:
8266    /// - Superblock v0 with root group STE
8267    /// - Root group object header (v1) with symbol table message
8268    /// - Local heap (header + data) with dataset name
8269    /// - B-tree v1 (group, leaf) pointing to one SNOD
8270    /// - SNOD with one entry for the dataset
8271    /// - Dataset object header (v1) with dataspace, datatype, layout messages
8272    /// - Raw dataset data (contiguous)
8273    fn build_v0_file(dataset_name: &str, dims: &[u64], data: &[u8]) -> Vec<u8> {
8274        let sa: usize = 8; // sizeof_addr
8275        let ss: usize = 8; // sizeof_size
8276        let ndims = dims.len();
8277        let element_size = data.len() as u64 / dims.iter().product::<u64>();
8278
8279        // We'll lay out the file regions in order, computing offsets as we go.
8280        let mut file = Vec::new();
8281
8282        // ---- Plan layout offsets ----
8283        // We need to know the addresses before writing, so let's compute them.
8284        // Superblock: starts at 0
8285        let sb_size = 8 + 8 + 4 + 4 * sa + (ss + sa + 4 + 4 + 16); // sig + header + flags + 4 addrs + STE
8286                                                                   // Pad to 8-byte alignment
8287        let sb_size_aligned = (sb_size + 7) & !7;
8288
8289        // Root group object header (v1): after superblock
8290        let root_ohdr_addr = sb_size_aligned as u64;
8291        // The root ohdr contains a symbol table message (type 0x11):
8292        //   btree_addr(8) + heap_addr(8) = 16 bytes
8293        // v1 message wire format: type(2) + size(2) + flags(1) + reserved(3) + data
8294        let stab_msg_data_size = 2 * sa; // btree + heap addr
8295        let stab_msg_wire = 8 + stab_msg_data_size;
8296        let stab_msg_wire_aligned = (stab_msg_wire + 7) & !7;
8297        let root_ohdr_data_size = stab_msg_wire_aligned;
8298        let root_ohdr_total = 16 + root_ohdr_data_size; // v1 16-byte prefix + messages
8299        let root_ohdr_total_aligned = (root_ohdr_total + 7) & !7;
8300
8301        // Local heap header: after root ohdr
8302        let heap_hdr_addr = root_ohdr_addr + root_ohdr_total_aligned as u64;
8303        let heap_hdr_size = 4 + 1 + 3 + ss + ss + sa;
8304        let heap_hdr_size_aligned = (heap_hdr_size + 7) & !7;
8305
8306        // Local heap data: after heap header
8307        let heap_data_addr = heap_hdr_addr + heap_hdr_size_aligned as u64;
8308        // Data: empty string at offset 0 (for root), then dataset_name at offset 1
8309        let name_bytes = dataset_name.as_bytes();
8310        let heap_data_content_size = 1 + name_bytes.len() + 1; // \0 + name + \0
8311        let heap_data_size = (heap_data_content_size + 7) & !7;
8312
8313        // B-tree v1 node: after heap data
8314        let btree_addr = heap_data_addr + heap_data_size as u64;
8315        // B-tree header: TREE(4) + type(1) + level(1) + entries_used(2) + left(sa) + right(sa)
8316        // Plus interleaved keys/children: key[0](ss), child[0](sa), key[1](ss)
8317        let btree_size = 4 + 1 + 1 + 2 + 2 * sa + 2 * ss + sa;
8318        let btree_size_aligned = (btree_size + 7) & !7;
8319
8320        // SNOD: after B-tree
8321        let snod_addr = btree_addr + btree_size_aligned as u64;
8322        // SNOD header: SNOD(4) + version(1) + reserved(1) + num_symbols(2)
8323        // + 1 entry: name_offset(ss) + obj_header_addr(sa) + cache_type(4) + reserved(4) + scratch(16)
8324        let entry_size = ss + sa + 4 + 4 + 16;
8325        let snod_size = 8 + entry_size;
8326        let snod_size_aligned = (snod_size + 7) & !7;
8327
8328        // Dataset object header (v1): after SNOD
8329        let ds_ohdr_addr = snod_addr + snod_size_aligned as u64;
8330        // Messages: dataspace(0x01), datatype(0x03), data_layout(0x08)
8331
8332        // Dataspace v1: version(1) + ndims(1) + flags(1) + reserved(1) + reserved(4) + ndims*ss
8333        let ds_msg_data_size = 8 + ndims * ss;
8334        let ds_msg_wire = 8 + ds_msg_data_size;
8335        let ds_msg_wire_aligned = (ds_msg_wire + 7) & !7;
8336
8337        // Datatype: for integer types, 12 bytes
8338        // Use i32: class=0, version=1, size=4, bit_offset=0, bit_precision=32, signed
8339        let dt_msg_data_size = 12;
8340        let dt_msg_wire = 8 + dt_msg_data_size;
8341        let dt_msg_wire_aligned = (dt_msg_wire + 7) & !7;
8342
8343        // Data layout v3 contiguous: version(1) + class(1) + addr(sa) + size(ss)
8344        let dl_msg_data_size = 2 + sa + ss;
8345        let dl_msg_wire = 8 + dl_msg_data_size;
8346        let dl_msg_wire_aligned = (dl_msg_wire + 7) & !7;
8347
8348        let ds_ohdr_data_size = ds_msg_wire_aligned + dt_msg_wire_aligned + dl_msg_wire_aligned;
8349        let ds_ohdr_total = 16 + ds_ohdr_data_size; // v1 16-byte prefix
8350        let ds_ohdr_total_aligned = (ds_ohdr_total + 7) & !7;
8351
8352        // Raw data: after dataset object header
8353        let raw_data_addr = ds_ohdr_addr + ds_ohdr_total_aligned as u64;
8354        let raw_data_size = data.len();
8355
8356        let eof = raw_data_addr + raw_data_size as u64;
8357
8358        // ---- Write the file ----
8359
8360        // 1. Superblock v0
8361        let sig: [u8; 8] = [0x89, 0x48, 0x44, 0x46, 0x0d, 0x0a, 0x1a, 0x0a];
8362        file.extend_from_slice(&sig);
8363        file.push(0); // version 0
8364        file.push(0); // free-space version
8365        file.push(0); // root group STE version
8366        file.push(0); // reserved
8367        file.push(0); // shared header version
8368        file.push(sa as u8); // sizeof_addr
8369        file.push(ss as u8); // sizeof_size
8370        file.push(0); // reserved
8371        file.extend_from_slice(&4u16.to_le_bytes()); // sym_leaf_k
8372        file.extend_from_slice(&32u16.to_le_bytes()); // btree_internal_k
8373        file.extend_from_slice(&0u32.to_le_bytes()); // file_consistency_flags
8374                                                     // base_addr
8375        write_le(&mut file, 0, sa);
8376        // extension_addr = UNDEF
8377        write_le(&mut file, UNDEF_ADDR, sa);
8378        // eof_addr
8379        write_le(&mut file, eof, sa);
8380        // driver_info_addr = UNDEF
8381        write_le(&mut file, UNDEF_ADDR, sa);
8382        // Root group STE:
8383        write_le(&mut file, 0, ss); // name_offset
8384        write_le(&mut file, root_ohdr_addr, sa); // obj_header_addr
8385        file.extend_from_slice(&1u32.to_le_bytes()); // cache_type = 1 (stab)
8386        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
8387                                                     // scratch pad: btree_addr + heap_addr
8388        write_le(&mut file, btree_addr, sa);
8389        write_le(&mut file, heap_hdr_addr, sa);
8390        // Pad superblock
8391        while file.len() < sb_size_aligned {
8392            file.push(0);
8393        }
8394
8395        // 2. Root group object header (v1, 16-byte prefix)
8396        assert_eq!(file.len(), root_ohdr_addr as usize);
8397        file.push(1); // version
8398        file.push(0); // reserved
8399        file.extend_from_slice(&1u16.to_le_bytes()); // num_messages = 1
8400        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
8401        file.extend_from_slice(&(root_ohdr_data_size as u32).to_le_bytes());
8402        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)
8403                                           // Symbol table message (type 0x0011)
8404        file.extend_from_slice(&0x0011u16.to_le_bytes()); // type
8405        file.extend_from_slice(&(stab_msg_data_size as u16).to_le_bytes()); // size
8406        file.push(0); // flags
8407        file.extend_from_slice(&[0u8; 3]); // reserved
8408        write_le(&mut file, btree_addr, sa);
8409        write_le(&mut file, heap_hdr_addr, sa);
8410        // Pad
8411        while file.len() < (root_ohdr_addr as usize + root_ohdr_total_aligned) {
8412            file.push(0);
8413        }
8414
8415        // 3. Local heap header
8416        assert_eq!(file.len(), heap_hdr_addr as usize);
8417        file.extend_from_slice(b"HEAP");
8418        file.push(0); // version
8419        file.extend_from_slice(&[0u8; 3]); // reserved
8420        write_le(&mut file, heap_data_size as u64, ss); // data_size
8421        write_le(&mut file, u64::MAX, ss); // free_list_offset (none)
8422        write_le(&mut file, heap_data_addr, sa); // data_addr
8423        while file.len() < (heap_hdr_addr as usize + heap_hdr_size_aligned) {
8424            file.push(0);
8425        }
8426
8427        // 4. Local heap data
8428        assert_eq!(file.len(), heap_data_addr as usize);
8429        file.push(0); // offset 0: empty string (root self-reference)
8430        file.extend_from_slice(name_bytes); // offset 1: dataset name
8431        file.push(0); // null terminator
8432        while file.len() < (heap_data_addr as usize + heap_data_size) {
8433            file.push(0);
8434        }
8435
8436        // 5. B-tree v1 (leaf, 1 entry)
8437        assert_eq!(file.len(), btree_addr as usize);
8438        file.extend_from_slice(b"TREE");
8439        file.push(0); // type = group
8440        file.push(0); // level = leaf
8441        file.extend_from_slice(&1u16.to_le_bytes()); // entries_used = 1
8442        write_le(&mut file, UNDEF_ADDR, sa); // left sibling
8443        write_le(&mut file, UNDEF_ADDR, sa); // right sibling
8444                                             // key[0] = 0 (first name offset)
8445        write_le(&mut file, 0, ss);
8446        // child[0] = snod_addr
8447        write_le(&mut file, snod_addr, sa);
8448        // key[1] = dataset name offset (after root)
8449        write_le(&mut file, 1, ss);
8450        while file.len() < (btree_addr as usize + btree_size_aligned) {
8451            file.push(0);
8452        }
8453
8454        // 6. SNOD with 1 entry
8455        assert_eq!(file.len(), snod_addr as usize);
8456        file.extend_from_slice(b"SNOD");
8457        file.push(1); // version
8458        file.push(0); // reserved
8459        file.extend_from_slice(&1u16.to_le_bytes()); // num_symbols = 1
8460                                                     // Entry: dataset
8461        write_le(&mut file, 1, ss); // name_offset = 1 (index into local heap)
8462        write_le(&mut file, ds_ohdr_addr, sa); // obj_header_addr
8463        file.extend_from_slice(&0u32.to_le_bytes()); // cache_type = 0 (not a group)
8464        file.extend_from_slice(&0u32.to_le_bytes()); // reserved
8465        file.extend_from_slice(&[0u8; 16]); // scratch pad (unused)
8466        while file.len() < (snod_addr as usize + snod_size_aligned) {
8467            file.push(0);
8468        }
8469
8470        // 7. Dataset object header (v1, 16-byte prefix)
8471        assert_eq!(file.len(), ds_ohdr_addr as usize);
8472        file.push(1); // version
8473        file.push(0); // reserved
8474        file.extend_from_slice(&3u16.to_le_bytes()); // num_messages = 3
8475        file.extend_from_slice(&1u32.to_le_bytes()); // obj_ref_count
8476        file.extend_from_slice(&(ds_ohdr_data_size as u32).to_le_bytes());
8477        file.extend_from_slice(&[0u8; 4]); // reserved padding (v1 alignment)
8478
8479        // Message 1: Dataspace (type 0x01) - version 1
8480        file.extend_from_slice(&0x0001u16.to_le_bytes());
8481        file.extend_from_slice(&(ds_msg_data_size as u16).to_le_bytes());
8482        file.push(0); // flags
8483        file.extend_from_slice(&[0u8; 3]); // reserved
8484                                           // Dataspace v1 payload:
8485        file.push(1); // version = 1
8486        file.push(ndims as u8);
8487        file.push(0); // flags (no max dims)
8488        file.push(0); // reserved
8489        file.extend_from_slice(&[0u8; 4]); // reserved (4 bytes)
8490        for &d in dims {
8491            write_le(&mut file, d, ss);
8492        }
8493        // Pad message
8494        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned;
8495        while file.len() < target {
8496            file.push(0);
8497        }
8498
8499        // Message 2: Datatype (type 0x03) - i32
8500        file.extend_from_slice(&0x0003u16.to_le_bytes());
8501        file.extend_from_slice(&(dt_msg_data_size as u16).to_le_bytes());
8502        file.push(0); // flags
8503        file.extend_from_slice(&[0u8; 3]); // reserved
8504                                           // Datatype payload: class=0 (fixed point), version=1
8505        file.push(0x10); // class(0) | version(1)<<4
8506        file.push(0x08); // byte_order=LE, signed=true (bit 3)
8507        file.push(0); // flags byte 1
8508        file.push(0); // flags byte 2
8509        file.extend_from_slice(&(element_size as u32).to_le_bytes()); // element size
8510        file.extend_from_slice(&0u16.to_le_bytes()); // bit_offset
8511        file.extend_from_slice(&((element_size * 8) as u16).to_le_bytes()); // bit_precision
8512        let target = ds_ohdr_addr as usize + 16 + ds_msg_wire_aligned + dt_msg_wire_aligned;
8513        while file.len() < target {
8514            file.push(0);
8515        }
8516
8517        // Message 3: Data Layout (type 0x08) - contiguous v3
8518        file.extend_from_slice(&0x0008u16.to_le_bytes());
8519        file.extend_from_slice(&(dl_msg_data_size as u16).to_le_bytes());
8520        file.push(0); // flags
8521        file.extend_from_slice(&[0u8; 3]); // reserved
8522                                           // Data layout payload:
8523        file.push(3); // version = 3
8524        file.push(1); // class = contiguous
8525        write_le(&mut file, raw_data_addr, sa); // address
8526        write_le(&mut file, raw_data_size as u64, ss); // size
8527        let target = ds_ohdr_addr as usize + ds_ohdr_total_aligned;
8528        while file.len() < target {
8529            file.push(0);
8530        }
8531
8532        // 8. Raw data
8533        assert_eq!(file.len(), raw_data_addr as usize);
8534        file.extend_from_slice(data);
8535
8536        assert_eq!(file.len(), eof as usize);
8537        file
8538    }
8539
8540    #[test]
8541    fn test_read_v0_file_with_one_dataset() {
8542        let dims = [3u64, 4];
8543        let values: Vec<i32> = (0..12).collect();
8544        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8545
8546        let file_bytes = build_v0_file("my_dataset", &dims, &raw_data);
8547
8548        // Write to a temp file
8549        let path = temp_path("v0_reader");
8550        {
8551            let mut f = std::fs::File::create(&path).unwrap();
8552            f.write_all(&file_bytes).unwrap();
8553            f.sync_all().unwrap();
8554        }
8555
8556        // Read it back
8557        let mut reader = Hdf5Reader::open(&path).unwrap();
8558        let names = reader.dataset_names();
8559        assert_eq!(names, vec!["my_dataset"]);
8560
8561        let shape = reader.dataset_shape("my_dataset").unwrap();
8562        assert_eq!(shape, vec![3, 4]);
8563
8564        let data = reader.read_dataset_raw("my_dataset").unwrap();
8565        assert_eq!(data, raw_data);
8566
8567        // Verify the values
8568        let read_values: Vec<i32> = data
8569            .chunks_exact(4)
8570            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
8571            .collect();
8572        assert_eq!(read_values, values);
8573
8574        std::fs::remove_file(&path).ok();
8575    }
8576
8577    #[test]
8578    fn test_read_v0_file_1d_dataset() {
8579        let dims = [5u64];
8580        let values: Vec<i32> = vec![100, 200, 300, 400, 500];
8581        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8582
8583        let file_bytes = build_v0_file("data_1d", &dims, &raw_data);
8584
8585        let path = temp_path("v0_1d");
8586        {
8587            let mut f = std::fs::File::create(&path).unwrap();
8588            f.write_all(&file_bytes).unwrap();
8589        }
8590
8591        let mut reader = Hdf5Reader::open(&path).unwrap();
8592        assert_eq!(reader.dataset_names(), vec!["data_1d"]);
8593        assert_eq!(reader.dataset_shape("data_1d").unwrap(), vec![5]);
8594
8595        let data = reader.read_dataset_raw("data_1d").unwrap();
8596        let read_values: Vec<i32> = data
8597            .chunks_exact(4)
8598            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
8599            .collect();
8600        assert_eq!(read_values, values);
8601
8602        std::fs::remove_file(&path).ok();
8603    }
8604
8605    #[test]
8606    fn test_detect_v2v3_still_works() {
8607        // Verify that opening a v3 file written by our writer still works
8608        let path = temp_path("detect_v3");
8609        {
8610            use crate::io::writer::Hdf5Writer;
8611            let writer = Hdf5Writer::create(&path).unwrap();
8612            let datatype = crate::format::messages::datatype::DatatypeMessage::i32_type();
8613            let idx = writer.create_dataset("test", datatype, &[4]).unwrap();
8614            let data = [1i32, 2, 3, 4];
8615            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
8616            writer.write_dataset_raw(idx, &raw).unwrap();
8617            writer.close().unwrap();
8618        }
8619
8620        let mut reader = Hdf5Reader::open(&path).unwrap();
8621        assert_eq!(reader.dataset_names(), vec!["test"]);
8622        let shape = reader.dataset_shape("test").unwrap();
8623        assert_eq!(shape, vec![4]);
8624
8625        let data = reader.read_dataset_raw("test").unwrap();
8626        let vals: Vec<i32> = data
8627            .chunks_exact(4)
8628            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
8629            .collect();
8630        assert_eq!(vals, vec![1, 2, 3, 4]);
8631
8632        std::fs::remove_file(&path).ok();
8633    }
8634
8635    /// Collect the (src_off, out_off, len) runs the coalescer emits.
8636    fn collect_runs(
8637        dims: &[u64],
8638        starts: &[u64],
8639        counts: &[u64],
8640        es: u64,
8641    ) -> Vec<(u64, usize, usize)> {
8642        let mut v = Vec::new();
8643        for_each_contiguous_run(dims, starts, counts, es, |s, o, l| {
8644            v.push((s, o, l));
8645            Ok(())
8646        })
8647        .unwrap();
8648        v
8649    }
8650
8651    #[test]
8652    fn coalesce_1d_is_single_run() {
8653        // 1-D selection is always one contiguous run.
8654        assert_eq!(collect_runs(&[10], &[2], &[3], 1), vec![(2, 0, 3)]);
8655        // element_size scales offsets and length.
8656        assert_eq!(collect_runs(&[10], &[2], &[3], 4), vec![(8, 0, 12)]);
8657    }
8658
8659    #[test]
8660    fn coalesce_full_last_dim_merges_into_one_run() {
8661        // 2-D, full last dim => the whole [r0:r1, :] block is one run.
8662        // dims=[4,5], select rows 1..3, all 5 columns.
8663        assert_eq!(collect_runs(&[4, 5], &[1, 0], &[2, 5], 1), vec![(5, 0, 10)]);
8664    }
8665
8666    #[test]
8667    fn coalesce_partial_last_dim_keeps_one_run_per_row() {
8668        // 2-D, partial last dim => no merge; one run per selected row.
8669        // dims=[4,5] strides=[5,1]; select rows 1..3, cols 1..4.
8670        assert_eq!(
8671            collect_runs(&[4, 5], &[1, 1], &[2, 3], 1),
8672            vec![(6, 0, 3), (11, 3, 3)]
8673        );
8674        // Same shape, element_size=4: strides=[20,4], inner_base=4.
8675        assert_eq!(
8676            collect_runs(&[4, 5], &[1, 1], &[2, 3], 4),
8677            vec![(24, 0, 12), (44, 12, 12)]
8678        );
8679    }
8680
8681    #[test]
8682    fn coalesce_3d_reported_case_one_run_per_outer_index() {
8683        // The reported workload: [:, r0:r1, :] of [nproj, nz, nx].
8684        // dims=[3,4,5] strides=[20,5,1]; select all of dim0, rows 1..3 of
8685        // dim1, all of dim2. Last dim full => merge dim1+dim2; dim1 partial
8686        // => one run per dim0 index (3 runs, not 3*2=6 rows).
8687        assert_eq!(
8688            collect_runs(&[3, 4, 5], &[0, 1, 0], &[3, 2, 5], 1),
8689            vec![(5, 0, 10), (25, 10, 10), (45, 20, 10)]
8690        );
8691    }
8692
8693    #[test]
8694    fn coalesce_3d_full_inner_dims_is_single_run() {
8695        // [r0:r1, :, :] => both inner dims full => one contiguous run.
8696        // dims=[3,4,5] strides=[20,5,1]; select rows 1..3 of dim0.
8697        assert_eq!(
8698            collect_runs(&[3, 4, 5], &[1, 0, 0], &[2, 4, 5], 1),
8699            vec![(20, 0, 40)]
8700        );
8701    }
8702
8703    /// Build a contiguous i32 dataset and verify `read_slice` returns the
8704    /// correct bytes for both coalesced and non-coalesced selections.
8705    #[test]
8706    fn read_slice_contiguous_3d_matches_naive_extraction() {
8707        let dims = [3u64, 4, 5];
8708        let total: usize = (dims[0] * dims[1] * dims[2]) as usize;
8709        let values: Vec<i32> = (0..total as i32).collect();
8710        let raw_data: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
8711        let file_bytes = build_v0_file("vol", &dims, &raw_data);
8712
8713        let path = temp_path("slice_3d_contig");
8714        {
8715            let mut f = std::fs::File::create(&path).unwrap();
8716            f.write_all(&file_bytes).unwrap();
8717            f.sync_all().unwrap();
8718        }
8719        let mut reader = Hdf5Reader::open(&path).unwrap();
8720
8721        // Naive row-major extraction for an arbitrary [starts, counts).
8722        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
8723            let mut out = Vec::new();
8724            for i in 0..counts[0] {
8725                for j in 0..counts[1] {
8726                    for k in 0..counts[2] {
8727                        let gi = starts[0] + i;
8728                        let gj = starts[1] + j;
8729                        let gk = starts[2] + k;
8730                        out.push(values[(gi * dims[1] * dims[2] + gj * dims[2] + gk) as usize]);
8731                    }
8732                }
8733            }
8734            out
8735        };
8736        let decode = |raw: Vec<u8>| -> Vec<i32> {
8737            raw.chunks_exact(4)
8738                .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
8739                .collect()
8740        };
8741
8742        // Mix of coalesced and non-coalesced selections.
8743        let cases: &[([u64; 3], [u64; 3])] = &[
8744            ([0, 1, 0], [3, 2, 5]), // [:, 1:3, :]  -> coalesced (3 runs)
8745            ([1, 0, 0], [2, 4, 5]), // [1:3, :, :]  -> single run
8746            ([0, 0, 1], [3, 4, 3]), // [:, :, 1:4]  -> partial last dim, no merge
8747            ([1, 2, 1], [2, 2, 4]), // interior block, partial all dims
8748            ([0, 0, 0], [3, 4, 5]), // whole dataset -> single run
8749            ([2, 3, 4], [1, 1, 1]), // single element
8750        ];
8751        for &(starts, counts) in cases {
8752            let got = decode(reader.read_slice("vol", &starts, &counts).unwrap());
8753            assert_eq!(
8754                got,
8755                expect(starts, counts),
8756                "slice starts={starts:?} counts={counts:?}"
8757            );
8758        }
8759
8760        let _ = std::fs::remove_file(&path);
8761    }
8762
8763    /// Run the collector against a header built by hand. The handle is only
8764    /// touched when a message sends the collector to the heap, which none of
8765    /// these do, so an empty file is enough of a file.
8766    fn collect_from(
8767        messages: Vec<crate::format::object_header::ObjectHeaderMessage>,
8768    ) -> Result<Vec<String>, String> {
8769        let path = temp_path("collect");
8770        std::fs::File::create(&path).unwrap();
8771        let mut handle = FileHandle::open_read(&path).unwrap();
8772        let ctx = FormatContext {
8773            sizeof_addr: 8,
8774            sizeof_size: 8,
8775        };
8776        let header = ObjectHeader {
8777            flags: 0x02,
8778            times: None,
8779            messages,
8780        };
8781        let attrs = collect_object_attributes(&mut handle, &ctx, &header);
8782        drop(handle);
8783        let _ = std::fs::remove_file(&path);
8784        attrs
8785            .complete("obj")
8786            .map(|e| e.iter().map(|a| a.name().to_string()).collect())
8787            .map_err(|e| e.to_string())
8788    }
8789
8790    fn msg(msg_type: u8, data: Vec<u8>) -> crate::format::object_header::ObjectHeaderMessage {
8791        crate::format::object_header::ObjectHeaderMessage {
8792            msg_type,
8793            flags: 0,
8794            data,
8795            creation_index: 0,
8796        }
8797    }
8798
8799    /// An attribute info message that will not decode takes the object's whole
8800    /// attribute set with it — the dense storage it names is where the
8801    /// attributes are. The listing must say so rather than come back short.
8802    #[test]
8803    fn an_undecodable_attribute_info_message_fails_the_listing() {
8804        // Version 9: `H5O_AINFO_VERSION_0` is the only one that exists.
8805        let err = collect_from(vec![msg(MSG_ATTR_INFO, vec![9, 0])]).unwrap_err();
8806        assert!(
8807            err.contains("attributes of 'obj' cannot be read whole")
8808                && err.contains("attribute info message"),
8809            "{err}"
8810        );
8811    }
8812
8813    /// An attribute message damaged past its own name has no name to be listed
8814    /// under, so it too is an object-level failure — the one case
8815    /// [`AttributeEntry::parse`] cannot name.
8816    #[test]
8817    fn an_unnameable_attribute_message_fails_the_listing() {
8818        let err = collect_from(vec![msg(MSG_ATTRIBUTE, vec![1, 0, 0])]).unwrap_err();
8819        assert!(
8820            err.contains("attributes of 'obj' cannot be read whole")
8821                && err.contains("attribute message"),
8822            "{err}"
8823        );
8824    }
8825
8826    /// The failure is the object's, not one name's: a set that reads whole
8827    /// still lists, and a compact attribute info message is not dense storage.
8828    #[test]
8829    fn a_whole_attribute_set_still_lists() {
8830        let ctx = FormatContext {
8831            sizeof_addr: 8,
8832            sizeof_size: 8,
8833        };
8834        let ainfo = crate::format::messages::attr_info::AttributeInfoMessage::compact();
8835        let names = collect_from(vec![msg(MSG_ATTR_INFO, ainfo.encode(&ctx))]).unwrap();
8836        assert!(names.is_empty(), "{names:?}");
8837    }
8838
8839    /// A space-padded fixed-length string attribute reads back without its
8840    /// padding: `H5T__conv_s_s` ends the value after the last non-space byte,
8841    /// and nothing else in the element marks where it stops.
8842    #[test]
8843    fn fixed_string_attr_value_honors_the_declared_pad() {
8844        use crate::format::messages::dataspace::DataspaceMessage;
8845        use crate::format::messages::datatype::DatatypeMessage;
8846
8847        let attr = |padding: u8, data: &[u8]| AttributeMessage {
8848            name: "units".to_string(),
8849            datatype: DatatypeMessage::FixedString {
8850                size: data.len() as u32,
8851                padding,
8852                charset: 0,
8853            },
8854            dataspace: DataspaceMessage::scalar(),
8855            data: data.to_vec(),
8856        };
8857
8858        // Space padded: no NUL anywhere, so truncating at the first NUL kept
8859        // the padding.
8860        assert_eq!(
8861            fixed_string_attr_value(&attr(2, b"volt    ")).unwrap(),
8862            "volt"
8863        );
8864        // Null terminated and null padded end at the first NUL, so trailing
8865        // spaces before it are content.
8866        assert_eq!(
8867            fixed_string_attr_value(&attr(0, b"volt  \0\0")).unwrap(),
8868            "volt  "
8869        );
8870        assert_eq!(
8871            fixed_string_attr_value(&attr(1, b"volt\0\0\0\0")).unwrap(),
8872            "volt"
8873        );
8874        // A reserved rule is named rather than guessed at.
8875        let err = fixed_string_attr_value(&attr(7, b"volt    ")).unwrap_err();
8876        assert!(
8877            err.to_string().contains("padding rule 7"),
8878            "unexpected error: {err}"
8879        );
8880    }
8881}
8882
8883#[cfg(test)]
8884mod h5py_debug_tests {
8885    use super::*;
8886
8887    #[test]
8888    fn debug_read_h5py() {
8889        let path = std::path::Path::new("/tmp/test_h5py_default.h5");
8890        if !path.exists() {
8891            return;
8892        }
8893
8894        let handle = FileHandle::open_read(path).unwrap();
8895        let sb_buf = handle.read_at_most(0, 1024).unwrap();
8896        let version = detect_superblock_version(&sb_buf).unwrap();
8897        eprintln!("Superblock version: {}", version);
8898
8899        let sb = SuperblockV0V1::decode(&sb_buf).unwrap();
8900        eprintln!(
8901            "sizeof_addr={}, sizeof_size={}",
8902            sb.sizeof_offsets, sb.sizeof_lengths
8903        );
8904        let (ste_btree, ste_heap) = sb
8905            .root_symbol_table_entry
8906            .cached_symbol_table()
8907            .unwrap_or((UNDEF_ADDR, UNDEF_ADDR));
8908        eprintln!(
8909            "STE: obj_header={}, cache={:?}, btree={}, heap={}",
8910            sb.root_symbol_table_entry.obj_header_addr,
8911            sb.root_symbol_table_entry.cache,
8912            ste_btree,
8913            ste_heap
8914        );
8915
8916        let ctx = FormatContext {
8917            sizeof_addr: sb.sizeof_offsets,
8918            sizeof_size: sb.sizeof_lengths,
8919        };
8920
8921        // Read local heap
8922        let heap_buf = handle.read_at_most(ste_heap, 128).unwrap();
8923        let heap_hdr = LocalHeapHeader::decode(
8924            &heap_buf,
8925            ctx.sizeof_addr as usize,
8926            ctx.sizeof_size as usize,
8927        )
8928        .unwrap();
8929        eprintln!(
8930            "Heap data_addr={}, data_size={}",
8931            heap_hdr.data_addr, heap_hdr.data_size
8932        );
8933
8934        let heap_data = handle
8935            .read_at(heap_hdr.data_addr, heap_hdr.data_size as usize)
8936            .unwrap();
8937        eprintln!(
8938            "Heap data bytes: {:?}",
8939            &heap_data[..std::cmp::min(64, heap_data.len())]
8940        );
8941
8942        // Read btree
8943        let btree_buf = handle.read_at_most(ste_btree, 8192).unwrap();
8944        let btree = BTreeV1Node::decode(
8945            &btree_buf,
8946            ctx.sizeof_addr as usize,
8947            ctx.sizeof_size as usize,
8948            BTreeV1Config::default().snode_max_entries(),
8949        )
8950        .unwrap();
8951        eprintln!(
8952            "BTree: type={}, level={}, entries={}, children={:?}",
8953            btree.node_type, btree.level, btree.entries_used, btree.children
8954        );
8955
8956        // Read SNOD
8957        for &child in &btree.children {
8958            let snod_buf = handle.read_at_most(child, 8192).unwrap();
8959            let snod = SymbolTableNode::decode(
8960                &snod_buf,
8961                ctx.sizeof_addr as usize,
8962                ctx.sizeof_size as usize,
8963                BTreeV1Config::default().sym_leaf_max_entries(),
8964            )
8965            .unwrap();
8966            eprintln!("SNOD at {}: {} entries", child, snod.entries.len());
8967            for entry in &snod.entries {
8968                let name = local_heap_get_string(&heap_data, entry.name_offset).unwrap();
8969                eprintln!(
8970                    "  entry: name='{}' (offset={}), obj_header={}, cache={:?}",
8971                    name, entry.name_offset, entry.obj_header_addr, entry.cache
8972                );
8973            }
8974        }
8975
8976        // Try full open
8977        let reader = Hdf5Reader::open(path).unwrap();
8978        eprintln!("Datasets found: {:?}", reader.dataset_names());
8979    }
8980
8981    // ====================================================================
8982    // Group/link discovery: continuation blocks, dense links, v0/v1 groups.
8983    //
8984    // These tests generate HDF5 fixtures with h5py (HDF5 2.0.0). If the
8985    // pinned Python interpreter is not present, the test skips so the suite
8986    // still runs in environments without it.
8987    // ====================================================================
8988
8989    const TEST_PYTHON: &str = "/Users/stevek/mamba/envs/bs2026.1/bin/python";
8990
8991    /// Per-call unique temp path (PID + atomic counter) to avoid collisions
8992    /// across concurrent test runs.
8993    fn temp_path(name: &str) -> std::path::PathBuf {
8994        use std::sync::atomic::{AtomicU64, Ordering};
8995        static COUNTER: AtomicU64 = AtomicU64::new(0);
8996        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
8997        std::env::temp_dir().join(format!(
8998            "rust_hdf5_gap_test_{}_{}_{}.h5",
8999            name,
9000            std::process::id(),
9001            n
9002        ))
9003    }
9004
9005    /// Run a Python snippet to generate a fixture; returns false if Python
9006    /// is unavailable so the caller can skip the test.
9007    fn gen_fixture(script: &str) -> bool {
9008        if !std::path::Path::new(TEST_PYTHON).exists() {
9009            return false;
9010        }
9011        let status = std::process::Command::new(TEST_PYTHON)
9012            .arg("-c")
9013            .arg(script)
9014            .status();
9015        matches!(status, Ok(s) if s.success())
9016    }
9017
9018    #[test]
9019    fn gap1_v2_root_continuation_block() {
9020        let path = temp_path("gap1_cont");
9021        let p = path.display().to_string();
9022        // ~6 datasets in a v2 root group forces an object-header
9023        // continuation block.
9024        let script = format!(
9025            "import h5py,numpy as np\n\
9026             f=h5py.File(r'{p}','w',libver='latest')\n\
9027             [f.create_dataset('ds_%d'%i,data=np.arange(i*10,i*10+10,dtype='int32')) for i in range(6)]\n\
9028             f.close()"
9029        );
9030        if !gen_fixture(&script) {
9031            eprintln!("skipping gap1: python unavailable");
9032            return;
9033        }
9034
9035        let mut reader = Hdf5Reader::open(&path).unwrap();
9036        let mut names = reader.dataset_names();
9037        names.sort();
9038        assert_eq!(
9039            names,
9040            vec!["ds_0", "ds_1", "ds_2", "ds_3", "ds_4", "ds_5"],
9041            "all 6 datasets must be found across the continuation block"
9042        );
9043        // Element-exact read of one dataset.
9044        let raw = reader.read_dataset_raw("ds_3").unwrap();
9045        let vals: Vec<i32> = raw
9046            .chunks_exact(4)
9047            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9048            .collect();
9049        assert_eq!(vals, (30..40).collect::<Vec<i32>>());
9050        let _ = std::fs::remove_file(&path);
9051    }
9052
9053    #[test]
9054    fn gap2_v2_dense_fractal_heap_links() {
9055        let path = temp_path("gap2_dense");
9056        let p = path.display().to_string();
9057        // 14 datasets in one v2 group forces dense (fractal-heap) link
9058        // storage.
9059        let script = format!(
9060            "import h5py,numpy as np\n\
9061             f=h5py.File(r'{p}','w',libver='latest')\n\
9062             g=f.create_group('dense')\n\
9063             [g.create_dataset('d%02d'%i,data=np.full(4,i,dtype='float64')) for i in range(14)]\n\
9064             f.close()"
9065        );
9066        if !gen_fixture(&script) {
9067            eprintln!("skipping gap2: python unavailable");
9068            return;
9069        }
9070
9071        let mut reader = Hdf5Reader::open(&path).unwrap();
9072        let mut names = reader.dataset_names();
9073        names.sort();
9074        let expected: Vec<String> = (0..14).map(|i| format!("dense/d{:02}", i)).collect();
9075        assert_eq!(
9076            names, expected,
9077            "all 14 dense-stored links must be recovered from the fractal heap"
9078        );
9079        // Element-exact read of one dense-stored dataset.
9080        let raw = reader.read_dataset_raw("dense/d07").unwrap();
9081        let vals: Vec<f64> = raw
9082            .chunks_exact(8)
9083            .map(|c| f64::from_le_bytes([c[0], c[1], c[2], c[3], c[4], c[5], c[6], c[7]]))
9084            .collect();
9085        assert_eq!(vals, vec![7.0; 4]);
9086        let _ = std::fs::remove_file(&path);
9087    }
9088
9089    #[test]
9090    fn gap3_v0v1_legacy_subgroups() {
9091        let path = temp_path("gap3_legacy");
9092        let p = path.display().to_string();
9093        // libver='earliest' => v0 superblock, symbol-table groups; datasets
9094        // nested inside subgroups.
9095        let script = format!(
9096            "import h5py,numpy as np\n\
9097             f=h5py.File(r'{p}','w',libver='earliest')\n\
9098             g1=f.create_group('grp1')\n\
9099             g1.create_dataset('a',data=np.arange(5,dtype='int16'))\n\
9100             g2=g1.create_group('sub')\n\
9101             g2.create_dataset('b',data=np.arange(7,dtype='int64'))\n\
9102             f.create_dataset('top',data=np.arange(3,dtype='int32'))\n\
9103             f.close()"
9104        );
9105        if !gen_fixture(&script) {
9106            eprintln!("skipping gap3: python unavailable");
9107            return;
9108        }
9109
9110        let mut reader = Hdf5Reader::open(&path).unwrap();
9111        let mut names = reader.dataset_names();
9112        names.sort();
9113        assert_eq!(
9114            names,
9115            vec!["grp1/a", "grp1/sub/b", "top"],
9116            "datasets nested in legacy symbol-table subgroups must be found"
9117        );
9118        // Element-exact read of a doubly-nested dataset.
9119        let raw = reader.read_dataset_raw("grp1/sub/b").unwrap();
9120        let vals: Vec<i64> = raw
9121            .chunks_exact(8)
9122            .map(|c| i64::from_le_bytes([c[0], c[1], c[2], c[3], c[4], c[5], c[6], c[7]]))
9123            .collect();
9124        assert_eq!(vals, (0..7).collect::<Vec<i64>>());
9125        let _ = std::fs::remove_file(&path);
9126    }
9127
9128    /// N-bit chunked datasets with non-zero bit offset and signed types with
9129    /// negative values must read back element-exact through the crate's
9130    /// chunked readers. The post-filter datatype conversion shifts/masks/
9131    /// sign-extends each element after the filter pipeline.
9132    #[test]
9133    fn nbit_chunked_post_filter_conversion() {
9134        let path = temp_path("nbit_conv");
9135        let p = path.display().to_string();
9136        // Build N-bit datasets with reduced precision + non-zero offset via
9137        // h5py's low-level filter API (h5py has no high-level N-bit knob).
9138        let script = format!(
9139            "import h5py,numpy as np\n\
9140             from h5py import h5t,h5p,h5s,h5d,h5f,h5z\n\
9141             fid=h5f.create(r'{p}'.encode())\n\
9142             def mk(name,bt,prec,off,npd,vals,chunk):\n\
9143            \x20dt=bt.copy();dt.set_precision(prec);dt.set_offset(off)\n\
9144            \x20arr=np.ascontiguousarray(np.asarray(vals,dtype=npd))\n\
9145            \x20sp=h5s.create_simple(arr.shape)\n\
9146            \x20dc=h5p.create(h5p.DATASET_CREATE);dc.set_chunk(chunk)\n\
9147            \x20dc.set_filter(h5z.FILTER_NBIT,h5z.FLAG_OPTIONAL,())\n\
9148            \x20ds=h5d.create(fid,name.encode(),dt,sp,dc)\n\
9149            \x20ds.write(h5s.ALL,h5s.ALL,arr);ds.close()\n\
9150             mk('u4_p17_o3',h5t.STD_U32LE,17,3,'u4',[0,1,1000,65535,131071,70000,42,99999],(4,))\n\
9151             mk('i4_p13_o5',h5t.STD_I32LE,13,5,'i4',[-5,-1,0,1,7,-4096,4095,-77,42,100,-100,3],(4,))\n\
9152             mk('i2_p9_o4',h5t.STD_I16LE,9,4,'i2',[-256,-1,0,1,255,-7,7,-200],(3,))\n\
9153             mk('i4_2d_p11_o6',h5t.STD_I32LE,11,6,'i4',np.array([[-1024,-1,0,5],[1023,-77,88,-3]],dtype='i4'),(1,4))\n\
9154             fid.close()"
9155        );
9156        if !gen_fixture(&script) {
9157            eprintln!("skipping nbit_chunked_post_filter_conversion: python unavailable");
9158            return;
9159        }
9160
9161        let mut reader = Hdf5Reader::open(&path).unwrap();
9162
9163        // Unsigned u4, precision 17, bit offset 3.
9164        let raw = reader.read_dataset_raw("u4_p17_o3").unwrap();
9165        let got: Vec<u32> = raw
9166            .chunks_exact(4)
9167            .map(|c| u32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9168            .collect();
9169        assert_eq!(
9170            got,
9171            vec![0u32, 1, 1000, 65535, 131071, 70000, 42, 99999],
9172            "u4 N-bit dataset must decode to exact unsigned values"
9173        );
9174
9175        // Signed i4 with negatives, precision 13, bit offset 5.
9176        let raw = reader.read_dataset_raw("i4_p13_o5").unwrap();
9177        let got: Vec<i32> = raw
9178            .chunks_exact(4)
9179            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9180            .collect();
9181        assert_eq!(
9182            got,
9183            vec![-5i32, -1, 0, 1, 7, -4096, 4095, -77, 42, 100, -100, 3],
9184            "i4 N-bit dataset must sign-extend negative values"
9185        );
9186
9187        // Signed i2 with negatives, precision 9, bit offset 4.
9188        let raw = reader.read_dataset_raw("i2_p9_o4").unwrap();
9189        let got: Vec<i16> = raw
9190            .chunks_exact(2)
9191            .map(|c| i16::from_le_bytes([c[0], c[1]]))
9192            .collect();
9193        assert_eq!(
9194            got,
9195            vec![-256i16, -1, 0, 1, 255, -7, 7, -200],
9196            "i2 N-bit dataset must sign-extend negative values"
9197        );
9198
9199        // 2D signed i4, precision 11, bit offset 6 (1-row chunks).
9200        let raw = reader.read_dataset_raw("i4_2d_p11_o6").unwrap();
9201        let got: Vec<i32> = raw
9202            .chunks_exact(4)
9203            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9204            .collect();
9205        assert_eq!(
9206            got,
9207            vec![-1024i32, -1, 0, 5, 1023, -77, 88, -3],
9208            "2D i4 N-bit dataset must decode element-exact"
9209        );
9210
9211        // read_slice path must also apply the conversion exactly once.
9212        let raw = reader.read_slice("i4_p13_o5", &[4], &[3]).unwrap();
9213        let got: Vec<i32> = raw
9214            .chunks_exact(4)
9215            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9216            .collect();
9217        assert_eq!(got, vec![7i32, -4096, 4095], "read_slice must convert too");
9218
9219        // 2D slice: second row, all columns.
9220        let raw = reader.read_slice("i4_2d_p11_o6", &[1, 0], &[1, 4]).unwrap();
9221        let got: Vec<i32> = raw
9222            .chunks_exact(4)
9223            .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9224            .collect();
9225        assert_eq!(
9226            got,
9227            vec![1023i32, -77, 88, -3],
9228            "2D read_slice must convert"
9229        );
9230
9231        let _ = std::fs::remove_file(&path);
9232    }
9233
9234    /// Partial-slice reads of chunked datasets must skip non-overlapping
9235    /// chunks yet return exactly the selected region, for every chunk index
9236    /// type: v1 B-tree (libver=earliest), single chunk, fixed array,
9237    /// extensible array (one unlimited dim), and v2 B-tree (>1 unlimited dim).
9238    ///
9239    /// The dataset is a 5×4×6 int32 `arange`, chunked 2×2×2 so the chunk grid
9240    /// is ragged (edge chunks) and most selections touch a strict subset of
9241    /// chunks. Each slice is checked against the row-major `arange` value so a
9242    /// dropped/misplaced chunk or a mis-sized output buffer is caught.
9243    #[test]
9244    fn read_slice_chunked_all_index_types() {
9245        let latest = temp_path("slice_chunk_latest");
9246        let earliest = temp_path("slice_chunk_earliest");
9247        let pl = latest.display().to_string();
9248        let pe = earliest.display().to_string();
9249        // libver=latest selects modern indices by maxshape: fixed -> Fixed
9250        // Array, one unlimited dim -> Extensible Array, >1 unlimited -> v2
9251        // B-tree, single chunk -> Single Chunk index. libver=earliest always
9252        // uses the v1 B-tree chunk index.
9253        let script = format!(
9254            "import h5py,numpy as np\n\
9255             a=np.arange(5*4*6,dtype='int32').reshape(5,4,6)\n\
9256             f=h5py.File(r'{pl}','w',libver='latest')\n\
9257             f.create_dataset('single',data=a,chunks=(5,4,6))\n\
9258             f.create_dataset('fa',data=a,chunks=(2,2,2))\n\
9259             f.create_dataset('ea',data=a,chunks=(2,2,2),maxshape=(None,4,6))\n\
9260             f.create_dataset('btv2',data=a,chunks=(2,2,2),maxshape=(None,None,6))\n\
9261             f.close()\n\
9262             g=h5py.File(r'{pe}','w',libver='earliest')\n\
9263             g.create_dataset('btv1',data=a,chunks=(2,2,2))\n\
9264             g.close()"
9265        );
9266        if !gen_fixture(&script) {
9267            eprintln!("skipping read_slice_chunked_all_index_types: python unavailable");
9268            return;
9269        }
9270
9271        let dims = [5u64, 4, 6];
9272        // Row-major value of element (i,j,k) in the arange dataset.
9273        let val = |i: u64, j: u64, k: u64| (i * dims[1] * dims[2] + j * dims[2] + k) as i32;
9274        let expect = |starts: [u64; 3], counts: [u64; 3]| -> Vec<i32> {
9275            let mut out = Vec::new();
9276            for i in 0..counts[0] {
9277                for j in 0..counts[1] {
9278                    for k in 0..counts[2] {
9279                        out.push(val(starts[0] + i, starts[1] + j, starts[2] + k));
9280                    }
9281                }
9282            }
9283            out
9284        };
9285        let decode = |raw: Vec<u8>| -> Vec<i32> {
9286            raw.chunks_exact(4)
9287                .map(|c| i32::from_le_bytes([c[0], c[1], c[2], c[3]]))
9288                .collect()
9289        };
9290        let cases: &[([u64; 3], [u64; 3])] = &[
9291            ([0, 1, 0], [5, 2, 6]), // full last dim, partial mid -> coalesced runs
9292            ([1, 0, 0], [3, 4, 6]), // partial dim0, inner dims full -> one run
9293            ([0, 0, 2], [5, 4, 3]), // partial last dim -> one run per (i,j) row
9294            ([2, 1, 3], [1, 2, 2]), // interior block spanning few chunks
9295            ([0, 0, 0], [5, 4, 6]), // whole dataset via read_slice
9296            ([4, 3, 5], [1, 1, 1]), // single element at the far edge chunk
9297            ([1, 1, 1], [3, 3, 4]), // straddles chunk boundaries on all axes
9298        ];
9299
9300        let mut reader_l = Hdf5Reader::open(&latest).unwrap();
9301        for name in ["single", "fa", "ea", "btv2"] {
9302            // Full read sanity first, then every slice.
9303            let full = decode(reader_l.read_dataset_raw(name).unwrap());
9304            assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "{name} full read");
9305            for &(starts, counts) in cases {
9306                let got = decode(reader_l.read_slice(name, &starts, &counts).unwrap());
9307                assert_eq!(
9308                    got,
9309                    expect(starts, counts),
9310                    "{name} slice starts={starts:?} counts={counts:?}"
9311                );
9312            }
9313        }
9314
9315        let mut reader_e = Hdf5Reader::open(&earliest).unwrap();
9316        let full = decode(reader_e.read_dataset_raw("btv1").unwrap());
9317        assert_eq!(full, expect([0, 0, 0], [5, 4, 6]), "btv1 full read");
9318        for &(starts, counts) in cases {
9319            let got = decode(reader_e.read_slice("btv1", &starts, &counts).unwrap());
9320            assert_eq!(
9321                got,
9322                expect(starts, counts),
9323                "btv1 slice starts={starts:?} counts={counts:?}"
9324            );
9325        }
9326
9327        let _ = std::fs::remove_file(&latest);
9328        let _ = std::fs::remove_file(&earliest);
9329    }
9330
9331    /// A slice read must place the same bytes whichever way a chunk reaches
9332    /// the output: read run by run straight out of the file — what unfiltered
9333    /// data allows, its stored bytes being the dataset's bytes — or copied out
9334    /// of a decoded whole-chunk image, which filtered data always needs and
9335    /// which runs too small to be worth a positioned read each fall back to.
9336    /// Both sinks are driven off one array here, so naive extraction is the
9337    /// shared oracle for the two of them and for every run shape in between.
9338    #[test]
9339    #[cfg(feature = "deflate")]
9340    fn slice_reads_agree_however_the_chunk_reaches_the_output() {
9341        let path = temp_path("slice_run_placement");
9342        let (rows, cols) = (8usize, 4096usize); // a chunk row is 32 KiB
9343        let data: Vec<f64> = (0..rows * cols).map(|i| i as f64).collect();
9344        let (rows, cols) = (rows as u64, cols as u64);
9345        {
9346            let file = crate::H5File::create(&path).unwrap();
9347            let ds = file
9348                .new_dataset::<f64>()
9349                .shape([rows as usize, cols as usize])
9350                .chunk(&[2, cols as usize])
9351                .create("plain")
9352                .unwrap();
9353            ds.write_raw(&data).unwrap();
9354            let ds = file
9355                .new_dataset::<f64>()
9356                .shape([rows as usize, cols as usize])
9357                .chunk(&[2, cols as usize])
9358                .deflate(1)
9359                .create("zipped")
9360                .unwrap();
9361            ds.write_raw(&data).unwrap();
9362            file.close().unwrap();
9363        }
9364
9365        let expect = |starts: [u64; 2], counts: [u64; 2]| -> Vec<f64> {
9366            let mut out = Vec::new();
9367            for i in 0..counts[0] {
9368                for j in 0..counts[1] {
9369                    out.push(data[((starts[0] + i) * cols + starts[1] + j) as usize]);
9370                }
9371            }
9372            out
9373        };
9374        let decode = |raw: Vec<u8>| -> Vec<f64> {
9375            raw.chunks_exact(8)
9376                .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
9377                .collect()
9378        };
9379        let cases: &[([u64; 2], [u64; 2])] = &[
9380            ([0, 0], [rows, cols]), // whole dataset: one run per chunk
9381            ([3, 0], [4, cols]),    // straddles chunk rows, full width
9382            ([1, 7], [5, 3]),       // 24-byte runs: not worth a read each
9383            ([2, 1000], [2, 2048]), // 16 KiB runs, off the chunk row origin
9384            ([7, 4095], [1, 1]),    // one element in the last chunk
9385        ];
9386
9387        let mut reader = Hdf5Reader::open(&path).unwrap();
9388        for name in ["plain", "zipped"] {
9389            let full = decode(reader.read_dataset_raw(name).unwrap());
9390            assert_eq!(full, data, "{name} full read");
9391            for &(starts, counts) in cases {
9392                let got = decode(reader.read_slice(name, &starts, &counts).unwrap());
9393                assert_eq!(
9394                    got,
9395                    expect(starts, counts),
9396                    "{name} slice starts={starts:?} counts={counts:?}"
9397                );
9398            }
9399        }
9400
9401        std::fs::remove_file(&path).ok();
9402    }
9403}