Skip to main content

rust_hdf5/io/
writer.rs

1//! HDF5 file writer.
2//!
3//! Produces a valid HDF5 file with superblock v3, a root group object header,
4//! and datasets with contiguous or chunked storage. The output is readable by `h5dump`.
5
6use std::collections::{HashMap, HashSet};
7use std::path::{Path, PathBuf};
8
9use crate::dataset::DatasetAccess;
10use crate::format::btree_v1::{BTreeV1Config, ChunkBTreeV1Node, ChunkBTreeV1Tree, ChunkKey};
11use crate::format::chunk_index::btree_v2::Bt2ChunkIndex;
12use crate::format::chunk_index::extensible_array::{
13    compute_chunk_size_len, compute_ndblk_addrs, compute_nsblk_addrs, EaDblkPath, EaGeometry,
14    EaLoc, ExtensibleArrayDataBlock, ExtensibleArrayHeader, ExtensibleArrayIndexBlock,
15    ExtensibleArraySuperBlock, FilteredChunkEntry, FilteredDataBlock, FilteredIndexBlock,
16    EA_CLS_CHUNK, EA_CLS_FILT_CHUNK,
17};
18use crate::format::chunk_index::fixed_array::{
19    decode_filtered_page, decode_unfiltered_page, encode_filtered_page, encode_unfiltered_page,
20    FixedArrayDataBlock, FixedArrayFilteredChunkElement, FixedArrayHeader, FixedArrayPagedPrefix,
21    FA_CLIENT_FILT_CHUNK,
22};
23use crate::format::creation_order::CreationOrder;
24use crate::format::dense_attr::build_dense_attributes;
25use crate::format::dense_link::build_dense_links;
26use crate::format::free_space::{
27    self, FreeSection, FreeSpaceClass, FreeSpaceHeader, FreeSpaceManager,
28};
29use crate::format::local_heap::{
30    local_heap_header_size, LocalHeapHeader, LocalHeapImage, LOCAL_HEAP_FREE_NULL,
31};
32use crate::format::messages::attr_info::{next_creation_index, AttributeInfoMessage};
33use crate::format::messages::attribute::{
34    AttributeEntry, AttributeMessage, ATTR_FLAG_SPACE_SHARED, ATTR_FLAG_TYPE_SHARED,
35};
36use crate::format::messages::data_layout::{
37    DataLayoutMessage, EarrayParams, FixedArrayParams, LAYOUT_VERSION_DEFAULT,
38};
39use crate::format::messages::dataspace::{DataspaceClass, DataspaceMessage};
40use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, ReferenceKind};
41use crate::format::messages::external_file_list::{ExternalFileListMessage, UNLIMITED};
42use crate::format::messages::fill_value::{
43    FillValueMessage, FILL_TIME_ALLOC, FILL_TIME_IFSET, FILL_TIME_NEVER,
44};
45use crate::format::messages::filter::{self, FilterPipeline};
46use crate::format::messages::group_info::GroupInfoMessage;
47use crate::format::messages::link::{CharacterSet, LinkMessage, LinkTarget};
48use crate::format::messages::link_info::LinkInfoMessage;
49use crate::format::messages::mod_time::ModificationTime;
50use crate::format::messages::superblock_ext::{
51    FileSpaceInfoMessage, FileSpaceStrategy, SharedMessageTableMessage,
52    DEFAULT_FILE_SPACE_PAGE_SIZE, FS_ADDR_COUNT_V1, PAGE_SIZE_MAX, PAGE_SIZE_MIN,
53};
54use crate::format::messages::virtual_mapping::{
55    parse_source_name, VirtualMapping, VirtualMappingList,
56};
57use crate::format::messages::*;
58use crate::format::object_header::{ObjectHeader, ObjectTimes, MAX_MESSAGE_SIZE};
59use crate::format::reference::{
60    encode_reference_element, encode_revised_blob, ReferenceElementImage, ReferenceTarget,
61    REVISED_BLOB_TOKEN_OFFSET,
62};
63use crate::format::selection::Selection;
64use crate::format::sohm::{
65    type_flag, SharedMessagePointer, MAX_SOHM_INDEXES, SOHM_HEAP_ID_LEN, SOHM_POINTER_HEAP_ID_AT,
66};
67use crate::format::sohm_write::{
68    build_shared_messages, NestedShare, SharedMessage, SohmIndexContent, SohmIndexSpec,
69};
70use crate::format::superblock::*;
71use crate::format::{FormatContext, LibverBound, ObjectFormat, UNDEF_ADDR};
72
73use crate::format::selection::check_hyperslab;
74use crate::io::allocator::{FileAllocator, FreeBlock};
75use crate::io::file_handle::FileHandle;
76use crate::io::hyperslab::{for_each_contiguous_run, for_each_dual_run};
77use crate::io::symbol_table_io::{free_stab, write_stab, Stab, StabExtents, StabLink, StabTarget};
78use crate::io::{FileMeta, IoResult};
79
80/// On-disk size in bytes of a fixed-array data block, for the layout (paged or
81/// flat) implied by `hdr`.
82///
83/// Mirrors `H5FA_DBLOCK_SIZE` (`H5FApkg.h`):
84///   - non-paged: `prefix + nelmts * raw_elmt_size + checksum`
85///   - paged: `prefix + page_init_bitmap + nelmts * raw_elmt_size
86///     + npages * checksum`, where the prefix checksum covers the bitmap.
87///
88/// `raw_elmt_size` is `sizeof_addr` for an unfiltered array, and
89/// `sizeof_addr + chunk_size_len + 4` (the filtered element: address +
90/// compressed size + filter mask) for a filtered array. libhdf5 carries this
91/// value as `hdr->cparam.raw_elmt_size`, i.e. exactly `hdr.element_size`.
92fn fixed_array_dblk_disk_size(ctx: &FormatContext, hdr: &FixedArrayHeader) -> u64 {
93    let elem_size = hdr.element_size as u64;
94    let sa = ctx.sizeof_addr as u64;
95    let nelmts = hdr.num_elmts;
96    // Common metadata prefix: signature(4) + version(1) + client_id(1) + header_addr(sa).
97    let meta_prefix = 4 + 1 + 1 + sa;
98    if hdr.is_paged() {
99        let npages = hdr.npages();
100        let bitmap_size = npages.div_ceil(8);
101        // prefix (incl. its own 4-byte checksum) + elements + per-page checksums.
102        (meta_prefix + bitmap_size + 4) + nelmts * elem_size + npages * 4
103    } else {
104        // prefix + elements + single 4-byte checksum.
105        meta_prefix + nelmts * elem_size + 4
106    }
107}
108
109/// A walk of a v2 B-tree: the file and node geometry the descent reads
110/// through, and the two collections it fills — every node's raw record
111/// bytes and every node block's address, the latter because `open_append`
112/// needs it so the reconstructed [`Bt2DatasetInfo::node_addrs`] pool owns
113/// the on-disk nodes (the next flush re-serializes the tree over them, and
114/// a delete frees them).
115///
116/// `record_size`, `node_size` and `geo` are constant for the whole walk, so
117/// [`descend`](Self::descend) takes only what changes per level: the node's
118/// address, its depth, and how many records it holds.
119struct Bt2Walk<'a> {
120    handle: &'a FileHandle,
121    ctx: &'a FormatContext,
122    record_size: u16,
123    node_size: u32,
124    geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
125    records: Vec<u8>,
126    node_addrs: Vec<u64>,
127}
128
129impl<'a> Bt2Walk<'a> {
130    fn new(
131        handle: &'a FileHandle,
132        ctx: &'a FormatContext,
133        record_size: u16,
134        node_size: u32,
135        geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
136    ) -> Self {
137        Self {
138            handle,
139            ctx,
140            record_size,
141            node_size,
142            geo,
143            records: Vec::new(),
144            node_addrs: Vec::new(),
145        }
146    }
147
148    /// Walk the subtree rooted at `addr`, at depth `depth` with `nrec`
149    /// records, collecting every node's raw record bytes and every node
150    /// block's address.
151    fn descend(&mut self, addr: u64, depth: u16, nrec: u16) -> IoResult<()> {
152        use crate::format::chunk_index::btree_v2::{Bt2InternalNode, Bt2LeafNode};
153
154        self.node_addrs.push(addr);
155        let buf = self.handle.read_at_most(addr, self.node_size as usize)?;
156        if depth == 0 {
157            let leaf = Bt2LeafNode::decode(&buf, nrec, self.record_size)?;
158            self.records.extend_from_slice(&leaf.record_data);
159        } else {
160            let node = Bt2InternalNode::decode(
161                &buf,
162                self.ctx,
163                depth,
164                nrec,
165                self.record_size,
166                self.geo.max_nrec_size,
167                self.geo.child_total_size(depth),
168            )?;
169            // In-order: an internal node's records separate its children, so each
170            // one belongs between the subtrees on either side of it.
171            let children: Vec<(u64, u16)> = node
172                .child_addrs
173                .iter()
174                .zip(node.child_nrecords.iter())
175                .map(|(&a, &n)| (a, n))
176                .collect();
177            let rec = self.record_size as usize;
178            for (i, (child_addr, child_nrec)) in children.into_iter().enumerate() {
179                self.descend(child_addr, depth - 1, child_nrec)?;
180                if let Some(record) = node.record_data.get(i * rec..(i + 1) * rec) {
181                    self.records.extend_from_slice(record);
182                }
183            }
184        }
185        Ok(())
186    }
187}
188
189/// A walk of a version-1 raw-data-chunk B-tree: the file and geometry the
190/// descent reads through, and the two collections it fills.
191///
192/// The v1 counterpart of [`Bt2Walk`], and for the same reason: the
193/// records are what [`BtreeV1DatasetInfo::build_tree`] bulk-loads on the next
194/// flush, and the addresses are the block pool that flush re-serializes over,
195/// so a reopened tree owns the nodes it found instead of leaking them and
196/// allocating a second set beside them.
197///
198/// One value rather than nine parameters threaded through the recursion: only
199/// `addr` and `depth` change between one level and the next, so they are what
200/// [`descend`](Self::descend) takes and everything else lives here.
201struct BtreeV1Walk<'a> {
202    handle: &'a FileHandle,
203    ctx: &'a FormatContext,
204    config: &'a BTreeV1Config,
205    /// The chunk edge lengths, *without* the trailing element-size dimension,
206    /// so `chunk_dims.len()` is the rank the node keys are decoded at.
207    chunk_dims: &'a [u64],
208    file_size: u64,
209    records: Vec<BtreeV1ChunkRecord>,
210    node_addrs: Vec<u64>,
211}
212
213impl<'a> BtreeV1Walk<'a> {
214    fn new(
215        handle: &'a FileHandle,
216        ctx: &'a FormatContext,
217        config: &'a BTreeV1Config,
218        chunk_dims: &'a [u64],
219        file_size: u64,
220    ) -> Self {
221        Self {
222            handle,
223            ctx,
224            config,
225            chunk_dims,
226            file_size,
227            records: Vec::new(),
228            node_addrs: Vec::new(),
229        }
230    }
231
232    /// Walk the subtree rooted at `addr`, collecting every leaf entry as a
233    /// [`BtreeV1ChunkRecord`] and every node block's address.
234    ///
235    /// Records come out in key order because a v1 B-tree's leaves are in key
236    /// order and this descends left to right, which is what
237    /// [`BtreeV1DatasetInfo::position`]'s binary search needs. The keys store
238    /// element offsets (`scaled * chunk_dim`, `H5D__btree_encode_key`), so the
239    /// grid position this records is the quotient.
240    fn descend(&mut self, addr: u64, depth: u32) -> IoResult<()> {
241        // The same bound the reader's walk uses: a node's level is one byte, so
242        // no honest tree is deeper than that, and a cyclic index stops here.
243        if depth > 256 {
244            return Err(crate::io::IoError::InvalidState(
245                "chunk B-tree v1 exceeds maximum depth".into(),
246            ));
247        }
248        if addr == UNDEF_ADDR || addr >= self.file_size {
249            return Ok(());
250        }
251        let rank = self.chunk_dims.len();
252        let sa = self.ctx.sizeof_addr as usize;
253        let node_size = self.config.chunk_btree_node_size(sa, rank);
254        let buf = self.handle.read_at_most(addr, node_size)?;
255        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.config.chunk_max_entries())?;
256        self.node_addrs.push(addr);
257
258        if node.level == 0 {
259            for (i, &child_addr) in node.children.iter().enumerate() {
260                let key = &node.keys[i];
261                let scaled: Vec<u64> = key.offsets[..rank]
262                    .iter()
263                    .zip(self.chunk_dims)
264                    .map(|(&offset, &dim)| offset.checked_div(dim).unwrap_or(0))
265                    .collect();
266                self.records.push(BtreeV1ChunkRecord {
267                    scaled,
268                    address: child_addr,
269                    nbytes: key.chunk_size,
270                    filter_mask: key.filter_mask,
271                });
272            }
273        } else {
274            for &child_addr in &node.children {
275                self.descend(child_addr, depth + 1)?;
276            }
277        }
278        Ok(())
279    }
280}
281
282/// Encode a fixed-array data block for the layout implied by `hdr`, using the
283/// chunk addresses held in `dblk.elements` (unfiltered) or the filtered chunk
284/// entries in `dblk.filtered_elements` (filtered, `client_id == 1`).
285///
286/// For the paged layout (`hdr.is_paged()`), emits the `FADB` prefix with a
287/// page-init bitmap followed by `npages` checksummed element pages. A page is
288/// marked initialized iff at least one of its chunk addresses is defined,
289/// mirroring libhdf5's lazy `H5FA__dblk_page_create`. Uninitialized pages are
290/// still written (all `UNDEF_ADDR`, valid checksum) so the file contains no
291/// uninitialized bytes; the reader skips them via the bitmap.
292fn encode_fixed_array_dblk(
293    ctx: &FormatContext,
294    hdr: &FixedArrayHeader,
295    dblk: &FixedArrayDataBlock,
296) -> Vec<u8> {
297    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
298    let sa = ctx.sizeof_addr as usize;
299    // chunk_size_len for filtered entries = element_size - sizeof_addr - 4.
300    // libhdf5 carries element_size = sizeof_addr + chunk_size_len + 4.
301    let chunk_size_len = (hdr.element_size as usize).saturating_sub(sa + 4);
302
303    if !hdr.is_paged() {
304        return if is_filtered {
305            dblk.encode_filtered(ctx, chunk_size_len)
306        } else {
307            dblk.encode_unfiltered(ctx)
308        };
309    }
310
311    let npages = hdr.npages() as usize;
312    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
313
314    // Build the page-init bitmap (MSB-first): a page is initialized iff any of
315    // its elements points at a defined address.
316    let mut bitmap = vec![0u8; npages.div_ceil(8)];
317    let nelmts = if is_filtered {
318        dblk.filtered_elements.len()
319    } else {
320        dblk.elements.len()
321    };
322    for p in 0..npages {
323        let start = p * dblk_page_nelmts;
324        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
325        let initialized = if is_filtered {
326            dblk.filtered_elements[start..end]
327                .iter()
328                .any(|e| e.address != UNDEF_ADDR)
329        } else {
330            dblk.elements[start..end].iter().any(|&a| a != UNDEF_ADDR)
331        };
332        if initialized {
333            bitmap[p / 8] |= 0x80u8 >> (p % 8);
334        }
335    }
336
337    let prefix = FixedArrayPagedPrefix {
338        client_id: hdr.client_id,
339        header_addr: dblk.header_addr,
340        page_init_bitmap: bitmap,
341        prefix_size: 4 + 1 + 1 + sa + npages.div_ceil(8) + 4,
342    };
343
344    let mut buf = prefix.encode(ctx);
345    debug_assert_eq!(buf.len(), prefix.prefix_size);
346
347    // Append each page: all pages use the full `dblk_page_nelmts` stride;
348    // only the last page holds fewer elements (libhdf5 H5FA.c).
349    for p in 0..npages {
350        let start = p * dblk_page_nelmts;
351        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
352        if is_filtered {
353            buf.extend_from_slice(&encode_filtered_page(
354                &dblk.filtered_elements[start..end],
355                ctx,
356                chunk_size_len,
357            ));
358        } else {
359            buf.extend_from_slice(&encode_unfiltered_page(&dblk.elements[start..end], ctx));
360        }
361    }
362    buf
363}
364
365/// Decode a fixed-array data block for the layout implied by `hdr` — the
366/// inverse of [`encode_fixed_array_dblk`], and the single decode dispatch
367/// over non-paged/paged × unfiltered/filtered.
368///
369/// For the paged layout, pages whose bitmap bit is clear are skipped, not
370/// decoded: libhdf5 never writes an uninitialized page, so its bytes are
371/// arbitrary and carry no valid checksum. Their elements stay at the
372/// undefined-address defaults, which is exactly what the bitmap means.
373fn decode_fixed_array_dblk(
374    ctx: &FormatContext,
375    hdr: &FixedArrayHeader,
376    buf: &[u8],
377    chunk_size_len: usize,
378) -> crate::format::FormatResult<FixedArrayDataBlock> {
379    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
380    let num_elmts = hdr.num_elmts as usize;
381
382    if !hdr.is_paged() {
383        return if is_filtered {
384            FixedArrayDataBlock::decode_filtered(buf, ctx, num_elmts, chunk_size_len)
385        } else {
386            FixedArrayDataBlock::decode_unfiltered(buf, ctx, num_elmts)
387        };
388    }
389
390    let npages = hdr.npages() as usize;
391    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
392    let prefix = FixedArrayPagedPrefix::decode(buf, ctx, npages as u64)?;
393
394    let mut dblk = if is_filtered {
395        FixedArrayDataBlock::new_filtered(prefix.header_addr, num_elmts)
396    } else {
397        FixedArrayDataBlock::new_unfiltered(prefix.header_addr, num_elmts)
398    };
399    dblk.client_id = hdr.client_id;
400
401    // Pages follow the prefix back to back; every page spans the full
402    // `dblk_page_nelmts` stride except the last, which holds the remainder.
403    let mut pos = prefix.prefix_size;
404    for p in 0..npages {
405        let start = p * dblk_page_nelmts;
406        let end = ((p + 1) * dblk_page_nelmts).min(num_elmts);
407        let nelmts = end - start;
408        if prefix.page_initialized(p) {
409            let page_buf = buf.get(pos..).unwrap_or(&[]);
410            if is_filtered {
411                let elems = decode_filtered_page(page_buf, ctx, nelmts, chunk_size_len)?;
412                dblk.filtered_elements[start..end].clone_from_slice(&elems);
413            } else {
414                let addrs = decode_unfiltered_page(page_buf, ctx, nelmts)?;
415                dblk.elements[start..end].copy_from_slice(&addrs);
416            }
417        }
418        pos += nelmts * hdr.element_size as usize + 4;
419    }
420    Ok(dblk)
421}
422
423/// Interior-mutability cell for per-dataset write state, selected by feature.
424///
425/// This is the §5-B "cfg-selected interior types" from
426/// `docs/threadsafe-fine-grained-locking.md`: the single-threaded build uses a
427/// `RefCell` (zero overhead, no atomics), while the `threadsafe` build uses a
428/// `Mutex` so two threads can write *different* datasets concurrently while the
429/// same dataset's writes serialize. Call sites are identical across both via
430/// [`Slot::lock`].
431#[cfg(not(feature = "threadsafe"))]
432pub(crate) struct Slot<T>(std::cell::RefCell<T>);
433
434#[cfg(not(feature = "threadsafe"))]
435impl<T> Slot<T> {
436    pub(crate) fn new(value: T) -> Self {
437        Slot(std::cell::RefCell::new(value))
438    }
439    /// Borrow the contents mutably (an uncontended `RefCell` borrow).
440    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, T> {
441        self.0.borrow_mut()
442    }
443}
444
445#[cfg(feature = "threadsafe")]
446pub(crate) struct Slot<T>(std::sync::Mutex<T>);
447
448#[cfg(feature = "threadsafe")]
449impl<T> Slot<T> {
450    pub(crate) fn new(value: T) -> Self {
451        Slot(std::sync::Mutex::new(value))
452    }
453    /// Lock the contents. Different datasets hold different slots, so this
454    /// only contends when two threads write the *same* dataset.
455    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, T> {
456        self.0.lock().unwrap()
457    }
458}
459
460/// Proof that the create gate (`create_lock`) is held and the new dataset's
461/// name passed the uniqueness check. Only [`Hdf5Writer::begin_create`]
462/// constructs one and [`Hdf5Writer::push_dataset`] demands one, so a creator
463/// cannot reach the dataset registry while skipping either step. Carries
464/// the canonical (link-resolved) name the creator must store, so the
465/// registry only ever holds tree paths.
466pub(crate) struct CreateGuard<'a> {
467    #[cfg(not(feature = "threadsafe"))]
468    _gate: std::cell::RefMut<'a, ()>,
469    #[cfg(feature = "threadsafe")]
470    _gate: std::sync::MutexGuard<'a, ()>,
471    /// The dataset name with every group hard link in it resolved.
472    pub(crate) name: String,
473    /// The group that will hold the new dataset's link, resolved from the
474    /// path components of `name`; `None` is the root group. Carried here so
475    /// [`Hdf5Writer::push_dataset`] registers the child itself and no creator
476    /// can leave a dataset whose name says one thing and whose parent group
477    /// says another.
478    pub(crate) parent: Option<usize>,
479}
480
481/// Reference-counted shared pointer, feature-selected. The single-thread
482/// build uses `Rc` (no atomics); the `threadsafe` build uses `Arc` so a
483/// dataset/group slot can be cloned out of the registry and locked on its
484/// own — letting writes to *different* datasets proceed concurrently without
485/// holding the registry lock. See `docs/threadsafe-fine-grained-locking.md`
486/// (Stage 3).
487#[cfg(not(feature = "threadsafe"))]
488pub(crate) type Shared<T> = std::rc::Rc<T>;
489#[cfg(feature = "threadsafe")]
490pub(crate) type Shared<T> = std::sync::Arc<T>;
491
492/// One dataset's cell in the registry: its metadata slot plus the operation
493/// lock that serializes whole logical operations on it. Both live in one
494/// allocation so they cannot fall out of step — every dataset has its op
495/// lock by construction.
496pub(crate) struct DatasetCell {
497    /// Serializes one *whole* logical operation on this dataset.
498    ///
499    /// The metadata slot below serializes each individual acquisition, but a
500    /// multi-acquisition operation — take the append buffer → write chunks →
501    /// re-buffer the tail → extend, or flush-then-overwrite in a slice write
502    /// — would interleave with a concurrent same-dataset operation *between*
503    /// its acquisitions under `threadsafe`. Public write entries take this
504    /// lock and delegate to `_inner` variants; `_inner` variants and the
505    /// `pub(crate)` write helpers require the caller to hold it (or to hold
506    /// the writer exclusively via `&mut`, as close and the SWMR wrapper do).
507    ///
508    /// Not reentrant: the single-thread build's `RefCell` panics instantly
509    /// on a nested acquisition, so a missed entry/inner split fails loudly
510    /// in every test run rather than deadlocking only under `threadsafe`.
511    ///
512    /// Lock order: `create_lock → op → registry spine → metadata slot`. An
513    /// op lock is never held across another dataset's op lock, and no
514    /// op-lock holder takes `create_lock`, so the order is acyclic.
515    pub(crate) op: Slot<()>,
516    info: Slot<DatasetInfo>,
517}
518
519impl DatasetCell {
520    pub(crate) fn new(info: DatasetInfo) -> Self {
521        DatasetCell {
522            op: Slot::new(()),
523            info: Slot::new(info),
524        }
525    }
526
527    /// Borrow the metadata slot (a single acquisition; see [`Self::op`] for
528    /// whole-operation serialization).
529    #[cfg(not(feature = "threadsafe"))]
530    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, DatasetInfo> {
531        self.info.lock()
532    }
533
534    /// Lock the metadata slot (a single acquisition; see [`Self::op`] for
535    /// whole-operation serialization).
536    #[cfg(feature = "threadsafe")]
537    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, DatasetInfo> {
538        self.info.lock()
539    }
540}
541
542/// A single dataset's [`DatasetCell`], reference-counted so a writer can
543/// clone it out of the registry (releasing the registry lock) and then lock
544/// just this one dataset. Two threads writing different datasets take
545/// different `DatasetRef` locks and never contend; the same dataset's writes
546/// serialize, which is required because one chunk index is not concurrently
547/// mutable.
548pub(crate) type DatasetRef = Shared<DatasetCell>;
549
550/// A single group's metadata behind its own [`Slot`], reference-counted like
551/// [`DatasetRef`].
552pub(crate) type GroupRef = Shared<Slot<GroupInfo>>;
553
554/// Appended frames held back until they complete a chunk.
555///
556/// The buffer is the sole authority for rows `base .. base + frames`: the
557/// file's chunks do not hold them yet, and any operation that writes those
558/// rows must go through [`Hdf5Writer::flush_append_buffer`] first. `base` is
559/// recorded when the frames are buffered — never derived from the current
560/// extent, which an `extend_dataset` can move independently.
561pub struct AppendBuffer {
562    /// Absolute row of the first buffered frame.
563    pub base: u64,
564    /// Number of buffered frames.
565    pub frames: u64,
566    /// The frames' bytes, `frames` whole rows, row-major.
567    pub bytes: Vec<u8>,
568}
569
570/// One file a dataset's raw data lives in, as the writer holds it: the name
571/// the I/O path opens, together with the local-heap offset the External File
572/// List message stores that name as.
573///
574/// The two halves are one entry rather than two parallel lists because they
575/// describe one slot — the message encodes `name_offset`, and every read or
576/// write of the slot's bytes opens `name`; splitting them is what lets a
577/// rewrite pair a name with another slot's offset.
578#[derive(Debug, Clone, PartialEq, Eq)]
579pub struct ExternalFile {
580    /// The file name exactly as the heap stores it. Resolved against
581    /// `HDF5_EXTFILE_PREFIX` at I/O time, never here — the same rule the read
582    /// side follows.
583    pub name: String,
584    /// Where `name` sits in the local heap at [`ExternalStorage::heap_addr`].
585    pub name_offset: u64,
586    /// Byte offset within `name` where this slot's region begins.
587    pub offset: u64,
588    /// Bytes of the dataset's raw data this slot holds.
589    pub size: u64,
590}
591
592/// A dataset whose contiguous raw data lives outside this file — the External
593/// File List message (`H5O_EFL_ID`) and the local heap its names are in.
594///
595/// The data layout message of such a dataset still says `Contiguous`, with
596/// its address left undefined: it is this message's presence that makes
597/// libhdf5 route the dataset's I/O through `H5D_LOPS_EFL` (H5Dlayout.c).
598#[derive(Debug, Clone)]
599pub struct ExternalStorage {
600    /// Address of the local heap header holding every slot's name.
601    pub heap_addr: u64,
602    /// The files, in the order their regions concatenate into the dataset's
603    /// logical byte range.
604    pub files: Vec<ExternalFile>,
605    /// The prefix every one of those names is joined against, and the open
606    /// that settled it. Lives here rather than on [`DatasetInfo`] so a
607    /// dataset with no external storage cannot carry a prefix and a dataset
608    /// with external storage cannot lack one.
609    prefix: EfilePrefix,
610}
611
612/// The expanded external file prefix in force for one dataset, and the open
613/// that decided it — libhdf5's `dset->shared->extfile_prefix`.
614///
615/// `H5D__build_file_prefix` runs it once per open of the shared info, from
616/// the dapl of `H5D__create` (H5Dint.c:1318) or of the `H5D__open` that
617/// found no shared info yet (:1537), and both `H5D__efl_read` and
618/// `H5D__efl_write` then join against that one answer (H5Defl.c:315-317,
619/// :429-431). Measured under libhdf5 1.14.6 and 2.0.0: `H5Dcreate2` with a
620/// dapl naming a directory creates the raw data file there at `H5Dwrite`,
621/// and `HDF5_EXTFILE_PREFIX` shadows that property on the write path exactly
622/// as it does on the read path.
623#[derive(Debug, Clone, Default)]
624struct EfilePrefix {
625    /// The expansion itself; `None` is "no prefix", which leaves a stored
626    /// name to resolve against the process's current directory.
627    expanded: Option<PathBuf>,
628    /// The open that decided [`expanded`](Self::expanded). An expired handle
629    /// means no open is holding the answer any more, so the next one settles
630    /// it afresh — which is the state a dataset this session reopened starts
631    /// in, `H5Fopen` opening no dataset of its own.
632    open: std::sync::Weak<()>,
633}
634
635impl ExternalStorage {
636    /// The message this storage encodes to (`H5O_efl_t`).
637    fn message(&self) -> ExternalFileListMessage {
638        ExternalFileListMessage {
639            heap_addr: self.heap_addr,
640            slots: self
641                .files
642                .iter()
643                .map(
644                    |f| crate::format::messages::external_file_list::ExternalFileSlot {
645                        name_offset: f.name_offset,
646                        offset: f.offset,
647                        size: f.size,
648                    },
649                )
650                .collect(),
651        }
652    }
653
654    /// Bytes the slots reserve in total (`H5O_efl_total_size`), saturating
655    /// rather than wrapping so an overflowing list reads as "as large as it
656    /// gets" and passes any size check instead of failing one.
657    fn total_size(&self) -> u64 {
658        self.files
659            .iter()
660            .fold(0u64, |acc, f| acc.saturating_add(f.size))
661    }
662}
663
664/// A dataset whose elements are read out of other datasets — the virtual
665/// layout message (`H5D_VIRTUAL`) and the mapping list it points at.
666///
667/// The mappings live in one global heap object rather than in the header
668/// (`H5D__virtual_store_layout`), so the layout message carries only its
669/// address and index; the list itself is kept here so a rewrite of the header
670/// can re-emit the message pointing at the same object.
671#[derive(Debug, Clone, PartialEq, Eq)]
672pub struct VirtualStorage {
673    /// Address of the global heap collection holding the mapping list.
674    pub heap_addr: u64,
675    /// Index of the mapping-list object within that collection.
676    pub heap_index: u32,
677    /// The mappings themselves, in the order they were declared — which is
678    /// the order libhdf5 resolves overlapping ones in.
679    pub mappings: Vec<VirtualMapping>,
680}
681
682/// Where a contiguous dataset's raw bytes live, read off its registry entry
683/// so the write itself can run with the slot unlocked.
684///
685/// The one place the local-versus-external-versus-nowhere choice is made; see
686/// [`DatasetInfo::contiguous_target`].
687enum ContiguousTarget {
688    /// A block in this file, starting at this address.
689    Local(u64),
690    /// The files an External File List names, in dataset order, and the
691    /// prefix in force for the open doing the writing — carried together
692    /// because a slot name means nothing without it.
693    External {
694        files: Vec<ExternalFile>,
695        prefix: Option<PathBuf>,
696    },
697    /// Nowhere: the dataset is virtual, and every element of it is stored in
698    /// whichever source dataset its mappings send that element to.
699    Virtual,
700}
701
702/// What a writer-mode `H5Dataset` handle is built from — the shape and
703/// element width it answers questions with, the chunk index it writes
704/// through, and the open it holds.
705pub(crate) struct DatasetHandleParts {
706    pub(crate) shape: Vec<usize>,
707    pub(crate) element_size: usize,
708    /// `None` for storage that is not chunked.
709    pub(crate) chunk_index: Option<ChunkIndexKind>,
710    /// Keeps this open alive; see [`Hdf5Writer::bind_efile_prefix`].
711    pub(crate) open: Option<crate::io::reader::DatasetOpenToken>,
712}
713
714impl ContiguousTarget {
715    /// Whether this target is storage bytes can be written into at all —
716    /// false only for [`ContiguousTarget::Virtual`], which names sources
717    /// rather than storage.
718    fn is_storage(&self) -> bool {
719        !matches!(self, Self::Virtual)
720    }
721}
722
723/// The one refusal of a write into a virtual dataset, so the two paths that
724/// can reach one — [`Hdf5Writer::write_contiguous_bytes`] and the pre-insert
725/// gate of [`Hdf5Writer::write_vlen_strings_slice`] — say the same thing.
726///
727/// libhdf5 does take this write, pushing each element through the mapping
728/// that covers it into the source dataset holding it (`H5D__virtual_write`);
729/// this writer never opens a source file, so it refuses rather than dropping
730/// the bytes somewhere they cannot be read back from.
731/// The legality checks `H5Pset_virtual` runs over one mapping —
732/// `H5D_virtual_check_mapping_pre` and `H5D_virtual_check_mapping_post`
733/// (H5Dvirtual.c).
734///
735/// The two upstream checks that need the *source dataset's* own extent (the
736/// limited/limited element-count match, and a printf mapping's single-block
737/// match) are not run here for the same reason upstream skips them when the
738/// source space status is `H5O_VIRTUAL_STATUS_INVALID`: a mapping may name a
739/// source that does not exist yet, and nothing here opens one.
740fn check_virtual_mapping(dataset: &str, m: &VirtualMapping) -> IoResult<()> {
741    for (which, sel) in [
742        ("virtual", &m.virtual_selection),
743        ("source", &m.source_selection),
744    ] {
745        if matches!(sel, Selection::Points(_)) {
746            return Err(crate::io::IoError::Unsupported(format!(
747                "virtual dataset '{dataset}' has a point {which} selection, which \
748                 H5D_virtual_check_mapping_pre refuses for every virtual dataset mapping \
749                 (\"point selections not currently supported with virtual datasets\")"
750            )));
751        }
752    }
753
754    let unlim_virtual = m.virtual_selection.unlim_dim().is_some();
755    let unlim_source = m.source_selection.unlim_dim().is_some();
756
757    // Both sides unbounded: the mapping grows with its source, so the slices
758    // they exchange must be the same shape whatever either extent becomes.
759    if unlim_virtual && unlim_source {
760        if let (Some(v), Some(sr)) = (
761            regular_hyperslab(&m.virtual_selection),
762            regular_hyperslab(&m.source_selection),
763        ) {
764            let (nv, ns) = (v.num_elem_non_unlim(), sr.num_elem_non_unlim());
765            if nv != ns {
766                return Err(crate::io::IoError::InvalidState(format!(
767                    "virtual dataset '{dataset}' maps an unlimited source selection onto an \
768                     unlimited virtual selection, but a slice of the non-unlimited \
769                     dimensions holds {ns:?} source elements and {nv:?} virtual ones"
770                )));
771            }
772        }
773    }
774
775    // `H5D_virtual_check_mapping_post`: an unlimited virtual selection over a
776    // limited source selection is the printf shape, where each block of the
777    // virtual selection is filled by a *different* source dataset named by
778    // substituting that block's index. It needs a `%b` to name them, and a
779    // hyperslab virtual selection to have blocks at all; every other shape
780    // needs the opposite, since a substitution with only one block to fill
781    // has nothing to vary over.
782    let nsubs = parse_source_name(&m.source_file_name)
783        .and_then(|f| Ok(f.nsubs() + parse_source_name(&m.source_dset_name)?.nsubs()))
784        .map_err(|e| {
785            crate::io::IoError::InvalidState(format!(
786                "virtual dataset '{dataset}' source name: {e}"
787            ))
788        })?;
789    if unlim_virtual && !unlim_source {
790        if nsubs == 0 {
791            return Err(crate::io::IoError::InvalidState(format!(
792                "virtual dataset '{dataset}' has an unlimited virtual selection, a limited \
793                 source selection, and no printf specifiers in source names"
794            )));
795        }
796        if !matches!(m.virtual_selection, Selection::Hyperslab { .. }) {
797            return Err(crate::io::IoError::InvalidState(format!(
798                "virtual dataset '{dataset}' has a printf mapping whose virtual selection is \
799                 not a hyperslab; the substitution runs over the blocks of that hyperslab"
800            )));
801        }
802    } else if nsubs > 0 {
803        return Err(crate::io::IoError::InvalidState(format!(
804            "virtual dataset '{dataset}' has printf specifier(s) in source name(s) without \
805             an unlimited virtual selection and limited source selection"
806        )));
807    }
808    Ok(())
809}
810
811/// The regular (start, stride, count, block) form behind a selection, or
812/// `None` — the only form that can carry `H5S_UNLIMITED`, so every unlimited
813/// check goes through it.
814fn regular_hyperslab(sel: &Selection) -> Option<&crate::format::selection::RegularHyperslab> {
815    match sel {
816        Selection::Hyperslab {
817            form: crate::format::selection::Hyperslab::Regular(r),
818            ..
819        } => Some(r),
820        _ => None,
821    }
822}
823
824fn virtual_write_refused() -> crate::io::IoError {
825    crate::io::IoError::Unsupported(
826        "cannot write into a virtual dataset: its elements live in the source datasets \
827         its mappings name, and this writer does not write through to them — write the \
828         source datasets themselves"
829            .into(),
830    )
831}
832
833/// Metadata for a dataset being written.
834///
835/// The whole struct lives behind a per-dataset [`Slot`] (via [`DatasetRef`]).
836/// The streaming write path locks it only briefly — compression runs *outside*
837/// the lock — so writes to different datasets do not contend, and a structural
838/// op (create/delete) that scans names only momentarily touches a sibling
839/// slot.
840pub struct DatasetInfo {
841    /// Link name within the root group.
842    pub name: String,
843    /// Element datatype.
844    pub datatype: DatatypeMessage,
845    /// The committed datatype this dataset shares, when it was created from
846    /// one. The type itself stays in [`datatype`](Self::datatype) — the
847    /// dataspace, the element width and every payload check need it — and
848    /// this says the header must store a pointer to that object instead of a
849    /// datatype message of its own.
850    pub committed_type: Option<CommittedTypeRef>,
851    /// Dataspace (dimensionality).
852    pub dataspace: DataspaceMessage,
853    /// The object format the reopen found this dataset's messages written in,
854    /// `None` for a dataset this session created.
855    ///
856    /// A rewrite re-encodes the whole header — the shared-message table is
857    /// laid out whole, so every heap ID moves and every header naming one has
858    /// to be written again. Re-deriving the message format from the reopened
859    /// session's bounds would upgrade messages the file already has, which
860    /// libhdf5 never does: it grows a header in place and leaves every
861    /// message it did not touch alone. The same rule the reopen already
862    /// applies to a group it found in a symbol table
863    /// ([`uses_symbol_table`](Hdf5Writer::uses_symbol_table)) — what the file
864    /// says governs, not what this session's bound would have chosen.
865    pub read_format: Option<ObjectFormat>,
866    /// File offset of the dataset's object header (set during finalize).
867    pub obj_header_addr: u64,
868    /// File offset of the raw data block (contiguous only).
869    pub data_addr: u64,
870    /// Size of the raw data in bytes (contiguous only).
871    pub data_size: u64,
872    /// The raw data itself, for a compact dataset — the whole image, which
873    /// [`build_dataset_header`](Hdf5Writer::build_dataset_header) puts inside
874    /// the data layout message rather than in a block of its own. `Some` is
875    /// what makes a dataset compact, and the buffer is created at its final
876    /// length (filled, as `H5D__compact_fill` does, before any write), so it
877    /// is also the dataset's byte count; `data_addr`/`data_size` stay at the
878    /// "no block in the file" values a compact dataset shares with a NULL one.
879    pub compact: Option<Vec<u8>>,
880    /// The files this dataset's contiguous raw data lives in, when it lives
881    /// outside this HDF5 file. `Some` is what makes a contiguous dataset
882    /// externally stored: its `data_addr` stays [`UNDEF_ADDR`] and every byte
883    /// goes to the files named here instead of to a block of this file's own.
884    pub external: Option<ExternalStorage>,
885    /// The source datasets this dataset's elements are read from, when it is
886    /// virtual. `Some` is what makes it virtual, and it stores nothing of its
887    /// own: `data_addr`/`data_size` keep the "no block in this file" values a
888    /// compact dataset also has.
889    pub virtual_storage: Option<VirtualStorage>,
890    /// Chunked storage info (None for contiguous).
891    pub chunked: Option<ChunkedDatasetInfo>,
892    /// Fixed array chunked storage info.
893    pub fixed_array: Option<FixedArrayDatasetInfo>,
894    /// B-tree v2 chunked storage info.
895    pub btree_v2: Option<Bt2DatasetInfo>,
896    /// Implicit (no structure) chunked storage info.
897    pub implicit: Option<ImplicitDatasetInfo>,
898    /// Single-chunk chunked storage info: the whole (fixed) dataspace is
899    /// exactly one chunk.
900    pub single_chunk: Option<SingleChunkDatasetInfo>,
901    /// Version-1 B-tree chunked storage info — the classic-format index.
902    pub btree_v1: Option<BtreeV1DatasetInfo>,
903    /// Appended frames not yet written to chunks, `None` when empty.
904    pub append: Option<AppendBuffer>,
905    /// Attributes attached to this dataset.
906    pub attributes: Vec<AttributeEntry>,
907    /// File offset where the dataset object header was written (for SWMR in-place rewrites).
908    pub obj_header_written_addr: Option<u64>,
909    /// Encoded size of the dataset object header (for verifying in-place rewrites fit).
910    /// Every block the object's on-disk header occupies, chunk 0 first, or
911    /// empty when it has none yet. A rewrite keeps chunk 0's block — its
912    /// address is what every reference to the object holds — and frees the
913    /// rest, so a continuation block left behind is space no free-space
914    /// manager records.
915    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
916    /// Filter pipeline for compressed chunks.
917    pub filter_pipeline: Option<FilterPipeline>,
918    /// Soft-deleted: excluded from finalize output.
919    pub deleted: bool,
920    /// The dataspace extent changed this session (`extend_dataset` /
921    /// `set_dataset_extent`). On a reopened dataset the finalize gate
922    /// otherwise infers "modified" from `chunks_written` alone, and a
923    /// session that only changed the extent would keep the old on-disk
924    /// header — silently dropping the new shape.
925    pub extent_dirty: bool,
926    /// Something the object header encodes changed this session without
927    /// touching the dataset's storage — an attribute set or removed, a fill
928    /// value defined. See [`header_stale`](DatasetInfo::header_stale).
929    pub header_dirty: bool,
930    /// The hard link count the on-disk header was written with, so finalize
931    /// can tell that this session changed it.
932    ///
933    /// A count, not a flag, because the count is what the header records and
934    /// the ways to change it are many: creating a link, unlinking one,
935    /// deleting a link's parent group, promoting a link to a primary name.
936    /// Comparing the value closes all of them at once, where a dirty flag
937    /// would have to be set at each and would be forgotten at the next one
938    /// added.
939    pub nlink_written: u32,
940    /// When the link naming this dataset was created; see
941    /// [`GroupInfo::creation_seq`].
942    pub creation_seq: u64,
943    /// How this dataset records creation order for its attributes — the
944    /// file's creation-order policy captured when the dataset was created,
945    /// the way libhdf5 captures the DCPL. A dataset holds no links, so only
946    /// the attribute half of [`TrackOrder`] applies to it.
947    pub track_attr_order: CreationOrder,
948    /// User-defined fill value bytes (exactly one element wide). `None`
949    /// means default zero-fill; `Some` is emitted as a `fill_defined = 2`
950    /// fill-value message in the dataset object header.
951    pub fill_value: Option<Vec<u8>>,
952    /// Fill value write time (`H5Pset_fill_time`'s `H5D_fill_time_t`, one of
953    /// [`FILL_TIME_ALLOC`], [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]),
954    /// emitted verbatim into the fill-value message's write-time field.
955    /// Defaults to `FILL_TIME_IFSET`, `H5D_CRT_FILL_TIME_DEF` — what a fresh
956    /// dataset creation property list carries until `set_dataset_fill_time`
957    /// says otherwise.
958    pub fill_time: u8,
959    /// Layout message version for chunked storage: 4, or 5 when the chunk
960    /// index encodes stored chunk sizes in a fixed `sizeof_size` field
961    /// (libhdf5 2.0). Chosen at create by `Hdf5Writer::chunk_layout_version`,
962    /// preserved from the file on reopen, and emitted verbatim at finalize.
963    /// Contiguous datasets ignore it.
964    pub layout_version: u8,
965    /// The times this object tracks: `Some` exactly when it was created with
966    /// `H5Pset_obj_track_times(true)`, `None` when it was not.
967    ///
968    /// One meaning on both header versions, which store them differently and
969    /// store different amounts of them: a version-2 header keeps all four in
970    /// its prefix, and a version-1 dataset keeps one, in an `H5O_MTIME_NEW`
971    /// message. [`touch_oh`] is the single place that turns this into either
972    /// of those, so the four fields are here whichever version the object
973    /// has, exactly as `H5O_t` carries `atime`/`mtime`/`ctime`/`btime` for a
974    /// version-1 header it never serialises them from.
975    pub times: Option<ObjectTimes>,
976}
977
978impl DatasetInfo {
979    /// Which chunk index this dataset uses, `None` for storage that is not
980    /// chunked — the one place the index-carrying fields are turned into an
981    /// answer.
982    ///
983    /// INVARIANT: a chunk index added to this struct is added here. A site
984    /// that spells the disjunction out itself is what classifies a new index
985    /// as contiguous storage, and contiguous storage is read and written at
986    /// [`data_addr`](Self::data_addr) — which a chunked dataset leaves
987    /// undefined, so the misclassification is a read or a write at
988    /// `UNDEF_ADDR` rather than an error.
989    pub(crate) fn chunk_index_kind(&self) -> Option<ChunkIndexKind> {
990        if self.chunked.is_some() {
991            Some(ChunkIndexKind::ExtensibleArray)
992        } else if self.fixed_array.is_some() {
993            Some(ChunkIndexKind::FixedArray)
994        } else if self.btree_v2.is_some() {
995            Some(ChunkIndexKind::BtreeV2)
996        } else if self.implicit.is_some() {
997            Some(ChunkIndexKind::Implicit)
998        } else if self.single_chunk.is_some() {
999            Some(ChunkIndexKind::SingleChunk)
1000        } else if self.btree_v1.is_some() {
1001            Some(ChunkIndexKind::BtreeV1)
1002        } else {
1003            None
1004        }
1005    }
1006
1007    /// Whether this dataset's raw data is stored in chunks — the question
1008    /// every storage-form test asks, asked in one place.
1009    pub(crate) fn is_chunked(&self) -> bool {
1010        self.chunk_index_kind().is_some()
1011    }
1012
1013    /// Where this dataset's contiguous raw bytes live, or `None` when it has
1014    /// no contiguous storage to write into at all — a chunked dataset, a
1015    /// compact one (whose bytes *are* the layout message), or one whose block
1016    /// was never allocated.
1017    ///
1018    /// INVARIANT: every write of a contiguous dataset's raw bytes picks its
1019    /// destination here and reaches it through
1020    /// [`Hdf5Writer::write_contiguous_bytes`]. A site that read `data_addr`
1021    /// itself would write an externally-stored dataset's data into this file
1022    /// — at [`UNDEF_ADDR`], the far end of the address space — instead of into
1023    /// the files its header names, and would do the same to a virtual one,
1024    /// whose bytes are not this file's to write at all.
1025    ///
1026    /// Chunked storage is excluded through
1027    /// [`chunk_index_kind`](Self::chunk_index_kind) rather than by naming the
1028    /// index-carrying fields, so an index added to this struct cannot arrive
1029    /// here as contiguous storage: an implicit-indexed dataset reads
1030    /// `data_addr` as the base of its chunk grid, which as a contiguous
1031    /// destination would take a raw write meant for one chunk and lay it over
1032    /// the whole grid.
1033    fn contiguous_target(&self) -> Option<ContiguousTarget> {
1034        if self.is_chunked() || self.compact.is_some() {
1035            return None;
1036        }
1037        if self.virtual_storage.is_some() {
1038            return Some(ContiguousTarget::Virtual);
1039        }
1040        match &self.external {
1041            Some(ext) => Some(ContiguousTarget::External {
1042                files: ext.files.clone(),
1043                prefix: ext.prefix.expanded.clone(),
1044            }),
1045            None => {
1046                (self.data_addr != UNDEF_ADDR).then_some(ContiguousTarget::Local(self.data_addr))
1047            }
1048        }
1049    }
1050
1051    /// The one run of file bytes an implicitly indexed dataset's chunk grid
1052    /// is — its start and its length — or `None` when the dataset is indexed
1053    /// some other way or its space is not allocated yet.
1054    ///
1055    /// That index has no per-chunk structure to hold an address in: every
1056    /// chunk sits at `data_addr + linear_index * chunk_bytes` and the grid is
1057    /// allocated whole at create (`H5D__none_idx_get_addr`, H5Dnone.c). So the
1058    /// run is file space this writer allocated, and it is the *only* storage a
1059    /// chunk of such a dataset can occupy — the builder refuses external and
1060    /// virtual storage together with chunked storage, which is why
1061    /// [`allocated_storage_run`](Self::allocated_storage_run) can name it
1062    /// [`ContiguousTarget::Local`] and no chunk write can reach the other two.
1063    fn implicit_grid(&self) -> Option<(u64, u64)> {
1064        let imp = self.implicit.as_ref()?;
1065        (imp.data_addr != UNDEF_ADDR).then_some((imp.data_addr, imp.data_size))
1066    }
1067
1068    /// The run of raw storage this writer *allocated* for the dataset — the
1069    /// target to initialise it through and its size — or `None` when it
1070    /// allocated none.
1071    ///
1072    /// The two storage forms that are one run of bytes: a contiguous
1073    /// dataset's data block, and an implicitly indexed dataset's chunk grid.
1074    /// A compact dataset is excluded (its bytes are its layout message) and so
1075    /// is every other chunk index, whose chunks are placed one at a time.
1076    ///
1077    /// External storage is excluded because this writer does not allocate it:
1078    /// `H5D__alloc_storage` skips its whole body — the space reservation and
1079    /// the `H5D__init_storage` that would tile the fill value into it — for a
1080    /// dataset with an external file list or an empty extent, "we assume that
1081    /// external storage is already allocated by the caller, or at least will
1082    /// be before I/O is performed" (H5Dint.c:2270-2274). Measured under
1083    /// libhdf5 1.14.6 and 2.0.0: a user fill value, `H5D_FILL_TIME_ALLOC` and
1084    /// `H5D_ALLOC_TIME_EARLY` together leave the raw data file uncreated at
1085    /// `H5Dcreate2`, and a read before any write fails with "unable to open
1086    /// external raw data file" rather than reporting the fill.
1087    ///
1088    /// INVARIANT: only storage whose bytes this file owns is initialised as
1089    /// one run, so the allocate-time fill cannot reach the files an external
1090    /// file list names or the sources a virtual dataset maps.
1091    fn allocated_storage_run(&self) -> Option<(ContiguousTarget, u64)> {
1092        match self.implicit_grid() {
1093            Some((addr, size)) => Some((ContiguousTarget::Local(addr), size)),
1094            // Not a fallthrough for an unallocated implicit grid:
1095            // `contiguous_target` answers `None` for every chunked dataset.
1096            None => match self.contiguous_target() {
1097                Some(t @ ContiguousTarget::Local(_)) => Some((t, self.data_size)),
1098                _ => None,
1099            },
1100        }
1101    }
1102
1103    /// Whether this session wrote chunk data or changed the extent, so the
1104    /// dataset's index structures have to be re-flushed.
1105    fn storage_dirty(&self) -> bool {
1106        self.chunked.as_ref().is_some_and(|c| c.chunks_written > 0)
1107            || self
1108                .fixed_array
1109                .as_ref()
1110                .is_some_and(|f| f.chunks_written > 0)
1111            || self.btree_v2.as_ref().is_some_and(|b| b.chunks_written > 0)
1112            || self.btree_v1.as_ref().is_some_and(|b| b.chunks_written > 0)
1113            || self
1114                .single_chunk
1115                .as_ref()
1116                .is_some_and(|s| s.chunks_written > 0)
1117            || self.extent_dirty
1118    }
1119
1120    /// Whether a reopened dataset's on-disk object header no longer describes
1121    /// it.
1122    ///
1123    /// INVARIANT: every mutation of something `build_dataset_header` encodes
1124    /// must show up here. Finalize keeps the original header when this is
1125    /// false, so a change this misses is not deferred — it is discarded, with
1126    /// no error to say so. Attributes were the case that proved it: they are
1127    /// invisible to the chunk-write counters, so an attribute set on a
1128    /// reopened dataset vanished at close.
1129    fn header_stale(&self) -> bool {
1130        self.storage_dirty() || self.header_dirty
1131    }
1132
1133    /// The same question for the one thing the dataset itself cannot see: how
1134    /// many hard links resolve to it. That count lives in the header — an
1135    /// Object Reference Count message in a version-2 header, the `nlink`
1136    /// prefix field of a version-1 one — but it is a property of the file's
1137    /// link graph, so the caller supplies today's value.
1138    fn header_stale_with(&self, nlink: u32) -> bool {
1139        self.header_stale() || nlink != self.nlink_written
1140    }
1141
1142    /// Record that this dataset's on-disk object header was just written with
1143    /// `nlink` in it.
1144    ///
1145    /// INVARIANT: every write of a dataset object header passes through here.
1146    /// [`header_stale_with`](Self::header_stale_with) is the one authority for
1147    /// "does what is on disk still describe this dataset?", and it answers by
1148    /// comparing against [`nlink_written`](Self::nlink_written) — so a site
1149    /// that writes a header without saying so leaves that answer describing an
1150    /// older write. There are three writers: `finalize`, `finalize_for_swmr`
1151    /// and `write_dataset_header_inplace`. The last recorded nothing; it could
1152    /// not drift today only because a count it could write is a count that
1153    /// makes the header outgrow its block, which it refuses. That is a
1154    /// property of the reference-count message's size, not a rule anything
1155    /// states, and it is not what the field's definition rests on.
1156    fn header_written(&mut self, nlink: u32) {
1157        self.nlink_written = nlink;
1158    }
1159}
1160
1161/// Runtime metadata for a chunked dataset.
1162pub struct ChunkedDatasetInfo {
1163    /// Chunk dimension sizes.
1164    pub chunk_dims: Vec<u64>,
1165    /// Extensible array parameters.
1166    pub earray_params: EarrayParams,
1167    /// File offset of the EA header.
1168    pub ea_header_addr: u64,
1169    /// File offset of the EA index block.
1170    pub ea_iblk_addr: u64,
1171    /// In-memory copy of the EA header (for updating statistics).
1172    pub ea_header: ExtensibleArrayHeader,
1173    /// In-memory copy of the EA index block (for unfiltered datasets).
1174    pub ea_iblk: ExtensibleArrayIndexBlock,
1175    /// Number of chunks written so far.
1176    pub chunks_written: u64,
1177    /// Filtered index block (for compressed datasets).
1178    pub filt_iblk: Option<FilteredIndexBlock>,
1179    /// chunk_size_len for filtered entries.
1180    pub chunk_size_len: u8,
1181}
1182
1183/// Where a newly-created EA data block's address must be recorded.
1184enum DblkParent {
1185    /// Slot `index_block.dblk_addrs[idx]`.
1186    IndexBlock(usize),
1187    /// Slot `super_block.dblk_addrs[local_dblk]` of the super block at `sblk_addr`.
1188    SuperBlock {
1189        sblk_addr: u64,
1190        ndblks_in_sblk: usize,
1191        local_dblk: usize,
1192    },
1193}
1194
1195/// Which attribute list an attribute operation targets: the root group's,
1196/// a group's (by full path), or a dataset's (by writer index).
1197#[derive(Clone, Copy)]
1198pub enum AttrTarget<'a> {
1199    /// The root group's (file-level) attributes.
1200    Root,
1201    /// A group's attributes, by full path.
1202    Group(&'a str),
1203    /// A dataset's attributes, by writer index.
1204    Dataset(usize),
1205}
1206
1207/// Which chunk index a dataset uses.
1208///
1209/// The five above the line are what `H5D__layout_set_latest_indexing`
1210/// (H5Dlayout.c) picks between once the file format allows a version-4 data
1211/// layout message, in this precedence: a v2 B-tree for two or more unlimited
1212/// dimensions, an extensible array for exactly one, and — for a fixed shape —
1213/// the single-chunk index whenever exactly one chunk covers the whole
1214/// dataspace (`dims == max_dims == chunk_dims`, checked before either
1215/// alternative below and taken regardless of filter or allocation-time), else
1216/// the implicit index when nothing has to be recorded per chunk (no filter,
1217/// early allocation), else a fixed array. [`BtreeV1`](Self::BtreeV1) is not
1218/// one of them: it belongs to the version-3 layout message, and a file whose
1219/// superblock is older than version 2 can carry no other.
1220#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1221pub(crate) enum ChunkIndexKind {
1222    ExtensibleArray,
1223    FixedArray,
1224    BtreeV2,
1225    Implicit,
1226    SingleChunk,
1227    BtreeV1,
1228}
1229
1230/// A chunked dataset's grid geometry, snapshotted out of its slot.
1231///
1232/// The single owner of chunk-grid arithmetic: how many chunks span each
1233/// dimension, where a coordinate sits in the row-major order the array
1234/// indices record, and how many bytes one chunk holds.
1235struct ChunkGeometry {
1236    kind: ChunkIndexKind,
1237    dims: Vec<u64>,
1238    max_dims: Option<Vec<u64>>,
1239    chunk_dims: Vec<u64>,
1240    element_size: u64,
1241}
1242
1243impl ChunkGeometry {
1244    /// Unfiltered byte size of one whole chunk.
1245    fn chunk_bytes(&self) -> u64 {
1246        self.chunk_dims.iter().product::<u64>() * self.element_size
1247    }
1248
1249    /// Row-major position of `coords` in the chunk grid — the linear index an
1250    /// extensible or fixed array records the chunk under, computed against
1251    /// the maximum-extent grid by [`crate::io::chunk_grid::linear_index`].
1252    fn linear_index(&self, coords: &[u64]) -> IoResult<u64> {
1253        crate::io::chunk_grid::linear_index(
1254            &self.dims,
1255            self.max_dims.as_deref(),
1256            &self.chunk_dims,
1257            coords,
1258        )
1259    }
1260}
1261
1262/// The refusal every attribute mutation gets while SWMR streaming is
1263/// active, from the two owners of attribute-list change
1264/// ([`Hdf5Writer::set_attribute`] and `evict_attr`).
1265fn swmr_attr_error(name: &str) -> crate::io::IoError {
1266    crate::io::IoError::InvalidState(format!(
1267        "cannot add or modify attribute '{name}' during SWMR streaming: object \
1268         headers are frozen while readers stream, and a superseded variable-length \
1269         value's heap storage could never be reclaimed; set attributes before \
1270         start_swmr (libhdf5 forbids attribute changes during SWMR writes too)"
1271    ))
1272}
1273
1274/// Where an attribute arriving at [`Hdf5Writer::insert_attribute`] came from.
1275///
1276/// The variable-length setters have to evict before they allocate — the
1277/// free-before-alloc order — so by the time the replacement is inserted the
1278/// list no longer holds the entry it replaces, and the ordinary "already
1279/// present, so keep its index" test cannot see it. `H5A__attr_write` does not
1280/// create the attribute again, so the index travels with the eviction rather
1281/// than being stamped afresh; without it a rewritten attribute takes the set's
1282/// running maximum and moves to the end of the creation order.
1283#[derive(Debug, Clone, Copy)]
1284enum AttrOrigin {
1285    /// A new attribute, which takes the set's next creation index.
1286    Created,
1287    /// A value written over an attribute this writer has just evicted, which
1288    /// keeps that attribute's creation index — `None` when the object tracks
1289    /// no order, and so records none. An eviction that found nothing to remove
1290    /// answers `Created`: what follows it is a create like any other.
1291    Rewritten(Option<u16>),
1292}
1293use AttrOrigin::{Created, Rewritten};
1294
1295/// Take an object's attributes into the append session, or refuse the reopen.
1296///
1297/// Append mode rebuilds every object header it touches out of the attributes
1298/// read from it, so what this returns is what the object will still have when
1299/// the session finalizes. An attribute set that could not be read whole —
1300/// `ObjectAttributes::into_complete` refuses it — would come back as the part
1301/// that did read, silently deleting the rest.
1302///
1303/// Left to surface at `finalize`, that failure would land after this session's
1304/// chunk data and indices had already been written past the allocation point
1305/// the superblock still records, leaving a file libhdf5 reads as truncated.
1306/// Refusing the open leaves it untouched.
1307///
1308/// Size is no longer a reason to refuse: an attribute too large for a header
1309/// message goes back out through dense storage, the form libhdf5 read it from.
1310///
1311/// The set comes back in creation-index order, which is the order the registry
1312/// holds attributes in for an object made in this session too. A dense set is
1313/// read through the name index, so the order it arrives in is the order a hash
1314/// walk took; sorting here is what makes "the list is in creation order" true
1315/// of a reopened object as well, without any later stage having to know which
1316/// storage form the attributes came out of. Attributes of an untracked object
1317/// carry no index and keep the order they were read in.
1318fn take_reopened_attributes(
1319    attrs: crate::io::reader::ObjectAttributes,
1320    owner: &str,
1321) -> IoResult<Vec<AttributeEntry>> {
1322    let mut attrs = attrs.into_complete(owner)?;
1323    attrs.sort_by_key(|a| a.creation_index());
1324    Ok(attrs)
1325}
1326
1327/// The creation-order policy an on-disk object header declares — the single
1328/// owner of the recovery rule, used for the root group, every reopened group
1329/// and (through its attribute half) every reopened dataset.
1330///
1331/// The two halves come from two different places, and reading one for both is
1332/// how a file that sets only one of them came back with both or neither:
1333///
1334///   * links — the `Link Info` message's flag bits, which is what
1335///     `H5Pget_link_creation_order` reads (`H5G__get_create_plist`). A group
1336///     with no such message (or one this crate cannot decode) tracks nothing;
1337///     so does every dataset, which has no links to order.
1338///   * attributes — the object header's own flag bits, which is what
1339///     `H5Pget_attr_creation_order` reads (`H5Pocpl.c`). The `Attribute Info`
1340///     message carries the same two bits, but the header is the authority
1341///     libhdf5 consults, and it is present even when the object has no
1342///     attributes yet.
1343fn recover_track_order(
1344    header: &crate::format::object_header::ObjectHeader,
1345    ctx: &FormatContext,
1346) -> TrackOrder {
1347    let links = header
1348        .messages
1349        .iter()
1350        .find(|m| m.msg_type == crate::format::messages::MSG_LINK_INFO)
1351        .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
1352        .map(|(info, _)| info.creation_order())
1353        .unwrap_or_default();
1354    TrackOrder {
1355        links,
1356        attrs: header.attribute_creation_order(),
1357    }
1358}
1359
1360/// `H5O_touch_oh` (H5Oint.c:1273): put an object's tracked times where its
1361/// header version keeps them.
1362///
1363/// INVARIANT: every object header this writer builds passes its times through
1364/// here. The version decides the storage and nothing else does — a caller that
1365/// set `ObjectHeader::times` itself would hand a version-1 encode a prefix
1366/// field that version has no room for, and one that added the message itself
1367/// would put a second copy in a version-2 header.
1368///
1369/// `force` is upstream's own parameter, and it is what splits datasets from
1370/// everything else: it creates the version-1 `H5O_MTIME_NEW` message when the
1371/// header has none, and only `H5D__update_oh_info` passes it true
1372/// (H5Dint.c:1022-1026). Every other caller passes false and so creates no
1373/// message at all, which is why a version-1 group or committed datatype
1374/// records no time even when it is tracking them. A version-2 header keeps all
1375/// four times in its prefix whatever `force` says.
1376fn touch_oh(
1377    header: &mut ObjectHeader,
1378    format: ObjectFormat,
1379    times: Option<ObjectTimes>,
1380    force: bool,
1381) {
1382    let Some(times) = touched_times(times) else {
1383        return;
1384    };
1385    match format {
1386        ObjectFormat::Modern => header.times = Some(times),
1387        ObjectFormat::Legacy if force => header.add_message(
1388            crate::format::messages::MSG_MOD_TIME,
1389            0x00,
1390            ModificationTime(times.change).encode(),
1391        ),
1392        ObjectFormat::Legacy => {}
1393    }
1394}
1395
1396/// The times a header being (re)written carries, given what the object had.
1397///
1398/// Every object header this writer emits is one it is writing *now*, which is
1399/// what `H5O_touch_oh` is called for: an object that stores times gets its
1400/// access and change time moved to now, and one that does not store them stays
1401/// that way — the flag belongs to the object's creation property list, and a
1402/// rewrite is not a creation.
1403fn touched_times(times: Option<ObjectTimes>) -> Option<ObjectTimes> {
1404    times.map(|t| t.touched(now_seconds()))
1405}
1406
1407/// Seconds since the epoch, as an object header stores them (`H5_now`).
1408///
1409/// Saturates rather than wrapping: the field is a 32-bit count, and a clock
1410/// past 2106 is better reported as the largest time the format can express
1411/// than as a time in 1970. A clock before the epoch yields 0, which is what
1412/// libhdf5 writes for "no time recorded".
1413fn now_seconds() -> u32 {
1414    std::time::SystemTime::now()
1415        .duration_since(std::time::UNIX_EPOCH)
1416        .map_or(0, |d| u32::try_from(d.as_secs()).unwrap_or(u32::MAX))
1417}
1418
1419/// The dense storage an on-disk object header names: the fractal heap and the
1420/// indices its `Attribute Info` and `Link Info` messages point at.
1421///
1422/// A rewrite of that header lays fresh storage out and stops naming this, so
1423/// what this returns is exactly what the rewrite supersedes and must free.
1424/// Compact storage names no heap and yields `None` — there is nothing to free
1425/// and nothing that could be freed twice.
1426fn superseded_dense(
1427    header: &crate::format::object_header::ObjectHeader,
1428    ctx: &FormatContext,
1429) -> (Option<AttributeInfoMessage>, Option<LinkInfoMessage>) {
1430    let decode = |msg_type: u8| {
1431        header
1432            .messages
1433            .iter()
1434            .find(|m| m.msg_type == msg_type)
1435            .map(|m| m.data.as_slice())
1436    };
1437    let attrs = decode(crate::format::messages::MSG_ATTR_INFO)
1438        .and_then(|d| AttributeInfoMessage::decode(d, ctx).ok())
1439        .map(|(info, _)| info)
1440        .filter(|info| info.is_dense());
1441    let links = decode(crate::format::messages::MSG_LINK_INFO)
1442        .and_then(|d| LinkInfoMessage::decode(d, ctx).ok())
1443        .map(|(info, _)| info)
1444        .filter(|info| info.is_dense());
1445    (attrs, links)
1446}
1447
1448/// One collection block with free space that a later vlen insert may
1449/// fill — an entry in the writer's CWFS list (libhdf5 `f->shared->cwfs`).
1450struct CwfsEntry {
1451    /// Block address of the collection.
1452    addr: u64,
1453    /// Declared block size; never changes after allocation.
1454    size: usize,
1455    /// Bytes its free-space marker owns, per
1456    /// [`GlobalHeapCollection::free_space_at`](crate::format::global_heap::GlobalHeapCollection::free_space_at).
1457    free: usize,
1458}
1459
1460/// Maximum CWFS entries tracked — libhdf5's `H5HG_NCWFS` (H5HGpkg.h).
1461const H5HG_NCWFS: usize = 16;
1462
1463/// Record a collection with `free` bytes in the CWFS list: update its
1464/// entry if present, append while the list is short, and otherwise
1465/// replace the entry with the least free space when this one has more —
1466/// the retention rule of libhdf5's `H5HG_insert`.
1467fn cwfs_note(cwfs: &mut Vec<CwfsEntry>, addr: u64, size: usize, free: usize) {
1468    if let Some(p) = cwfs.iter().position(|e| e.addr == addr) {
1469        cwfs[p].free = free;
1470        return;
1471    }
1472    if cwfs.len() < H5HG_NCWFS {
1473        cwfs.insert(0, CwfsEntry { addr, size, free });
1474        return;
1475    }
1476    if let Some(p) = (0..cwfs.len()).min_by_key(|&p| cwfs[p].free) {
1477        if free > cwfs[p].free {
1478            cwfs[p] = CwfsEntry { addr, size, free };
1479        }
1480    }
1481}
1482
1483/// The uniform rejection for `delete_dataset` / `delete_group` while SWMR
1484/// streaming is active: deleting frees the object's blocks, and a live
1485/// reader may hold any of their addresses.
1486fn swmr_delete_error(name: &str) -> crate::io::IoError {
1487    crate::io::IoError::InvalidState(format!(
1488        "cannot delete '{name}' during SWMR streaming: a reader may hold the \
1489         object's header and storage addresses (libhdf5 forbids link deletion \
1490         during SWMR writes too)"
1491    ))
1492}
1493
1494/// Whether the chunk at grid `coords` lies entirely at or beyond `extent` in
1495/// some dimension — no element of it would survive a shrink to that extent.
1496fn chunk_outside_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1497    coords
1498        .iter()
1499        .zip(chunk_dims)
1500        .zip(extent)
1501        .any(|((&c, &cd), &e)| c.saturating_mul(cd) >= e)
1502}
1503
1504/// Whether the chunk at grid `coords` keeps elements under `extent` but
1505/// extends past it in some dimension — a shrink must refill its
1506/// out-of-extent region with the fill value.
1507fn chunk_straddles_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1508    !chunk_outside_extent(coords, chunk_dims, extent)
1509        && coords
1510            .iter()
1511            .zip(chunk_dims)
1512            .zip(extent)
1513            .any(|((&c, &cd), &e)| (c + 1).saturating_mul(cd) > e)
1514}
1515
1516/// Overwrite, in `data` (one whole chunk, unfiltered, row-major), every
1517/// element at or beyond `extent` with the matching bytes of `fill` — a
1518/// same-sized buffer tiled with the fill value. The caller guarantees the
1519/// chunk at `coords` straddles `extent`, so every dimension keeps at least
1520/// one element. Returns the replaced bytes, so a vlen dataset's dead
1521/// heap references can be released rather than stranded.
1522fn refill_chunk_beyond_extent(
1523    data: &mut [u8],
1524    fill: &[u8],
1525    coords: &[u64],
1526    chunk_dims: &[u64],
1527    extent: &[u64],
1528    element_size: usize,
1529) -> Vec<u8> {
1530    let ndims = chunk_dims.len();
1531    let keep: Vec<usize> = (0..ndims)
1532        .map(|d| {
1533            let origin = coords[d] * chunk_dims[d];
1534            chunk_dims[d].min(extent[d].saturating_sub(origin)) as usize
1535        })
1536        .collect();
1537    // Row-major walk: for every row (all dimensions but the last),
1538    // overwrite the whole row when its prefix is outside the keep box,
1539    // else only the row's out-of-extent tail.
1540    let row_elems = chunk_dims[ndims - 1] as usize;
1541    let keep_last = keep[ndims - 1];
1542    let nrows: u64 = chunk_dims[..ndims - 1].iter().product();
1543    let mut replaced = Vec::new();
1544    for r in 0..nrows {
1545        let mut rem = r;
1546        let mut in_keep = true;
1547        for d in (0..ndims - 1).rev() {
1548            let c = rem % chunk_dims[d];
1549            rem /= chunk_dims[d];
1550            if c as usize >= keep[d] {
1551                in_keep = false;
1552            }
1553        }
1554        let start = if in_keep { keep_last } else { 0 };
1555        if start == row_elems {
1556            continue;
1557        }
1558        let a = (r as usize * row_elems + start) * element_size;
1559        let b = (r as usize + 1) * row_elems * element_size;
1560        replaced.extend_from_slice(&data[a..b]);
1561        data[a..b].copy_from_slice(&fill[a..b]);
1562    }
1563    replaced
1564}
1565
1566/// Validate caller-supplied chunk geometry at dataset definition, the rule
1567/// libhdf5 applies in `H5D__chunk_construct` (H5Dchunk.c): the chunk rank
1568/// must match the dataspace rank, no chunk dimension may be zero, and a
1569/// chunk dimension may not exceed a fixed maximum dimension — except in a
1570/// dimension whose current size is zero, which libhdf5 exempts.
1571fn validate_chunk_geometry(dims: &[u64], max_dims: &[u64], chunk_dims: &[u64]) -> IoResult<()> {
1572    let ndims = dims.len();
1573    if chunk_dims.len() != ndims {
1574        return Err(crate::io::IoError::InvalidState(format!(
1575            "chunk shape has {} dimensions but the dataspace has {}",
1576            chunk_dims.len(),
1577            ndims
1578        )));
1579    }
1580    if max_dims.len() != ndims {
1581        return Err(crate::io::IoError::InvalidState(format!(
1582            "maximum shape has {} dimensions but the dataspace has {}",
1583            max_dims.len(),
1584            ndims
1585        )));
1586    }
1587    for d in 0..ndims {
1588        if chunk_dims[d] == 0 {
1589            return Err(crate::io::IoError::InvalidState(format!(
1590                "chunk dimension {d} is zero"
1591            )));
1592        }
1593        if dims[d] != 0 && max_dims[d] != u64::MAX && max_dims[d] < chunk_dims[d] {
1594            return Err(crate::io::IoError::InvalidState(format!(
1595                "chunk dimension {} is {} but the maximum dimension size is {}",
1596                d, chunk_dims[d], max_dims[d]
1597            )));
1598        }
1599    }
1600    Ok(())
1601}
1602
1603/// An extensible-array index requires at most one unlimited dimension —
1604/// `H5D__chunk_construct` (H5Dchunk.c) only selects this index for exactly
1605/// one — at any position: `chunk_grid::linear_index` seeds the unlimited
1606/// dimension into the slot no down-chunks multiplier touches, the same
1607/// address libhdf5 reaches by swizzling it to the slowest position
1608/// (`H5VM_swizzle_coords`, H5Dearray.c). Two or more unlimited dimensions
1609/// have no finite grid at all; that shape needs a v2 B-tree index instead.
1610fn ensure_at_most_one_unlimited(max_dims: &[u64]) -> IoResult<()> {
1611    let unlimited: Vec<usize> = max_dims
1612        .iter()
1613        .enumerate()
1614        .filter(|&(_, &m)| m == u64::MAX)
1615        .map(|(d, _)| d)
1616        .collect();
1617    if unlimited.len() > 1 {
1618        return Err(crate::io::IoError::InvalidState(format!(
1619            "an extensible-array index supports at most one unlimited dimension, \
1620             but dimensions {unlimited:?} are all unlimited; a v2 B-tree index \
1621             handles two or more"
1622        )));
1623    }
1624    Ok(())
1625}
1626
1627/// Reject strings the dataset's declared character set cannot label.
1628///
1629/// A Rust `&str` is always UTF-8, so only an ASCII declaration (charset 0)
1630/// can be violated. libhdf5 stores the bytes unvalidated — its vlen write
1631/// path has no cset check anywhere — which mislabels them for every reader
1632/// that trusts the declaration (h5py raises on the same mismatch).
1633fn ensure_vlen_charset(charset: u8, strings: &[&str]) -> IoResult<()> {
1634    if charset == 0 {
1635        if let Some((i, s)) = strings.iter().enumerate().find(|(_, s)| !s.is_ascii()) {
1636            return Err(crate::io::IoError::InvalidState(format!(
1637                "string {i} ({s:?}) is not ASCII, but the dataset's character set is"
1638            )));
1639        }
1640    }
1641    Ok(())
1642}
1643
1644/// Runtime metadata for a fixed-array-indexed chunked dataset.
1645pub struct FixedArrayDatasetInfo {
1646    /// Chunk dimension sizes.
1647    pub chunk_dims: Vec<u64>,
1648    /// File offset of the FA header.
1649    pub fa_header_addr: u64,
1650    /// File offset of the FA data block.
1651    pub fa_dblk_addr: u64,
1652    /// In-memory copy of the FA header.
1653    pub fa_header: FixedArrayHeader,
1654    /// In-memory copy of the FA data block.
1655    pub fa_dblk: FixedArrayDataBlock,
1656    /// Number of chunks written so far.
1657    pub chunks_written: u64,
1658}
1659
1660/// Runtime metadata for an implicitly indexed chunked dataset — the index
1661/// that is no structure at all (`H5Dnone.c`).
1662///
1663/// Every chunk of the maximum-extent grid is allocated at create in one
1664/// contiguous run, in the row-major order [`crate::io::chunk_grid`] defines,
1665/// so a chunk's address is `data_addr + linear_index * chunk_bytes` and
1666/// nothing has to be recorded when one is written. libhdf5 picks this index
1667/// only when that arithmetic is total: no filter (every chunk is exactly
1668/// `chunk_bytes` long), no unlimited dimension (the run has a finite length),
1669/// and early allocation (the run exists before any write).
1670pub struct ImplicitDatasetInfo {
1671    /// Chunk dimension sizes.
1672    pub chunk_dims: Vec<u64>,
1673    /// File offset of the first chunk — the layout message's index address.
1674    pub data_addr: u64,
1675    /// Byte length of the whole chunk run: `nchunks * chunk_bytes`.
1676    pub data_size: u64,
1677}
1678
1679/// Runtime metadata for a single-chunk indexed dataset (`H5Dsingle.c`): a
1680/// fixed dataspace exactly one chunk wide in every dimension
1681/// (`dims == max_dims == chunk_dims`), so there is exactly one chunk and its
1682/// address — and, when filtered, its stored size and filter mask — are held
1683/// directly in the layout message rather than in any index structure.
1684///
1685/// libhdf5 selects this index ahead of the implicit and fixed-array indexes
1686/// whenever the shape qualifies, whether or not the dataset is filtered or
1687/// early-allocated (`H5D__layout_set_latest_indexing`, H5Dlayout.c).
1688pub struct SingleChunkDatasetInfo {
1689    /// Chunk dimension sizes (equal to the dataspace's `dims`).
1690    pub chunk_dims: Vec<u64>,
1691    /// File offset of the chunk, [`UNDEF_ADDR`] until the chunk is written
1692    /// (or immediately, for an unfiltered dataset created with early
1693    /// allocation).
1694    pub data_addr: u64,
1695    /// The chunk's full unfiltered byte length — `chunk_dims.product() *
1696    /// element_size`, fixed for the dataset's lifetime.
1697    pub data_size: u64,
1698    /// Stored (on-disk) byte length: equal to `data_size` when the dataset
1699    /// carries no filter pipeline; the filtered length once the chunk has
1700    /// been written, 0 before then.
1701    pub nbytes: u64,
1702    /// Filter mask recorded for the stored chunk (bit *i* set means filter
1703    /// *i* was skipped); meaningful only when the dataset is filtered.
1704    pub filter_mask: u32,
1705    /// Chunks written this session (0 or 1) — `storage_dirty`'s signal that
1706    /// the layout message's address/size/mask fields must be re-flushed.
1707    pub chunks_written: u64,
1708    /// Whether this dataset was created with early allocation
1709    /// (`H5D_ALLOC_TIME_EARLY`) — distinct from `data_addr` being defined,
1710    /// which also becomes true the moment an incrementally allocated
1711    /// dataset's one chunk is written; `build_dataset_header` needs this to
1712    /// tell the two apart when it reports the fill-value message's
1713    /// allocation time. Only ever set for an unfiltered dataset: a filtered
1714    /// chunk's stored length is not known until it is compressed, so there
1715    /// is nothing to allocate ahead of that write regardless of alloc time
1716    /// (the same gap `create_fixed_array_dataset_with_pipeline` has).
1717    pub early_alloc: bool,
1718}
1719
1720/// One chunk as the version-1 B-tree records it — the key libhdf5 stores
1721/// (`H5D_btree_key_t`) plus the address it keys.
1722pub struct BtreeV1ChunkRecord {
1723    /// Grid position of the chunk. The key's element offsets are derived from
1724    /// it at encode time (`scaled * chunk_dim`), so this is the one place the
1725    /// position is stored and the sort order is over these coordinates.
1726    pub scaled: Vec<u64>,
1727    /// File offset of the chunk's bytes.
1728    pub address: u64,
1729    /// Stored byte length — the filtered length when the dataset is filtered,
1730    /// the full chunk otherwise. `u32` because the key's field is.
1731    pub nbytes: u32,
1732    /// Filter mask: bit `i` set means filter `i` was skipped for this chunk.
1733    pub filter_mask: u32,
1734}
1735
1736/// Runtime metadata for a chunked dataset indexed by a version-1 B-tree —
1737/// the classic-format chunk index (`H5Dbtree.c`), and the only one a
1738/// version-0/1 superblock file can carry.
1739pub struct BtreeV1DatasetInfo {
1740    /// Chunk dimension sizes.
1741    pub chunk_dims: Vec<u64>,
1742    /// Maximum dimensions (u64::MAX = unlimited).
1743    pub max_dims: Vec<u64>,
1744    /// The file's v1-B-tree "K" ranks. Every node's width is derived from
1745    /// them, and they are recorded only in the superblock this file was
1746    /// opened with — so they are carried rather than re-derived.
1747    pub config: BTreeV1Config,
1748    /// The chunks, in key order (`scaled` ascending, lexicographically).
1749    pub records: Vec<BtreeV1ChunkRecord>,
1750    /// Pool of node-size blocks holding the tree's nodes, on the same terms
1751    /// as [`Bt2DatasetInfo::node_addrs`]: a flush re-serializes the whole
1752    /// bulk-loaded tree over them and allocates only the shortfall, so no
1753    /// flush can orphan a block it replaced.
1754    pub node_addrs: Vec<u64>,
1755    /// Address of the tree's root node — what the version-3 data layout
1756    /// message carries. `UNDEF_ADDR` until a flush puts a node in the file,
1757    /// which is the state libhdf5 leaves a chunked dataset in until its first
1758    /// chunk is written.
1759    pub root_addr: u64,
1760    /// Number of chunks written so far.
1761    pub chunks_written: u64,
1762}
1763
1764impl BtreeV1DatasetInfo {
1765    /// The chunk shape a key's offsets are scaled by: the chunk dimensions
1766    /// with the element size appended, which is also what the layout message
1767    /// stores.
1768    fn key_dims(&self, element_size: u64) -> Vec<u64> {
1769        let mut dims = self.chunk_dims.clone();
1770        dims.push(element_size);
1771        dims
1772    }
1773
1774    /// Bulk-load the tree this index's records describe.
1775    fn build_tree(&self, element_size: u64, sizeof_addr: usize) -> ChunkBTreeV1Tree {
1776        let dims = self.key_dims(element_size);
1777        let entries: Vec<(ChunkKey, u64)> = self
1778            .records
1779            .iter()
1780            .map(|r| {
1781                (
1782                    ChunkKey::for_chunk(&r.scaled, &dims, r.nbytes, r.filter_mask),
1783                    r.address,
1784                )
1785            })
1786            .collect();
1787        // The right boundary closes the tree past its greatest key, which is
1788        // the last record's — the records are kept in key order.
1789        let last = self
1790            .records
1791            .last()
1792            .map_or_else(|| vec![0; self.chunk_dims.len()], |r| r.scaled.clone());
1793        ChunkBTreeV1Tree::build(
1794            &entries,
1795            ChunkKey::right_bound(&last, &dims),
1796            &self.config,
1797            sizeof_addr,
1798        )
1799    }
1800
1801    /// Where `scaled` sits in [`records`](Self::records): `Ok` at its record,
1802    /// `Err` at the position one would be inserted at.
1803    fn position(&self, scaled: &[u64]) -> Result<usize, usize> {
1804        self.records
1805            .binary_search_by(|r| r.scaled.as_slice().cmp(scaled))
1806    }
1807}
1808
1809/// Runtime metadata for a B-tree v2 indexed chunked dataset.
1810pub struct Bt2DatasetInfo {
1811    /// Chunk dimension sizes.
1812    pub chunk_dims: Vec<u64>,
1813    /// File offset of the BT2 header.
1814    pub bt2_header_addr: u64,
1815    /// Pool of node-size blocks (the index's
1816    /// [`node_size`](Bt2ChunkIndex::node_size) bytes each) holding the tree's
1817    /// nodes, in the order [`Bt2Tree::encode`] emits them.
1818    ///
1819    /// The single owner of the tree's node addresses: a flush re-serializes the
1820    /// whole tree over these blocks and allocates only the shortfall, so no
1821    /// flush can orphan a block it replaced. Every node is the same size, so a
1822    /// block stays usable however the tree reshapes.
1823    ///
1824    /// The pool holds exactly one block per node after every flush, in both
1825    /// directions: a taller tree allocates the shortfall, a smaller one frees
1826    /// the surplus. Nothing here depends on the record count only ever rising,
1827    /// so a record-removal path can be added to [`Bt2ChunkIndex`] without the
1828    /// blocks it drops going unreachable.
1829    pub node_addrs: Vec<u64>,
1830    /// In-memory chunk index.
1831    pub index: Bt2ChunkIndex,
1832    /// Number of chunks written so far.
1833    pub chunks_written: u64,
1834}
1835
1836/// Metadata for a group being written.
1837pub struct GroupInfo {
1838    /// Full path of this group (e.g. "/detector" or "/detector/raw").
1839    pub name: String,
1840    /// Index of the parent group in the groups vec, or None for root-level groups.
1841    pub parent: Option<usize>,
1842    /// Indices of child datasets (into `datasets` vec).
1843    pub child_datasets: Vec<usize>,
1844    /// Indices of child groups (into `groups` vec).
1845    pub child_groups: Vec<usize>,
1846    /// File offset of this group's object header (set during finalize).
1847    pub obj_header_addr: u64,
1848    /// File offset of the on-disk header a reopen found for this group, so
1849    /// finalize can free the block it supersedes.
1850    pub obj_header_written_addr: Option<u64>,
1851    /// Encoded size of that on-disk header (first block).
1852    /// Every block the object's on-disk header occupies, chunk 0 first, or
1853    /// empty when it has none yet. A rewrite keeps chunk 0's block — its
1854    /// address is what every reference to the object holds — and frees the
1855    /// rest, so a continuation block left behind is space no free-space
1856    /// manager records.
1857    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
1858    /// Soft-deleted: excluded from finalize output.
1859    pub deleted: bool,
1860    /// Attributes attached to this group (e.g. NeXus `NX_class`).
1861    pub attributes: Vec<AttributeEntry>,
1862    /// When the link naming this group was created, on the writer's single
1863    /// monotonic sequence. Groups, datasets and hard links share it, so a
1864    /// parent can order its links the way they were actually made.
1865    pub creation_seq: u64,
1866    /// How this group records creation order for its links and, separately,
1867    /// for its attributes. Creation-order tracking is a property of the
1868    /// object's creation property list in libhdf5, so it is captured here
1869    /// when the group is created rather than read from the writer at
1870    /// finalize: a later change of policy must not rewrite an object already
1871    /// made.
1872    pub track_order: TrackOrder,
1873    /// The times this group tracks, on the same terms as
1874    /// [`DatasetInfo::times`]. A version-1 group header records none of them:
1875    /// nothing calls `H5O_touch_oh` with `force` for a group, so the message a
1876    /// version-1 dataset gets is never created for one.
1877    pub times: Option<ObjectTimes>,
1878}
1879
1880/// One object's creation-order policy, with the two subsystems libhdf5 keeps
1881/// apart kept apart here too.
1882///
1883/// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` are separate
1884/// calls reading back out of separate places on disk — the Link Info message
1885/// and the object header's own flag bits — and a file may set either alone.
1886/// Carrying them as one flag made a reopen give a one-of-two file both or
1887/// neither.
1888#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
1889pub struct TrackOrder {
1890    /// Creation order of the links this group holds. Meaningless for a
1891    /// dataset, which is why `DatasetInfo` keeps only the attribute half.
1892    pub links: CreationOrder,
1893    /// Creation order of the attributes attached to this object.
1894    pub attrs: CreationOrder,
1895}
1896
1897impl TrackOrder {
1898    /// The policy the crate's single `track_order` knob selects: both
1899    /// subsystems tracked *and* indexed, or neither — the pair h5py's
1900    /// `File(track_order=True)` writes.
1901    pub fn uniform(track: bool) -> Self {
1902        let order = if track {
1903            CreationOrder::Indexed
1904        } else {
1905            CreationOrder::Untracked
1906        };
1907        Self {
1908            links: order,
1909            attrs: order,
1910        }
1911    }
1912}
1913
1914/// The object a [`HardLink`] resolves to.
1915#[derive(Clone, Copy)]
1916pub enum HardLinkTarget {
1917    /// Index into the writer's `datasets` vec.
1918    Dataset(usize),
1919    /// Index into the writer's `groups` vec.
1920    Group(usize),
1921}
1922
1923/// A user-created hard link: an additional name, in some group, for an
1924/// object that already exists under its own name.
1925///
1926/// The HDF5 file format makes every group entry a `name -> object header
1927/// address` mapping, so a hard link is just a second such entry pointing at
1928/// an already-written object. No data is copied.
1929#[derive(Clone)]
1930pub struct HardLink {
1931    /// Parent group index (`None` = the root group).
1932    pub parent: Option<usize>,
1933    /// Leaf name of the link within the parent group.
1934    pub name: String,
1935    /// Object this link resolves to.
1936    pub target: HardLinkTarget,
1937    /// When this link was created; see [`GroupInfo::creation_seq`].
1938    pub creation_seq: u64,
1939}
1940
1941/// A user-created symbolic link: a name in a group whose value is a path
1942/// rather than an object header address.
1943///
1944/// A soft link holds a path within this file; an external link holds a file
1945/// name and a path within that file. Neither names an object this writer
1946/// owns, so — unlike [`HardLink`] — nothing about it is resolved: the link is
1947/// stored as written and answered at traversal time, exactly as `H5Lcreate_soft`
1948/// and `H5Lcreate_external` store theirs.
1949#[derive(Clone)]
1950pub struct SymbolicLink {
1951    /// Parent group index (`None` = the root group).
1952    pub parent: Option<usize>,
1953    /// Leaf name of the link within the parent group.
1954    pub name: String,
1955    /// The path (and, for an external link, the file) this link names.
1956    pub target: LinkTarget,
1957    /// When this link was created; see [`GroupInfo::creation_seq`].
1958    pub creation_seq: u64,
1959}
1960
1961/// A committed (named) datatype: an object header holding one datatype
1962/// message and nothing else, reached by a link like any other object.
1963///
1964/// `H5Tcommit2` makes the type an object in its own right so several datasets
1965/// can declare they share it; each of those datasets then stores a pointer to
1966/// this object header in place of its own datatype message. The object's
1967/// reference count is therefore the links naming it *plus* the datasets
1968/// sharing it — `H5O__shared_link_adj` counts a share as a link — and an
1969/// object no link and no dataset reaches is not written at all.
1970#[derive(Clone)]
1971pub struct CommittedDatatype {
1972    /// Full path with no leading `/`, the form dataset names take.
1973    pub name: String,
1974    /// Parent group index (`None` = the root group).
1975    pub parent: Option<usize>,
1976    /// The committed type.
1977    pub datatype: DatatypeMessage,
1978    /// When the link naming it was created; see [`GroupInfo::creation_seq`].
1979    pub creation_seq: u64,
1980    /// The times it tracks, on the same terms as [`DatasetInfo::times`]. A
1981    /// version-1 committed datatype header records none of them, for the same
1982    /// reason a version-1 group's does not.
1983    pub times: Option<ObjectTimes>,
1984    /// File offset of its object header (set during finalize).
1985    pub obj_header_addr: u64,
1986}
1987
1988/// Where the object header a dataset's shared datatype pointer must name
1989/// comes from.
1990///
1991/// A dataset built on a committed type stores no datatype message: it stores
1992/// the address of the type's object header. Only the address matters at
1993/// encode time, but it is knowable at two different moments — a type this
1994/// session commits has no address until finalize lays the file out, while one
1995/// a reopen found is already at an address this session will not move. Naming
1996/// both here keeps [`build_dataset_header`](Hdf5Writer::build_dataset_header)
1997/// the one place that turns a share into a pointer, whichever way the share
1998/// arrived.
1999#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2000pub enum CommittedTypeRef {
2001    /// A type committed in this session, by its index in
2002    /// [`committed_datatypes`](Hdf5Writer::committed_datatypes); its address
2003    /// is read from that registry once finalize has stamped one.
2004    Session(usize),
2005    /// A committed datatype a reopen kept by its bytes, at the object header
2006    /// address it already occupies.
2007    Preserved(u64),
2008}
2009
2010/// A link a reopened file already held that this writer cannot express.
2011///
2012/// Soft, external and user-defined links have no creation, retarget or delete
2013/// operation here — only hard links do — so a header rewrite that emits what
2014/// the registry models would erase them. Their encoded `Link` message rides
2015/// along instead and is written back byte for byte, which preserves every
2016/// field (name character set, creation order, the link value) without this
2017/// writer having to model any of them.
2018///
2019/// A *hard* link is preserved the same way when the object it names is one
2020/// the reopen could not model: writing the link back unchanged leaves that
2021/// object's header exactly where it is, which is the only way the rewrite can
2022/// keep what it cannot rebuild.
2023#[derive(Clone)]
2024pub struct PreservedLink {
2025    /// Parent group index (`None` = the root group).
2026    pub parent: Option<usize>,
2027    /// Leaf name of the link within the parent group.
2028    pub name: String,
2029    /// The link's class, decoded once at collection so listings can report
2030    /// it. Never the source of what gets written — `encoded` is.
2031    pub class: crate::io::reader::LinkClass,
2032    /// The encoded `Link` message body, exactly as read from the file.
2033    pub encoded: Vec<u8>,
2034    /// Why the object this link names could not be modelled, for the callers
2035    /// that ask for it by name. `None` when the link's own class — not its
2036    /// target — is what this writer cannot express.
2037    pub reason: Option<String>,
2038    /// What the object this link names is, when the walk could tell. A
2039    /// listing asks this; `reason` is prose for the caller that asks why.
2040    pub kind: PreservedKind,
2041}
2042
2043/// Every link a reopen walk met, split by what the writer can do with it.
2044/// A header rewrite emits both halves, so a link in neither half is a link
2045/// the close would destroy.
2046#[derive(Default)]
2047struct CollectedLinks {
2048    /// Hard links whose target the reopen modelled, with the plan that says
2049    /// how to rebuild it.
2050    hard: Vec<(HardEntry, CollectedObject)>,
2051    /// Links written back unchanged: the class this writer cannot express,
2052    /// and the hard links whose object it cannot model.
2053    preserved: Vec<PreservedEntry>,
2054}
2055
2056/// One hard link the reopen walk met: what it names, and the exact message
2057/// that names it.
2058#[derive(Clone)]
2059struct HardEntry {
2060    /// Full link path, in the no-leading-`/` form the registry uses.
2061    path: String,
2062    /// Object header address the link names.
2063    address: u64,
2064    /// The encoded `Link` message body, exactly as read from the file.
2065    encoded: Vec<u8>,
2066}
2067
2068/// A link the rewrite writes back exactly as it read it.
2069struct PreservedEntry {
2070    path: String,
2071    class: crate::io::reader::LinkClass,
2072    encoded: Vec<u8>,
2073    /// Why the object it names could not be modelled; `None` when the link's
2074    /// own class is what this writer cannot express.
2075    reason: Option<String>,
2076    /// What the object is, when the walk could tell.
2077    kind: PreservedKind,
2078}
2079
2080/// What a reopen can do with one object it reached.
2081///
2082/// A header rewrite emits a modelled object out of the registry, so the
2083/// registry may hold an object only when *every* message the model consumes
2084/// decoded. A partial read is not a smaller object, it is a different one:
2085/// before this rule a dataset whose datatype message did not decode was
2086/// registered as a group, and the close rewrote its header as one.
2087enum ObjectPlan {
2088    /// A dataset the rewrite can rebuild.
2089    Dataset(Box<DatasetParts>),
2090    /// A group the rewrite can rebuild, and the links it holds.
2091    Group(GroupParts),
2092    /// An object this writer cannot model, and why. Its header is never
2093    /// rewritten and never freed; the link naming it is written back byte for
2094    /// byte, so the object stays exactly as the file already had it — what
2095    /// libhdf5 does with the parts of a file it does not understand.
2096    ///
2097    /// `kind` is what the walk could still tell about the object it is
2098    /// keeping. Not modelling an object is not the same as not knowing what
2099    /// it is, and answering the second question with the first is what made
2100    /// `named_datatype_names` deny, in write mode, a datatype the same file
2101    /// reports in read mode.
2102    Preserve { why: String, kind: PreservedKind },
2103}
2104
2105/// What a preserved object is, as far as the reopen walk could tell.
2106///
2107/// Deliberately not a copy of the reader's `ObjectKind`: that one carries the
2108/// decoded object, and a preserved object is precisely the one whose contents
2109/// the writer does not decode. This says only what a listing needs.
2110#[derive(Clone, Copy, PartialEq, Eq, Debug)]
2111pub enum PreservedKind {
2112    /// The walk did not classify it — or the link's own class, not its
2113    /// target, is what could not be expressed.
2114    Unclassified,
2115    /// A committed (named) datatype, by
2116    /// [`header_is_committed_datatype`](crate::io::reader::header_is_committed_datatype).
2117    NamedDatatype,
2118}
2119
2120impl ObjectPlan {
2121    /// An object kept by its bytes, of a kind the walk did not classify.
2122    ///
2123    /// Every reason that is a *failure* to read reaches this: a message that
2124    /// did not decode says nothing about what the object was.
2125    fn preserve(why: impl Into<String>) -> Self {
2126        ObjectPlan::Preserve {
2127            why: why.into(),
2128            kind: PreservedKind::Unclassified,
2129        }
2130    }
2131}
2132
2133/// The messages a dataset's rewrite is built from, all decoded.
2134struct DatasetParts {
2135    /// Every block the header chain occupies, chunk 0 first. All of them are
2136    /// superseded: the rewrite re-encodes the whole chain into one fresh
2137    /// chunk, so a continuation left unfreed is space nothing claims.
2138    header_blocks: crate::io::object_header_io::HeaderBlocks,
2139    datatype: DatatypeMessage,
2140    /// The committed datatype object header `datatype` was read *through*,
2141    /// when the header stores a pointer instead of a message of its own.
2142    ///
2143    /// The literal type is in `datatype` either way, because the read resolves
2144    /// the pointer before anything decodes it; this is what a rewrite needs to
2145    /// put the pointer back rather than inline a copy of the named type and
2146    /// leave `H5Tcommitted` false.
2147    committed_type: Option<u64>,
2148    dataspace: crate::format::messages::dataspace::DataspaceMessage,
2149    /// The object format the reopen found this dataset's messages written in,
2150    /// read from the dataspace message's own version byte.
2151    ///
2152    /// A version-2 superblock does not settle it: `H5F__super_init` raises the
2153    /// superblock for a shared-message table or non-default file-space
2154    /// properties without touching `H5F_LOW_BOUND` (H5Fsuper.c:1135, :1144), so
2155    /// a file created at the earliest bound with either can hold version-1
2156    /// messages under a version-2 superblock — which is what
2157    /// `tests/fixtures/sohm_*.h5` are.
2158    read_format: ObjectFormat,
2159    layout: crate::format::messages::data_layout::DataLayoutMessage,
2160    filter_pipeline: Option<FilterPipeline>,
2161    fill_value: Option<Vec<u8>>,
2162    /// The fill-value message's write-time byte, preserved across a
2163    /// rewrite the same way `fill_value` is — an appended-to dataset must
2164    /// keep the policy libhdf5 (or this writer) declared for it, not fall
2165    /// back to the `H5D_CRT_FILL_TIME_DEF` a fresh dataset gets.
2166    fill_write_time: u8,
2167    attributes: Vec<AttributeEntry>,
2168    /// The creation-order policy the on-disk header declares; a rewrite that
2169    /// read it from the writer instead would stamp this session's policy onto
2170    /// an object libhdf5 created under another.
2171    track_order: TrackOrder,
2172    /// The times the on-disk header records, for the same reason: whether an
2173    /// object tracks them is settled when it is created, not when it is
2174    /// rewritten. Recovered by [`ObjectHeader::recorded_times`].
2175    times: Option<ObjectTimes>,
2176    /// The dense storage the rewrite supersedes and must free.
2177    dense: DenseCarry,
2178    /// The External File List the header carries, with each slot's name
2179    /// already read back out of the local heap the message points at. `None`
2180    /// for a dataset whose raw data is in this file.
2181    ///
2182    /// Carried rather than re-derived because the rewrite has to re-emit the
2183    /// message: a contiguous layout with an undefined address and no EFL
2184    /// beside it is a dataset with no data at all, so dropping this on a
2185    /// header rewrite would silently unlink every external byte.
2186    external: Option<ExternalStorage>,
2187}
2188
2189/// The same for a group, plus the links it holds — decoded once, with the
2190/// bytes they came from, so the walk and the rewrite agree on its contents.
2191struct GroupParts {
2192    header_blocks: crate::io::object_header_io::HeaderBlocks,
2193    attributes: Vec<AttributeEntry>,
2194    links: Vec<(crate::format::messages::link::LinkMessage, Vec<u8>)>,
2195    track_order: TrackOrder,
2196    times: Option<ObjectTimes>,
2197    dense: DenseCarry,
2198    /// The symbol-table storage a classic group's header names — the blocks
2199    /// the rewrite supersedes. `None` for a link-message group, which has
2200    /// none. Its links are already in `links`: the walk turns each symbol
2201    /// table entry into the link message it stands for, so nothing downstream
2202    /// has to know which of the two forms the group was in.
2203    stab: Option<StabExtents>,
2204}
2205
2206/// The dense storage one reopened object's header names, which the rewrite of
2207/// that header stops naming and therefore has to free. Both halves are read
2208/// back before this is built — a heap that could not be read makes the object
2209/// [`ObjectPlan::Preserve`], so nothing here describes storage whose contents
2210/// were lost.
2211#[derive(Default)]
2212struct DenseCarry {
2213    attrs: Option<AttributeInfoMessage>,
2214    links: Option<LinkInfoMessage>,
2215}
2216
2217/// A modelled object, as the walk hands it to the registry rebuild. A group's
2218/// links are not here: the walk followed them, and each child is an entry of
2219/// its own.
2220enum CollectedObject {
2221    Dataset(Box<DatasetParts>),
2222    Group {
2223        header_blocks: crate::io::object_header_io::HeaderBlocks,
2224        attributes: Vec<AttributeEntry>,
2225        track_order: TrackOrder,
2226        times: Option<ObjectTimes>,
2227        dense: DenseCarry,
2228        stab: Option<StabExtents>,
2229    },
2230}
2231
2232/// The reopen's discovery pass: one walk that classifies every object it
2233/// reaches and descends into the groups among them.
2234///
2235/// Every object the close will touch is decided here and nowhere else, so
2236/// "modelled or preserved" is a property of the walk rather than of whatever
2237/// each later stage happened to be able to decode.
2238struct ReopenWalk<'a> {
2239    handle: &'a mut FileHandle,
2240    meta: &'a crate::io::FileMeta,
2241    out: CollectedLinks,
2242    /// Object headers already descended into, so hard-link cycles end.
2243    visited: std::collections::HashSet<u64>,
2244}
2245
2246impl<'a> ReopenWalk<'a> {
2247    fn new(handle: &'a mut FileHandle, meta: &'a crate::io::FileMeta) -> Self {
2248        Self {
2249            handle,
2250            meta,
2251            out: CollectedLinks::default(),
2252            visited: std::collections::HashSet::new(),
2253        }
2254    }
2255
2256    /// Everything the walk found.
2257    fn finish(self) -> CollectedLinks {
2258        self.out
2259    }
2260
2261    /// Decide what the reopen can do with the object at `addr`.
2262    ///
2263    /// The single gate: every object the rewrite touches is classified here,
2264    /// and an object is modelled only when each message the model consumes
2265    /// decoded. See [`ObjectPlan`] for why anything else must keep its bytes.
2266    fn plan(&mut self, addr: u64) -> IoResult<ObjectPlan> {
2267        let (handle, meta) = (&mut *self.handle, self.meta);
2268        let ctx = &meta.ctx;
2269        use crate::format::messages::data_layout::DataLayoutMessage;
2270        use crate::format::messages::dataspace::DataspaceMessage;
2271        use crate::format::messages::link::{CharacterSet, LinkMessage};
2272        use crate::format::messages::link_info::LinkInfoMessage;
2273        use crate::format::messages::shared::MSG_FLAG_SHARED;
2274        use crate::format::messages::{
2275            MSG_ATTRIBUTE, MSG_DATASPACE, MSG_DATATYPE, MSG_DATA_LAYOUT, MSG_EXTERNAL_FILE_LIST,
2276            MSG_FILL_VALUE, MSG_FILTER_PIPELINE, MSG_LINK, MSG_LINK_INFO, MSG_SYMBOL_TABLE,
2277        };
2278
2279        // The whole chain, messages and blocks alike: a filter pipeline or an
2280        // attribute that spilled into a continuation is one the rewrite would
2281        // otherwise drop, and a continuation block it does not know about is
2282        // one the rewrite would orphan.
2283        let (header, header_blocks) =
2284            match crate::io::object_header_io::read_object_header_with_blocks(handle, meta, addr) {
2285                Ok(h) => h,
2286                Err(e) => {
2287                    return Ok(ObjectPlan::preserve(format!(
2288                        "its object header chain does not read: {e}"
2289                    )))
2290                }
2291            };
2292
2293        // The policy, the times and the storage the header declares, read once
2294        // from the whole chain: all three are properties of the object, not of
2295        // any one message the loop below happens to reach.
2296        let track_order = recover_track_order(&header, ctx);
2297        let times = header.recorded_times();
2298        let (dense_attrs, dense_links) = superseded_dense(&header, ctx);
2299
2300        // Attributes come from the reader's collector rather than from the
2301        // loop below, so compact, dense and shared attributes all reach the
2302        // rewrite by the one path that knows how to read each of them. An
2303        // object whose set did not read whole is preserved: a short set here
2304        // would be a rewrite deleting the attributes it could not read.
2305        let attributes = match take_reopened_attributes(
2306            crate::io::reader::collect_object_attributes(handle, ctx, &header),
2307            &format!("the object at {addr:#x}"),
2308        ) {
2309            Ok(a) => a,
2310            Err(e) => {
2311                return Ok(ObjectPlan::preserve(format!(
2312                    "its attributes do not read back whole: {e}"
2313                )))
2314            }
2315        };
2316
2317        let mut datatype = None;
2318        let mut dataspace = None;
2319        let mut layout = None;
2320        let mut filter_pipeline = None;
2321        let mut fill_value = None;
2322        // No fill-value message at all is the library default, the same
2323        // convention the reader-side decode (`Hdf5Reader::dataset_info`)
2324        // uses for `fill_defined`.
2325        let mut fill_write_time: u8 = FILL_TIME_IFSET;
2326        let mut external = None;
2327        let mut links = Vec::new();
2328        let mut stab = None;
2329        // A datatype, dataspace or layout message says the object is not a
2330        // group, whether or not the three a dataset needs are all there.
2331        let mut dataset_shaped = false;
2332
2333        for msg in &header.messages {
2334            let consumed = matches!(
2335                msg.msg_type,
2336                MSG_DATATYPE
2337                    | MSG_DATASPACE
2338                    | MSG_DATA_LAYOUT
2339                    | MSG_FILTER_PIPELINE
2340                    | MSG_FILL_VALUE
2341                    | MSG_EXTERNAL_FILE_LIST
2342                    | MSG_ATTRIBUTE
2343                    | MSG_LINK
2344                    | MSG_LINK_INFO
2345                    | MSG_SYMBOL_TABLE
2346            );
2347            // A shared message holds a reference to where its body lives, not
2348            // the body. Decoding those bytes as one does not fail loudly — the
2349            // reference's version byte reads as a version and a class of its
2350            // own — so the guard is the only thing between a shared datatype
2351            // and a rewrite that invents a type for it.
2352            if consumed && msg.flags & MSG_FLAG_SHARED != 0 {
2353                return Ok(ObjectPlan::preserve(format!(
2354                    "its message of type {:#04x} is a shared-message reference, which this \
2355                     writer does not resolve",
2356                    msg.msg_type
2357                )));
2358            }
2359            macro_rules! consume {
2360                ($decode:expr, $what:literal) => {
2361                    match $decode {
2362                        Ok(v) => v,
2363                        Err(e) => {
2364                            return Ok(ObjectPlan::preserve(format!(
2365                                "its {} message does not decode: {e}",
2366                                $what
2367                            )))
2368                        }
2369                    }
2370                };
2371            }
2372            match msg.msg_type {
2373                // The pre-1.6 modification time, a formatted date string
2374                // (`H5O_MTIME`, type 0x0E). `recorded_times` reads only the
2375                // modern form, and a rewrite emits only that, so an object
2376                // carrying this one would come back out with the time it
2377                // recorded gone. Keeping its bytes is the same answer an
2378                // undecodable message already gets.
2379                crate::format::messages::MSG_MOD_TIME_OLD => {
2380                    return Ok(ObjectPlan::preserve(
2381                        "it carries a pre-1.6 modification time message, which this writer \
2382                         reads but does not write",
2383                    ))
2384                }
2385                MSG_DATATYPE => {
2386                    dataset_shaped = true;
2387                    let (dt, _) = consume!(DatatypeMessage::decode(&msg.data, ctx), "datatype");
2388                    datatype = Some(dt);
2389                }
2390                MSG_DATASPACE => {
2391                    dataset_shaped = true;
2392                    let version = msg.data.first().copied().unwrap_or(1);
2393                    let (ds, _) = consume!(DataspaceMessage::decode(&msg.data, ctx), "dataspace");
2394                    dataspace = Some((ds, version));
2395                }
2396                MSG_DATA_LAYOUT => {
2397                    dataset_shaped = true;
2398                    let (dl, _) =
2399                        consume!(DataLayoutMessage::decode(&msg.data, ctx), "data layout");
2400                    layout = Some(dl);
2401                }
2402                MSG_FILTER_PIPELINE => {
2403                    let (p, _) = consume!(FilterPipeline::decode(&msg.data), "filter pipeline");
2404                    if !p.filters.is_empty() {
2405                        filter_pipeline = Some(p);
2406                    }
2407                }
2408                MSG_FILL_VALUE => {
2409                    let (fv, _) = consume!(FillValueMessage::decode(&msg.data), "fill value");
2410                    if fv.fill_defined == 2 {
2411                        fill_value = fv.fill_value;
2412                    }
2413                    fill_write_time = fv.fill_write_time;
2414                }
2415                MSG_EXTERNAL_FILE_LIST => {
2416                    dataset_shaped = true;
2417                    let (efl, _) = consume!(
2418                        ExternalFileListMessage::decode(&msg.data, ctx),
2419                        "external file list"
2420                    );
2421                    // The names live in a local heap of their own, so the
2422                    // rewrite cannot re-emit the message from its bytes alone
2423                    // — it has to be able to point at the same strings. A heap
2424                    // that does not read back leaves the object preserved,
2425                    // which is what keeps its data reachable.
2426                    let resolved = match crate::io::reader::Hdf5Reader::resolve_external_file_slots(
2427                        handle, ctx, &efl,
2428                    ) {
2429                        Ok(r) => r,
2430                        Err(e) => {
2431                            return Ok(ObjectPlan::preserve(format!(
2432                                "its external file list names do not read back: {e}"
2433                            )))
2434                        }
2435                    };
2436                    external = Some(ExternalStorage {
2437                        heap_addr: efl.heap_addr,
2438                        // `H5Fopen` opens no dataset, so nothing has read a
2439                        // dapl for this one yet; the first handle it hands
2440                        // out settles the prefix.
2441                        prefix: EfilePrefix::default(),
2442                        files: efl
2443                            .slots
2444                            .iter()
2445                            .zip(resolved)
2446                            .map(|(slot, seg)| ExternalFile {
2447                                name: seg.name,
2448                                name_offset: slot.name_offset,
2449                                offset: slot.offset,
2450                                size: slot.size,
2451                            })
2452                            .collect(),
2453                    });
2454                }
2455                MSG_LINK => {
2456                    let (l, _) = consume!(LinkMessage::decode(&msg.data, ctx), "link");
2457                    links.push((l, msg.data.clone()));
2458                }
2459                MSG_LINK_INFO => {
2460                    let (li, _) = consume!(LinkInfoMessage::decode(&msg.data, ctx), "link info");
2461                    // Once a group holds enough links libhdf5 moves them into
2462                    // the fractal heap this message names and writes no `Link`
2463                    // messages at all. Reading them back is what makes the
2464                    // rewrite emit the group with its children; a rewrite from
2465                    // the header messages alone emitted it empty, orphaning
2466                    // every object below it.
2467                    if li.fractal_heap_address != UNDEF_ADDR {
2468                        let dense = match crate::io::reader::Hdf5Reader::read_dense_links(
2469                            handle,
2470                            ctx,
2471                            li.fractal_heap_address,
2472                        ) {
2473                            Ok(l) => l,
2474                            Err(e) => {
2475                                return Ok(ObjectPlan::preserve(format!(
2476                                    "its dense link storage does not read: {e}"
2477                                )))
2478                            }
2479                        };
2480                        // Re-encoded rather than carried as bytes: a heap
2481                        // object is not a header message, so there are no
2482                        // message bytes to carry. The encoding round-trips
2483                        // through the same decoder that just read it.
2484                        links.extend(dense.into_iter().map(|l| {
2485                            let bytes = l.encode(ctx);
2486                            (l, bytes)
2487                        }));
2488                    }
2489                }
2490                MSG_SYMBOL_TABLE => {
2491                    // A classic group keeps no link message at all: its links
2492                    // are symbol table entries in the B-tree this message
2493                    // names. Turning each into the link message it stands for
2494                    // is what lets the rest of the reopen — the walk, the
2495                    // registry, the preserve path — work on one link model
2496                    // whichever form the group is in.
2497                    let Some(s) = Stab::decode(&msg.data, ctx) else {
2498                        return Ok(ObjectPlan::preserve(
2499                            "its symbol table message is shorter than the two addresses it \
2500                             must carry",
2501                        ));
2502                    };
2503                    let contents = match crate::io::symbol_table_io::read_stab(handle, meta, s) {
2504                        Ok(c) => c,
2505                        Err(e) => {
2506                            return Ok(ObjectPlan::preserve(format!(
2507                                "its symbol table does not read: {e}"
2508                            )))
2509                        }
2510                    };
2511                    stab = Some(contents.extents);
2512                    links.extend(contents.links.into_iter().map(|l| {
2513                        let msg = match l.target {
2514                            StabTarget::Hard { addr, .. } => LinkMessage::hard(&l.name, addr),
2515                            StabTarget::Soft { value } => LinkMessage::soft(&l.name, &value),
2516                        };
2517                        // An entry carries no character set field, so the link
2518                        // it stands for has the file default whatever its name
2519                        // looks like (`H5G__ent_to_link`, H5Gent.c:372).
2520                        // Deriving one from the name would take a group
2521                        // libhdf5 wrote with a high-byte ASCII name out of its
2522                        // symbol table on the rewrite.
2523                        let msg = msg.with_cset(CharacterSet::Ascii);
2524                        let bytes = msg.encode(ctx);
2525                        (msg, bytes)
2526                    }));
2527                }
2528                _ => {}
2529            }
2530        }
2531
2532        // libhdf5 refuses a layout that disagrees with its sibling dataspace
2533        // and datatype as the dataset opens (`H5O__layout_decode` for the
2534        // chunk rank, `H5D__compact_init` for the compact size); modelled
2535        // anyway, the disagreement would be read at the wrong rank or past
2536        // the compact payload, so the dataset keeps its bytes, exactly as
2537        // unreadable as the file already had it.
2538        if let (Some((ds, _)), Some(dt), Some(dl)) = (&dataspace, &datatype, &layout) {
2539            if let Err(e) = dl.check_against_dataset(ds, dt, ctx) {
2540                return Ok(ObjectPlan::preserve(format!(
2541                    "its layout doesn't fit its dataspace and datatype: {e}"
2542                )));
2543            }
2544        }
2545
2546        match (datatype, dataspace, layout) {
2547            // A layout `rebuild_dataset` has no arm for leaves the registry
2548            // entry with an undefined data address, and the close then rewrites
2549            // the header as a contiguous, unallocated dataset — every element
2550            // gone, silently. Only the layouts that rebuild are modelled; the
2551            // rest keep their bytes, as an undecodable message already does.
2552            // The virtual layout is this.
2553            (Some(_), Some(_), Some(layout)) if !layout_rebuilds(&layout) => {
2554                Ok(ObjectPlan::preserve(format!(
2555                    "its data layout is {}, which this writer reads but does not build",
2556                    layout.describe()
2557                )))
2558            }
2559            (Some(datatype), Some((dataspace, dataspace_version)), Some(layout)) => {
2560                // Asked of the raw chain, not of `header`: the read above has
2561                // already put the named type's message in place of the pointer.
2562                let committed_type = match crate::io::object_header_io::committed_datatype_address(
2563                    handle, meta, addr,
2564                ) {
2565                    Ok(c) => c,
2566                    Err(e) => {
2567                        return Ok(ObjectPlan::preserve(format!(
2568                            "its shared datatype pointer does not decode: {e}"
2569                        )))
2570                    }
2571                };
2572                Ok(ObjectPlan::Dataset(Box::new(DatasetParts {
2573                    header_blocks,
2574                    datatype,
2575                    committed_type,
2576                    dataspace,
2577                    read_format: if dataspace_version <= 1 {
2578                        ObjectFormat::Legacy
2579                    } else {
2580                        ObjectFormat::Modern
2581                    },
2582                    layout,
2583                    filter_pipeline,
2584                    fill_value,
2585                    fill_write_time,
2586                    attributes,
2587                    track_order,
2588                    times,
2589                    dense: DenseCarry {
2590                        attrs: dense_attrs,
2591                        links: dense_links,
2592                    },
2593                    external,
2594                })))
2595            }
2596            // A committed (named) datatype has a datatype message and neither
2597            // of the other two; so does a dataset whose header this crate only
2598            // half understands. Neither is a group, and modelling either as
2599            // one is what rewrote them into empty groups. They part company
2600            // here and nowhere else: the datatype is kept by its bytes like
2601            // the other, but a listing can still name it.
2602            _ if crate::io::reader::header_is_committed_datatype(&header) => {
2603                Ok(ObjectPlan::Preserve {
2604                    why: "it is a committed (named) datatype, which this writer carries by \
2605                          its bytes rather than re-encoding"
2606                        .into(),
2607                    kind: PreservedKind::NamedDatatype,
2608                })
2609            }
2610            _ if dataset_shaped => Ok(ObjectPlan::preserve(
2611                "it carries a datatype, dataspace or layout message but not the three a \
2612                 dataset is built from; this writer models only groups and datasets",
2613            )),
2614            _ => Ok(ObjectPlan::Group(GroupParts {
2615                header_blocks,
2616                attributes,
2617                links,
2618                track_order,
2619                times,
2620                dense: DenseCarry {
2621                    attrs: dense_attrs,
2622                    links: dense_links,
2623                },
2624                stab,
2625            })),
2626        }
2627    }
2628
2629    /// Walk `links` (one group's, already decoded), classifying every object
2630    /// they name and descending into the groups among them.
2631    fn group(
2632        &mut self,
2633        links: &[(crate::format::messages::link::LinkMessage, Vec<u8>)],
2634        prefix: &str,
2635        depth: usize,
2636    ) -> IoResult<()> {
2637        // Bound nesting depth so a pathologically deep group chain cannot
2638        // overflow the stack (the `visited` set bounds total work but not
2639        // recursion depth).
2640        if depth > 256 {
2641            return Ok(());
2642        }
2643        use crate::format::messages::link::LinkTarget;
2644        for (link, encoded) in links {
2645            let full_name = if prefix.is_empty() {
2646                link.name.clone()
2647            } else {
2648                format!("{}/{}", prefix, link.name)
2649            };
2650
2651            // Only a hard link names an object this writer can rebuild. Every
2652            // other class is kept by its bytes, because a close that emitted
2653            // only what the registry models would drop it from the file.
2654            let LinkTarget::Hard { address } = &link.target else {
2655                self.out.preserved.push(PreservedEntry {
2656                    path: full_name,
2657                    class: crate::io::reader::LinkClass::from_target(&link.target),
2658                    encoded: encoded.clone(),
2659                    reason: None,
2660                    kind: PreservedKind::Unclassified,
2661                });
2662                continue;
2663            };
2664            let entry = HardEntry {
2665                path: full_name.clone(),
2666                address: *address,
2667                encoded: encoded.clone(),
2668            };
2669
2670            match self.plan(*address)? {
2671                // Kept by its bytes, exactly as a link class this writer
2672                // cannot express is: writing the link back unchanged is what
2673                // leaves the object's header where the file already has it.
2674                ObjectPlan::Preserve { why, kind } => self.out.preserved.push(PreservedEntry {
2675                    path: full_name,
2676                    class: crate::io::reader::LinkClass::Hard,
2677                    encoded: entry.encoded,
2678                    reason: Some(why),
2679                    kind,
2680                }),
2681                ObjectPlan::Dataset(parts) => {
2682                    self.out.hard.push((entry, CollectedObject::Dataset(parts)));
2683                }
2684                ObjectPlan::Group(parts) => {
2685                    self.out.hard.push((
2686                        entry,
2687                        CollectedObject::Group {
2688                            header_blocks: parts.header_blocks,
2689                            attributes: parts.attributes,
2690                            track_order: parts.track_order,
2691                            times: parts.times,
2692                            dense: parts.dense,
2693                            stab: parts.stab,
2694                        },
2695                    ));
2696                    // Recurse only into a group's header we have not entered
2697                    // before — breaks hard-link cycles.
2698                    if self.visited.insert(*address) {
2699                        self.group(&parts.links, &full_name, depth + 1)?;
2700                    }
2701                }
2702            }
2703        }
2704        Ok(())
2705    }
2706}
2707
2708/// Rebuild one reopened dataset's in-memory registry entry, storage and
2709/// all, from the header messages the walk decoded.
2710///
2711/// Fails when the chunk index the file names does not read back. The
2712/// caller answers that by preserving the object rather than registering
2713/// a dataset whose index has forgotten where its chunks are: the close
2714/// rewrites what the registry holds, so an index rebuilt from the part of
2715/// it that decoded would strand every chunk it could not read.
2716fn rebuild_dataset(
2717    handle: &mut FileHandle,
2718    meta: &FileMeta,
2719    file_size: u64,
2720    name: String,
2721    obj_addr: u64,
2722    parts: DatasetParts,
2723) -> IoResult<DatasetInfo> {
2724    let ctx = &meta.ctx;
2725    let DatasetParts {
2726        header_blocks,
2727        datatype: dt,
2728        committed_type,
2729        dataspace: ds,
2730        read_format,
2731        layout: dl,
2732        filter_pipeline: fp,
2733        fill_value,
2734        fill_write_time,
2735        attributes: attrs,
2736        track_order,
2737        times,
2738        dense: _,
2739        external,
2740    } = parts;
2741
2742    let mut info = DatasetInfo {
2743        name,
2744        datatype: dt,
2745        // The named type's own object is preserved by its bytes, so the
2746        // address the walk read the pointer from is the address it will still
2747        // be at when this header is written back.
2748        committed_type: committed_type.map(CommittedTypeRef::Preserved),
2749        read_format: Some(read_format),
2750        external,
2751        virtual_storage: None,
2752        dataspace: ds,
2753        obj_header_addr: obj_addr,
2754        data_addr: UNDEF_ADDR,
2755        data_size: 0,
2756        compact: None,
2757        chunked: None,
2758        fixed_array: None,
2759        implicit: None,
2760        single_chunk: None,
2761        btree_v1: None,
2762        btree_v2: None,
2763        append: None,
2764        attributes: attrs,
2765        obj_header_written_addr: Some(obj_addr),
2766        obj_header_blocks: header_blocks,
2767        filter_pipeline: fp,
2768        deleted: false,
2769        extent_dirty: false,
2770        header_dirty: false,
2771        // Stamped by the caller once the whole link graph is registered: it
2772        // is the count of links reaching this object, which one dataset's
2773        // parts cannot see.
2774        nlink_written: 1,
2775        // Stamped by the caller, which knows the order the walk met each
2776        // object; the rebuild sees one dataset at a time.
2777        creation_seq: 0,
2778        track_attr_order: track_order.attrs,
2779        fill_value,
2780        fill_time: fill_write_time,
2781        // Preserve the on-disk layout version so finalize re-encodes
2782        // what it read: a v5 file reopened and appended to must not be
2783        // silently downgraded to v4 (the filtered indexes keep their
2784        // 8-byte size fields, which v4 readers would mis-derive).
2785        layout_version: match &dl {
2786            DataLayoutMessage::ChunkedV4 { version, .. } => *version,
2787            // The classic index has no version above its own: a version-3
2788            // message is the whole of `H5D__chunk_set_info`'s MAX below the
2789            // version-4 gate, and re-encoding it any higher would name an
2790            // index the message cannot carry.
2791            DataLayoutMessage::ChunkedV3 { .. } => LAYOUT_VERSION_DEFAULT,
2792            _ => 4,
2793        },
2794        times,
2795    };
2796
2797    // Reconstruct storage-specific metadata
2798    debug_assert!(
2799        layout_rebuilds(&dl),
2800        "ReopenWalk::plan must preserve a layout this has no arm for"
2801    );
2802    match &dl {
2803        DataLayoutMessage::Contiguous { address, size } => {
2804            info.data_addr = *address;
2805            info.data_size = *size;
2806        }
2807        // The image is the layout message, so the rebuild carries it out of
2808        // the header it came from: anything that makes this dataset's header
2809        // stale rewrites the layout message from `compact`, and a rebuild
2810        // that left it empty would rewrite the dataset as an unallocated
2811        // contiguous one — dropping every byte.
2812        DataLayoutMessage::Compact { data } => {
2813            info.compact = Some(data.clone());
2814        }
2815        // The classic chunk index, reconstructed into the same
2816        // `BtreeV1DatasetInfo` a chunked dataset *created* in this format
2817        // gets, so the one set of machinery — `build_tree`, the flush's block
2818        // pool, `write_chunk`, `extend_dataset`, the prune a delete runs —
2819        // drives a reopened dataset and a fresh one alike. `root_addr` is what
2820        // the layout message carries and stays undefined for a dataset whose
2821        // chunks were never written, exactly as libhdf5 leaves it.
2822        DataLayoutMessage::ChunkedV3 {
2823            chunk_dims,
2824            b_tree_address,
2825        } => {
2826            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2827            let mut walk = BtreeV1Walk::new(handle, ctx, &meta.btree, &real_chunk_dims, file_size);
2828            walk.descend(*b_tree_address, 0)?;
2829            let BtreeV1Walk {
2830                records,
2831                node_addrs,
2832                ..
2833            } = walk;
2834            let max_dims = info
2835                .dataspace
2836                .max_dims
2837                .clone()
2838                .unwrap_or_else(|| info.dataspace.dims.clone());
2839            info.btree_v1 = Some(BtreeV1DatasetInfo {
2840                chunk_dims: real_chunk_dims,
2841                max_dims,
2842                // The file's own "K" ranks, not this session's defaults: they
2843                // set every node's width, so a tree bulk-loaded under the
2844                // wrong ones would re-serialize over blocks of the wrong size.
2845                config: meta.btree,
2846                records,
2847                node_addrs,
2848                root_addr: *b_tree_address,
2849                chunks_written: 0,
2850            });
2851        }
2852        DataLayoutMessage::ChunkedV4 {
2853            chunk_dims,
2854            index_address,
2855            index_type,
2856            earray_params,
2857            single_chunk_filter,
2858            ..
2859        } => {
2860            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2861
2862            if *index_type == crate::format::messages::data_layout::ChunkIndexType::ExtensibleArray
2863            {
2864                if let Some(params) = earray_params {
2865                    let ep = EarrayParams {
2866                        max_nelmts_bits: params.max_nelmts_bits,
2867                        idx_blk_elmts: params.idx_blk_elmts,
2868                        sup_blk_min_data_ptrs: params.sup_blk_min_data_ptrs,
2869                        data_blk_min_elmts: params.data_blk_min_elmts,
2870                        max_dblk_page_nelmts_bits: params.max_dblk_page_nelmts_bits,
2871                    };
2872                    let ndblk_addrs = compute_ndblk_addrs(ep.sup_blk_min_data_ptrs)?;
2873                    let nsblk_addrs = compute_nsblk_addrs(
2874                        ep.idx_blk_elmts,
2875                        ep.data_blk_min_elmts,
2876                        ep.sup_blk_min_data_ptrs,
2877                        ep.max_nelmts_bits,
2878                    )?;
2879
2880                    // Read EA header
2881                    let hdr_buf = handle.read_at_most(*index_address, 256)?;
2882                    let ea_header = ExtensibleArrayHeader::decode(&hdr_buf, ctx)?;
2883
2884                    let is_filtered = ea_header.class_id
2885                        == crate::format::chunk_index::extensible_array::EA_CLS_FILT_CHUNK;
2886                    let chunk_size_len = if is_filtered {
2887                        ea_header.raw_elmt_size - ctx.sizeof_addr - 4
2888                    } else {
2889                        0
2890                    };
2891
2892                    // Read the EA index block. Filtered datasets
2893                    // store a `FilteredIndexBlock`; unfiltered ones a
2894                    // plain `ExtensibleArrayIndexBlock`. Both must be
2895                    // reconstructed so a reopened dataset can append
2896                    // (write_chunk consults whichever applies).
2897                    let ea_iblk_addr = ea_header.idx_blk_addr;
2898                    let (ea_iblk, filt_iblk) = if is_filtered {
2899                        let placeholder = ExtensibleArrayIndexBlock::new(
2900                            *index_address,
2901                            ep.idx_blk_elmts,
2902                            ndblk_addrs,
2903                            nsblk_addrs,
2904                        );
2905                        let fib = if ea_iblk_addr != UNDEF_ADDR {
2906                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2907                            FilteredIndexBlock::decode(
2908                                &iblk_buf,
2909                                ctx,
2910                                ep.idx_blk_elmts as usize,
2911                                ndblk_addrs,
2912                                nsblk_addrs,
2913                                chunk_size_len,
2914                            )?
2915                        } else {
2916                            FilteredIndexBlock::new(
2917                                *index_address,
2918                                ep.idx_blk_elmts,
2919                                ndblk_addrs,
2920                                nsblk_addrs,
2921                            )
2922                        };
2923                        (placeholder, Some(fib))
2924                    } else {
2925                        let eib = if ea_iblk_addr != UNDEF_ADDR {
2926                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2927                            ExtensibleArrayIndexBlock::decode(
2928                                &iblk_buf,
2929                                ctx,
2930                                ep.idx_blk_elmts as usize,
2931                                ndblk_addrs,
2932                                nsblk_addrs,
2933                            )?
2934                        } else {
2935                            ExtensibleArrayIndexBlock::new(
2936                                *index_address,
2937                                ep.idx_blk_elmts,
2938                                ndblk_addrs,
2939                                nsblk_addrs,
2940                            )
2941                        };
2942                        (eib, None)
2943                    };
2944
2945                    info.chunked = Some(ChunkedDatasetInfo {
2946                        chunk_dims: real_chunk_dims,
2947                        earray_params: ep,
2948                        ea_header_addr: *index_address,
2949                        ea_iblk_addr,
2950                        ea_header,
2951                        ea_iblk,
2952                        chunks_written: 0,
2953                        filt_iblk,
2954                        chunk_size_len,
2955                    });
2956                }
2957            } else if *index_type
2958                == crate::format::messages::data_layout::ChunkIndexType::FixedArray
2959            {
2960                // Read the FA header and data block back so a
2961                // reopened dataset is writable and deletable, not
2962                // re-link only — a placeholder made a delete free
2963                // just the header and leak every chunk plus the
2964                // index. Paged data blocks (any FA with more than
2965                // dblk_page_nelmts chunks, libhdf5 default 1024)
2966                // reconstruct through the same decode owner; only
2967                // pages the bitmap marks initialized are decoded.
2968                let hdr_buf = handle.read_at_most(*index_address, 256)?;
2969                let fa_header = FixedArrayHeader::decode(&hdr_buf, ctx)?;
2970                let is_filtered = fa_header.client_id == FA_CLIENT_FILT_CHUNK;
2971                let chunk_size_len = if is_filtered {
2972                    (fa_header.element_size as usize)
2973                        .checked_sub(ctx.sizeof_addr as usize + 4)
2974                        .ok_or_else(|| {
2975                            crate::io::IoError::InvalidState(
2976                                "fixed array filtered element_size too small".into(),
2977                            )
2978                        })?
2979                } else {
2980                    0
2981                };
2982                if fa_header.data_blk_addr != UNDEF_ADDR && chunk_size_len <= 8 {
2983                    let dblk_size = fixed_array_dblk_disk_size(ctx, &fa_header) as usize;
2984                    let dblk_buf = handle.read_at_most(fa_header.data_blk_addr, dblk_size)?;
2985                    let fa_dblk =
2986                        decode_fixed_array_dblk(ctx, &fa_header, &dblk_buf, chunk_size_len)?;
2987                    info.fixed_array = Some(FixedArrayDatasetInfo {
2988                        chunk_dims: real_chunk_dims,
2989                        fa_header_addr: *index_address,
2990                        fa_dblk_addr: fa_header.data_blk_addr,
2991                        fa_header,
2992                        fa_dblk,
2993                        // Chunks written this session, matching the
2994                        // EA reconstruction above.
2995                        chunks_written: 0,
2996                    });
2997                }
2998            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::BTreeV2 {
2999                use crate::format::chunk_index::btree_v2::{
3000                    Bt2Geometry, Bt2Header, BT2_TYPE_CHUNK_FILT, BT2_TYPE_CHUNK_UNFILT,
3001                };
3002
3003                // Walk the tree back into the in-memory index and
3004                // adopt its node blocks as the flush pool. The pool
3005                // re-serializes at the header's node_size, whatever
3006                // it is — libhdf5 sizes every node from
3007                // hdr->node_size (H5B2leaf.c, H5B2internal.c) — so
3008                // a foreign size reopens too. Only a record type
3009                // that is not a chunk record, or a node size below
3010                // the bulk loader's few-records-per-node floor
3011                // (the same bound creation enforces), stays
3012                // re-link only.
3013                let hdr_buf = handle.read_at_most(*index_address, 256)?;
3014                let bt2_hdr = Bt2Header::decode(&hdr_buf, ctx)?;
3015                let ndims = real_chunk_dims.len();
3016                let is_filt = match bt2_hdr.record_type {
3017                    BT2_TYPE_CHUNK_UNFILT => Some(false),
3018                    BT2_TYPE_CHUNK_FILT => Some(true),
3019                    _ => None,
3020                };
3021                if let (Some(is_filt), true) = (
3022                    is_filt,
3023                    bt2_hdr.node_size as usize >= 10 + 3 * bt2_hdr.record_size as usize,
3024                ) {
3025                    let mut index = if is_filt {
3026                        let csl = (bt2_hdr.record_size as usize)
3027                            .checked_sub(ctx.sizeof_addr as usize + 4 + ndims * 8)
3028                            .filter(|&c| c <= 8)
3029                            .ok_or_else(|| {
3030                                crate::io::IoError::InvalidState(
3031                                    "v2 B-tree filtered record size does not fit \
3032                                     its rank and address width"
3033                                        .into(),
3034                                )
3035                            })?;
3036                        Bt2ChunkIndex::new_filtered(ndims, csl as u8)
3037                    } else {
3038                        Bt2ChunkIndex::new_unfiltered(ndims)
3039                    };
3040                    // Re-serialize with the creator's parameters:
3041                    // node blocks keep their size and the rewritten
3042                    // header keeps its declared split/merge.
3043                    index.node_size = bt2_hdr.node_size;
3044                    index.split_percent = bt2_hdr.split_percent;
3045                    index.merge_percent = bt2_hdr.merge_percent;
3046                    let mut node_addrs = Vec::new();
3047                    if bt2_hdr.root_node_addr != UNDEF_ADDR && bt2_hdr.total_num_records > 0 {
3048                        let geo = Bt2Geometry::new(
3049                            bt2_hdr.node_size,
3050                            bt2_hdr.record_size,
3051                            bt2_hdr.depth,
3052                            ctx.sizeof_addr,
3053                        );
3054                        let mut walk =
3055                            Bt2Walk::new(handle, ctx, bt2_hdr.record_size, bt2_hdr.node_size, &geo);
3056                        walk.descend(
3057                            bt2_hdr.root_node_addr,
3058                            bt2_hdr.depth,
3059                            bt2_hdr.num_records_in_root,
3060                        )?;
3061                        node_addrs = walk.node_addrs;
3062                        let record_bytes = walk.records;
3063                        let total = if bt2_hdr.record_size > 0 {
3064                            record_bytes.len() / bt2_hdr.record_size as usize
3065                        } else {
3066                            0
3067                        };
3068                        if is_filt {
3069                            for r in Bt2ChunkIndex::decode_filtered_records(
3070                                &record_bytes,
3071                                total,
3072                                ndims,
3073                                bt2_hdr.record_size,
3074                                ctx,
3075                            )? {
3076                                index.insert_filtered(
3077                                    r.scaled_offsets,
3078                                    r.chunk_address,
3079                                    r.chunk_size,
3080                                    r.filter_mask,
3081                                );
3082                            }
3083                        } else {
3084                            for r in Bt2ChunkIndex::decode_unfiltered_records(
3085                                &record_bytes,
3086                                total,
3087                                ndims,
3088                                ctx,
3089                            )? {
3090                                index.insert(r.scaled_offsets, r.chunk_address);
3091                            }
3092                        }
3093                    }
3094                    info.btree_v2 = Some(Bt2DatasetInfo {
3095                        chunk_dims: real_chunk_dims,
3096                        bt2_header_addr: *index_address,
3097                        node_addrs,
3098                        index,
3099                        chunks_written: 0,
3100                    });
3101                }
3102            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::Implicit
3103            {
3104                // Nothing to read back: the index *is* the run of chunk space
3105                // at `index_address`, and its length is the chunk grid times
3106                // the chunk size. Reconstructing that length is what lets a
3107                // delete free the storage and a write address it — a rebuild
3108                // that left this empty would rewrite the dataset as an
3109                // unallocated contiguous one, dropping every byte.
3110                let mut nchunks: u64 = 1;
3111                for g in crate::io::chunk_grid::index_grid(
3112                    &info.dataspace.dims,
3113                    info.dataspace.max_dims.as_deref(),
3114                    &real_chunk_dims,
3115                )? {
3116                    nchunks = nchunks.checked_mul(g).ok_or_else(|| {
3117                        crate::io::IoError::InvalidState("chunk count overflows u64".into())
3118                    })?;
3119                }
3120                let data_size = nchunks
3121                    .checked_mul(chunk_dims.iter().product::<u64>())
3122                    .ok_or_else(|| {
3123                        crate::io::IoError::InvalidState(
3124                            "implicit chunk storage overflows u64".into(),
3125                        )
3126                    })?;
3127                info.implicit = Some(ImplicitDatasetInfo {
3128                    chunk_dims: real_chunk_dims,
3129                    data_addr: *index_address,
3130                    data_size,
3131                });
3132            } else if *index_type
3133                == crate::format::messages::data_layout::ChunkIndexType::SingleChunk
3134            {
3135                // No index structure to read back either: the one chunk's
3136                // address, and its stored size and filter mask if the
3137                // layout's filtered flag is set, are the whole of the
3138                // layout message. `chunk_dims` already includes the
3139                // trailing element-size dimension, so its product is the
3140                // chunk's unfiltered byte length directly (see `data_size`
3141                // in the Implicit arm above).
3142                let data_size = chunk_dims.iter().product::<u64>();
3143                let (nbytes, filter_mask) = match single_chunk_filter {
3144                    Some(scf) => (scf.nbytes, scf.filter_mask),
3145                    None => (data_size, 0),
3146                };
3147                info.single_chunk = Some(SingleChunkDatasetInfo {
3148                    chunk_dims: real_chunk_dims,
3149                    data_addr: *index_address,
3150                    data_size,
3151                    nbytes,
3152                    filter_mask,
3153                    chunks_written: 0,
3154                    // Whether this was created with early allocation isn't
3155                    // recoverable here: `fill_value` above is only the
3156                    // decoded fill bytes, not the fill-value message's
3157                    // `alloc_time` byte the layout was chosen under. A
3158                    // reopened dataset that later gets a header rewrite
3159                    // therefore reports incremental allocation regardless
3160                    // of how it was actually created — the same
3161                    // imprecision a reopened `fixed_array`/`btree_v2`
3162                    // dataset already has, for the same reason.
3163                    early_alloc: false,
3164                });
3165            }
3166        }
3167        // Unreachable by `layout_rebuilds`, which is the gate
3168        // `ReopenWalk::plan` consults before it ever calls this.
3169        _ => {}
3170    }
3171
3172    Ok(info)
3173}
3174
3175/// Write `data` at *dataset-relative* byte offset `skip` into an external file
3176/// list, walking slots by cumulative declared size exactly like
3177/// `H5D__efl_write` (H5Defl.c).
3178///
3179/// Each slot's file is opened create-if-missing and never truncated, so a
3180/// write touches only the byte range that slot owns. A write past the *total*
3181/// declared size of the list is an error, matching upstream's "write past
3182/// logical end of file" check.
3183fn write_external_file_bytes(
3184    files: &[ExternalFile],
3185    extfile_prefix: Option<&Path>,
3186    mut skip: u64,
3187    data: &[u8],
3188) -> IoResult<()> {
3189    // `H5D__efl_write`'s slot walk: an `H5O_EFL_UNLIMITED` slot matches every
3190    // remaining offset (`skip >= u64::MAX` is never true), so the search stops
3191    // there and the write below takes the whole rest of the data.
3192    let mut slot_idx = 0usize;
3193    while slot_idx < files.len() && skip >= files[slot_idx].size {
3194        skip -= files[slot_idx].size;
3195        slot_idx += 1;
3196    }
3197
3198    let mut written = 0usize;
3199    while written < data.len() {
3200        let Some(slot) = files.get(slot_idx) else {
3201            return Err(crate::io::IoError::InvalidState(
3202                "write past the logical end of the external file list".into(),
3203            ));
3204        };
3205        let full_path = crate::io::reader::combine_prefixed_path(extfile_prefix, &slot.name);
3206        let ext_handle = FileHandle::open_or_create_readwrite_with_locking(
3207            &full_path,
3208            crate::io::locking::FileLocking::Disabled,
3209        )
3210        .map_err(|e| {
3211            crate::io::IoError::InvalidState(format!(
3212                "unable to open external raw data file {} for writing: {e}",
3213                full_path.display()
3214            ))
3215        })?;
3216        let this_write = (slot.size - skip).min((data.len() - written) as u64) as usize;
3217        let at = slot.offset.checked_add(skip).ok_or_else(|| {
3218            crate::io::IoError::InvalidState(format!(
3219                "external file '{}' slot offset {} overflows {skip} bytes into the slot",
3220                slot.name, slot.offset
3221            ))
3222        })?;
3223        ext_handle.write_at(at, &data[written..written + this_write])?;
3224        // This handle is dropped at the end of the iteration, and `Drop` can
3225        // only print a flush failure. Empty the accumulator here instead, so a
3226        // full disk on an external raw-data file reaches the caller.
3227        ext_handle.flush()?;
3228
3229        written += this_write;
3230        skip = 0;
3231        slot_idx += 1;
3232    }
3233    Ok(())
3234}
3235
3236/// The directory the HDF5 file at `path` sits in — libhdf5's `H5F_t::extpath`,
3237/// which `H5D__build_file_prefix` expands `${ORIGIN}` to.
3238///
3239/// Canonicalized, so the value survives the process changing directory and so
3240/// a writer and a reader of the same file agree on it. Called once per open,
3241/// never per I/O, for exactly that reason.
3242fn source_dir_of(path: &Path) -> IoResult<PathBuf> {
3243    let canonical = std::fs::canonicalize(path)?;
3244    Ok(canonical
3245        .parent()
3246        .map(Path::to_path_buf)
3247        .unwrap_or_default())
3248}
3249
3250/// Whether [`rebuild_dataset`] has an arm that reconstructs this layout.
3251///
3252/// The single list: `ReopenWalk::plan` preserves an object whose layout this
3253/// says no to, so a layout added to one side and not the other cannot happen.
3254/// Keeping two lists is what would rewrite a modelled dataset as unallocated
3255/// contiguous storage, or preserve one the writer can now build.
3256fn layout_rebuilds(layout: &DataLayoutMessage) -> bool {
3257    matches!(
3258        layout,
3259        DataLayoutMessage::Contiguous { .. }
3260            | DataLayoutMessage::Compact { .. }
3261            | DataLayoutMessage::ChunkedV3 { .. }
3262            | DataLayoutMessage::ChunkedV4 { .. }
3263    )
3264}
3265
3266/// Encode an Object Reference Count message (type 0x16) body: a version
3267/// byte (`H5O_REFCOUNT_VERSION` = 0) followed by the little-endian u32
3268/// count. Emitted on objects reached by more than one hard link.
3269fn encode_refcount(refcount: u32) -> Vec<u8> {
3270    let mut v = Vec::with_capacity(5);
3271    v.push(0u8);
3272    v.extend_from_slice(&refcount.to_le_bytes());
3273    v
3274}
3275
3276/// The symbol-table storage of every group that has one, and the single owner
3277/// of which groups those are.
3278///
3279/// A group stores its links in a symbol table because the file was *made* that
3280/// way — `H5F_LIBVER_EARLIEST` is the one bound `H5G__obj_create_real`
3281/// (H5Gobj.c:179) writes them at — or because it already had one when the file
3282/// was reopened. The second is not the first: `H5G_obj_insert` inserts into
3283/// whatever storage the group is in and converts only when a link will not fit
3284/// an entry (H5Gobj.c:512), so a symbol table survives a reopen at any bound.
3285/// A file with shared messages is where the two come apart, because its
3286/// superblock extension forces a version-2 superblock over symbol-table groups
3287/// (H5Fsuper.c:1135) and `H5F__super_read` then raises the low bound to
3288/// `H5F_LIBVER_V18` on reopen — new objects are the modern generation while the
3289/// groups already there stay symbol tables.
3290struct SymbolTables {
3291    /// The scopes the reopen found a Symbol Table message on. Fixed for the
3292    /// session: a group already in that storage stays in it, whatever bound
3293    /// the objects added beside it are written at.
3294    found: HashSet<LinkScope>,
3295    /// The symbol-table storage each group's header already names, by the
3296    /// scope whose rewrite supersedes it.
3297    ///
3298    /// INVARIANT: every entry is freed exactly once, by
3299    /// [`Hdf5Writer::prepare_symbol_tables`], which removes it as it frees.
3300    superseded: Slot<HashMap<LinkScope, StabExtents>>,
3301    /// The storage that same pass laid out, read by the header builders.
3302    ///
3303    /// INVARIANT: an entry exists here only after every block of that group's
3304    /// heap and B-tree is on disk. `build_group_header` reads it and never
3305    /// builds — a header is sized and then written by two separate calls, so a
3306    /// build that allocated would allocate twice.
3307    written: Slot<HashMap<LinkScope, Stab>>,
3308}
3309
3310impl SymbolTables {
3311    /// What a file being created starts from: no group found in a symbol table
3312    /// because none was read, and nothing on disk to free.
3313    fn none_found() -> Self {
3314        Self {
3315            found: HashSet::new(),
3316            superseded: Slot::new(HashMap::new()),
3317            written: Slot::new(HashMap::new()),
3318        }
3319    }
3320}
3321
3322/// Everything a version-0/1 (symbol-table) file carries that a version-2/3 one
3323/// does not.
3324///
3325/// Its presence *is* the generation switch — [`Hdf5Writer::message_format`]
3326/// reads nothing else: libhdf5 at `H5F_LIBVER_EARLIEST` writes a version-0/1
3327/// superblock over version-1 object headers over symbol-table groups. Which
3328/// groups are symbol tables is the separate question [`SymbolTables`] answers,
3329/// because a reopen at a newer bound keeps the ones it finds.
3330///
3331/// Two things put one here, and only two: reopening a file that already is in
3332/// that format, and creating one at that bound
3333/// ([`LegacyFile::created`]). Neither is distinguished afterwards — a file is
3334/// classic or it is not, and every encoder asks only that.
3335struct LegacyFile {
3336    /// The superblock as it was read, or as [`LegacyFile::created`] built it.
3337    /// The close re-emits it with only the end of file and the root symbol
3338    /// table entry recomputed: the "K" ranks in particular are recorded
3339    /// nowhere else, and every node width in the file is derived from them.
3340    superblock: SuperblockV0V1,
3341}
3342
3343impl LegacyFile {
3344    /// The classic-format state a file created at `H5F_LIBVER_EARLIEST`
3345    /// starts from.
3346    ///
3347    /// A new file has no symbol table on disk to free and none laid out, so
3348    /// its [`SymbolTables`] starts empty and every group it makes takes that
3349    /// storage from the bound rather than from what was found.
3350    ///
3351    /// The superblock is the one `H5F__super_init` writes at that bound: the
3352    /// library-default "K" ranks (`H5F_CRT_SYM_LEAF_DEF`,
3353    /// `HDF5_BTREE_SNODE_IK_DEF`), no free-space info and no driver info. The
3354    /// root entry's object header address and cached symbol table are stamped
3355    /// in by [`Hdf5Writer::write_superblock`] once the root group has one;
3356    /// its name offset is the empty string at the front of every local heap.
3357    ///
3358    /// Version 0, not 1: a version-1 superblock exists only to carry a
3359    /// non-default chunked-storage "K" value (H5Fsuper.c:1150), and this
3360    /// writer has no property to set one.
3361    fn created(ctx: FormatContext, base_address: u64) -> Self {
3362        let btree = BTreeV1Config::default();
3363        Self {
3364            superblock: SuperblockV0V1 {
3365                version: SUPERBLOCK_V0,
3366                sizeof_offsets: ctx.sizeof_addr,
3367                sizeof_lengths: ctx.sizeof_size,
3368                file_consistency_flags: 0,
3369                sym_leaf_k: btree.sym_leaf_k,
3370                btree_internal_k: btree.snode_internal_k,
3371                indexed_storage_k: None,
3372                base_address,
3373                superblock_extension_address: UNDEF_ADDR,
3374                end_of_file_address: 0,
3375                driver_info_address: UNDEF_ADDR,
3376                root_symbol_table_entry: SymbolTableEntry {
3377                    name_offset: 0,
3378                    obj_header_addr: UNDEF_ADDR,
3379                    cache: SymbolTableCache::Nothing,
3380                },
3381            },
3382        }
3383    }
3384}
3385
3386/// The superblock extension a reopen found, and the single owner of the one
3387/// this file's close writes back.
3388///
3389/// The extension is external truth: it is where a file records the things its
3390/// superblock has no field for — non-default v1 B-tree "K" ranks, a driver's
3391/// settings, the file space strategy and its persisted free-space managers,
3392/// and the shared object header message table. `H5F__super_ext_write_msg`
3393/// modifies one message of it and leaves the rest alone, so a close that lays
3394/// a fresh extension out from what *this writer* models drops everything it
3395/// does not — and the K ranks are not decoration: a chunked dataset's version-1
3396/// B-tree nodes are sized from `chunk_internal_k`, so a reader that has lost
3397/// the message reads the tree at the default rank and fails outright.
3398///
3399/// INVARIANT: every message of the extension read is re-emitted by
3400/// [`Hdf5Writer::write_superblock_extension`], byte for byte, except the
3401/// shared-message table — the one message naming storage this session lays out
3402/// afresh, which [`SohmState`] recomputes. Nothing else here is interpreted,
3403/// so a message this crate does not model survives exactly as a modelled one
3404/// does.
3405struct CarriedExtension {
3406    /// Every block the extension header occupied — chunk 0 and each
3407    /// continuation it named — freed once the replacement is laid out. Empty
3408    /// for a file with no extension, and for one whose extension this session
3409    /// is the first to write. A rewrite re-encodes the whole chain into one
3410    /// chunk, so freeing only the first would leave the rest as space no
3411    /// free-space manager records and no object claims.
3412    superseded: crate::io::object_header_io::HeaderBlocks,
3413    /// Every message that header held — the shared-message table,
3414    /// continuations and null padding excepted. The first two are structure
3415    /// rather than content; the third is free space.
3416    carried: Vec<crate::io::object_header_io::ExtensionMessage>,
3417    /// Where [`Hdf5Writer::write_superblock_extension`] put the replacement,
3418    /// and the only value the superblock's extension address is read from.
3419    /// `None` until that pass runs, and for a file that needs no extension.
3420    addr: Slot<Option<u64>>,
3421}
3422
3423/// What a reopen learns from a file's free-space managers, split by who owns
3424/// it: the sections go to the allocator and the rest stays with the writer.
3425struct ReopenedFreeSpace {
3426    /// `None` for a file this writer records no free space for.
3427    state: Option<Box<FileSpaceState>>,
3428    /// Every section the managers held, each tagged with the manager it came
3429    /// out of and merged only within it, address-ordered. Empty whenever
3430    /// `state` is `None`.
3431    sections: Vec<FreeBlock>,
3432}
3433
3434/// The file-space info message this session is responsible for, and the
3435/// manager blocks it supersedes.
3436///
3437/// A file whose message says `persist` records the space its own edits
3438/// released in one free-space manager per allocation type: a header block
3439/// (`FSHD`) naming a sections block (`FSSE`) that lists every free region.
3440/// Nothing else in the file says those regions are free, so a session that
3441/// rewrites the file without reading them either leaks the space it frees or
3442/// hands out space a manager still claims.
3443///
3444/// Present for a file this writer *created* with non-default file-space
3445/// properties as well, where there is nothing to read and the message is this
3446/// session's to write. `None` — the field, not this struct — is the third
3447/// case: a reopened file whose message this session must not touch, which the
3448/// carried extension re-emits byte for byte.
3449///
3450/// INVARIANT: the sections read are handed to [`FileAllocator`] and tracked
3451/// there alone, so there is one account of the file's free space and not two.
3452/// What stays here is only what the allocator has no place for: the message to
3453/// write, and the managers' own blocks, which are not free space until the
3454/// close that replaces them frees them.
3455struct FileSpaceState {
3456    /// The message, as read or as the creation options declared it. It is the
3457    /// only place the manager addresses are recorded, so the close that moves
3458    /// them rewrites this message.
3459    info: FileSpaceInfoMessage,
3460    /// The manager blocks themselves — one header, and one sections block per
3461    /// manager that had any sections. Freed by the close that lays their
3462    /// replacements out, the rule every other superseded structure follows.
3463    /// Empty for a created file, which supersedes nothing.
3464    superseded: Vec<(u64, u64)>,
3465}
3466
3467impl FileSpaceState {
3468    /// Whether this file keeps free-space managers on disk. Both strategies
3469    /// that have managers do — paged aggregation has the same managers plus a
3470    /// large one — while the two aggregator-only strategies and
3471    /// `persist: false` still carry the message with nothing to write into it.
3472    fn records_free_space(&self) -> bool {
3473        self.info.persist
3474            && matches!(
3475                self.info.strategy,
3476                FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
3477            )
3478    }
3479}
3480
3481/// One free-space manager that has been given its own two blocks, and the
3482/// sections it will write into them.
3483///
3484/// Produced by
3485/// [`settle_free_space_managers`](Hdf5Writer::settle_free_space_managers).
3486/// Both blocks are ordinary allocations out of the same [`FileAllocator`] the
3487/// rest of the file uses, because upstream's are too:
3488/// `H5FS_vfd_alloc_hdr_and_section_info_if_needed` calls `H5MF_alloc`
3489/// (H5FSsection.c:2352, 2406).
3490struct PlacedManager {
3491    /// Which of the file's managers this is; its message slot names it in the
3492    /// file-space info message.
3493    manager: FreeSpaceManager,
3494    /// Header block address.
3495    hdr_addr: u64,
3496    /// Sections block address.
3497    sect_addr: u64,
3498    /// Bytes the sections block occupies. What the header records as both
3499    /// `sect_size` and `alloc_sect_size`, so an image shorter than the block
3500    /// is padded rather than reported short.
3501    sect_size: u64,
3502    /// The sections this manager records, in serialization order. Filled on
3503    /// the settling round, once no allocation can change them.
3504    sections: Vec<FreeSection>,
3505}
3506
3507/// The manager header for `sections`, before its own blocks have addresses.
3508///
3509/// Every width the section encoding uses comes from here, and the only one
3510/// that varies with the content is `serial_sections` — it decides how many
3511/// bytes a per-size run count takes — so sizing a layout and encoding it must
3512/// go through this one function or the two disagree.
3513fn manager_header(sections: &[FreeSection]) -> FreeSpaceHeader {
3514    FreeSpaceHeader {
3515        client: free_space::CLIENT_FILE,
3516        total_space: sections.iter().map(|s| s.len).sum(),
3517        total_sections: sections.len() as u64,
3518        // Every class the file client registers is serializable; only a
3519        // fractal heap's manager has ghost sections.
3520        serial_sections: sections.len() as u64,
3521        ghost_sections: 0,
3522        nclasses: free_space::FILE_SECT_CLASSES,
3523        shrink_percent: free_space::SHRINK_PERCENT,
3524        expand_percent: free_space::EXPAND_PERCENT,
3525        max_sect_addr: free_space::SEC2_MAX_SECT_ADDR,
3526        max_sect_size: free_space::SEC2_MAXADDR,
3527        sect_addr: UNDEF_ADDR,
3528        sect_size: 0,
3529        alloc_sect_size: 0,
3530    }
3531}
3532
3533impl Default for CarriedExtension {
3534    /// What a file with no extension carries: nothing to free, nothing to
3535    /// re-emit, and no address until a shared-message table gives it one.
3536    fn default() -> Self {
3537        Self {
3538            superseded: Vec::new(),
3539            carried: Vec::new(),
3540            addr: Slot::new(None),
3541        }
3542    }
3543}
3544
3545/// Where a file's superblock version comes from — the two cases libhdf5 keeps
3546/// strictly apart, and this writer's single source for both the version it
3547/// writes back and the generation it writes new structures in.
3548///
3549/// INVARIANT: reopening a file never changes its superblock version, and every
3550/// structure appended to it is written at a library-version bound of at least
3551/// the row that version belongs to.
3552///
3553/// libhdf5 splits the same way. `H5F__super_init` is the only place a version
3554/// is *decided* — content first, then `MAX(super_vers,
3555/// HDF5_superblock_ver_bounds[low_bound])` (H5Fsuper.c:1128-1154).
3556/// `H5F__super_read` never recomputes one; it validates what it read and
3557/// raises the file's low bound to match, version 2 to at least
3558/// `H5F_LIBVER_V18` and version 3 to at least `H5F_LIBVER_V110`
3559/// (hdf5_1.14.6 H5Fsuper.c:460-466). One direction only: the version bounds
3560/// the structures, the structures never bound the version back.
3561///
3562/// Two variants rather than one number with a rule attached, because the
3563/// number means different things on the two paths — a floor to raise on the
3564/// create path, a fixed value on the reopen path — and a single field would
3565/// have every reader re-derive which.
3566#[derive(Debug, Clone, Copy)]
3567enum SuperblockVersion {
3568    /// A file this writer created. The version its creation options start
3569    /// from, which [`superblock_version_for`](Hdf5Writer::superblock_version_for)
3570    /// raises to what the content and the named bound need. Nothing is on
3571    /// disk yet, so nothing floors the bound.
3572    Chosen(u8),
3573    /// A file this writer reopened: the version already in the file. Written
3574    /// back unchanged, and the floor under every bound this session writes at.
3575    Existing(u8),
3576}
3577
3578impl SuperblockVersion {
3579    /// The oldest library-version bound this file may be written at.
3580    ///
3581    /// `H5F__super_read`'s upgrade, as a table rather than two `if`s: the
3582    /// oldest row of `HDF5_superblock_ver_bounds` (H5Fsuper.c:68) whose entry
3583    /// is the version on disk. A created file has no superblock on disk, so
3584    /// its floor is the oldest bound there is.
3585    ///
3586    /// `Existing(0..=1)` and `Hdf5Writer::legacy` say the same thing from two
3587    /// directions and cannot disagree: `open_append_with_locking` builds the
3588    /// `LegacyFile` from exactly the versions this arm covers.
3589    fn libver_floor(self) -> LibverBound {
3590        match self {
3591            Self::Chosen(_) => LibverBound::Earliest,
3592            Self::Existing(0..=1) => LibverBound::Earliest,
3593            Self::Existing(2) => LibverBound::V18,
3594            Self::Existing(_) => LibverBound::V110,
3595        }
3596    }
3597}
3598
3599/// A registry entry that has held some name.
3600///
3601/// Datasets, groups and committed datatypes keep stable indices — their
3602/// registries only grow, deletion being a flag — so the index can name the
3603/// exact entry. The link registries shrink as links are unlinked, and a
3604/// link's path is derived from its parent group's current name, so for those
3605/// the index records only that the kind once claimed the name and the (short)
3606/// list itself answers.
3607#[derive(Clone, Copy, PartialEq, Eq)]
3608enum NameHit {
3609    Dataset(usize),
3610    Group(usize),
3611    Datatype(usize),
3612    HardLink,
3613    SymbolicLink,
3614    PreservedLink,
3615}
3616
3617/// Which names the file model already holds, so creating an object does not
3618/// have to walk every registry to find out.
3619///
3620/// INVARIANT: while `map` is `Some`, every name a registry entry currently
3621/// holds has an entry in `map` covering that entry. The converse is not
3622/// required: a hit whose object was since deleted, or whose name has since
3623/// changed, stays in the map and is filtered out by
3624/// [`Hdf5Writer::name_holder`], which re-runs the very predicates the linear
3625/// scan used. The index may therefore answer "maybe", never "free" for a name
3626/// that is taken.
3627///
3628/// MUST NOT: no code may give a registry entry a name, or move the path a
3629/// link is emitted under, without either registering the new name through
3630/// [`Hdf5Writer::register_name`] or dropping the index through
3631/// [`Hdf5Writer::forget_name_index`]. State a constructor puts straight into
3632/// the registries needs neither — `map` starts `None`, and the first query
3633/// builds it from the registries as they then stand.
3634struct NameIndex {
3635    map: Option<HashMap<String, Vec<NameHit>>>,
3636    /// Bumped whenever the registries move under a build in flight, so that
3637    /// build's result is discarded instead of being installed stale.
3638    epoch: u64,
3639}
3640
3641impl NameIndex {
3642    fn new() -> Self {
3643        NameIndex {
3644            map: None,
3645            epoch: 0,
3646        }
3647    }
3648
3649    /// Record that `hit` holds `name`. With no map built there is nothing to
3650    /// record, but the registries have moved, so any build in flight is
3651    /// invalidated rather than trusted.
3652    fn insert(&mut self, name: &str, hit: NameHit) {
3653        match self.map.as_mut() {
3654            None => self.epoch += 1,
3655            Some(map) => {
3656                let hits = map.entry(name.to_string()).or_default();
3657                if !hits.contains(&hit) {
3658                    hits.push(hit);
3659                }
3660            }
3661        }
3662    }
3663
3664    /// Throw the index away: the next query rebuilds it from the registries.
3665    fn forget(&mut self) {
3666        self.map = None;
3667        self.epoch += 1;
3668    }
3669}
3670
3671/// HDF5 file writer.
3672///
3673/// Usage:
3674/// 1. `Hdf5Writer::create(path)` to create a new file.
3675/// 2. `create_dataset(name, datatype, dims)` to define datasets.
3676/// 3. `write_dataset_raw(index, data)` to write raw data.
3677/// 4. `close()` to finalize the file (writes superblock, headers, etc.).
3678pub struct Hdf5Writer {
3679    handle: FileHandle,
3680    allocator: FileAllocator,
3681    ctx: FormatContext,
3682    /// Dataset registry. The outer [`Slot`] guards the spine (push on create,
3683    /// index/clone on access) and is held only briefly; each [`DatasetRef`]
3684    /// carries one dataset's metadata behind its own lock. A writer clones
3685    /// the `DatasetRef` out (releasing this lock) before doing the long
3686    /// per-dataset work, so a create never blocks an in-flight write.
3687    pub(crate) datasets: Slot<Vec<DatasetRef>>,
3688    /// Group registry, same shape as [`Self::datasets`].
3689    pub(crate) groups: Slot<Vec<GroupRef>>,
3690    /// User-created hard links (additional names for existing objects),
3691    /// resolved and emitted during finalize.
3692    pub(crate) hard_links: Slot<Vec<HardLink>>,
3693    /// User-created soft and external links. Held apart from
3694    /// [`Self::hard_links`] because they name a path rather than an object:
3695    /// nothing resolves them, and no object's reference count counts them.
3696    pub(crate) symbolic_links: Slot<Vec<SymbolicLink>>,
3697    /// Datatypes committed this session, each an object of its own; see
3698    /// [`CommittedDatatype`].
3699    pub(crate) committed_datatypes: Slot<Vec<CommittedDatatype>>,
3700    /// Links a reopened file held that this writer cannot express, carried
3701    /// through every header rewrite by their encoded bytes. Always empty for
3702    /// a freshly created file; see [`PreservedLink`].
3703    pub(crate) preserved_links: Slot<Vec<PreservedLink>>,
3704    /// Which names the registries above already hold; see [`NameIndex`].
3705    /// Boxed so this side table costs the writer one pointer: inline, its
3706    /// map shifted every field after it and cost the attribute path ~5%.
3707    name_index: Slot<Box<NameIndex>>,
3708    /// Attributes attached to the root group (file-level attributes).
3709    pub(crate) root_attributes: Slot<Vec<crate::format::messages::attribute::AttributeEntry>>,
3710    /// Serializes object creation so name-uniqueness check and registry insert
3711    /// happen atomically.
3712    ///
3713    /// INVARIANT: no two emitted links share a full-path name. Under
3714    /// `threadsafe`, create methods run on the shared read guard, so without
3715    /// this gate two threads could both pass the duplicate-name check (which
3716    /// snapshots a registry and drops its lock) and both push, writing an
3717    /// invalid HDF5 file with two same-named links. A create holds this lock
3718    /// across its check *and* its push; the streaming write path never takes
3719    /// it, so writes to existing datasets stay fully concurrent. It is the
3720    /// outermost lock a create acquires (create_lock → spine → slot), and no
3721    /// write path takes it, so it cannot deadlock with the registry locks.
3722    pub(crate) create_lock: Slot<()>,
3723    /// The low `H5Pset_libver_bounds` bound the *caller named*, or `None`
3724    /// when none was: the oldest libhdf5 a file this writer creates must stay
3725    /// readable by. It is the one switch the version-bearing messages read —
3726    /// the datatype message version (`H5O_dtype_ver_bounds`), the data layout
3727    /// message version (`H5O_layout_ver_bounds`) and with it the chunk index,
3728    /// and the superblock floor (`HDF5_superblock_ver_bounds`).
3729    ///
3730    /// `None` is not `Some(Earliest)`. No single libhdf5 bound describes this
3731    /// crate's default file: it takes the earliest row of the datatype and
3732    /// superblock tables (version-1 datatypes, a version-2 superblock raised
3733    /// to 3 only by what the content needs) over the v1.10 chunk indexes,
3734    /// which is the `H5F_LIBVER_V110` row of the layout table. Naming a bound
3735    /// asks for one whole libhdf5 generation instead, so the two cannot share
3736    /// a field.
3737    ///
3738    /// Nothing reads this directly:
3739    /// [`session_libver`](Hdf5Writer::session_libver) is the only reader, and
3740    /// it is where the default meets the floor the file's own superblock puts
3741    /// under it (see [`SuperblockVersion`]). A default is a bound the *writer*
3742    /// picks, and on a reopened file the writer has no say — which is exactly
3743    /// the difference this field cannot express on its own.
3744    libver: Option<LibverBound>,
3745    closed: bool,
3746    /// Set once `finalize_for_swmr` has published a readable file.
3747    ///
3748    /// A SWMR reader may hold a chunk index that still points at a block this
3749    /// writer has since replaced, so from that point on a relocated chunk's
3750    /// old block is kept rather than released for reuse — the same rule as
3751    /// libhdf5's `H5D__chunk_file_alloc`, which skips `H5MF_xfree` under
3752    /// `H5F_ACC_SWMR_WRITE`.
3753    swmr_active: bool,
3754    /// Collections with free space — libhdf5's `f->shared->cwfs` list. A
3755    /// vlen insert fills these partially-filled collection blocks before
3756    /// creating a new one, so many small writes share 4096-byte blocks
3757    /// instead of each taking their own. Entries hold `(addr, block size,
3758    /// free bytes)` hints; the block on disk stays the single truth for
3759    /// contents, and only the two functions that rewrite collection blocks
3760    /// ([`insert_vlen_objects`](Self::insert_vlen_objects) and
3761    /// [`release_vlen_references`](Self::release_vlen_references)) may
3762    /// update this list. In-memory only, like the allocator's free list:
3763    /// a reopened file's free space is rediscovered as releases touch its
3764    /// collections. Capped at [`H5HG_NCWFS`] entries.
3765    cwfs: Slot<Vec<CwfsEntry>>,
3766    /// Address of the root group object header (set after first finalize).
3767    root_group_addr: Option<u64>,
3768    /// Size of the encoded root group object header (for in-place rewrites).
3769    /// The on-disk root header block a reopen found, `(addr, len)`, so
3770    /// finalize can free the block its rewrite supersedes.
3771    superseded_root_header: crate::io::object_header_io::HeaderBlocks,
3772    /// Where this file's superblock version comes from. The single owner of
3773    /// both halves of the reopen invariant — see [`SuperblockVersion`],
3774    /// [`superblock_version_for`](Self::superblock_version_for) and
3775    /// [`libver_floor`](Self::libver_floor).
3776    superblock_version: SuperblockVersion,
3777    /// Objects whose attributes this finalize spilled to dense storage, and
3778    /// the `Attribute Info` message naming what was written for each.
3779    ///
3780    /// INVARIANT: an entry exists here only after every block of that
3781    /// object's heap and name index is on disk, and only
3782    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes) may add
3783    /// one. `emit_attributes` reads it and never builds — a header is sized
3784    /// and then written by two separate `build_*_header` calls, so a build
3785    /// that allocated would allocate twice and leave the sized-for blocks
3786    /// stranded.
3787    dense_attributes: Slot<HashMap<AttrScope, AttributeInfoMessage>>,
3788    /// Groups whose links this finalize spilled to dense storage, and the
3789    /// `Link Info` message naming what was written for each.
3790    ///
3791    /// INVARIANT: an entry exists here only after every block of that group's
3792    /// heap and name index is on disk, and only
3793    /// [`prepare_dense_links`](Self::prepare_dense_links) may add one.
3794    dense_links: Slot<HashMap<LinkScope, LinkInfoMessage>>,
3795    /// The dense storage the reopened object headers already name — the heaps
3796    /// and indices this session's rewrites and deletes supersede.
3797    ///
3798    /// `None` for a file this session created: every block such a file will
3799    /// hold was allocated here, so there is nothing on disk to supersede and
3800    /// nothing to allocate for the bookkeeping either.
3801    ///
3802    /// INVARIANT: every entry is freed exactly once, by
3803    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs)
3804    /// or [`release_superseded_dense_links`](Self::release_superseded_dense_links),
3805    /// which remove it as they free. Nothing else may remove one: an entry
3806    /// that leaves without reaching the allocator is a leaked heap, and one
3807    /// that reaches it twice hands the same blocks to two objects.
3808    superseded_dense: Slot<Option<Box<SupersededDense>>>,
3809    /// The creation-order policy in force: whether an object created from
3810    /// now on records creation order for its links and its attributes. The
3811    /// h5py `track_order` analogue; see
3812    /// [`set_track_order`](Self::set_track_order). Each object captures this
3813    /// at creation, so changing it never rewrites an object already made.
3814    track_order: TrackOrder,
3815    /// Whether an object created from now on records the times its header can
3816    /// hold — `H5Pset_obj_track_times`, whose default is on
3817    /// (`H5O_CRT_OHDR_FLAGS_DEF` is `H5O_HDR_STORE_TIMES`, H5Opkg.h:74).
3818    /// Captured by each object at creation for the same reason
3819    /// [`track_order`](Self::track_order) is: it belongs to the creation
3820    /// property list, so a later change must not rewrite an object already
3821    /// made.
3822    track_times: bool,
3823    /// The root group's own captured policy. The root is created with the
3824    /// file, so its value comes from
3825    /// [`create_with_options`](Self::create_with_options) — or, on reopen,
3826    /// from the header already on disk.
3827    root_track_order: TrackOrder,
3828    /// The root group's stored times, on the same terms as
3829    /// [`GroupInfo::times`]: whatever a reopened file's root header had, and
3830    /// `None` for a file this writer created.
3831    root_times: Option<ObjectTimes>,
3832    /// Hands out the creation sequence numbers that order a group's links.
3833    next_creation_seq: Slot<u64>,
3834    /// Object-reference elements waiting for their target's object header
3835    /// address, which only exists once finalize has placed every header.
3836    pending_object_references: Slot<Vec<PendingObjectReference>>,
3837    /// Heap-backed reference objects waiting for the same address — the
3838    /// pre-1.12 region form and every 1.12 form whose element is a blob id.
3839    pending_heap_references: Slot<Vec<PendingHeapReference>>,
3840    /// What each object-reference attribute's value *means*, so
3841    /// [`object_attributes`](Hdf5Writer::object_attributes) can say it in
3842    /// addresses every time an object header is built.
3843    attribute_references: Slot<Vec<AttributeReferenceValue>>,
3844    /// Set when this file is in the classic (version-0/1 superblock) format,
3845    /// whether it was reopened in it or created at `H5F_LIBVER_EARLIEST`.
3846    /// See [`LegacyFile`]; [`is_legacy`](Self::is_legacy) is the only reader
3847    /// of whether it is there.
3848    legacy: Option<Box<LegacyFile>>,
3849    /// Which groups keep their links in a symbol table, and the storage each
3850    /// of them has. Empty for a file whose groups all store links in messages;
3851    /// see [`SymbolTables`], which owns the question.
3852    symbol_tables: SymbolTables,
3853    /// The v1 B-tree "K" ranks every node width in this file is derived from,
3854    /// after the superblock extension has had its say. A property of the file
3855    /// rather than of its generation: a version-2 superblock records no ranks
3856    /// of its own but its extension may, and a rewrite that used the library
3857    /// defaults there would write nodes of the wrong width.
3858    /// [`btree_v1_config`](Hdf5Writer::btree_v1_config) is the only reader.
3859    btree: BTreeV1Config,
3860    /// The superblock extension this file carries, and where the replacement
3861    /// went; see [`CarriedExtension`].
3862    extension: Box<CarriedExtension>,
3863    /// The free-space managers a reopened `persist: true` file carries; see
3864    /// [`FileSpaceState`]. `None` for every other file — one with no
3865    /// file-space info message, one that does not persist, one under paged
3866    /// aggregation, and every file this session created — and those files get
3867    /// no free-space manager written either.
3868    free_space: Option<Box<FileSpaceState>>,
3869    /// The file's shared-message indexes, when it was created with any.
3870    /// `None` — the default — is a file with no shared-message table, where
3871    /// [`share_message`](Self::share_message) is the identity.
3872    sohm: Option<Box<SohmState>>,
3873    /// The directory holding this HDF5 file, resolved once when it was opened
3874    /// — libhdf5's `H5F_t::extpath`, and the same value the read side keeps.
3875    /// External raw-data file names are joined against it when
3876    /// `HDF5_EXTFILE_PREFIX` names `${ORIGIN}`, so a write and a later read of
3877    /// the same dataset must resolve a relative name identically; capturing it
3878    /// at open time rather than reading the process's current directory per
3879    /// write is what makes that hold.
3880    source_dir: PathBuf,
3881}
3882
3883/// A file's shared object header messages, from creation to the table on disk.
3884///
3885/// INVARIANT: a message body reaches the file either literally or as a pointer
3886/// to exactly one heap object, never both, and the reference count of that
3887/// object is the number of headers that hold the pointer.
3888/// [`share_message`](Hdf5Writer::share_message) is the only place a body is
3889/// offered to an index, and
3890/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) is the
3891/// only place the phase changes — so counting and substituting are two passes
3892/// over the same call site rather than two pieces of logic that must agree.
3893struct SohmState {
3894    /// The indexes the file was created with, in table order.
3895    indexes: Vec<SohmIndexSpec>,
3896    /// What `share_message` does to an eligible message right now.
3897    phase: Slot<SohmPhase>,
3898    /// Address of the master table this session laid out, once it has one.
3899    /// Also the once-only latch on the layout: a second finalize keeps the
3900    /// table the first one published, and
3901    /// [`Hdf5Writer::write_superblock_extension`] reads it to name that table
3902    /// in the extension.
3903    table_addr: Slot<Option<u64>>,
3904    /// The blocks the table a reopen found occupies — the master table and
3905    /// each index's heap and index structure — taken by the finalize that
3906    /// replaces them. Empty for a file this session created.
3907    ///
3908    /// The table is laid out whole from the whole message set, so a reopen
3909    /// replaces it rather than inserting into it, and every header holding a
3910    /// pointer into the old one is rewritten in the same finalize
3911    /// ([`Hdf5Writer::rebuilds_shared_messages`]).
3912    superseded: Slot<Vec<(u64, u64)>>,
3913}
3914
3915/// The passes `share_message` runs in, and the state between them.
3916enum SohmPhase {
3917    /// Outside a finalize: every message stays literal.
3918    Idle,
3919    /// Measuring headers, before the bodies they will hold are final. A
3920    /// shareable message answers at the width of a heap pointer over a heap
3921    /// object that does not exist yet, which is the width the one it ends up
3922    /// pointing at has: a `H5O_shared_t` in heap form is the same size
3923    /// whatever it names. Nothing this pass produces is written — it exists so
3924    /// [`allocate_object_headers`](Hdf5Writer::allocate_object_headers) can
3925    /// reserve a block for a header whose messages are shared before the
3926    /// content phase has decided which heap object each one shares.
3927    ///
3928    /// The set is [`FirstCopies`], and it is why this pass has state at all:
3929    /// a message left literal is *wider* than a pointer, so a header can only
3930    /// be measured by making the same first-copy decision the substituting
3931    /// pass will make.
3932    Predict(FirstCopies),
3933    /// Counting the bodies the file will share. Messages still go in
3934    /// literally, so nothing this pass builds is written.
3935    Collect(SohmCollector),
3936    /// Substituting. A body the collect pass never saw stays literal, which
3937    /// is a valid file: the record it would have shared simply keeps a
3938    /// reference count one higher than the pointers that reach it.
3939    Resolve {
3940        /// Heap ID per body, from the table this finalize laid out.
3941        ids: HashMap<(u8, Vec<u8>), [u8; SOHM_HEAP_ID_LEN]>,
3942        /// The first copies this pass has already handed out; see
3943        /// [`FirstCopies`].
3944        first: FirstCopies,
3945    },
3946}
3947
3948/// The bodies a pass has already left literal in the header that offered them
3949/// first (`H5SM_IN_OH`, H5SM.c:1400-1417).
3950///
3951/// INVARIANT: the three passes walk the same object headers in the same order
3952/// — [`allocate_object_headers`](Hdf5Writer::allocate_object_headers),
3953/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) and
3954/// [`write_object_headers`](Hdf5Writer::write_object_headers) each build every
3955/// dataset in `datasets` order, then every group, then the root — so "the
3956/// header that offered this body first" is the same header in all three. Each
3957/// pass keeps its own set rather than sharing one, so a pass that does not run
3958/// cannot leave a stale decision behind for the next one. A divergence would
3959/// make a header wider than the block reserved for it, which
3960/// [`check_header_size`] refuses rather than writing.
3961type FirstCopies = std::collections::HashSet<(u8, Vec<u8>)>;
3962
3963/// The object header a message is being written into — `H5SM_try_share`'s
3964/// `open_oh` argument, which is what decides whether a first copy has a header
3965/// to stay literal in at all.
3966#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3967enum ShareOwner {
3968    /// `H5SM_try_share(f, NULL, ...)`: the message belongs to no object header
3969    /// of its own. An attribute's datatype and dataspace are offered this way
3970    /// (H5Aint.c:375-377) — they live inside the attribute's body, so there is
3971    /// no header message for a record to name and the body goes to the heap on
3972    /// first use however shareable its class is.
3973    Detached,
3974    /// `H5SM_try_share(f, oh, ...)`: the message is a message of the object
3975    /// header at this address (`H5O__msg_alloc`, H5Omessage.c:1735).
3976    Header(u64),
3977}
3978
3979impl SohmState {
3980    /// A file's indexes, plus the blocks of the table they were read out of
3981    /// when the file was reopened (empty when it was created this session).
3982    fn new(indexes: Vec<SohmIndexSpec>, superseded: Vec<(u64, u64)>) -> Self {
3983        Self {
3984            indexes,
3985            phase: Slot::new(SohmPhase::Idle),
3986            table_addr: Slot::new(None),
3987            superseded: Slot::new(superseded),
3988        }
3989    }
3990
3991    /// The index that would take a `msg_type` message of `body_len` bytes,
3992    /// as `H5SM_try_share` resolves one: the first index whose type mask
3993    /// covers the class, and then only if the message reaches that index's
3994    /// minimum. A message too small for its index is not offered to another —
3995    /// `H5SM__get_index` picks by type alone and the size check comes after.
3996    fn index_for(&self, msg_type: u8, body_len: usize) -> Option<usize> {
3997        let flag = type_flag(msg_type)?;
3998        let (at, spec) = self
3999            .indexes
4000            .iter()
4001            .enumerate()
4002            .find(|(_, spec)| spec.mesg_types & flag != 0)?;
4003        (body_len as u64 >= u64::from(spec.min_mesg_size)).then_some(at)
4004    }
4005
4006    /// Whether any index takes attribute messages, which is what makes the
4007    /// file record message creation indices — `H5SM_init` sets
4008    /// `store_msg_crt_idx` on exactly this condition (H5SM.c:220).
4009    fn shares_attributes(&self) -> bool {
4010        let Some(flag) = type_flag(MSG_ATTRIBUTE) else {
4011            return false;
4012        };
4013        self.indexes.iter().any(|spec| spec.mesg_types & flag != 0)
4014    }
4015}
4016
4017/// What decides whether two offers are the same shared message: the class,
4018/// the bytes, and the messages the bytes will end up pointing at.
4019type CollectedKey = (u8, Vec<u8>, Vec<NestedShare>);
4020
4021/// The shareable message bodies of one collect pass, in first-seen order.
4022struct SohmCollector {
4023    /// Per index, its bodies with the number of headers holding each.
4024    messages: Vec<Vec<SharedMessage>>,
4025    /// Where a body sits: `(index, position in that index's messages)`, keyed
4026    /// by everything that decides what will be stored — the class, the bytes,
4027    /// and the messages the bytes will end up pointing at.
4028    seen: HashMap<CollectedKey, (usize, usize)>,
4029}
4030
4031impl SohmCollector {
4032    fn new(nindexes: usize) -> Self {
4033        Self {
4034            messages: vec![Vec::new(); nindexes],
4035            seen: HashMap::new(),
4036        }
4037    }
4038
4039    /// Count one message against `index`, adding the body the first time it
4040    /// is seen, and say whether that body is new.
4041    ///
4042    /// `ohdr` is the header this offer would leave the body literal in when it
4043    /// is the first — `None` when the class cannot be shared in an object
4044    /// header or the offer names none. It is recorded only for a first copy:
4045    /// once a body is in the heap, later offers of it are pointers whatever
4046    /// header they come from.
4047    ///
4048    /// Two bodies are the same message only if their nesting agrees as well:
4049    /// the heap IDs a nesting body will hold are still zero here, so two
4050    /// attributes that differ only in their datatype are the same bytes at
4051    /// this point and different bytes on disk.
4052    fn record(
4053        &mut self,
4054        index: usize,
4055        msg_type: u8,
4056        body: &[u8],
4057        nested: &[NestedShare],
4058        ohdr: Option<u64>,
4059    ) -> bool {
4060        let key = (msg_type, body.to_vec(), nested.to_vec());
4061        match self.seen.get(&key) {
4062            Some(&(at, pos)) => {
4063                self.messages[at][pos].ref_count += 1;
4064                false
4065            }
4066            None => {
4067                let pos = self.messages[index].len();
4068                self.messages[index].push(SharedMessage {
4069                    msg_type,
4070                    body: body.to_vec(),
4071                    nested: nested.to_vec(),
4072                    ref_count: 1,
4073                    ohdr_addr: ohdr,
4074                });
4075                self.seen.insert(key, (index, pos));
4076                true
4077            }
4078        }
4079    }
4080
4081    /// Give back the reference [`record`](Self::record) took for a body whose
4082    /// container turned out to be a copy of one already here.
4083    ///
4084    /// A body reached only through a shared container is referenced once per
4085    /// container *record*, not once per object that has one: the pointer to
4086    /// it lives in the container's heap object, which exists once however
4087    /// many headers name it. `H5O__attr_create` reaches the same count from
4088    /// the other side, by building each attribute's components shared and
4089    /// then calling `H5O__attr_delete` — which decrements exactly the
4090    /// datatype and dataspace (H5Oattr.c:568-585) — whenever the attribute it
4091    /// built was not the first copy (H5Oattribute.c:331-366).
4092    fn release(&mut self, msg_type: u8, body: &[u8]) {
4093        if let Some(&(at, pos)) = self.seen.get(&(msg_type, body.to_vec(), Vec::new())) {
4094            let count = &mut self.messages[at][pos].ref_count;
4095            *count = count.saturating_sub(1);
4096        }
4097    }
4098}
4099
4100/// The file-creation properties a brand-new file is made with.
4101///
4102/// libhdf5 splits these across the file creation and file access property
4103/// lists (`H5Pset_userblock`, `H5Pset_link_creation_order`,
4104/// `H5Pset_libver_bounds`, the locking property); what they have in common is
4105/// that they are read once, when the file is created, and cannot be changed
4106/// afterwards without rewriting it. Options that *can* change mid-session —
4107/// the bound for objects created later, the creation-order policy for later
4108/// objects — have their own setters.
4109#[derive(Debug, Clone, Copy, Default)]
4110pub struct FileCreateOptions {
4111    /// OS-level locking policy for the new file.
4112    pub locking: crate::io::locking::FileLocking,
4113    /// Creation-order policy for the root group, and the default for every
4114    /// object created afterwards; see [`Hdf5Writer::set_track_order`].
4115    pub track_order: bool,
4116    /// Time-tracking policy for the root group, and the default for every
4117    /// object created afterwards; see [`Hdf5Writer::set_track_times`].
4118    pub track_times: bool,
4119    /// The file's low library-version bound (`H5Pset_libver_bounds`'s `low`),
4120    /// or `None` when the caller named none.
4121    ///
4122    /// The distinction is not decoration. `Some(LibverBound::Earliest)` is a
4123    /// request for the format libhdf5 writes at `H5F_LIBVER_EARLIEST` — a
4124    /// version-0 superblock over symbol-table groups and version-1 object
4125    /// headers, which is what [`ObjectFormat::Legacy`] encodes. `None` keeps
4126    /// what this crate has always written for a file whose creator said
4127    /// nothing: the version-2 superblock and link-message groups of the v1.8
4128    /// format, with the earliest bound's message versions where they can
4129    /// express the content. That combination is this crate's own, not one
4130    /// libhdf5 writes, so it cannot be spelled as a bound.
4131    pub libver: Option<LibverBound>,
4132    /// Bytes reserved in front of the superblock for the application's own
4133    /// use (`H5Pset_userblock`). Zero, the default, places the superblock at
4134    /// offset 0; otherwise a power of two of at least
4135    /// [`MIN_USERBLOCK`] bytes, since a reader finds the
4136    /// superblock by doubling its search offset from there.
4137    pub userblock: u64,
4138    /// Shared object header message indexes; see [`SharedMessageConfig`].
4139    pub shared_messages: SharedMessageConfig,
4140    /// How the file manages its own space; see [`FileSpaceConfig`].
4141    pub file_space: FileSpaceConfig,
4142}
4143
4144/// The file-space handling properties a new file is created with — the three
4145/// arguments of `H5Pset_file_space_strategy` and the one of
4146/// `H5Pset_file_space_page_size`.
4147///
4148/// The four together are what `H5F__super_init` compares against the library
4149/// defaults to decide whether the file needs a file-space info message at all
4150/// (H5Fsuper.c:1092-1097), which is why the page size belongs here even though
4151/// only paged aggregation allocates by it: a file that names a page size and
4152/// nothing else still carries the message.
4153#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4154pub struct FileSpaceConfig {
4155    /// `H5F_fspace_strategy_t`.
4156    pub strategy: FileSpaceStrategy,
4157    /// Whether the free-space managers are written to the file on close.
4158    pub persist: bool,
4159    /// The smallest section a manager records; a block freed below it is
4160    /// space the file leaks rather than tracks.
4161    pub threshold: u64,
4162    /// `H5Pset_file_space_page_size`: the file-space page every allocation of
4163    /// a paged file is shaped by, and the value the message carries whatever
4164    /// the strategy.
4165    pub page_size: u64,
4166}
4167
4168impl Default for FileSpaceConfig {
4169    /// `H5F_FILE_SPACE_STRATEGY_DEF`, `H5F_FREE_SPACE_PERSIST_DEF`,
4170    /// `H5F_FREE_SPACE_THRESHOLD_DEF` and `H5F_FILE_SPACE_PAGE_SIZE_DEF`
4171    /// (H5Fprivate.h:326-336).
4172    fn default() -> Self {
4173        Self {
4174            strategy: FileSpaceStrategy::FsmAggr,
4175            persist: false,
4176            threshold: 1,
4177            page_size: DEFAULT_FILE_SPACE_PAGE_SIZE,
4178        }
4179    }
4180}
4181
4182impl FileSpaceConfig {
4183    /// The properties as `H5P__set_file_space_strategy` (H5Pfcpl.c:1176)
4184    /// stores them: `persist` and `threshold` are set only for the two
4185    /// strategies that have free-space managers to persist, and keep their
4186    /// defaults for the two that do not.
4187    pub fn new(strategy: FileSpaceStrategy, persist: bool, threshold: u64) -> Self {
4188        let uses_managers = matches!(
4189            strategy,
4190            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
4191        );
4192        Self {
4193            strategy,
4194            persist: uses_managers && persist,
4195            threshold: if uses_managers {
4196                threshold
4197            } else {
4198                Self::default().threshold
4199            },
4200            ..Self::default()
4201        }
4202    }
4203
4204    /// `H5Pset_file_space_page_size`, the fourth file-space property and the
4205    /// one libhdf5 sets on its own call.
4206    ///
4207    /// Independent of the strategy, as upstream is: the value reaches the
4208    /// file-space info message whatever the strategy is, and only paged
4209    /// aggregation allocates by it. Out-of-range sizes are refused where the
4210    /// file is created ([`validate`](Self::validate)) rather than here, so a
4211    /// builder chain stays a builder chain.
4212    pub fn with_page_size(mut self, page_size: u64) -> Self {
4213        self.page_size = page_size;
4214        self
4215    }
4216
4217    /// Whether the file has to say any of this on disk. `H5F__super_init`
4218    /// writes the file-space info message only for a file that differs from
4219    /// the library defaults in one of the four properties (H5Fsuper.c:1092),
4220    /// and raises such a file's superblock to version 2 so it has an
4221    /// extension to write it into (H5Fsuper.c:1144).
4222    pub fn is_default(&self) -> bool {
4223        *self == Self::default()
4224    }
4225
4226    /// Refuse what this writer cannot make. `H5Pset_file_space_strategy`
4227    /// itself only refuses a strategy outside the enum (H5Pfcpl.c:1223), and
4228    /// `H5Pset_file_space_page_size` a page size outside `[512, 1 GiB]`
4229    /// (H5Pfcpl.c:1389-1393) — no power of two required, only the bounds.
4230    fn validate(&self) -> IoResult<()> {
4231        if !(PAGE_SIZE_MIN..=PAGE_SIZE_MAX).contains(&self.page_size) {
4232            return Err(crate::io::IoError::InvalidState(format!(
4233                "a file-space page size is between {PAGE_SIZE_MIN} bytes and \
4234                 {PAGE_SIZE_MAX}, not {}",
4235                self.page_size
4236            )));
4237        }
4238        match self.strategy {
4239            FileSpaceStrategy::FsmAggr
4240            | FileSpaceStrategy::Aggr
4241            | FileSpaceStrategy::None
4242            | FileSpaceStrategy::Page => Ok(()),
4243            FileSpaceStrategy::Unknown(b) => Err(crate::io::IoError::InvalidState(format!(
4244                "invalid file-space strategy {b}"
4245            ))),
4246        }
4247    }
4248
4249    /// The message a created file carries, before anything is allocated:
4250    /// every manager address undefined and no end-of-allocation recorded,
4251    /// which is what `H5F__super_init` writes (H5Fsuper.c:1369-1382).
4252    fn message(&self) -> FileSpaceInfoMessage {
4253        FileSpaceInfoMessage {
4254            // `H5O_fsinfo_set_version` starts at version 1 and only ever
4255            // raises it, so a created file never carries the version-0 form
4256            // however low its version bounds are.
4257            version: 1,
4258            strategy: self.strategy,
4259            persist: self.persist,
4260            threshold: self.threshold,
4261            page_size: self.page_size,
4262            pgend_meta_thres: 0,
4263            eoa_pre_fsm_fsalloc: UNDEF_ADDR,
4264            fs_addr: vec![UNDEF_ADDR; FS_ADDR_COUNT_V1],
4265        }
4266    }
4267}
4268
4269/// The shared object header message indexes a new file is created with.
4270///
4271/// libhdf5 sets these with three calls on the file creation property list:
4272/// `H5Pset_shared_mesg_nindexes` fixes how many indexes there are,
4273/// `H5Pset_shared_mesg_index` gives each one the message types it covers and
4274/// the smallest message it will take, and `H5Pset_shared_mesg_phase_change`
4275/// sets the list/B-tree thresholds for all of them at once. The default —
4276/// no indexes — is a file with no shared-message table, which is what every
4277/// file this crate wrote before the option existed.
4278#[derive(Debug, Clone, Copy, PartialEq)]
4279pub struct SharedMessageConfig {
4280    /// Indexes in table order; only the first `count` are in use.
4281    indexes: [SohmIndexSpec; MAX_SOHM_INDEXES],
4282    /// How many indexes the caller asked for. Kept even when it is more than
4283    /// the array holds, so file creation can refuse the count the way
4284    /// `H5Pset_shared_mesg_nindexes` does rather than silently drop indexes.
4285    count: usize,
4286}
4287
4288impl Default for SharedMessageConfig {
4289    fn default() -> Self {
4290        Self {
4291            indexes: [SohmIndexSpec {
4292                mesg_types: 0,
4293                min_mesg_size: 0,
4294                list_max: DEFAULT_SOHM_LIST_MAX,
4295                btree_min: DEFAULT_SOHM_BTREE_MIN,
4296            }; MAX_SOHM_INDEXES],
4297            count: 0,
4298        }
4299    }
4300}
4301
4302impl SharedMessageConfig {
4303    /// One index per `(mesg_types, min_mesg_size)` pair — the arguments
4304    /// `H5Pset_shared_mesg_index` takes, where `mesg_types` is the bit mask
4305    /// [`type_flag`](crate::format::sohm::type_flag) builds — with the
4306    /// file-wide phase change `H5Pset_shared_mesg_phase_change` sets: above
4307    /// `list_max` an index is a v2 B-tree, below `btree_min` it is a list
4308    /// again, and `list_max == 0` makes it a B-tree from its first message.
4309    ///
4310    /// Nothing is validated here; [`Hdf5Writer::create_with_options`] refuses
4311    /// a configuration libhdf5 would refuse, so an invalid one is reported
4312    /// where the file is made rather than where the value is typed.
4313    pub fn new(indexes: &[(u16, u32)], list_max: u16, btree_min: u16) -> Self {
4314        let mut config = Self {
4315            count: indexes.len(),
4316            ..Self::default()
4317        };
4318        for (slot, &(mesg_types, min_mesg_size)) in config.indexes.iter_mut().zip(indexes) {
4319            *slot = SohmIndexSpec {
4320                mesg_types,
4321                min_mesg_size,
4322                list_max,
4323                btree_min,
4324            };
4325        }
4326        config
4327    }
4328
4329    /// The indexes in use, in table order.
4330    pub(crate) fn specs(&self) -> &[SohmIndexSpec] {
4331        &self.indexes[..self.count.min(MAX_SOHM_INDEXES)]
4332    }
4333
4334    /// Refuse a configuration `H5Pset_shared_mesg_nindexes` or
4335    /// `H5Pset_shared_mesg_phase_change` would refuse.
4336    fn validate(&self) -> IoResult<()> {
4337        if self.count > MAX_SOHM_INDEXES {
4338            return Err(crate::io::IoError::InvalidState(format!(
4339                "a file may declare at most {MAX_SOHM_INDEXES} shared-message \
4340                 indexes, not {}",
4341                self.count
4342            )));
4343        }
4344        for spec in self.specs() {
4345            // The two thresholds must not overlap, or an index would convert
4346            // back and forth on every insert.
4347            if u32::from(spec.btree_min) > u32::from(spec.list_max) + 1 {
4348                return Err(crate::io::IoError::InvalidState(format!(
4349                    "shared-message phase change needs btree_min ({}) at most one \
4350                     past list_max ({}), or an index converts on every insert",
4351                    spec.btree_min, spec.list_max
4352                )));
4353            }
4354            if spec.mesg_types == 0 {
4355                return Err(crate::io::IoError::InvalidState(
4356                    "a shared-message index covering no message type would never \
4357                     be used; give it a type mask or drop it"
4358                        .into(),
4359                ));
4360            }
4361        }
4362        Ok(())
4363    }
4364}
4365
4366/// One object-reference element written before its value could be known.
4367///
4368/// An `H5R_OBJECT1` element is the target's object header address, and
4369/// addresses are assigned during finalize, so a write records the target by
4370/// path here and [`Hdf5Writer::write_object_reference_values`] puts the address
4371/// down once every header has one.
4372pub(crate) struct PendingObjectReference {
4373    /// Dataset holding the element.
4374    dataset: usize,
4375    /// Element index within that dataset.
4376    element: u64,
4377    /// Path of the object the element names; `/` is the root group.
4378    target: String,
4379}
4380
4381/// One heap-backed reference object written before its target's address could
4382/// be known.
4383///
4384/// The *element* of a `H5R_DATASET_REGION1`, and of every 1.12 reference whose
4385/// encoding does not fit inline, is final at write time — it is the global-heap
4386/// id of the object the write inserted. What waits is the `sizeof_addr` bytes
4387/// of that heap object holding the target's object header address, which
4388/// [`Hdf5Writer::write_heap_reference_values`] stamps in.
4389pub(crate) struct PendingHeapReference {
4390    /// Address of the global-heap collection holding the object.
4391    collection: u64,
4392    /// The object's index within that collection.
4393    index: u16,
4394    /// Where the target's token sits inside that object. The pre-1.12 region
4395    /// form leads with it (`H5R__encode_token_region_compat`); every 1.12 form
4396    /// puts the token's length byte first (`H5R__encode_obj_token`).
4397    token_offset: usize,
4398    /// What the reference names, and how strictly its path must resolve.
4399    target: PendingHeapTarget,
4400}
4401
4402/// What the path of a heap-backed reference must resolve to.
4403///
4404/// The two rules `H5R` applies: a region reference names a *dataset*, since
4405/// `H5Rcreate_region` takes one dataset's dataspace and every reader
4406/// dereferences it as one, while an attribute reference names the attribute's
4407/// owner, which `H5Rcreate_attr` lets be any object.
4408#[derive(Debug, Clone)]
4409pub(crate) enum PendingHeapTarget {
4410    Dataset(String),
4411    Object(String),
4412}
4413
4414/// The value of an attribute whose elements are object references, kept as
4415/// what it means rather than as what it encodes to.
4416///
4417/// An attribute's value is part of its object header message, so it cannot be
4418/// stamped after the fact the way a dataset element can — the header is one
4419/// block, written once. What is stored instead is the paths, and
4420/// [`Hdf5Writer::object_attributes`] turns them into addresses every time the
4421/// attribute set is built: the measuring pass reads the zeros of objects that
4422/// have no address yet, the content pass reads the addresses the file will
4423/// have, and the two agree in length because an address is a fixed-width
4424/// field. The entry in the object's attribute list carries an image with
4425/// zeros where the addresses go and is never itself written.
4426///
4427/// The address of `targets[i]` lands at byte `i * stride` of that image: the
4428/// whole element when the attribute is an array of references, the leading
4429/// member when each element is a compound that carries other fields beside
4430/// the reference (`REFERENCE_LIST`'s `dimension`), which the stored image
4431/// already holds.
4432pub(crate) struct AttributeReferenceValue {
4433    /// The object the attribute hangs on.
4434    scope: AttrScope,
4435    /// The attribute's name within that object.
4436    name: String,
4437    /// Paths of the objects the elements name, in element order; `/` is the
4438    /// root group.
4439    targets: Vec<String>,
4440    /// Bytes from one element's address to the next: the element size.
4441    stride: usize,
4442}
4443
4444/// The attribute naming the scales attached to each axis of a dataset.
4445pub(crate) const DIMENSION_LIST: &str = "DIMENSION_LIST";
4446/// The attribute naming every (dataset, axis) a dimension scale is attached to.
4447pub(crate) const REFERENCE_LIST: &str = "REFERENCE_LIST";
4448/// The `CLASS` a dimension scale carries.
4449const DIMENSION_SCALE_CLASS: &str = "DIMENSION_SCALE";
4450
4451/// A dataset's `CLASS` attribute as `H5DS` reads it.
4452enum ClassAttr {
4453    /// A fixed-length string, with what `H5DSis_scale` checks beside the text.
4454    Fixed {
4455        size: u32,
4456        null_terminated: bool,
4457        text: String,
4458    },
4459    /// A variable-length string.
4460    VarLen(String),
4461    /// Not a string at all.
4462    NotString,
4463}
4464
4465/// `bytes` read as a C string: everything before the first NUL.
4466fn c_string(bytes: &[u8]) -> String {
4467    let end = bytes.iter().position(|&b| b == 0).unwrap_or(bytes.len());
4468    String::from_utf8_lossy(&bytes[..end]).into_owned()
4469}
4470
4471/// Refuse an object header body that is not the length its block was reserved
4472/// at.
4473///
4474/// The one check standing behind
4475/// [`HeaderLayout`]'s premise that measuring a header before its content is
4476/// final gives the same length as encoding it after. `what` names the object
4477/// only when the check fails, so the caller pays for the lookup only then.
4478fn check_header_size(
4479    encoded: &[u8],
4480    reserved: usize,
4481    what: impl FnOnce() -> String,
4482) -> IoResult<()> {
4483    if encoded.len() == reserved {
4484        return Ok(());
4485    }
4486    Err(crate::io::IoError::InvalidState(format!(
4487        "the object header of {} encodes to {} bytes but was measured at {}; \
4488         a message in it changed length once the addresses it names were known",
4489        what(),
4490        encoded.len(),
4491        reserved
4492    )))
4493}
4494
4495/// Where one object header goes: chunk 0's block and, when the header does
4496/// not fit it, a continuation block of its own.
4497///
4498/// Produced by [`Hdf5Writer::place_header`] and consumed by
4499/// [`Hdf5Writer::encode_header_in`]; between the two, everything the header
4500/// names is built against the address it records. The sizes travel with the
4501/// addresses because they are what the blocks were reserved at: the writing
4502/// pass checks each image against them rather than trusting that the two
4503/// passes agreed.
4504#[derive(Debug, Clone, Copy)]
4505struct HeaderPlacement {
4506    /// Chunk 0's address.
4507    addr: u64,
4508    /// Bytes reserved at `addr`. For a fresh header that is the whole image,
4509    /// a continuation chunk included, since one is laid directly behind
4510    /// chunk 0 in the same block.
4511    size: usize,
4512    /// Whether the block is one the object's existing header already
4513    /// occupied, which chunk 0 is then held to the size of; a fresh block is
4514    /// an exact fit.
4515    kept: bool,
4516    /// A continuation block of its own, `(address, size)`: what a kept block
4517    /// too small for every message spills into.
4518    continuation: Option<(u64, usize)>,
4519}
4520
4521impl HeaderPlacement {
4522    /// A block of `size` bytes at `addr` holding the whole header.
4523    fn fresh(addr: u64, size: usize) -> Self {
4524        Self {
4525            addr,
4526            size,
4527            kept: false,
4528            continuation: None,
4529        }
4530    }
4531
4532    /// The placement as the registry records a written header: chunk 0's
4533    /// block, then the continuation block when there is one.
4534    fn blocks(&self) -> crate::io::object_header_io::HeaderBlocks {
4535        std::iter::once((self.addr, self.size as u64))
4536            .chain(self.continuation.map(|(a, s)| (a, s as u64)))
4537            .collect()
4538    }
4539
4540    /// The placement a written header's recorded blocks describe, to write
4541    /// it back over: chunk 0 held to its block, and the continuation chunk,
4542    /// if it has one, to its own.
4543    fn over(blocks: &[(u64, u64)]) -> Option<Self> {
4544        match blocks {
4545            [(addr, size)] => Some(Self {
4546                addr: *addr,
4547                size: *size as usize,
4548                kept: true,
4549                continuation: None,
4550            }),
4551            [(addr, size), (cont, cont_size)] => Some(Self {
4552                addr: *addr,
4553                size: *size as usize,
4554                kept: true,
4555                continuation: Some((*cont, *cont_size as usize)),
4556            }),
4557            _ => None,
4558        }
4559    }
4560}
4561
4562/// Where every object header this finalize writes goes.
4563///
4564/// Produced by [`Hdf5Writer::allocate_object_headers`] and consumed by
4565/// [`Hdf5Writer::write_object_headers`].
4566struct HeaderLayout {
4567    /// `(dataset index, placement)`, in write order.
4568    datasets: Vec<(usize, HeaderPlacement)>,
4569    /// `(group index, placement)`, in write order.
4570    groups: Vec<(usize, HeaderPlacement)>,
4571    /// The root group's placement.
4572    root: HeaderPlacement,
4573}
4574
4575/// The chunk-0 blocks existing object headers keep across a rewrite, by
4576/// object: `(address, length)` of each, as the open-time walk read it.
4577///
4578/// Filled by [`Hdf5Writer::supersede_headers`] from the registry's
4579/// `obj_header_blocks` and consumed by
4580/// [`Hdf5Writer::allocate_object_headers`].
4581#[derive(Default)]
4582struct KeptChunks {
4583    datasets: std::collections::HashMap<usize, (u64, u64)>,
4584    groups: std::collections::HashMap<usize, (u64, u64)>,
4585    root: Option<(u64, u64)>,
4586}
4587
4588/// Refuse a region-reference selection the target dataset's extent does not
4589/// admit — libhdf5's `H5S_select_valid`, which `H5Rcreate` applies before it
4590/// serializes anything.
4591///
4592/// The rank check comes from [`Selection::to_boxes`], which also refuses a
4593/// regular hyperslab with an unlimited count or block; a region reference has
4594/// no growable extent to resolve one against.
4595fn validate_region_selection(selection: &Selection, dims: &[u64], path: &str) -> IoResult<()> {
4596    let boxes = selection.to_boxes(dims).map_err(|e| {
4597        crate::io::IoError::InvalidState(format!("region reference over '{path}': {e}"))
4598    })?;
4599    for (start, count) in boxes {
4600        for (d, (&s, &c)) in start.iter().zip(&count).enumerate() {
4601            if s.checked_add(c).is_none_or(|end| end > dims[d]) {
4602                return Err(crate::io::IoError::InvalidState(format!(
4603                    "region reference over '{path}' selects {s}..{} in dimension {d}, \
4604                     outside the dataset's extent of {}",
4605                    s.saturating_add(c),
4606                    dims[d]
4607                )));
4608            }
4609        }
4610    }
4611    Ok(())
4612}
4613
4614/// What a reopen found already on disk in dense form, by the scope whose
4615/// header names it.
4616///
4617/// Both halves together because they are found together — one walk of the
4618/// reopened headers fills both — and released together only in the delete
4619/// path; a finalize supersedes attribute storage before it lays object
4620/// headers out and link storage after, so each half has its own owner.
4621#[derive(Debug, Default)]
4622struct SupersededDense {
4623    attrs: HashMap<AttrScope, AttributeInfoMessage>,
4624    links: HashMap<LinkScope, LinkInfoMessage>,
4625}
4626
4627/// Which object's attribute list a prepared dense layout belongs to.
4628#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4629pub(crate) enum AttrScope {
4630    Root,
4631    Group(usize),
4632    Dataset(usize),
4633}
4634
4635/// Which group's link list a prepared dense layout belongs to.
4636#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4637pub(crate) enum LinkScope {
4638    Root,
4639    Group(usize),
4640}
4641
4642/// Attributes an object header keeps before libhdf5 spills the whole set to
4643/// dense storage (`H5O_CRT_ATTR_MAX_COMPACT_DEF`).
4644const MAX_COMPACT_ATTRS: usize = 8;
4645
4646/// Links `H5G__obj_create_real` sizes a new group's object header for
4647/// (`H5G_CRT_GINFO_EST_NUM_ENTRIES`), and the name length it assumes for each
4648/// (`H5G_CRT_GINFO_EST_NAME_LEN`). Together with the link info and group info
4649/// messages they are the whole of chunk 0 — see
4650/// [`chunk0_capacity`](Hdf5Writer::chunk0_capacity).
4651const EST_LINK_COUNT: usize = 4;
4652/// See [`EST_LINK_COUNT`].
4653const EST_LINK_NAME_LEN: usize = 8;
4654
4655/// Messages a shared-message index keeps in list form before it becomes a v2
4656/// B-tree (`H5F_CRT_SHMSG_LIST_MAX_DEF`).
4657const DEFAULT_SOHM_LIST_MAX: u16 = 50;
4658
4659/// Messages a shared-message B-tree index drops to before it reverts to a
4660/// list (`H5F_CRT_SHMSG_BTREE_MIN_DEF`).
4661const DEFAULT_SOHM_BTREE_MIN: u16 = 40;
4662
4663/// Links a group header keeps before libhdf5 spills the whole set to dense
4664/// storage (`H5G_CRT_GINFO_MAX_COMPACT`). This writer emits no phase-change
4665/// values in the Group Info message, so the default is what applies.
4666const MAX_COMPACT_LINKS: usize = 8;
4667
4668/// Bytes a compact dataset's raw image may occupy.
4669///
4670/// `H5D__compact_construct` bounds it by `H5O_MESG_MAX_SIZE` less the layout
4671/// message's own four bytes (version, class, and the 2-byte data length).
4672/// The constant it subtracts from is 65536, one past what the object header's
4673/// 2-byte message size field can express, so the ceiling here is taken from
4674/// [`MAX_MESSAGE_SIZE`] — the largest message that actually encodes — and is
4675/// one byte below libhdf5's.
4676pub const MAX_COMPACT_DATA: usize = MAX_MESSAGE_SIZE - 4;
4677
4678/// Smallest userblock a file can be created with, and the granularity of
4679/// every larger one: `H5Pset_userblock` takes 0 or a power of two from here
4680/// up, because `H5FD_locate_signature` looks for the superblock at 0 and then
4681/// at this offset doubled repeatedly.
4682pub const MIN_USERBLOCK: u64 = 512;
4683
4684impl Hdf5Writer {
4685    /// Create a new HDF5 file at `path` using the env-var-derived locking
4686    /// policy (controlled by `HDF5_USE_FILE_LOCKING`).
4687    ///
4688    /// The superblock (48 bytes for v3 with 8-byte offsets) is reserved at
4689    /// offset 0 and written during `close()`.
4690    pub fn create(path: &Path) -> IoResult<Self> {
4691        Self::create_with_locking(
4692            path,
4693            crate::io::locking::FileLocking::from_env_or(Default::default()),
4694        )
4695    }
4696
4697    /// Create a new HDF5 file at `path` with an explicit locking policy.
4698    pub fn create_with_locking(
4699        path: &Path,
4700        locking: crate::io::locking::FileLocking,
4701    ) -> IoResult<Self> {
4702        Self::create_with_options(
4703            path,
4704            FileCreateOptions {
4705                locking,
4706                ..Default::default()
4707            },
4708        )
4709    }
4710
4711    /// Create a new HDF5 file at `path` with explicit file-creation options.
4712    pub fn create_with_options(path: &Path, options: FileCreateOptions) -> IoResult<Self> {
4713        let FileCreateOptions {
4714            locking,
4715            track_order,
4716            track_times,
4717            libver,
4718            userblock,
4719            shared_messages,
4720            file_space,
4721        } = options;
4722        shared_messages.validate()?;
4723        file_space.validate()?;
4724        if userblock != 0 && (userblock < MIN_USERBLOCK || !userblock.is_power_of_two()) {
4725            return Err(crate::io::IoError::InvalidState(format!(
4726                "a userblock is {MIN_USERBLOCK} bytes or a power of two above it, \
4727                 not {userblock}: a reader locates the superblock by doubling its \
4728                 search offset from {MIN_USERBLOCK}, so no other size can hold one"
4729            )));
4730        }
4731        let policy = free_space::SpacePolicy::for_message(&file_space.message());
4732        // `H5F__super_init` (H5Fsuper.c:1182-1192) refuses a userblock that is
4733        // not a whole number of allocation units, which for a paged file is
4734        // the file-space page: everything after the userblock is addressed
4735        // from its end, so a userblock that is not a page multiple would put
4736        // every page boundary off the file's own grid.
4737        if let Some(page) = policy.page() {
4738            if userblock != 0 && userblock % page != 0 {
4739                return Err(crate::io::IoError::InvalidState(format!(
4740                    "a paged file's userblock is a multiple of its {page}-byte \
4741                     file-space page, not {userblock}"
4742                )));
4743            }
4744        }
4745        let mut handle = FileHandle::create_with_locking(path, locking)?;
4746        if userblock != 0 {
4747            // Written while the handle is still unbased, so offset 0 is the
4748            // start of the file: the block belongs to the application, not to
4749            // the HDF5 address space that begins where it ends. libhdf5 zeroes
4750            // it the same way (`H5F__super_init`), leaving a file whose first
4751            // `userblock` bytes are the application's to overwrite.
4752            handle.write_at(0, &vec![0u8; userblock as usize])?;
4753            handle.set_base(userblock);
4754        }
4755        let ctx = FormatContext::default_v3();
4756
4757        // `H5F_LIBVER_EARLIEST` is the one bound under which libhdf5 writes
4758        // the classic generation — the version-1 rows of every
4759        // message-version table, the symbol-table group form
4760        // (`H5G__obj_create_real`, H5Gobj.c:179) and the version-0 superblock
4761        // row of `HDF5_superblock_ver_bounds`.
4762        //
4763        // Shared object header messages move the last of those three and
4764        // nothing else. Their master table lives in a superblock extension,
4765        // which only a version-2 superblock has, so `H5F__super_init` raises
4766        // the superblock to version 2 whatever the low bound says
4767        // (H5Fsuper.c:1135) — but it does not touch `H5F_LOW_BOUND`, which is
4768        // what every other rule reads. So such a file is a version-2
4769        // superblock over symbol-table groups and version-1 messages, which
4770        // is what the `tests/fixtures/sohm_*.h5` files libhdf5 itself wrote
4771        // are.
4772        let classic = libver == Some(LibverBound::Earliest);
4773        let legacy = classic.then(|| Box::new(LegacyFile::created(ctx, userblock)));
4774        // Non-default file-space properties raise the superblock the same way
4775        // a shared-message table does, and for the same reason: the message
4776        // that declares them lives in an extension, and only a version-2
4777        // superblock has one (H5Fsuper.c:1144).
4778        let superblock_version = SuperblockVersion::Chosen(
4779            if classic && shared_messages.specs().is_empty() && file_space.is_default() {
4780                SUPERBLOCK_V0
4781            } else {
4782                SUPERBLOCK_V2
4783            },
4784        );
4785
4786        // Reserve the superblock at offset 0. Which version it gets is only
4787        // known once the file's content is (see `superblock_version_for`),
4788        // but the two a version-2 file can reach — 2 and 3 — encode to the
4789        // same size, so the reservation follows the base version alone.
4790        let superblock_size = match legacy.as_deref() {
4791            Some(l) if matches!(superblock_version, SuperblockVersion::Chosen(v) if v < SUPERBLOCK_V2) => {
4792                l.superblock.encoded_size()
4793            }
4794            _ => SuperblockV2V3::size_for(ctx.sizeof_addr),
4795        };
4796        // The superblock is an ordinary allocation, not a reservation: under
4797        // paged aggregation it takes the whole of page zero and leaves the
4798        // rest of that page as a section of the metadata manager, which is
4799        // what `H5F__super_init` gets from `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`
4800        // going through `H5MF__alloc_pagefs`. Unpaged it returns offset zero
4801        // and moves the end of the file to `superblock_size`, which is what
4802        // reserving it did.
4803        let allocator = FileAllocator::with_policy(0, policy);
4804        allocator.allocate(superblock_size as u64, FreeSpaceClass::Metadata);
4805
4806        Ok(Self {
4807            handle,
4808            allocator,
4809            ctx,
4810            datasets: Slot::new(Vec::new()),
4811            groups: Slot::new(Vec::new()),
4812            hard_links: Slot::new(Vec::new()),
4813            symbolic_links: Slot::new(Vec::new()),
4814            committed_datatypes: Slot::new(Vec::new()),
4815            preserved_links: Slot::new(Vec::new()),
4816            name_index: Slot::new(Box::new(NameIndex::new())),
4817            root_attributes: Slot::new(Vec::new()),
4818            create_lock: Slot::new(()),
4819            libver,
4820            closed: false,
4821            swmr_active: false,
4822            cwfs: Slot::new(Vec::new()),
4823            root_group_addr: None,
4824            superseded_root_header: Vec::new(),
4825            // A new file starts at the oldest superblock the generation it was
4826            // created in allows, and finalize raises it if the content needs a
4827            // newer one.
4828            superblock_version,
4829            dense_attributes: Slot::new(HashMap::new()),
4830            dense_links: Slot::new(HashMap::new()),
4831            superseded_dense: Slot::new(None),
4832            track_order: TrackOrder::uniform(track_order),
4833            track_times,
4834            root_track_order: TrackOrder::uniform(track_order),
4835            // The root group is created with the file, so it captures the
4836            // policy the same instant every other field of it is settled.
4837            root_times: track_times.then(|| ObjectTimes::created_at(now_seconds())),
4838            next_creation_seq: Slot::new(0),
4839            pending_object_references: Slot::new(Vec::new()),
4840            pending_heap_references: Slot::new(Vec::new()),
4841            attribute_references: Slot::new(Vec::new()),
4842            legacy,
4843            symbol_tables: SymbolTables::none_found(),
4844            // A created file has no extension to carry and no ranks but the
4845            // library defaults: `H5Pset_sym_k`/`H5Pset_istore_k` have no
4846            // equivalent on this writer's creation path.
4847            btree: BTreeV1Config::default(),
4848            extension: Box::default(),
4849            // A file created at the library defaults declares no file-space
4850            // strategy, so it has no message to write and no manager to keep;
4851            // one created with any other properties owns both.
4852            free_space: (!file_space.is_default()).then(|| {
4853                Box::new(FileSpaceState {
4854                    info: file_space.message(),
4855                    superseded: Vec::new(),
4856                })
4857            }),
4858            sohm: (!shared_messages.specs().is_empty())
4859                .then(|| Box::new(SohmState::new(shared_messages.specs().to_vec(), Vec::new()))),
4860            source_dir: source_dir_of(path)?,
4861        })
4862    }
4863
4864    /// Target the libhdf5 2.0 file format for datasets created after this
4865    /// call: filtered chunked datasets get layout message version 5, whose
4866    /// chunk indexes store chunk sizes in a fixed `sizeof_size`-byte field
4867    /// with no overflow limit (see [`Self::chunk_layout_version`]). Off by
4868    /// default, because readers older than libhdf5 2.0 — including the
4869    /// 1.14-based h5py wheels — reject version 5.
4870    ///
4871    /// `false` names `H5F_LIBVER_EARLIEST`, the far end of the same table,
4872    /// rather than un-naming the bound: it is `set_libver_bound`'s contract
4873    /// that applies, chunk index included.
4874    pub fn set_libver_latest(&mut self, latest: bool) -> IoResult<()> {
4875        self.set_libver_bound(if latest {
4876            LibverBound::V200
4877        } else {
4878            LibverBound::Earliest
4879        })
4880    }
4881
4882    /// Bytes this file reserves in front of its superblock
4883    /// (`H5Pget_userblock`).
4884    ///
4885    /// The same value for a file created with one and for a file reopened
4886    /// through [`open_append_with_locking`](Self::open_append_with_locking),
4887    /// which takes it from where the signature turned up: it is the base of
4888    /// the handle's address space either way.
4889    pub fn userblock_size(&self) -> u64 {
4890        self.handle.base()
4891    }
4892
4893    /// Set the file's low libver bound, the equivalent of
4894    /// `H5Pset_libver_bounds`'s `low` argument. Objects created after this
4895    /// call encode their messages at the versions that bound calls for.
4896    ///
4897    /// On a reopened file the bound is raised to the row the file's superblock
4898    /// version belongs to if it names an older one, exactly as
4899    /// `H5F__super_read` raises the fapl's value — see
4900    /// [`libver_floor`](Self::libver_floor). Only a bound the file's format
4901    /// cannot express at all is refused.
4902    pub fn set_libver_bound(&mut self, libver: LibverBound) -> IoResult<()> {
4903        // A classic file cannot honour a newer bound: every encoder in it
4904        // reads `H5F_LOW_BOUND`, and raising that is what makes libhdf5 write
4905        // the version-2/3 superblock this file does not have. Refused rather
4906        // than pinned silently, so the caller learns the bound did not take.
4907        if libver != LibverBound::Earliest && self.is_legacy() {
4908            return Err(crate::io::IoError::Unsupported(format!(
4909                "cannot set the library-version bound to {libver:?} on this file: it is                  in the classic (version-0/1 superblock) format, which libhdf5 writes                  only at H5F_LIBVER_EARLIEST"
4910            )));
4911        }
4912        self.libver = Some(libver);
4913        Ok(())
4914    }
4915
4916    /// The generation the *message* encoders follow — dataspace, datatype,
4917    /// fill value, attribute.
4918    ///
4919    /// A property of the file, not of the object: `H5S__set_version`,
4920    /// `H5O__fill_set_version`, `H5A__set_version` and `H5T_set_version` all
4921    /// read `H5F_LOW_BOUND(f)` and nothing about the object they are encoding
4922    /// for. So a creation-order-tracking group in a classic file still gets
4923    /// version-1 dataspaces and version-1 attribute messages, even though its
4924    /// own header is version 2.
4925    fn message_format(&self) -> ObjectFormat {
4926        match self.legacy {
4927            Some(_) => ObjectFormat::Legacy,
4928            None => ObjectFormat::Modern,
4929        }
4930    }
4931
4932    /// The object header version an object with this creation-order policy
4933    /// gets — `H5O__set_version` (H5Oint.c:251).
4934    ///
4935    /// Version 1 is the floor a classic file's low bound sets, but tracking
4936    /// creation order of *either* kind raises the object past it: the link
4937    /// creation index lives in the message envelope and the attribute tracking
4938    /// flags live in the header prefix, and version 1 has neither. This is a
4939    /// per-object question in a classic file, which is why the format is not
4940    /// one switch for the whole file — libhdf5 writes version-2 headers inside
4941    /// a version-0 superblock whenever the creation property list asks for
4942    /// creation order.
4943    fn header_format(&self, track: TrackOrder) -> ObjectFormat {
4944        let attrs = self.header_attr_order(track.attrs);
4945        if self.legacy.is_some() && !track.links.is_tracked() && !attrs.is_tracked() {
4946            ObjectFormat::Legacy
4947        } else {
4948            ObjectFormat::Modern
4949        }
4950    }
4951
4952    /// The attribute creation-order policy an object header records, given
4953    /// what the object's creation property list asked for.
4954    ///
4955    /// A file whose shared-message configuration covers attributes records a
4956    /// creation index on every object header message: a shared attribute is
4957    /// found again through it, so `H5SM_init` sets `store_msg_crt_idx`
4958    /// (H5SM.c:220) and `H5O__create_ohdr` then raises every header it creates
4959    /// to version 2 and ORs `H5O_HDR_ATTR_CRT_ORDER_TRACKED` into its flags
4960    /// (H5Oint.c:364, H5Oint.c:442) whatever the property list says. So on
4961    /// such a file the floor is `Tracked` — this is the only place that floor
4962    /// is applied, and both the header version and the header flags come
4963    /// through here.
4964    fn header_attr_order(&self, requested: CreationOrder) -> CreationOrder {
4965        if requested.is_tracked() || !self.tracks_message_creation_index() {
4966            return requested;
4967        }
4968        CreationOrder::Tracked
4969    }
4970
4971    /// Whether this finalize replaces the file's shared-message table.
4972    ///
4973    /// It does whenever the file has indexes and no table has been published
4974    /// this session — every finalize of a file created with them, and the
4975    /// first finalize after a reopen. `build_shared_messages` lays a table out
4976    /// whole from the whole message set rather than inserting into an existing
4977    /// one, so a reopen's table is a *replacement*: every heap ID in the file
4978    /// is reassigned, which makes every object header that holds one stale
4979    /// however little else about it changed. A second finalize (a SWMR close)
4980    /// keeps the table the first published and answers `false`.
4981    fn rebuilds_shared_messages(&self) -> bool {
4982        self.sohm
4983            .as_deref()
4984            .is_some_and(|s| s.table_addr.lock().is_none())
4985    }
4986
4987    /// Whether every object header this writer emits records message creation
4988    /// indices.
4989    fn tracks_message_creation_index(&self) -> bool {
4990        self.sohm
4991            .as_deref()
4992            .is_some_and(SohmState::shares_attributes)
4993    }
4994
4995    /// Whether the group at `scope` stores its links in a symbol table —
4996    /// `H5G__obj_create_real` (H5Gobj.c:129) and the conversion
4997    /// `H5G_obj_insert` performs (H5Gobj.c:512).
4998    ///
4999    /// The new group format is used unconditionally from `H5F_LIBVER_V18` up
5000    /// *for a group being created*, and below it only when the group tracks
5001    /// link creation order: a symbol table entry has no room for a creation
5002    /// index. The two axes are independent — a group that tracks only
5003    /// *attribute* creation order gets a version-2 header over a symbol table,
5004    /// which is what libhdf5 writes for it.
5005    ///
5006    /// A group the reopen found in a symbol table is not being created, and
5007    /// `H5G_obj_insert` never moves an existing group to the new format for
5008    /// the bound's sake. So [`SymbolTables::found`] answers for it whatever
5009    /// generation the rest of this session writes at.
5010    ///
5011    /// The content of the group is the third axis. A symbol table entry has
5012    /// three cache types and no room for a fourth, so an external or
5013    /// user-defined link cannot go in one; libhdf5 answers by converting that
5014    /// one group to link messages the moment such a link is inserted, leaving
5015    /// the superblock version, the object header version and every other group
5016    /// in the file alone. This writer builds each group's storage once at
5017    /// finalize rather than link by link, so the same rule reads as a question
5018    /// about the finished set.
5019    fn uses_symbol_table(&self, scope: LinkScope, links: CreationOrder) -> bool {
5020        (self.legacy.is_some() || self.symbol_tables.found.contains(&scope))
5021            && !links.is_tracked()
5022            && self.links_fit_symbol_table(scope, links)
5023    }
5024
5025    /// Whether every link `scope` holds is one a symbol table entry can
5026    /// express — `H5G_obj_insert`'s `obj_lnk->cset != H5T_CSET_ASCII ||
5027    /// obj_lnk->type > H5L_TYPE_BUILTIN_MAX` test (H5Gobj.c:514), asked of the
5028    /// whole set.
5029    ///
5030    /// A link a reopen carried through verbatim counts too, and one this
5031    /// writer cannot even decode counts as not fitting: the entry would have
5032    /// to be built from the decoded form, while a link message is re-emitted
5033    /// byte for byte.
5034    fn links_fit_symbol_table(&self, scope: LinkScope, order: CreationOrder) -> bool {
5035        self.group_links(scope, order)
5036            .iter()
5037            .all(LinkMessage::fits_symbol_table)
5038            && self.preserved_links_for(scope).iter().all(|encoded| {
5039                LinkMessage::decode(encoded, &self.ctx)
5040                    .is_ok_and(|(link, _)| link.fits_symbol_table())
5041            })
5042    }
5043
5044    /// The header format of the registered dataset at `index`.
5045    ///
5046    /// A dataset has no links, so only the attribute half of the policy can
5047    /// raise it past version 1.
5048    fn dataset_header_format(&self, index: usize) -> ObjectFormat {
5049        let ds = self.ds(index);
5050        let attrs = ds.lock().track_attr_order;
5051        self.header_format(TrackOrder {
5052            links: CreationOrder::default(),
5053            attrs,
5054        })
5055    }
5056
5057    /// The header format of the registered group at `index`.
5058    fn group_header_format(&self, index: usize) -> ObjectFormat {
5059        let grp = self.grp(index);
5060        let track = grp.lock().track_order;
5061        self.header_format(track)
5062    }
5063
5064    /// The oldest bound this file may be written at, and the single owner of
5065    /// the reopen half of the [`SuperblockVersion`] invariant.
5066    ///
5067    /// A reopened file's superblock version is the only thing on disk that
5068    /// says which generation the file is, and `H5F__super_read` reads it as
5069    /// exactly that: it raises `H5F_LOW_BOUND` to the row that version belongs
5070    /// to (hdf5_1.14.6 H5Fsuper.c:460-466). Every version-selecting site below
5071    /// goes through [`session_libver`](Self::session_libver) rather than
5072    /// reading the `libver` field, so none of them can hand a reopened file a
5073    /// structure older than the file already claims to hold.
5074    ///
5075    /// A floor, not a ceiling. `H5Fopen` takes a fapl like `H5Fcreate` does,
5076    /// and a bound named above this one applies: libhdf5 1.14.6 writes a
5077    /// version-4 layout message into a version-2 superblock when asked at
5078    /// `H5F_LIBVER_V110`, leaving the superblock version alone. The ceiling is
5079    /// the separate question [`set_libver_bound`](Self::set_libver_bound)
5080    /// answers — no bound but `Earliest` may be named on a classic file.
5081    fn libver_floor(&self) -> LibverBound {
5082        self.superblock_version.libver_floor()
5083    }
5084
5085    /// The bound one family of encoders is written at, and the single reader
5086    /// of the `libver` field.
5087    ///
5088    /// Three inputs, in the order libhdf5 applies them. A bound the caller
5089    /// named is the fapl's `low`, raised to the floor exactly as
5090    /// `H5F__super_read` raises it. With no bound named the answer depends on
5091    /// which superblock this file has:
5092    ///
5093    /// * A file this writer created has none yet, so the writer picks —
5094    ///   `create_default`, which differs per family because this crate's
5095    ///   default file is two rows rather than one bound (see the `libver`
5096    ///   field, [`encoding_libver`] and [`layout_version_bound`]). The
5097    ///   superblock is then written to match what was picked.
5098    /// * A reopened file has already said which generation it is, and its
5099    ///   superblock cannot be rewritten to match a newer pick. So the floor is
5100    ///   the whole answer — the same value `H5F_LOW_BOUND` has after
5101    ///   `H5F__super_read` under a default fapl.
5102    ///
5103    /// [`encoding_libver`]: Self::encoding_libver
5104    /// [`layout_version_bound`]: Self::layout_version_bound
5105    fn session_libver(&self, create_default: LibverBound) -> LibverBound {
5106        let floor = self.libver_floor();
5107        let bound = match (self.libver, self.superblock_version) {
5108            (Some(named), _) => named.max(floor),
5109            (None, SuperblockVersion::Existing(_)) => floor,
5110            (None, SuperblockVersion::Chosen(_)) => create_default,
5111        };
5112        match self.message_format() {
5113            // `H5F_LIBVER_EARLIEST` is the only low bound under which libhdf5
5114            // writes a version-0/1 superblock at all, so a newer structure
5115            // inside one is a combination no libhdf5 produces. Refused where
5116            // the caller asks for it (`set_libver_bound`) rather than silently
5117            // dropped; capping here is what keeps the encoders honest if a
5118            // path ever misses that gate.
5119            ObjectFormat::Legacy => bound.min(LibverBound::Earliest),
5120            ObjectFormat::Modern => bound,
5121        }
5122    }
5123
5124    /// The bound the message encoders see — dataspace, datatype, fill value,
5125    /// attribute.
5126    fn encoding_libver(&self) -> LibverBound {
5127        self.session_libver(LibverBound::Earliest)
5128    }
5129
5130    /// The data layout message version this file's bound calls for —
5131    /// `H5O_layout_ver_bounds[H5F_LOW_BOUND(f)]` (H5Dlayout.c:44), the term
5132    /// `H5D__chunk_set_info` weighs against the version a chunk *requires*
5133    /// (H5Dchunk.c:936, :1046).
5134    ///
5135    /// With no bound named the row is `H5F_LIBVER_V110`'s: this crate's
5136    /// default file uses the v1.10 chunk indexes, which is exactly what that
5137    /// row says and what no other row does (see the `libver` field for why the
5138    /// default is not `Earliest` here even though the datatype and superblock
5139    /// tables read it that way). A file whose superblock already places it on
5140    /// an older row takes that row instead — a reopened version-2 superblock
5141    /// is the `V18` row, whose layout version of 3 has no index-type field at
5142    /// all, so its appended chunked datasets go on the version-1 B-tree.
5143    fn layout_version_bound(&self) -> u8 {
5144        self.session_libver(LibverBound::V110).layout_version()
5145    }
5146
5147    /// The data layout version a chunk of `chunk_bytes` *requires* whatever
5148    /// the bound says — `version_req` in `H5D__chunk_set_info` (H5Dchunk.c:909).
5149    ///
5150    /// Only one thing raises it: a chunk over 4 GiB does not fit the version-4
5151    /// message's 32-bit stored-size field. The floor is the default the
5152    /// creation property list carries, `H5O_LAYOUT_VERSION_DEFAULT`
5153    /// (H5Oprivate.h:451), which is why a classic file's chunked dataset is a
5154    /// version-3 message rather than the version-1 its bound's row names.
5155    fn required_chunk_layout_version(chunk_bytes: u64) -> u8 {
5156        if chunk_bytes > u32::MAX as u64 {
5157            5
5158        } else {
5159            LAYOUT_VERSION_DEFAULT
5160        }
5161    }
5162
5163    /// Whether a new chunked dataset of this chunk size is indexed by one of
5164    /// the v1.10 indexes — extensible array, fixed array, v2 B-tree, single
5165    /// chunk or implicit — rather than by the version-1 B-tree.
5166    ///
5167    /// The gate `H5D__chunk_set_info` puts in front of the whole
5168    /// index-selection block (H5Dchunk.c:936): the bound's layout version
5169    /// reaches 4, or the chunk requires a version that does. Only inside it
5170    /// does the dataspace get to pick between the five; below it the layout
5171    /// message has no index-type field and the chunks go on the version-1
5172    /// B-tree. So the format decides before the shape does — a fixed shape
5173    /// covered by exactly one chunk takes the single-chunk index only on the
5174    /// near side of this gate.
5175    pub(crate) fn uses_v110_chunk_indexing(&self, chunk_bytes: u64) -> bool {
5176        self.layout_version_bound() >= 4 || Self::required_chunk_layout_version(chunk_bytes) >= 4
5177    }
5178
5179    /// Refuse an SWMR session this file's format cannot record.
5180    ///
5181    /// The two checks `H5F__start_swmr_write` opens with: the superblock must
5182    /// be at least version 3 (H5Fint.c:3814, hdf5_1.14.6 H5Fint.c:3751) — the
5183    /// only version with the status-flags field that says a writer is attached
5184    /// — and the low bound must be at least `H5F_LIBVER_V110` (H5Fint.c:3818),
5185    /// the oldest bound whose `HDF5_superblock_ver_bounds` row reaches version
5186    /// 3.
5187    ///
5188    /// Which of the two applies is the [`SuperblockVersion`] question. A
5189    /// reopened file already has its version and reopening never rewrites one,
5190    /// so the first check decides and the second cannot fail after it: the
5191    /// version-3 floor is `V110`. A file this writer created has no version on
5192    /// disk yet, so only the second is askable — and a caller who named no
5193    /// bound at all passes it, because nothing in such a file says the
5194    /// superblock may not be version 3 and SWMR is what makes it one.
5195    ///
5196    /// Named, not silently upgraded. libhdf5 upgrades in the one case where
5197    /// SWMR is asked for at *create* time (`H5F_ACC_SWMR_WRITE` raises the
5198    /// bound to V110 in `H5F__super_init`, H5Fsuper.c:1131); on the reopen
5199    /// path it refuses instead, and so does this.
5200    fn reject_swmr(&self) -> IoResult<()> {
5201        let why = match self.superblock_version {
5202            SuperblockVersion::Existing(version) if version >= SUPERBLOCK_V3 => return Ok(()),
5203            SuperblockVersion::Existing(version) => format!(
5204                "its superblock is version {version}, and reopening a file never \
5205                 rewrites that"
5206            ),
5207            SuperblockVersion::Chosen(_) if self.is_legacy() => {
5208                "it is in the classic (version-0/1 superblock) format that \
5209                 H5F_LIBVER_EARLIEST selects"
5210                    .to_string()
5211            }
5212            SuperblockVersion::Chosen(_) if self.libver.is_some_and(|b| b < LibverBound::V110) => {
5213                "it was asked for at a library-version bound below H5F_LIBVER_V110, \
5214                 whose superblock row is version 2"
5215                    .to_string()
5216            }
5217            SuperblockVersion::Chosen(_) => return Ok(()),
5218        };
5219        Err(crate::io::IoError::Unsupported(format!(
5220            "cannot start an SWMR session on this file: {why}, and SWMR needs a \
5221             version-3 superblock to record that a writer is attached; create the \
5222             file at H5F_LIBVER_V110 or newer"
5223        )))
5224    }
5225
5226    /// Whether this file is in the classic (version-0/1 superblock) format,
5227    /// whose groups store their links in symbol tables — either because it
5228    /// was reopened in it or because it was created at
5229    /// `H5F_LIBVER_EARLIEST`.
5230    pub(crate) fn is_legacy(&self) -> bool {
5231        self.legacy.is_some()
5232    }
5233
5234    /// The v1-B-tree "K" ranks in force for this file, from which every v1
5235    /// node's width is derived.
5236    ///
5237    /// A version-0/1 superblock records them in a field of its own and a
5238    /// version-2/3 one in a B-tree-K message in its superblock extension, so
5239    /// the file's generation says nothing about whether they are the defaults
5240    /// — `H5F__super_read` reads both into the same `H5F_shared_t`, and so
5241    /// does the reopen, into `btree`.
5242    fn btree_v1_config(&self) -> BTreeV1Config {
5243        self.btree
5244    }
5245
5246    /// Track and index creation order for the links and the attributes of
5247    /// every object created after this call — the equivalent of setting
5248    /// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` to
5249    /// `H5P_CRT_ORDER_TRACKED | H5P_CRT_ORDER_INDEXED` on the creation
5250    /// property lists those objects are made with.
5251    ///
5252    /// Objects already created keep the policy they were made under, exactly
5253    /// as libhdf5 keeps what their creation property list said. The root
5254    /// group is created with the file, so its policy comes from
5255    /// [`create_with_options`](Self::create_with_options) instead.
5256    pub fn set_track_order(&mut self, track: bool) {
5257        self.track_order = TrackOrder::uniform(track);
5258    }
5259
5260    /// Record the times of every object created after this call —
5261    /// `H5Pset_obj_track_times` on the creation property lists those objects
5262    /// are made with.
5263    ///
5264    /// Off by default, which is h5py's default and not libhdf5's: h5py's
5265    /// high-level API sets `track_times=False` on every object it makes
5266    /// (`_hl/files.py:189`, `_hl/dataset.py:39`, `_hl/group.py:42`), while a
5267    /// bare creation property list leaves it on (`H5O_CRT_OHDR_FLAGS_DEF` is
5268    /// `H5O_HDR_STORE_TIMES`, H5Opkg.h:74). A caller after libhdf5's own
5269    /// bytes turns it on here.
5270    ///
5271    /// Objects already created keep the policy they were made under, and the
5272    /// root group takes its own from
5273    /// [`create_with_options`](Self::create_with_options) — the same split
5274    /// [`set_track_order`](Self::set_track_order) has, and for the same
5275    /// reason: this is a creation property, not a file-wide setting.
5276    pub fn set_track_times(&mut self, track: bool) {
5277        self.track_times = track;
5278    }
5279
5280    /// The times an object created right now records — all four set to the
5281    /// current time, as `H5O_apply_ohdr` initialises them (H5Oint.c:411-414),
5282    /// or `None` when this session is not tracking times.
5283    ///
5284    /// INVARIANT: every object this writer registers takes its `times` from
5285    /// here. The policy belongs to the creation property list, so reading
5286    /// [`track_times`](Self::track_times) at any later moment — a finalize, a
5287    /// header rewrite — would stamp a policy the object was not made under.
5288    fn created_object_times(&self) -> Option<ObjectTimes> {
5289        self.track_times
5290            .then(|| ObjectTimes::created_at(now_seconds()))
5291    }
5292
5293    /// Layout message version for a new chunked dataset on one of the v1.10
5294    /// indexes — `H5D__chunk_set_info`'s closing
5295    /// `MAX3(layout->version, version_req, MIN(bound, version_perf))`
5296    /// (H5Dchunk.c:1046).
5297    ///
5298    /// Version 5 is *required* for a chunk over 4 GiB (pre-2.0 readers cannot
5299    /// handle one even though the v4 wire format could express it) and
5300    /// *preferred* for filtered chunks, which is why it takes the file's
5301    /// bound to get there: the preference is capped by the bound's own row,
5302    /// so only the 2.0 format lets it through. Everything else stays at
5303    /// version 4, which every 1.10+ reader accepts.
5304    fn chunk_layout_version(&self, filtered: bool, chunk_bytes: u64) -> u8 {
5305        // `version_perf`: 4 for the v1.10 indexes as such, 5 when a filter
5306        // can make a chunk expand past what version 4 can record.
5307        let preferred = if filtered { 5 } else { 4 };
5308        Self::required_chunk_layout_version(chunk_bytes)
5309            .max(self.layout_version_bound().min(preferred))
5310            .max(LAYOUT_VERSION_DEFAULT)
5311    }
5312
5313    /// Width of the stored-chunk-size field in a filtered chunk index:
5314    /// version 5 uses the fixed `sizeof_size`; version 4 derives it from the
5315    /// uncompressed chunk byte count (one spare byte included), the
5316    /// `H5D_*_COMPUTE_CHUNK_SIZE_LEN` rule shared by the extensible-array,
5317    /// fixed-array and v2-B-tree indexes.
5318    fn chunk_size_len_for(&self, layout_version: u8, chunk_bytes: u64) -> u8 {
5319        if layout_version >= 5 {
5320            self.ctx.sizeof_size
5321        } else {
5322            compute_chunk_size_len(chunk_bytes)
5323        }
5324    }
5325
5326    /// Provide public access to the format context.
5327    pub fn ctx(&self) -> &FormatContext {
5328        &self.ctx
5329    }
5330
5331    /// Number of dataset slots in the registry (including soft-deleted ones).
5332    pub(crate) fn dataset_count(&self) -> usize {
5333        self.datasets.lock().len()
5334    }
5335
5336    /// Clone out the [`DatasetRef`] for `index`, releasing the registry lock
5337    /// immediately. Lock the returned ref to read or mutate that one dataset.
5338    ///
5339    /// Panics on an out-of-range index, exactly like the `Vec` indexing it
5340    /// replaces; bounds-checking callers consult [`Self::dataset_count`] first.
5341    ///
5342    /// MUST NOT be called while the registry [`Slot`] is already locked (it
5343    /// would deadlock the `threadsafe` mutex / panic the single-thread
5344    /// `RefCell`): collect the refs you need, drop the registry guard, then work.
5345    pub(crate) fn ds(&self, index: usize) -> DatasetRef {
5346        Shared::clone(&self.datasets.lock()[index])
5347    }
5348
5349    /// Number of group slots in the registry (including soft-deleted ones).
5350    pub(crate) fn group_count(&self) -> usize {
5351        self.groups.lock().len()
5352    }
5353
5354    /// Clone out the [`GroupRef`] for `index`. Same contract as [`Self::ds`].
5355    pub(crate) fn grp(&self, index: usize) -> GroupRef {
5356        Shared::clone(&self.groups.lock()[index])
5357    }
5358
5359    /// Enter the create gate: take `create_lock` and check that `name` is not
5360    /// already taken. The returned witness is what [`Self::push_dataset`]
5361    /// requires, so the uniqueness check and the registry push are atomic
5362    /// (see `create_lock`) at every creator by construction.
5363    pub(crate) fn begin_create(&self, name: &str) -> IoResult<CreateGuard<'_>> {
5364        let gate = self.create_lock.lock();
5365        // A creation path through hard links lands in the link's target
5366        // group, as HDF5 traversal does. Canonicalizing here — the one
5367        // entry every creator passes — keeps alias forms out of the
5368        // registry.
5369        let name = self.canonical_dataset_path(name);
5370        // A path that leaves this file, or that runs into an object the
5371        // reopen kept verbatim, is refused here rather than at each creator:
5372        // this is the one gate every creation passes, so a creator added
5373        // later cannot forget the check. Both run before the parent lookup,
5374        // which would otherwise report the group such a path names as absent
5375        // instead of naming what stops the path. Uniqueness comes first among
5376        // them: a name already in the file is taken whatever holds it.
5377        self.reject_external_traversal(&name)?;
5378        self.ensure_name_free(&name)?;
5379        self.reject_preserved_object(&name)?;
5380        let (parent, _leaf) = self.split_parent(&name)?;
5381        Ok(CreateGuard {
5382            _gate: gate,
5383            name,
5384            parent,
5385        })
5386    }
5387
5388    /// Split an object path into the group that will hold its link and the
5389    /// leaf link name, resolving every component through the group registry.
5390    ///
5391    /// `path` is the registry form — no leading `/`, e.g. `"grp/sub/late"`.
5392    /// This is what keeps a `/` out of a link name: HDF5 link names are
5393    /// single path components (`H5G_traverse` splits on `/` before it ever
5394    /// reaches `H5L_link`), so a name that carries a path must name a group
5395    /// that exists, or be refused.
5396    ///
5397    /// A missing component is an error rather than an implicit group: the
5398    /// default link creation property list has `H5Pset_create_intermediate_group`
5399    /// off, and this writer exposes no property list to turn it on with.
5400    fn split_parent(&self, path: &str) -> IoResult<(Option<usize>, String)> {
5401        let (parent_path, leaf) = path.rsplit_once('/').unwrap_or(("", path));
5402        if leaf.is_empty() {
5403            return Err(crate::io::IoError::InvalidState(format!(
5404                "'{path}' does not end in a link name"
5405            )));
5406        }
5407        if parent_path.is_empty() {
5408            return Ok((None, leaf.to_string()));
5409        }
5410        let abs = format!("/{parent_path}");
5411        let groups = self.group_refs();
5412        let idx = groups
5413            .iter()
5414            .position(|g| {
5415                let gg = g.lock();
5416                gg.name == abs && !gg.deleted
5417            })
5418            .ok_or_else(|| {
5419                crate::io::IoError::NotFound(format!(
5420                    "cannot create '{path}': group '{abs}' does not exist"
5421                ))
5422            })?;
5423        Ok((Some(idx), leaf.to_string()))
5424    }
5425
5426    /// Push a freshly-built dataset into the registry and return its index.
5427    /// Takes the registry lock only for the push, so it does not block an
5428    /// in-flight write that already cloned its own [`DatasetRef`] out.
5429    /// The [`CreateGuard`] proves the caller entered through
5430    /// [`Self::begin_create`] and still holds the gate.
5431    pub(crate) fn push_dataset(&self, create: &CreateGuard<'_>, info: DatasetInfo) -> usize {
5432        let name = info.name.clone();
5433        let idx = {
5434            let mut reg = self.datasets.lock();
5435            let idx = reg.len();
5436            reg.push(Shared::new(DatasetCell::new(info)));
5437            idx
5438        };
5439        self.register_name(&name, NameHit::Dataset(idx));
5440        // The spine guard is dropped before the group slot is taken: the lock
5441        // order is spine -> slot and never the reverse.
5442        if let Some(pidx) = create.parent {
5443            self.grp(pidx).lock().child_datasets.push(idx);
5444        }
5445        idx
5446    }
5447
5448    /// Push a freshly-built group into the registry and return its index.
5449    pub(crate) fn push_group(&self, info: GroupInfo) -> usize {
5450        let name = info.name.trim_start_matches('/').to_string();
5451        let idx = {
5452            let mut reg = self.groups.lock();
5453            let idx = reg.len();
5454            reg.push(Shared::new(Slot::new(info)));
5455            idx
5456        };
5457        self.register_name(&name, NameHit::Group(idx));
5458        idx
5459    }
5460
5461    /// Snapshot every [`DatasetRef`] (spine lock held only for the clone).
5462    /// Iterate the snapshot to lock each dataset one at a time — this keeps
5463    /// the lock order *spine → slot* and never reacquires the spine while a
5464    /// slot is held, which is what makes the registry deadlock-free.
5465    pub(crate) fn dataset_refs(&self) -> Vec<DatasetRef> {
5466        self.datasets.lock().iter().map(Shared::clone).collect()
5467    }
5468
5469    /// Snapshot every [`GroupRef`]; see [`Self::dataset_refs`].
5470    pub(crate) fn group_refs(&self) -> Vec<GroupRef> {
5471        self.groups.lock().iter().map(Shared::clone).collect()
5472    }
5473
5474    /// Snapshot the hard-link list (the lock is held only for the clone), so
5475    /// callers can resolve each link's target/parent — which locks dataset and
5476    /// group slots — without holding the hard-link lock.
5477    /// The next creation sequence number.
5478    ///
5479    /// One monotonic counter for datasets, groups and hard links alike: a
5480    /// group orders its links by it, so an interleaved run of `create_group`
5481    /// and `create_dataset` comes back out in the order it was made rather
5482    /// than grouped by kind.
5483    fn take_creation_seq(&self) -> u64 {
5484        let mut next = self.next_creation_seq.lock();
5485        let seq = *next;
5486        *next += 1;
5487        seq
5488    }
5489
5490    pub(crate) fn hard_links_vec(&self) -> Vec<HardLink> {
5491        self.hard_links.lock().clone()
5492    }
5493
5494    /// Snapshot the symbolic-link list; see [`Self::hard_links_vec`].
5495    pub(crate) fn symbolic_links_vec(&self) -> Vec<SymbolicLink> {
5496        self.symbolic_links.lock().clone()
5497    }
5498
5499    /// Open an existing HDF5 file for appending new datasets, using the
5500    /// env-var-derived locking policy.
5501    ///
5502    /// Reads existing dataset object headers fully, reconstructing metadata
5503    /// for chunked datasets so that `write_chunk` and `extend_dataset` work
5504    /// on reopened datasets.
5505    pub fn open_append(path: &Path) -> IoResult<Self> {
5506        Self::open_append_with_locking(
5507            path,
5508            crate::io::locking::FileLocking::from_env_or(Default::default()),
5509        )
5510    }
5511
5512    /// Carry a reopened file's shared-message table into the writer's model:
5513    /// the index specifications the file was created with, and every block the
5514    /// table occupies so the finalize that replaces it can give them back.
5515    ///
5516    /// `H5SM_init` fixes the index count, each index's type mask, its minimum
5517    /// message size and the file-wide phase-change pair when the file is
5518    /// created, and nothing afterwards changes any of them — they are file
5519    /// creation properties. So the master table on disk *is* the
5520    /// [`SharedMessageConfig`] the file was made with, read back.
5521    ///
5522    /// Returns `None` for a file with no shared-message table, which is every
5523    /// file libhdf5 writes without `H5Pset_shared_mesg_nindexes`.
5524    /// Read the free-space managers a reopened file persists, if it does.
5525    ///
5526    /// `H5F__super_read` copies the file-space info message's addresses into
5527    /// `f->shared->fs_addr[]` and the library opens each manager lazily; this
5528    /// reads them all at once, because the writer needs the whole section set
5529    /// before it allocates anything.
5530    ///
5531    /// Returns `None` — nothing read, nothing to write back — for a file with
5532    /// no file-space info message, one that does not persist, and one whose
5533    /// strategy keeps no managers at all.
5534    fn reopen_free_space(
5535        handle: &mut FileHandle,
5536        meta: &crate::io::FileMeta,
5537        ext: &crate::io::reader::SuperblockExtension,
5538    ) -> IoResult<ReopenedFreeSpace> {
5539        let none = || ReopenedFreeSpace {
5540            state: None,
5541            sections: Vec::new(),
5542        };
5543        let Some(info) = ext.file_space_info.as_ref().filter(|i| i.persist) else {
5544            return Ok(none());
5545        };
5546        if !matches!(
5547            info.strategy,
5548            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
5549        ) {
5550            return Ok(none());
5551        }
5552        let found = crate::io::free_space_io::read_managers(handle, &meta.ctx, info)?;
5553        Ok(ReopenedFreeSpace {
5554            state: Some(Box::new(FileSpaceState {
5555                info: info.clone(),
5556                superseded: found.blocks,
5557            })),
5558            sections: found.sections,
5559        })
5560    }
5561
5562    fn reopen_shared_messages(
5563        handle: &mut FileHandle,
5564        meta: &crate::io::FileMeta,
5565        ext: &crate::io::reader::SuperblockExtension,
5566    ) -> IoResult<Option<Box<SohmState>>> {
5567        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
5568        use crate::format::fractal_heap::collect_heap_extents;
5569        use crate::format::sohm::{list_size, SohmMasterTable, SOHM_INDEX_LIST};
5570
5571        let (Some(table), Some(smt)) = (
5572            meta.sohm.as_ref().filter(|t| !t.indexes.is_empty()),
5573            ext.shared_message_table.as_ref(),
5574        ) else {
5575            return Ok(None);
5576        };
5577        let ctx = &meta.ctx;
5578
5579        // The extension header itself is superseded by `CarriedExtension`,
5580        // which owns it whether or not the file has shared messages; what is
5581        // superseded here is only the storage the table message names.
5582        let mut superseded = Vec::new();
5583        superseded.push((
5584            smt.table_address,
5585            SohmMasterTable::encoded_size(ctx, smt.nindexes) as u64,
5586        ));
5587
5588        let mut specs = Vec::with_capacity(table.indexes.len());
5589        for index in &table.indexes {
5590            specs.push(SohmIndexSpec {
5591                mesg_types: index.mesg_types,
5592                min_mesg_size: index.min_mesg_size,
5593                list_max: index.list_max,
5594                btree_min: index.btree_min,
5595            });
5596            let mut reader = crate::io::reader::HandleBlockReader { handle };
5597            if index.heap_addr != UNDEF_ADDR {
5598                superseded.extend(collect_heap_extents(index.heap_addr, ctx, &mut reader)?);
5599            }
5600            if index.index_addr != UNDEF_ADDR {
5601                if index.index_type == SOHM_INDEX_LIST {
5602                    // `H5SM_LIST_SIZE`: the block is sized for `list_max`
5603                    // records however few are in it.
5604                    superseded.push((index.index_addr, list_size(ctx, index.list_max) as u64));
5605                } else {
5606                    superseded.extend(collect_btree_v2_extents(
5607                        index.index_addr,
5608                        ctx,
5609                        &mut reader,
5610                    )?);
5611                }
5612            }
5613        }
5614        Ok(Some(Box::new(SohmState::new(specs, superseded))))
5615    }
5616
5617    /// Open an existing HDF5 file for appending with an explicit locking
5618    /// policy.
5619    pub fn open_append_with_locking(
5620        path: &Path,
5621        locking: crate::io::locking::FileLocking,
5622    ) -> IoResult<Self> {
5623        let mut handle = FileHandle::open_readwrite_with_locking(path, locking)?;
5624        // The same `H5FD_locate_signature` search the read path makes, through
5625        // the same handle mechanism: the offset it finds is the file's base
5626        // address, so the allocator's end-of-file, every write and the
5627        // superblock rewrite all work in the HDF5 address space, and the
5628        // userblock in `[0, base)` is not addressable from this writer at all.
5629        let super_addr = handle
5630            .locate_signature()?
5631            .ok_or(crate::format::FormatError::InvalidSignature)?;
5632        handle.set_base(super_addr);
5633        let file_size = handle.file_size()?;
5634
5635        let sb_buf = handle.read_at_most(0, 256)?;
5636        // Which generation the file is decides everything the close then
5637        // writes back: version-1 object headers and symbol-table groups over a
5638        // version-0/1 superblock, or version-2 headers and link-message groups
5639        // over a version-2/3 one. libhdf5 writes those two combinations and no
5640        // mixture of them, so the branch is taken once, here, and carried as
5641        // `legacy`.
5642        let version = crate::format::superblock::detect_superblock_version(&sb_buf)?;
5643        let (ctx, sb_btree, root_addr, ext_addr, legacy) = if version <= 1 {
5644            let sb = SuperblockV0V1::decode(&sb_buf)?;
5645            let ctx = FormatContext {
5646                sizeof_addr: sb.sizeof_offsets,
5647                sizeof_size: sb.sizeof_lengths,
5648            };
5649            // Unlike a v2/v3 superblock, a classic one carries the "K" ranks
5650            // itself; every v1-B-tree and symbol-table node width in the file
5651            // comes from them.
5652            let btree = crate::format::btree_v1::BTreeV1Config {
5653                sym_leaf_k: sb.sym_leaf_k,
5654                snode_internal_k: sb.btree_internal_k,
5655                chunk_internal_k: sb.indexed_storage_k.unwrap_or(32),
5656            };
5657            let root = sb.root_symbol_table_entry.obj_header_addr;
5658            let ext = sb.superblock_extension_address;
5659            (ctx, btree, root, ext, Some(sb))
5660        } else {
5661            let sb = SuperblockV2V3::decode(&sb_buf)?;
5662            let ctx = FormatContext {
5663                sizeof_addr: sb.sizeof_offsets,
5664                sizeof_size: sb.sizeof_lengths,
5665            };
5666            (
5667                ctx,
5668                crate::format::btree_v1::BTreeV1Config::default(),
5669                sb.root_group_object_header_address,
5670                sb.superblock_extension_address,
5671                None,
5672            )
5673        };
5674
5675        // The reopen reads object headers exactly as the reader does, so it
5676        // needs the same file-level parameters: a v2/v3 superblock carries no
5677        // B-tree K values, and only the extension can override the defaults.
5678        let (meta, ext) = crate::io::reader::Hdf5Reader::read_extension_and_meta(
5679            &mut handle,
5680            ctx,
5681            sb_btree,
5682            ext_addr,
5683        )?;
5684
5685        // A file with shared object header messages keeps datatypes,
5686        // dataspaces and attributes in a fractal heap per index, and each
5687        // object header holds a heap ID pointing at one. The table is laid out
5688        // whole from the whole message set (`build_shared_messages`), never
5689        // grown insert by insert, so a reopen carries the indexes and the
5690        // bodies forward and the next finalize lays a new table out over the
5691        // old one's blocks — which is sound exactly while no header keeping
5692        // its bytes still points into the old heap. The walk below is what
5693        // settles that.
5694        let sohm = Self::reopen_shared_messages(&mut handle, &meta, &ext)?;
5695
5696        // The extension is external truth this close rewrites, so what it held
5697        // is captured whole here — before anything else reads the file — and
5698        // re-emitted by `write_superblock_extension`. Read from the raw chain
5699        // rather than from `ext`, which keeps only the messages this crate
5700        // models.
5701        let extension = if ext_addr == UNDEF_ADDR || ext_addr == 0 {
5702            Box::<CarriedExtension>::default()
5703        } else {
5704            let (carried, blocks) = crate::io::object_header_io::superblock_extension_messages(
5705                &mut handle,
5706                &meta,
5707                ext_addr,
5708            )?;
5709            Box::new(CarriedExtension {
5710                superseded: blocks,
5711                carried,
5712                addr: Slot::new(None),
5713            })
5714        };
5715
5716        // The managers that extension's file-space info message names, read
5717        // before anything allocates: the sections they hold are file space
5718        // this session may hand out, and the close rewrites them.
5719        let reopened_free_space = Self::reopen_free_space(&mut handle, &meta, &ext)?;
5720
5721        // Discover links from root group (and subgroups recursively). Every
5722        // object is classified before it is registered, and the root is the
5723        // one object with no alternative: its header must be rewritten to
5724        // hold anything new, so an unmodellable root is refused here rather
5725        // than rewritten into whatever this writer could read of it.
5726        let mut walk = ReopenWalk::new(&mut handle, &meta);
5727        let root = match walk.plan(root_addr)? {
5728            ObjectPlan::Group(parts) => parts,
5729            ObjectPlan::Dataset(_) => {
5730                return Err(crate::io::IoError::InvalidState(
5731                    "cannot open this file for appending: its root object is a dataset, \
5732                     not a group"
5733                        .into(),
5734                ))
5735            }
5736            ObjectPlan::Preserve { why, .. } => {
5737                return Err(crate::io::IoError::Unsupported(format!(
5738                    "cannot open this file for appending: {why}. Every append rewrites the \
5739                     root group's header, and this writer will not rewrite it from the part \
5740                     of it that it can read"
5741                )));
5742            }
5743        };
5744        let root_header_blocks = root.header_blocks;
5745        let root_attributes = root.attributes;
5746        let root_track_order = root.track_order;
5747        let root_times = root.times;
5748        let root_dense = root.dense;
5749        let root_stab = root.stab;
5750
5751        walk.group(&root.links, "", 0)?;
5752        let collected = walk.finish();
5753        let mut link_entries = collected.hard;
5754        let mut preserved = collected.preserved;
5755        // Objects the loop below could not rebuild, by header address, so the
5756        // other links to one are preserved with it rather than left pointing
5757        // at a registry entry that is no longer there.
5758        let mut unrebuilt: std::collections::HashMap<u64, String> = Default::default();
5759
5760        // Two link entries can share one object header — hard links. Only
5761        // the first-walked path becomes the object; the rest are rebuilt
5762        // as hard-link registry entries further down. Without this split
5763        // every alias came back as its own DatasetInfo carrying the same
5764        // storage addresses, so deleting (or finalizing) one freed blocks
5765        // the others still referenced.
5766        let mut seen_header_addrs = std::collections::HashSet::new();
5767        let mut alias_entries: Vec<HardEntry> = Vec::new();
5768        link_entries.retain(|(entry, _)| {
5769            if seen_header_addrs.insert(entry.address) {
5770                true
5771            } else {
5772                alias_entries.push(entry.clone());
5773                false
5774            }
5775        });
5776
5777        // The order the walk met each object, kept before the loop below
5778        // consumes the entries: `ensure_groups_for` needs parents to precede
5779        // children.
5780        let walk_order: Vec<String> = link_entries.iter().map(|(e, _)| e.path.clone()).collect();
5781
5782        let mut existing_datasets = Vec::new();
5783        // Non-dataset link targets (groups): the header's chunk-0 address and
5784        // every block its chain occupies, by link path — so finalize can free
5785        // the blocks its rewrite supersedes — plus the attributes the header
5786        // carries, which the group registry below must keep or finalize
5787        // rewrites the group without them.
5788        type GroupHeaderInfo = (
5789            u64,
5790            crate::io::object_header_io::HeaderBlocks,
5791            Vec<AttributeEntry>,
5792            TrackOrder,
5793            Option<ObjectTimes>,
5794        );
5795        let mut group_headers: std::collections::HashMap<String, GroupHeaderInfo> =
5796            Default::default();
5797        // The dense storage each rebuilt dataset's header named, by registry
5798        // index, so finalize frees exactly what its rewrite supersedes. Keyed
5799        // after the rebuild succeeded: a preserved dataset keeps its header,
5800        // and freeing the heap that header still names would strand it.
5801        let mut dataset_dense: Vec<(usize, AttributeInfoMessage)> = Vec::new();
5802        let mut group_dense: Vec<(String, DenseCarry)> = Vec::new();
5803        // The same, for the symbol-table storage a classic group's header
5804        // names: keyed by path here, by registry index once every group has
5805        // one.
5806        let mut group_stabs: Vec<(String, StabExtents)> = Vec::new();
5807        for (entry, object) in link_entries {
5808            let HardEntry {
5809                path: name,
5810                address: obj_addr,
5811                encoded,
5812            } = entry;
5813            let parts = match object {
5814                CollectedObject::Group {
5815                    header_blocks,
5816                    attributes,
5817                    track_order,
5818                    times,
5819                    dense,
5820                    stab,
5821                } => {
5822                    group_dense.push((name.clone(), dense));
5823                    if let Some(stab) = stab {
5824                        group_stabs.push((name.clone(), stab));
5825                    }
5826                    group_headers.insert(
5827                        name,
5828                        (obj_addr, header_blocks, attributes, track_order, times),
5829                    );
5830                    continue;
5831                }
5832                CollectedObject::Dataset(parts) => *parts,
5833            };
5834            let dense_attrs = parts.dense.attrs.clone();
5835            match rebuild_dataset(&mut handle, &meta, file_size, name.clone(), obj_addr, parts) {
5836                Ok(info) => {
5837                    if let Some(ainfo) = dense_attrs {
5838                        dataset_dense.push((existing_datasets.len(), ainfo));
5839                    }
5840                    existing_datasets.push(info);
5841                }
5842                // Kept by its bytes for the same reason a header this walk
5843                // could not decode is: the rewrite would otherwise emit an
5844                // object whose chunk index no longer names its chunks.
5845                Err(e) => {
5846                    let why = format!("this writer could not rebuild its chunk index: {e}");
5847                    unrebuilt.insert(obj_addr, why.clone());
5848                    preserved.push(PreservedEntry {
5849                        path: name,
5850                        class: crate::io::reader::LinkClass::Hard,
5851                        encoded,
5852                        reason: Some(why),
5853                        // A dataset whose chunk index would not rebuild: the
5854                        // walk classified it, and it is not a datatype.
5855                        kind: PreservedKind::Unclassified,
5856                    });
5857                }
5858            }
5859        }
5860
5861        // Reconstruct the group registry. Every group is a link entry of its
5862        // own, whether or not a dataset lives under it, so the registry is
5863        // built from the discovered links — rebuilding it from dataset paths
5864        // alone made attribute-only and empty groups vanish at close, and
5865        // dropped the attributes of the groups that survived.
5866        let mut groups: Vec<GroupInfo> = Vec::new();
5867        let mut group_index_map: std::collections::HashMap<String, usize> =
5868            std::collections::HashMap::new();
5869
5870        // Register the chain of groups "/a", "/a/b", … for the link-style
5871        // path `link_path` ("a/b"), taking each one's on-disk header block
5872        // and attributes out of `group_headers` when the link walk saw it.
5873        fn ensure_groups_for(
5874            link_path: &str,
5875            groups: &mut Vec<GroupInfo>,
5876            group_index_map: &mut std::collections::HashMap<String, usize>,
5877            group_headers: &mut std::collections::HashMap<String, GroupHeaderInfo>,
5878        ) {
5879            let mut path = String::new();
5880            for part in link_path.split('/') {
5881                let parent_path = if path.is_empty() {
5882                    "/".to_string()
5883                } else {
5884                    path.clone()
5885                };
5886                if path.is_empty() {
5887                    path = format!("/{}", part);
5888                } else {
5889                    path = format!("{}/{}", path, part);
5890                }
5891                if group_index_map.contains_key(&path) {
5892                    continue;
5893                }
5894                let parent = if parent_path == "/" {
5895                    None
5896                } else {
5897                    group_index_map.get(&parent_path).copied()
5898                };
5899                let gidx = groups.len();
5900                let (obj_header_written_addr, obj_header_blocks, attributes, track_order, times) =
5901                    group_headers.remove(path.trim_start_matches('/')).map_or(
5902                        (None, Vec::new(), Vec::new(), TrackOrder::default(), None),
5903                        |(addr, blocks, attrs, track, times)| {
5904                            (Some(addr), blocks, attrs, track, times)
5905                        },
5906                    );
5907                groups.push(GroupInfo {
5908                    name: path.clone(),
5909                    parent,
5910                    creation_seq: 0,
5911                    track_order,
5912                    times,
5913                    child_datasets: Vec::new(),
5914                    child_groups: Vec::new(),
5915                    obj_header_addr: 0,
5916                    obj_header_written_addr,
5917                    obj_header_blocks,
5918                    deleted: false,
5919                    attributes,
5920                });
5921                if let Some(pidx) = parent {
5922                    groups[pidx].child_groups.push(gidx);
5923                }
5924                group_index_map.insert(path.clone(), gidx);
5925            }
5926        }
5927
5928        // Every linked group, in link-walk order (parents precede children).
5929        for name in &walk_order {
5930            if group_headers.contains_key(name.as_str()) {
5931                ensure_groups_for(name, &mut groups, &mut group_index_map, &mut group_headers);
5932            }
5933        }
5934
5935        // Assign each dataset to its immediate parent group, creating any
5936        // group the link walk could not decode (its chain stays placeholder).
5937        for (di, ds) in existing_datasets.iter().enumerate() {
5938            let parts: Vec<&str> = ds.name.split('/').collect();
5939            if parts.len() <= 1 {
5940                continue; // root-level dataset, no group
5941            }
5942            let parent_link_path = parts[..parts.len() - 1].join("/");
5943            ensure_groups_for(
5944                &parent_link_path,
5945                &mut groups,
5946                &mut group_index_map,
5947                &mut group_headers,
5948            );
5949            let gidx = group_index_map[&format!("/{}", parent_link_path)];
5950            groups[gidx].child_datasets.push(di);
5951        }
5952
5953        // An object the rebuild above gave up on is preserved by its bytes,
5954        // so the other links to it are preserved too: there is no registry
5955        // entry for them to name.
5956        alias_entries.retain(|entry| match unrebuilt.get(&entry.address) {
5957            None => true,
5958            Some(why) => {
5959                preserved.push(PreservedEntry {
5960                    path: entry.path.clone(),
5961                    class: crate::io::reader::LinkClass::Hard,
5962                    encoded: entry.encoded.clone(),
5963                    reason: Some(why.clone()),
5964                    kind: PreservedKind::Unclassified,
5965                });
5966                false
5967            }
5968        });
5969
5970        // The one thing a rebuilt shared-message table can break: an object
5971        // kept by its bytes keeps the heap IDs its header holds, and the
5972        // finalize gives the heap those IDs name back to the allocator. Every
5973        // object the registry holds is rewritten instead
5974        // ([`rebuilds_shared_messages`](Self::rebuilds_shared_messages)), so
5975        // this asks only the preserved ones, and names the object rather than
5976        // the feature — the file is appendable the moment nothing preserved
5977        // holds a heap ID or hides a subtree that might.
5978        if sohm.is_some() {
5979            for entry in &preserved {
5980                if !matches!(entry.class, crate::io::reader::LinkClass::Hard) {
5981                    continue;
5982                }
5983                let Ok((link, _)) = LinkMessage::decode(&entry.encoded, &meta.ctx) else {
5984                    continue;
5985                };
5986                let LinkTarget::Hard { address } = link.target else {
5987                    continue;
5988                };
5989                if let Some(blocks) = crate::io::object_header_io::blocks_shared_message_rebuild(
5990                    &mut handle,
5991                    &meta,
5992                    address,
5993                )? {
5994                    let why = entry
5995                        .reason
5996                        .as_deref()
5997                        .unwrap_or("this writer cannot model it");
5998                    return Err(crate::io::IoError::Unsupported(format!(
5999                        "cannot open this file for appending: '{}' {blocks}, but {why}, so \
6000                         its header keeps the bytes it has while the append lays the \
6001                         shared-message table out afresh",
6002                        entry.path
6003                    )));
6004                }
6005            }
6006        }
6007
6008        // Rebuild the hard-link registry from the alias entries set aside
6009        // above, so the H5Ldelete semantics survive a reopen. An alias whose
6010        // target the walk could not model is not here at all: it was
6011        // preserved by its own bytes, exactly as the first link to that
6012        // object was.
6013        let mut hard_links: Vec<HardLink> = Vec::new();
6014        for HardEntry {
6015            path,
6016            address: addr,
6017            ..
6018        } in alias_entries
6019        {
6020            let target = if let Some(di) = existing_datasets
6021                .iter()
6022                .position(|d| d.obj_header_addr == addr)
6023            {
6024                HardLinkTarget::Dataset(di)
6025            } else if let Some(gi) = groups
6026                .iter()
6027                .position(|g| g.obj_header_written_addr == Some(addr))
6028            {
6029                HardLinkTarget::Group(gi)
6030            } else {
6031                continue;
6032            };
6033            let (parent, link_name) = match path.rsplit_once('/') {
6034                None => (None, path),
6035                Some((dir, leaf)) => {
6036                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
6037                    (
6038                        group_index_map.get(&format!("/{dir}")).copied(),
6039                        leaf.to_string(),
6040                    )
6041                }
6042            };
6043            hard_links.push(HardLink {
6044                parent,
6045                name: link_name,
6046                target,
6047                creation_seq: 0,
6048            });
6049        }
6050
6051        // Attach every link the writer cannot express to the group that
6052        // holds it, so the rewrite of that group's header emits it again.
6053        // `ensure_groups_for` registers the parent chain, which matters for
6054        // a group whose only content is such a link: nothing else would put
6055        // it in the registry, and the close would drop group and link alike.
6056        let mut preserved_links: Vec<PreservedLink> = Vec::new();
6057        for PreservedEntry {
6058            path,
6059            class,
6060            encoded,
6061            reason,
6062            kind,
6063        } in preserved
6064        {
6065            let (parent, link_name) = match path.rsplit_once('/') {
6066                None => (None, path),
6067                Some((dir, leaf)) => {
6068                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
6069                    (
6070                        group_index_map.get(&format!("/{dir}")).copied(),
6071                        leaf.to_string(),
6072                    )
6073                }
6074            };
6075            preserved_links.push(PreservedLink {
6076                parent,
6077                name: link_name,
6078                class,
6079                encoded,
6080                reason,
6081                kind,
6082            });
6083        }
6084
6085        // Stamp the creation sequence a reopened file cannot supply. Nothing
6086        // on disk says which link was made first unless the group tracked
6087        // creation order, and this reader does not carry that back out, so
6088        // discovery order is what there is: datasets, then groups, then the
6089        // hard links found beside them — the order the writer emitted links
6090        // in before it ordered them at all.
6091        let mut creation_seq = 0u64;
6092        for d in &mut existing_datasets {
6093            d.creation_seq = creation_seq;
6094            creation_seq += 1;
6095        }
6096        for g in &mut groups {
6097            g.creation_seq = creation_seq;
6098            creation_seq += 1;
6099        }
6100        for l in &mut hard_links {
6101            l.creation_seq = creation_seq;
6102            creation_seq += 1;
6103        }
6104
6105        // The strategy is the file's, not this session's: a paged file
6106        // allocates on its own page grid however it was opened, `persist`
6107        // deciding only whether the managers survive the close.
6108        let allocator = FileAllocator::with_policy(
6109            file_size,
6110            ext.file_space_info
6111                .as_ref()
6112                .map_or(free_space::SpacePolicy::Aggr, |info| {
6113                    free_space::SpacePolicy::for_message(info)
6114                }),
6115        );
6116        // The sections the file's own managers recorded are free space, so
6117        // they are what this session allocates from first — `H5MF_alloc` asks
6118        // the free-space manager before it bumps the end of the file, and a
6119        // reopen that skipped this would grow a file that had room.
6120        allocator.reset_free_list(&reopened_free_space.sections);
6121
6122        // Now that every object has its registry index, key the dense storage
6123        // found on disk by the scope that will supersede it. A group the link
6124        // walk saw but never registered is not rewritten either, so leaving it
6125        // out is what keeps its storage referenced.
6126        let mut superseded = SupersededDense {
6127            attrs: dataset_dense
6128                .into_iter()
6129                .map(|(di, ainfo)| (AttrScope::Dataset(di), ainfo))
6130                .collect(),
6131            links: HashMap::new(),
6132        };
6133        superseded
6134            .attrs
6135            .extend(root_dense.attrs.map(|a| (AttrScope::Root, a)));
6136        superseded
6137            .links
6138            .extend(root_dense.links.map(|l| (LinkScope::Root, l)));
6139        for (name, dense) in group_dense {
6140            let Some(&gidx) = group_index_map.get(&format!("/{name}")) else {
6141                continue;
6142            };
6143            superseded
6144                .attrs
6145                .extend(dense.attrs.map(|a| (AttrScope::Group(gidx), a)));
6146            superseded
6147                .links
6148                .extend(dense.links.map(|l| (LinkScope::Group(gidx), l)));
6149        }
6150        let superseded = (!superseded.attrs.is_empty() || !superseded.links.is_empty())
6151            .then(|| Box::new(superseded));
6152
6153        // The same keying for the symbol-table storage. Built from the headers
6154        // alone, not from the superblock version: a group whose header carried
6155        // no Symbol Table message contributes nothing — what happens to a group
6156        // libhdf5 wrote at a newer bound inside an otherwise classic file — and
6157        // one that carried it keeps its storage even where the superblock is
6158        // version 2, which is what a file with shared messages is.
6159        let mut stabs: HashMap<LinkScope, StabExtents> = HashMap::new();
6160        stabs.extend(root_stab.map(|s| (LinkScope::Root, s)));
6161        for (name, extents) in group_stabs {
6162            if let Some(&gidx) = group_index_map.get(&format!("/{name}")) {
6163                stabs.insert(LinkScope::Group(gidx), extents);
6164            }
6165        }
6166        let symbol_tables = SymbolTables {
6167            found: stabs.keys().copied().collect(),
6168            superseded: Slot::new(stabs),
6169            written: Slot::new(HashMap::new()),
6170        };
6171
6172        // The superblock the close re-emits, and the generation every message
6173        // this session encodes belongs to.
6174        let legacy = legacy.map(|superblock| Box::new(LegacyFile { superblock }));
6175
6176        // Wrap the reconstructed plain vecs into the per-slot registry. The
6177        // reconstruction logic above runs single-threaded on local `Vec`s;
6178        // only the final hand-off needs the `Shared<Slot<_>>` shape.
6179        let datasets = existing_datasets
6180            .into_iter()
6181            .map(|i| Shared::new(DatasetCell::new(i)))
6182            .collect();
6183        let groups = groups
6184            .into_iter()
6185            .map(|g| Shared::new(Slot::new(g)))
6186            .collect();
6187
6188        let writer = Self {
6189            handle,
6190            allocator,
6191            ctx,
6192            datasets: Slot::new(datasets),
6193            groups: Slot::new(groups),
6194            hard_links: Slot::new(hard_links),
6195            // A reopen carries the soft and external links it found as
6196            // `preserved_links`, byte for byte; this list holds only the ones
6197            // created in this session.
6198            symbolic_links: Slot::new(Vec::new()),
6199            committed_datatypes: Slot::new(Vec::new()),
6200            preserved_links: Slot::new(preserved_links),
6201            name_index: Slot::new(Box::new(NameIndex::new())),
6202            root_attributes: Slot::new(root_attributes),
6203            create_lock: Slot::new(()),
6204            // A reopen names no bound: the file already is whichever
6205            // generation it is, and the version in its superblock is what
6206            // says so — see `libver_floor`. `set_libver_bound` is where a
6207            // caller asks for a newer one, exactly as `H5Fopen` takes a fapl.
6208            libver: None,
6209            closed: false,
6210            swmr_active: false,
6211            cwfs: Slot::new(Vec::new()),
6212            root_group_addr: None,
6213            superseded_root_header: root_header_blocks,
6214            // The version the file already has. It is written back unchanged
6215            // and it floors every bound this session writes at, so the append
6216            // hands the file back in the generation it found it in.
6217            superblock_version: SuperblockVersion::Existing(version),
6218            // The reopened file's own policy, so objects added in this
6219            // session are made the way the file already declares.
6220            root_track_order,
6221            root_times,
6222            dense_attributes: Slot::new(HashMap::new()),
6223            dense_links: Slot::new(HashMap::new()),
6224            superseded_dense: Slot::new(superseded),
6225            track_order: root_track_order,
6226            // Not recovered from the file the way the creation-order policy
6227            // is: a version-1 header leaves no trace of whether the object was
6228            // tracking times, so there is nothing on disk to read the policy
6229            // back from. An object added to a reopened file gets this writer's
6230            // own default, the same one a created file starts at.
6231            track_times: false,
6232            next_creation_seq: Slot::new(creation_seq),
6233            pending_object_references: Slot::new(Vec::new()),
6234            pending_heap_references: Slot::new(Vec::new()),
6235            attribute_references: Slot::new(Vec::new()),
6236            legacy,
6237            symbol_tables,
6238            // The ranks the superblock or its extension declared, which every
6239            // v1-B-tree and symbol-table node this session writes is sized by.
6240            btree: meta.btree,
6241            extension,
6242            free_space: reopened_free_space.state,
6243            // The indexes the file was created with, and the blocks its
6244            // current table occupies; the next finalize lays a new table out
6245            // over them from the whole message set.
6246            sohm,
6247            source_dir: source_dir_of(path)?,
6248        };
6249        // The link graph is complete only now, so this is the first point the
6250        // count each on-disk header was written with can be read off it: in a
6251        // well-formed file the links the walk found reaching an object *are*
6252        // that count, so nothing has to be decoded out of the headers.
6253        for i in 0..writer.dataset_count() {
6254            let nlink = writer.object_link_count(HardLinkTarget::Dataset(i));
6255            writer.ds(i).lock().nlink_written = nlink;
6256        }
6257        Ok(writer)
6258    }
6259
6260    /// Return the names of all datasets created so far.
6261    pub fn dataset_names(&self) -> Vec<String> {
6262        self.dataset_refs()
6263            .iter()
6264            .filter_map(|d| {
6265                let g = d.lock();
6266                (!g.deleted).then(|| g.name.clone())
6267            })
6268            .collect()
6269    }
6270
6271    /// Find a dataset index by name. Like `H5Dopen`, the name may be any
6272    /// link path to the dataset: a user hard link's path — or a path
6273    /// whose group components pass through such links — resolves to its
6274    /// target.
6275    pub fn dataset_index(&self, name: &str) -> Option<usize> {
6276        let name = self.canonical_dataset_path(name);
6277        self.dataset_refs()
6278            .iter()
6279            .position(|d| {
6280                let g = d.lock();
6281                g.name == name && !g.deleted
6282            })
6283            .or_else(|| {
6284                self.hard_links_vec().iter().find_map(|l| match l.target {
6285                    HardLinkTarget::Dataset(i)
6286                        if self.hard_link_emitted(l) && self.hard_link_full_path(l) == name =>
6287                    {
6288                        Some(i)
6289                    }
6290                    _ => None,
6291                })
6292            })
6293    }
6294
6295    /// Reconstruct the fields a writer-mode `H5Dataset` handle needs for the
6296    /// dataset at `index`, and open it under `access`. Single owner of this
6297    /// mapping so `H5File::dataset_writer`, `H5Group::dataset_writer`, and
6298    /// the vlen-string helpers all agree — including on
6299    /// [`bind_efile_prefix`](Self::bind_efile_prefix), which no handle site
6300    /// can then forget to run.
6301    pub(crate) fn dataset_handle_parts(
6302        &self,
6303        index: usize,
6304        access: &DatasetAccess,
6305    ) -> IoResult<DatasetHandleParts> {
6306        let open = self.bind_efile_prefix(index, access)?;
6307        let ds = self.ds(index);
6308        let g = ds.lock();
6309        Ok(DatasetHandleParts {
6310            shape: g.dataspace.dims.iter().map(|&d| d as usize).collect(),
6311            element_size: g.datatype.element_size() as usize,
6312            chunk_index: g.chunk_index_kind(),
6313            open,
6314        })
6315    }
6316
6317    /// Put `access`'s external file prefix in force for the dataset at
6318    /// `index`, or join the open that already settled one.
6319    ///
6320    /// INVARIANT: every write of an externally stored dataset's raw bytes
6321    /// joins its slot names against the prefix an *open* settled, and this is
6322    /// the only place that settles one. `write_contiguous_bytes` reads it and
6323    /// nothing else writes it, so a write cannot resolve a prefix of its own
6324    /// and land bytes where a read under the same properties would not look
6325    /// for them.
6326    ///
6327    /// First open wins, and a joining open may not disagree: `H5D__open_name`
6328    /// compares its own expanded prefix against the open dataset's and fails
6329    /// when they differ (H5Dint.c:1533-1545). Measured under libhdf5 1.14.6
6330    /// and 2.0.0, with a dataset created through a dapl naming a directory
6331    /// and its handle still alive: a second open naming another directory is
6332    /// refused, one naming the same directory joins, one naming none is
6333    /// refused too, and with `HDF5_EXTFILE_PREFIX` set — which shadows every
6334    /// property, so all three expand alike — none of them is. Dropping every
6335    /// handle releases the answer and the next open settles it afresh, which
6336    /// the same measurement confirms.
6337    ///
6338    /// Returns the token that keeps the open alive, `None` for a dataset
6339    /// whose raw data is in this file and which therefore has no prefix to
6340    /// agree about.
6341    pub(crate) fn bind_efile_prefix(
6342        &self,
6343        index: usize,
6344        access: &DatasetAccess,
6345    ) -> IoResult<Option<crate::io::reader::DatasetOpenToken>> {
6346        let ds = self.ds(index);
6347        let mut g = ds.lock();
6348        let source_dir = &self.source_dir;
6349        let Some(ext) = g.external.as_mut() else {
6350            return Ok(None);
6351        };
6352        let want =
6353            crate::io::reader::resolve_extfile_prefix(access.efile_prefix_value(), source_dir);
6354        if let Some(open) = ext.prefix.open.upgrade() {
6355            if ext.prefix.expanded != want {
6356                let name = g.name.clone();
6357                return Err(crate::io::IoError::InvalidState(format!(
6358                    "dataset {name:?} is already open under a different external file                      prefix, and libhdf5 refuses to join an open that disagrees about one"
6359                )));
6360            }
6361            return Ok(Some(open));
6362        }
6363        let token: crate::io::reader::DatasetOpenToken = std::sync::Arc::new(());
6364        ext.prefix = EfilePrefix {
6365            expanded: want,
6366            open: std::sync::Arc::downgrade(&token),
6367        };
6368        Ok(Some(token))
6369    }
6370
6371    /// Reject a name some other link in the file already occupies.
6372    ///
6373    /// `name` is the registry's full-path form, with no leading `/`. HDF5
6374    /// requires link names to be unique within their group, and every kind of
6375    /// link this writer can emit competes for the same name: a dataset's own
6376    /// link, a group's, a user hard link, a soft or external link, and a link
6377    /// a reopen is carrying through verbatim. This is the one place that list
6378    /// is written down, so a creator cannot be blind to a kind it does not
6379    /// itself make — nor a kind added after it.
6380    fn ensure_name_free(&self, name: &str) -> IoResult<()> {
6381        let holder = self.name_holder(name);
6382        // The index is a filter over the registries, not a second copy of
6383        // them, so a debug build re-derives the answer on every create: a
6384        // name it failed to record surfaces as a failing assertion in the
6385        // suite rather than as two links of one name in somebody's file.
6386        #[cfg(debug_assertions)]
6387        assert_eq!(
6388            holder,
6389            self.scan_name_holder(name),
6390            "the name index disagrees with the registries for '{name}'"
6391        );
6392        match holder {
6393            None => Ok(()),
6394            Some(kind) => Err(crate::io::IoError::InvalidState(format!(
6395                "a {kind} named '{name}' already exists"
6396            ))),
6397        }
6398    }
6399
6400    /// What already holds `name`, or `None` if it is free.
6401    ///
6402    /// The kinds answer in a fixed order — dataset, group, committed
6403    /// datatype, hard link, symbolic link, preserved link — because the
6404    /// refusal names the first one that holds it. [`NameIndex`] narrows each
6405    /// kind to the entries that ever took this name; every candidate is then
6406    /// put through the same predicate the full scan used, so a hit left
6407    /// behind by a delete or a rename answers exactly as an absent one does.
6408    fn name_holder(&self, name: &str) -> Option<&'static str> {
6409        self.build_name_index();
6410        let hits: Vec<NameHit> = {
6411            let index = self.name_index.lock();
6412            index.map.as_ref().and_then(|m| m.get(name))?.clone()
6413        };
6414        for hit in &hits {
6415            if let NameHit::Dataset(i) = *hit {
6416                let ds = self.ds(i);
6417                let d = ds.lock();
6418                if !d.deleted && d.name == name {
6419                    return Some("dataset");
6420                }
6421            }
6422        }
6423        for hit in &hits {
6424            if let NameHit::Group(i) = *hit {
6425                let grp = self.grp(i);
6426                let g = grp.lock();
6427                if !g.deleted && g.name.trim_start_matches('/') == name {
6428                    return Some("group");
6429                }
6430            }
6431        }
6432        for hit in &hits {
6433            if let NameHit::Datatype(i) = *hit {
6434                // The registry lock goes before `parent_alive` takes a group
6435                // slot, never across it.
6436                let (parent, held) = {
6437                    let reg = self.committed_datatypes.lock();
6438                    (reg[i].parent, reg[i].name == name)
6439                };
6440                if held && self.parent_alive(parent) {
6441                    return Some("committed datatype");
6442                }
6443            }
6444        }
6445        if hits.contains(&NameHit::HardLink)
6446            && self
6447                .hard_links_vec()
6448                .iter()
6449                .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6450        {
6451            return Some("hard link");
6452        }
6453        if hits.contains(&NameHit::SymbolicLink)
6454            && self
6455                .symbolic_links_vec()
6456                .iter()
6457                .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6458        {
6459            return Some("link");
6460        }
6461        // A preserved link occupies its name in the group just as a modelled
6462        // one does; both are emitted, and two link messages of one name in a
6463        // group is an invalid file.
6464        if hits.contains(&NameHit::PreservedLink)
6465            && self.preserved_link_paths().iter().any(|(p, _)| *p == name)
6466        {
6467            return Some("link");
6468        }
6469        None
6470    }
6471
6472    /// The same answer read straight off the registries, which is what the
6473    /// index is checked against in a debug build.
6474    #[cfg(debug_assertions)]
6475    fn scan_name_holder(&self, name: &str) -> Option<&'static str> {
6476        if self.dataset_refs().iter().any(|d| {
6477            let g = d.lock();
6478            !g.deleted && g.name == name
6479        }) {
6480            return Some("dataset");
6481        }
6482        if self.group_refs().iter().any(|g| {
6483            let gg = g.lock();
6484            !gg.deleted && gg.name.trim_start_matches('/') == name
6485        }) {
6486            return Some("group");
6487        }
6488        if self
6489            .committed_datatypes_vec()
6490            .iter()
6491            .any(|c| self.parent_alive(c.parent) && c.name == name)
6492        {
6493            return Some("committed datatype");
6494        }
6495        if self
6496            .hard_links_vec()
6497            .iter()
6498            .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6499        {
6500            return Some("hard link");
6501        }
6502        if self
6503            .symbolic_links_vec()
6504            .iter()
6505            .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6506        {
6507            return Some("link");
6508        }
6509        if self.preserved_link_paths().iter().any(|(p, _)| *p == name) {
6510            return Some("link");
6511        }
6512        None
6513    }
6514
6515    /// Build the name index unless it is already built.
6516    ///
6517    /// The walk takes the registry spines and their slots, so it runs with no
6518    /// index lock held — the writer never holds one lock across another — and
6519    /// the result is kept only if nothing renamed, created or unlinked
6520    /// anything while it ran.
6521    fn build_name_index(&self) {
6522        let epoch = {
6523            let index = self.name_index.lock();
6524            if index.map.is_some() {
6525                return;
6526            }
6527            index.epoch
6528        };
6529        let mut map: HashMap<String, Vec<NameHit>> = HashMap::new();
6530        for (i, ds) in self.dataset_refs().iter().enumerate() {
6531            let d = ds.lock();
6532            if !d.deleted {
6533                map.entry(d.name.clone())
6534                    .or_default()
6535                    .push(NameHit::Dataset(i));
6536            }
6537        }
6538        for (i, grp) in self.group_refs().iter().enumerate() {
6539            let g = grp.lock();
6540            if !g.deleted {
6541                map.entry(g.name.trim_start_matches('/').to_string())
6542                    .or_default()
6543                    .push(NameHit::Group(i));
6544            }
6545        }
6546        for (i, c) in self.committed_datatypes_vec().iter().enumerate() {
6547            map.entry(c.name.clone())
6548                .or_default()
6549                .push(NameHit::Datatype(i));
6550        }
6551        for l in self.hard_links_vec().iter() {
6552            map.entry(self.hard_link_full_path(l))
6553                .or_default()
6554                .push(NameHit::HardLink);
6555        }
6556        for l in self.symbolic_links_vec().iter() {
6557            map.entry(self.symbolic_link_full_path(l))
6558                .or_default()
6559                .push(NameHit::SymbolicLink);
6560        }
6561        for (path, _) in self.preserved_link_paths() {
6562            map.entry(path).or_default().push(NameHit::PreservedLink);
6563        }
6564        let mut index = self.name_index.lock();
6565        if index.map.is_none() && index.epoch == epoch {
6566            index.map = Some(map);
6567        }
6568    }
6569
6570    /// Record that `hit` now holds `name` — the one way a new name enters the
6571    /// index, called from every push that gives a registry entry a name.
6572    fn register_name(&self, name: &str, hit: NameHit) {
6573        self.name_index.lock().insert(name, hit);
6574    }
6575
6576    /// Drop the index because something moved names wholesale (a group
6577    /// rename carries its subtree and every link path under it).
6578    fn forget_name_index(&self) {
6579        self.name_index.lock().forget();
6580    }
6581
6582    /// Delete a dataset name, with libhdf5's `H5Ldelete` semantics: a name
6583    /// is only a link. If `name` is a user hard link's path, just that
6584    /// link is removed and the object is untouched. If it is the tree name
6585    /// and a user hard link still names the object, the object survives
6586    /// under it — the link becomes the primary name and nothing is freed.
6587    /// Only deleting the *last* name soft-deletes the object and frees the
6588    /// file space it owned: its chunk blocks and chunk-index structures
6589    /// (or contiguous data block), the global-heap objects of its
6590    /// variable-length data and attributes, and — on a reopened file — the
6591    /// on-disk object header block. The freed space is reused by later
6592    /// allocations in this session; the file does not shrink.
6593    ///
6594    /// Refused while SWMR streaming is active: a live reader may hold any
6595    /// of those addresses (libhdf5 forbids link deletion during SWMR
6596    /// writes too).
6597    pub fn delete_dataset(&self, name: &str) -> IoResult<()> {
6598        if self.swmr_active {
6599            return Err(swmr_delete_error(name));
6600        }
6601        self.reject_external_traversal(name)?;
6602        // The gate keeps the link list and child lists still while this
6603        // delete reads and rewrites them (create_lock → op → slot order,
6604        // the same as every creator).
6605        let _create = self.create_lock.lock();
6606        // `H5Ldelete` resolves the path through links only *up to* the
6607        // leaf — the leaf is what gets deleted, so a leaf naming a user
6608        // link must stay literal and be unlinked, not its target.
6609        let name = match name.rsplit_once('/') {
6610            None => name.to_string(),
6611            Some((dir, leaf)) => format!(
6612                "{}/{leaf}",
6613                self.canonical_group_path(&format!("/{dir}"))
6614                    .trim_start_matches('/')
6615            ),
6616        };
6617        let refs = self.dataset_refs();
6618        let idx = match refs.iter().position(|d| {
6619            let g = d.lock();
6620            g.name == name && !g.deleted
6621        }) {
6622            Some(i) => i,
6623            None => {
6624                // Not a tree name — the path may name a user hard link,
6625                // and deleting a link path unlinks just that link (the
6626                // creation collision checks keep the two namespaces
6627                // disjoint, so the order of the lookups cannot matter).
6628                let link = self.hard_links_vec().iter().position(|l| {
6629                    self.hard_link_emitted(l)
6630                        && matches!(l.target, HardLinkTarget::Dataset(_))
6631                        && self.hard_link_full_path(l) == name
6632                });
6633                let Some(pos) = link else {
6634                    return Err(crate::io::IoError::NotFound(name));
6635                };
6636                self.hard_links.lock().remove(pos);
6637                return Ok(());
6638            }
6639        };
6640        // A surviving hard link keeps the object: promote the first one to
6641        // the primary name and delete nothing.
6642        let promote = self.hard_links_vec().iter().position(|l| {
6643            self.hard_link_emitted(l) && matches!(l.target, HardLinkTarget::Dataset(i) if i == idx)
6644        });
6645        if let Some(pos) = promote {
6646            self.promote_dataset_to_link(idx, pos);
6647            return Ok(());
6648        }
6649        refs[idx].lock().deleted = true;
6650        // Remove from parent group's child_datasets
6651        for grp in self.group_refs() {
6652            grp.lock().child_datasets.retain(|&di| di != idx);
6653        }
6654        self.purge_dead_links();
6655        let ds = self.ds(idx);
6656        let _op = ds.op.lock();
6657        self.release_dataset_storage(idx)
6658    }
6659
6660    /// Soft-delete a group and all its child datasets and sub-groups,
6661    /// freeing every deleted object's file space the way
6662    /// [`delete_dataset`](Self::delete_dataset) does — with the same
6663    /// `H5Ldelete` semantics: a `name` that is a user hard link's path
6664    /// unlinks just that link, and hard links from *outside* the subtree
6665    /// keep their targets. A dataset or group such a link names survives,
6666    /// re-homed under the link (a group brings its whole subtree with
6667    /// it); a link naming the deleted group itself turns the call into a
6668    /// pure rename and nothing is freed. Refused while SWMR streaming is
6669    /// active, same rule as `delete_dataset`.
6670    pub fn delete_group(&self, name: &str) -> IoResult<()> {
6671        if self.swmr_active {
6672            return Err(swmr_delete_error(name));
6673        }
6674        self.reject_external_traversal(name)?;
6675        // Same gate as `delete_dataset`: the pre-scan below and the
6676        // promotions must see a still link list and child lists.
6677        let _create = self.create_lock.lock();
6678        let name = if name.starts_with('/') {
6679            name.to_string()
6680        } else {
6681            format!("/{}", name)
6682        };
6683        // Leaf stays literal, directory resolves through links — the
6684        // same `H5Ldelete` rule as `delete_dataset`.
6685        let name = match name.rsplit_once('/') {
6686            Some((dir, leaf)) if !dir.is_empty() => {
6687                format!("{}/{leaf}", self.canonical_group_path(dir))
6688            }
6689            _ => name,
6690        };
6691        let groups = self.group_refs();
6692        let gidx = match groups.iter().position(|g| {
6693            let gg = g.lock();
6694            gg.name == name && !gg.deleted
6695        }) {
6696            Some(i) => i,
6697            None => {
6698                // Same `H5Ldelete` rule as `delete_dataset`: a path naming
6699                // a user hard link to a group unlinks just that link.
6700                let trimmed = name.trim_start_matches('/');
6701                let link = self.hard_links_vec().iter().position(|l| {
6702                    self.hard_link_emitted(l)
6703                        && matches!(l.target, HardLinkTarget::Group(_))
6704                        && self.hard_link_full_path(l) == trimmed
6705                });
6706                let Some(pos) = link else {
6707                    return Err(crate::io::IoError::NotFound(name.clone()));
6708                };
6709                self.hard_links.lock().remove(pos);
6710                return Ok(());
6711            }
6712        };
6713
6714        // A link is "outside" when its parent group does not die with the
6715        // subtree; only outside links can keep their targets alive.
6716        fn outside(parent: Option<usize>, doomed_gs: &[usize]) -> bool {
6717            match parent {
6718                None => true,
6719                Some(pi) => !doomed_gs.contains(&pi),
6720            }
6721        }
6722        // A group an outside link names survives, re-homed with its whole
6723        // subtree under the link. Each promotion moves that subtree out of
6724        // the doomed set — and can turn a link inside it into an outside
6725        // one — so rescan from scratch until no promotable group is left.
6726        // Promoting `gidx` itself makes the delete a pure rename: return.
6727        let mut doomed_ds = Vec::new();
6728        let mut doomed_gs = Vec::new();
6729        loop {
6730            doomed_ds.clear();
6731            doomed_gs.clear();
6732            self.collect_live_subtree(gidx, &mut doomed_ds, &mut doomed_gs);
6733            let promote = self
6734                .hard_links_vec()
6735                .iter()
6736                .enumerate()
6737                .find_map(|(pos, l)| match l.target {
6738                    HardLinkTarget::Group(gi)
6739                        if self.hard_link_emitted(l)
6740                            && outside(l.parent, &doomed_gs)
6741                            && doomed_gs.contains(&gi) =>
6742                    {
6743                        Some((pos, gi))
6744                    }
6745                    _ => None,
6746                });
6747            let Some((pos, gi)) = promote else { break };
6748            self.promote_group_to_link(gi, pos);
6749            if gi == gidx {
6750                return Ok(());
6751            }
6752        }
6753        // A dataset an outside link names survives its container: re-home
6754        // it under the link now, so the marking pass below never sees it.
6755        for di in doomed_ds {
6756            let promote = self.hard_links_vec().iter().position(|l| {
6757                self.hard_link_emitted(l)
6758                    && outside(l.parent, &doomed_gs)
6759                    && matches!(l.target, HardLinkTarget::Dataset(i) if i == di)
6760            });
6761            if let Some(pos) = promote {
6762                self.promote_dataset_to_link(di, pos);
6763            }
6764        }
6765
6766        let mut ds_deleted = Vec::new();
6767        let mut gs_deleted = Vec::new();
6768        self.delete_group_recursive(gidx, &mut ds_deleted, &mut gs_deleted);
6769        // Remove from parent's child_groups
6770        let parent = groups[gidx].lock().parent;
6771        if let Some(pidx) = parent {
6772            groups[pidx].lock().child_groups.retain(|&gi| gi != gidx);
6773        }
6774        self.purge_dead_links();
6775        // Free storage only after the whole subtree is marked: the lists
6776        // hold each object exactly once (the marking pass skips anything
6777        // already deleted), so nothing is freed twice.
6778        for di in ds_deleted {
6779            let ds = self.ds(di);
6780            let _op = ds.op.lock();
6781            self.release_dataset_storage(di)?;
6782        }
6783        for gi in gs_deleted {
6784            self.release_group_storage(gi)?;
6785        }
6786        Ok(())
6787    }
6788
6789    /// Collect the live (not soft-deleted) members of `gidx`'s subtree,
6790    /// each exactly once, without changing anything — the read-only twin
6791    /// of [`delete_group_recursive`](Self::delete_group_recursive), for
6792    /// the pre-scan that must run before any marking.
6793    fn collect_live_subtree(&self, gidx: usize, ds_out: &mut Vec<usize>, gs_out: &mut Vec<usize>) {
6794        if gs_out.contains(&gidx) {
6795            return;
6796        }
6797        let (child_ds, child_gs) = {
6798            let grp = self.grp(gidx);
6799            let g = grp.lock();
6800            if g.deleted {
6801                return;
6802            }
6803            (g.child_datasets.clone(), g.child_groups.clone())
6804        };
6805        gs_out.push(gidx);
6806        for di in child_ds {
6807            if !self.ds(di).lock().deleted && !ds_out.contains(&di) {
6808                ds_out.push(di);
6809            }
6810        }
6811        for gi in child_gs {
6812            self.collect_live_subtree(gi, ds_out, gs_out);
6813        }
6814    }
6815
6816    /// Re-home dataset `idx` under the hard link at `pos` in the link
6817    /// list — the surviving half of `H5Ldelete`: the link leaves the user
6818    /// list and becomes the dataset's primary (tree) name, in the link's
6819    /// parent group. Storage is untouched; any further links to the
6820    /// dataset stay in the list and keep resolving.
6821    fn promote_dataset_to_link(&self, idx: usize, pos: usize) {
6822        let link = self.hard_links.lock().remove(pos);
6823        let new_name = self.hard_link_full_path(&link);
6824        for grp in self.group_refs() {
6825            grp.lock().child_datasets.retain(|&di| di != idx);
6826        }
6827        if let Some(pi) = link.parent {
6828            self.grp(pi).lock().child_datasets.push(idx);
6829        }
6830        self.ds(idx).lock().name = new_name.clone();
6831        self.register_name(&new_name, NameHit::Dataset(idx));
6832    }
6833
6834    /// The group counterpart of
6835    /// [`promote_dataset_to_link`](Self::promote_dataset_to_link): re-home
6836    /// group `gidx` under the hard link at `pos`, bringing its whole
6837    /// subtree with it. Names are stored as full paths, so every live
6838    /// descendant is renamed by prefix.
6839    fn promote_group_to_link(&self, gidx: usize, pos: usize) {
6840        let link = self.hard_links.lock().remove(pos);
6841        let new_name = format!("/{}", self.hard_link_full_path(&link));
6842        let old_name = self.grp(gidx).lock().name.clone();
6843        for grp in self.group_refs() {
6844            grp.lock().child_groups.retain(|&g| g != gidx);
6845        }
6846        {
6847            let grp = self.grp(gidx);
6848            let mut g = grp.lock();
6849            g.parent = link.parent;
6850            g.name = new_name.clone();
6851        }
6852        if let Some(pi) = link.parent {
6853            self.grp(pi).lock().child_groups.push(gidx);
6854        }
6855
6856        let mut ds_in = Vec::new();
6857        let mut gs_in = Vec::new();
6858        self.collect_live_subtree(gidx, &mut ds_in, &mut gs_in);
6859        // Group names carry a leading '/' ("/a/b"), dataset names none
6860        // ("a/b/ds") — two prefix forms of the same rename.
6861        let old_grp_prefix = format!("{old_name}/");
6862        let new_grp_prefix = format!("{new_name}/");
6863        let old_ds_prefix = old_grp_prefix.trim_start_matches('/').to_string();
6864        let new_ds_prefix = new_grp_prefix.trim_start_matches('/').to_string();
6865        for gi in gs_in {
6866            if gi == gidx {
6867                continue;
6868            }
6869            let grp = self.grp(gi);
6870            let mut g = grp.lock();
6871            let renamed = g
6872                .name
6873                .strip_prefix(&old_grp_prefix)
6874                .map(|rest| format!("{new_grp_prefix}{rest}"));
6875            if let Some(n) = renamed {
6876                g.name = n;
6877            }
6878        }
6879        for di in ds_in {
6880            let ds = self.ds(di);
6881            let mut d = ds.lock();
6882            let renamed = d
6883                .name
6884                .strip_prefix(&old_ds_prefix)
6885                .map(|rest| format!("{new_ds_prefix}{rest}"));
6886            if let Some(n) = renamed {
6887                d.name = n;
6888            }
6889        }
6890        // A group carries its subtree and every link path under it, so far
6891        // more names moved than this function can enumerate: start over.
6892        self.forget_name_index();
6893    }
6894
6895    /// Drop link entries that can no longer be emitted — their parent group
6896    /// or, for a hard link, their target object was just deleted — so the
6897    /// lists mirror what the file will hold instead of carrying suppressed
6898    /// zombies. Both kinds are purged here so a delete cannot clear one list
6899    /// and leave the other holding a name in a group that is gone.
6900    fn purge_dead_links(&self) {
6901        let dead: Vec<usize> = self
6902            .hard_links_vec()
6903            .iter()
6904            .enumerate()
6905            .filter(|(_, l)| !self.hard_link_emitted(l))
6906            .map(|(p, _)| p)
6907            .collect();
6908        let mut links = self.hard_links.lock();
6909        for p in dead.into_iter().rev() {
6910            links.remove(p);
6911        }
6912        drop(links);
6913
6914        let dead: Vec<usize> = self
6915            .symbolic_links_vec()
6916            .iter()
6917            .enumerate()
6918            .filter(|(_, l)| !self.symbolic_link_emitted(l))
6919            .map(|(p, _)| p)
6920            .collect();
6921        let mut links = self.symbolic_links.lock();
6922        for p in dead.into_iter().rev() {
6923            links.remove(p);
6924        }
6925    }
6926
6927    /// Mark `gidx` and its subtree deleted, appending each newly-deleted
6928    /// object's index to `ds_out` / `gs_out` exactly once — the caller
6929    /// frees their storage, and an object reachable twice (or a subtree
6930    /// already deleted) must not be freed twice.
6931    fn delete_group_recursive(
6932        &self,
6933        gidx: usize,
6934        ds_out: &mut Vec<usize>,
6935        gs_out: &mut Vec<usize>,
6936    ) {
6937        // Mark deleted and snapshot the child lists, releasing the group lock
6938        // before locking any dataset/child-group slot (spine → slot order).
6939        let (child_ds, child_gs) = {
6940            let grp = self.grp(gidx);
6941            let mut g = grp.lock();
6942            if g.deleted {
6943                return;
6944            }
6945            g.deleted = true;
6946            (g.child_datasets.clone(), g.child_groups.clone())
6947        };
6948        gs_out.push(gidx);
6949        for di in child_ds {
6950            let ds = self.ds(di);
6951            let mut d = ds.lock();
6952            if !d.deleted {
6953                d.deleted = true;
6954                ds_out.push(di);
6955            }
6956        }
6957        for gi in child_gs {
6958            self.delete_group_recursive(gi, ds_out, gs_out);
6959        }
6960    }
6961
6962    /// Free everything a soft-deleted dataset owned. The single owner of
6963    /// delete-time reclamation, called only from the two delete paths with
6964    /// the dataset already marked deleted and its op lock held.
6965    ///
6966    /// A deleted dataset contributes nothing to finalize (the header,
6967    /// index-flush and append-flush loops all skip it), so nothing in the
6968    /// finalized file can reference the blocks freed here. Never runs under
6969    /// SWMR — the delete entry points refuse first.
6970    fn release_dataset_storage(&self, index: usize) -> IoResult<()> {
6971        use crate::format::messages::datatype::DatatypeMessage;
6972        let (indexed, ndims, contiguous, is_vlen, attrs, header_blocks, mapping_list) = {
6973            let ds = self.ds(index);
6974            let mut m = ds.lock();
6975            // Buffered rows were never written to a chunk; they die with
6976            // the dataset instead of being flushed at close.
6977            m.append = None;
6978            let indexed = m.is_chunked();
6979            let contiguous = (!indexed && m.data_addr != UNDEF_ADDR && m.data_size > 0)
6980                .then_some((m.data_addr, m.data_size));
6981            m.data_addr = UNDEF_ADDR;
6982            m.data_size = 0;
6983            // The external files themselves are the application's, not this
6984            // file's, and neither is the name heap freed: `H5O_MSG_EFL`
6985            // installs no file-delete method, so libhdf5 leaves the heap block
6986            // behind too. Dropping the list is what stops a deleted dataset
6987            // still claiming storage.
6988            m.external = None;
6989            // The mapping list is this file's own metadata, so unlike the
6990            // external files above it *is* freed — `H5D__virtual_delete`
6991            // removes the heap object. The source datasets it named are
6992            // another file's and are left alone.
6993            let mapping_list = m
6994                .virtual_storage
6995                .take()
6996                .and_then(|v| u16::try_from(v.heap_index).ok().map(|i| (v.heap_addr, i)));
6997            let is_vlen = matches!(
6998                m.datatype,
6999                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
7000            );
7001            let attrs = std::mem::take(&mut m.attributes);
7002            m.obj_header_written_addr = None;
7003            let header_blocks = std::mem::take(&mut m.obj_header_blocks);
7004            (
7005                indexed,
7006                m.dataspace.dims.len(),
7007                contiguous,
7008                is_vlen,
7009                attrs,
7010                header_blocks,
7011                mapping_list,
7012            )
7013        };
7014        if let Some((addr, idx)) = mapping_list {
7015            self.remove_heap_objects([(addr, vec![idx])].into_iter().collect())?;
7016        }
7017        if indexed {
7018            // Prune to a zero extent: every stored chunk is entirely beyond
7019            // it, so the walk frees each chunk block and collects the vlen
7020            // references its bytes held (released inside).
7021            self.prune_chunks_beyond(index, &vec![0; ndims])?;
7022            self.free_chunk_index(index)?;
7023        } else if let Some((addr, size)) = contiguous {
7024            if is_vlen {
7025                let data = self.handle.read_at(addr, size as usize)?;
7026                self.release_vlen_references(&data)?;
7027            }
7028            self.allocator.free(addr, size, FreeSpaceClass::RawData);
7029        }
7030        for attr in &attrs {
7031            self.release_attr_vlen(attr)?;
7032        }
7033        self.release_superseded_dense_attrs(AttrScope::Dataset(index))?;
7034        for (addr, size) in header_blocks {
7035            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
7036        }
7037        Ok(())
7038    }
7039
7040    /// Free a deleted group's file space: its attributes' global-heap
7041    /// objects and, on a reopened file, the on-disk header block. The
7042    /// group counterpart of
7043    /// [`release_dataset_storage`](Self::release_dataset_storage).
7044    fn release_group_storage(&self, gidx: usize) -> IoResult<()> {
7045        let (attrs, header_blocks) = {
7046            let grp = self.grp(gidx);
7047            let mut g = grp.lock();
7048            let attrs = std::mem::take(&mut g.attributes);
7049            g.obj_header_written_addr = None;
7050            (attrs, std::mem::take(&mut g.obj_header_blocks))
7051        };
7052        for attr in &attrs {
7053            self.release_attr_vlen(attr)?;
7054        }
7055        self.release_superseded_dense_attrs(AttrScope::Group(gidx))?;
7056        self.release_superseded_dense_links(LinkScope::Group(gidx))?;
7057        for (addr, size) in header_blocks {
7058            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
7059        }
7060        Ok(())
7061    }
7062
7063    /// Free the dense attribute storage a reopened header names, once, when
7064    /// this session stops naming it — because the header is being rewritten
7065    /// around fresh storage, or because the object was deleted.
7066    ///
7067    /// The single owner of that transition: nothing else removes an attribute
7068    /// entry from [`superseded_dense`](Self::superseded_dense), and this
7069    /// removes it as it frees, so no heap is freed twice or left half freed.
7070    /// An object whose storage was compact, or whose header this session
7071    /// keeps, has no entry and nothing happens.
7072    ///
7073    /// Never under SWMR: a live reader may still be walking the storage the
7074    /// published headers name, the same rule the superseded-header and
7075    /// relocated-chunk paths follow. The entry stays in place, unfreed.
7076    fn release_superseded_dense_attrs(&self, scope: AttrScope) -> IoResult<()> {
7077        if self.swmr_active {
7078            return Ok(());
7079        }
7080        let taken = self
7081            .superseded_dense
7082            .lock()
7083            .as_mut()
7084            .and_then(|s| s.attrs.remove(&scope));
7085        let Some(ainfo) = taken else {
7086            return Ok(());
7087        };
7088        self.release_dense_storage(
7089            ainfo.fractal_heap_address,
7090            ainfo.name_btree_address,
7091            ainfo.creation_order_btree_address,
7092        )
7093    }
7094
7095    /// The link counterpart of
7096    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs),
7097    /// under the same invariant and the same SWMR rule. Split from it because
7098    /// the two are superseded at different points of a finalize: attribute
7099    /// storage before the object headers are laid out, link storage after
7100    /// every one of them has an address.
7101    fn release_superseded_dense_links(&self, scope: LinkScope) -> IoResult<()> {
7102        if self.swmr_active {
7103            return Ok(());
7104        }
7105        let taken = self
7106            .superseded_dense
7107            .lock()
7108            .as_mut()
7109            .and_then(|s| s.links.remove(&scope));
7110        let Some(linfo) = taken else {
7111            return Ok(());
7112        };
7113        self.release_dense_storage(
7114            linfo.fractal_heap_address,
7115            linfo.name_btree_address,
7116            linfo.creation_order_btree_address,
7117        )
7118    }
7119
7120    /// Return one dense storage's file space to the allocator: the fractal
7121    /// heap in full, its name index, and the creation-order index when the
7122    /// object had one.
7123    ///
7124    /// The extents come from walking the structures themselves rather than
7125    /// from re-deriving what a writer would have allocated, so storage
7126    /// libhdf5 laid out is freed as accurately as storage this crate wrote.
7127    /// Every walk here already ran once this session — the reopen read every
7128    /// attribute out of this heap through the same index — so a failure means
7129    /// the file changed underneath us, and surfacing it beats freeing a
7130    /// partial extent list.
7131    fn release_dense_storage(
7132        &self,
7133        heap_addr: u64,
7134        name_bt2_addr: u64,
7135        corder_bt2_addr: Option<u64>,
7136    ) -> IoResult<()> {
7137        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
7138        use crate::format::fractal_heap::collect_heap_extents;
7139
7140        let mut reader = crate::io::reader::HandleBlockReader {
7141            handle: &self.handle,
7142        };
7143        let mut extents = Vec::new();
7144        if heap_addr != UNDEF_ADDR {
7145            extents.extend(collect_heap_extents(heap_addr, &self.ctx, &mut reader)?);
7146        }
7147        for addr in [Some(name_bt2_addr), corder_bt2_addr]
7148            .into_iter()
7149            .flatten()
7150            .filter(|&a| a != UNDEF_ADDR)
7151        {
7152            extents.extend(collect_btree_v2_extents(addr, &self.ctx, &mut reader)?);
7153        }
7154        for (addr, len) in extents {
7155            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
7156        }
7157        Ok(())
7158    }
7159
7160    /// Free a deleted dataset's chunk-index structures, after the chunks
7161    /// themselves were freed by a zero-extent prune. Takes the index info
7162    /// out of the slot, so the dataset no longer claims chunked storage.
7163    ///
7164    /// Every block's size is recovered the way its allocation computed it:
7165    /// re-encoding the in-memory copy (EA header and index block, FA
7166    /// header and data block, BT2 header) or sizing a same-shape dummy
7167    /// from the array geometry (EA data blocks, whose element counts come
7168    /// from [`EaGeometry`]; BT2 nodes are all `node_size`).
7169    fn free_chunk_index(&self, index: usize) -> IoResult<()> {
7170        let ds = self.ds(index);
7171        let mut m = ds.lock();
7172        let is_filtered = m.filter_pipeline.is_some();
7173        if let Some(c) = m.chunked.take() {
7174            let p = &c.earray_params;
7175            let bits = p.max_nelmts_bits;
7176            let csl = c.chunk_size_len;
7177            let geo = EaGeometry::new(
7178                p.idx_blk_elmts,
7179                p.data_blk_min_elmts,
7180                p.sup_blk_min_data_ptrs,
7181                bits,
7182                p.max_dblk_page_nelmts_bits,
7183            )?;
7184            let dblk_size = |nelmts: u64| -> u64 {
7185                if is_filtered {
7186                    FilteredDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7187                        .encode(&self.ctx, bits, csl)
7188                        .len() as u64
7189                } else {
7190                    ExtensibleArrayDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7191                        .encoded_size(&self.ctx, bits) as u64
7192                }
7193            };
7194            let (dblk_addrs, sblk_addrs, iblk_size) = if is_filtered {
7195                let f = c.filt_iblk.as_ref().unwrap();
7196                (
7197                    f.dblk_addrs.clone(),
7198                    f.sblk_addrs.clone(),
7199                    f.encode(&self.ctx, csl).len() as u64,
7200                )
7201            } else {
7202                (
7203                    c.ea_iblk.dblk_addrs.clone(),
7204                    c.ea_iblk.sblk_addrs.clone(),
7205                    c.ea_iblk.encoded_size(&self.ctx) as u64,
7206                )
7207            };
7208            // Data blocks addressed from the index block belong to the
7209            // first `iblock_nsblks` super blocks; each of those defines the
7210            // element count (and so the disk size) of its data blocks.
7211            let mut g = 0usize;
7212            'direct: for s in geo.sblk.iter().take(geo.iblock_nsblks) {
7213                for _ in 0..s.ndblks {
7214                    let Some(&a) = dblk_addrs.get(g) else {
7215                        break 'direct;
7216                    };
7217                    g += 1;
7218                    if a == UNDEF_ADDR {
7219                        continue;
7220                    }
7221                    if s.dblk_nelmts > geo.dblk_page_nelmts {
7222                        return Err(crate::io::IoError::InvalidState(
7223                            "cannot free a paged extensible-array data block, \
7224                             which is not yet supported"
7225                                .into(),
7226                        ));
7227                    }
7228                    self.allocator
7229                        .free(a, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7230                }
7231            }
7232            for (off, &sa) in sblk_addrs.iter().enumerate() {
7233                if sa == UNDEF_ADDR {
7234                    continue;
7235                }
7236                let s = geo.sblk[geo.iblock_nsblks + off];
7237                if s.dblk_nelmts > geo.dblk_page_nelmts {
7238                    return Err(crate::io::IoError::InvalidState(
7239                        "cannot free a paged extensible-array data block, \
7240                         which is not yet supported"
7241                            .into(),
7242                    ));
7243                }
7244                let buf = self.handle.read_at_most(sa, 65536)?;
7245                let sb =
7246                    ExtensibleArraySuperBlock::decode(&buf, &self.ctx, bits, s.ndblks as usize, 0)?;
7247                for &da in &sb.dblk_addrs {
7248                    if da != UNDEF_ADDR {
7249                        self.allocator
7250                            .free(da, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7251                    }
7252                }
7253                self.allocator.free(
7254                    sa,
7255                    sb.encode(&self.ctx, bits).len() as u64,
7256                    FreeSpaceClass::Metadata,
7257                );
7258            }
7259            self.allocator
7260                .free(c.ea_iblk_addr, iblk_size, FreeSpaceClass::Metadata);
7261            self.allocator.free(
7262                c.ea_header_addr,
7263                c.ea_header.encoded_size(&self.ctx) as u64,
7264                FreeSpaceClass::Metadata,
7265            );
7266            return Ok(());
7267        }
7268        if let Some(fa) = m.fixed_array.take() {
7269            self.allocator.free(
7270                fa.fa_dblk_addr,
7271                fixed_array_dblk_disk_size(&self.ctx, &fa.fa_header),
7272                FreeSpaceClass::Metadata,
7273            );
7274            self.allocator.free(
7275                fa.fa_header_addr,
7276                fa.fa_header.encode(&self.ctx).len() as u64,
7277                FreeSpaceClass::Metadata,
7278            );
7279            return Ok(());
7280        }
7281        // The implicit index has no structure to free, only the one run of
7282        // chunk space it was given at create — which is the whole of its
7283        // storage, so nothing else can be leaked or double-freed here.
7284        if let Some(imp) = m.implicit.take() {
7285            self.allocator
7286                .free(imp.data_addr, imp.data_size, FreeSpaceClass::RawData);
7287            return Ok(());
7288        }
7289        // The single-chunk index has no structure of its own either: its one
7290        // chunk is the whole of its storage, addressed directly from the
7291        // layout message rather than any index this function's doc comment's
7292        // "chunks already freed by a zero-extent prune" applies to — so
7293        // freeing it here, if it was ever allocated, is the only place it
7294        // happens.
7295        if let Some(sc) = m.single_chunk.take() {
7296            if sc.data_addr != UNDEF_ADDR {
7297                let len = if is_filtered { sc.nbytes } else { sc.data_size };
7298                self.allocator
7299                    .free(sc.data_addr, len, FreeSpaceClass::RawData);
7300            }
7301            return Ok(());
7302        }
7303        // The version-1 B-tree owns nothing but its node blocks: the header
7304        // every other index has is, here, the root pointer inside the layout
7305        // message.
7306        if let Some(bt1) = m.btree_v1.take() {
7307            let element_size = m.datatype.element_size() as u64;
7308            let node_size = bt1
7309                .build_tree(element_size, self.ctx.sizeof_addr as usize)
7310                .node_size();
7311            for &a in &bt1.node_addrs {
7312                self.allocator
7313                    .free(a, node_size as u64, FreeSpaceClass::Metadata);
7314            }
7315            return Ok(());
7316        }
7317        if let Some(bt2) = m.btree_v2.take() {
7318            let tree = bt2.index.build_tree(&self.ctx);
7319            for &a in &bt2.node_addrs {
7320                self.allocator
7321                    .free(a, tree.node_size as u64, FreeSpaceClass::Metadata);
7322            }
7323            self.allocator.free(
7324                bt2.bt2_header_addr,
7325                tree.header(UNDEF_ADDR).encode(&self.ctx).len() as u64,
7326                FreeSpaceClass::Metadata,
7327            );
7328        }
7329        Ok(())
7330    }
7331
7332    /// Return the chunk dimensions for a dataset, if chunked.
7333    ///
7334    /// Returns an owned `Vec` because the chunk geometry now lives behind the
7335    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7336    pub fn dataset_chunk_dims(&self, index: usize) -> Option<Vec<u64>> {
7337        let ds = self.ds(index);
7338        let m = ds.lock();
7339        m.chunk_index_kind().map(|kind| match kind {
7340            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
7341            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
7342            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
7343            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
7344            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
7345            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
7346        })
7347    }
7348
7349    /// Return the current dimensions of a dataset.
7350    ///
7351    /// Returns an owned `Vec` because the dataspace now lives behind the
7352    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7353    pub fn dataset_dims(&self, index: usize) -> Vec<u64> {
7354        self.ds(index).lock().dataspace.dims.clone()
7355    }
7356
7357    /// Return the maximum extent a dataset declares, per dimension.
7358    ///
7359    /// An absent maximum shape means the shape is fixed at its current extent
7360    /// (libhdf5 defaults maxdims to dims at creation), so the current
7361    /// dimensions are returned; `H5S_UNLIMITED` is `u64::MAX`.
7362    pub fn dataset_max_dims(&self, index: usize) -> Vec<u64> {
7363        let ds = self.ds(index);
7364        let m = ds.lock();
7365        m.dataspace
7366            .max_dims
7367            .clone()
7368            .unwrap_or_else(|| m.dataspace.dims.clone())
7369    }
7370
7371    /// Whether a dataset stores its raw data through a filter pipeline.
7372    ///
7373    /// The write paths ask before choosing how to hand a chunk over: an
7374    /// unfiltered chunk's bytes go to the file exactly as the caller holds
7375    /// them, while a filtered one has to be compressed first.
7376    pub(crate) fn dataset_is_filtered(&self, index: usize) -> bool {
7377        self.ds(index).lock().filter_pipeline.is_some()
7378    }
7379
7380    /// Return the datatype a dataset declares on disk.
7381    ///
7382    /// The typed write paths need it to store bytes in the declared byte
7383    /// order; a reopened dataset handle has no copy of its own, and a cached
7384    /// one could disagree with what the header will say.
7385    pub fn dataset_datatype(&self, index: usize) -> DatatypeMessage {
7386        self.ds(index).lock().datatype.clone()
7387    }
7388
7389    /// Create a group in the file hierarchy.
7390    ///
7391    /// `parent_path` is the full path of the parent group (e.g., "/" for root).
7392    /// `name` is the name of the new group (e.g., "detector").
7393    ///
7394    /// Returns the group index in the writer's group list.
7395    pub fn create_group(&self, parent_path: &str, name: &str) -> IoResult<usize> {
7396        // Hold the create gate across the uniqueness check and the registry
7397        // push so the two are atomic (see `create_lock`).
7398        let _create = self.create_lock.lock();
7399        // A parent path through hard links creates in the link's target,
7400        // as HDF5 traversal does.
7401        let parent_path = self.canonical_group_path(parent_path);
7402        let parent_path = parent_path.as_str();
7403        let full_name = if parent_path == "/" {
7404            format!("/{}", name)
7405        } else {
7406            format!("{}/{}", parent_path, name)
7407        };
7408        // Same rule as dataset creation: a path through a carried external
7409        // link names a group in the other file, which this writer cannot make.
7410        self.reject_external_traversal(&full_name)?;
7411        // `name` may itself carry path components; resolving the whole thing
7412        // is what keeps a '/' out of the link this group will be reached by.
7413        let (parent_idx, _leaf) = self.split_parent(full_name.trim_start_matches('/'))?;
7414
7415        self.ensure_name_free(full_name.trim_start_matches('/'))?;
7416
7417        let group_idx = self.push_group(GroupInfo {
7418            name: full_name,
7419            parent: parent_idx,
7420            creation_seq: self.take_creation_seq(),
7421            track_order: self.track_order,
7422            times: self.created_object_times(),
7423            child_datasets: Vec::new(),
7424            child_groups: Vec::new(),
7425            obj_header_addr: 0,
7426            obj_header_written_addr: None,
7427            obj_header_blocks: Vec::new(),
7428            deleted: false,
7429            attributes: Vec::new(),
7430        });
7431
7432        // Register this group as a child of its parent
7433        if let Some(pidx) = parent_idx {
7434            self.grp(pidx).lock().child_groups.push(group_idx);
7435        }
7436
7437        Ok(group_idx)
7438    }
7439
7440    /// Register a dataset as belonging to a group.
7441    ///
7442    /// `group_path` is the full path of the group (e.g., "/detector").
7443    /// `ds_index` is the dataset index returned by `create_dataset`.
7444    pub fn assign_dataset_to_group(&self, group_path: &str, ds_index: usize) -> IoResult<()> {
7445        let group_path = self.canonical_group_path(group_path);
7446        let group_path = group_path.as_str();
7447        let groups = self.group_refs();
7448        let group_idx = groups
7449            .iter()
7450            .position(|g| {
7451                let gg = g.lock();
7452                gg.name == group_path && !gg.deleted
7453            })
7454            .ok_or_else(|| {
7455                crate::io::IoError::NotFound(format!("group '{}' not found", group_path))
7456            })?;
7457        // A move, not an addition: the create gate has already placed every
7458        // dataset from the path components of its name, so appending here
7459        // would leave one dataset linked from two groups at once.
7460        for g in &groups {
7461            g.lock().child_datasets.retain(|&d| d != ds_index);
7462        }
7463        groups[group_idx].lock().child_datasets.push(ds_index);
7464        Ok(())
7465    }
7466
7467    /// Create a hard link: an additional name for an object that already
7468    /// exists in the file.
7469    ///
7470    /// No data is copied — the link and its target share one object header,
7471    /// exactly as `h5py` / libhdf5 hard links do.
7472    ///
7473    /// * `parent_group_path` — full path of the group that will hold the
7474    ///   link (`"/"` for the root group).
7475    /// * `link_name` — leaf name of the new link within that group.
7476    /// * `target_path` — full path of an existing dataset or group, with or
7477    ///   without a leading `/`.
7478    pub fn create_hard_link(
7479        &self,
7480        parent_group_path: &str,
7481        link_name: &str,
7482        target_path: &str,
7483    ) -> IoResult<()> {
7484        if link_name.is_empty() || link_name.contains('/') {
7485            return Err(crate::io::IoError::InvalidState(format!(
7486                "hard link name '{link_name}' must be a non-empty leaf name"
7487            )));
7488        }
7489
7490        // Neither end may sit across a carried external link: the target
7491        // would be an object in the other file, and the link itself would be
7492        // a name in a group this writer does not own.
7493        self.reject_external_traversal(target_path)?;
7494        self.reject_external_traversal(&format!(
7495            "{}/{link_name}",
7496            parent_group_path.trim_end_matches('/')
7497        ))?;
7498
7499        // Hold the create gate across the collision check and the hard-link
7500        // push so the two are atomic (see `create_lock`).
7501        let _create = self.create_lock.lock();
7502        // Both paths resolve through hard links, as HDF5 traversal does.
7503        let parent_group_path = self.canonical_group_path(parent_group_path);
7504        let parent_group_path = parent_group_path.as_str();
7505
7506        // Resolve the parent group (None == root).
7507        let parent = if parent_group_path == "/" {
7508            None
7509        } else {
7510            Some(
7511                self.group_refs()
7512                    .iter()
7513                    .position(|g| {
7514                        let gg = g.lock();
7515                        gg.name == parent_group_path && !gg.deleted
7516                    })
7517                    .ok_or_else(|| {
7518                        crate::io::IoError::NotFound(format!(
7519                            "parent group '{parent_group_path}' not found"
7520                        ))
7521                    })?,
7522            )
7523        };
7524
7525        // Resolve the target. Dataset names are stored without a leading
7526        // '/', group names with one — compare on the trimmed form. A
7527        // trailing '/' is tolerated too.
7528        let target_rel = self.canonical_dataset_path(target_path.trim_matches('/'));
7529        let target_rel = target_rel.as_str();
7530        if target_rel.is_empty() {
7531            return Err(crate::io::IoError::InvalidState(
7532                "cannot hard-link the root group".into(),
7533            ));
7534        }
7535        let target = self.resolve_object(target_rel).ok_or_else(|| {
7536            crate::io::IoError::NotFound(format!("hard link target '{target_path}' not found"))
7537        })?;
7538
7539        // Reject a name already taken in the parent group.
7540        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7541
7542        self.hard_links.lock().push(HardLink {
7543            parent,
7544            name: link_name.to_string(),
7545            target,
7546            creation_seq: self.take_creation_seq(),
7547        });
7548        self.register_name(&self.link_full_path(parent, link_name), NameHit::HardLink);
7549        Ok(())
7550    }
7551
7552    /// Whether a hard link will actually be emitted: both its parent group
7553    /// and its target object must still be present (not soft-deleted).
7554    fn hard_link_emitted(&self, link: &HardLink) -> bool {
7555        let parent_ok = self.parent_alive(link.parent);
7556        let target_ok = match link.target {
7557            HardLinkTarget::Dataset(i) => !self.ds(i).lock().deleted,
7558            HardLinkTarget::Group(i) => !self.grp(i).lock().deleted,
7559        };
7560        parent_ok && target_ok
7561    }
7562
7563    /// The full path a link occupies, with no leading `/` — the same form
7564    /// dataset names are stored in. The one place a parent index and a leaf
7565    /// name become a path, so every link kind answers the collision check in
7566    /// the same spelling.
7567    fn link_full_path(&self, parent: Option<usize>, name: &str) -> String {
7568        match parent {
7569            None => name.to_string(),
7570            Some(pi) => format!(
7571                "{}/{name}",
7572                self.grp(pi).lock().name.trim_start_matches('/')
7573            ),
7574        }
7575    }
7576
7577    /// The full path a hard link occupies; see [`Self::link_full_path`].
7578    fn hard_link_full_path(&self, link: &HardLink) -> String {
7579        self.link_full_path(link.parent, &link.name)
7580    }
7581
7582    /// Whether a symbolic link will actually be emitted: its parent group
7583    /// must still be present. There is no target to check — a soft or
7584    /// external link is allowed to dangle, and `H5Lcreate_soft` does not look
7585    /// at the path it stores.
7586    fn symbolic_link_emitted(&self, link: &SymbolicLink) -> bool {
7587        self.parent_alive(link.parent)
7588    }
7589
7590    /// Whether the group that would hold a link still exists; `None` is the
7591    /// root group, which cannot be deleted.
7592    ///
7593    /// A deleted group's header is never written, so nothing it would have
7594    /// held is in the file — and the name is free again. Every registry
7595    /// decides that the same way, through here.
7596    fn parent_alive(&self, parent: Option<usize>) -> bool {
7597        match parent {
7598            None => true,
7599            Some(pi) => !self.grp(pi).lock().deleted,
7600        }
7601    }
7602
7603    /// The full path a symbolic link occupies; see [`Self::link_full_path`].
7604    fn symbolic_link_full_path(&self, link: &SymbolicLink) -> String {
7605        self.link_full_path(link.parent, &link.name)
7606    }
7607
7608    /// Create a soft or external link: a name in a group whose value is a
7609    /// path rather than an object.
7610    ///
7611    /// The single owner of symbolic-link creation — `H5Lcreate_soft` and
7612    /// `H5Lcreate_external` differ only in the value they store, and the
7613    /// name, parent and collision rules they share are all here.
7614    ///
7615    /// * `parent_group_path` — full path of the group that will hold the
7616    ///   link (`"/"` for the root group).
7617    /// * `link_name` — leaf name of the new link within that group.
7618    /// * `target` — the path this link names, and for an external link the
7619    ///   file holding it. Neither is resolved or required to exist: HDF5
7620    ///   answers a symbolic link at traversal time, so a dangling one is a
7621    ///   legal file.
7622    pub fn create_symbolic_link(
7623        &self,
7624        parent_group_path: &str,
7625        link_name: &str,
7626        target: LinkTarget,
7627    ) -> IoResult<()> {
7628        if link_name.is_empty() || link_name.contains('/') {
7629            return Err(crate::io::IoError::InvalidState(format!(
7630                "link name '{link_name}' must be a non-empty leaf name"
7631            )));
7632        }
7633        // `H5Lcreate_external` refuses an empty file or object name, and
7634        // stores the object path normalized; a link written here and one
7635        // libhdf5 writes from the same arguments then hold the same bytes.
7636        let target = match target {
7637            LinkTarget::External { file, path } => {
7638                if file.is_empty() || path.is_empty() {
7639                    return Err(crate::io::IoError::InvalidState(
7640                        "an external link needs both a file name and an object path".into(),
7641                    ));
7642                }
7643                LinkTarget::External {
7644                    file,
7645                    path: crate::format::messages::link::normalize_object_path(&path),
7646                }
7647            }
7648            other => other,
7649        };
7650        // The link itself would be a name in a group that lives in another
7651        // file; its *value* may name anything, including a path this writer
7652        // cannot follow, because nothing follows it here.
7653        self.reject_external_traversal(&format!(
7654            "{}/{link_name}",
7655            parent_group_path.trim_end_matches('/')
7656        ))?;
7657
7658        let _create = self.create_lock.lock();
7659        let parent_group_path = self.canonical_group_path(parent_group_path);
7660        let parent_group_path = parent_group_path.as_str();
7661        let parent = if parent_group_path == "/" {
7662            None
7663        } else {
7664            Some(
7665                self.group_refs()
7666                    .iter()
7667                    .position(|g| {
7668                        let gg = g.lock();
7669                        gg.name == parent_group_path && !gg.deleted
7670                    })
7671                    .ok_or_else(|| {
7672                        crate::io::IoError::NotFound(format!(
7673                            "parent group '{parent_group_path}' not found"
7674                        ))
7675                    })?,
7676            )
7677        };
7678
7679        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7680        self.symbolic_links.lock().push(SymbolicLink {
7681            parent,
7682            name: link_name.to_string(),
7683            target,
7684            creation_seq: self.take_creation_seq(),
7685        });
7686        self.register_name(
7687            &self.link_full_path(parent, link_name),
7688            NameHit::SymbolicLink,
7689        );
7690        Ok(())
7691    }
7692
7693    // ---------------------------------------------------------- committed types
7694
7695    /// Snapshot the committed-datatype list; see [`Self::hard_links_vec`].
7696    pub(crate) fn committed_datatypes_vec(&self) -> Vec<CommittedDatatype> {
7697        self.committed_datatypes.lock().clone()
7698    }
7699
7700    /// The paths of every committed datatype a name still reaches, in
7701    /// creation order. One inside a deleted group is not among them: no link
7702    /// to it is emitted, so the file will not hold that name.
7703    ///
7704    /// Both halves of the file answer. A datatype an earlier session
7705    /// committed is carried by its bytes, not re-encoded, so it lives in the
7706    /// preserved-link list rather than the registry — and listing only the
7707    /// registry is what made this answer `[]` for a file whose every named
7708    /// type was committed before it was opened, while a reader of the same
7709    /// file named them all.
7710    pub(crate) fn committed_datatype_names(&self) -> Vec<String> {
7711        let mut out: Vec<String> = self
7712            .committed_datatypes_vec()
7713            .iter()
7714            .filter(|c| self.parent_alive(c.parent))
7715            .map(|c| c.name.clone())
7716            .collect();
7717        out.extend(
7718            self.preserved_links
7719                .lock()
7720                .iter()
7721                .filter(|l| l.kind == PreservedKind::NamedDatatype)
7722                .map(|l| self.preserved_link_full_path(l)),
7723        );
7724        out
7725    }
7726
7727    /// Commit `datatype` as an object of its own under `name` —
7728    /// `H5Tcommit2`. Returns its index in the committed-datatype registry.
7729    ///
7730    /// The object holds one datatype message and nothing else. It goes
7731    /// through [`begin_create`](Self::begin_create) like a dataset, so its
7732    /// name is resolved to a real parent group, refused if taken, and refused
7733    /// if it would cross a carried external link.
7734    pub fn commit_datatype(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
7735        let create = self.begin_create(name.trim_start_matches('/'))?;
7736        let entry = CommittedDatatype {
7737            name: create.name.clone(),
7738            parent: create.parent,
7739            datatype,
7740            creation_seq: self.take_creation_seq(),
7741            times: self.created_object_times(),
7742            obj_header_addr: 0,
7743        };
7744        let name = entry.name.clone();
7745        let idx = {
7746            let mut reg = self.committed_datatypes.lock();
7747            let idx = reg.len();
7748            reg.push(entry);
7749            idx
7750        };
7751        self.register_name(&name, NameHit::Datatype(idx));
7752        Ok(idx)
7753    }
7754
7755    /// Resolve a committed datatype's path to its registry index and the type
7756    /// it holds — the pair a dataset needs to be built on it.
7757    ///
7758    /// Returned together so the caller cannot pair one committed type's index
7759    /// with another's datatype: the dataset's element width, dataspace and
7760    /// payload checks all come from the type, and its header names the index.
7761    pub(crate) fn committed_datatype_for_share(
7762        &self,
7763        name: &str,
7764    ) -> IoResult<(usize, DatatypeMessage)> {
7765        let name = self.canonical_dataset_path(name.trim_start_matches('/'));
7766        let all = self.committed_datatypes_vec();
7767        all.iter()
7768            .position(|c| self.parent_alive(c.parent) && c.name == name)
7769            .map(|i| (i, all[i].datatype.clone()))
7770            .ok_or_else(|| {
7771                crate::io::IoError::NotFound(format!("no committed datatype named '{name}'"))
7772            })
7773    }
7774
7775    /// Record that dataset `dataset` stores its datatype as a pointer to the
7776    /// committed datatype `committed`.
7777    ///
7778    /// Takes an index [`committed_datatype_for_share`](Self::committed_datatype_for_share)
7779    /// produced, alongside the datatype from the same call, so the two cannot
7780    /// disagree and there is nothing here that can fail after the dataset
7781    /// exists.
7782    pub(crate) fn share_committed_type(&self, dataset: usize, committed: usize) {
7783        debug_assert!(committed < self.committed_datatypes.lock().len());
7784        self.ds(dataset).lock().committed_type = Some(CommittedTypeRef::Session(committed));
7785    }
7786
7787    /// How many names reach the committed datatype `index`: the link that
7788    /// gave it its name, plus every live dataset that shares it.
7789    ///
7790    /// `H5O__shared_link_adj` counts a share as a link, which is why a type
7791    /// h5py commits and then builds one dataset on reports `rc == 2`. Zero
7792    /// means nothing reaches it at all — the group holding its name was
7793    /// deleted and no dataset shares it — and then it is not written.
7794    fn committed_datatype_refcount(&self, index: usize) -> u32 {
7795        let linked = {
7796            let parent = self.committed_datatypes.lock()[index].parent;
7797            u32::from(self.parent_alive(parent))
7798        };
7799        let shares = self
7800            .dataset_refs()
7801            .iter()
7802            .filter(|d| {
7803                let m = d.lock();
7804                !m.deleted && m.committed_type == Some(CommittedTypeRef::Session(index))
7805            })
7806            .count() as u32;
7807        linked + shares
7808    }
7809
7810    /// Append the link naming each committed datatype whose parent group is
7811    /// `parent`. A committed datatype is reached by an ordinary hard link —
7812    /// what makes it a datatype rather than a group or a dataset is the one
7813    /// message in the header it points at.
7814    ///
7815    /// Only a live group's links are collected, and a live parent is itself a
7816    /// reference, so every address named here belongs to a header
7817    /// `write_committed_datatype_headers` wrote.
7818    fn push_committed_datatypes(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
7819        for cd in self.committed_datatypes_vec() {
7820            if cd.parent != parent {
7821                continue;
7822            }
7823            let leaf = cd.name.rsplit('/').next().unwrap_or(&cd.name);
7824            links.push((cd.creation_seq, LinkMessage::hard(leaf, cd.obj_header_addr)));
7825        }
7826    }
7827
7828    /// Rewrite a group path that passes through hard links into the tree
7829    /// path of the group it reaches — HDF5 traversal, where any link in a
7830    /// path component resolves to its target. Group-name form (leading
7831    /// `/`). Repeats because a substituted target's subtree can hold
7832    /// further links; bounded like libhdf5's link-traversal limit, so a
7833    /// link cycle cannot loop forever. A path with no link components
7834    /// (including one naming nothing at all) comes back unchanged.
7835    pub(crate) fn canonical_group_path(&self, path: &str) -> String {
7836        let mut path = path.to_string();
7837        for _ in 0..64 {
7838            // The longest emitted group-link path that is the whole of
7839            // `path` or a '/'-boundary prefix of it.
7840            let mut best: Option<(usize, usize)> = None; // (prefix len, target)
7841            for l in self.hard_links_vec() {
7842                let HardLinkTarget::Group(gi) = l.target else {
7843                    continue;
7844                };
7845                if !self.hard_link_emitted(&l) {
7846                    continue;
7847                }
7848                let lp = format!("/{}", self.hard_link_full_path(&l));
7849                let covers = path == lp || path.starts_with(&format!("{lp}/"));
7850                if covers && best.is_none_or(|(len, _)| lp.len() > len) {
7851                    best = Some((lp.len(), gi));
7852                }
7853            }
7854            let Some((len, gi)) = best else { break };
7855            let target_name = self.grp(gi).lock().name.clone();
7856            path = format!("{}{}", target_name, &path[len..]);
7857        }
7858        path
7859    }
7860
7861    /// [`canonical_group_path`](Self::canonical_group_path) in the
7862    /// dataset-name form (no leading `/`): the leaf is a dataset, so only
7863    /// group links can appear as components and the whole path can go
7864    /// through the group rewrite unchanged.
7865    fn canonical_dataset_path(&self, name: &str) -> String {
7866        self.canonical_group_path(&format!("/{name}"))
7867            .trim_start_matches('/')
7868            .to_string()
7869    }
7870
7871    /// Total number of hard links resolving to an object: its own tree link
7872    /// plus every emitted user-created hard link pointing at it.
7873    fn object_link_count(&self, target: HardLinkTarget) -> u32 {
7874        let same = |a: HardLinkTarget, b: HardLinkTarget| -> bool {
7875            matches!(
7876                (a, b),
7877                (HardLinkTarget::Dataset(x), HardLinkTarget::Dataset(y))
7878                    | (HardLinkTarget::Group(x), HardLinkTarget::Group(y))
7879                if x == y
7880            )
7881        };
7882        1 + self
7883            .hard_links_vec()
7884            .iter()
7885            .filter(|l| self.hard_link_emitted(l) && same(l.target, target))
7886            .count() as u32
7887    }
7888
7889    /// The object a path names, or `None` when nothing in the file does.
7890    ///
7891    /// `path` is the trimmed, hard-link-canonical form (no leading or
7892    /// trailing `/`) that dataset and group names compare against. The single
7893    /// owner of path→object resolution on the write side: hard links and
7894    /// object references must agree on what a path means, including that a
7895    /// path may itself be a user hard link — links have no chain (each points
7896    /// straight at the object header, as in libhdf5), so the existing link's
7897    /// target is the answer.
7898    pub(crate) fn resolve_object(&self, path: &str) -> Option<HardLinkTarget> {
7899        if let Some(idx) = self.dataset_refs().iter().position(|d| {
7900            let g = d.lock();
7901            !g.deleted && g.name.trim_start_matches('/') == path
7902        }) {
7903            return Some(HardLinkTarget::Dataset(idx));
7904        }
7905        if let Some(idx) = self.group_refs().iter().position(|g| {
7906            let gg = g.lock();
7907            !gg.deleted && gg.name.trim_start_matches('/') == path
7908        }) {
7909            return Some(HardLinkTarget::Group(idx));
7910        }
7911        self.hard_links_vec().iter().find_map(|l| {
7912            (self.hard_link_emitted(l) && self.hard_link_full_path(l) == path).then_some(l.target)
7913        })
7914    }
7915
7916    /// The address of dataset `index`'s own contiguous block, for the two
7917    /// writers that stamp single elements into it by file offset — object and
7918    /// region references, whose values are only known once finalize has placed
7919    /// every object header.
7920    ///
7921    /// Refuses, rather than handing back an address that is not one, every
7922    /// dataset that has no such block: chunked, compact, unallocated, or with
7923    /// its raw data in files outside this one.
7924    fn local_element_block(&self, index: usize, what: &str) -> IoResult<u64> {
7925        let ds = self.ds(index);
7926        let m = ds.lock();
7927        match m.contiguous_target() {
7928            Some(ContiguousTarget::Local(addr)) => Ok(addr),
7929            Some(ContiguousTarget::External { .. }) => {
7930                Err(crate::io::IoError::InvalidState(format!(
7931                    "{what} are stamped into the dataset's own contiguous block, and \
7932                 dataset '{}' has none: its raw data lives in external files",
7933                    m.name
7934                )))
7935            }
7936            Some(ContiguousTarget::Virtual) => Err(crate::io::IoError::InvalidState(format!(
7937                "{what} are stamped into the dataset's own contiguous block, and \
7938                 dataset '{}' has none: it is virtual, and its elements come from \
7939                 the source datasets its mappings name",
7940                m.name
7941            ))),
7942            None => Err(crate::io::IoError::InvalidState(format!(
7943                "{what} are stamped into contiguous storage; create the dataset \
7944                 without chunking"
7945            ))),
7946        }
7947    }
7948
7949    /// Store object references naming `paths` into the elements of dataset
7950    /// `index` starting at `start`.
7951    ///
7952    /// The value of an `H5R_OBJECT1` element is its target's object header
7953    /// address, which finalize assigns, so what lands here is the target path;
7954    /// [`Self::write_object_reference_values`] writes the addresses. Elements
7955    /// never written keep the zero image libhdf5 reads back as a null
7956    /// reference.
7957    pub fn write_object_references(
7958        &self,
7959        index: usize,
7960        start: u64,
7961        paths: &[&str],
7962    ) -> IoResult<()> {
7963        let elements = {
7964            let ds = self.ds(index);
7965            let m = ds.lock();
7966            match &m.datatype {
7967                // Both generations of object reference: `H5T_STD_REF_OBJ` and
7968                // the 1.12 `H5T_STD_REF`. They differ only in the element
7969                // image, which `encode_reference_element` owns.
7970                DatatypeMessage::Reference {
7971                    kind: ReferenceKind::Object1 | ReferenceKind::Object2,
7972                    ..
7973                } => {}
7974                other => {
7975                    return Err(crate::io::IoError::InvalidState(format!(
7976                        "dataset '{}' has datatype {other}, not an object reference",
7977                        m.name
7978                    )))
7979                }
7980            }
7981            m.dataspace
7982                .dims
7983                .iter()
7984                .fold(1u64, |a, &d| a.saturating_mul(d))
7985        };
7986        // Refused here as well as at fixup time, so a dataset whose storage
7987        // cannot hold stamped elements is reported at the call that chose it.
7988        self.local_element_block(index, "object references")?;
7989        let end = start.saturating_add(paths.len() as u64);
7990        if end > elements {
7991            return Err(crate::io::IoError::InvalidState(format!(
7992                "elements {start}..{end} are outside the dataset's {elements}"
7993            )));
7994        }
7995        // Resolve now as well as at fixup time, so a path that names nothing
7996        // is reported at the call that got it wrong.
7997        for path in paths {
7998            self.object_reference_target(path)?;
7999        }
8000        let mut pending = self.pending_object_references.lock();
8001        for (i, path) in paths.iter().enumerate() {
8002            pending.push(PendingObjectReference {
8003                dataset: index,
8004                element: start + i as u64,
8005                target: (*path).to_string(),
8006            });
8007        }
8008        Ok(())
8009    }
8010
8011    /// Record a hard link count of `rc` in `header`, if this file's format
8012    /// needs a message to carry it.
8013    ///
8014    /// A version-2 header carries the count in an Object Reference Count
8015    /// message, and only when more than one link reaches the object. A
8016    /// version-1 header carries it in its prefix and gets no message at all —
8017    /// `H5O_link_oh` gates every refcount-message operation on
8018    /// `oh->version > H5O_VERSION_1` (H5Oint.c:851), so a version-1 header
8019    /// holding one is a shape libhdf5 never writes.
8020    ///
8021    /// The message carries `H5O_MSG_FLAG_DONTSHARE`, which both refcount
8022    /// operations pass (H5Oint.c:874 append, H5Oint.c:864 write): the count is
8023    /// a property of this one object header, so a shared-message index that
8024    /// pointed several headers at one copy would make every object with the
8025    /// same link count share a single number.
8026    fn emit_refcount(&self, header: &mut ObjectHeader, rc: u32, format: ObjectFormat) {
8027        if rc > 1 && format == ObjectFormat::Modern {
8028            header.add_message(MSG_OBJ_REF_COUNT, MSG_FLAG_DONTSHARE, encode_refcount(rc));
8029        }
8030    }
8031
8032    /// Encode `header` as `placement` lays it out, at the version this file's
8033    /// format calls for and with `rc` as the object's hard link count: every
8034    /// `(address, image)` pair to write, chunk 0 first.
8035    ///
8036    /// The count is passed rather than read off the header because the two
8037    /// versions carry it in different places — the version-1 prefix's `nlink`
8038    /// field, the version-2 Reference Count message
8039    /// [`emit_refcount`](Self::emit_refcount) already added — and only the
8040    /// caller knows it.
8041    ///
8042    /// INVARIANT: an object header's chunk 0 never moves once something in the
8043    /// file has named its address. A written header is rewritten over the
8044    /// chunk-0 block it already has, padded when the messages shrank and
8045    /// spilling into a continuation block of its own when they grew — the way
8046    /// `H5O__alloc_new_chunk` (H5Oalloc.c) grows a header libhdf5 cannot
8047    /// extend in place. That is what keeps every object reference already in
8048    /// the file — in a reference dataset, an attribute, a `REFERENCE_LIST`,
8049    /// whoever wrote them — resolving after this session. The one exception is
8050    /// a block too small to hold even the message naming a continuation, which
8051    /// [`place_header`](Self::place_header) gives up and replaces.
8052    ///
8053    /// A fresh header lives in the one block its address and encoded size
8054    /// describe: one whose messages overflow chunk 0 gets its continuation
8055    /// chunk immediately behind it in that same block, so the address is
8056    /// enough to free or supersede the whole header. libhdf5 would have grown
8057    /// chunk 0 into space that free rather than chaining onto it, but it reads
8058    /// a continuation chunk by the address and length its message states and
8059    /// cares nothing for where that lands.
8060    fn encode_header_in(
8061        &self,
8062        header: &ObjectHeader,
8063        rc: u32,
8064        format: ObjectFormat,
8065        placement: &HeaderPlacement,
8066    ) -> IoResult<Vec<(u64, Vec<u8>)>> {
8067        let plan = if placement.kept {
8068            header.plan_chunks_in(format, placement.size, &self.ctx)?
8069        } else {
8070            header.plan_chunks(format, self.chunk0_capacity(header, format), &self.ctx)?
8071        };
8072        let continuation_addr = match placement.continuation {
8073            Some((addr, _)) => addr,
8074            None => placement.addr + plan.chunk0_size as u64,
8075        };
8076        let (mut chunk0, continuation) =
8077            header.encode_chunked(&plan, format, &self.ctx, continuation_addr, rc)?;
8078        match (placement.continuation, continuation) {
8079            (Some((addr, _)), Some(image)) => Ok(vec![(placement.addr, chunk0), (addr, image)]),
8080            (None, Some(image)) => {
8081                chunk0.extend_from_slice(&image);
8082                Ok(vec![(placement.addr, chunk0)])
8083            }
8084            (None, None) => Ok(vec![(placement.addr, chunk0)]),
8085            (Some((addr, size)), None) => Err(crate::io::IoError::InvalidState(format!(
8086                "an object header was placed with a {size}-byte continuation block at \
8087                 {addr:#x} that it no longer needs; a message in it changed length \
8088                 once the addresses it names were known"
8089            ))),
8090        }
8091    }
8092
8093    /// Reserve the blocks `header` will be written over, keeping `kept` — the
8094    /// chunk-0 block the object's existing header occupies — when there is
8095    /// one it can be written over.
8096    ///
8097    /// A header's layout does not depend on the addresses it carries, which is
8098    /// what lets the group pass hand every group header an address before it
8099    /// writes any of their content: every address is a fixed-width field.
8100    ///
8101    /// A kept block is given up only when it cannot describe the header at
8102    /// all: too narrow for the message naming a continuation chunk, or not a
8103    /// shape the header's version can pad (see `ObjectHeader::plan_chunks_in`).
8104    /// Then it is freed and the header gets a fresh block, exactly as a new
8105    /// object does — and the references naming it are the caller's to
8106    /// restamp, which the writer does for every one it registered.
8107    fn place_header(
8108        &mut self,
8109        header: &ObjectHeader,
8110        format: ObjectFormat,
8111        kept: Option<(u64, u64)>,
8112    ) -> IoResult<HeaderPlacement> {
8113        if let Some((addr, len)) = kept {
8114            let plan = usize::try_from(len)
8115                .ok()
8116                .and_then(|len| header.plan_chunks_in(format, len, &self.ctx).ok());
8117            match plan {
8118                Some(plan) => {
8119                    let continuation = (plan.continuation_size > 0).then(|| {
8120                        let size = plan.continuation_size;
8121                        let addr = self
8122                            .allocator
8123                            .allocate(size as u64, FreeSpaceClass::Metadata);
8124                        (addr, size)
8125                    });
8126                    return Ok(HeaderPlacement {
8127                        addr,
8128                        size: len as usize,
8129                        kept: true,
8130                        continuation,
8131                    });
8132                }
8133                // A block a SWMR reader may be walking stays allocated, as
8134                // everywhere else under `swmr_active`.
8135                None if !self.swmr_active => {
8136                    self.allocator.free(addr, len, FreeSpaceClass::Metadata);
8137                }
8138                None => {}
8139            }
8140        }
8141        let plan = header.plan_chunks(format, self.chunk0_capacity(header, format), &self.ctx)?;
8142        let size = plan.chunk0_size + plan.continuation_size;
8143        let addr = self
8144            .allocator
8145            .allocate(size as u64, FreeSpaceClass::Metadata);
8146        Ok(HeaderPlacement::fresh(addr, size))
8147    }
8148
8149    /// How many bytes of messages `header`'s chunk 0 holds before the rest
8150    /// spill into a continuation chunk.
8151    ///
8152    /// libhdf5 sizes chunk 0 once, when the object header is created, and can
8153    /// only grow it while the space behind it is still free — so an object
8154    /// whose creation-time estimate covered every message it would ever hold
8155    /// keeps one chunk, and one whose estimate was a guess does not. A dataset
8156    /// or a committed datatype is created from messages already in hand
8157    /// (`H5D__update_oh_info`, `H5T__commit`), so its estimate is exact and
8158    /// this writer's exact fit is the same answer.
8159    ///
8160    /// A group is the exception: `H5G__obj_create_real` (H5Gobj.c:219) sizes
8161    /// its header for the link info and group info messages plus
8162    /// `H5G_CRT_GINFO_EST_NUM_ENTRIES` links of `H5G_CRT_GINFO_EST_NAME_LEN`
8163    /// characters, and nothing else — attributes above all — is in that
8164    /// estimate. The Link Info message is what identifies one: it is the
8165    /// message that makes an object a new-format group, and
8166    /// `H5G__obj_get_linfo` uses it for exactly this question.
8167    ///
8168    /// A version-1 header is written as one chunk whatever it holds: its
8169    /// groups keep their links in a symbol table, not in the header, so the
8170    /// estimate that makes a version-2 group spill never applies to one.
8171    fn chunk0_capacity(&self, header: &ObjectHeader, format: ObjectFormat) -> usize {
8172        if format == ObjectFormat::Legacy {
8173            return usize::MAX;
8174        }
8175        let envelope = header.message_envelope_size();
8176        let sized = |msg_type: u8| {
8177            header
8178                .messages
8179                .iter()
8180                .find(|m| m.msg_type == msg_type)
8181                .map(|m| envelope + m.data.len())
8182        };
8183        let Some(link_info) = sized(MSG_LINK_INFO) else {
8184            return usize::MAX;
8185        };
8186        // One estimated hard link: version, flags, a one-byte name length for
8187        // a name this short, the name, and the object header address.
8188        let link = envelope + 1 + 1 + 1 + EST_LINK_NAME_LEN + self.ctx.sizeof_addr as usize;
8189        link_info + sized(MSG_GROUP_INFO).unwrap_or(0) + EST_LINK_COUNT * link
8190    }
8191
8192    /// The object an object reference's path names, as a hard-link target;
8193    /// `None` for the root group, which has no registry slot.
8194    fn object_reference_target(&self, path: &str) -> IoResult<Option<HardLinkTarget>> {
8195        let rel = self.canonical_dataset_path(path.trim_matches('/'));
8196        if rel.is_empty() {
8197            return Ok(None);
8198        }
8199        self.resolve_object(&rel)
8200            .map(Some)
8201            .ok_or_else(|| crate::io::IoError::NotFound(format!("reference target '{path}'")))
8202    }
8203
8204    /// The object header address an object reference's `path` names, or zero
8205    /// when that object has not been given one yet.
8206    ///
8207    /// Zero is where the superblock sits, so it is never an object header's
8208    /// address. It is what every object reads as before
8209    /// [`allocate_object_headers`](Self::allocate_object_headers) runs, which
8210    /// is what lets the pass that measures a header stand in for the pass that
8211    /// writes it: an address is a fixed-width field, so the placeholder is the
8212    /// same size as the answer.
8213    fn object_reference_address(&self, path: &str) -> IoResult<u64> {
8214        Ok(match self.object_reference_target(path)? {
8215            Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8216            Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8217            None => self.root_group_addr.unwrap_or(0),
8218        })
8219    }
8220
8221    /// `scope`'s attributes as this finalize will write them: the stored set,
8222    /// with every object-reference attribute's value said in the object header
8223    /// addresses assigned so far.
8224    ///
8225    /// The single owner of a reference attribute's value, and the only source
8226    /// an object header build may take an attribute set from. Nothing stored
8227    /// is mutated, so the pass that measures a header and the pass that writes
8228    /// it cannot disagree about anything but the addresses — which they cannot
8229    /// disagree about in length.
8230    ///
8231    /// INVARIANT: the stored attribute list is what says which attributes
8232    /// exist; a recorded reference value can only give a value to one already
8233    /// in it. So a value left behind by an object whose list was emptied — a
8234    /// deleted group or dataset — cannot put the attribute back, and a value
8235    /// whose attribute was replaced by one of another type is dropped at the
8236    /// replacement instead of reaching it (see
8237    /// [`forget_attribute_reference`](Self::forget_attribute_reference)).
8238    fn object_attributes(&self, scope: AttrScope) -> IoResult<Vec<AttributeEntry>> {
8239        let mut attrs = match scope {
8240            AttrScope::Root => self.root_attributes.lock().clone(),
8241            AttrScope::Group(gi) => self.grp(gi).lock().attributes.clone(),
8242            AttrScope::Dataset(i) => self.ds(i).lock().attributes.clone(),
8243        };
8244        // Snapshot first: resolving a path locks group and dataset slots.
8245        let values: Vec<(String, Vec<String>, usize)> = self
8246            .attribute_references
8247            .lock()
8248            .iter()
8249            .filter(|r| r.scope == scope)
8250            .map(|r| (r.name.clone(), r.targets.clone(), r.stride))
8251            .collect();
8252        let width = self.ctx.sizeof_addr as usize;
8253        for (name, targets, stride) in values {
8254            let Some(pos) = attrs.iter().position(|a| a.name() == name) else {
8255                continue;
8256            };
8257            let Some(msg) = attrs[pos].readable() else {
8258                continue;
8259            };
8260            let mut msg = msg.clone();
8261            for (i, target) in targets.iter().enumerate() {
8262                let at = i * stride;
8263                let held = msg.data.len();
8264                let slot = msg.data.get_mut(at..at + width).ok_or_else(|| {
8265                    crate::io::IoError::InvalidState(format!(
8266                        "attribute '{name}' holds {held} bytes, too few for reference {i} at {at}"
8267                    ))
8268                })?;
8269                slot.copy_from_slice(
8270                    &self.object_reference_address(target)?.to_le_bytes()[..width],
8271                );
8272            }
8273            attrs[pos] = AttributeEntry::from(msg).with_creation_index(attrs[pos].creation_index());
8274        }
8275        Ok(attrs)
8276    }
8277
8278    /// The registry scope `target` names — the same object
8279    /// [`with_attr_list`](Self::with_attr_list) reaches, as the key the
8280    /// reference-value registry is indexed by. Refuses what that accessor
8281    /// refuses, and for the same reasons.
8282    fn attr_scope(&self, target: AttrTarget<'_>) -> IoResult<AttrScope> {
8283        match target {
8284            AttrTarget::Root => Ok(AttrScope::Root),
8285            AttrTarget::Group(path) => {
8286                let path = self.canonical_group_path(path);
8287                self.group_refs()
8288                    .iter()
8289                    .position(|g| {
8290                        let gg = g.lock();
8291                        gg.name == path && !gg.deleted
8292                    })
8293                    .map(AttrScope::Group)
8294                    .ok_or_else(|| {
8295                        crate::io::IoError::NotFound(format!("group '{path}' not found"))
8296                    })
8297            }
8298            AttrTarget::Dataset(index) => {
8299                let count = self.dataset_count();
8300                if index >= count {
8301                    return Err(crate::io::IoError::InvalidState(format!(
8302                        "dataset index {index} out of range (have {count})"
8303                    )));
8304                }
8305                Ok(AttrScope::Dataset(index))
8306            }
8307        }
8308    }
8309
8310    /// Drop the reference value recorded for `scope`'s attribute `name`.
8311    ///
8312    /// Called by both owners of attribute-list mutation —
8313    /// [`insert_attribute`](Self::insert_attribute) and
8314    /// [`evict_attr`](Self::evict_attr) — so an attribute that is replaced or
8315    /// removed cannot leave its value behind for whatever takes its name next.
8316    /// A string attribute written over a reference attribute is the case that
8317    /// needs it: without this the string's bytes would be overwritten with
8318    /// addresses at finalize.
8319    fn forget_attribute_reference(&self, scope: AttrScope, name: &str) {
8320        self.attribute_references
8321            .lock()
8322            .retain(|r| !(r.scope == scope && r.name == name));
8323    }
8324
8325    /// Write every pending object reference element as its target's object
8326    /// header address.
8327    ///
8328    /// INVARIANT: a reference element on disk holds its target's header
8329    /// address. Reached through [`write_reference_values`](Self::write_reference_values),
8330    /// which places it after every header has an address; a target that no
8331    /// longer resolves fails the finalize rather than leaving a placeholder
8332    /// behind.
8333    fn write_object_reference_values(&mut self) -> IoResult<()> {
8334        // Snapshot rather than drain: a SWMR session finalizes twice, and the
8335        // close-time finalize rebuilds every header at a fresh address, so the
8336        // elements must be stamped again with the addresses that survive.
8337        let pending: Vec<(usize, u64, String)> = self
8338            .pending_object_references
8339            .lock()
8340            .iter()
8341            .map(|p| (p.dataset, p.element, p.target.clone()))
8342            .collect();
8343        for (dataset, element, target) in &pending {
8344            let addr = match self.object_reference_target(target)? {
8345                Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8346                Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8347                None => self.root_group_addr.ok_or_else(|| {
8348                    crate::io::IoError::InvalidState(
8349                        "root group header address is not assigned yet".into(),
8350                    )
8351                })?,
8352            };
8353            // The element image is the dataset's own datatype's business: the
8354            // pre-1.12 and 1.12 forms differ in width and in layout, and the
8355            // dataset says which it holds.
8356            let (kind, width) = {
8357                let ds = self.ds(*dataset);
8358                let m = ds.lock();
8359                let DatatypeMessage::Reference { kind, size } = &m.datatype else {
8360                    return Err(crate::io::IoError::InvalidState(format!(
8361                        "dataset '{}' is no longer a reference dataset",
8362                        m.name
8363                    )));
8364                };
8365                (*kind, *size as usize)
8366            };
8367            let image = match kind {
8368                ReferenceKind::Object1 => ReferenceElementImage::Legacy(addr),
8369                ReferenceKind::Object2 => ReferenceElementImage::Inline(addr),
8370                other => {
8371                    return Err(crate::io::IoError::InvalidState(format!(
8372                        "dataset {dataset} now holds {other:?} elements, not object references"
8373                    )))
8374                }
8375            };
8376            let image = encode_reference_element(&image, width, &self.ctx)?;
8377            let data_addr = self.local_element_block(*dataset, "object references")?;
8378            let at = data_addr + element * width as u64;
8379            self.handle.write_at(at, &image)?;
8380        }
8381        Ok(())
8382    }
8383
8384    /// Store region references over `targets` into the elements of dataset
8385    /// `index` starting at `start`.
8386    ///
8387    /// Each target is the path of a dataset and a selection over it. What the
8388    /// element holds is a global-heap id — collection address then object index
8389    /// (`H5R__encode_heap`) — and the heap object it names is the target's
8390    /// object header address followed by the serialized selection
8391    /// (`H5R__encode_token_region_compat`). Both the object and the element are
8392    /// written here; only the address inside the object waits for
8393    /// [`Self::write_heap_reference_values`]. Elements never written keep the
8394    /// zero image libhdf5 reads back as a null reference.
8395    pub fn write_region_references(
8396        &self,
8397        index: usize,
8398        start: u64,
8399        targets: &[(&str, Selection)],
8400    ) -> IoResult<()> {
8401        let elements = {
8402            let ds = self.ds(index);
8403            let m = ds.lock();
8404            match &m.datatype {
8405                DatatypeMessage::Reference {
8406                    kind: ReferenceKind::DatasetRegion1,
8407                    ..
8408                } => {}
8409                other => {
8410                    return Err(crate::io::IoError::InvalidState(format!(
8411                        "dataset '{}' has datatype {other}, not a region reference",
8412                        m.name
8413                    )))
8414                }
8415            }
8416            m.dataspace
8417                .dims
8418                .iter()
8419                .fold(1u64, |a, &d| a.saturating_mul(d))
8420        };
8421        let data_addr = self.local_element_block(index, "region references")?;
8422        let end = start.saturating_add(targets.len() as u64);
8423        if end > elements {
8424            return Err(crate::io::IoError::InvalidState(format!(
8425                "elements {start}..{end} are outside the dataset's {elements}"
8426            )));
8427        }
8428
8429        // Build every heap object before inserting any: a path that names no
8430        // dataset, or a selection its extent does not admit, is reported at the
8431        // call that got it wrong rather than after half the batch is on disk.
8432        let sa = self.ctx.sizeof_addr as usize;
8433        let mut blobs = Vec::with_capacity(targets.len());
8434        for (path, selection) in targets {
8435            let target = self.region_reference_target(path)?;
8436            let dims = self.ds(target).lock().dataspace.dims.clone();
8437            validate_region_selection(selection, &dims, path)?;
8438            let mut blob = vec![0u8; sa];
8439            blob.extend_from_slice(&selection.encode()?);
8440            blobs.push(blob);
8441        }
8442        let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
8443        let placements = self.insert_vlen_objects(&items)?;
8444
8445        let width = (sa + 4) as u64;
8446        let mut pending = self.pending_heap_references.lock();
8447        for (i, &(collection, obj_index)) in placements.iter().enumerate() {
8448            let mut elem = Vec::with_capacity(width as usize);
8449            elem.extend_from_slice(&collection.to_le_bytes()[..sa]);
8450            elem.extend_from_slice(&u32::from(obj_index).to_le_bytes());
8451            self.handle
8452                .write_at(data_addr + (start + i as u64) * width, &elem)?;
8453            pending.push(PendingHeapReference {
8454                collection,
8455                index: obj_index,
8456                token_offset: 0,
8457                target: PendingHeapTarget::Dataset(targets[i].0.to_string()),
8458            });
8459        }
8460        Ok(())
8461    }
8462
8463    /// Store 1.12 references over `targets` into the elements of dataset
8464    /// `index` starting at `start` — the `H5T_STD_REF` trio.
8465    ///
8466    /// One datatype holds all three kinds, because a 1.12 element leads with
8467    /// the kind it holds; which is why this takes a [`ReferenceTarget`] per
8468    /// element rather than a fixed kind. `H5R_OBJECT2` needs nothing but the
8469    /// target's address, so its element is written inline by the same finalize
8470    /// pass every object reference goes through. The other two encode a
8471    /// selection or an attribute name alongside the token, which does not fit
8472    /// an element, so what is stored is a global-heap blob and the element is
8473    /// its id (`H5T__ref_disk_write`). Elements never written keep the zero
8474    /// image `H5T__ref_disk_isnull` reads back as a null reference.
8475    pub fn write_revised_references(
8476        &self,
8477        index: usize,
8478        start: u64,
8479        targets: &[(&str, ReferenceTarget)],
8480    ) -> IoResult<()> {
8481        let (width, elements) = {
8482            let ds = self.ds(index);
8483            let m = ds.lock();
8484            match &m.datatype {
8485                DatatypeMessage::Reference {
8486                    kind: ReferenceKind::Object2,
8487                    size,
8488                } => (
8489                    *size as u64,
8490                    m.dataspace
8491                        .dims
8492                        .iter()
8493                        .fold(1u64, |a, &d| a.saturating_mul(d)),
8494                ),
8495                other => {
8496                    return Err(crate::io::IoError::InvalidState(format!(
8497                        "dataset '{}' has datatype {other}, not the 1.12 H5T_STD_REF",
8498                        m.name
8499                    )))
8500                }
8501            }
8502        };
8503        let data_addr = self.local_element_block(index, "references")?;
8504        let end = start.saturating_add(targets.len() as u64);
8505        if end > elements {
8506            return Err(crate::io::IoError::InvalidState(format!(
8507                "elements {start}..{end} are outside the dataset's {elements}"
8508            )));
8509        }
8510
8511        // Build every blob before inserting any, so a path that names nothing,
8512        // a selection an extent does not admit or an attribute that does not
8513        // exist is reported at the call that got it wrong rather than after
8514        // half the batch is on disk.
8515        let mut blobs: Vec<(u64, ReferenceKind, PendingHeapTarget, Vec<u8>)> = Vec::new();
8516        let mut inline: Vec<(u64, String)> = Vec::new();
8517        for (i, (path, target)) in targets.iter().enumerate() {
8518            let element = start + i as u64;
8519            // The rank of the extent the selection is over, which only a region
8520            // reference encodes and takes from the target's dataspace.
8521            let mut extent_rank = 0;
8522            let (kind, pending) = match target {
8523                ReferenceTarget::Object => {
8524                    self.object_reference_target(path)?;
8525                    inline.push((element, (*path).to_string()));
8526                    continue;
8527                }
8528                ReferenceTarget::Region(selection) => {
8529                    let ds = self.region_reference_target(path)?;
8530                    let dims = self.ds(ds).lock().dataspace.dims.clone();
8531                    validate_region_selection(selection, &dims, path)?;
8532                    extent_rank = dims.len();
8533                    (
8534                        ReferenceKind::DatasetRegion2,
8535                        PendingHeapTarget::Dataset((*path).to_string()),
8536                    )
8537                }
8538                ReferenceTarget::Attribute(name) => {
8539                    let scope = match self.object_reference_target(path)? {
8540                        Some(HardLinkTarget::Dataset(i)) => AttrScope::Dataset(i),
8541                        Some(HardLinkTarget::Group(i)) => AttrScope::Group(i),
8542                        None => AttrScope::Root,
8543                    };
8544                    if !self
8545                        .object_attributes(scope)?
8546                        .iter()
8547                        .any(|a| a.name() == name)
8548                    {
8549                        return Err(crate::io::IoError::NotFound(format!(
8550                            "attribute '{name}' of reference target '{path}'"
8551                        )));
8552                    }
8553                    (
8554                        ReferenceKind::Attr,
8555                        PendingHeapTarget::Object((*path).to_string()),
8556                    )
8557                }
8558            };
8559            blobs.push((
8560                element,
8561                kind,
8562                pending,
8563                encode_revised_blob(0, target, extent_rank, &self.ctx)?,
8564            ));
8565        }
8566
8567        let items: Vec<&[u8]> = blobs.iter().map(|(_, _, _, b)| b.as_slice()).collect();
8568        let placements = self.insert_vlen_objects(&items)?;
8569
8570        let mut pending = self.pending_heap_references.lock();
8571        for ((element, kind, target, blob), &(collection, obj_index)) in
8572            blobs.iter().zip(&placements)
8573        {
8574            // The size the element declares is the heap object's own byte
8575            // count: `H5VL__native_blob_get` refuses to read one whose size
8576            // does not match what the element says.
8577            let image = encode_reference_element(
8578                &ReferenceElementImage::Blob {
8579                    kind: *kind,
8580                    size: blob.len() as u32,
8581                    collection,
8582                    index: u32::from(obj_index),
8583                },
8584                width as usize,
8585                &self.ctx,
8586            )?;
8587            self.handle.write_at(data_addr + element * width, &image)?;
8588            pending.push(PendingHeapReference {
8589                collection,
8590                index: obj_index,
8591                token_offset: REVISED_BLOB_TOKEN_OFFSET,
8592                target: target.clone(),
8593            });
8594        }
8595        drop(pending);
8596
8597        let mut pending = self.pending_object_references.lock();
8598        for (element, path) in inline {
8599            pending.push(PendingObjectReference {
8600                dataset: index,
8601                element,
8602                target: path,
8603            });
8604        }
8605        Ok(())
8606    }
8607
8608    /// The dataset a region reference's path names.
8609    ///
8610    /// A region reference names a *dataset*: `H5Rcreate` with
8611    /// `H5R_DATASET_REGION` takes the dataspace of one, and every reader
8612    /// dereferences it as one. A path that resolves to a group — or to the root
8613    /// group, which has no registry slot — is refused here rather than stored
8614    /// as a reference nothing can dereference.
8615    fn region_reference_target(&self, path: &str) -> IoResult<usize> {
8616        match self.object_reference_target(path)? {
8617            Some(HardLinkTarget::Dataset(i)) => Ok(i),
8618            _ => Err(crate::io::IoError::InvalidState(format!(
8619                "region reference target '{path}' is not a dataset"
8620            ))),
8621        }
8622    }
8623
8624    /// Stamp every pending heap-backed reference's object with its target's
8625    /// object header address.
8626    ///
8627    /// The references that are still stamped rather than written once: the
8628    /// *element* is a global-heap id, so the heap object has to exist at the
8629    /// call that stores the reference, long before any address does. The object
8630    /// was inserted with its token zeroed, so its size does not change here:
8631    /// each collection is read once, patched, and rewritten at its own declared
8632    /// size, which leaves every element's heap id valid — and leaves the
8633    /// object's byte count equal to the size the 1.12 element declares, which
8634    /// `H5VL__native_blob_get` refuses to read past.
8635    fn write_heap_reference_values(&mut self) -> IoResult<()> {
8636        use crate::format::global_heap::GlobalHeapCollection;
8637
8638        // Snapshot rather than drain, for the same reason the object-reference
8639        // pass does: a SWMR session finalizes twice and the close-time finalize
8640        // rebuilds every header at a fresh address.
8641        let pending: Vec<(u64, u16, usize, PendingHeapTarget)> = self
8642            .pending_heap_references
8643            .lock()
8644            .iter()
8645            .map(|p| (p.collection, p.index, p.token_offset, p.target.clone()))
8646            .collect();
8647        if pending.is_empty() {
8648            return Ok(());
8649        }
8650        let sa = self.ctx.sizeof_addr as usize;
8651        // Group by collection so one holding several references is read and
8652        // rewritten once.
8653        let mut per_collection: std::collections::BTreeMap<u64, Vec<(u16, usize, u64)>> =
8654            Default::default();
8655        for (collection, index, token_offset, target) in &pending {
8656            let addr = match target {
8657                PendingHeapTarget::Dataset(path) => {
8658                    let ds = self.region_reference_target(path)?;
8659                    self.ds(ds).lock().obj_header_addr
8660                }
8661                PendingHeapTarget::Object(path) => self.object_reference_address(path)?,
8662            };
8663            per_collection
8664                .entry(*collection)
8665                .or_default()
8666                .push((*index, *token_offset, addr));
8667        }
8668        for (collection, patches) in per_collection {
8669            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
8670            // exactly that, so one read usually covers the whole image.
8671            let mut image = self.handle.read_at_most(collection, 4096)?;
8672            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
8673            if declared > image.len() {
8674                image = self.handle.read_at(collection, declared)?;
8675            }
8676            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
8677            for (index, token_offset, addr) in patches {
8678                let token = gcol
8679                    .objects
8680                    .iter_mut()
8681                    .find(|o| o.index == index)
8682                    .and_then(|o| o.data.get_mut(token_offset..token_offset + sa))
8683                    .ok_or_else(|| {
8684                        crate::io::IoError::InvalidState(format!(
8685                            "object {index} of global heap collection {collection:#x} is no \
8686                             longer the reference written into it"
8687                        ))
8688                    })?;
8689                token.copy_from_slice(&addr.to_le_bytes()[..sa]);
8690            }
8691            let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
8692            self.handle.write_at(collection, &rewritten)?;
8693        }
8694        Ok(())
8695    }
8696
8697    /// Give every reference written this session its target's object header
8698    /// address.
8699    ///
8700    /// INVARIANT: no file is closed holding a reference whose target address is
8701    /// still the placeholder its write left. Both finalize paths call this in
8702    /// the content phase — after
8703    /// [`allocate_object_headers`](Self::allocate_object_headers), so every
8704    /// address exists, and before any object header is written — and this is
8705    /// the only caller of the per-kind passes, so a reference kind added later
8706    /// is written at both finalize sites or at neither. A target that no longer
8707    /// resolves fails the finalize rather than leaving a placeholder behind.
8708    ///
8709    /// This covers the two reference kinds whose value lives outside an object
8710    /// header. An attribute's value lives *inside* one, so it has no pass here:
8711    /// [`object_attributes`](Self::object_attributes) says it in addresses as
8712    /// the header is built.
8713    fn write_reference_values(&mut self) -> IoResult<()> {
8714        self.write_object_reference_values()?;
8715        self.write_heap_reference_values()
8716    }
8717
8718    /// Append every user-created hard link whose parent group is `parent`
8719    /// (`None` == the root group). Called while collecting a group's links,
8720    /// once every object's header address has been assigned.
8721    fn push_hard_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8722        for link in self.hard_links_vec() {
8723            if link.parent != parent || !self.hard_link_emitted(&link) {
8724                continue;
8725            }
8726            let addr = match link.target {
8727                HardLinkTarget::Dataset(i) => self.ds(i).lock().obj_header_addr,
8728                HardLinkTarget::Group(i) => self.grp(i).lock().obj_header_addr,
8729            };
8730            links.push((link.creation_seq, LinkMessage::hard(&link.name, addr)));
8731        }
8732    }
8733
8734    /// Append every user-created symbolic link whose parent group is `parent`
8735    /// (`None` == the root group).
8736    ///
8737    /// Nothing here waits on the layout pass — the link's value is a path, not
8738    /// an address — but it is collected with the rest so it takes its place in
8739    /// creation order and counts toward the phase change.
8740    fn push_symbolic_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8741        for link in self.symbolic_links_vec() {
8742            if link.parent != parent || !self.symbolic_link_emitted(&link) {
8743                continue;
8744            }
8745            links.push((
8746                link.creation_seq,
8747                LinkMessage {
8748                    name: link.name.clone(),
8749                    target: link.target.clone(),
8750                    creation_order: None,
8751                    cset: CharacterSet::for_name(&link.name),
8752                },
8753            ));
8754        }
8755    }
8756
8757    /// Refuse a caller path that would have to leave this file through one of
8758    /// the external links a reopened file brought in.
8759    ///
8760    /// The reader follows such a path into the file the link names; the writer
8761    /// cannot, because it models one file and would have to write into
8762    /// another. Saying which link stops the path — rather than reporting the
8763    /// name as absent, or worse, creating a second link of that name beside
8764    /// it — is the whole of what write mode does here.
8765    pub(crate) fn reject_external_traversal(&self, path: &str) -> IoResult<()> {
8766        let path = path.trim_start_matches('/');
8767        let crossing = self.preserved_link_paths().into_iter().find(|(p, class)| {
8768            matches!(class, crate::io::reader::LinkClass::External { .. })
8769                && (path == p || path.starts_with(&format!("{p}/")))
8770        });
8771        match crossing {
8772            None => Ok(()),
8773            Some((link, crate::io::reader::LinkClass::External { file, path: target })) => {
8774                Err(crate::io::IoError::Unsupported(format!(
8775                    "'{path}' resolves through the external link '{link}' to '{target}' in \
8776                     '{file}'; this writer carries external links through a rewrite but does \
8777                     not open the file they name"
8778                )))
8779            }
8780            // `find` matched on the External arm, so no other class reaches here.
8781            Some(_) => Ok(()),
8782        }
8783    }
8784
8785    /// Resolve `name` to a live dataset index, reporting *why* it does not
8786    /// resolve rather than collapsing every cause into absence.
8787    ///
8788    /// The write-mode counterpart of [`Hdf5Reader::open_dataset`]: the single
8789    /// gate every by-name dataset lookup in write mode goes through.
8790    ///
8791    /// [`Hdf5Reader::open_dataset`]: crate::io::reader::Hdf5Reader::open_dataset
8792    pub(crate) fn open_dataset_index(&self, name: &str) -> IoResult<usize> {
8793        self.reject_external_traversal(name)?;
8794        self.reject_preserved_object(name)?;
8795        self.dataset_index(name)
8796            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
8797    }
8798
8799    /// Refuse a caller path that names an object the reopen kept by its bytes
8800    /// rather than modelling.
8801    ///
8802    /// Such an object is in the file and stays in it, but this writer holds
8803    /// none of what it would need to read or rewrite it. Saying so — with the
8804    /// reason the classification recorded — is the difference between an
8805    /// object the writer will not touch and a name the file does not have.
8806    pub(crate) fn reject_preserved_object(&self, path: &str) -> IoResult<()> {
8807        let path = path.trim_start_matches('/');
8808        let objects: Vec<(String, String)> = {
8809            let preserved = self.preserved_links.lock();
8810            preserved
8811                .iter()
8812                .filter_map(|l| {
8813                    l.reason
8814                        .as_ref()
8815                        .map(|why| (self.preserved_link_full_path(l), why.clone()))
8816                })
8817                .collect()
8818        };
8819        match objects
8820            .into_iter()
8821            .find(|(full, _)| path == full || path.starts_with(&format!("{full}/")))
8822        {
8823            None => Ok(()),
8824            Some((link, why)) => Err(crate::io::IoError::Unsupported(format!(
8825                "'{path}' is, or is inside, the object '{link}', which this file's reopen \
8826                 kept exactly as it found it because {why}"
8827            ))),
8828        }
8829    }
8830
8831    /// Every link this writer will emit that names a *path* rather than an
8832    /// object, with the class a listing reports for it: the soft and external
8833    /// links created this session, and the ones a reopen is carrying through.
8834    ///
8835    /// The object listings answer for hard links, so a write-mode link
8836    /// listing is this plus those; keeping both sources in one place is what
8837    /// stops a listing from seeing a kind the class lookup does not, or the
8838    /// reverse.
8839    pub(crate) fn path_link_classes(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8840        let mut out: Vec<(String, crate::io::reader::LinkClass)> = self
8841            .symbolic_links_vec()
8842            .iter()
8843            .filter(|l| self.symbolic_link_emitted(l))
8844            .map(|l| {
8845                (
8846                    self.symbolic_link_full_path(l),
8847                    crate::io::reader::LinkClass::from_target(&l.target),
8848                )
8849            })
8850            .collect();
8851        out.extend(self.preserved_link_paths());
8852        out
8853    }
8854
8855    /// Every link this writer is carrying but cannot express, by full path.
8856    pub(crate) fn preserved_link_paths(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8857        self.preserved_links
8858            .lock()
8859            .iter()
8860            .map(|l| (self.preserved_link_full_path(l), l.class.clone()))
8861            .collect()
8862    }
8863
8864    /// The full path of a preserved link: its parent group's path plus its
8865    /// leaf name, in the no-leading-`/` form the registry uses.
8866    fn preserved_link_full_path(&self, link: &PreservedLink) -> String {
8867        match link.parent {
8868            None => link.name.clone(),
8869            Some(gi) => {
8870                let group = self.grp(gi).lock().name.clone();
8871                format!("{}/{}", group.trim_start_matches('/'), link.name)
8872            }
8873        }
8874    }
8875
8876    /// The single owner of "which links does this group hold", in the order
8877    /// they were created and, when the file tracks creation order, stamped
8878    /// with it.
8879    ///
8880    /// Both the compact form (one `MSG_LINK` per link) and the dense form (the
8881    /// same messages inside a fractal heap) are built from this one list, so
8882    /// the phase-change decision, the storage it selects and the creation
8883    /// order recorded in either can never disagree about what the group
8884    /// contains.
8885    fn group_links(&self, scope: LinkScope, order: CreationOrder) -> Vec<LinkMessage> {
8886        let mut links: Vec<(u64, LinkMessage)> = Vec::new();
8887        match scope {
8888            LinkScope::Root => {
8889                // Datasets that belong to a subgroup are that group's links,
8890                // not the root's. Each group slot is locked one at a time.
8891                let mut datasets_in_subgroups: std::collections::HashSet<usize> =
8892                    std::collections::HashSet::new();
8893                for grp in self.group_refs() {
8894                    let g = grp.lock();
8895                    if g.deleted {
8896                        continue;
8897                    }
8898                    datasets_in_subgroups.extend(g.child_datasets.iter().copied());
8899                }
8900                // `dataset_refs` preserves registry order, so `enumerate`
8901                // yields each dataset's true index.
8902                for (i, ds) in self.dataset_refs().into_iter().enumerate() {
8903                    let m = ds.lock();
8904                    if m.deleted || datasets_in_subgroups.contains(&i) {
8905                        continue;
8906                    }
8907                    // The leaf, never the registry path: a link name is one
8908                    // path component, and `H5G_traverse` would split a '/'
8909                    // in it before `H5L_link` ever saw the name.
8910                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8911                    links.push((
8912                        m.creation_seq,
8913                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8914                    ));
8915                }
8916                for grp in self.group_refs() {
8917                    let g = grp.lock();
8918                    if g.deleted || g.parent.is_some() {
8919                        continue;
8920                    }
8921                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8922                    links.push((
8923                        g.creation_seq,
8924                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8925                    ));
8926                }
8927                self.push_hard_links(&mut links, None);
8928                self.push_symbolic_links(&mut links, None);
8929                self.push_committed_datatypes(&mut links, None);
8930            }
8931            LinkScope::Group(group_idx) => {
8932                // Snapshot the child lists, then drop the slot guard: the
8933                // per-child reads below re-lock dataset and group slots
8934                // (including this one).
8935                let (child_datasets, child_groups) = {
8936                    let grp = self.grp(group_idx);
8937                    let g = grp.lock();
8938                    (g.child_datasets.clone(), g.child_groups.clone())
8939                };
8940                for ds_idx in child_datasets {
8941                    let ds = self.ds(ds_idx);
8942                    let m = ds.lock();
8943                    if m.deleted {
8944                        continue;
8945                    }
8946                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8947                    links.push((
8948                        m.creation_seq,
8949                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8950                    ));
8951                }
8952                for child_idx in child_groups {
8953                    let child_grp = self.grp(child_idx);
8954                    let g = child_grp.lock();
8955                    if g.deleted {
8956                        continue;
8957                    }
8958                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8959                    links.push((
8960                        g.creation_seq,
8961                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8962                    ));
8963                }
8964                self.push_hard_links(&mut links, Some(group_idx));
8965                self.push_symbolic_links(&mut links, Some(group_idx));
8966                self.push_committed_datatypes(&mut links, Some(group_idx));
8967            }
8968        }
8969        // Creation order, not order by kind: a run of create_group and
8970        // create_dataset draws from one counter, so this is the order the
8971        // caller made them in. `H5G_obj_insert` numbers from zero within the
8972        // group, so the rank here is the link's creation order.
8973        links.sort_by_key(|(seq, _)| *seq);
8974        links
8975            .into_iter()
8976            .enumerate()
8977            .map(|(rank, (_, link))| {
8978                if order.is_tracked() {
8979                    link.with_creation_order(rank as i64)
8980                } else {
8981                    link
8982                }
8983            })
8984            .collect()
8985    }
8986
8987    /// Whether `links` must live in dense storage rather than in the group's
8988    /// object header — the `H5G_obj_insert` phase-change rule, applied to the
8989    /// whole set at once because this writer builds each header from scratch
8990    /// rather than inserting one link at a time.
8991    ///
8992    /// libhdf5 converts when the count *reaches* `max_compact` and another
8993    /// link arrives, so a set of exactly `max_compact` is still compact; and
8994    /// separately when one message would not fit the 16-bit size field an
8995    /// object header message has.
8996    ///
8997    /// The answer depends only on the link names and kinds, never on the
8998    /// addresses they point at, which is what lets a group header be sized
8999    /// before [`prepare_dense_links`](Self::prepare_dense_links) has run.
9000    fn links_need_dense(&self, links: &[LinkMessage]) -> bool {
9001        links.len() > MAX_COMPACT_LINKS
9002            || links
9003                .iter()
9004                .any(|l| l.encode(&self.ctx).len() > MAX_MESSAGE_SIZE)
9005    }
9006
9007    /// The single owner of link emission into a group object header: the Link
9008    /// Info and Group Info messages, and then either one `MSG_LINK` per link
9009    /// or nothing at all when the set has spilled to dense storage.
9010    ///
9011    /// The two storage forms are exclusive (`H5G_obj_insert` moves the whole
9012    /// set at once), and a header carrying both would report every link twice.
9013    ///
9014    /// A group whose links are dense but not yet laid out gets a compact Link
9015    /// Info message here. That is deliberate: the message encodes to the same
9016    /// length either way — two addresses, defined or not — so the sizing pass
9017    /// that runs before `prepare_dense_links` still reserves the right number
9018    /// of bytes, and the write pass that runs after it emits the real heap and
9019    /// index addresses. It is the same two-pass rule the child link addresses
9020    /// already follow.
9021    fn emit_links(
9022        &self,
9023        header: &mut ObjectHeader,
9024        scope: LinkScope,
9025        links: &[LinkMessage],
9026        order: CreationOrder,
9027    ) {
9028        // A symbol-table group holds no link messages at all: its links are the
9029        // entries of the symbol table `prepare_symbol_tables` laid out, and
9030        // the header carries only the two addresses naming it. Link Info and
9031        // Group Info are version-1.8 messages and have no business in a
9032        // version-1 header — `H5G__stab_valid` reads the Symbol Table message
9033        // and nothing else.
9034        if self.uses_symbol_table(scope, order) {
9035            // Sizing runs before the tables are laid out; the message is the
9036            // same two addresses wide either way, so the placeholder reserves
9037            // exactly what the real one needs. Same two-pass rule the child
9038            // link addresses already follow.
9039            let stab = self
9040                .symbol_tables
9041                .written
9042                .lock()
9043                .get(&scope)
9044                .copied()
9045                .unwrap_or(Stab {
9046                    btree_addr: UNDEF_ADDR,
9047                    heap_addr: UNDEF_ADDR,
9048                });
9049            header.add_message(MSG_SYMBOL_TABLE, 0x00, stab.encode(&self.ctx));
9050            return;
9051        }
9052        // Links a reopen carried through verbatim because this writer cannot
9053        // express them. They are emitted here rather than by a second caller
9054        // so that no header-rewrite path can drop them, and their presence
9055        // pins the group to compact storage: dense storage would have to
9056        // re-encode each link into the heap, which is exactly the byte
9057        // fidelity preserving them is for.
9058        let preserved = self.preserved_links_for(scope);
9059        let dense = preserved.is_empty() && self.links_need_dense(links);
9060        let link_info = self.dense_links.lock().get(&scope).cloned();
9061        let link_info = link_info.unwrap_or_else(|| {
9062            let mut info = LinkInfoMessage::compact();
9063            if order.is_tracked() {
9064                // `H5G__obj_insert` post-increments `max_corder`, so a group
9065                // holding n links reports n.
9066                info.max_creation_order = Some(links.len() as u64);
9067            }
9068            if order.is_indexed() {
9069                // The index address stays undefined while the links live in
9070                // the header, but the message must still carry the field:
9071                // `H5Pget_link_creation_order` reads INDEXED off this flag,
9072                // not off the address.
9073                info.creation_order_btree_address = Some(UNDEF_ADDR);
9074            }
9075            info
9076        });
9077        header.add_message(MSG_LINK_INFO, 0x00, link_info.encode(&self.ctx));
9078        // The link info message takes no flags and the group info message
9079        // takes `H5O_MSG_FLAG_CONSTANT`, exactly as `H5G__obj_create_real`
9080        // creates the pair (H5Gobj.c:255, :259) and as
9081        // `H5G__obj_insert`'s phase change re-creates it (H5Gobj.c:526). The
9082        // asymmetry is real: the link info message records the group's
9083        // storage and its creation-order counter, both of which change as
9084        // links come and go, while the group info message holds the phase
9085        // change and estimated-name-length constants of the creation property
9086        // list, which nothing after creation rewrites.
9087        header.add_message(
9088            MSG_GROUP_INFO,
9089            MSG_FLAG_CONSTANT,
9090            GroupInfoMessage::default().encode(),
9091        );
9092        if dense {
9093            return;
9094        }
9095        for link in links {
9096            header.add_message(MSG_LINK, 0x00, link.encode(&self.ctx));
9097        }
9098        for encoded in preserved {
9099            header.add_message(MSG_LINK, 0x00, encoded);
9100        }
9101    }
9102
9103    /// The verbatim link bodies a reopen carried into `scope`.
9104    fn preserved_links_for(&self, scope: LinkScope) -> Vec<Vec<u8>> {
9105        let parent = match scope {
9106            LinkScope::Root => None,
9107            LinkScope::Group(i) => Some(i),
9108        };
9109        self.preserved_links
9110            .lock()
9111            .iter()
9112            .filter(|l| l.parent == parent)
9113            .map(|l| l.encoded.clone())
9114            .collect()
9115    }
9116
9117    /// Lay out and write dense link storage for every group that needs it,
9118    /// recording the resulting `Link Info` message per group.
9119    ///
9120    /// The sole owner of that transition. It must run after every object
9121    /// header address is assigned — the heap holds encoded link messages, and
9122    /// those name their targets — and before any group header is written.
9123    ///
9124    /// Every group whose header this finalize rewrites passes through here,
9125    /// dense or not: the storage a reopened header named is superseded by the
9126    /// rewrite whichever form the new link set takes, and freeing it first is
9127    /// what lets the replacement reuse those blocks.
9128    fn prepare_dense_links(&self) -> IoResult<()> {
9129        let mut scopes: Vec<(LinkScope, Vec<LinkMessage>, CreationOrder)> = Vec::new();
9130        for gi in 0..self.group_count() {
9131            let (deleted, order) = {
9132                let grp = self.grp(gi);
9133                let g = grp.lock();
9134                (g.deleted, g.track_order.links)
9135            };
9136            // A symbol-table group is `prepare_symbol_tables`' business; it
9137            // has no Link Info message to hold a fractal heap address, and it
9138            // never had dense storage to release.
9139            if deleted || self.uses_symbol_table(LinkScope::Group(gi), order) {
9140                continue;
9141            }
9142            self.release_superseded_dense_links(LinkScope::Group(gi))?;
9143            let links = self.group_links(LinkScope::Group(gi), order);
9144            if self.links_need_dense(&links) {
9145                scopes.push((LinkScope::Group(gi), links, order));
9146            }
9147        }
9148        let root_order = self.root_track_order.links;
9149        if !self.uses_symbol_table(LinkScope::Root, root_order) {
9150            self.release_superseded_dense_links(LinkScope::Root)?;
9151            let root_links = self.group_links(LinkScope::Root, root_order);
9152            if self.links_need_dense(&root_links) {
9153                scopes.push((LinkScope::Root, root_links, root_order));
9154            }
9155        }
9156
9157        for (scope, links, order) in scopes {
9158            // `close` after `start_swmr` finalizes a second time over the same
9159            // groups, so rebuilding here would allocate a whole second heap
9160            // and strand the one the published headers already name.
9161            if self.dense_links.lock().contains_key(&scope) {
9162                continue;
9163            }
9164            let dense = build_dense_links(&links, &self.ctx, order, &mut |len| {
9165                self.allocator.allocate(len, FreeSpaceClass::Metadata)
9166            })?;
9167            for block in &dense.blocks {
9168                self.handle.write_at(block.addr, &block.image)?;
9169            }
9170            self.dense_links.lock().insert(scope, dense.linfo);
9171        }
9172        Ok(())
9173    }
9174
9175    /// Lay out whichever of the two forms of link storage this file uses,
9176    /// before any group header is written.
9177    ///
9178    /// The two are exclusive because the formats are: a classic group has no
9179    /// Link Info message to put a fractal heap address in, and a link-message
9180    /// group has no symbol table.
9181    fn prepare_link_storage(&self) -> IoResult<()> {
9182        self.prepare_dense_links()?;
9183        self.prepare_symbol_tables()
9184    }
9185
9186    /// Lay out and write the symbol table of every classic group, and free the
9187    /// storage each rewrite supersedes. A no-op on a link-message file.
9188    ///
9189    /// The classic counterpart of [`prepare_dense_links`](Self::prepare_dense_links),
9190    /// and the sole owner of that transition. The same two placement rules
9191    /// apply for the same two reasons: it runs after every object header has
9192    /// an address, because a symbol table entry names its target's header, and
9193    /// before any group header is written, because the header carries the
9194    /// Symbol Table message naming what this laid out.
9195    ///
9196    /// Deepest group first, root last. A hard link to a group caches that
9197    /// group's own B-tree and heap in the entry's scratch pad
9198    /// (`H5G__link_to_ent`), so the child's table must exist before the
9199    /// parent's is built; `H5G__stab_valid` checks the root entry's cache
9200    /// against the root header's Symbol Table message, so a stale pair there
9201    /// is not a slow lookup but a file `H5Fopen` rejects.
9202    ///
9203    /// Every classic group is rebuilt on every pass — there is no "already
9204    /// done" short-circuit like the dense one, because the only way this runs
9205    /// twice is a `Drop` retry after a failed `close`, and the entries of the
9206    /// first pass name header addresses the second pass has moved. (A SWMR
9207    /// session, the other double-finalize, cannot reach here: SWMR needs a
9208    /// version-3 superblock, so `start_swmr` refuses a classic file.)
9209    fn prepare_symbol_tables(&self) -> IoResult<()> {
9210        // Depth by parent chain, not by counting separators in the registry
9211        // path: the chain is what actually says which table has to exist first.
9212        let mut scopes: Vec<(usize, LinkScope, CreationOrder)> = Vec::new();
9213        for gi in 0..self.group_count() {
9214            let (deleted, order, mut parent) = {
9215                let grp = self.grp(gi);
9216                let g = grp.lock();
9217                (g.deleted, g.track_order.links, g.parent)
9218            };
9219            if deleted || !self.uses_symbol_table(LinkScope::Group(gi), order) {
9220                continue;
9221            }
9222            let mut depth = 1usize;
9223            while let Some(p) = parent {
9224                depth += 1;
9225                parent = self.grp(p).lock().parent;
9226            }
9227            scopes.push((depth, LinkScope::Group(gi), order));
9228        }
9229        scopes.sort_by_key(|&(depth, ..)| std::cmp::Reverse(depth));
9230        let root_order = self.root_track_order.links;
9231        if self.uses_symbol_table(LinkScope::Root, root_order) {
9232            scopes.push((0, LinkScope::Root, root_order));
9233        }
9234
9235        let meta = self.stab_meta();
9236        for (_, scope, order) in scopes {
9237            // Freed before the replacement is laid out, so a rewrite reuses
9238            // the same blocks instead of growing the file on every open/close
9239            // cycle — the rule `prepare_dense_links` and the header rewrite
9240            // already follow. Removed as it is freed, so no second pass can
9241            // free it twice.
9242            let superseded = self.symbol_tables.superseded.lock().remove(&scope);
9243            if let Some(extents) = superseded {
9244                free_stab(&self.allocator, &extents);
9245            }
9246            let links = self.stab_links_for(scope, order)?;
9247            let stab = write_stab(&self.handle, &self.allocator, &meta, &links)?;
9248            self.symbol_tables.written.lock().insert(scope, stab);
9249        }
9250        Ok(())
9251    }
9252
9253    /// The file-level parameters every symbol-table node width is derived from
9254    /// — the address/length widths and the B-tree "K" ranks. Only a version-0/1
9255    /// superblock records ranks of its own; [`btree_v1_config`] is the one
9256    /// place that decides whether this file has any.
9257    ///
9258    /// [`btree_v1_config`]: Self::btree_v1_config
9259    fn stab_meta(&self) -> FileMeta {
9260        FileMeta {
9261            ctx: self.ctx,
9262            btree: self.btree_v1_config(),
9263            sohm: None,
9264        }
9265    }
9266
9267    /// `scope`'s links as symbol table entries.
9268    ///
9269    /// A link a reopen carried through verbatim is decoded back out of its
9270    /// encoded Link message here, because a classic group has no link message
9271    /// to preserve it into. Nothing is lost in the round trip: the walk built
9272    /// that message from a symbol table entry in the first place, and the two
9273    /// forms carry the same three facts.
9274    fn stab_links_for(&self, scope: LinkScope, order: CreationOrder) -> IoResult<Vec<StabLink>> {
9275        let groups = self.group_header_scopes();
9276        let mut out = Vec::new();
9277        for link in self.group_links(scope, order) {
9278            out.push(self.stab_link(&link, &groups)?);
9279        }
9280        for encoded in self.preserved_links_for(scope) {
9281            let (link, _) = LinkMessage::decode(&encoded, &self.ctx)?;
9282            out.push(self.stab_link(&link, &groups)?);
9283        }
9284        Ok(out)
9285    }
9286
9287    /// Where each group's object header now sits, so a hard link that lands on
9288    /// one can cache that group's symbol table in its scratch pad.
9289    fn group_header_scopes(&self) -> HashMap<u64, LinkScope> {
9290        let mut map = HashMap::new();
9291        for gi in 0..self.group_count() {
9292            let grp = self.grp(gi);
9293            let g = grp.lock();
9294            if !g.deleted {
9295                map.insert(g.obj_header_addr, LinkScope::Group(gi));
9296            }
9297        }
9298        map
9299    }
9300
9301    /// One link as a symbol table entry.
9302    ///
9303    /// The scratch pad caches the target group's B-tree and heap when the
9304    /// target is a group this pass has already laid out — what
9305    /// `H5G__link_to_ent` does, and what lets `H5G__stab_lookup` walk a path
9306    /// without opening each header on the way. For anything else the pad stays
9307    /// `H5G_NOTHING_CACHED`, the value libhdf5 itself writes whenever the
9308    /// target has no Symbol Table message to read.
9309    fn stab_link(
9310        &self,
9311        link: &LinkMessage,
9312        groups: &HashMap<u64, LinkScope>,
9313    ) -> IoResult<StabLink> {
9314        let target = match &link.target {
9315            LinkTarget::Hard { address } => {
9316                let cached = groups
9317                    .get(address)
9318                    .and_then(|scope| self.symbol_tables.written.lock().get(scope).copied());
9319                StabTarget::Hard {
9320                    addr: *address,
9321                    cached,
9322                }
9323            }
9324            LinkTarget::Soft { target } => StabTarget::Soft {
9325                value: target.clone(),
9326            },
9327            // Unreachable by construction: a group holding one of these is
9328            // not a symbol-table group at all
9329            // ([`LinkMessage::fits_symbol_table`] is what
9330            // [`Hdf5Writer::uses_symbol_table`] asks), so this pass never
9331            // visits it. Reported rather than panicked so a future caller
9332            // that skips that gate learns which link it lost.
9333            LinkTarget::External { .. } | LinkTarget::UserDefined { .. } => {
9334                return Err(crate::io::IoError::InvalidState(format!(
9335                    "cannot store the link {:?} in a symbol table: it holds only \
9336                     hard and soft links, and this group was not converted to link \
9337                     messages the way `H5G_obj_insert` converts it",
9338                    link.name
9339                )))
9340            }
9341        };
9342        Ok(StabLink {
9343            name: link.name.clone(),
9344            target,
9345        })
9346    }
9347
9348    /// The single owner of attribute emission into an object header: appends
9349    /// the Attribute Info message and then one `MSG_ATTRIBUTE` per attribute.
9350    ///
9351    /// On a version-2 object header the two are inseparable.
9352    /// `H5O__attr_count_real` derives `H5Oget_info().num_attrs` from the
9353    /// Attribute Info message alone — with no such message the count reads as
9354    /// zero however many attribute messages follow, which is what made every
9355    /// rust-written file report `num_attrs == 0` to libhdf5 while
9356    /// `H5Aiterate2` still yielded the attributes. The message carries no
9357    /// count of its own: `H5A__get_ainfo` fills `nattrs` from the attribute
9358    /// messages the header loader actually saw, so compact storage needs
9359    /// nothing but the message's presence.
9360    ///
9361    /// When [`prepare_dense_attributes`](Self::prepare_dense_attributes) has
9362    /// spilled `scope`'s attributes to a fractal heap, the same message names
9363    /// that heap instead and *no* attribute message follows: the two storage
9364    /// forms are exclusive (`H5O__attr_create` moves the whole set at once),
9365    /// and a header carrying both would report every attribute twice.
9366    fn emit_attributes(
9367        &self,
9368        header: &mut ObjectHeader,
9369        scope: AttrScope,
9370        attributes: &[AttributeEntry],
9371        order: CreationOrder,
9372        format: ObjectFormat,
9373        owner: ShareOwner,
9374    ) {
9375        // `H5Pget_attr_creation_order` reads the object header's own flags,
9376        // not the Attribute Info message, so this is what makes the object
9377        // report creation-ordered attributes — and tracking widens every
9378        // message envelope by the creation index below.
9379        let order = self.header_attr_order(order);
9380        header.set_attribute_creation_order(order);
9381        if attributes.is_empty() {
9382            return;
9383        }
9384        // A version-1 object header gets the attribute messages alone.
9385        // `H5O__attr_create` gates every mention of the Attribute Info message
9386        // on `oh->version > H5O_VERSION_1` (H5Oattribute.c:218), and so does
9387        // `H5O__attr_count_real`, which is why the count still reads correctly
9388        // without it: on a version-1 header libhdf5 counts the messages.
9389        if format == ObjectFormat::Legacy {
9390            for attr in attributes {
9391                header.add_message(MSG_ATTRIBUTE, 0x00, self.encode_attribute(attr));
9392            }
9393            return;
9394        }
9395        // Whether the set spills is a property of the set alone, so it is the
9396        // same answer in the pass that measures this header and in the pass
9397        // that writes it — even though the storage itself is laid out between
9398        // the two, because it can only be laid out once every object header
9399        // has an address. Sizing therefore falls back to a placeholder message
9400        // of the same width: only the creation-order flags change the
9401        // Attribute Info message's length, so the header measured here holds
9402        // the header written against the storage that replaces it. Same
9403        // two-pass rule `emit_links` follows for dense links and symbol
9404        // tables.
9405        let dense = self.attributes_need_dense(attributes, format);
9406        let stored = self.dense_attributes.lock().get(&scope).cloned();
9407        let ainfo = stored.unwrap_or_else(|| {
9408            let mut ainfo = AttributeInfoMessage::compact();
9409            if order.is_tracked() {
9410                ainfo.max_creation_index = Some(next_creation_index(attributes));
9411            }
9412            if order.is_indexed() {
9413                // Compact storage has no index B-tree, but the message still
9414                // announces one so that its flags match the header's
9415                // (`H5O__attr_create` asserts they agree).
9416                ainfo.creation_order_btree_address = Some(UNDEF_ADDR);
9417            }
9418            ainfo
9419        });
9420        header.add_message(MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, ainfo.encode(&self.ctx));
9421        if dense {
9422            return;
9423        }
9424        // Each attribute states its own creation index — the one it was
9425        // created with here, or the one the file it was read from records. An
9426        // attribute with none belongs to an object that tracks no order, where
9427        // the field is not encoded at all.
9428        for attr in attributes {
9429            let (flags, body) = self.share_attribute(attr, format, owner);
9430            header.add_message_indexed(
9431                MSG_ATTRIBUTE,
9432                flags,
9433                body,
9434                attr.creation_index().unwrap_or(0),
9435            );
9436        }
9437    }
9438
9439    /// One attribute message body, at the version this file's low library
9440    /// bound calls for (`H5A__set_version`, which reads the bound and nothing
9441    /// about the object the attribute hangs on).
9442    fn encode_attribute(&self, attr: &AttributeEntry) -> Vec<u8> {
9443        attr.encode_for(&self.ctx, self.encoding_libver(), self.message_format())
9444    }
9445
9446    /// What a header stores for one attribute: the message flags and the body,
9447    /// with the attribute's own datatype and dataspace shared wherever an
9448    /// index covers them.
9449    ///
9450    /// `H5A__create` offers both to `H5SM_try_share` (H5Aint.c:375-377) before
9451    /// `H5O__attr_create` offers the attribute itself (H5Oattribute.c:726), so
9452    /// the attribute body that reaches the heap already holds their pointers
9453    /// and says which fields they are in its own flags byte
9454    /// (`H5O_ATTR_FLAG_TYPE_SHARED` / `H5O_ATTR_FLAG_SPACE_SHARED`,
9455    /// H5Oattr.c:358-359). Both offers go through
9456    /// [`share_message`](Self::share_message) like any other, so the pass that
9457    /// counts references and the pass that substitutes see the same three
9458    /// messages.
9459    fn share_attribute(
9460        &self,
9461        attr: &AttributeEntry,
9462        format: ObjectFormat,
9463        owner: ShareOwner,
9464    ) -> (u8, Vec<u8>) {
9465        let libver = self.encoding_libver();
9466        // Only a readable attribute has pieces to offer: an unreadable one is
9467        // the bytes it was read from, put back as they were. Version 1 has no
9468        // flags byte to record a shared field in — `H5O__attr_encode` writes a
9469        // reserved zero there — so a classic file shares the attribute whole
9470        // or not at all.
9471        let Some(message) = attr.readable().filter(|_| format.attribute_version() >= 2) else {
9472            return self.share_message(
9473                owner,
9474                MSG_ATTRIBUTE,
9475                0x00,
9476                attr.encode_for(&self.ctx, libver, format),
9477            );
9478        };
9479
9480        let datatype = message.datatype.encode_at(&self.ctx, libver);
9481        let dataspace = message.dataspace.encode_for(&self.ctx, format);
9482        // `H5A__create` passes no open header for either (H5Aint.c:375-377):
9483        // both live inside the attribute's body, so neither has a header
9484        // message a `H5SM_IN_OH` record could name and both reach the heap on
9485        // first use.
9486        let (dt_flags, dt_field) =
9487            self.share_message(ShareOwner::Detached, MSG_DATATYPE, 0x00, datatype.clone());
9488        let (ds_flags, ds_field) =
9489            self.share_message(ShareOwner::Detached, MSG_DATASPACE, 0x00, dataspace.clone());
9490
9491        let mut attr_flags = 0u8;
9492        if dt_flags & MSG_FLAG_SHARED != 0 {
9493            attr_flags |= ATTR_FLAG_TYPE_SHARED;
9494        }
9495        if ds_flags & MSG_FLAG_SHARED != 0 {
9496            attr_flags |= ATTR_FLAG_SPACE_SHARED;
9497        }
9498        let encoded = message.encode_with_fields(attr_flags, &dt_field, &ds_field);
9499
9500        // Each shared field's heap ID sits two bytes into the pointer that
9501        // replaced it; the body offered below carries whatever
9502        // `share_message` just produced, which is a zeroed ID in the pass that
9503        // counts and the real one in the pass that substitutes.
9504        let mut nested = Vec::new();
9505        if attr_flags & ATTR_FLAG_TYPE_SHARED != 0 {
9506            nested.push(NestedShare {
9507                heap_id_at: encoded.datatype_at + SOHM_POINTER_HEAP_ID_AT,
9508                target: (MSG_DATATYPE, datatype),
9509            });
9510        }
9511        if attr_flags & ATTR_FLAG_SPACE_SHARED != 0 {
9512            nested.push(NestedShare {
9513                heap_id_at: encoded.dataspace_at + SOHM_POINTER_HEAP_ID_AT,
9514                target: (MSG_DATASPACE, dataspace),
9515            });
9516        }
9517        self.share_nesting_message(owner, MSG_ATTRIBUTE, 0x00, encoded.body, nested)
9518    }
9519
9520    /// Whether `attributes` must live in dense storage rather than in the
9521    /// object header — the `H5O__attr_create` phase-change rule, applied to
9522    /// the whole set at once because this writer builds each header from
9523    /// scratch rather than inserting one attribute at a time.
9524    ///
9525    /// libhdf5 converts when the count *reaches* `max_compact` and another
9526    /// attribute arrives, so a set of exactly `max_compact` is still compact;
9527    /// and separately when one message would not fit the 16-bit size field an
9528    /// object header message has.
9529    ///
9530    /// Never in a classic file. Dense attribute storage is a fractal heap
9531    /// reached through an Attribute Info message, both introduced in the 1.8
9532    /// format; at `H5F_LIBVER_EARLIEST` libhdf5 keeps every attribute in the
9533    /// header however many there are (`H5O__attr_create` reaches the phase
9534    /// change only when the object header version allows it). An attribute
9535    /// too large for the 16-bit size field is then an error, which
9536    /// `ObjectHeader::encode_v1` raises, rather than a reason to spill.
9537    fn attributes_need_dense(&self, attributes: &[AttributeEntry], format: ObjectFormat) -> bool {
9538        if format == ObjectFormat::Legacy {
9539            return false;
9540        }
9541        attributes.len() > MAX_COMPACT_ATTRS
9542            || attributes
9543                .iter()
9544                .any(|a| self.encode_attribute(a).len() > MAX_MESSAGE_SIZE)
9545    }
9546
9547    /// Every object whose attributes this finalize re-lays-out, with the
9548    /// creation-order policy each one's storage must follow.
9549    ///
9550    /// `datasets` lists the datasets whose headers this finalize will
9551    /// actually write. A reopened dataset that took no writes keeps its
9552    /// original header — and with it whatever storage that header already
9553    /// names — so touching its attribute storage would strand every block of
9554    /// it.
9555    ///
9556    /// The policy is the one the *header* records, not the one the object's
9557    /// creation property list asked for: those differ on a file whose
9558    /// shared-message configuration covers attributes, where
9559    /// [`header_attr_order`](Self::header_attr_order) raises every object to
9560    /// tracked. Storage laid out against the property list would then omit the
9561    /// creation indices the header says are there — and, since the Attribute
9562    /// Info message carries a maximum creation index only when tracked, would
9563    /// be two bytes shorter than the message the sizing pass measured.
9564    fn attribute_scopes(&self, datasets: &[usize]) -> Vec<(AttrScope, CreationOrder)> {
9565        let order_of = |requested| self.header_attr_order(requested);
9566        let mut scopes = vec![(AttrScope::Root, order_of(self.root_track_order.attrs))];
9567        for gi in 0..self.group_count() {
9568            if self.grp(gi).lock().deleted {
9569                continue;
9570            }
9571            let order = self.grp(gi).lock().track_order.attrs;
9572            scopes.push((AttrScope::Group(gi), order_of(order)));
9573        }
9574        for &i in datasets {
9575            let order = self.ds(i).lock().track_attr_order;
9576            scopes.push((AttrScope::Dataset(i), order_of(order)));
9577        }
9578        scopes
9579    }
9580
9581    /// Lay out and write dense attribute storage for every object that needs
9582    /// it, recording the resulting `Attribute Info` message per object.
9583    ///
9584    /// The sole owner of that transition. It runs after every object header
9585    /// has an address — an attribute may hold an object reference, and the
9586    /// heap holds the encoded attribute messages — and before any object
9587    /// header is written, because the header carries the Attribute Info
9588    /// message naming what this laid out. Every block is on disk before the
9589    /// map naming it is populated, so a header written from that map can only
9590    /// point at bytes that exist. The same placement rule, for the same two
9591    /// reasons, as [`prepare_dense_links`](Self::prepare_dense_links).
9592    ///
9593    /// Which objects spill is not decided here: `emit_attributes` asks
9594    /// [`attributes_need_dense`](Self::attributes_need_dense) itself, so the
9595    /// header measured before this ran and the header written after it agree
9596    /// without either consulting the other.
9597    fn prepare_dense_attributes(&self, datasets: &[usize]) -> IoResult<()> {
9598        for (scope, order) in self.attribute_scopes(datasets) {
9599            // Every scope here has its header rewritten, so the storage a
9600            // reopen found on it is superseded whether or not the new set is
9601            // dense again — a free driven by "the new set needs a heap" would
9602            // never reach an object that dropped back to compact. Freed
9603            // immediately before its replacement is laid out, so the rewrite
9604            // lands in the blocks it just gave back instead of growing the
9605            // file on every open/close cycle.
9606            self.release_superseded_dense_attrs(scope)?;
9607            // `close` after `start_swmr` finalizes a second time over the same
9608            // attribute sets — SWMR refuses every attribute mutation — so
9609            // rebuilding here would allocate a whole second heap and strand
9610            // the one the published headers already name.
9611            if self.dense_attributes.lock().contains_key(&scope) {
9612                continue;
9613            }
9614            let attributes = self.object_attributes(scope)?;
9615            if !self.attributes_need_dense(&attributes, self.attr_scope_format(scope)) {
9616                continue;
9617            }
9618            let dense = build_dense_attributes(&attributes, &self.ctx, order, &mut |len| {
9619                self.allocator.allocate(len, FreeSpaceClass::Metadata)
9620            })?;
9621            for block in &dense.blocks {
9622                self.handle.write_at(block.addr, &block.image)?;
9623            }
9624            self.dense_attributes.lock().insert(scope, dense.ainfo);
9625        }
9626        Ok(())
9627    }
9628
9629    /// The object header format `scope`'s owner is written at, which is what
9630    /// decides whether its attributes may spill at all.
9631    fn attr_scope_format(&self, scope: AttrScope) -> ObjectFormat {
9632        match scope {
9633            AttrScope::Root => self.header_format(self.root_track_order),
9634            AttrScope::Group(gi) => self.group_header_format(gi),
9635            AttrScope::Dataset(i) => self.dataset_header_format(i),
9636        }
9637    }
9638
9639    /// Whether a dataset's datatype message may be offered to a
9640    /// shared-message index at all.
9641    ///
9642    /// The datatype is the one message class carrying a `can_share` callback
9643    /// (`H5O__dtype_can_share`, H5Odtype.c:99), and `H5SM__can_share_common`
9644    /// asks it before any index is consulted (H5SM.c:895-899). It refuses an
9645    /// immutable type and a committed one (H5Odtype.c:1893-1901); the
9646    /// committed half is already answered by address at the call site.
9647    ///
9648    /// A dataset's type reaches that predicate still immutable only when
9649    /// `H5D__init_type` kept the caller's own `H5T_t` rather than copying it,
9650    /// which it does exactly when the type is immutable, is not relocatable,
9651    /// and the low bound this dataset's messages are written at is below
9652    /// `H5F_LIBVER_V18` (H5Dint.c:569-572) — the bound the dataset was
9653    /// *created* under, which for a dataset a reopen found is not this
9654    /// session's.
9655    /// Any of the three failing produces an `H5T_COPY_ALL` copy, which is
9656    /// `H5T_STATE_RDONLY` rather than immutable (H5T.c:4461-4462) and so is
9657    /// shareable — which is why `H5Tcopy(H5T_STD_I32LE)` shares where
9658    /// `H5T_STD_I32LE` itself does not (tests/fixtures/gen_sohm.c).
9659    ///
9660    /// An attribute has no such branch: `H5A__create` copies unconditionally
9661    /// (H5Aint.c:341), so its datatype is always eligible and
9662    /// [`share_attribute`](Self::share_attribute) offers it without asking.
9663    fn dataset_datatype_shareable(&self, datatype: &DatatypeMessage, libver: LibverBound) -> bool {
9664        !datatype.is_predefined() || datatype.is_relocatable() || libver >= LibverBound::V18
9665    }
9666
9667    /// Whether the first copy of a `msg_type` message may stay literal in the
9668    /// object header that writes it.
9669    ///
9670    /// `H5O_msg_can_share_in_ohdr` reads the class's `H5O_SHARE_IN_OHDR` flag
9671    /// (H5Omessage.c:1426); the five classes that carry it are datatype
9672    /// (H5Odtype.c:89), dataspace (H5Osdspace.c:61), both fill value messages
9673    /// (H5Ofill.c:106 and :130) and the filter pipeline (H5Opline.c:65). The
9674    /// attribute class does not, which is why an attribute reaches the heap on
9675    /// its first use.
9676    const fn shares_in_ohdr(msg_type: u8) -> bool {
9677        matches!(
9678            msg_type,
9679            MSG_DATASPACE
9680                | MSG_DATATYPE
9681                | MSG_FILL_VALUE
9682                | MSG_FILL_VALUE_OLD
9683                | MSG_FILTER_PIPELINE
9684        )
9685    }
9686
9687    /// What a header stores for a message a shared-message index may cover:
9688    /// the body itself, or a pointer into the shared-message heap.
9689    ///
9690    /// The single point at which a message is offered to an index. Every
9691    /// header builder routes its shareable messages through here, so the pass
9692    /// that counts references and the pass that substitutes pointers walk
9693    /// exactly the same set — the counting and the substituting cannot drift
9694    /// apart, because they are one call site in two phases.
9695    ///
9696    /// `owner` is `H5SM_try_share`'s `open_oh`: the header this message
9697    /// belongs to, or [`ShareOwner::Detached`] for a body that is part of
9698    /// another message rather than a message of a header.
9699    ///
9700    /// Outside a finalize, and in any file created without indexes, this is
9701    /// the identity.
9702    fn share_message(
9703        &self,
9704        owner: ShareOwner,
9705        msg_type: u8,
9706        flags: u8,
9707        body: Vec<u8>,
9708    ) -> (u8, Vec<u8>) {
9709        self.share_nesting_message(owner, msg_type, flags, body, Vec::new())
9710    }
9711
9712    /// [`share_message`](Self::share_message) for a body that itself holds
9713    /// shared-message pointers.
9714    ///
9715    /// `nested` names each heap ID inside `body`, which is zero until the
9716    /// table is laid out. Two bodies that differ only in what they point at
9717    /// are the same bytes here and different bytes on disk, so the count and
9718    /// the substitute are keyed on the pair.
9719    fn share_nesting_message(
9720        &self,
9721        owner: ShareOwner,
9722        msg_type: u8,
9723        flags: u8,
9724        body: Vec<u8>,
9725        nested: Vec<NestedShare>,
9726    ) -> (u8, Vec<u8>) {
9727        let Some(sohm) = self.sohm.as_deref() else {
9728            return (flags, body);
9729        };
9730        // A message already carrying a pointer — a committed datatype — is
9731        // shared by address and must not be shared again, and the message
9732        // classes libhdf5 marks `H5O_MSG_FLAG_DONTSHARE` never reach an index.
9733        if flags & (MSG_FLAG_SHARED | MSG_FLAG_DONTSHARE) != 0 {
9734            return (flags, body);
9735        }
9736        let Some(index) = sohm.index_for(msg_type, body.len()) else {
9737            return (flags, body);
9738        };
9739        // `share_in_ohdr && open_oh` (H5SM.c:1400): the first copy of one of
9740        // these classes stays where it was written, marked shareable, and only
9741        // a second use moves the body to the heap.
9742        let ohdr = match owner {
9743            ShareOwner::Header(addr) if Self::shares_in_ohdr(msg_type) => Some(addr),
9744            _ => None,
9745        };
9746        // What a pointer to this body looks like: a zeroed heap ID until the
9747        // table exists, which is the width the real one has.
9748        let pointer = |id| {
9749            (
9750                flags | MSG_FLAG_SHARED,
9751                SharedMessagePointer::encode_sohm(id),
9752            )
9753        };
9754        match &mut *sohm.phase.lock() {
9755            SohmPhase::Idle => (flags, body),
9756            SohmPhase::Predict(first) => {
9757                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9758                    return (flags | MSG_FLAG_SHAREABLE, body);
9759                }
9760                pointer([0u8; SOHM_HEAP_ID_LEN])
9761            }
9762            // The same substitution `Predict` makes, so that what the collect
9763            // pass builds around a shared message is the width the resolve
9764            // pass will build — which is what lets an attribute body assembled
9765            // in this pass be the body assembled in that one, bar the heap IDs
9766            // it is here recording a need for.
9767            SohmPhase::Collect(collector) => {
9768                let first = collector.record(index, msg_type, &body, &nested, ohdr);
9769                if !first && !nested.is_empty() {
9770                    // This body is already here, so the pointers it holds
9771                    // already exist in the heap and the offers that built
9772                    // this copy of it must not count a second time.
9773                    for share in &nested {
9774                        collector.release(share.target.0, &share.target.1);
9775                    }
9776                }
9777                if ohdr.is_some() && first {
9778                    return (flags | MSG_FLAG_SHAREABLE, body);
9779                }
9780                pointer([0u8; SOHM_HEAP_ID_LEN])
9781            }
9782            SohmPhase::Resolve { ids, first } => {
9783                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9784                    return (flags | MSG_FLAG_SHAREABLE, body);
9785                }
9786                let key = (msg_type, body);
9787                match ids.get(&key) {
9788                    Some(&id) => pointer(id),
9789                    // The collect pass never saw this body — a dataspace a
9790                    // SWMR extend changed after the table was laid out, say.
9791                    // Left literal, which leaves a heap object counted for one
9792                    // reference more than reaches it and nothing else.
9793                    None => (flags, key.1),
9794                }
9795            }
9796        }
9797    }
9798
9799    /// Answer every shareable message at a heap pointer's width for the rest
9800    /// of this finalize's allocation phase.
9801    ///
9802    /// Half of the bracket [`prepare_shared_messages`](Self::prepare_shared_messages)
9803    /// closes, and the reason the two can sit on opposite sides of the
9804    /// allocation: a header cannot be measured until it is known which of its
9805    /// messages are pointers, and a body cannot be counted until every address
9806    /// it names exists. Only the width is knowable in the first phase, and the
9807    /// width is all the measurement needs.
9808    ///
9809    /// A finalize that will not lay a table out — a `finalize_for_swmr`, a
9810    /// second finalize over a table already published — leaves the phase where
9811    /// it found it, so what that pass measures is what it writes.
9812    fn begin_shared_message_layout(&self) {
9813        let Some(sohm) = self.sohm.as_deref() else {
9814            return;
9815        };
9816        let mut phase = sohm.phase.lock();
9817        if matches!(*phase, SohmPhase::Idle) && sohm.table_addr.lock().is_none() {
9818            *phase = SohmPhase::Predict(FirstCopies::default());
9819        }
9820    }
9821
9822    /// Lay out the file's shared-message table: count the bodies every header
9823    /// this finalize writes would share, put them in their index's heap, and
9824    /// arm the substitution the header builders then apply.
9825    ///
9826    /// The sole owner of the transition to `Resolve`. It runs last in the
9827    /// content phase, after
9828    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes),
9829    /// [`prepare_link_storage`](Self::prepare_link_storage) and
9830    /// [`write_reference_values`](Self::write_reference_values), because a
9831    /// body is only counted once it is the body the file will hold: an
9832    /// attribute that spilled into dense storage is not in a header to be
9833    /// shared at all, and one holding an object reference says an object
9834    /// header address that exists only after the allocation phase. Counting
9835    /// either of them earlier would count a body no header ends up carrying,
9836    /// and leave the header that carries the real one literal — which
9837    /// [`check_header_size`] would then refuse, the block having been
9838    /// reserved at a pointer's width.
9839    ///
9840    /// Once per file: a second finalize (a SWMR session's close) keeps the
9841    /// table the first one published rather than allocating a second one and
9842    /// stranding the first.
9843    fn prepare_shared_messages(&self, datasets: &[usize]) -> IoResult<()> {
9844        let Some(sohm) = self.sohm.as_deref() else {
9845            return Ok(());
9846        };
9847        if sohm.table_addr.lock().is_some() {
9848            return Ok(());
9849        }
9850
9851        // Collect: build every header this finalize will write and throw it
9852        // away, keeping only what its shareable messages were.
9853        *sohm.phase.lock() = SohmPhase::Collect(SohmCollector::new(sohm.indexes.len()));
9854        for &i in datasets {
9855            self.build_dataset_header(i)?;
9856        }
9857        for gi in 0..self.group_count() {
9858            if self.grp(gi).lock().deleted {
9859                continue;
9860            }
9861            self.build_group_header(gi)?;
9862        }
9863        self.build_root_group_header()?;
9864        let SohmPhase::Collect(collector) =
9865            std::mem::replace(&mut *sohm.phase.lock(), SohmPhase::Idle)
9866        else {
9867            return Err(crate::io::IoError::InvalidState(
9868                "the shared-message collect pass did not finish in the collect phase".into(),
9869            ));
9870        };
9871
9872        let indexes: Vec<SohmIndexContent> = sohm
9873            .indexes
9874            .iter()
9875            .zip(collector.messages)
9876            .map(|(&spec, messages)| SohmIndexContent { spec, messages })
9877            .collect();
9878        // The table a reopen found is superseded whole by the one below, and
9879        // every header that pointed into it is in this finalize's rewrite set
9880        // — so its blocks go back immediately before the replacement is laid
9881        // out, and the new table lands in them instead of growing the file on
9882        // every open/close cycle. Taken, not read: a second finalize must not
9883        // free the same blocks twice.
9884        for (addr, len) in std::mem::take(&mut *sohm.superseded.lock()) {
9885            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9886        }
9887        let built = build_shared_messages(&indexes, &self.ctx, &mut |len| {
9888            self.allocator.allocate(len, FreeSpaceClass::Metadata)
9889        })?;
9890        for block in &built.blocks {
9891            self.handle.write_at(block.addr, &block.image)?;
9892        }
9893
9894        // Only now, with every block on disk: from here the header builders
9895        // substitute pointers, and `write_superblock_extension` names the
9896        // table this laid out.
9897        *sohm.phase.lock() = SohmPhase::Resolve {
9898            ids: built.heap_ids,
9899            first: FirstCopies::default(),
9900        };
9901        *sohm.table_addr.lock() = Some(built.table_addr);
9902        Ok(())
9903    }
9904
9905    /// Write the file's free-space managers over the space this close leaves
9906    /// free, and return the file-space info message body naming them.
9907    ///
9908    /// Called from [`write_superblock_extension`](Self::write_superblock_extension)
9909    /// once every other block of the file has an address, which is what makes
9910    /// the allocator's free list the file's *final* free space: a block
9911    /// allocated after this point would land in space a manager still claims.
9912    ///
9913    /// INVARIANT: from the moment this returns, every byte the allocator holds
9914    /// free is a byte some sections block records, and the two blocks each
9915    /// manager itself occupies are held by neither. Nothing may allocate
9916    /// between here and the superblock write; `write_object_headers` writes
9917    /// over blocks reserved in an earlier phase and is the only thing that
9918    /// runs in between.
9919    ///
9920    /// Returns `None` for a file with no message of its own to write — a
9921    /// reopen whose carried message this session must not touch, and a file
9922    /// created at the library defaults — which leaves both byte-identical to
9923    /// what the same close wrote before free space was recorded at all. A file
9924    /// that carries the message but keeps no managers (either non-manager
9925    /// strategy, or `persist: false`) gets the message back with every address
9926    /// undefined, which is what `H5F__super_init` writes for it.
9927    fn write_free_space_managers(&self) -> IoResult<Option<Vec<u8>>> {
9928        let Some(fs) = self.free_space.as_deref() else {
9929            return Ok(None);
9930        };
9931        if !fs.records_free_space() {
9932            return Ok(Some(fs.info.encode(&self.ctx)?));
9933        }
9934        // The managers a reopen found are superseded whole by the ones below,
9935        // so their blocks go back before anything is laid out: the space the
9936        // old manager occupied is free space the new one records, and the new
9937        // one may be laid out in it.
9938        for &(addr, len) in &fs.superseded {
9939            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9940        }
9941
9942        let hdr_size = FreeSpaceHeader::encoded_size(&self.ctx) as u64;
9943        let settled = self.settle_free_space_managers(hdr_size, fs.info.threshold)?;
9944
9945        let mut info = fs.info.clone();
9946        info.fs_addr = vec![UNDEF_ADDR; info.fs_addr.len()];
9947        for placed in &settled {
9948            let mut header = manager_header(&placed.sections);
9949            // The settle loop sized the block; that the encode agrees is the
9950            // invariant that makes `sect_size` a length a reader can trust.
9951            let needed = free_space::sinfo_encoded_size(&header, &placed.sections, &self.ctx);
9952            if needed > placed.sect_size {
9953                return Err(crate::io::IoError::InvalidState(format!(
9954                    "the free-space sections need {needed} bytes, not the {} laid out",
9955                    placed.sect_size
9956                )));
9957            }
9958            header.sect_addr = placed.sect_addr;
9959            header.sect_size = placed.sect_size;
9960            header.alloc_sect_size = placed.sect_size;
9961            self.handle.write_at(
9962                placed.sect_addr,
9963                &free_space::encode_sections(
9964                    &header,
9965                    placed.hdr_addr,
9966                    &placed.sections,
9967                    placed.sect_size as usize,
9968                    &self.ctx,
9969                ),
9970            )?;
9971            self.handle
9972                .write_at(placed.hdr_addr, &header.encode(&self.ctx))?;
9973            // `H5MF__close_delete_fstype` leaves a manager with no sections
9974            // without an address, so only the ones written name themselves.
9975            info.fs_addr[placed.manager.message_slot()] = placed.hdr_addr;
9976        }
9977        // The end of the file *after* the settle above, not before it, which
9978        // the field's name denies: it is 1.10 vintage, where two EOAs were
9979        // kept — one taken before the self-referential managers were placed
9980        // and one after (H5MF.c:3305 and 3382 in 1.10.11) — and the message
9981        // carried the first (1.10.11 H5MF.c:1833, 1999). 1.14 keeps one,
9982        // `f->shared->eoa_fsm_fsalloc`, read once the allocation loop has run
9983        // (H5MF.c:3234-3240) and encoded into this field by both close paths
9984        // (H5MF.c:1759, 1923); H5Fsuper.c:826 names it "the final eoa". A
9985        // 1.10 reader wants that value and not the older one: equal EOAs are
9986        // the case `H5MF_tidy_self_referential_fsm_hack` returns on
9987        // (1.10.11 H5MF.c:3620-3622), which is what leaves the managers this
9988        // close wrote in place.
9989        info.eoa_pre_fsm_fsalloc = self.allocator.eof();
9990        Ok(Some(info.encode(&self.ctx)?))
9991    }
9992
9993    /// The file's free space as each manager will record it: address-ordered
9994    /// per manager, tagged with the section class that manager writes, and
9995    /// with everything below `threshold` left out.
9996    ///
9997    /// The allocator is the single owner of merging — `H5FS__sect_merge`'s
9998    /// rules, per manager and, on a paged file, per page — so nothing merges
9999    /// here; overlap is checked because two overlapping sections would be a
10000    /// manager claiming space another structure holds.
10001    fn free_sections(&self, threshold: u64) -> IoResult<Vec<(FreeSpaceManager, Vec<FreeSection>)>> {
10002        let policy = self.allocator.policy();
10003        let extents = self.allocator.free_extents();
10004        let mut sets = Vec::new();
10005        for manager in FreeSpaceManager::ALL {
10006            let mut sections: Vec<FreeSection> = extents
10007                .iter()
10008                .filter(|b| b.manager == manager)
10009                // `H5FS_sect_add` refuses a section below the file's
10010                // threshold, so a block smaller than it is space the file
10011                // leaks rather than records — the same trade the threshold is
10012                // there to make.
10013                .filter(|b| b.len >= threshold)
10014                .map(|b| FreeSection {
10015                    addr: b.addr,
10016                    len: b.len,
10017                    class: policy.section_class(manager),
10018                })
10019                .collect();
10020            sections.sort_unstable_by_key(|s| s.addr);
10021            if let Some(bad) = sections
10022                .windows(2)
10023                .find(|w| w[0].addr + w[0].len > w[1].addr)
10024            {
10025                return Err(crate::io::IoError::InvalidState(format!(
10026                    "this session freed overlapping blocks: {:#x}+{} overlaps {:#x}",
10027                    bad[0].addr, bad[0].len, bad[1].addr
10028                )));
10029            }
10030            sets.push((manager, sections));
10031        }
10032        Ok(sets)
10033    }
10034
10035    /// Give every manager that records anything its own header and sections
10036    /// blocks, and return what each will write.
10037    ///
10038    /// Self-referential, which is the whole difficulty: a manager's two blocks
10039    /// come out of the free space the managers record, and taking them changes
10040    /// that space, which changes how many bytes the sections block needs.
10041    /// Upstream reruns the allocation pass until no manager allocates anything
10042    /// further — the `do { ... } while (continue_alloc_fsm)` loop in
10043    /// `H5MF_settle_meta_data_fsm` (H5MF.c:3213-3247) around
10044    /// `H5FS_vfd_alloc_hdr_and_section_info_if_needed`, which allocates
10045    /// through `H5MF_alloc` like everything else. So does this: the blocks
10046    /// come out of the same [`FileAllocator`], under the same strategy, so a
10047    /// paged file's manager blocks land in pages and their page remainders are
10048    /// recorded like any others.
10049    ///
10050    /// Two rules make it terminate. A manager, once placed, stays placed: were
10051    /// its blocks released because its sections had been consumed, freeing
10052    /// them would put those sections back and the next round would place it
10053    /// again. And a sections block only ever grows: upstream frees a block
10054    /// that turned out too small and reallocates it next round
10055    /// (H5FSsection.c:2418-2423), and a size that only rises reaches its
10056    /// bound.
10057    fn settle_free_space_managers(
10058        &self,
10059        hdr_size: u64,
10060        threshold: u64,
10061    ) -> IoResult<Vec<PlacedManager>> {
10062        /// Rounds before the layout is called divergent. A round either places
10063        /// a manager or grows one sections block, and there are three
10064        /// managers, so a file that needs more than this is not converging.
10065        const ROUNDS: usize = 16;
10066
10067        // Raw data first and metadata last, in `H5MF_settle_raw_data_fsm`'s
10068        // order (H5C.c:689-696): every manager's own blocks are metadata
10069        // allocations, so the metadata manager funds all of them and is the
10070        // one whose section set the others change.
10071        const ORDER: [FreeSpaceManager; 3] = [
10072            FreeSpaceManager::RawData,
10073            FreeSpaceManager::Large,
10074            FreeSpaceManager::Metadata,
10075        ];
10076
10077        let size_of = |sections: &[FreeSection]| {
10078            let ordered = free_space::serialization_order(sections);
10079            free_space::sinfo_encoded_size(&manager_header(&ordered), &ordered, &self.ctx)
10080        };
10081        let mut placed: Vec<PlacedManager> = Vec::new();
10082        for _ in 0..ROUNDS {
10083            let sets = self.free_sections(threshold)?;
10084            let sections_of = |manager: FreeSpaceManager| {
10085                sets.iter()
10086                    .find(|(m, _)| *m == manager)
10087                    .map(|(_, s)| s.as_slice())
10088                    .unwrap_or_default()
10089            };
10090
10091            let mut changed = false;
10092            for manager in ORDER {
10093                let sections = sections_of(manager);
10094                if sections.is_empty() || placed.iter().any(|p| p.manager == manager) {
10095                    continue;
10096                }
10097                let sect_size = size_of(sections);
10098                let hdr_addr = self.allocator.allocate(hdr_size, FreeSpaceClass::Metadata);
10099                let sect_addr = self.allocator.allocate(sect_size, FreeSpaceClass::Metadata);
10100                placed.push(PlacedManager {
10101                    manager,
10102                    hdr_addr,
10103                    sect_addr,
10104                    sect_size,
10105                    sections: Vec::new(),
10106                });
10107                changed = true;
10108            }
10109            if !changed {
10110                for p in &mut placed {
10111                    let needed = size_of(sections_of(p.manager));
10112                    if needed > p.sect_size {
10113                        self.allocator
10114                            .free(p.sect_addr, p.sect_size, FreeSpaceClass::Metadata);
10115                        p.sect_size = needed;
10116                        p.sect_addr = self.allocator.allocate(needed, FreeSpaceClass::Metadata);
10117                        changed = true;
10118                    }
10119                }
10120            }
10121            if !changed {
10122                for p in &mut placed {
10123                    p.sections = free_space::serialization_order(sections_of(p.manager));
10124                }
10125                return Ok(placed);
10126            }
10127        }
10128        Err(crate::io::IoError::InvalidState(format!(
10129            "the free-space managers did not settle in {ROUNDS} rounds"
10130        )))
10131    }
10132
10133    /// Write the file's superblock extension, and the sole owner of that
10134    /// object header.
10135    ///
10136    /// Runs after [`prepare_shared_messages`](Self::prepare_shared_messages),
10137    /// whose table it names, and before the superblock that names it. What it
10138    /// writes is [`CarriedExtension`] — every message the reopened file's
10139    /// extension held — plus the shared-message table message, which is the
10140    /// one message whose content this session owns: the table moved, so the
10141    /// message read is stale and the message written names the new address.
10142    ///
10143    /// A file with neither carried messages nor shared messages gets no
10144    /// extension, which is what libhdf5 writes for it: `H5F__super_ext_create`
10145    /// is called only when there is a message to put in one.
10146    ///
10147    /// Version 1, holding its messages in one chunk: the extension is created
10148    /// before anything raises the file's object header version
10149    /// (`H5F__super_ext_create` passes `H5O_HDR_STORE_TIMES` off and takes the
10150    /// version-1 path), so an extension of any generation of file looks the
10151    /// same.
10152    fn write_superblock_extension(&self) -> IoResult<()> {
10153        if self.extension.addr.lock().is_some() {
10154            return Ok(());
10155        }
10156        let table = self.sohm.as_deref().and_then(|sohm| {
10157            sohm.table_addr
10158                .lock()
10159                .map(|addr| (sohm.indexes.len(), addr))
10160        });
10161        // A file with file-space properties of its own needs an extension
10162        // too: the message that declares them is the only place they are
10163        // recorded, and a file created with them carries nothing else.
10164        if self.extension.carried.is_empty() && table.is_none() && self.free_space.is_none() {
10165            return Ok(());
10166        }
10167
10168        let mut messages: Vec<crate::io::object_header_io::ExtensionMessage> =
10169            self.extension.carried.clone();
10170        if let Some(fs) = self.free_space.as_deref() {
10171            // The declared message, at exactly the length the one written
10172            // below will have — every field of it is fixed-width, and only
10173            // `persist` and the message version change the count of
10174            // addresses, neither of which the close alters. The image is sized
10175            // and its block allocated before the managers can be laid out, so
10176            // the message has to reach its final *length* here even though its
10177            // content is settled later.
10178            let declared = fs.info.encode(&self.ctx)?;
10179            match messages
10180                .iter_mut()
10181                .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10182            {
10183                Some(msg) => msg.body = declared,
10184                None => messages.push(crate::io::object_header_io::ExtensionMessage {
10185                    msg_type: MSG_FILE_SPACE_INFO,
10186                    flags: MSG_FLAG_DONTSHARE | MSG_FLAG_MARK_IF_UNKNOWN,
10187                    body: declared,
10188                }),
10189            }
10190        }
10191        if let Some((nindexes, table_addr)) = table {
10192            let nindexes = u8::try_from(nindexes).map_err(|_| {
10193                crate::io::IoError::InvalidState(format!("{nindexes} shared-message indexes"))
10194            })?;
10195            messages.push(crate::io::object_header_io::ExtensionMessage {
10196                msg_type: MSG_SHARED_MESSAGE_TABLE,
10197                flags: MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
10198                body: SharedMessageTableMessage {
10199                    version: 0,
10200                    table_address: table_addr,
10201                    nindexes,
10202                }
10203                .encode(&self.ctx),
10204            });
10205        }
10206        let encode = |messages: &[crate::io::object_header_io::ExtensionMessage]| {
10207            let mut extension = ObjectHeader::new();
10208            for msg in messages {
10209                extension.add_message(msg.msg_type, msg.flags, msg.body.clone());
10210            }
10211            extension.encode_v1(1)
10212        };
10213        let image = encode(&messages)?;
10214        // Freed before the replacement is placed, so a reopen reuses the block
10215        // instead of stranding one per open/close cycle — the rule every other
10216        // superseded structure follows.
10217        for &(addr, len) in &self.extension.superseded {
10218            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
10219        }
10220        let addr = self
10221            .allocator
10222            .allocate(image.len() as u64, FreeSpaceClass::Metadata);
10223
10224        // Every block of this file now has an address, so the allocator holds
10225        // exactly the file's free space: settle the free-space managers over
10226        // it and say in this extension where they went.
10227        let image = match self.write_free_space_managers()? {
10228            None => image,
10229            Some(body) => {
10230                let msg = messages
10231                    .iter_mut()
10232                    .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10233                    .ok_or_else(|| {
10234                        crate::io::IoError::InvalidState(
10235                            "a persisting file lost its file-space info message".into(),
10236                        )
10237                    })?;
10238                // Same length as the declared body put in above, so the
10239                // image measured before the block was allocated still fits.
10240                if body.len() != msg.body.len() {
10241                    return Err(crate::io::IoError::InvalidState(format!(
10242                        "the file-space info message was laid out at {} bytes and \
10243                         written back at {}",
10244                        msg.body.len(),
10245                        body.len()
10246                    )));
10247                }
10248                msg.body = body;
10249                encode(&messages)?
10250            }
10251        };
10252        self.handle.write_at(addr, &image)?;
10253        *self.extension.addr.lock() = Some(addr);
10254        Ok(())
10255    }
10256
10257    /// Define a new contiguous dataset. Returns the dataset index (used with
10258    /// `write_dataset_raw`).
10259    ///
10260    /// The raw-data region is allocated immediately so that
10261    /// `write_dataset_raw` can be called at any time before `close()`.
10262    pub fn create_dataset(
10263        &self,
10264        name: &str,
10265        datatype: DatatypeMessage,
10266        dims: &[u64],
10267    ) -> IoResult<usize> {
10268        let create = self.begin_create(name)?;
10269        let name = create.name.as_str();
10270        let total_elements: u64 = if dims.is_empty() {
10271            1
10272        } else {
10273            dims.iter().product()
10274        };
10275        let element_size = datatype.element_size() as u64;
10276        let data_size = total_elements * element_size;
10277
10278        // Allocate space for the raw data.
10279        let data_addr = if data_size > 0 {
10280            self.allocator.allocate(data_size, FreeSpaceClass::RawData)
10281        } else {
10282            UNDEF_ADDR
10283        };
10284
10285        let dataspace = if dims.is_empty() {
10286            DataspaceMessage::scalar()
10287        } else {
10288            DataspaceMessage::simple(dims)
10289        };
10290
10291        let idx = self.push_dataset(
10292            &create,
10293            DatasetInfo {
10294                name: name.to_string(),
10295                datatype,
10296                committed_type: None,
10297                external: None,
10298                virtual_storage: None,
10299                dataspace,
10300                read_format: None,
10301                obj_header_addr: 0, // set during finalize
10302                data_addr,
10303                data_size,
10304                compact: None,
10305                chunked: None,
10306                fixed_array: None,
10307                implicit: None,
10308                single_chunk: None,
10309                btree_v1: None,
10310                btree_v2: None,
10311                append: None,
10312                attributes: Vec::new(),
10313                obj_header_written_addr: None,
10314                obj_header_blocks: Vec::new(),
10315                filter_pipeline: None,
10316                deleted: false,
10317                extent_dirty: false,
10318                header_dirty: false,
10319                nlink_written: 1,
10320                creation_seq: self.take_creation_seq(),
10321                track_attr_order: self.track_order.attrs,
10322                fill_value: None,
10323                fill_time: FILL_TIME_IFSET,
10324                layout_version: 4,
10325                times: self.created_object_times(),
10326            },
10327        );
10328
10329        Ok(idx)
10330    }
10331
10332    /// Define a new dataset whose raw data lives in files outside this one —
10333    /// `H5Pset_external`, h5py's `external=[(name, offset, size)]`.
10334    ///
10335    /// Each entry names a file, the byte offset in it where that entry's
10336    /// region starts, and how many bytes of the dataset the region holds; the
10337    /// entries concatenate, in order, into the dataset's logical byte range,
10338    /// and together must cover it. Nothing is allocated in this file: the data
10339    /// layout message says contiguous storage at an undefined address, and it
10340    /// is the External File List beside it that says where the bytes are
10341    /// (`H5D__layout_oh_create`).
10342    ///
10343    /// A named file is created on first write and never truncated, so several
10344    /// slots — or several datasets — may own disjoint ranges of one file, the
10345    /// way `H5D__efl_write` opens them.
10346    ///
10347    /// The last slot may take the unlimited size `H5O_EFL_UNLIMITED`, which
10348    /// makes it absorb however many bytes the dataset comes to hold; a
10349    /// dataset whose dataspace is unlimited must have one, since nothing
10350    /// finite could cover it (`H5D__efl_construct`: "unlimited dataspace but
10351    /// finite storage"). Only the first dimension may be extendible, which is
10352    /// the same function's other rule.
10353    pub fn create_external_dataset(
10354        &self,
10355        name: &str,
10356        datatype: DatatypeMessage,
10357        dims: &[u64],
10358        max_dims: Option<&[u64]>,
10359        files: &[(&str, u64, u64)],
10360    ) -> IoResult<usize> {
10361        if files.is_empty() {
10362            return Err(crate::io::IoError::InvalidState(format!(
10363                "external dataset '{name}' names no files; external storage is defined by \
10364                 the files it lives in, so at least one is required"
10365            )));
10366        }
10367        let create = self.begin_create(name)?;
10368        let name = create.name.as_str();
10369        let total_elements: u64 = if dims.is_empty() {
10370            1
10371        } else {
10372            dims.iter().product()
10373        };
10374        let data_size = total_elements * datatype.element_size() as u64;
10375
10376        let mut heap = LocalHeapImage::with_empty_string();
10377        let mut entries = Vec::with_capacity(files.len());
10378        for (i, &(file_name, offset, size)) in files.iter().enumerate() {
10379            if file_name.is_empty() {
10380                return Err(crate::io::IoError::InvalidState(format!(
10381                    "external dataset '{name}' has a slot with an empty file name"
10382                )));
10383            }
10384            // `H5Pset_external` refuses to add a slot behind an unlimited one
10385            // ("previous file size is unlimited"): the unlimited slot already
10386            // owns every byte from its own start onwards, so nothing after it
10387            // could ever be reached.
10388            if size == UNLIMITED && i + 1 != files.len() {
10389                return Err(crate::io::IoError::InvalidState(format!(
10390                    "external dataset '{name}' gives slot {i} ('{file_name}') the unlimited \
10391                     size H5O_EFL_UNLIMITED with {} slot(s) behind it; an unlimited slot \
10392                     absorbs the rest of the dataset, so it can only be the last",
10393                    files.len() - i - 1
10394                )));
10395            }
10396            if offset.checked_add(size).is_none() {
10397                return Err(crate::io::IoError::InvalidState(format!(
10398                    "external dataset '{name}' slot '{file_name}' spans offset {offset} \
10399                     plus {size} bytes, past the end of the 64-bit address space"
10400                )));
10401            }
10402            entries.push(ExternalFile {
10403                name: file_name.to_string(),
10404                name_offset: heap.insert_str(file_name),
10405                offset,
10406                size,
10407            });
10408        }
10409        let external = ExternalStorage {
10410            // Filled in below, once the heap the names went into has an
10411            // address; the names' offsets within it are already final.
10412            heap_addr: UNDEF_ADDR,
10413            files: entries,
10414            // Settled by the open this create hands a handle out for, which
10415            // is `H5D__create` reading the dapl at H5Dint.c:1318.
10416            prefix: EfilePrefix::default(),
10417        };
10418        // `H5D__efl_construct`, over the dataset's *maximum* extent: the
10419        // slots must reserve at least every byte the dataset could come to
10420        // hold, and an unlimited extent can only be covered by an unlimited
10421        // last slot ("unlimited dataspace but finite storage").
10422        let max_dims = max_dims.unwrap_or(dims);
10423        if max_dims.len() != dims.len() {
10424            return Err(crate::io::IoError::InvalidState(format!(
10425                "external dataset '{name}' has {} dimensions but {} maximum ones",
10426                dims.len(),
10427                max_dims.len()
10428            )));
10429        }
10430        for (d, (&max, &cur)) in max_dims.iter().zip(dims).enumerate().skip(1) {
10431            if max > cur {
10432                return Err(crate::io::IoError::InvalidState(format!(
10433                    "external dataset '{name}' makes dimension {d} extendible ({cur} of \
10434                     {max}); only the first dimension can be extendible for external storage"
10435                )));
10436            }
10437        }
10438        let reserved = external.total_size();
10439        if max_dims.contains(&u64::MAX) {
10440            if reserved != UNLIMITED {
10441                return Err(crate::io::IoError::InvalidState(format!(
10442                    "external dataset '{name}' has an unlimited dataspace but its files \
10443                     reserve only {reserved} bytes; the last slot must take the unlimited \
10444                     size H5O_EFL_UNLIMITED"
10445                )));
10446            }
10447        } else {
10448            let max_bytes = max_dims
10449                .iter()
10450                .try_fold(datatype.element_size() as u64, |acc, &d| acc.checked_mul(d))
10451                .ok_or_else(|| {
10452                    crate::io::IoError::InvalidState(format!(
10453                        "external dataset '{name}' maximum extent times its element size \
10454                         overflows 64 bits"
10455                    ))
10456                })?;
10457            if reserved < max_bytes {
10458                return Err(crate::io::IoError::InvalidState(format!(
10459                    "external dataset '{name}' needs {max_bytes} bytes but its files reserve \
10460                     only {reserved}"
10461                )));
10462            }
10463        }
10464
10465        // The names' heap, written now: it is ordinary metadata of this file,
10466        // and the message the header carries is only an address into it.
10467        let sa = self.ctx.sizeof_addr as usize;
10468        let ss = self.ctx.sizeof_size as usize;
10469        let heap_bytes = heap.as_bytes().to_vec();
10470        let heap_addr = self.allocator.allocate(
10471            local_heap_header_size(sa, ss) as u64,
10472            FreeSpaceClass::Metadata,
10473        );
10474        let heap_data_addr = self
10475            .allocator
10476            .allocate(heap_bytes.len() as u64, FreeSpaceClass::Metadata);
10477        let heap_hdr = LocalHeapHeader {
10478            data_size: heap_bytes.len() as u64,
10479            // Sized to hold exactly these names, so no block of it is free.
10480            free_list_offset: LOCAL_HEAP_FREE_NULL,
10481            data_addr: heap_data_addr,
10482        };
10483        self.handle.write_at(heap_addr, &heap_hdr.encode(sa, ss))?;
10484        self.handle.write_at(heap_data_addr, &heap_bytes)?;
10485        let external = ExternalStorage {
10486            heap_addr,
10487            ..external
10488        };
10489
10490        let dataspace = if dims.is_empty() {
10491            DataspaceMessage::scalar()
10492        } else {
10493            let mut ds = DataspaceMessage::simple(dims);
10494            if max_dims != dims {
10495                ds.max_dims = Some(max_dims.to_vec());
10496            }
10497            ds
10498        };
10499
10500        let idx = self.push_dataset(
10501            &create,
10502            DatasetInfo {
10503                name: name.to_string(),
10504                datatype,
10505                committed_type: None,
10506                external: Some(external),
10507                virtual_storage: None,
10508                dataspace,
10509                read_format: None,
10510                obj_header_addr: 0, // set during finalize
10511                // No block of this file's own: the layout message declares
10512                // contiguous storage at an undefined address, which is what
10513                // sends a reader to the external file list instead.
10514                data_addr: UNDEF_ADDR,
10515                data_size,
10516                compact: None,
10517                chunked: None,
10518                fixed_array: None,
10519                btree_v2: None,
10520                implicit: None,
10521                single_chunk: None,
10522                btree_v1: None,
10523                append: None,
10524                attributes: Vec::new(),
10525                obj_header_written_addr: None,
10526                obj_header_blocks: Vec::new(),
10527                filter_pipeline: None,
10528                deleted: false,
10529                extent_dirty: false,
10530                header_dirty: false,
10531                nlink_written: 1,
10532                creation_seq: self.take_creation_seq(),
10533                track_attr_order: self.track_order.attrs,
10534                fill_value: None,
10535                fill_time: FILL_TIME_IFSET,
10536                layout_version: 4,
10537                times: self.created_object_times(),
10538            },
10539        );
10540
10541        Ok(idx)
10542    }
10543
10544    /// Define a new virtual dataset — `H5Pset_virtual`, h5py's
10545    /// `create_virtual_dataset(name, VirtualLayout)`.
10546    ///
10547    /// Each mapping says which elements of this dataset (`virtual_selection`)
10548    /// are read from which elements (`source_selection`) of a dataset in
10549    /// another file; the sources are never opened here, and a mapping naming
10550    /// one that does not exist yet is perfectly legal — libhdf5 resolves each
10551    /// at read time, filling from the fill value where nothing maps.
10552    ///
10553    /// The mappings do not live in the object header: they are serialized
10554    /// into one global heap object and the layout message carries only its
10555    /// address and index (`H5D__virtual_store_layout`), which is why this
10556    /// allocates a heap object and nothing else.
10557    ///
10558    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: the
10559    /// mapping grows with its source, and the virtual dataset's extent in
10560    /// that dimension is whatever the sources reachable at read time supply
10561    /// (`H5D__virtual_set_extent_unlim`). A `printf`-style source name is
10562    /// written as one too: `%b` substitutes the block index, so one mapping
10563    /// stands for the family of source datasets that fill the successive
10564    /// blocks of an unlimited virtual selection.
10565    pub fn create_virtual_dataset(
10566        &self,
10567        name: &str,
10568        datatype: DatatypeMessage,
10569        dims: &[u64],
10570        max_dims: Option<&[u64]>,
10571        mappings: &[VirtualMapping],
10572    ) -> IoResult<usize> {
10573        if mappings.is_empty() {
10574            return Err(crate::io::IoError::InvalidState(format!(
10575                "virtual dataset '{name}' names no mappings; a virtual dataset is defined \
10576                 by the source datasets it maps, so at least one is required"
10577            )));
10578        }
10579        for m in mappings {
10580            check_virtual_mapping(name, m)?;
10581        }
10582
10583        let create = self.begin_create(name)?;
10584        let name = create.name.as_str();
10585
10586        // The mapping list is ordinary file metadata, written now: the header
10587        // built at finalize carries only the heap address and object index it
10588        // lands at.
10589        let block = VirtualMappingList {
10590            mappings: mappings.to_vec(),
10591        }
10592        .encode(&self.ctx)?;
10593        let (heap_addr, heap_index) = self.insert_vlen_objects(&[&block])?[0];
10594
10595        let dataspace = if dims.is_empty() {
10596            DataspaceMessage::scalar()
10597        } else {
10598            let mut ds = DataspaceMessage::simple(dims);
10599            // A caller that named no maximum gets the current dimensions, the
10600            // maximum `simple` already filled in: `H5Screate_simple(rank,
10601            // dims, NULL)` reaches the encoder with `extent.max` set
10602            // (H5S.c:1293-1299), so leaving it absent here would write a
10603            // message no upstream API call can produce.
10604            if let Some(max) = max_dims {
10605                ds.max_dims = Some(max.to_vec());
10606            }
10607            ds
10608        };
10609
10610        let idx = self.push_dataset(
10611            &create,
10612            DatasetInfo {
10613                name: name.to_string(),
10614                datatype,
10615                committed_type: None,
10616                external: None,
10617                virtual_storage: Some(VirtualStorage {
10618                    heap_addr,
10619                    heap_index: heap_index as u32,
10620                    mappings: mappings.to_vec(),
10621                }),
10622                dataspace,
10623                read_format: None,
10624                obj_header_addr: 0, // set during finalize
10625                // Not a block of this file at all: every element is read out
10626                // of a source dataset, so there is nothing here to allocate
10627                // and nothing to free when the dataset is deleted.
10628                data_addr: UNDEF_ADDR,
10629                data_size: 0,
10630                compact: None,
10631                chunked: None,
10632                fixed_array: None,
10633                btree_v2: None,
10634                implicit: None,
10635                single_chunk: None,
10636                btree_v1: None,
10637                append: None,
10638                attributes: Vec::new(),
10639                obj_header_written_addr: None,
10640                obj_header_blocks: Vec::new(),
10641                filter_pipeline: None,
10642                deleted: false,
10643                extent_dirty: false,
10644                header_dirty: false,
10645                nlink_written: 1,
10646                creation_seq: self.take_creation_seq(),
10647                track_attr_order: self.track_order.attrs,
10648                fill_value: None,
10649                fill_time: FILL_TIME_IFSET,
10650                layout_version: 4,
10651                times: self.created_object_times(),
10652            },
10653        );
10654
10655        Ok(idx)
10656    }
10657
10658    /// Define a new compact dataset — `H5Pset_layout(dcpl, H5D_COMPACT)`.
10659    ///
10660    /// The raw data lives inside the data layout message in the dataset's own
10661    /// object header, so it costs no block of its own and no extra seek to
10662    /// read; the price is the ceiling, and that the whole image is rewritten
10663    /// whenever the header is. The buffer is created at its final length and
10664    /// zero-filled, which is what `H5D__compact_fill` does at create time, so
10665    /// a dataset never written still reads back as its fill value.
10666    ///
10667    /// Errors when the image exceeds [`MAX_COMPACT_DATA`].
10668    pub fn create_compact_dataset(
10669        &self,
10670        name: &str,
10671        datatype: DatatypeMessage,
10672        dims: &[u64],
10673    ) -> IoResult<usize> {
10674        let total_elements: u64 = if dims.is_empty() {
10675            1
10676        } else {
10677            dims.iter().product()
10678        };
10679        let data_size = total_elements * datatype.element_size() as u64;
10680        if data_size > MAX_COMPACT_DATA as u64 {
10681            return Err(crate::io::IoError::InvalidState(format!(
10682                "compact dataset '{name}' needs {data_size} bytes, above the \
10683                 {MAX_COMPACT_DATA}-byte ceiling a data layout message can hold; \
10684                 use contiguous or chunked storage"
10685            )));
10686        }
10687
10688        let create = self.begin_create(name)?;
10689        let name = create.name.as_str();
10690        let dataspace = if dims.is_empty() {
10691            DataspaceMessage::scalar()
10692        } else {
10693            DataspaceMessage::simple(dims)
10694        };
10695
10696        let idx = self.push_dataset(
10697            &create,
10698            DatasetInfo {
10699                name: name.to_string(),
10700                datatype,
10701                committed_type: None,
10702                external: None,
10703                virtual_storage: None,
10704                dataspace,
10705                read_format: None,
10706                obj_header_addr: 0, // set during finalize
10707                data_addr: UNDEF_ADDR,
10708                data_size: 0,
10709                compact: Some(vec![0u8; data_size as usize]),
10710                chunked: None,
10711                fixed_array: None,
10712                implicit: None,
10713                single_chunk: None,
10714                btree_v1: None,
10715                btree_v2: None,
10716                append: None,
10717                attributes: Vec::new(),
10718                obj_header_written_addr: None,
10719                obj_header_blocks: Vec::new(),
10720                filter_pipeline: None,
10721                deleted: false,
10722                extent_dirty: false,
10723                header_dirty: false,
10724                nlink_written: 1,
10725                creation_seq: self.take_creation_seq(),
10726                track_attr_order: self.track_order.attrs,
10727                fill_value: None,
10728                fill_time: FILL_TIME_IFSET,
10729                layout_version: 4,
10730                times: self.created_object_times(),
10731            },
10732        );
10733
10734        Ok(idx)
10735    }
10736
10737    /// Define a new dataset with the NULL dataspace: no elements at all.
10738    ///
10739    /// Distinct from a scalar dataset (`create_dataset` with `dims == []`),
10740    /// which holds exactly one element — a NULL dataspace holds zero, so
10741    /// there is no raw image to allocate: `data_addr` stays `UNDEF_ADDR` and
10742    /// `data_size` stays 0 permanently, the same terminal state
10743    /// `create_dataset` already reaches for a zero-length dimension.
10744    pub fn create_null_dataset(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
10745        let create = self.begin_create(name)?;
10746        let name = create.name.as_str();
10747
10748        let idx = self.push_dataset(
10749            &create,
10750            DatasetInfo {
10751                name: name.to_string(),
10752                datatype,
10753                committed_type: None,
10754                external: None,
10755                virtual_storage: None,
10756                dataspace: DataspaceMessage::null(),
10757                read_format: None,
10758                obj_header_addr: 0, // set during finalize
10759                data_addr: UNDEF_ADDR,
10760                data_size: 0,
10761                compact: None,
10762                chunked: None,
10763                fixed_array: None,
10764                implicit: None,
10765                single_chunk: None,
10766                btree_v1: None,
10767                btree_v2: None,
10768                append: None,
10769                attributes: Vec::new(),
10770                obj_header_written_addr: None,
10771                obj_header_blocks: Vec::new(),
10772                filter_pipeline: None,
10773                deleted: false,
10774                extent_dirty: false,
10775                header_dirty: false,
10776                nlink_written: 1,
10777                creation_seq: self.take_creation_seq(),
10778                track_attr_order: self.track_order.attrs,
10779                fill_value: None,
10780                fill_time: FILL_TIME_IFSET,
10781                layout_version: 4,
10782                times: self.created_object_times(),
10783            },
10784        );
10785
10786        Ok(idx)
10787    }
10788
10789    /// Define a new chunked dataset with an extensible array index.
10790    ///
10791    /// Returns the dataset index. The dataset starts empty (dims[0] = 0 if
10792    /// the first dimension is unlimited). Use `write_chunk` and
10793    /// `extend_dataset` to add data.
10794    pub fn create_chunked_dataset(
10795        &self,
10796        name: &str,
10797        datatype: DatatypeMessage,
10798        dims: &[u64],
10799        max_dims: &[u64],
10800        chunk_dims: &[u64],
10801    ) -> IoResult<usize> {
10802        let create = self.begin_create(name)?;
10803        let name = create.name.as_str();
10804        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
10805        ensure_at_most_one_unlimited(max_dims)?;
10806        let chunk_bytes = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
10807        let layout_version = self.chunk_layout_version(false, chunk_bytes);
10808        let earray_params = EarrayParams::default_params();
10809        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
10810        let nsblk_addrs = compute_nsblk_addrs(
10811            earray_params.idx_blk_elmts,
10812            earray_params.data_blk_min_elmts,
10813            earray_params.sup_blk_min_data_ptrs,
10814            earray_params.max_nelmts_bits,
10815        )?;
10816
10817        // Create EA header
10818        let mut ea_header = ExtensibleArrayHeader::new_for_chunks(&self.ctx);
10819        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
10820        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
10821        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
10822        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
10823        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
10824
10825        // Allocate and write EA header (placeholder, will be updated)
10826        let hdr_encoded = ea_header.encode(&self.ctx);
10827        let ea_header_addr = self
10828            .allocator
10829            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
10830
10831        // Create EA index block with pre-allocated super block address slots
10832        let ea_iblk = ExtensibleArrayIndexBlock::new(
10833            ea_header_addr,
10834            earray_params.idx_blk_elmts,
10835            ndblk_addrs,
10836            nsblk_addrs,
10837        );
10838
10839        // Allocate and write EA index block
10840        let iblk_encoded = ea_iblk.encode(&self.ctx);
10841        let ea_iblk_addr = self
10842            .allocator
10843            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
10844
10845        // Update header with index block address
10846        ea_header.idx_blk_addr = ea_iblk_addr;
10847
10848        // Write both to disk
10849        let hdr_encoded = ea_header.encode(&self.ctx);
10850        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
10851        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
10852
10853        // Build dataspace with max dims
10854        let dataspace = DataspaceMessage {
10855            // Chunked storage always requires at least one dimension, so
10856            // this is never Scalar or Null.
10857            class: DataspaceClass::Simple,
10858            dims: dims.to_vec(),
10859            max_dims: Some(max_dims.to_vec()),
10860        };
10861
10862        let idx = self.push_dataset(
10863            &create,
10864            DatasetInfo {
10865                name: name.to_string(),
10866                datatype,
10867                committed_type: None,
10868                external: None,
10869                virtual_storage: None,
10870                dataspace,
10871                read_format: None,
10872                obj_header_addr: 0,
10873                data_addr: UNDEF_ADDR,
10874                data_size: 0,
10875                compact: None,
10876                attributes: Vec::new(),
10877                obj_header_written_addr: None,
10878                obj_header_blocks: Vec::new(),
10879                filter_pipeline: None,
10880                deleted: false,
10881                extent_dirty: false,
10882                header_dirty: false,
10883                nlink_written: 1,
10884                creation_seq: self.take_creation_seq(),
10885                track_attr_order: self.track_order.attrs,
10886                fill_value: None,
10887                fill_time: FILL_TIME_IFSET,
10888                layout_version,
10889                times: self.created_object_times(),
10890                fixed_array: None,
10891                implicit: None,
10892                single_chunk: None,
10893                btree_v1: None,
10894                btree_v2: None,
10895                chunked: Some(ChunkedDatasetInfo {
10896                    chunk_dims: chunk_dims.to_vec(),
10897                    earray_params,
10898                    ea_header_addr,
10899                    ea_iblk_addr,
10900                    ea_header,
10901                    ea_iblk,
10902                    chunks_written: 0,
10903                    filt_iblk: None,
10904                    chunk_size_len: 0,
10905                }),
10906                append: None,
10907            },
10908        );
10909
10910        Ok(idx)
10911    }
10912
10913    /// Write `data` into a contiguous dataset's raw storage at *dataset-
10914    /// relative* byte offset `off`.
10915    ///
10916    /// The single owner of a contiguous raw-data write. Which storage that is
10917    /// — a block of this file, or the files an External File List names — is
10918    /// decided once, by [`DatasetInfo::contiguous_target`], and never at a
10919    /// call site.
10920    fn write_contiguous_bytes(
10921        &self,
10922        target: &ContiguousTarget,
10923        off: u64,
10924        data: &[u8],
10925    ) -> IoResult<()> {
10926        match target {
10927            ContiguousTarget::Local(addr) => Ok(self.handle.write_at(addr + off, data)?),
10928            ContiguousTarget::External { files, prefix } => {
10929                // The prefix the open settled, not one resolved here:
10930                // `H5D__efl_write` joins against `dset->shared->extfile_prefix`
10931                // (H5Defl.c:429-431), the same field `H5D__efl_read` joins
10932                // against, so a relative name lands where a later read looks.
10933                write_external_file_bytes(files, prefix.as_deref(), off, data)
10934            }
10935            ContiguousTarget::Virtual => Err(virtual_write_refused()),
10936        }
10937    }
10938
10939    /// Write raw bytes to a contiguous dataset identified by `index`.
10940    ///
10941    /// The caller is responsible for providing data in the correct byte order
10942    /// and layout. The length must match the total data size declared at
10943    /// creation time.
10944    pub fn write_dataset_raw(&self, index: usize, data: &[u8]) -> IoResult<()> {
10945        let ds = self.ds(index);
10946        let _op = ds.op.lock();
10947        let target = {
10948            let mut g = ds.lock();
10949            if g.is_chunked() {
10950                return Err(crate::io::IoError::InvalidState(
10951                    "use write_chunk for chunked datasets".into(),
10952                ));
10953            }
10954            // A compact dataset's raw image is its layout message, so the
10955            // write lands in the buffer the header is built from rather than
10956            // at a file offset, and the header it is built into is now stale.
10957            if let Some(image) = g.compact.as_mut() {
10958                if data.len() != image.len() {
10959                    return Err(crate::io::IoError::InvalidState(format!(
10960                        "data size mismatch: expected {} bytes, got {}",
10961                        image.len(),
10962                        data.len()
10963                    )));
10964                }
10965                image.copy_from_slice(data);
10966                g.header_dirty = true;
10967                return Ok(());
10968            }
10969            let Some(target) = g.contiguous_target() else {
10970                return Err(crate::io::IoError::InvalidState(
10971                    "dataset has no data allocated".into(),
10972                ));
10973            };
10974            // A dataset that stores nothing of its own has no byte count to
10975            // check a write against — `write_contiguous_bytes` refuses it by
10976            // name below, which is the answer the caller needs.
10977            if target.is_storage() && data.len() as u64 != g.data_size {
10978                return Err(crate::io::IoError::InvalidState(format!(
10979                    "data size mismatch: expected {} bytes, got {}",
10980                    g.data_size,
10981                    data.len()
10982                )));
10983            }
10984            target
10985        };
10986        self.write_contiguous_bytes(&target, 0, data)
10987    }
10988
10989    /// Write a chunk of data to a chunked dataset.
10990    ///
10991    /// `chunk_offset` is the chunk coordinates (e.g., [frame_idx] for a 1D-chunked
10992    /// streaming dataset where chunk_dims = [1, H, W]).
10993    /// Only the first (unlimited) dimension index is used for EA indexing.
10994    ///
10995    /// `data` must be exactly chunk_size bytes (product of chunk_dims * element_size).
10996    pub fn write_chunk(&self, index: usize, chunk_idx: u64, data: &[u8]) -> IoResult<()> {
10997        let ds = self.ds(index);
10998        let _op = ds.op.lock();
10999        self.write_chunk_inner(index, chunk_idx, data)
11000    }
11001
11002    /// [`Self::write_chunk`] body; the caller holds the dataset's op lock or
11003    /// the writer exclusively.
11004    pub(crate) fn write_chunk_inner(
11005        &self,
11006        index: usize,
11007        chunk_idx: u64,
11008        data: &[u8],
11009    ) -> IoResult<()> {
11010        let ds = self.ds(index);
11011        // Read the chunk geometry and filter pipeline under one brief lock,
11012        // then drop it: compression runs *outside* the lock, and
11013        // `record_ea_chunk` re-locks the same slot, so the guard must not be
11014        // held across either.
11015        let (chunk_bytes, pipeline) = {
11016            let g = ds.lock();
11017            let element_size = g.datatype.element_size() as u64;
11018            let chunked = g
11019                .chunked
11020                .as_ref()
11021                .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?;
11022            (
11023                chunked.chunk_dims.iter().product::<u64>() * element_size,
11024                g.filter_pipeline.clone(),
11025            )
11026        };
11027
11028        if data.len() as u64 != chunk_bytes {
11029            return Err(crate::io::IoError::InvalidState(format!(
11030                "chunk data size mismatch: expected {} bytes, got {}",
11031                chunk_bytes,
11032                data.len()
11033            )));
11034        }
11035
11036        // Apply compression if filter pipeline is set
11037        let compressed;
11038        let write_data = if let Some(ref pipeline) = pipeline {
11039            compressed = filter::apply_filters(pipeline, data)?;
11040            &compressed
11041        } else {
11042            data
11043        };
11044        // filter_mask = 0: this path runs the whole pipeline, so no filter is
11045        // skipped for the chunk.
11046        self.record_ea_chunk(index, chunk_idx, write_data, 0)
11047    }
11048
11049    /// Decide where a chunk's bytes belong and put them there, returning the
11050    /// address to record in the index.
11051    ///
11052    /// `old` is the chunk's current `(address, stored length)` if the index
11053    /// already holds an entry for it. This is the single owner of the
11054    /// rewrite-placement rule, mirroring libhdf5's `H5D__chunk_file_alloc`
11055    /// (`H5Dchunk.c`): a chunk whose stored size is unchanged is overwritten
11056    /// where it already lives, and only a chunk that no longer fits moves,
11057    /// releasing its old block. Without this every rewrite would abandon the
11058    /// old block and grow the file.
11059    fn place_chunk(&self, old: Option<(u64, u64)>, new_len: u64) -> u64 {
11060        match old {
11061            // Same stored size: overwrite in place. This is every unfiltered
11062            // rewrite (the stored size is fixed by the chunk shape) and every
11063            // filtered rewrite that compressed to the same length.
11064            Some((addr, len)) if addr != UNDEF_ADDR && len == new_len => addr,
11065            Some((addr, len)) if addr != UNDEF_ADDR => {
11066                // The chunk has to move. Under SWMR a reader may still hold an
11067                // index that points at the old block, so libhdf5 keeps it
11068                // (H5D__chunk_file_alloc skips H5MF_xfree when the file is
11069                // open for SWMR writing); do the same.
11070                if !self.swmr_active {
11071                    self.allocator.free(addr, len, FreeSpaceClass::RawData);
11072                }
11073                self.allocator.allocate(new_len, FreeSpaceClass::RawData)
11074            }
11075            _ => self.allocator.allocate(new_len, FreeSpaceClass::RawData),
11076        }
11077    }
11078
11079    /// Place a chunk's already-final bytes (filtered if the dataset is
11080    /// filtered) in the file and record them in the extensible-array index —
11081    /// in the index block, a data block, or a super block per the EA geometry.
11082    /// Shared by write_chunk and write_compressed_chunk.
11083    ///
11084    /// The index lookup happens *before* the bytes are placed, because the
11085    /// entry it finds is what tells [`place_chunk`](Self::place_chunk) whether
11086    /// this is a rewrite that can stay put.
11087    fn record_ea_chunk(
11088        &self,
11089        index: usize,
11090        chunk_idx: u64,
11091        final_bytes: &[u8],
11092        filter_mask: u32,
11093    ) -> IoResult<()> {
11094        let compressed_size = final_bytes.len() as u64;
11095        let ds = self.ds(index);
11096        // Hold one slot guard for the whole method: every dataset-state access
11097        // below goes through `m`, while `self.handle`/`self.allocator`/`self.ctx`
11098        // are disjoint fields safe to touch with the guard held.
11099        let mut m = ds.lock();
11100        let is_filtered = m.filter_pipeline.is_some();
11101        // For a filtered dataset the chunk's stored size is encoded in the
11102        // `chunk_size_len`-byte field of each filtered EA entry
11103        // (`FilteredChunkEntry::encode` writes `nbytes[..chunk_size_len]`,
11104        // which truncates silently). Reject a size that would not fit, the way
11105        // libhdf5's H5D_CHUNK_ENCODE_SIZE_CHECK does, instead of corrupting the
11106        // index. The compress path never exceeds this (chunk_size_len holds the
11107        // uncompressed chunk size); a direct/raw write with caller-supplied
11108        // bytes can.
11109        if is_filtered {
11110            let chunk_size_len = m.chunked.as_ref().unwrap().chunk_size_len as usize;
11111            if chunk_size_len < 8 && compressed_size >= (1u64 << (chunk_size_len * 8)) {
11112                return Err(crate::io::IoError::InvalidState(format!(
11113                    "filtered chunk size {compressed_size} does not fit in the \
11114                     {chunk_size_len}-byte extensible-array chunk-size field"
11115                )));
11116            }
11117        }
11118        let idx_blk_elmts = {
11119            let c = m.chunked.as_ref().unwrap();
11120            c.earray_params.idx_blk_elmts as u64
11121        };
11122
11123        if chunk_idx < idx_blk_elmts {
11124            let chunked = m.chunked.as_mut().unwrap();
11125            if is_filtered {
11126                if let Some(ref mut fiblk) = chunked.filt_iblk {
11127                    let old = fiblk.elements[chunk_idx as usize];
11128                    let chunk_addr =
11129                        self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11130                    self.handle.write_at(chunk_addr, final_bytes)?;
11131                    fiblk.elements[chunk_idx as usize] = FilteredChunkEntry {
11132                        addr: chunk_addr,
11133                        nbytes: compressed_size,
11134                        filter_mask,
11135                    };
11136                }
11137            } else {
11138                // An unfiltered chunk's stored size is fixed by the chunk
11139                // shape, so a rewrite always fits where it already is.
11140                let old = chunked.ea_iblk.elements[chunk_idx as usize];
11141                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11142                self.handle.write_at(chunk_addr, final_bytes)?;
11143                chunked.ea_iblk.elements[chunk_idx as usize] = chunk_addr;
11144            }
11145            chunked.chunks_written += 1;
11146            if chunk_idx + 1 > chunked.ea_header.max_idx_set {
11147                chunked.ea_header.max_idx_set = chunk_idx + 1;
11148            }
11149            if chunked.ea_header.num_elmts_realized < idx_blk_elmts {
11150                chunked.ea_header.num_elmts_realized = idx_blk_elmts;
11151            }
11152        } else {
11153            // chunk_idx >= idx_blk_elmts: place the chunk through the EA
11154            // data-block / super-block hierarchy (libhdf5-compatible geometry).
11155            let (geo, max_nelmts_bits, chunk_size_len, ea_header_addr) = {
11156                let c = m.chunked.as_ref().unwrap();
11157                let p = &c.earray_params;
11158                (
11159                    EaGeometry::new(
11160                        p.idx_blk_elmts,
11161                        p.data_blk_min_elmts,
11162                        p.sup_blk_min_data_ptrs,
11163                        p.max_nelmts_bits,
11164                        p.max_dblk_page_nelmts_bits,
11165                    )?,
11166                    p.max_nelmts_bits,
11167                    c.chunk_size_len,
11168                    c.ea_header_addr,
11169                )
11170            };
11171            let loc = match geo.locate(chunk_idx)? {
11172                EaLoc::Dblk(l) => l,
11173                EaLoc::Index { .. } => unreachable!("chunk_idx >= idx_blk_elmts"),
11174            };
11175            if loc.paged {
11176                return Err(crate::io::IoError::InvalidState(format!(
11177                    "chunk index {} needs a paged extensible-array data block, \
11178                     which is not yet supported",
11179                    chunk_idx
11180                )));
11181            }
11182            let class_id = if is_filtered {
11183                EA_CLS_FILT_CHUNK
11184            } else {
11185                EA_CLS_CHUNK
11186            };
11187            let dblk_nelmts = loc.dblk_nelmts as usize;
11188
11189            // Resolve the data block's current address and its parent slot,
11190            // creating the owning super block on demand.
11191            let parent: DblkParent;
11192            let mut dblk_addr: u64;
11193            match loc.path {
11194                EaDblkPath::Direct { idx: di } => {
11195                    let c = m.chunked.as_ref().unwrap();
11196                    dblk_addr = if is_filtered {
11197                        c.filt_iblk.as_ref().unwrap().dblk_addrs[di]
11198                    } else {
11199                        c.ea_iblk.dblk_addrs[di]
11200                    };
11201                    parent = DblkParent::IndexBlock(di);
11202                }
11203                EaDblkPath::ViaSblk {
11204                    sblk_off,
11205                    local_dblk,
11206                    ndblks_in_sblk,
11207                    sblk_block_offset,
11208                } => {
11209                    let mut sblk_addr = {
11210                        let c = m.chunked.as_ref().unwrap();
11211                        if is_filtered {
11212                            c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
11213                        } else {
11214                            c.ea_iblk.sblk_addrs[sblk_off]
11215                        }
11216                    };
11217                    if sblk_addr == UNDEF_ADDR {
11218                        let sb = ExtensibleArraySuperBlock::new(
11219                            class_id,
11220                            ea_header_addr,
11221                            sblk_block_offset,
11222                            ndblks_in_sblk,
11223                        );
11224                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11225                        sblk_addr = self
11226                            .allocator
11227                            .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11228                        self.handle.write_at(sblk_addr, &enc)?;
11229                        let c = m.chunked.as_mut().unwrap();
11230                        if is_filtered {
11231                            c.filt_iblk.as_mut().unwrap().sblk_addrs[sblk_off] = sblk_addr;
11232                        } else {
11233                            c.ea_iblk.sblk_addrs[sblk_off] = sblk_addr;
11234                        }
11235                        c.ea_header.num_sblks_created += 1;
11236                        c.ea_header.size_sblks_created += enc.len() as u64;
11237                    }
11238                    let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
11239                    // The writer never creates paged super blocks (it errors
11240                    // before the paging threshold), so page_init_total is 0.
11241                    let sb = ExtensibleArraySuperBlock::decode(
11242                        &sb_buf,
11243                        &self.ctx,
11244                        max_nelmts_bits,
11245                        ndblks_in_sblk,
11246                        0,
11247                    )?;
11248                    dblk_addr = sb.dblk_addrs[local_dblk];
11249                    parent = DblkParent::SuperBlock {
11250                        sblk_addr,
11251                        ndblks_in_sblk,
11252                        local_dblk,
11253                    };
11254                }
11255            }
11256
11257            // Create or update the data block holding this chunk's entry.
11258            let created = dblk_addr == UNDEF_ADDR;
11259            if is_filtered {
11260                let mut dblk = if created {
11261                    FilteredDataBlock::new(ea_header_addr, loc.dblk_block_offset, dblk_nelmts)
11262                } else {
11263                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11264                    FilteredDataBlock::decode(
11265                        &buf,
11266                        &self.ctx,
11267                        max_nelmts_bits,
11268                        dblk_nelmts,
11269                        chunk_size_len,
11270                    )?
11271                };
11272                // A freshly created data block holds only undefined addresses,
11273                // so this reads as "no previous chunk" without a special case.
11274                let old = dblk.elements[loc.offset_in_dblk as usize];
11275                let chunk_addr = self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11276                self.handle.write_at(chunk_addr, final_bytes)?;
11277                let entry = FilteredChunkEntry {
11278                    addr: chunk_addr,
11279                    nbytes: compressed_size,
11280                    filter_mask,
11281                };
11282                dblk.elements[loc.offset_in_dblk as usize] = entry;
11283                let enc = dblk.encode(&self.ctx, max_nelmts_bits, chunk_size_len);
11284                if created {
11285                    dblk_addr = self
11286                        .allocator
11287                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11288                }
11289                self.handle.write_at(dblk_addr, &enc)?;
11290                if created {
11291                    let c = m.chunked.as_mut().unwrap();
11292                    c.ea_header.num_dblks_created += 1;
11293                    c.ea_header.size_dblks_created += enc.len() as u64;
11294                }
11295            } else {
11296                let mut dblk = if created {
11297                    ExtensibleArrayDataBlock::new(
11298                        ea_header_addr,
11299                        loc.dblk_block_offset,
11300                        dblk_nelmts,
11301                    )
11302                } else {
11303                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11304                    ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, dblk_nelmts)?
11305                };
11306                // Unfiltered: the stored size is fixed by the chunk shape, so
11307                // a rewrite always fits its old block. A freshly created data
11308                // block holds undefined addresses and falls through to a new
11309                // allocation.
11310                let old = dblk.elements[loc.offset_in_dblk as usize];
11311                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11312                self.handle.write_at(chunk_addr, final_bytes)?;
11313                dblk.elements[loc.offset_in_dblk as usize] = chunk_addr;
11314                let enc = dblk.encode(&self.ctx, max_nelmts_bits);
11315                if created {
11316                    dblk_addr = self
11317                        .allocator
11318                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11319                }
11320                self.handle.write_at(dblk_addr, &enc)?;
11321                if created {
11322                    let c = m.chunked.as_mut().unwrap();
11323                    c.ea_header.num_dblks_created += 1;
11324                    c.ea_header.size_dblks_created += enc.len() as u64;
11325                }
11326            }
11327
11328            // Record a newly-created data block's address in its parent.
11329            if created {
11330                match parent {
11331                    DblkParent::IndexBlock(di) => {
11332                        let c = m.chunked.as_mut().unwrap();
11333                        if is_filtered {
11334                            c.filt_iblk.as_mut().unwrap().dblk_addrs[di] = dblk_addr;
11335                        } else {
11336                            c.ea_iblk.dblk_addrs[di] = dblk_addr;
11337                        }
11338                    }
11339                    DblkParent::SuperBlock {
11340                        sblk_addr,
11341                        ndblks_in_sblk,
11342                        local_dblk,
11343                    } => {
11344                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
11345                        let mut sb = ExtensibleArraySuperBlock::decode(
11346                            &buf,
11347                            &self.ctx,
11348                            max_nelmts_bits,
11349                            ndblks_in_sblk,
11350                            0,
11351                        )?;
11352                        sb.dblk_addrs[local_dblk] = dblk_addr;
11353                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11354                        self.handle.write_at(sblk_addr, &enc)?;
11355                    }
11356                }
11357            }
11358
11359            // Statistics.
11360            let c = m.chunked.as_mut().unwrap();
11361            c.chunks_written += 1;
11362            if chunk_idx + 1 > c.ea_header.max_idx_set {
11363                c.ea_header.max_idx_set = chunk_idx + 1;
11364            }
11365            if created {
11366                c.ea_header.num_elmts_realized += loc.dblk_nelmts;
11367            }
11368        }
11369        Ok(())
11370    }
11371
11372    /// Write a slice (hyperslab) of data to a dataset, contiguous or chunked.
11373    ///
11374    /// `starts` and `counts` define the N-dimensional selection.
11375    /// `data` must be exactly `product(counts) * element_size` bytes.
11376    ///
11377    /// The selection is validated once here and then handed to the layout's
11378    /// own writer, so a caller never has to know which storage the dataset
11379    /// uses.
11380    pub fn write_slice(
11381        &self,
11382        index: usize,
11383        starts: &[u64],
11384        counts: &[u64],
11385        data: &[u8],
11386    ) -> IoResult<()> {
11387        let ds = self.ds(index);
11388        let _op = ds.op.lock();
11389        self.write_slice_inner(index, starts, counts, data)
11390    }
11391
11392    /// [`Self::write_slice`] body; the caller holds the dataset's op lock or
11393    /// the writer exclusively.
11394    pub(crate) fn write_slice_inner(
11395        &self,
11396        index: usize,
11397        starts: &[u64],
11398        counts: &[u64],
11399        data: &[u8],
11400    ) -> IoResult<()> {
11401        let ds_ref = self.ds(index);
11402        let ds = ds_ref.lock();
11403        let is_chunked = ds.is_chunked();
11404
11405        let dims = &ds.dataspace.dims;
11406        let element_size = ds.datatype.element_size() as u64;
11407        let ndims = dims.len();
11408
11409        // Every hyperslab edge must stay inside the dataset; without this an
11410        // out-of-bounds selection writes raw bytes over neighbouring data.
11411        check_hyperslab(dims, starts, counts)?;
11412        if ndims == 0 {
11413            return Err(crate::io::IoError::InvalidState(
11414                "write_slice does not support scalar datasets; use write_dataset_raw".into(),
11415            ));
11416        }
11417
11418        let out_elems: u64 = counts.iter().product();
11419        if data.len() as u64 != out_elems * element_size {
11420            return Err(crate::io::IoError::InvalidState(format!(
11421                "data size mismatch: expected {} bytes, got {}",
11422                out_elems * element_size,
11423                data.len()
11424            )));
11425        }
11426
11427        // `dims` borrows the dataset slot; collect what the writers below need
11428        // so the guard can be dropped before they re-lock it.
11429        let dims = dims.clone();
11430        let target = ds.contiguous_target();
11431        drop(ds);
11432
11433        if is_chunked {
11434            // Rows the append buffer holds are not in the chunks yet; writing
11435            // them there anyway would be undone when the buffer flushes at
11436            // close. Hand them to the chunks first.
11437            self.flush_append_buffer_if_intersecting(index, starts[0], starts[0] + counts[0])?;
11438            return self.write_slice_chunked(index, starts, counts, data);
11439        }
11440        let Some(target) = target else {
11441            return Err(crate::io::IoError::InvalidState(
11442                "dataset has no data allocated".into(),
11443            ));
11444        };
11445
11446        // Write each maximal contiguous run in one write. Trailing
11447        // full-selected dimensions coalesce, mirroring the read path: a slice
11448        // with a full last axis becomes one write per outer index instead of
11449        // one write per last-axis row.
11450        for_each_contiguous_run(
11451            &dims,
11452            starts,
11453            counts,
11454            element_size,
11455            |dst_off, src_off, len| {
11456                self.write_contiguous_bytes(&target, dst_off, &data[src_off..src_off + len])
11457            },
11458        )?;
11459
11460        Ok(())
11461    }
11462
11463    /// Write a hyperslab into a chunked dataset, one chunk at a time.
11464    ///
11465    /// The selection is already validated by [`write_slice`](Self::write_slice).
11466    /// For each chunk the selection touches, the chunk's share of `data` is
11467    /// scattered into a whole-chunk buffer and the chunk is rewritten:
11468    ///
11469    /// - a chunk the selection covers completely is built from `data` alone —
11470    ///   nothing needs reading back (libhdf5 takes the same shortcut with the
11471    ///   `relax` flag of `H5D__chunk_lock`);
11472    /// - a chunk covered only in part starts from what is already stored, or
11473    ///   from a fill-value buffer when the chunk has never been written, so
11474    ///   neighbouring elements survive and untouched ones read as fill.
11475    ///
11476    /// An edge chunk that hangs past the dataset extent is always the partial
11477    /// case, so the region beyond the extent keeps its fill value.
11478    fn write_slice_chunked(
11479        &self,
11480        index: usize,
11481        starts: &[u64],
11482        counts: &[u64],
11483        data: &[u8],
11484    ) -> IoResult<()> {
11485        if counts.contains(&0) {
11486            return Ok(());
11487        }
11488        let geo = self.chunk_geometry(index)?;
11489        let ndims = geo.dims.len();
11490        if geo.chunk_dims.len() != ndims {
11491            return Err(crate::io::IoError::InvalidState(format!(
11492                "dataset chunk shape has {} dimensions but the dataspace has {}",
11493                geo.chunk_dims.len(),
11494                ndims
11495            )));
11496        }
11497        if geo.chunk_dims.contains(&0) {
11498            return Err(crate::io::IoError::InvalidState(
11499                "chunk shape has a zero-length dimension".into(),
11500            ));
11501        }
11502        let chunk_bytes = geo.chunk_bytes() as usize;
11503
11504        // Grid range the selection touches, inclusive on both ends.
11505        let first: Vec<u64> = (0..ndims).map(|d| starts[d] / geo.chunk_dims[d]).collect();
11506        let last: Vec<u64> = (0..ndims)
11507            .map(|d| (starts[d] + counts[d] - 1) / geo.chunk_dims[d])
11508            .collect();
11509
11510        let mut coords = first.clone();
11511        loop {
11512            // Intersect the selection with this chunk. `in_chunk` is the
11513            // region's origin inside the chunk, `in_data` its origin inside
11514            // the caller's counts-shaped buffer, `extent` its size.
11515            let mut in_chunk = vec![0u64; ndims];
11516            let mut in_data = vec![0u64; ndims];
11517            let mut extent = vec![0u64; ndims];
11518            let mut covers_whole_chunk = true;
11519            for d in 0..ndims {
11520                let chunk_origin = coords[d] * geo.chunk_dims[d];
11521                let lo = starts[d].max(chunk_origin);
11522                let hi = (starts[d] + counts[d]).min(chunk_origin + geo.chunk_dims[d]);
11523                in_chunk[d] = lo - chunk_origin;
11524                in_data[d] = lo - starts[d];
11525                extent[d] = hi - lo;
11526                if in_chunk[d] != 0 || extent[d] != geo.chunk_dims[d] {
11527                    covers_whole_chunk = false;
11528                }
11529            }
11530
11531            let mut buf = if covers_whole_chunk {
11532                // Every byte is overwritten below.
11533                vec![0u8; chunk_bytes]
11534            } else {
11535                match self.read_chunk_at_coords(index, &coords)? {
11536                    Some(existing) => {
11537                        if existing.len() != chunk_bytes {
11538                            return Err(crate::io::IoError::InvalidState(format!(
11539                                "stored chunk at {coords:?} is {} bytes but the chunk shape \
11540                                 needs {chunk_bytes}",
11541                                existing.len()
11542                            )));
11543                        }
11544                        existing
11545                    }
11546                    None => self.new_write_chunk_buffer(index, chunk_bytes),
11547                }
11548            };
11549
11550            for_each_dual_run(
11551                &geo.chunk_dims,
11552                &in_chunk,
11553                counts,
11554                &in_data,
11555                &extent,
11556                geo.element_size,
11557                |dst_off, src_off, len| {
11558                    let dst = dst_off as usize;
11559                    let src = src_off as usize;
11560                    buf[dst..dst + len].copy_from_slice(&data[src..src + len]);
11561                    Ok(())
11562                },
11563            )?;
11564            self.write_chunk_at_coords(index, &coords, &buf)?;
11565
11566            // Odometer over the touched grid range.
11567            let mut d = ndims;
11568            loop {
11569                if d == 0 {
11570                    return Ok(());
11571                }
11572                d -= 1;
11573                if coords[d] < last[d] {
11574                    coords[d] += 1;
11575                    break;
11576                }
11577                coords[d] = first[d];
11578            }
11579        }
11580    }
11581
11582    /// Add an attribute to the root group (file-level attribute), replacing
11583    /// a same-name attribute. See [`set_attribute`](Self::set_attribute).
11584    pub fn add_root_attribute(&self, attr: AttributeMessage) -> IoResult<()> {
11585        self.set_attribute(AttrTarget::Root, attr)
11586    }
11587
11588    /// Insert `attr` into the attribute list `target` names, replacing a
11589    /// same-name attribute.
11590    ///
11591    /// The single owner of attribute-list mutation: an `AttributeMessage`
11592    /// that leaves a list here has its vlen global-heap objects released, so
11593    /// no replacement — vlen over vlen, numeric over vlen — can strand heap
11594    /// space (the attribute counterpart of issue #10's dataset fix).
11595    ///
11596    /// Under SWMR every attribute mutation is refused, matching libhdf5's
11597    /// rule for SWMR writes. Object headers are frozen once streaming
11598    /// starts — a change was committed at close only when the header
11599    /// happened to be rebuilt (group attrs always, dataset attrs only if
11600    /// the dataset also got chunk writes) and silently dropped otherwise —
11601    /// and a replacement's superseded vlen value could never be reclaimed,
11602    /// since a streaming reader may hold its heap references.
11603    pub fn set_attribute(&self, target: AttrTarget<'_>, attr: AttributeMessage) -> IoResult<()> {
11604        self.insert_attribute(target, attr, Created)
11605    }
11606
11607    /// The body of [`set_attribute`](Self::set_attribute), told whether the
11608    /// attribute it is inserting is genuinely new — see [`AttrOrigin`].
11609    fn insert_attribute(
11610        &self,
11611        target: AttrTarget<'_>,
11612        attr: AttributeMessage,
11613        origin: AttrOrigin,
11614    ) -> IoResult<()> {
11615        if self.swmr_active {
11616            return Err(swmr_attr_error(&attr.name));
11617        }
11618        // Whatever this name meant before, it means the incoming message now.
11619        self.forget_attribute_reference(self.attr_scope(target)?, &attr.name);
11620        // No size gate: an attribute whose message is too large for the
11621        // 16-bit size field an object header message has spills the object's
11622        // whole attribute set to dense storage at finalize, exactly as
11623        // `H5O__attr_create` does. See `attributes_need_dense`.
11624        let mut entry = AttributeEntry::from(attr);
11625        let old = self.with_attr_list(target, |attrs| {
11626            if let Some(pos) = attrs.iter().position(|a| a.name() == entry.name()) {
11627                // `H5O__attr_write` replaces an existing attribute's value and
11628                // leaves its `crt_idx` alone: the attribute was not created
11629                // again, so its creation index does not move.
11630                entry.set_creation_index(attrs[pos].creation_index());
11631                Some(std::mem::replace(&mut attrs[pos], entry))
11632            } else {
11633                // `H5O__attr_create` stamps the set's running maximum onto the
11634                // new attribute and post-increments it — but only a create
11635                // reaches for it.
11636                entry.set_creation_index(match origin {
11637                    Created => Some(next_creation_index(attrs)),
11638                    Rewritten(kept) => kept,
11639                });
11640                attrs.push(entry);
11641                None
11642            }
11643        })?;
11644        match old {
11645            Some(old) => self.release_attr_vlen(&old),
11646            None => Ok(()),
11647        }
11648    }
11649
11650    /// Set a variable-length string attribute on `target`, replacing any
11651    /// same-name attribute.
11652    ///
11653    /// Owns the whole replacement sequence: the superseded attribute is
11654    /// removed and its heap objects released *before* the new value's
11655    /// collection is allocated — the free-before-alloc order (issue #10)
11656    /// that lets a reopen-replace loop land in the block it just freed
11657    /// instead of growing the file every session. The cost, as on the
11658    /// dataset path: a failure between the eviction and the insert below
11659    /// loses the attribute rather than leaking its heap space.
11660    pub fn set_vlen_string_attribute(
11661        &self,
11662        target: AttrTarget<'_>,
11663        name: &str,
11664        value: &str,
11665    ) -> IoResult<()> {
11666        let origin = self.evict_attr(target, name)?;
11667        let attr = self.vlen_string_attribute(name, value)?;
11668        self.insert_attribute(target, attr, origin)
11669    }
11670
11671    /// The array counterpart of
11672    /// [`set_vlen_string_attribute`](Self::set_vlen_string_attribute).
11673    pub fn set_vlen_string_array_attribute(
11674        &self,
11675        target: AttrTarget<'_>,
11676        name: &str,
11677        values: &[&str],
11678        dims: &[u64],
11679    ) -> IoResult<()> {
11680        let origin = self.evict_attr(target, name)?;
11681        let attr = self.vlen_string_array_attribute(name, values, dims)?;
11682        self.insert_attribute(target, attr, origin)
11683    }
11684
11685    /// Set an attribute on `target` whose value is the object references
11686    /// naming `paths` — h5py's `obj.attrs['ref'] = f['/target'].ref`.
11687    ///
11688    /// `dims` is the attribute's dataspace: empty for the scalar shape a
11689    /// single reference takes, `&[n]` for an array of them. Each path names a
11690    /// dataset or a group (`/` is the root group) and must already exist. What
11691    /// reaches the file is each target's object header address, which finalize
11692    /// assigns — so the paths are what is stored, and the attribute's message
11693    /// is built from them every time an object header is
11694    /// ([`object_attributes`](Self::object_attributes)). The message carries a
11695    /// zero image of the final width until then.
11696    pub fn set_object_reference_attribute(
11697        &self,
11698        target: AttrTarget<'_>,
11699        name: &str,
11700        paths: &[&str],
11701        dims: &[u64],
11702    ) -> IoResult<()> {
11703        let scope = self.attr_scope(target)?;
11704        // An empty `dims` is the scalar shape, whose one element the empty
11705        // product already reports.
11706        let elements: u64 = dims.iter().product();
11707        if elements != paths.len() as u64 {
11708            return Err(crate::io::IoError::InvalidState(format!(
11709                "attribute '{name}' shape {dims:?} needs {elements} references, got {}",
11710                paths.len()
11711            )));
11712        }
11713        // Resolve now as well as at finalize, so a path that names nothing is
11714        // reported at the call that got it wrong.
11715        for path in paths {
11716            self.object_reference_target(path)?;
11717        }
11718        let datatype = DatatypeMessage::object_reference(&self.ctx);
11719        let image = vec![0u8; paths.len() * datatype.element_size() as usize];
11720        let attr = if dims.is_empty() {
11721            AttributeMessage::scalar_numeric(name, datatype, image)
11722        } else {
11723            AttributeMessage::array_numeric(name, datatype, dims, image)
11724        };
11725        // Through the same owner as every other attribute, which is also what
11726        // drops any value this name carried before.
11727        self.set_attribute(target, attr)?;
11728        self.attribute_references
11729            .lock()
11730            .push(AttributeReferenceValue {
11731                scope,
11732                name: name.to_string(),
11733                targets: paths.iter().map(|p| (*p).to_string()).collect(),
11734                stride: self.ctx.sizeof_addr as usize,
11735            });
11736        Ok(())
11737    }
11738
11739    // -----------------------------------------------------------------------
11740    // Dimension scales — the H5DS high-level API (hl/src/H5DS.c)
11741    // -----------------------------------------------------------------------
11742
11743    /// Mark dataset `dsid` as a dimension scale — `H5DSset_scale`.
11744    ///
11745    /// Writes `CLASS` as the fixed-length null-terminated ASCII string
11746    /// `DIMENSION_SCALE` and, when `name` is given, `NAME` the same way: the
11747    /// `H5LT_set_attribute_string` form, one byte longer than the text so the
11748    /// terminator is stored, which is what `H5DSis_scale` requires of a scale
11749    /// (a 16-byte null-terminated `CLASS`). Either attribute already there is
11750    /// deleted and created anew, as `H5LT_set_attribute_string` does, so it
11751    /// takes a fresh creation index. A dataset with scales of its own
11752    /// (`DIMENSION_LIST`) is refused, as upstream refuses it.
11753    pub fn set_dimension_scale(&self, dsid: usize, name: Option<&str>) -> IoResult<()> {
11754        let scale_path = self.dataset_name(dsid)?;
11755        if self.dataset_attribute(dsid, DIMENSION_LIST)?.is_some() {
11756            return Err(crate::io::IoError::InvalidState(format!(
11757                "dataset '{scale_path}' has dimension scales attached and cannot become one"
11758            )));
11759        }
11760        self.set_fixed_string_attribute(dsid, "CLASS", DIMENSION_SCALE_CLASS)?;
11761        if let Some(name) = name {
11762            self.set_fixed_string_attribute(dsid, "NAME", name)?;
11763        }
11764        Ok(())
11765    }
11766
11767    /// Attach dataset `dsid` as a dimension scale of axis `idx` of dataset
11768    /// `did` — `H5DSattach_scale`.
11769    ///
11770    /// Two attributes record the attachment: `DIMENSION_LIST` on `did`, one
11771    /// variable-length sequence of object references per axis (a scalar
11772    /// dataset counts as rank 1), and `REFERENCE_LIST` on `dsid`, an array
11773    /// of `{dataset: H5T_STD_REF_OBJ, dimension: uint}` compounds naming
11774    /// every (dataset, axis) the scale is attached to. `dsid` is then made a
11775    /// scale if it is not one already ([`set_dimension_scale`] with no name).
11776    /// Both lists are rewritten whole; what an existing list holds is read
11777    /// back as paths (registered this session, or resolved from the file's
11778    /// addresses), so an attach in an append session keeps earlier
11779    /// attachments and every reference is stamped with the address its
11780    /// target ends up at.
11781    ///
11782    /// Refused, as upstream refuses them: `did == dsid`; a `did` that is a
11783    /// scale or carries a reserved `CLASS` (`IMAGE`, `PALETTE`, `TABLE`); a
11784    /// `dsid` that has scales of its own; an axis beyond `did`'s rank.
11785    ///
11786    /// Attaching a scale already attached to that axis changes nothing. This
11787    /// is stricter than upstream, which leaves `DIMENSION_LIST` as it is but
11788    /// still appends a duplicate `REFERENCE_LIST` entry; a second entry for
11789    /// the same (dataset, axis) tells `H5DSis_attached` nothing the first
11790    /// does not.
11791    ///
11792    /// [`set_dimension_scale`]: Self::set_dimension_scale
11793    pub fn attach_dimension_scale(&self, did: usize, dsid: usize, idx: usize) -> IoResult<()> {
11794        let data_path = self.dataset_name(did)?;
11795        let scale_path = self.dataset_name(dsid)?;
11796        if did == dsid {
11797            return Err(crate::io::IoError::InvalidState(format!(
11798                "dataset '{data_path}' cannot be its own dimension scale"
11799            )));
11800        }
11801        if self.is_dimension_scale(did)? {
11802            return Err(crate::io::IoError::InvalidState(format!(
11803                "dataset '{data_path}' is a dimension scale and cannot have scales attached"
11804            )));
11805        }
11806        if self.dataset_attribute(dsid, DIMENSION_LIST)?.is_some() {
11807            return Err(crate::io::IoError::InvalidState(format!(
11808                "dataset '{scale_path}' has dimension scales attached and cannot be one"
11809            )));
11810        }
11811        if self.has_reserved_class(did)? {
11812            return Err(crate::io::IoError::InvalidState(format!(
11813                "dataset '{data_path}' holds an image, palette or table and cannot have \
11814                 dimension scales"
11815            )));
11816        }
11817        let rank = self.ds(did).lock().dataspace.dims.len().max(1);
11818        if idx >= rank {
11819            return Err(crate::io::IoError::InvalidState(format!(
11820                "axis {idx} is out of range for the rank-{rank} dataset '{data_path}'"
11821            )));
11822        }
11823
11824        let mut lists = match self.dimension_list(did)? {
11825            Some(lists) => lists,
11826            None => vec![Vec::new(); rank],
11827        };
11828        if lists.len() != rank {
11829            return Err(crate::io::IoError::InvalidState(format!(
11830                "DIMENSION_LIST of '{data_path}' has {} entries for a rank-{rank} dataset",
11831                lists.len()
11832            )));
11833        }
11834        if lists[idx].contains(&scale_path) {
11835            return Ok(());
11836        }
11837        lists[idx].push(scale_path);
11838        self.write_dimension_list(did, &lists)?;
11839
11840        let mut entries = self.reference_list(dsid)?;
11841        entries.push((data_path, idx as u32));
11842        self.write_reference_list(dsid, &entries)?;
11843
11844        if !self.is_dimension_scale(dsid)? {
11845            self.set_dimension_scale(dsid, None)?;
11846        }
11847        Ok(())
11848    }
11849
11850    /// The registry name of live dataset `index`, or why there is none.
11851    fn dataset_name(&self, index: usize) -> IoResult<String> {
11852        let count = self.dataset_count();
11853        if index >= count {
11854            return Err(crate::io::IoError::InvalidState(format!(
11855                "dataset index {index} out of range (have {count})"
11856            )));
11857        }
11858        let ds = self.ds(index);
11859        let m = ds.lock();
11860        if m.deleted {
11861            return Err(crate::io::IoError::NotFound(format!(
11862                "dataset '{}' has been deleted",
11863                m.name
11864            )));
11865        }
11866        Ok(m.name.clone())
11867    }
11868
11869    /// The stored attribute `name` of dataset `index`, without marking the
11870    /// header dirty the way [`with_attr_list`](Self::with_attr_list) must.
11871    fn dataset_attribute(&self, index: usize, name: &str) -> IoResult<Option<AttributeEntry>> {
11872        self.dataset_name(index)?;
11873        Ok(self
11874            .ds(index)
11875            .lock()
11876            .attributes
11877            .iter()
11878            .find(|a| a.name() == name)
11879            .cloned())
11880    }
11881
11882    /// Write the scalar fixed-length string attribute `name` = `value` on
11883    /// dataset `index` — `H5LT_set_attribute_string`: the string is stored
11884    /// null-terminated in `strlen + 1` bytes, and an attribute of that name
11885    /// is deleted first rather than written over.
11886    fn set_fixed_string_attribute(&self, index: usize, name: &str, value: &str) -> IoResult<()> {
11887        if value.as_bytes().contains(&0) {
11888            return Err(crate::io::IoError::InvalidState(format!(
11889                "attribute '{name}' value holds an interior NUL"
11890            )));
11891        }
11892        let size = u32::try_from(value.len() + 1).map_err(|_| {
11893            crate::io::IoError::InvalidState(format!(
11894                "attribute '{name}' value of {} bytes exceeds the fixed-string width field",
11895                value.len()
11896            ))
11897        })?;
11898        let mut data = value.as_bytes().to_vec();
11899        data.push(0);
11900        let attr =
11901            AttributeMessage::scalar_numeric(name, DatatypeMessage::fixed_string(size), data);
11902        let target = AttrTarget::Dataset(index);
11903        self.evict_attr(target, name)?;
11904        self.insert_attribute(target, attr, Created)
11905    }
11906
11907    /// The `CLASS` attribute of dataset `index`, read the way `H5DS` reads
11908    /// it: as a C string, up to the first NUL.
11909    fn class_attribute(&self, index: usize) -> IoResult<Option<ClassAttr>> {
11910        use crate::format::global_heap::decode_vlen_reference;
11911
11912        let Some(entry) = self.dataset_attribute(index, "CLASS")? else {
11913            return Ok(None);
11914        };
11915        let msg = entry.decoded().map_err(|reason| {
11916            crate::io::IoError::InvalidState(format!(
11917                "CLASS attribute of '{}' cannot be decoded: {reason}",
11918                self.ds(index).lock().name
11919            ))
11920        })?;
11921        Ok(Some(match &msg.datatype {
11922            DatatypeMessage::FixedString { size, padding, .. } => {
11923                let avail = (*size as usize).min(msg.data.len());
11924                ClassAttr::Fixed {
11925                    size: *size,
11926                    null_terminated: *padding == 0,
11927                    text: c_string(&msg.data[..avail]),
11928                }
11929            }
11930            DatatypeMessage::VarLenString { .. } => {
11931                let (_, addr, obj_idx) = decode_vlen_reference(&msg.data, &self.ctx)?;
11932                let bytes = if addr == 0 || addr == UNDEF_ADDR {
11933                    Vec::new()
11934                } else {
11935                    let obj_idx = u16::try_from(obj_idx).map_err(|_| {
11936                        crate::io::IoError::InvalidState(format!(
11937                            "global heap object index {obj_idx} does not fit the 16-bit on-disk \
11938                             field"
11939                        ))
11940                    })?;
11941                    self.read_heap_object(addr, obj_idx)?
11942                };
11943                ClassAttr::VarLen(c_string(&bytes))
11944            }
11945            _ => ClassAttr::NotString,
11946        }))
11947    }
11948
11949    /// `H5DSis_scale`: a `CLASS` that is a string saying `DIMENSION_SCALE` —
11950    /// and, for a fixed-length string, null-terminated and exactly 16 bytes
11951    /// wide, the width the spec gives the attribute.
11952    fn is_dimension_scale(&self, index: usize) -> IoResult<bool> {
11953        Ok(match self.class_attribute(index)? {
11954            None | Some(ClassAttr::NotString) => false,
11955            Some(ClassAttr::Fixed {
11956                size,
11957                null_terminated,
11958                text,
11959            }) => null_terminated && size == 16 && text == DIMENSION_SCALE_CLASS,
11960            Some(ClassAttr::VarLen(text)) => text == DIMENSION_SCALE_CLASS,
11961        })
11962    }
11963
11964    /// `H5DS_is_reserved`: a `CLASS` naming an image, palette or table — the
11965    /// datasets the other high-level APIs own. A `CLASS` that is not a string
11966    /// is an error here, where [`is_dimension_scale`](Self::is_dimension_scale)
11967    /// reads it as "not a scale", because that is how upstream splits them.
11968    fn has_reserved_class(&self, index: usize) -> IoResult<bool> {
11969        Ok(match self.class_attribute(index)? {
11970            None => false,
11971            Some(ClassAttr::NotString) => {
11972                return Err(crate::io::IoError::InvalidState(format!(
11973                    "CLASS attribute of '{}' is not a string",
11974                    self.ds(index).lock().name
11975                )))
11976            }
11977            Some(ClassAttr::Fixed { text, .. }) | Some(ClassAttr::VarLen(text)) => {
11978                matches!(text.as_str(), "IMAGE" | "PALETTE" | "TABLE")
11979            }
11980        })
11981    }
11982
11983    /// The bytes of object `index` in the global heap collection at
11984    /// `collection` — `H5HG_read`.
11985    fn read_heap_object(&self, collection: u64, index: u16) -> IoResult<Vec<u8>> {
11986        use crate::format::global_heap::GlobalHeapCollection;
11987
11988        let mut image = self.handle.read_at_most(collection, 4096)?;
11989        let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
11990        if declared > image.len() {
11991            image = self.handle.read_at(collection, declared)?;
11992        }
11993        let (gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
11994        gcol.get_object(index).map(<[u8]>::to_vec).ok_or_else(|| {
11995            crate::io::IoError::InvalidState(format!(
11996                "global heap collection {collection:#x} has no object {index}"
11997            ))
11998        })
11999    }
12000
12001    /// The path of the object whose header is at `addr` in the file as it
12002    /// was opened — what an object reference read back from an append
12003    /// session's existing attributes names.
12004    fn path_of_header_address(&self, addr: u64) -> IoResult<String> {
12005        for ds in self.dataset_refs() {
12006            let m = ds.lock();
12007            if !m.deleted && m.obj_header_written_addr == Some(addr) {
12008                return Ok(m.name.clone());
12009            }
12010        }
12011        for grp in self.group_refs() {
12012            let g = grp.lock();
12013            if !g.deleted && g.obj_header_written_addr == Some(addr) {
12014                return Ok(g.name.clone());
12015            }
12016        }
12017        Err(crate::io::IoError::InvalidState(format!(
12018            "object reference to header {addr:#x} names no dataset or group of this file"
12019        )))
12020    }
12021
12022    /// The path a reference slot inside a global heap object names: the one
12023    /// registered for stamping when this session wrote the slot, else the
12024    /// one the address on disk resolves to.
12025    fn heap_reference_path(
12026        &self,
12027        collection: u64,
12028        index: u16,
12029        token_offset: usize,
12030        on_disk: &[u8],
12031    ) -> IoResult<String> {
12032        let registered = self
12033            .pending_heap_references
12034            .lock()
12035            .iter()
12036            .find(|p| {
12037                p.collection == collection && p.index == index && p.token_offset == token_offset
12038            })
12039            .map(|p| match &p.target {
12040                PendingHeapTarget::Dataset(path) | PendingHeapTarget::Object(path) => path.clone(),
12041            });
12042        if let Some(path) = registered {
12043            return Ok(path);
12044        }
12045        let mut raw = [0u8; 8];
12046        raw[..on_disk.len()].copy_from_slice(on_disk);
12047        self.path_of_header_address(u64::from_le_bytes(raw))
12048    }
12049
12050    /// Dataset `did`'s `DIMENSION_LIST` as the paths of the scales on each
12051    /// axis, or `None` when it has no such attribute.
12052    fn dimension_list(&self, did: usize) -> IoResult<Option<Vec<Vec<String>>>> {
12053        use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
12054
12055        let Some(entry) = self.dataset_attribute(did, DIMENSION_LIST)? else {
12056            return Ok(None);
12057        };
12058        let name = || self.ds(did).lock().name.clone();
12059        let msg = entry.decoded().map_err(|reason| {
12060            crate::io::IoError::InvalidState(format!(
12061                "DIMENSION_LIST of '{}' cannot be decoded: {reason}",
12062                name()
12063            ))
12064        })?;
12065        match &msg.datatype {
12066            DatatypeMessage::VarLenSequence { base }
12067                if matches!(
12068                    **base,
12069                    DatatypeMessage::Reference {
12070                        kind: ReferenceKind::Object1,
12071                        ..
12072                    }
12073                ) => {}
12074            other => {
12075                return Err(crate::io::IoError::InvalidState(format!(
12076                    "DIMENSION_LIST of '{}' is {other}; only a sequence of H5T_STD_REF_OBJ \
12077                     references is supported",
12078                    name()
12079                )))
12080            }
12081        }
12082        let sa = self.ctx.sizeof_addr as usize;
12083        let ref_size = vlen_reference_size(&self.ctx);
12084        let mut lists = Vec::new();
12085        for elem in msg.data.chunks_exact(ref_size) {
12086            let (seq_len, addr, obj_idx) = decode_vlen_reference(elem, &self.ctx)?;
12087            let seq_len = seq_len as usize;
12088            let mut paths = Vec::with_capacity(seq_len);
12089            if seq_len > 0 {
12090                let index = u16::try_from(obj_idx).map_err(|_| {
12091                    crate::io::IoError::InvalidState(format!(
12092                        "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
12093                    ))
12094                })?;
12095                let bytes = self.read_heap_object(addr, index)?;
12096                if bytes.len() < seq_len * sa {
12097                    return Err(crate::io::IoError::InvalidState(format!(
12098                        "DIMENSION_LIST of '{}' names {seq_len} scales in a {}-byte heap object",
12099                        name(),
12100                        bytes.len()
12101                    )));
12102                }
12103                for k in 0..seq_len {
12104                    paths.push(self.heap_reference_path(
12105                        addr,
12106                        index,
12107                        k * sa,
12108                        &bytes[k * sa..(k + 1) * sa],
12109                    )?);
12110                }
12111            }
12112            lists.push(paths);
12113        }
12114        Ok(Some(lists))
12115    }
12116
12117    /// Store `lists` — the scales attached to each axis — as dataset `did`'s
12118    /// `DIMENSION_LIST`, replacing the one it has.
12119    ///
12120    /// Each axis is one global heap object of `sizeof_addr` bytes per scale,
12121    /// zero until finalize stamps the scale's header address in through
12122    /// [`write_heap_reference_values`](Self::write_heap_reference_values);
12123    /// an axis with no scale is an empty heap object, as libhdf5's
12124    /// `H5VL__native_blob_put` stores an empty sequence. The attribute's
12125    /// value is the vlen reference to each object, final at write time.
12126    fn write_dimension_list(&self, did: usize, lists: &[Vec<String>]) -> IoResult<()> {
12127        use crate::format::global_heap::{
12128            encode_vlen_reference, vlen_reference_size, vlen_seq_len,
12129        };
12130
12131        let target = AttrTarget::Dataset(did);
12132        let origin = self.evict_attr(target, DIMENSION_LIST)?;
12133        let sa = self.ctx.sizeof_addr as usize;
12134        let blobs: Vec<Vec<u8>> = lists.iter().map(|l| vec![0u8; l.len() * sa]).collect();
12135        let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
12136        let placements = self.insert_vlen_objects(&items)?;
12137
12138        let mut data = Vec::with_capacity(lists.len() * vlen_reference_size(&self.ctx));
12139        let mut pending = self.pending_heap_references.lock();
12140        for (axis, &(collection, index)) in placements.iter().enumerate() {
12141            for (k, path) in lists[axis].iter().enumerate() {
12142                pending.push(PendingHeapReference {
12143                    collection,
12144                    index,
12145                    token_offset: k * sa,
12146                    target: PendingHeapTarget::Object(path.clone()),
12147                });
12148            }
12149            data.extend_from_slice(&encode_vlen_reference(
12150                vlen_seq_len(lists[axis].len())?,
12151                collection,
12152                u32::from(index),
12153                &self.ctx,
12154            ));
12155        }
12156        drop(pending);
12157
12158        let attr = AttributeMessage {
12159            name: DIMENSION_LIST.to_string(),
12160            datatype: DatatypeMessage::VarLenSequence {
12161                base: Box::new(DatatypeMessage::object_reference(&self.ctx)),
12162            },
12163            dataspace: DataspaceMessage::simple(&[lists.len() as u64]),
12164            data,
12165        };
12166        self.insert_attribute(target, attr, origin)
12167    }
12168
12169    /// Scale `dsid`'s `REFERENCE_LIST` as (dataset path, axis) pairs; empty
12170    /// when it has no such attribute.
12171    fn reference_list(&self, dsid: usize) -> IoResult<Vec<(String, u32)>> {
12172        let Some(entry) = self.dataset_attribute(dsid, REFERENCE_LIST)? else {
12173            return Ok(Vec::new());
12174        };
12175        let name = || self.ds(dsid).lock().name.clone();
12176        let msg = entry.decoded().map_err(|reason| {
12177            crate::io::IoError::InvalidState(format!(
12178                "REFERENCE_LIST of '{}' cannot be decoded: {reason}",
12179                name()
12180            ))
12181        })?;
12182        let unsupported = |why: String| {
12183            crate::io::IoError::InvalidState(format!(
12184                "REFERENCE_LIST of '{}' is {}; {why}",
12185                name(),
12186                msg.datatype
12187            ))
12188        };
12189        let DatatypeMessage::Compound { size, members } = &msg.datatype else {
12190            return Err(unsupported("a compound is required".into()));
12191        };
12192        let member = |m: &str| {
12193            members
12194                .iter()
12195                .find(|c| c.name == m)
12196                .ok_or_else(|| unsupported(format!("member '{m}' is missing")))
12197        };
12198        let dataset = member("dataset")?;
12199        let dimension = member("dimension")?;
12200        let sa = self.ctx.sizeof_addr as usize;
12201        if !matches!(
12202            dataset.datatype,
12203            DatatypeMessage::Reference {
12204                kind: ReferenceKind::Object1,
12205                ..
12206            }
12207        ) {
12208            return Err(unsupported(
12209                "only an H5T_STD_REF_OBJ 'dataset' member is supported".into(),
12210            ));
12211        }
12212        let DatatypeMessage::FixedPoint {
12213            size: 4,
12214            byte_order,
12215            ..
12216        } = dimension.datatype
12217        else {
12218            return Err(unsupported(
12219                "a 4-byte integer 'dimension' member is required".into(),
12220            ));
12221        };
12222        let stride = *size as usize;
12223        let registered: Option<Vec<String>> = self
12224            .attribute_references
12225            .lock()
12226            .iter()
12227            .find(|r| r.scope == AttrScope::Dataset(dsid) && r.name == REFERENCE_LIST)
12228            .map(|r| r.targets.clone());
12229        let mut entries = Vec::with_capacity(msg.data.len() / stride);
12230        for (i, elem) in msg.data.chunks_exact(stride).enumerate() {
12231            let at = |offset: u32, len: usize| {
12232                elem.get(offset as usize..offset as usize + len)
12233                    .ok_or_else(|| unsupported(format!("element {i} is too short for its members")))
12234            };
12235            let path = match &registered {
12236                Some(targets) => targets.get(i).cloned().ok_or_else(|| {
12237                    crate::io::IoError::InvalidState(format!(
12238                        "REFERENCE_LIST of '{}' entry {i} has no registered target",
12239                        name()
12240                    ))
12241                })?,
12242                None => {
12243                    let mut raw = [0u8; 8];
12244                    raw[..sa].copy_from_slice(at(dataset.offset, sa)?);
12245                    self.path_of_header_address(u64::from_le_bytes(raw))?
12246                }
12247            };
12248            let dim: [u8; 4] = at(dimension.offset, 4)?.try_into().expect("4 bytes");
12249            let dim = match byte_order {
12250                ByteOrder::LittleEndian => u32::from_le_bytes(dim),
12251                ByteOrder::BigEndian => u32::from_be_bytes(dim),
12252            };
12253            entries.push((path, dim));
12254        }
12255        Ok(entries)
12256    }
12257
12258    /// Store `entries` as scale `dsid`'s `REFERENCE_LIST`, replacing the one
12259    /// it has — deleted and created anew, as upstream does, so it takes a
12260    /// fresh creation index.
12261    ///
12262    /// The element is libhdf5's `ds_list_t` as it lands on disk: the
12263    /// reference at offset 0, `dimension` right after it, and the struct's
12264    /// trailing padding — 16 bytes over 8-byte addresses. The addresses are
12265    /// stamped at finalize through [`object_attributes`](Self::object_attributes)
12266    /// like any reference attribute's; the `dimension` fields are final here.
12267    fn write_reference_list(&self, dsid: usize, entries: &[(String, u32)]) -> IoResult<()> {
12268        use crate::format::messages::datatype::CompoundMember;
12269
12270        let target = AttrTarget::Dataset(dsid);
12271        self.evict_attr(target, REFERENCE_LIST)?;
12272        let sa = self.ctx.sizeof_addr as usize;
12273        let stride = sa + 8;
12274        let datatype = DatatypeMessage::compound(
12275            stride as u32,
12276            vec![
12277                CompoundMember {
12278                    name: "dataset".to_string(),
12279                    offset: 0,
12280                    datatype: DatatypeMessage::object_reference(&self.ctx),
12281                },
12282                CompoundMember {
12283                    name: "dimension".to_string(),
12284                    offset: sa as u32,
12285                    datatype: DatatypeMessage::u32_type(),
12286                },
12287            ],
12288        );
12289        let mut data = vec![0u8; entries.len() * stride];
12290        for (i, (_, dim)) in entries.iter().enumerate() {
12291            data[i * stride + sa..i * stride + sa + 4].copy_from_slice(&dim.to_le_bytes());
12292        }
12293        let attr = AttributeMessage::array_numeric(
12294            REFERENCE_LIST,
12295            datatype,
12296            &[entries.len() as u64],
12297            data,
12298        );
12299        self.insert_attribute(target, attr, Created)?;
12300        self.attribute_references
12301            .lock()
12302            .push(AttributeReferenceValue {
12303                scope: AttrScope::Dataset(dsid),
12304                name: REFERENCE_LIST.to_string(),
12305                targets: entries.iter().map(|(p, _)| p.clone()).collect(),
12306                stride,
12307            });
12308        Ok(())
12309    }
12310
12311    /// Take the attribute `name` off `target`'s list, releasing its heap
12312    /// objects. No-op when absent. Refused under SWMR — see
12313    /// [`set_attribute`](Self::set_attribute).
12314    ///
12315    /// What it answers is what the insert that follows it must be told: an
12316    /// attribute that was there is being rewritten and keeps its creation
12317    /// index, and one that was not is created.
12318    fn evict_attr(&self, target: AttrTarget<'_>, name: &str) -> IoResult<AttrOrigin> {
12319        if self.swmr_active {
12320            return Err(swmr_attr_error(name));
12321        }
12322        self.forget_attribute_reference(self.attr_scope(target)?, name);
12323        let old = self.with_attr_list(target, |attrs| {
12324            attrs
12325                .iter()
12326                .position(|a| a.name() == name)
12327                .map(|pos| attrs.remove(pos))
12328        })?;
12329        match old {
12330            Some(old) => {
12331                let origin = Rewritten(old.creation_index());
12332                self.release_attr_vlen(&old)?;
12333                Ok(origin)
12334            }
12335            None => Ok(Created),
12336        }
12337    }
12338
12339    /// Release the global-heap objects a superseded attribute owned.
12340    /// Recognizes top-level vlen datatypes only: a *compound* attribute
12341    /// with vlen members — which this crate cannot write, only a foreign
12342    /// file can carry — keeps its members' heap objects when replaced or
12343    /// deleted, the storage cost the foreign writer accepted. Every other
12344    /// class stores its value inline in the message. Per-object removal
12345    /// keeps collections shared with other refs (libhdf5-written files)
12346    /// intact.
12347    fn release_attr_vlen(&self, old: &AttributeEntry) -> IoResult<()> {
12348        use crate::format::messages::datatype::DatatypeMessage;
12349        // An attribute whose message this crate could not decode keeps
12350        // whatever heap space it references: releasing objects named by bytes
12351        // we cannot interpret would free storage that is still live.
12352        let Some(old) = old.readable() else {
12353            return Ok(());
12354        };
12355        if matches!(
12356            old.datatype,
12357            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
12358        ) {
12359            self.release_vlen_references(&old.data)?;
12360        }
12361        Ok(())
12362    }
12363
12364    /// Run `f` on the attribute list `target` names — the accessor every
12365    /// attribute mutation shares.
12366    fn with_attr_list<R>(
12367        &self,
12368        target: AttrTarget<'_>,
12369        f: impl FnOnce(&mut Vec<AttributeEntry>) -> R,
12370    ) -> IoResult<R> {
12371        match target {
12372            AttrTarget::Root => Ok(f(&mut self.root_attributes.lock())),
12373            AttrTarget::Group(path) => {
12374                let path = self.canonical_group_path(path);
12375                for grp in self.group_refs() {
12376                    let mut g = grp.lock();
12377                    if g.name == path && !g.deleted {
12378                        return Ok(f(&mut g.attributes));
12379                    }
12380                }
12381                Err(crate::io::IoError::NotFound(format!(
12382                    "group '{path}' not found"
12383                )))
12384            }
12385            AttrTarget::Dataset(index) => {
12386                let count = self.dataset_count();
12387                if index >= count {
12388                    return Err(crate::io::IoError::InvalidState(format!(
12389                        "dataset index {index} out of range (have {count})"
12390                    )));
12391                }
12392                let ds = self.ds(index);
12393                let mut m = ds.lock();
12394                // Every caller of this mutates the list, and a reopened
12395                // dataset's header is rewritten only when it is marked stale.
12396                m.header_dirty = true;
12397                Ok(f(&mut m.attributes))
12398            }
12399        }
12400    }
12401
12402    /// Store each of `items` as a global heap object and return its
12403    /// placement `(collection address, object index)`, in input order —
12404    /// the writer side of libhdf5's `H5HG_insert`.
12405    ///
12406    /// Placement follows libhdf5: a collection from the CWFS list takes an
12407    /// item when its free space holds the object *and* a residual
12408    /// free-space marker header (`encode_at_size` always emits the
12409    /// marker); what no listed collection can take goes into a fresh
12410    /// collection, spilling into another at the 65535-object index cap.
12411    /// One batch may therefore span several collections — invisible to
12412    /// readers, which resolve each reference's own collection address. An
12413    /// empty batch allocates nothing: an empty collection still encodes
12414    /// to the 4096-byte `H5HG_MINALLOC` minimum, a block nothing would
12415    /// reference. libhdf5 additionally tries to extend a nearly-full
12416    /// collection's block in place (`H5MF_try_extend`); this writer does
12417    /// not — an oversized item always starts a fresh collection.
12418    ///
12419    /// The `cwfs` lock is held across every read-modify-rewrite of a
12420    /// listed collection block: it serializes concurrent inserts (two
12421    /// datasets' writers can pack the same block) and inserts against
12422    /// [`release_vlen_references`](Self::release_vlen_references), which
12423    /// rewrites the same blocks when objects are freed.
12424    ///
12425    /// Under SWMR the CWFS list is neither consulted nor updated and every
12426    /// batch gets fresh collections: packing rewrites a block a streaming
12427    /// reader may be mid-walk on — the same reason `place_chunk` keeps a
12428    /// relocated chunk's old block.
12429    fn insert_vlen_objects(&self, items: &[&[u8]]) -> IoResult<Vec<(u64, u16)>> {
12430        use crate::format::global_heap::{GlobalHeapCollection, GlobalHeapObject};
12431
12432        if items.is_empty() {
12433            return Ok(Vec::new());
12434        }
12435        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
12436        let mut placements = Vec::with_capacity(items.len());
12437        let mut i = 0;
12438
12439        // Pack into listed collections while one can take the next item.
12440        if !self.swmr_active {
12441            let mut cwfs = self.cwfs.lock();
12442            while i < items.len() {
12443                let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
12444                let Some(pos) = cwfs.iter().position(|e| e.free >= need + objhdr) else {
12445                    // Second pass of libhdf5's H5F_cwfs_find_free_heap: no
12446                    // listed collection has room, so try to grow one in
12447                    // place before falling back to a fresh collection.
12448                    if self.extend_listed_collection(&mut cwfs, need + objhdr)? {
12449                        continue;
12450                    }
12451                    break;
12452                };
12453                let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
12454                let image = self.handle.read_at(addr, size)?;
12455                let (mut gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
12456                // The disk is the truth for free space; the entry is a hint.
12457                let Some(mut free) = gcol.free_space_at(&self.ctx, size) else {
12458                    cwfs.remove(pos);
12459                    continue;
12460                };
12461                let mut next_idx = gcol.max_index();
12462                let mut took = false;
12463                while i < items.len() && next_idx < u16::MAX {
12464                    let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
12465                    if free < need + objhdr {
12466                        break;
12467                    }
12468                    next_idx += 1;
12469                    gcol.objects.push(GlobalHeapObject {
12470                        index: next_idx,
12471                        ref_count: 0,
12472                        data: items[i].to_vec(),
12473                    });
12474                    placements.push((addr, next_idx));
12475                    free -= need;
12476                    took = true;
12477                    i += 1;
12478                }
12479                if took {
12480                    let rewritten = gcol.encode_at_size(&self.ctx, size)?;
12481                    self.handle.write_at(addr, &rewritten)?;
12482                    // Correct the entry to the measured free space and move
12483                    // it to the front — libhdf5 keeps `cwfs` in
12484                    // most-recently-used order.
12485                    let mut e = cwfs.remove(pos);
12486                    e.free = free;
12487                    cwfs.insert(0, e);
12488                } else if next_idx == u16::MAX {
12489                    // At the index cap nothing can be inserted no matter the
12490                    // free space; drop the entry or the scan re-picks it
12491                    // forever. (A removal can lower the top index again, and
12492                    // the release side re-lists the collection then.)
12493                    cwfs.remove(pos);
12494                } else {
12495                    // The hint overstated the block's free space — shrink it
12496                    // to the measured value so the scan moves on.
12497                    cwfs[pos].free = free;
12498                }
12499            }
12500        }
12501
12502        // What remains goes into fresh collections.
12503        while i < items.len() {
12504            let mut gcol = GlobalHeapCollection::new();
12505            // Objects are pushed with a running index: `add_object` rescans
12506            // for the max index per call, O(n²) across a spill-sized batch.
12507            let mut next_idx: u16 = 0;
12508            while i < items.len() && next_idx < u16::MAX {
12509                next_idx += 1;
12510                gcol.objects.push(GlobalHeapObject {
12511                    index: next_idx,
12512                    ref_count: 0,
12513                    data: items[i].to_vec(),
12514                });
12515                i += 1;
12516            }
12517            let encoded = gcol.encode(&self.ctx);
12518            let addr = self
12519                .allocator
12520                .allocate(encoded.len() as u64, FreeSpaceClass::RawData);
12521            self.handle.write_at(addr, &encoded)?;
12522            for idx in 1..=next_idx {
12523                placements.push((addr, idx));
12524            }
12525            // List the block's leftover free space for later inserts — the
12526            // minimum-size padding of a small batch is most of 4096 bytes.
12527            // Below two object headers not even an empty object fits.
12528            if !self.swmr_active {
12529                if let Some(free) = gcol.free_space_at(&self.ctx, encoded.len()) {
12530                    if free >= 2 * objhdr {
12531                        cwfs_note(&mut self.cwfs.lock(), addr, encoded.len(), free);
12532                    }
12533                }
12534            }
12535        }
12536        Ok(placements)
12537    }
12538
12539    /// Try to extend one listed collection in place so it can take an
12540    /// object needing `want` bytes of free space — the second pass of
12541    /// libhdf5's `H5F_cwfs_find_free_heap`: grow the file allocation
12542    /// ([`FileAllocator::try_extend`], mirroring `H5MF_try_extend`) and then
12543    /// the collection itself (`H5HG_extend`: a larger declared size and a
12544    /// free-space marker covering the new tail — here by re-encoding at the
12545    /// grown size, which writes exactly those two things).
12546    ///
12547    /// Extension size is `max(collection_size, shortfall)` — at least a
12548    /// doubling — capped so the result stays within [`GCOL_MAX_SIZE`], both
12549    /// as upstream computes them. On success the grown entry moves to the
12550    /// front of the list and the caller's scan re-picks it; the free-space
12551    /// measurement is taken from the block on disk, not the list's hint, so
12552    /// the rewrite and the entry agree.
12553    ///
12554    /// Caller holds the `cwfs` lock (it passes the guarded list), which is
12555    /// what serializes this read-modify-rewrite against concurrent inserts
12556    /// and releases.
12557    fn extend_listed_collection(&self, cwfs: &mut Vec<CwfsEntry>, want: usize) -> IoResult<bool> {
12558        use crate::format::global_heap::{GlobalHeapCollection, GCOL_MAX_SIZE};
12559
12560        let mut pos = 0;
12561        while pos < cwfs.len() {
12562            let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
12563            let image = self.handle.read_at(addr, size)?;
12564            let (gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
12565            // The disk is the truth for free space; the entry is a hint.
12566            let Some(free) = gcol.free_space_at(&self.ctx, size) else {
12567                cwfs.remove(pos);
12568                continue;
12569            };
12570            // A hint can understate the block (upstream's FREE_SIZE is its
12571            // in-memory truth and cannot): if the block already has room,
12572            // correct the hint instead of doubling the collection.
12573            if free >= want {
12574                cwfs[pos].free = free;
12575                return Ok(true);
12576            }
12577            let new_need = size.max(want.saturating_sub(free));
12578            if size + new_need > GCOL_MAX_SIZE
12579                || !self.allocator.try_extend(
12580                    addr,
12581                    size as u64,
12582                    new_need as u64,
12583                    FreeSpaceClass::RawData,
12584                )
12585            {
12586                pos += 1;
12587                continue;
12588            }
12589            let new_size = size + new_need;
12590            let rewritten = gcol.encode_at_size(&self.ctx, new_size)?;
12591            self.handle.write_at(addr, &rewritten)?;
12592            let mut e = cwfs.remove(pos);
12593            e.size = new_size;
12594            e.free = free + new_need;
12595            cwfs.insert(0, e);
12596            return Ok(true);
12597        }
12598        Ok(false)
12599    }
12600
12601    /// Create a variable-length string dataset and write string data.
12602    ///
12603    /// Stores strings in the global heap. The dataset raw data consists of
12604    /// vlen references (collection_addr + object_index pairs).
12605    ///
12606    /// `charset` is the datatype's declared character set (0 = ASCII,
12607    /// 1 = UTF-8); the strings are checked against it before anything is
12608    /// written, so the type never misdescribes the bytes under it.
12609    pub fn create_vlen_string_dataset(
12610        &self,
12611        name: &str,
12612        strings: &[&str],
12613        charset: u8,
12614    ) -> IoResult<usize> {
12615        use crate::format::global_heap::encode_vlen_reference;
12616        use crate::format::messages::datatype::DatatypeMessage;
12617
12618        ensure_vlen_charset(charset, strings)?;
12619
12620        let create = self.begin_create(name)?;
12621        let name = create.name.as_str();
12622        let num_strings = strings.len() as u64;
12623
12624        // Store the strings as heap objects; a batch that fits an earlier
12625        // collection's free space shares its block.
12626        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12627        let placements = self.insert_vlen_objects(&items)?;
12628
12629        // Build raw data: vlen references
12630        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12631        let data_size = (num_strings as usize) * ref_size;
12632        let mut raw_data = Vec::with_capacity(data_size);
12633        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12634            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12635            raw_data.extend_from_slice(&encode_vlen_reference(
12636                seq_len,
12637                gcol_addr,
12638                obj_idx as u32,
12639                &self.ctx,
12640            ));
12641        }
12642
12643        // Allocate and write raw data
12644        let data_addr = self
12645            .allocator
12646            .allocate(data_size as u64, FreeSpaceClass::RawData);
12647        self.handle.write_at(data_addr, &raw_data)?;
12648
12649        // Create the dataset with vlen string datatype
12650        let datatype = DatatypeMessage::VarLenString {
12651            padding: 0,
12652            charset,
12653        };
12654        let dataspace =
12655            crate::format::messages::dataspace::DataspaceMessage::simple(&[num_strings]);
12656
12657        let idx = self.push_dataset(
12658            &create,
12659            DatasetInfo {
12660                name: name.to_string(),
12661                datatype,
12662                committed_type: None,
12663                external: None,
12664                virtual_storage: None,
12665                dataspace,
12666                read_format: None,
12667                obj_header_addr: 0,
12668                data_addr,
12669                data_size: data_size as u64,
12670                compact: None,
12671                attributes: Vec::new(),
12672                obj_header_written_addr: None,
12673                obj_header_blocks: Vec::new(),
12674                filter_pipeline: None,
12675                deleted: false,
12676                extent_dirty: false,
12677                header_dirty: false,
12678                nlink_written: 1,
12679                creation_seq: self.take_creation_seq(),
12680                track_attr_order: self.track_order.attrs,
12681                fill_value: None,
12682                fill_time: FILL_TIME_IFSET,
12683                layout_version: 4,
12684                times: self.created_object_times(),
12685                chunked: None,
12686                fixed_array: None,
12687                implicit: None,
12688                single_chunk: None,
12689                btree_v1: None,
12690                btree_v2: None,
12691                append: None,
12692            },
12693        );
12694
12695        Ok(idx)
12696    }
12697
12698    /// Create a 1-D variable-length byte-array dataset.
12699    ///
12700    /// The `u8` case of [`create_vlen_sequence_dataset`], where an item's
12701    /// byte image and its element count are the same number.
12702    ///
12703    /// [`create_vlen_sequence_dataset`]: Self::create_vlen_sequence_dataset
12704    ///
12705    /// Superseded in production by [`write_vlen_numeric`](crate::H5File::write_vlen_numeric)
12706    /// (`H5Group::write_vlen_bytes` routes through it, not through here);
12707    /// kept as a direct entry point for this crate's own white-box tests.
12708    #[cfg(test)]
12709    pub fn create_vlen_bytes_dataset(&self, name: &str, items: &[&[u8]]) -> IoResult<usize> {
12710        use crate::format::messages::datatype::DatatypeMessage;
12711
12712        self.create_vlen_sequence_dataset(name, DatatypeMessage::u8_type(), items)
12713    }
12714
12715    /// Create a 1-D variable-length sequence dataset over `base`.
12716    ///
12717    /// Each item is the encoded image of one sequence — `n * base.element_size()`
12718    /// bytes in the base type's own byte order — and is stored as a global-heap
12719    /// object; the dataset holds one vlen reference per item, the same on-disk
12720    /// shape a vlen string dataset has. The `H5T_VLEN` length field counts base
12721    /// elements rather than bytes, so an image whose length is not a whole
12722    /// number of elements is refused here rather than stored under a length
12723    /// that misreads it.
12724    pub fn create_vlen_sequence_dataset(
12725        &self,
12726        name: &str,
12727        base: DatatypeMessage,
12728        items: &[&[u8]],
12729    ) -> IoResult<usize> {
12730        use crate::format::global_heap::encode_vlen_reference;
12731        use crate::format::messages::datatype::DatatypeMessage;
12732
12733        let elem_size = base.element_size() as usize;
12734        if elem_size == 0 {
12735            return Err(crate::io::IoError::InvalidState(format!(
12736                "vlen base datatype {base} has no element size"
12737            )));
12738        }
12739        for (i, item) in items.iter().enumerate() {
12740            if !item.len().is_multiple_of(elem_size) {
12741                return Err(crate::io::IoError::InvalidState(format!(
12742                    "sequence {i} is {} bytes, not a whole number of {elem_size}-byte elements",
12743                    item.len()
12744                )));
12745            }
12746        }
12747
12748        let create = self.begin_create(name)?;
12749        let name = create.name.as_str();
12750        let num_items = items.len() as u64;
12751
12752        // Store the sequence images as heap objects, sharing collection
12753        // blocks as `create_vlen_string_dataset` does.
12754        let placements = self.insert_vlen_objects(items)?;
12755
12756        // Build raw data: one vlen reference per item.
12757        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12758        let data_size = (num_items as usize) * ref_size;
12759        let mut raw_data = Vec::with_capacity(data_size);
12760        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12761            let seq_len = crate::format::global_heap::vlen_seq_len(items[i].len() / elem_size)?;
12762            raw_data.extend_from_slice(&encode_vlen_reference(
12763                seq_len,
12764                gcol_addr,
12765                obj_idx as u32,
12766                &self.ctx,
12767            ));
12768        }
12769
12770        // Allocate and write raw data.
12771        let data_addr = self
12772            .allocator
12773            .allocate(data_size as u64, FreeSpaceClass::RawData);
12774        self.handle.write_at(data_addr, &raw_data)?;
12775
12776        let datatype = DatatypeMessage::VarLenSequence {
12777            base: Box::new(base),
12778        };
12779        let dataspace = crate::format::messages::dataspace::DataspaceMessage::simple(&[num_items]);
12780
12781        let idx = self.push_dataset(
12782            &create,
12783            DatasetInfo {
12784                name: name.to_string(),
12785                datatype,
12786                committed_type: None,
12787                external: None,
12788                virtual_storage: None,
12789                dataspace,
12790                read_format: None,
12791                obj_header_addr: 0,
12792                data_addr,
12793                data_size: data_size as u64,
12794                compact: None,
12795                attributes: Vec::new(),
12796                obj_header_written_addr: None,
12797                obj_header_blocks: Vec::new(),
12798                filter_pipeline: None,
12799                deleted: false,
12800                extent_dirty: false,
12801                header_dirty: false,
12802                nlink_written: 1,
12803                creation_seq: self.take_creation_seq(),
12804                track_attr_order: self.track_order.attrs,
12805                fill_value: None,
12806                fill_time: FILL_TIME_IFSET,
12807                layout_version: 4,
12808                times: self.created_object_times(),
12809                chunked: None,
12810                fixed_array: None,
12811                implicit: None,
12812                single_chunk: None,
12813                btree_v1: None,
12814                btree_v2: None,
12815                append: None,
12816            },
12817        );
12818
12819        Ok(idx)
12820    }
12821
12822    /// Create a chunked, compressed variable-length string dataset.
12823    ///
12824    /// Strings are stored in the global heap (same as `create_vlen_string_dataset`),
12825    /// but the vlen references are stored in chunked layout with the given filter
12826    /// pipeline (e.g., deflate, zstd). `chunk_size` is the number of strings per chunk.
12827    pub fn create_vlen_string_dataset_compressed(
12828        &self,
12829        name: &str,
12830        strings: &[&str],
12831        chunk_size: usize,
12832        pipeline: FilterPipeline,
12833    ) -> IoResult<usize> {
12834        use crate::format::global_heap::encode_vlen_reference;
12835        use crate::format::messages::datatype::DatatypeMessage;
12836
12837        let create = self.begin_create(name)?;
12838        let name = create.name.as_str();
12839        let num_strings = strings.len() as u64;
12840        validate_chunk_geometry(&[num_strings], &[num_strings], &[chunk_size as u64])?;
12841
12842        // Store the strings as heap objects; the geometry validation above
12843        // must precede this so a refused call allocates nothing.
12844        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12845        let placements = self.insert_vlen_objects(&items)?;
12846
12847        // Build raw data: vlen references
12848        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12849        let data_size = (num_strings as usize) * ref_size;
12850        let mut raw_data = Vec::with_capacity(data_size);
12851        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12852            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12853            raw_data.extend_from_slice(&encode_vlen_reference(
12854                seq_len,
12855                gcol_addr,
12856                obj_idx as u32,
12857                &self.ctx,
12858            ));
12859        }
12860
12861        // Set up chunked compressed layout
12862        let datatype = DatatypeMessage::vlen_string_utf8();
12863        let element_size = datatype.element_size_ctx(&self.ctx) as u64;
12864        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12865        let dims: Vec<u64> = vec![num_strings];
12866        let max_dims: Vec<u64> = vec![num_strings];
12867        let chunk_bytes = chunk_size as u64 * element_size;
12868        let layout_version = self.chunk_layout_version(true, chunk_bytes);
12869        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
12870
12871        let earray_params = EarrayParams::default_params();
12872        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
12873        let nsblk_addrs = compute_nsblk_addrs(
12874            earray_params.idx_blk_elmts,
12875            earray_params.data_blk_min_elmts,
12876            earray_params.sup_blk_min_data_ptrs,
12877            earray_params.max_nelmts_bits,
12878        )?;
12879
12880        // Create filtered EA header
12881        let mut ea_header =
12882            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
12883        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
12884        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
12885        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
12886        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
12887        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
12888
12889        let hdr_encoded = ea_header.encode(&self.ctx);
12890        let ea_header_addr = self
12891            .allocator
12892            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
12893
12894        // Create filtered index block
12895        let filt_iblk = FilteredIndexBlock::new(
12896            ea_header_addr,
12897            earray_params.idx_blk_elmts,
12898            ndblk_addrs,
12899            nsblk_addrs,
12900        );
12901        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
12902        let ea_iblk_addr = self
12903            .allocator
12904            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
12905
12906        ea_header.idx_blk_addr = ea_iblk_addr;
12907
12908        let hdr_encoded = ea_header.encode(&self.ctx);
12909        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
12910        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
12911
12912        let dataspace = DataspaceMessage {
12913            // Chunked storage always requires at least one dimension, so
12914            // this is never Scalar or Null.
12915            class: DataspaceClass::Simple,
12916            dims: dims.to_vec(),
12917            max_dims: Some(max_dims.to_vec()),
12918        };
12919
12920        let ea_iblk = ExtensibleArrayIndexBlock::new(
12921            ea_header_addr,
12922            earray_params.idx_blk_elmts,
12923            ndblk_addrs,
12924            nsblk_addrs,
12925        );
12926
12927        let idx = self.push_dataset(
12928            &create,
12929            DatasetInfo {
12930                name: name.to_string(),
12931                datatype,
12932                committed_type: None,
12933                external: None,
12934                virtual_storage: None,
12935                dataspace,
12936                read_format: None,
12937                obj_header_addr: 0,
12938                data_addr: UNDEF_ADDR,
12939                data_size: 0,
12940                compact: None,
12941                attributes: Vec::new(),
12942                obj_header_written_addr: None,
12943                obj_header_blocks: Vec::new(),
12944                filter_pipeline: Some(pipeline),
12945                deleted: false,
12946                extent_dirty: false,
12947                header_dirty: false,
12948                nlink_written: 1,
12949                creation_seq: self.take_creation_seq(),
12950                track_attr_order: self.track_order.attrs,
12951                fill_value: None,
12952                fill_time: FILL_TIME_IFSET,
12953                layout_version,
12954                times: self.created_object_times(),
12955                fixed_array: None,
12956                implicit: None,
12957                single_chunk: None,
12958                btree_v1: None,
12959                btree_v2: None,
12960                chunked: Some(ChunkedDatasetInfo {
12961                    chunk_dims: chunk_dims.clone(),
12962                    earray_params,
12963                    ea_header_addr,
12964                    ea_iblk_addr,
12965                    ea_header,
12966                    ea_iblk,
12967                    chunks_written: 0,
12968                    filt_iblk: Some(filt_iblk),
12969                    chunk_size_len,
12970                }),
12971                append: None,
12972            },
12973        );
12974
12975        // Write chunks of vlen references with compression
12976        let chunk_byte_size = chunk_bytes as usize;
12977        let num_chunks = raw_data.len().div_ceil(chunk_byte_size);
12978        for chunk_i in 0..num_chunks {
12979            let start = chunk_i * chunk_byte_size;
12980            let end = (start + chunk_byte_size).min(raw_data.len());
12981            let chunk_data = if end - start < chunk_byte_size {
12982                // Pad last chunk to full size (vlen datasets carry no user
12983                // fill value, so this resolves to zero = null vlen reference).
12984                let mut padded = self.new_chunk_buffer(idx, chunk_byte_size);
12985                padded[..end - start].copy_from_slice(&raw_data[start..end]);
12986                padded
12987            } else {
12988                raw_data[start..end].to_vec()
12989            };
12990            self.write_chunk(idx, chunk_i as u64, &chunk_data)?;
12991        }
12992
12993        Ok(idx)
12994    }
12995
12996    /// Create an empty chunked vlen string dataset ready for incremental appends.
12997    ///
12998    /// The dataset starts with `dims = [0]` and `max_dims = [unlimited]`.
12999    /// Use `append_vlen_strings` to add data.
13000    pub fn create_appendable_vlen_string_dataset(
13001        &self,
13002        name: &str,
13003        chunk_size: usize,
13004        pipeline: Option<FilterPipeline>,
13005    ) -> IoResult<usize> {
13006        let datatype = DatatypeMessage::vlen_string_utf8();
13007        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
13008        let dims: Vec<u64> = vec![0];
13009        let max_dims: Vec<u64> = vec![u64::MAX];
13010
13011        if let Some(ref pl) = pipeline {
13012            self.create_chunked_dataset_with_pipeline(
13013                name,
13014                datatype,
13015                &dims,
13016                &max_dims,
13017                &chunk_dims,
13018                pl.clone(),
13019            )
13020        } else {
13021            self.create_chunked_dataset(name, datatype, &dims, &max_dims, &chunk_dims)
13022        }
13023    }
13024
13025    /// Append variable-length strings to an existing chunked vlen string dataset.
13026    ///
13027    /// Creates a new global heap collection for the strings, builds vlen
13028    /// references, and appends them as new chunks to the dataset.
13029    pub fn append_vlen_strings(&self, ds_index: usize, strings: &[&str]) -> IoResult<()> {
13030        use crate::format::global_heap::encode_vlen_reference;
13031        use crate::format::messages::datatype::DatatypeMessage;
13032
13033        if strings.is_empty() {
13034            return Ok(());
13035        }
13036
13037        // Whole-operation guard: buffer take, frame writes, re-buffer and
13038        // extend below are separate slot acquisitions that a concurrent
13039        // same-dataset append must not interleave with.
13040        let cell = self.ds(ds_index);
13041        let _op = cell.op.lock();
13042
13043        // The elements about to be written are vlen references; any other
13044        // element type would be overwritten with them as raw bytes.
13045        let charset = {
13046            let ds = self.ds(ds_index);
13047            let m = ds.lock();
13048            match m.datatype {
13049                DatatypeMessage::VarLenString { charset, .. } => charset,
13050                _ => {
13051                    return Err(crate::io::IoError::InvalidState(
13052                        "append_vlen_strings is only for variable-length string datasets".into(),
13053                    ))
13054                }
13055            }
13056        };
13057        ensure_vlen_charset(charset, strings)?;
13058
13059        // Every deterministic rejection must precede the heap write below:
13060        // a collection written for a batch the append then refuses (a
13061        // contiguous dataset, or a reopened dataset whose chunk index was
13062        // not reconstructed) is a 4096-byte orphan nothing references.
13063        let chunk_dims = self
13064            .dataset_chunk_dims(ds_index)
13065            .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?
13066            .to_vec();
13067        let dims = self.dataset_dims(ds_index).to_vec();
13068
13069        // Store the batch's strings as heap objects; a batch that fits an
13070        // earlier collection's free space shares its block.
13071        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
13072        let placements = self.insert_vlen_objects(&items)?;
13073
13074        // Build raw vlen reference bytes
13075        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
13076        let mut raw = Vec::with_capacity(strings.len() * ref_size);
13077        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13078            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
13079            raw.extend_from_slice(&encode_vlen_reference(
13080                seq_len,
13081                gcol_addr,
13082                obj_idx as u32,
13083                &self.ctx,
13084            ));
13085        }
13086
13087        let n_new_frames = strings.len();
13088        let current_dim0 = dims[0] as usize;
13089        let chunk_dim0 = chunk_dims[0] as usize;
13090        let frame_bytes = ref_size;
13091
13092        // Merge the buffer with the new frames when it is the dataset's tail;
13093        // a buffer left mid-extent (the extent moved past it) keeps its
13094        // recorded place — flush it and start fresh at the current end.
13095        let taken = { self.ds(ds_index).lock().append.take() };
13096        let (base_dim0, buffered_frames, mut combined) = match taken {
13097            Some(b) if b.base + b.frames == current_dim0 as u64 => {
13098                (b.base as usize, b.frames as usize, b.bytes)
13099            }
13100            Some(b) => {
13101                self.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
13102                (current_dim0, 0, Vec::new())
13103            }
13104            None => (current_dim0, 0, Vec::new()),
13105        };
13106        combined.extend_from_slice(&raw);
13107
13108        let total_frames = buffered_frames + n_new_frames;
13109
13110        // Rows up to the last chunk boundary are written now; the tail that
13111        // does not complete a chunk goes back in the buffer for the next
13112        // append (or the flush at close). The boundary can precede
13113        // `base_dim0` — a reopened file's flushed partial chunk leaves the
13114        // base mid-chunk — in which case everything is tail.
13115        let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
13116        let write_frames = last_boundary.saturating_sub(base_dim0);
13117        let tail_frames = total_frames - write_frames;
13118        if write_frames > 0 {
13119            self.write_append_frames(
13120                ds_index,
13121                base_dim0 as u64,
13122                write_frames as u64,
13123                &combined[..write_frames * frame_bytes],
13124            )?;
13125        }
13126        if tail_frames > 0 {
13127            let ds = self.ds(ds_index);
13128            let mut m = ds.lock();
13129            m.append = Some(AppendBuffer {
13130                base: (base_dim0 + write_frames) as u64,
13131                frames: tail_frames as u64,
13132                bytes: combined[write_frames * frame_bytes..].to_vec(),
13133            });
13134        }
13135
13136        // Extend dims
13137        let logical_dim0 = base_dim0 + total_frames;
13138        let mut new_dims = dims;
13139        new_dims[0] = logical_dim0 as u64;
13140        self.extend_dataset_inner(ds_index, &new_dims)?;
13141
13142        Ok(())
13143    }
13144
13145    /// Replace elements `start .. start + strings.len()` of a 1-D
13146    /// variable-length string dataset, leaving its extent and every other
13147    /// element alone.
13148    ///
13149    /// The replacements go into the global heap and only the vlen
13150    /// references of the named elements are rewritten, so the cost is the
13151    /// new strings plus the chunks those references live in — not the column.
13152    /// The objects the old references pointed at are freed *before* the
13153    /// replacement is allocated, so repeated updates reuse space instead of
13154    /// growing the file — including across close/reopen cycles, where the
13155    /// in-memory free list starts empty and only this free-first order lets
13156    /// the session reuse the block it just released. This is what libhdf5
13157    /// does: `H5T__vlen_disk_write` deletes the reference it read into the
13158    /// conversion background buffer before storing the new one.
13159    ///
13160    /// Elements the append buffer still holds are flushed to their chunks
13161    /// first, so the whole range is on disk and one write path covers it.
13162    pub fn write_vlen_strings_slice(
13163        &self,
13164        ds_index: usize,
13165        start: u64,
13166        strings: &[&str],
13167    ) -> IoResult<()> {
13168        use crate::format::global_heap::{encode_vlen_reference, vlen_reference_size};
13169        use crate::format::messages::datatype::DatatypeMessage;
13170
13171        // An empty batch is a no-op: nothing to replace, nothing to free.
13172        if strings.is_empty() {
13173            return Ok(());
13174        }
13175
13176        // Whole-operation guard: the flush, the old-reference reads and the
13177        // slice write below must not interleave with a concurrent
13178        // same-dataset operation.
13179        let cell = self.ds(ds_index);
13180        let _op = cell.op.lock();
13181
13182        // Snapshot what the write needs, then drop the guard: `write_slice`
13183        // below re-locks the same slot.
13184        let (charset, dims, writable) = {
13185            let ds = self.ds(ds_index);
13186            let m = ds.lock();
13187            let charset = match m.datatype {
13188                DatatypeMessage::VarLenString { charset, .. } => charset,
13189                _ => {
13190                    return Err(crate::io::IoError::InvalidState(
13191                        "write_vlen_strings_slice is only for variable-length string datasets"
13192                            .into(),
13193                    ))
13194                }
13195            };
13196            let writable = if m.is_chunked() {
13197                Ok(())
13198            } else {
13199                match m.contiguous_target() {
13200                    Some(ContiguousTarget::Virtual) => Err(virtual_write_refused()),
13201                    Some(_) => Ok(()),
13202                    None => Err(crate::io::IoError::InvalidState(
13203                        "dataset has no data allocated".into(),
13204                    )),
13205                }
13206            };
13207            (charset, m.dataspace.dims.clone(), writable)
13208        };
13209
13210        // `write_slice_inner` rejects a dataset with neither chunk machinery
13211        // nor allocated data (a reopened dataset whose index was not
13212        // reconstructed), and refuses a virtual one outright — those
13213        // rejections must come before the heap write below, or every failed
13214        // call orphans a 4096-byte collection.
13215        writable?;
13216
13217        if dims.len() != 1 {
13218            return Err(crate::io::IoError::InvalidState(format!(
13219                "write_vlen_strings_slice is only for 1-dimension datasets, this one has {}",
13220                dims.len()
13221            )));
13222        }
13223        let end = start + strings.len() as u64;
13224        if end > dims[0] {
13225            return Err(crate::io::IoError::InvalidState(format!(
13226                "elements {start}..{end} are outside the dataset's {} elements",
13227                dims[0]
13228            )));
13229        }
13230        ensure_vlen_charset(charset, strings)?;
13231
13232        let ref_size = vlen_reference_size(&self.ctx);
13233
13234        // Elements the append buffer holds are not in the chunks yet: hand
13235        // them to the chunks first so the whole range is on disk and the one
13236        // write path below covers it.
13237        self.flush_append_buffer_if_intersecting(ds_index, start, end)?;
13238
13239        // The on-disk references about to be overwritten, read before anything
13240        // moves. libhdf5 reads the same bytes into the conversion background
13241        // buffer (`H5D__scatgath_write` gathers the file's current elements
13242        // when `need_bkg` is set) and hands them to `H5T__vlen_disk_write`,
13243        // which deletes them before storing the new reference.
13244        let superseded = self.current_element_bytes(ds_index, start, end - start, ref_size)?;
13245
13246        // Free the superseded objects *before* allocating the replacement,
13247        // the order `H5T__vlen_disk_write` uses. The freed block satisfies
13248        // the allocation below within this same session, so a reopen-and-
13249        // replace loop keeps the file flat — no persisted free-space
13250        // information exists to carry it across sessions (issue #10). The
13251        // cost, shared with libhdf5: a failure between here and the ref
13252        // write below leaves the dataset's old references dangling.
13253        self.release_vlen_references(&superseded)?;
13254
13255        // The insert comes after the release above so the space the release
13256        // recovered — a freed block, or in-collection bytes the release just
13257        // listed in `cwfs` — can satisfy this batch.
13258        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
13259        let placements = self.insert_vlen_objects(&items)?;
13260
13261        let mut refs = Vec::with_capacity(strings.len() * ref_size);
13262        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13263            refs.extend_from_slice(&encode_vlen_reference(
13264                crate::format::global_heap::vlen_seq_len(strings[i].len())?,
13265                gcol_addr,
13266                obj_idx as u32,
13267                &self.ctx,
13268            ));
13269        }
13270
13271        self.write_slice_inner(ds_index, &[start], &[strings.len() as u64], &refs)?;
13272
13273        Ok(())
13274    }
13275
13276    /// The bytes elements `start .. start + count` of a 1-D dataset currently
13277    /// hold, whichever layout stores them.
13278    ///
13279    /// Elements no write has reached yet read as zeros — for a vlen dataset
13280    /// that is the nil reference, which names no heap object.
13281    fn current_element_bytes(
13282        &self,
13283        ds_index: usize,
13284        start: u64,
13285        count: u64,
13286        element_size: usize,
13287    ) -> IoResult<Vec<u8>> {
13288        let mut out = vec![0u8; count as usize * element_size];
13289        if count == 0 {
13290            return Ok(out);
13291        }
13292
13293        let (is_chunked, data_addr) = {
13294            let ds = self.ds(ds_index);
13295            let m = ds.lock();
13296            (m.is_chunked(), m.data_addr)
13297        };
13298
13299        if !is_chunked {
13300            if data_addr != UNDEF_ADDR {
13301                // `read_at_most`, not `read_at`: a contiguous dataset's block is
13302                // reserved when it is created, so the file can still be shorter
13303                // than the block until something writes it. What is missing has
13304                // never been written, which is the zeros above.
13305                let at = data_addr + start * element_size as u64;
13306                let got = self.handle.read_at_most(at, out.len())?;
13307                out[..got.len()].copy_from_slice(&got);
13308            }
13309            return Ok(out);
13310        }
13311
13312        let geo = self.chunk_geometry(ds_index)?;
13313        let per_chunk = geo.chunk_dims[0];
13314        // Only a corrupt or crafted file declares a zero-length chunk
13315        // dimension; the divisions below must reject it the way
13316        // `write_slice` does, not panic.
13317        if per_chunk == 0 {
13318            return Err(crate::io::IoError::InvalidState(
13319                "chunk shape has a zero-length dimension".into(),
13320            ));
13321        }
13322        let end = start + count;
13323        for c in (start / per_chunk)..=((end - 1) / per_chunk) {
13324            let origin = c * per_chunk;
13325            let lo = start.max(origin);
13326            let hi = end.min(origin + per_chunk);
13327            // A chunk with no block yet leaves this span as the zeros above.
13328            let Some(chunk) = self.read_chunk_at_coords(ds_index, &[c])? else {
13329                continue;
13330            };
13331            let src = ((lo - origin) as usize) * element_size;
13332            let dst = ((lo - start) as usize) * element_size;
13333            let len = ((hi - lo) as usize) * element_size;
13334            if src + len > chunk.len() {
13335                return Err(crate::io::IoError::InvalidState(format!(
13336                    "chunk {c} is {} bytes, too short for elements {lo}..{hi}",
13337                    chunk.len()
13338                )));
13339            }
13340            out[dst..dst + len].copy_from_slice(&chunk[src..src + len]);
13341        }
13342        Ok(out)
13343    }
13344
13345    /// Free the global heap objects `refs` names, so replacing a vlen element
13346    /// does not strand what it used to point at.
13347    ///
13348    /// Callers pass refs only for *top-level* vlen datatypes (the
13349    /// `collect_refs` / `is_vlen` decisions at the prune, delete and
13350    /// attribute-release sites all match `VarLenString`/`VarLenSequence`).
13351    /// A compound datatype with vlen members — writable only by a foreign
13352    /// library, never by this crate — keeps its members' heap objects when
13353    /// its storage is pruned, deleted or replaced.
13354    ///
13355    /// This is libhdf5's `H5HG_remove` reached through `H5T__vlen_disk_delete`:
13356    /// the object leaves its collection, the collection is rewritten at its
13357    /// existing size with the recovered bytes given to the free-space marker,
13358    /// and a collection that ends up empty returns its block to the allocator.
13359    /// A rewritten collection's recovered space is listed in `cwfs` for
13360    /// [`insert_vlen_objects`](Self::insert_vlen_objects) to pack into; a
13361    /// freed block leaves the list.
13362    /// A nil reference (address 0 or `UNDEF_ADDR`) names no object. The
13363    /// address decides, not the sequence length: this crate's writers store
13364    /// even the empty string as a real heap object, so a zero-length reference
13365    /// with a defined address still holds one that must be released. libhdf5
13366    /// diverges here against itself — `H5T__vlen_disk_delete` returns before
13367    /// `H5HG_remove` when the sequence length is zero, yet its write path
13368    /// (`H5VL__native_blob_put`) inserts a heap object even for an empty
13369    /// sequence, stranding it forever. The address rule frees those objects.
13370    ///
13371    /// Heap objects carry no reference count on this path, matching libhdf5:
13372    /// its vlen code never calls `H5HG_link` (only the virtual-dataset layer
13373    /// does). Releasing the same reference twice is absorbed by the
13374    /// missing-index check below, but a crafted file in which two elements
13375    /// share one heap object would lose it for the survivor when either is
13376    /// replaced — the same exposure the file has under libhdf5. This crate's
13377    /// writers never share: each element write inserts its own object.
13378    ///
13379    /// Under SWMR nothing is freed and no collection is rewritten: a reader may
13380    /// be following those references, the same reason `place_chunk` keeps a
13381    /// relocated chunk's old block.
13382    fn release_vlen_references(&self, refs: &[u8]) -> IoResult<()> {
13383        use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
13384
13385        let ref_size = vlen_reference_size(&self.ctx);
13386        if ref_size == 0 || refs.len() < ref_size {
13387            return Ok(());
13388        }
13389
13390        // Group by collection so one holding several replaced objects is read,
13391        // rewritten and judged empty exactly once.
13392        let mut per_collection: std::collections::BTreeMap<u64, Vec<u16>> = Default::default();
13393        for r in refs.chunks_exact(ref_size) {
13394            let (_seq_len, addr, obj_idx) = decode_vlen_reference(r, &self.ctx)?;
13395            if addr == 0 || addr == UNDEF_ADDR {
13396                continue;
13397            }
13398            let Ok(idx) = u16::try_from(obj_idx) else {
13399                return Err(crate::io::IoError::InvalidState(format!(
13400                    "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
13401                )));
13402            };
13403            per_collection.entry(addr).or_default().push(idx);
13404        }
13405        self.remove_heap_objects(per_collection)
13406    }
13407
13408    /// Remove global heap objects — `H5HG_remove` — given the object indices
13409    /// grouped by the collection they live in.
13410    ///
13411    /// The single owner of heap-object removal: the vlen release path above
13412    /// reaches it with the objects a replaced element used to name, and
13413    /// [`release_dataset_storage`](Self::release_dataset_storage) with the
13414    /// one mapping-list object a deleted virtual dataset owned, which is what
13415    /// `H5D__virtual_delete` frees the same way.
13416    fn remove_heap_objects(
13417        &self,
13418        per_collection: std::collections::BTreeMap<u64, Vec<u16>>,
13419    ) -> IoResult<()> {
13420        use crate::format::global_heap::GlobalHeapCollection;
13421
13422        if self.swmr_active {
13423            return Ok(());
13424        }
13425
13426        // An object on its way out can hold no stamp: a reference this
13427        // session wrote into it would otherwise be stamped into whatever a
13428        // later insert puts at the same index. Pruned here, by the one owner
13429        // of removal, so no release path — attribute replacement, element
13430        // rewrite, dataset deletion — can leave one behind.
13431        self.pending_heap_references.lock().retain(|p| {
13432            !per_collection
13433                .get(&p.collection)
13434                .is_some_and(|indices| indices.contains(&p.index))
13435        });
13436
13437        // The `cwfs` lock is held across the sweep: it serializes these
13438        // collection-block rewrites (and frees) against
13439        // `insert_vlen_objects`, which may be packing new objects into the
13440        // same blocks.
13441        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
13442        let mut cwfs = self.cwfs.lock();
13443        for (addr, indices) in per_collection {
13444            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
13445            // exactly that, so one read usually covers the whole image; only
13446            // an oversized collection needs a second read at its declared size.
13447            let mut image = self.handle.read_at_most(addr, 4096)?;
13448            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
13449            if declared > image.len() {
13450                image = self.handle.read_at(addr, declared)?;
13451            }
13452            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
13453            let mut removed_any = false;
13454            for idx in indices {
13455                removed_any |= gcol.remove_object(idx);
13456            }
13457            // Every index already gone (a stale or duplicate reference):
13458            // leave the image alone. Rewriting is not just wasted I/O — a
13459            // 100%-full collection written by libhdf5 has no free-space
13460            // marker, so re-encoding it at its declared size cannot fit one
13461            // and the whole element update would fail.
13462            if !removed_any {
13463                continue;
13464            }
13465            if gcol.is_empty() {
13466                self.allocator
13467                    .free(addr, declared as u64, FreeSpaceClass::RawData);
13468                // The block is gone; a lingering entry would let an insert
13469                // pack into space the allocator can hand to anything.
13470                cwfs.retain(|e| e.addr != addr);
13471            } else {
13472                let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
13473                self.handle.write_at(addr, &rewritten)?;
13474                // The recovered bytes are packable now — list them, the way
13475                // libhdf5's `H5HG_remove` adds the heap to `cwfs`.
13476                if let Some(free) = gcol.free_space_at(&self.ctx, declared) {
13477                    if free >= 2 * objhdr {
13478                        cwfs_note(&mut cwfs, addr, declared, free);
13479                    }
13480                }
13481            }
13482        }
13483        Ok(())
13484    }
13485
13486    /// Add an attribute to a dataset.
13487    ///
13488    /// The attribute will be written as a message in the dataset's object
13489    /// header when the file is finalized.
13490    pub fn add_dataset_attribute(&self, ds_index: usize, attr: AttributeMessage) -> IoResult<()> {
13491        self.set_attribute(AttrTarget::Dataset(ds_index), attr)
13492    }
13493
13494    /// Build a variable-length UTF-8 string attribute message.
13495    ///
13496    /// The string is stored as one object in a global heap collection and the
13497    /// returned [`AttributeMessage`] carries the vlen reference as its data,
13498    /// with a vlen-string datatype and scalar dataspace. h5py reads the value
13499    /// back as a Python `str` (not `bytes`).
13500    ///
13501    /// This is the single owner of vlen-string-attribute construction: every
13502    /// public string-attribute setter (dataset, group, root, and the SWMR
13503    /// equivalents) routes through it, so a `VarLenUnicode` /
13504    /// `set_attr_string` value is always stored as a true variable-length
13505    /// string rather than the fixed-length string it used to be.
13506    ///
13507    /// The string's heap object is placed by
13508    /// [`insert_vlen_objects`](Self::insert_vlen_objects), so consecutive
13509    /// attributes pack into a shared collection instead of each paying the
13510    /// 4096-byte `H5HG_MINALLOC` minimum for a block that holds one string.
13511    fn vlen_string_attribute(&self, name: &str, value: &str) -> IoResult<AttributeMessage> {
13512        use crate::format::global_heap::encode_vlen_reference;
13513        use crate::format::messages::dataspace::DataspaceMessage;
13514        use crate::format::messages::datatype::DatatypeMessage;
13515
13516        let (gcol_addr, obj_idx) = self.insert_vlen_objects(&[value.as_bytes()])?[0];
13517        let seq_len = crate::format::global_heap::vlen_seq_len(value.len())?;
13518        let data = encode_vlen_reference(seq_len, gcol_addr, obj_idx as u32, &self.ctx);
13519        Ok(AttributeMessage {
13520            name: name.to_string(),
13521            datatype: DatatypeMessage::vlen_string_utf8(),
13522            dataspace: DataspaceMessage::scalar(),
13523            data,
13524        })
13525    }
13526
13527    /// Build a variable-length UTF-8 string **array** attribute message.
13528    ///
13529    /// The N-dimensional counterpart of
13530    /// [`vlen_string_attribute`](Self::vlen_string_attribute): every element
13531    /// string is stored as one object in a single global heap collection, and
13532    /// the attribute data is the row-major concatenation of one vlen reference
13533    /// per element. The datatype is the same vlen-string datatype; the dataspace
13534    /// is the simple dataspace described by `shape` (an empty `shape` is a
13535    /// scalar). h5py reads the value back as a numpy array of Python `str` with
13536    /// that shape.
13537    ///
13538    /// The caller owns the invariant that `values.len()` equals the product of
13539    /// `shape` (the public setters validate it before calling). The element
13540    /// objects are placed by
13541    /// [`insert_vlen_objects`](Self::insert_vlen_objects) — a zero-element
13542    /// array allocates nothing, and each reference carries its element's
13543    /// own collection address.
13544    fn vlen_string_array_attribute(
13545        &self,
13546        name: &str,
13547        values: &[&str],
13548        shape: &[u64],
13549    ) -> IoResult<AttributeMessage> {
13550        use crate::format::global_heap::encode_vlen_reference;
13551        use crate::format::messages::dataspace::DataspaceMessage;
13552        use crate::format::messages::datatype::DatatypeMessage;
13553
13554        debug_assert_eq!(
13555            values.len() as u64,
13556            shape.iter().product::<u64>(),
13557            "vlen_string_array_attribute values.len() must equal product(shape)"
13558        );
13559
13560        let items: Vec<&[u8]> = values.iter().map(|v| v.as_bytes()).collect();
13561        let placements = self.insert_vlen_objects(&items)?;
13562
13563        let mut data = Vec::with_capacity(values.len() * 16);
13564        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13565            data.extend_from_slice(&encode_vlen_reference(
13566                crate::format::global_heap::vlen_seq_len(values[i].len())?,
13567                gcol_addr,
13568                obj_idx as u32,
13569                &self.ctx,
13570            ));
13571        }
13572        Ok(AttributeMessage {
13573            name: name.to_string(),
13574            datatype: DatatypeMessage::vlen_string_utf8(),
13575            dataspace: DataspaceMessage::simple(shape),
13576            data,
13577        })
13578    }
13579
13580    /// Set a user-defined fill value for a dataset.
13581    ///
13582    /// `bytes` must be exactly one element wide (matching the dataset's
13583    /// datatype). The value is emitted as a `fill_defined = 2` fill-value
13584    /// message in the dataset object header when the file is finalized.
13585    ///
13586    /// IMPORTANT: for a *contiguous* dataset this also immediately writes
13587    /// the tiled fill value across the whole data block, so it must be
13588    /// called BEFORE any `write_dataset_raw` / `write_slice` — otherwise the
13589    /// fill write clobbers data already written. (The high-level builder
13590    /// always calls this right after creating the dataset.)
13591    pub fn set_dataset_fill_value(&self, ds_index: usize, bytes: Vec<u8>) -> IoResult<()> {
13592        let count = self.dataset_count();
13593        if ds_index >= count {
13594            return Err(crate::io::IoError::InvalidState(format!(
13595                "dataset index {} out of range",
13596                ds_index
13597            )));
13598        }
13599        let ds_ref = self.ds(ds_index);
13600        let mut ds = ds_ref.lock();
13601        let es = ds.datatype.element_size() as usize;
13602        if bytes.len() != es {
13603            return Err(crate::io::IoError::InvalidState(format!(
13604                "fill value is {} bytes but dataset element size is {}",
13605                bytes.len(),
13606                es
13607            )));
13608        }
13609        // For a dataset with no per-chunk fill path the fill-value message
13610        // only declares fill-on-allocation — tile the fill value across the
13611        // storage itself now, so unwritten elements read back as the fill
13612        // value. Which storage that is depends on the layout: a compact
13613        // dataset's is the image inside its layout message, a contiguous
13614        // one's is its data block. (The high-level builder calls this
13615        // immediately after create, before any data is written; a subsequent
13616        // write_raw/write_slice overwrites its region.)
13617        // An implicitly indexed dataset is filled here too, and for the same
13618        // reason: that index has no per-chunk fill path because it has no
13619        // per-chunk anything — its whole chunk grid is one run of space,
13620        // allocated and filled at create like a contiguous block. So the test
13621        // is not "is it chunked" but "does something else fill its chunks".
13622        let fills_per_chunk = ds
13623            .chunk_index_kind()
13624            .is_some_and(|k| k != ChunkIndexKind::Implicit);
13625        // `H5D_FILL_TIME_NEVER` means exactly this: the library never writes
13626        // the fill value into allocated storage. Call `set_dataset_fill_time`
13627        // before this method to have it observed here — the storage this
13628        // would otherwise tile keeps whatever zero bytes its allocation
13629        // already gave it.
13630        if !fills_per_chunk && ds.fill_time != FILL_TIME_NEVER {
13631            if let Some(len) = ds.compact.as_ref().map(Vec::len) {
13632                ds.compact = Some(crate::format::messages::fill_value::tiled_fill(
13633                    len,
13634                    Some(&bytes),
13635                ));
13636            } else {
13637                // An implicit index's chunk grid is filled as one run, the
13638                // same way a contiguous block is, and storage this file did
13639                // not allocate is not filled at all; `allocated_storage_run`
13640                // is where both of those are decided.
13641                let run = ds.allocated_storage_run();
13642                if let Some((target, data_size)) = run.filter(|&(_, size)| size > 0) {
13643                    let filled = crate::format::messages::fill_value::tiled_fill(
13644                        data_size as usize,
13645                        Some(&bytes),
13646                    );
13647                    self.write_contiguous_bytes(&target, 0, &filled)?;
13648                }
13649            }
13650        }
13651
13652        ds.fill_value = Some(bytes);
13653        ds.header_dirty = true;
13654        Ok(())
13655    }
13656
13657    /// Set when the fill value is written into allocated storage —
13658    /// `H5Pset_fill_time`. `time` is one of [`FILL_TIME_ALLOC`],
13659    /// [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]; anything else is rejected
13660    /// the way `H5Pset_fill_time` rejects an out-of-range `H5D_fill_time_t`.
13661    ///
13662    /// Call this before [`set_dataset_fill_value`](Self::set_dataset_fill_value)
13663    /// so that a `FILL_TIME_NEVER` policy is in place before that call
13664    /// decides whether to eager-tile the value into storage. (The
13665    /// high-level builder always calls it first.)
13666    pub fn set_dataset_fill_time(&self, ds_index: usize, time: u8) -> IoResult<()> {
13667        if !matches!(time, FILL_TIME_ALLOC | FILL_TIME_NEVER | FILL_TIME_IFSET) {
13668            return Err(crate::io::IoError::InvalidState(format!(
13669                "invalid fill time {time}; must be {FILL_TIME_ALLOC} (alloc), \
13670                 {FILL_TIME_NEVER} (never) or {FILL_TIME_IFSET} (if-set)"
13671            )));
13672        }
13673        let count = self.dataset_count();
13674        if ds_index >= count {
13675            return Err(crate::io::IoError::InvalidState(format!(
13676                "dataset index {} out of range",
13677                ds_index
13678            )));
13679        }
13680        let ds_ref = self.ds(ds_index);
13681        let mut ds = ds_ref.lock();
13682        ds.fill_time = time;
13683        ds.header_dirty = true;
13684        Ok(())
13685    }
13686
13687    /// Allocate a `chunk_bytes`-sized buffer pre-filled with dataset
13688    /// `ds_index`'s fill value (tiled one element wide), or zeros when no
13689    /// user-defined fill value exists.
13690    ///
13691    /// Every partial chunk the writer emits must be built on top of a
13692    /// buffer from this method, so that the unwritten element region of an
13693    /// allocated chunk reads back as the fill value rather than zero.
13694    ///
13695    /// Unconditional: a shrink's straddler refill
13696    /// (`refill_chunk_beyond_extent`) calls this to repair data about to
13697    /// become reachable again, which libhdf5's `H5D__chunk_prune_fill` does
13698    /// regardless of the fill-time policy. [`new_write_chunk_buffer`](Self::new_write_chunk_buffer)
13699    /// is the gated counterpart for a chunk touched for the first time
13700    /// during a write, where the policy does apply.
13701    pub(crate) fn new_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
13702        let ds = self.ds(ds_index);
13703        let m = ds.lock();
13704        let fv = m.fill_value.as_deref();
13705        crate::format::messages::fill_value::tiled_fill(chunk_bytes, fv)
13706    }
13707
13708    /// The buffer a chunk gets the first time a write touches it — this
13709    /// dataset's allocation-time fill gate. `H5D__chunk_lock`'s cache-miss
13710    /// path (H5Dchunk.c:4894) fills such a buffer only for `ALLOC`, or for
13711    /// `IFSET` with a fill value defined; `NEVER` leaves it as the zeros a
13712    /// fresh buffer already has. Everything else about the buffer is
13713    /// [`new_chunk_buffer`](Self::new_chunk_buffer)'s.
13714    fn new_write_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
13715        let never = {
13716            let ds = self.ds(ds_index);
13717            let m = ds.lock();
13718            m.fill_time == FILL_TIME_NEVER
13719        };
13720        if never {
13721            vec![0u8; chunk_bytes]
13722        } else {
13723            self.new_chunk_buffer(ds_index, chunk_bytes)
13724        }
13725    }
13726
13727    /// Write `n_frames` whole frames whose first row is `base_frame`, for
13728    /// whichever chunk index the dataset uses and whatever its chunk shape.
13729    ///
13730    /// The single owner of an append's chunk writes. The frames are one
13731    /// hyperslab — rows `base_frame .. base_frame + n_frames` over the full
13732    /// row shape — so the write goes through
13733    /// [`write_slice_chunked`](Self::write_slice_chunked), the same engine
13734    /// `write_slice` uses: a chunk the span covers completely is written
13735    /// straight through, a partial one is read-modify-write on top of what
13736    /// is stored (or the fill value), and a chunk row narrower or wider
13737    /// than the frame row is scattered at the chunk stride. The previous
13738    /// owner required the extensible-array index and packed rows at the
13739    /// frame stride, so appends to a fixed-array or v2 B-tree dataset
13740    /// failed at close and lost the buffered rows.
13741    ///
13742    /// The caller holds the dataset's op lock or the writer exclusively.
13743    pub(crate) fn write_append_frames(
13744        &self,
13745        ds_index: usize,
13746        base_frame: u64,
13747        n_frames: u64,
13748        frames: &[u8],
13749    ) -> IoResult<()> {
13750        if n_frames == 0 {
13751            return Ok(());
13752        }
13753        let geo = self.chunk_geometry(ds_index)?;
13754        let mut starts = vec![0u64; geo.dims.len()];
13755        starts[0] = base_frame;
13756        let mut counts = geo.dims.clone();
13757        counts[0] = n_frames;
13758        let expected = counts.iter().product::<u64>() * geo.element_size;
13759        if frames.len() as u64 != expected {
13760            return Err(crate::io::IoError::InvalidState(format!(
13761                "{n_frames} frames at rows {base_frame}.. need {expected} bytes, got {}",
13762                frames.len()
13763            )));
13764        }
13765        self.write_slice_chunked(ds_index, &starts, &counts, frames)
13766    }
13767
13768    /// Write the dataset's append buffer (if any) into its chunks and clear
13769    /// it. The single owner of the buffer-to-chunks transition: the flush at
13770    /// close, an append meeting a non-contiguous buffer, and any operation
13771    /// about to write rows the buffer holds all come through here.
13772    ///
13773    /// The caller holds the dataset's op lock or the writer exclusively —
13774    /// the take and the frame writes are separate acquisitions.
13775    pub(crate) fn flush_append_buffer(&self, ds_index: usize) -> IoResult<()> {
13776        let taken = { self.ds(ds_index).lock().append.take() };
13777        match taken {
13778            Some(b) => self.write_append_frames(ds_index, b.base, b.frames, &b.bytes),
13779            None => Ok(()),
13780        }
13781    }
13782
13783    /// Flush the append buffer when rows `start_row .. end_row` intersect
13784    /// the buffered range — those rows' current content is the buffer, and
13785    /// writing them on disk while the buffer still holds them would be
13786    /// undone by the flush at close.
13787    ///
13788    /// The caller holds the dataset's op lock or the writer exclusively.
13789    pub(crate) fn flush_append_buffer_if_intersecting(
13790        &self,
13791        ds_index: usize,
13792        start_row: u64,
13793        end_row: u64,
13794    ) -> IoResult<()> {
13795        let intersects = {
13796            let ds = self.ds(ds_index);
13797            let m = ds.lock();
13798            m.append
13799                .as_ref()
13800                .is_some_and(|b| start_row < b.base + b.frames && end_row > b.base)
13801        };
13802        if intersects {
13803            self.flush_append_buffer(ds_index)
13804        } else {
13805            Ok(())
13806        }
13807    }
13808
13809    /// Read an already-written chunk's *decompressed* bytes when the chunk
13810    /// is allocated and resolvable from the in-memory extensible-array
13811    /// index. Handles index-block and data-block chunks, filtered and
13812    /// unfiltered.
13813    ///
13814    /// Returns `Ok(None)` only when the chunk has never been written
13815    /// (address `UNDEF`) or the index genuinely does not reach it, which for
13816    /// a read-modify-write means the chunk's content is the fill value.
13817    pub(crate) fn read_chunk_if_present(
13818        &self,
13819        ds_index: usize,
13820        chunk_idx: u64,
13821    ) -> IoResult<Option<Vec<u8>>> {
13822        // Phase 1: resolve the chunk's location from the in-memory index.
13823        // Hold the slot guard through Phase 1: `chunked` borrows it, while the
13824        // `self.handle`/`self.ctx` reads below touch disjoint fields.
13825        let ds = self.ds(ds_index);
13826        let m = ds.lock();
13827        let element_size = m.datatype.element_size() as u64;
13828        let pipeline = m.filter_pipeline.clone();
13829        let Some(chunked) = m.chunked.as_ref() else {
13830            return Ok(None);
13831        };
13832        let chunk_bytes = chunked.chunk_dims.iter().product::<u64>() * element_size;
13833        let max_nelmts_bits = chunked.earray_params.max_nelmts_bits;
13834        let chunk_size_len = chunked.chunk_size_len;
13835        let is_filtered = chunked.filt_iblk.is_some();
13836
13837        // The chunk entry is either read straight from an index block, or
13838        // located via a data block that must itself be read from disk.
13839        enum Loc {
13840            Direct(u64, u64, u32),
13841            DataBlock {
13842                dblk_addr: u64,
13843                offset: usize,
13844                nelmts: usize,
13845            },
13846        }
13847
13848        // Resolve the chunk's location with the libhdf5-compatible EA
13849        // geometry (super-block-grouped data blocks), matching `record_ea_chunk`.
13850        let ea_loc = {
13851            let p = &chunked.earray_params;
13852            EaGeometry::new(
13853                p.idx_blk_elmts,
13854                p.data_blk_min_elmts,
13855                p.sup_blk_min_data_ptrs,
13856                p.max_nelmts_bits,
13857                p.max_dblk_page_nelmts_bits,
13858            )?
13859            .locate(chunk_idx)?
13860        };
13861        let loc = match ea_loc {
13862            EaLoc::Index { elem } => {
13863                if is_filtered {
13864                    let e = &chunked.filt_iblk.as_ref().unwrap().elements[elem];
13865                    Loc::Direct(e.addr, e.nbytes, e.filter_mask)
13866                } else {
13867                    Loc::Direct(chunked.ea_iblk.elements[elem], chunk_bytes, 0)
13868                }
13869            }
13870            EaLoc::Dblk(l) => {
13871                if l.paged {
13872                    return Err(crate::io::IoError::InvalidState(format!(
13873                        "chunk index {} lives in a paged extensible-array data \
13874                         block, which is not yet supported for read-modify-write",
13875                        chunk_idx
13876                    )));
13877                }
13878                let dblk_addr = match l.path {
13879                    EaDblkPath::Direct { idx } => {
13880                        if is_filtered {
13881                            chunked.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
13882                        } else {
13883                            chunked.ea_iblk.dblk_addrs[idx]
13884                        }
13885                    }
13886                    EaDblkPath::ViaSblk {
13887                        sblk_off,
13888                        local_dblk,
13889                        ndblks_in_sblk,
13890                        ..
13891                    } => {
13892                        let sblk_addr = if is_filtered {
13893                            chunked.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
13894                        } else {
13895                            chunked.ea_iblk.sblk_addrs[sblk_off]
13896                        };
13897                        if sblk_addr == UNDEF_ADDR {
13898                            return Ok(None);
13899                        }
13900                        let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
13901                        let sb = ExtensibleArraySuperBlock::decode(
13902                            &sb_buf,
13903                            &self.ctx,
13904                            max_nelmts_bits,
13905                            ndblks_in_sblk,
13906                            0,
13907                        )?;
13908                        sb.dblk_addrs[local_dblk]
13909                    }
13910                };
13911                if dblk_addr == UNDEF_ADDR {
13912                    return Ok(None);
13913                }
13914                Loc::DataBlock {
13915                    dblk_addr,
13916                    offset: l.offset_in_dblk as usize,
13917                    nelmts: l.dblk_nelmts as usize,
13918                }
13919            }
13920        };
13921
13922        // Phase 2: resolve through the data block (if needed) and read. The
13923        // mask is the chunk's filter mask (0 for unfiltered), so a chunk
13924        // written via a direct chunk write with a skipped filter is reversed
13925        // correctly during read-modify-write.
13926        let (addr, nbytes, mask) = match loc {
13927            Loc::Direct(a, n, m) => (a, n, m),
13928            Loc::DataBlock {
13929                dblk_addr,
13930                offset,
13931                nelmts,
13932            } => {
13933                let buf = self.handle.read_at_most(dblk_addr, 65536)?;
13934                if is_filtered {
13935                    let dblk = FilteredDataBlock::decode(
13936                        &buf,
13937                        &self.ctx,
13938                        max_nelmts_bits,
13939                        nelmts,
13940                        chunk_size_len,
13941                    )?;
13942                    let e = &dblk.elements[offset];
13943                    (e.addr, e.nbytes, e.filter_mask)
13944                } else {
13945                    let dblk =
13946                        ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, nelmts)?;
13947                    (dblk.elements[offset], chunk_bytes, 0)
13948                }
13949            }
13950        };
13951        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13952    }
13953
13954    /// Read one stored chunk block and undo its filters.
13955    ///
13956    /// `nbytes` is the *stored* length and `mask` the chunk's filter mask, so
13957    /// a chunk written by a direct chunk write with a skipped filter is
13958    /// reversed correctly. `Ok(None)` means the chunk has no block yet — the
13959    /// single place that judgement is made, shared by every chunk index.
13960    fn read_chunk_block(
13961        &self,
13962        pipeline: Option<&FilterPipeline>,
13963        addr: u64,
13964        nbytes: u64,
13965        mask: u32,
13966    ) -> IoResult<Option<Vec<u8>>> {
13967        if addr == UNDEF_ADDR || nbytes == 0 {
13968            return Ok(None);
13969        }
13970        let raw = self.handle.read_at(addr, nbytes as usize)?;
13971        match pipeline {
13972            Some(pl) => Ok(Some(filter::reverse_filters_masked(pl, &raw, mask)?)),
13973            None => Ok(Some(raw)),
13974        }
13975    }
13976
13977    /// Read the *decompressed* bytes of the chunk at `chunk_coords`, whichever
13978    /// chunk index the dataset uses, or `Ok(None)` when that chunk has never
13979    /// been written.
13980    ///
13981    /// This is the read half of a partial-chunk read-modify-write: a hyperslab
13982    /// write that covers only part of a chunk must start from what is already
13983    /// there. Keeping one entry point for all three index types is what lets
13984    /// [`write_slice`](Self::write_slice) stay index-agnostic.
13985    pub(crate) fn read_chunk_at_coords(
13986        &self,
13987        ds_index: usize,
13988        chunk_coords: &[u64],
13989    ) -> IoResult<Option<Vec<u8>>> {
13990        let geo = self.chunk_geometry(ds_index)?;
13991        // Only the linearly-addressed indexes compute a slot; a v2 B-tree is
13992        // keyed by the coordinates themselves (and may hold unlimited inner
13993        // dimensions, which have no linear slot).
13994        match geo.kind {
13995            ChunkIndexKind::ExtensibleArray => {
13996                let linear = geo.linear_index(chunk_coords)?;
13997                self.read_chunk_if_present(ds_index, linear)
13998            }
13999            ChunkIndexKind::FixedArray => {
14000                let linear = geo.linear_index(chunk_coords)?;
14001                let ds = self.ds(ds_index);
14002                let m = ds.lock();
14003                let pipeline = m.filter_pipeline.clone();
14004                let fa = m.fixed_array.as_ref().unwrap();
14005                let lidx = linear as usize;
14006                let (addr, nbytes, mask) = if pipeline.is_some() {
14007                    match fa.fa_dblk.filtered_elements.get(lidx) {
14008                        Some(e) => (e.address, e.chunk_size, e.filter_mask),
14009                        None => return Ok(None),
14010                    }
14011                } else {
14012                    match fa.fa_dblk.elements.get(lidx) {
14013                        Some(&a) => (a, geo.chunk_bytes(), 0),
14014                        None => return Ok(None),
14015                    }
14016                };
14017                drop(m);
14018                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14019            }
14020            ChunkIndexKind::BtreeV2 => {
14021                let ds = self.ds(ds_index);
14022                let m = ds.lock();
14023                let pipeline = m.filter_pipeline.clone();
14024                let bt2 = m.btree_v2.as_ref().unwrap();
14025                // A filtered index records the stored size and mask per chunk;
14026                // an unfiltered one stores whole chunks, so their size is the
14027                // chunk shape and no filter ran.
14028                let found = if bt2.index.filtered {
14029                    bt2.index
14030                        .lookup_filtered(chunk_coords)
14031                        .map(|r| (r.chunk_address, r.chunk_size, r.filter_mask))
14032                } else {
14033                    bt2.index
14034                        .lookup(chunk_coords)
14035                        .map(|r| (r.chunk_address, geo.chunk_bytes(), 0))
14036                };
14037                drop(m);
14038                match found {
14039                    Some((addr, nbytes, mask)) => {
14040                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14041                    }
14042                    None => Ok(None),
14043                }
14044            }
14045            // Every chunk of an implicitly indexed dataset exists from the
14046            // moment the dataset does, so there is no "never written" answer
14047            // to give: an untouched chunk reads back as the fill value the
14048            // create wrote there.
14049            ChunkIndexKind::Implicit => {
14050                let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
14051                self.read_chunk_block(None, grid + offset, geo.chunk_bytes(), 0)
14052            }
14053            // A single-chunk dataset's one chunk is never written until its
14054            // first write (unless the dataset was early-allocated and
14055            // unfiltered, in which case create already gave it an address) —
14056            // unlike Implicit, `UNDEF_ADDR` here is a real "never written".
14057            ChunkIndexKind::SingleChunk => {
14058                let ds = self.ds(ds_index);
14059                let m = ds.lock();
14060                let pipeline = m.filter_pipeline.clone();
14061                let sc = m.single_chunk.as_ref().unwrap();
14062                if sc.data_addr == UNDEF_ADDR {
14063                    return Ok(None);
14064                }
14065                let (addr, nbytes, mask) = if pipeline.is_some() {
14066                    (sc.data_addr, sc.nbytes, sc.filter_mask)
14067                } else {
14068                    (sc.data_addr, geo.chunk_bytes(), 0)
14069                };
14070                drop(m);
14071                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14072            }
14073            ChunkIndexKind::BtreeV1 => {
14074                let ds = self.ds(ds_index);
14075                let m = ds.lock();
14076                let pipeline = m.filter_pipeline.clone();
14077                let bt1 = m.btree_v1.as_ref().unwrap();
14078                let found = bt1
14079                    .position(chunk_coords)
14080                    .ok()
14081                    .map(|i| &bt1.records[i])
14082                    .map(|r| (r.address, r.nbytes as u64, r.filter_mask));
14083                drop(m);
14084                match found {
14085                    Some((addr, nbytes, mask)) => {
14086                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14087                    }
14088                    None => Ok(None),
14089                }
14090            }
14091        }
14092    }
14093
14094    /// The slot one chunk of an implicitly indexed dataset occupies: the
14095    /// address its whole chunk grid starts at, and the chunk's offset within
14096    /// that grid. `data_addr + linear_index * chunk_bytes` is the whole of
14097    /// that index (`H5D__none_idx_get_addr`, H5Dnone.c).
14098    ///
14099    /// The one place a chunk of such a dataset is placed — read and write both
14100    /// come through here, so the bounds check below covers both. The grid it
14101    /// names is [`DatasetInfo::implicit_grid`], which is why the write side
14102    /// can hand [`ContiguousTarget::Local`] to
14103    /// [`write_contiguous_bytes`](Self::write_contiguous_bytes) without asking
14104    /// anything: the external and virtual destinations that owner also knows
14105    /// about are unreachable from a chunked dataset.
14106    fn implicit_chunk_slot(
14107        &self,
14108        ds_index: usize,
14109        geo: &ChunkGeometry,
14110        chunk_coords: &[u64],
14111    ) -> IoResult<(u64, u64)> {
14112        let linear = geo.linear_index(chunk_coords)?;
14113        let ds = self.ds(ds_index);
14114        let m = ds.lock();
14115        let (grid, grid_size) = m.implicit_grid().ok_or_else(|| {
14116            crate::io::IoError::InvalidState("no implicitly indexed chunk grid".into())
14117        })?;
14118        let offset = linear.checked_mul(geo.chunk_bytes()).ok_or_else(|| {
14119            crate::io::IoError::InvalidState("implicit chunk offset overflows u64".into())
14120        })?;
14121        if offset + geo.chunk_bytes() > grid_size {
14122            return Err(crate::io::IoError::InvalidState(format!(
14123                "chunk {chunk_coords:?} lies outside the {grid_size} bytes of chunk space \
14124                 this implicitly indexed dataset was created with"
14125            )));
14126        }
14127        Ok((grid, offset))
14128    }
14129
14130    /// Write one whole chunk addressed by its grid coordinates, whichever
14131    /// chunk index the dataset uses. `data` is the chunk's unfiltered bytes;
14132    /// the dataset's filter pipeline (if any) runs here.
14133    ///
14134    /// The write half of the pair with
14135    /// [`read_chunk_at_coords`](Self::read_chunk_at_coords). Unlike the
14136    /// dataset-level `write_chunk_at`, this never grows the dataspace — a
14137    /// hyperslab write is bounded by the current extent by definition.
14138    ///
14139    /// The caller holds the dataset's op lock or the writer exclusively.
14140    pub(crate) fn write_chunk_at_coords(
14141        &self,
14142        ds_index: usize,
14143        chunk_coords: &[u64],
14144        data: &[u8],
14145    ) -> IoResult<()> {
14146        let geo = self.chunk_geometry(ds_index)?;
14147        match geo.kind {
14148            ChunkIndexKind::ExtensibleArray => {
14149                let linear = geo.linear_index(chunk_coords)?;
14150                self.write_chunk_inner(ds_index, linear, data)
14151            }
14152            ChunkIndexKind::FixedArray => {
14153                self.write_chunk_fixed_array_inner(ds_index, chunk_coords, data)
14154            }
14155            ChunkIndexKind::BtreeV2 => {
14156                self.write_chunk_btree_v2_inner(ds_index, chunk_coords, data)
14157            }
14158            ChunkIndexKind::Implicit => {
14159                self.write_chunk_implicit_inner(ds_index, chunk_coords, data)
14160            }
14161            ChunkIndexKind::SingleChunk => {
14162                self.write_chunk_single_chunk_inner(ds_index, chunk_coords, data)
14163            }
14164            ChunkIndexKind::BtreeV1 => {
14165                self.write_chunk_btree_v1_inner(ds_index, chunk_coords, data)
14166            }
14167        }
14168    }
14169
14170    /// Write one whole chunk to a dataset indexed by a version-1 B-tree.
14171    ///
14172    /// `chunk_coords` is the chunk's grid position. `data` is the chunk's
14173    /// unfiltered bytes; the dataset's filter pipeline runs here if it has
14174    /// one, and the key records the stored size and mask the way libhdf5's
14175    /// does (`H5D__btree_new_node`).
14176    ///
14177    /// The caller holds the dataset's op lock or the writer exclusively.
14178    pub(crate) fn write_chunk_btree_v1_inner(
14179        &self,
14180        ds_index: usize,
14181        chunk_coords: &[u64],
14182        data: &[u8],
14183    ) -> IoResult<()> {
14184        // Read what the write needs under a brief guard, then filter OUTSIDE
14185        // the lock, as every other index's write path does.
14186        let ds = self.ds(ds_index);
14187        let (chunk_bytes, pipeline) = {
14188            let m = ds.lock();
14189            let element_size = m.datatype.element_size() as u64;
14190            let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
14191                crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
14192            })?;
14193            (
14194                bt1.chunk_dims.iter().product::<u64>() * element_size,
14195                m.filter_pipeline.clone(),
14196            )
14197        };
14198        if data.len() as u64 != chunk_bytes {
14199            return Err(crate::io::IoError::InvalidState(format!(
14200                "chunk data size mismatch: expected {} bytes, got {}",
14201                chunk_bytes,
14202                data.len()
14203            )));
14204        }
14205
14206        let filtered;
14207        let stored = match pipeline {
14208            Some(ref pl) => {
14209                filtered = filter::apply_filters(pl, data)?;
14210                &filtered[..]
14211            }
14212            None => data,
14213        };
14214        self.record_btree_v1_chunk(ds_index, chunk_coords, stored, 0)
14215    }
14216
14217    /// Write a pre-filtered chunk verbatim to a version-1 B-tree dataset,
14218    /// recording the caller-supplied `filter_mask` — the classic-index half
14219    /// of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
14220    ///
14221    /// The caller holds the dataset's op lock or the writer exclusively.
14222    pub(crate) fn write_compressed_chunk_btree_v1_inner(
14223        &self,
14224        ds_index: usize,
14225        chunk_coords: &[u64],
14226        data: &[u8],
14227        filter_mask: u32,
14228    ) -> IoResult<()> {
14229        if self.ds(ds_index).lock().filter_pipeline.is_none() {
14230            return Err(crate::io::IoError::InvalidState(
14231                "write_chunk_raw requires a filtered dataset (an unfiltered chunk \
14232                 is stored at its full size, so there is nothing for a stored size \
14233                 or a filter mask to say)"
14234                    .into(),
14235            ));
14236        }
14237        self.record_btree_v1_chunk(ds_index, chunk_coords, data, filter_mask)
14238    }
14239
14240    /// Place a chunk's already-final bytes in the file and record them in the
14241    /// version-1 B-tree under the caller-supplied `filter_mask`.
14242    ///
14243    /// Shared by the two writes above, so both reach the index through one
14244    /// placement rule. The records are kept in key order here — the bulk load
14245    /// at flush walks them in that order and a lookup bisects them.
14246    fn record_btree_v1_chunk(
14247        &self,
14248        ds_index: usize,
14249        chunk_coords: &[u64],
14250        final_bytes: &[u8],
14251        filter_mask: u32,
14252    ) -> IoResult<()> {
14253        let stored_len = final_bytes.len() as u64;
14254        // The key's size field is 32 bits wide (`H5D_btree_key_t::nbytes`),
14255        // which is also libhdf5's limit on a chunk in this index.
14256        let Ok(nbytes) = u32::try_from(stored_len) else {
14257            return Err(crate::io::IoError::InvalidState(format!(
14258                "stored chunk size {stored_len} does not fit in the 32-bit size \
14259                 field of a version-1 B-tree chunk key"
14260            )));
14261        };
14262        let ds = self.ds(ds_index);
14263        let mut m = ds.lock();
14264        let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
14265            crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
14266        })?;
14267        if chunk_coords.len() != bt1.chunk_dims.len() {
14268            return Err(crate::io::IoError::InvalidState(format!(
14269                "chunk_coords has {} entries but the dataset has {} dimensions",
14270                chunk_coords.len(),
14271                bt1.chunk_dims.len()
14272            )));
14273        }
14274        // A coordinate past the maximum extent has no chunk to be: unlike the
14275        // array indexes there is no slot to run out of, so the bound is
14276        // checked here or not at all. An unlimited dimension has none.
14277        for (d, ((&c, &cd), &max)) in chunk_coords
14278            .iter()
14279            .zip(&bt1.chunk_dims)
14280            .zip(&bt1.max_dims)
14281            .enumerate()
14282        {
14283            if max != u64::MAX && c.saturating_mul(cd) >= max {
14284                return Err(crate::io::IoError::InvalidState(format!(
14285                    "chunk coordinate {c} in dimension {d} is outside the maximum \
14286                     extent {max}"
14287                )));
14288            }
14289        }
14290        let slot = bt1.position(chunk_coords);
14291        let old = slot.ok().map(|i| {
14292            let r = &bt1.records[i];
14293            (r.address, r.nbytes as u64)
14294        });
14295        // A rewrite whose stored size is unchanged stays where it is (always
14296        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
14297        let address = self.place_chunk(old, stored_len);
14298        self.handle.write_at(address, final_bytes)?;
14299
14300        let bt1 = m.btree_v1.as_mut().unwrap();
14301        let record = BtreeV1ChunkRecord {
14302            scaled: chunk_coords.to_vec(),
14303            address,
14304            nbytes,
14305            filter_mask,
14306        };
14307        match slot {
14308            Ok(i) => bt1.records[i] = record,
14309            Err(i) => bt1.records.insert(i, record),
14310        }
14311        bt1.chunks_written += 1;
14312        Ok(())
14313    }
14314
14315    /// Write one whole chunk of an implicitly indexed dataset into the slot
14316    /// its coordinates name. There is no index to record anything in — the
14317    /// slot is where it always was — so this is the write in full.
14318    ///
14319    /// The bytes go through [`write_contiguous_bytes`](Self::write_contiguous_bytes),
14320    /// the one owner of a raw-byte write, against the grid
14321    /// [`implicit_chunk_slot`](Self::implicit_chunk_slot) names.
14322    ///
14323    /// The caller holds the dataset's op lock or the writer exclusively.
14324    pub(crate) fn write_chunk_implicit_inner(
14325        &self,
14326        ds_index: usize,
14327        chunk_coords: &[u64],
14328        data: &[u8],
14329    ) -> IoResult<()> {
14330        let geo = self.chunk_geometry(ds_index)?;
14331        let chunk_bytes = geo.chunk_bytes();
14332        if data.len() as u64 != chunk_bytes {
14333            return Err(crate::io::IoError::InvalidState(format!(
14334                "chunk data size mismatch: expected {} bytes, got {}",
14335                chunk_bytes,
14336                data.len()
14337            )));
14338        }
14339        let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
14340        self.write_contiguous_bytes(&ContiguousTarget::Local(grid), offset, data)
14341    }
14342
14343    /// Snapshot the geometry needed to address a chunked dataset's grid.
14344    ///
14345    /// Taken under one brief slot guard so the callers below — which re-lock
14346    /// the slot through `write_chunk`/`read_chunk_*` — never hold it across
14347    /// compression or I/O.
14348    fn chunk_geometry(&self, ds_index: usize) -> IoResult<ChunkGeometry> {
14349        let ds = self.ds(ds_index);
14350        let m = ds.lock();
14351        let Some(kind) = m.chunk_index_kind() else {
14352            return Err(crate::io::IoError::InvalidState(
14353                "not a chunked dataset".into(),
14354            ));
14355        };
14356        let chunk_dims = match kind {
14357            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
14358            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
14359            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
14360            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
14361            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
14362            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
14363        };
14364        Ok(ChunkGeometry {
14365            kind,
14366            dims: m.dataspace.dims.clone(),
14367            max_dims: m.dataspace.max_dims.clone(),
14368            chunk_dims,
14369            element_size: m.datatype.element_size() as u64,
14370        })
14371    }
14372
14373    /// Index-grid slot of the chunk at grid `coords` (see
14374    /// [`crate::io::chunk_grid`]).
14375    pub(crate) fn chunk_slot(&self, ds_index: usize, coords: &[u64]) -> IoResult<u64> {
14376        self.chunk_geometry(ds_index)?.linear_index(coords)
14377    }
14378
14379    /// Grid coordinates of the chunk recorded under index-grid slot `linear`
14380    /// — the inverse of [`Self::chunk_slot`].
14381    pub(crate) fn chunk_coords_from_slot(
14382        &self,
14383        ds_index: usize,
14384        linear: u64,
14385    ) -> IoResult<Vec<u64>> {
14386        let geo = self.chunk_geometry(ds_index)?;
14387        crate::io::chunk_grid::coords_of(
14388            &geo.dims,
14389            geo.max_dims.as_deref(),
14390            &geo.chunk_dims,
14391            linear,
14392        )
14393    }
14394
14395    /// Define a chunked dataset indexed by a fixed array, fixed at its
14396    /// current shape (`max_dims == dims`). `chunk_dims` defines the chunk
14397    /// shape. Returns the dataset index.
14398    pub fn create_fixed_array_dataset(
14399        &self,
14400        name: &str,
14401        datatype: DatatypeMessage,
14402        dims: &[u64],
14403        chunk_dims: &[u64],
14404    ) -> IoResult<usize> {
14405        self.create_fixed_array_dataset_with_max(name, datatype, dims, dims, chunk_dims, None)
14406    }
14407
14408    /// Define a fixed-shape compressed chunked dataset indexed by a
14409    /// *filtered* Fixed Array (`max_dims == dims`).
14410    ///
14411    /// Like `create_fixed_array_dataset`, but the FA header carries the filtered
14412    /// client id and a `chunk_size_len`-wide compressed-size field per chunk
14413    /// (`FixedArrayFilteredChunkElement`), and the dataset gets a filter
14414    /// pipeline. Chunks written via `write_chunk_fixed_array` are compressed and
14415    /// their compressed size + filter mask are recorded in the data block.
14416    ///
14417    /// A convenience over [`create_fixed_array_dataset_with_max`]'s own
14418    /// pipeline argument; production dataset creation calls that directly,
14419    /// so this is kept as a direct entry point for this crate's own
14420    /// white-box tests.
14421    ///
14422    /// [`create_fixed_array_dataset_with_max`]: Self::create_fixed_array_dataset_with_max
14423    #[cfg(all(test, feature = "deflate"))]
14424    pub fn create_fixed_array_dataset_with_pipeline(
14425        &self,
14426        name: &str,
14427        datatype: DatatypeMessage,
14428        dims: &[u64],
14429        chunk_dims: &[u64],
14430        pipeline: FilterPipeline,
14431    ) -> IoResult<usize> {
14432        self.create_fixed_array_dataset_with_max(
14433            name,
14434            datatype,
14435            dims,
14436            dims,
14437            chunk_dims,
14438            Some(pipeline),
14439        )
14440    }
14441
14442    /// Define a chunked dataset indexed by a fixed array, growable up to
14443    /// `max_dims` (every maximum finite — libhdf5 picks this index exactly
14444    /// when no dimension is unlimited).
14445    ///
14446    /// The array is sized for the chunk grid of the *maximum* extent, the
14447    /// libhdf5 rule (`H5D__farray_idx_create` uses `max_nchunks`), so the
14448    /// dataset can be extended to `max_dims` without re-indexing chunks.
14449    pub fn create_fixed_array_dataset_with_max(
14450        &self,
14451        name: &str,
14452        datatype: DatatypeMessage,
14453        dims: &[u64],
14454        max_dims: &[u64],
14455        chunk_dims: &[u64],
14456        pipeline: Option<FilterPipeline>,
14457    ) -> IoResult<usize> {
14458        let create = self.begin_create(name)?;
14459        let name = create.name.as_str();
14460        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14461        if max_dims.contains(&u64::MAX) {
14462            return Err(crate::io::IoError::InvalidState(
14463                "a fixed-array index requires a fixed maximum shape (no unlimited dimension)"
14464                    .into(),
14465            ));
14466        }
14467        let mut num_chunks: u64 = 1;
14468        for g in crate::io::chunk_grid::index_grid(dims, Some(max_dims), chunk_dims)? {
14469            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
14470                crate::io::IoError::InvalidState("chunk count overflows u64".into())
14471            })?;
14472        }
14473
14474        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14475        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
14476
14477        // Create the FA header. For a filtered FA, chunk_size_len is sized
14478        // the same way the filtered Extensible Array path computes it:
14479        // derived from the uncompressed chunk byte count under layout v4,
14480        // the fixed `sizeof_size` under layout v5.
14481        let mut fa_header = if pipeline.is_some() {
14482            let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
14483            FixedArrayHeader::new_for_filtered_chunks(&self.ctx, num_chunks, chunk_size_len)
14484        } else {
14485            FixedArrayHeader::new_for_chunks(&self.ctx, num_chunks)
14486        };
14487        let hdr_encoded = fa_header.encode(&self.ctx);
14488        let fa_header_addr = self
14489            .allocator
14490            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14491
14492        // Create the FA data block. libhdf5 switches to a paged layout once
14493        // num_elmts exceeds dblk_page_nelmts; both layouts allocate space
14494        // for `num_chunks` entries up front, but the paged layout also
14495        // reserves the page-init bitmap and a per-page checksum.
14496        let fa_dblk = if pipeline.is_some() {
14497            FixedArrayDataBlock::new_filtered(fa_header_addr, num_chunks as usize)
14498        } else {
14499            FixedArrayDataBlock::new_unfiltered(fa_header_addr, num_chunks as usize)
14500        };
14501        let dblk_size = fixed_array_dblk_disk_size(&self.ctx, &fa_header);
14502        let fa_dblk_addr = self.allocator.allocate(dblk_size, FreeSpaceClass::Metadata);
14503
14504        // Update header with data block address
14505        fa_header.data_blk_addr = fa_dblk_addr;
14506
14507        // Write both. The data block content is finalized in `flush_dataset`
14508        // once all chunk addresses are known; here we just reserve space and
14509        // write the header so the file is structurally consistent.
14510        let hdr_encoded = fa_header.encode(&self.ctx);
14511        self.handle.write_at(fa_header_addr, &hdr_encoded)?;
14512        let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa_header, &fa_dblk);
14513        debug_assert_eq!(dblk_encoded.len() as u64, dblk_size);
14514        self.handle.write_at(fa_dblk_addr, &dblk_encoded)?;
14515
14516        // The maximum is stored even when it equals the dims: it is what
14517        // `extend_dataset` checks growth against, and the FA capacity above
14518        // is exactly its chunk grid.
14519        let dataspace = DataspaceMessage {
14520            // Chunked storage always requires at least one dimension, so
14521            // this is never Scalar or Null.
14522            class: DataspaceClass::Simple,
14523            dims: dims.to_vec(),
14524            max_dims: Some(max_dims.to_vec()),
14525        };
14526
14527        let idx = self.push_dataset(
14528            &create,
14529            DatasetInfo {
14530                name: name.to_string(),
14531                datatype,
14532                committed_type: None,
14533                external: None,
14534                virtual_storage: None,
14535                dataspace,
14536                read_format: None,
14537                obj_header_addr: 0,
14538                data_addr: UNDEF_ADDR,
14539                data_size: 0,
14540                compact: None,
14541                attributes: Vec::new(),
14542                obj_header_written_addr: None,
14543                obj_header_blocks: Vec::new(),
14544                filter_pipeline: pipeline,
14545                deleted: false,
14546                extent_dirty: false,
14547                header_dirty: false,
14548                nlink_written: 1,
14549                creation_seq: self.take_creation_seq(),
14550                track_attr_order: self.track_order.attrs,
14551                fill_value: None,
14552                fill_time: FILL_TIME_IFSET,
14553                layout_version,
14554                times: self.created_object_times(),
14555                chunked: None,
14556                btree_v2: None,
14557                implicit: None,
14558                single_chunk: None,
14559                btree_v1: None,
14560                fixed_array: Some(FixedArrayDatasetInfo {
14561                    chunk_dims: chunk_dims.to_vec(),
14562                    fa_header_addr,
14563                    fa_dblk_addr,
14564                    fa_header,
14565                    fa_dblk,
14566                    chunks_written: 0,
14567                }),
14568                append: None,
14569            },
14570        );
14571
14572        Ok(idx)
14573    }
14574
14575    /// Define a chunked dataset with the *implicit* index: no index structure
14576    /// at all, every chunk of the grid allocated at create in one contiguous
14577    /// run, addressed by arithmetic (`H5Dnone.c`).
14578    ///
14579    /// libhdf5 picks this index only where that arithmetic is total, and this
14580    /// enforces the same three conditions
14581    /// (`H5D__layout_set_latest_indexing`, H5Dlayout.c): no filter — a
14582    /// filtered chunk is not `chunk_bytes` long, so the run would not be a
14583    /// grid; no unlimited dimension — the run has to have a length; and early
14584    /// allocation, which is what this creator *does* rather than something it
14585    /// checks. The dataset's fill-value message says so
14586    /// (`build_dataset_header`), because a file claiming incremental
14587    /// allocation is one libhdf5 would never have chosen this index for.
14588    pub fn create_implicit_dataset(
14589        &self,
14590        name: &str,
14591        datatype: DatatypeMessage,
14592        dims: &[u64],
14593        chunk_dims: &[u64],
14594    ) -> IoResult<usize> {
14595        let create = self.begin_create(name)?;
14596        let name = create.name.as_str();
14597        validate_chunk_geometry(dims, dims, chunk_dims)?;
14598        let mut num_chunks: u64 = 1;
14599        for g in crate::io::chunk_grid::index_grid(dims, None, chunk_dims)? {
14600            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
14601                crate::io::IoError::InvalidState("chunk count overflows u64".into())
14602            })?;
14603        }
14604        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14605        let data_size = num_chunks.checked_mul(chunk_bytes).ok_or_else(|| {
14606            crate::io::IoError::InvalidState("implicit chunk storage overflows u64".into())
14607        })?;
14608        let layout_version = self.chunk_layout_version(false, chunk_bytes);
14609
14610        // Early allocation is the whole of this index: the run exists, and
14611        // holds the fill value, before any chunk is written. It is written
14612        // out rather than merely reserved because the file's end-of-file
14613        // address is what libhdf5 checks a file's completeness against — a
14614        // reserved-but-absent tail is a truncated file to it.
14615        let data_addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
14616        self.handle.write_at(
14617            data_addr,
14618            &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
14619        )?;
14620
14621        let dataspace = DataspaceMessage {
14622            // Chunked storage always requires at least one dimension, so
14623            // this is never Scalar or Null.
14624            class: DataspaceClass::Simple,
14625            dims: dims.to_vec(),
14626            max_dims: Some(dims.to_vec()),
14627        };
14628
14629        let idx = self.push_dataset(
14630            &create,
14631            DatasetInfo {
14632                name: name.to_string(),
14633                datatype,
14634                committed_type: None,
14635                external: None,
14636                virtual_storage: None,
14637                dataspace,
14638                read_format: None,
14639                obj_header_addr: 0,
14640                data_addr: UNDEF_ADDR,
14641                data_size: 0,
14642                compact: None,
14643                attributes: Vec::new(),
14644                obj_header_written_addr: None,
14645                obj_header_blocks: Vec::new(),
14646                filter_pipeline: None,
14647                deleted: false,
14648                extent_dirty: false,
14649                header_dirty: false,
14650                nlink_written: 1,
14651                creation_seq: self.take_creation_seq(),
14652                track_attr_order: self.track_order.attrs,
14653                fill_value: None,
14654                fill_time: FILL_TIME_IFSET,
14655                layout_version,
14656                times: self.created_object_times(),
14657                chunked: None,
14658                btree_v2: None,
14659                fixed_array: None,
14660                implicit: Some(ImplicitDatasetInfo {
14661                    chunk_dims: chunk_dims.to_vec(),
14662                    data_addr,
14663                    data_size,
14664                }),
14665                single_chunk: None,
14666                btree_v1: None,
14667                append: None,
14668            },
14669        );
14670
14671        Ok(idx)
14672    }
14673
14674    /// Define a chunked dataset indexed by the single-chunk index: a fixed
14675    /// shape covered by exactly one whole chunk (`chunk_dims == dims`), its
14676    /// address — and, once written, size and filter mask if filtered — held
14677    /// directly in the layout message instead of any index structure
14678    /// (`H5Dsingle.c`). libhdf5 selects this index ahead of both Implicit and
14679    /// Fixed Array whenever the shape qualifies, filtered or not, early
14680    /// allocation or not (`H5D__layout_set_latest_indexing`).
14681    ///
14682    /// `early_alloc` mirrors [`create_implicit_dataset`](Self::create_implicit_dataset):
14683    /// when true, the chunk's storage is allocated and filled with the fill
14684    /// value immediately, matching an early-allocated unfiltered dataset
14685    /// whose one chunk covers the whole shape. When false, the chunk has no
14686    /// address until its first write, the same as an unfiltered Fixed Array
14687    /// element.
14688    pub fn create_single_chunk_dataset(
14689        &self,
14690        name: &str,
14691        datatype: DatatypeMessage,
14692        dims: &[u64],
14693        chunk_dims: &[u64],
14694        early_alloc: bool,
14695    ) -> IoResult<usize> {
14696        let create = self.begin_create(name)?;
14697        let name = create.name.as_str();
14698        validate_chunk_geometry(dims, dims, chunk_dims)?;
14699        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14700        let layout_version = self.chunk_layout_version(false, data_size);
14701
14702        let data_addr = if early_alloc {
14703            // Same reasoning as `create_implicit_dataset`: the fill-value
14704            // bytes are written now, not merely reserved, because the
14705            // file's end-of-file address is what libhdf5 checks a file's
14706            // completeness against.
14707            let addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
14708            self.handle.write_at(
14709                addr,
14710                &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
14711            )?;
14712            addr
14713        } else {
14714            UNDEF_ADDR
14715        };
14716
14717        let dataspace = DataspaceMessage {
14718            // Chunked storage always requires at least one dimension, so
14719            // this is never Scalar or Null.
14720            class: DataspaceClass::Simple,
14721            dims: dims.to_vec(),
14722            max_dims: Some(dims.to_vec()),
14723        };
14724
14725        let idx = self.push_dataset(
14726            &create,
14727            DatasetInfo {
14728                name: name.to_string(),
14729                datatype,
14730                committed_type: None,
14731                external: None,
14732                virtual_storage: None,
14733                dataspace,
14734                read_format: None,
14735                obj_header_addr: 0,
14736                data_addr: UNDEF_ADDR,
14737                data_size: 0,
14738                compact: None,
14739                attributes: Vec::new(),
14740                obj_header_written_addr: None,
14741                obj_header_blocks: Vec::new(),
14742                filter_pipeline: None,
14743                deleted: false,
14744                extent_dirty: false,
14745                header_dirty: false,
14746                nlink_written: 1,
14747                creation_seq: self.take_creation_seq(),
14748                track_attr_order: self.track_order.attrs,
14749                fill_value: None,
14750                fill_time: FILL_TIME_IFSET,
14751                layout_version,
14752                times: self.created_object_times(),
14753                chunked: None,
14754                btree_v2: None,
14755                fixed_array: None,
14756                implicit: None,
14757                single_chunk: Some(SingleChunkDatasetInfo {
14758                    chunk_dims: chunk_dims.to_vec(),
14759                    data_addr,
14760                    data_size,
14761                    nbytes: if early_alloc { data_size } else { 0 },
14762                    filter_mask: 0,
14763                    chunks_written: 0,
14764                    early_alloc,
14765                }),
14766                btree_v1: None,
14767                append: None,
14768            },
14769        );
14770
14771        Ok(idx)
14772    }
14773
14774    /// Define a fixed-shape compressed chunked dataset — of exactly one
14775    /// whole chunk — indexed by a *filtered* single-chunk index
14776    /// (`H5O_LAYOUT_CHUNK_SINGLE_INDEX_WITH_FILTER`, H5Dsingle.c). The
14777    /// chunk's stored size and filter mask are recorded inline in the
14778    /// layout message once the chunk is written.
14779    ///
14780    /// Like [`create_fixed_array_dataset_with_pipeline`](Self::create_fixed_array_dataset_with_pipeline),
14781    /// there is nothing to allocate ahead of that first write — a filtered
14782    /// chunk's stored length isn't known until it is compressed — so this
14783    /// dataset is always incrementally allocated regardless of the caller's
14784    /// requested allocation time.
14785    pub fn create_single_chunk_dataset_with_pipeline(
14786        &self,
14787        name: &str,
14788        datatype: DatatypeMessage,
14789        dims: &[u64],
14790        chunk_dims: &[u64],
14791        pipeline: FilterPipeline,
14792    ) -> IoResult<usize> {
14793        let create = self.begin_create(name)?;
14794        let name = create.name.as_str();
14795        validate_chunk_geometry(dims, dims, chunk_dims)?;
14796        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14797        let layout_version = self.chunk_layout_version(true, data_size);
14798
14799        let dataspace = DataspaceMessage {
14800            // Chunked storage always requires at least one dimension, so
14801            // this is never Scalar or Null.
14802            class: DataspaceClass::Simple,
14803            dims: dims.to_vec(),
14804            max_dims: Some(dims.to_vec()),
14805        };
14806
14807        let idx = self.push_dataset(
14808            &create,
14809            DatasetInfo {
14810                name: name.to_string(),
14811                datatype,
14812                committed_type: None,
14813                external: None,
14814                virtual_storage: None,
14815                dataspace,
14816                read_format: None,
14817                obj_header_addr: 0,
14818                data_addr: UNDEF_ADDR,
14819                data_size: 0,
14820                compact: None,
14821                attributes: Vec::new(),
14822                obj_header_written_addr: None,
14823                obj_header_blocks: Vec::new(),
14824                filter_pipeline: Some(pipeline),
14825                deleted: false,
14826                extent_dirty: false,
14827                header_dirty: false,
14828                nlink_written: 1,
14829                creation_seq: self.take_creation_seq(),
14830                track_attr_order: self.track_order.attrs,
14831                fill_value: None,
14832                fill_time: FILL_TIME_IFSET,
14833                layout_version,
14834                times: self.created_object_times(),
14835                chunked: None,
14836                btree_v2: None,
14837                fixed_array: None,
14838                implicit: None,
14839                single_chunk: Some(SingleChunkDatasetInfo {
14840                    chunk_dims: chunk_dims.to_vec(),
14841                    data_addr: UNDEF_ADDR,
14842                    data_size,
14843                    nbytes: 0,
14844                    filter_mask: 0,
14845                    chunks_written: 0,
14846                    early_alloc: false,
14847                }),
14848                btree_v1: None,
14849                append: None,
14850            },
14851        );
14852
14853        Ok(idx)
14854    }
14855
14856    /// Define a chunked dataset indexed by a version-1 B-tree — the classic
14857    /// chunk index, and the only one a version-0/1 superblock file can carry.
14858    ///
14859    /// The tree itself is not created here: libhdf5 leaves the layout
14860    /// message's address undefined until the first chunk is inserted
14861    /// (`H5D__btree_idx_create` runs on that insert), and so does this — the
14862    /// flush that bulk-loads the records is what puts a node in the file.
14863    ///
14864    /// Unlike the array indexes this one has no grid to size, so it takes any
14865    /// number of unlimited dimensions: a key *is* the chunk's position, and
14866    /// the tree is ordered by it.
14867    pub fn create_btree_v1_dataset(
14868        &self,
14869        name: &str,
14870        datatype: DatatypeMessage,
14871        dims: &[u64],
14872        max_dims: &[u64],
14873        chunk_dims: &[u64],
14874        pipeline: Option<FilterPipeline>,
14875    ) -> IoResult<usize> {
14876        let create = self.begin_create(name)?;
14877        let name = create.name.as_str();
14878        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14879        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14880        if chunk_bytes > u32::MAX as u64 {
14881            return Err(crate::io::IoError::InvalidState(format!(
14882                "a {chunk_bytes}-byte chunk does not fit the 32-bit size field of a \
14883                 version-1 B-tree chunk key"
14884            )));
14885        }
14886
14887        let dataspace = DataspaceMessage {
14888            // Chunked storage always requires at least one dimension, so
14889            // this is never Scalar or Null.
14890            class: DataspaceClass::Simple,
14891            dims: dims.to_vec(),
14892            max_dims: Some(max_dims.to_vec()),
14893        };
14894
14895        let idx = self.push_dataset(
14896            &create,
14897            DatasetInfo {
14898                name: name.to_string(),
14899                datatype,
14900                committed_type: None,
14901                external: None,
14902                virtual_storage: None,
14903                dataspace,
14904                read_format: None,
14905                obj_header_addr: 0,
14906                data_addr: UNDEF_ADDR,
14907                data_size: 0,
14908                compact: None,
14909                attributes: Vec::new(),
14910                obj_header_written_addr: None,
14911                obj_header_blocks: Vec::new(),
14912                filter_pipeline: pipeline,
14913                deleted: false,
14914                extent_dirty: false,
14915                header_dirty: false,
14916                nlink_written: 1,
14917                creation_seq: self.take_creation_seq(),
14918                track_attr_order: self.track_order.attrs,
14919                fill_value: None,
14920                fill_time: FILL_TIME_IFSET,
14921                // The version-3 data layout message this index encodes as:
14922                // `H5O_LAYOUT_VERSION_DEFAULT`, which is the floor of
14923                // `H5D__chunk_set_info`'s final MAX and the whole of it below
14924                // the version-4 gate — a bound whose row is lower does not
14925                // push the message down, it only keeps the v1.10 indexes out.
14926                layout_version: LAYOUT_VERSION_DEFAULT,
14927                times: self.created_object_times(),
14928                chunked: None,
14929                fixed_array: None,
14930                btree_v2: None,
14931                implicit: None,
14932                single_chunk: None,
14933                btree_v1: Some(BtreeV1DatasetInfo {
14934                    chunk_dims: chunk_dims.to_vec(),
14935                    max_dims: max_dims.to_vec(),
14936                    config: self.btree_v1_config(),
14937                    records: Vec::new(),
14938                    node_addrs: Vec::new(),
14939                    root_addr: UNDEF_ADDR,
14940                    chunks_written: 0,
14941                }),
14942                append: None,
14943            },
14944        );
14945
14946        Ok(idx)
14947    }
14948
14949    /// Define a chunked dataset indexed by a B-tree v2 (multiple unlimited dimensions).
14950    ///
14951    /// Returns the dataset index.
14952    pub fn create_btree_v2_dataset(
14953        &self,
14954        name: &str,
14955        datatype: DatatypeMessage,
14956        dims: &[u64],
14957        max_dims: &[u64],
14958        chunk_dims: &[u64],
14959    ) -> IoResult<usize> {
14960        self.create_btree_v2_dataset_inner(name, datatype, dims, max_dims, chunk_dims, None)
14961    }
14962
14963    /// Define a *filtered* chunked dataset indexed by a B-tree v2.
14964    ///
14965    /// The v2 B-tree counterpart of
14966    /// [`create_chunked_dataset_with_pipeline`](Self::create_chunked_dataset_with_pipeline):
14967    /// chunks are compressed on write and the index records each chunk's
14968    /// stored size and filter mask (record type 11), the same shape libhdf5
14969    /// builds when a multi-unlimited-dimension dataset has a filter pipeline
14970    /// (`H5Dbtree2.c`, `H5D_BT2_FILT`).
14971    pub fn create_btree_v2_dataset_with_pipeline(
14972        &self,
14973        name: &str,
14974        datatype: DatatypeMessage,
14975        dims: &[u64],
14976        max_dims: &[u64],
14977        chunk_dims: &[u64],
14978        pipeline: FilterPipeline,
14979    ) -> IoResult<usize> {
14980        self.create_btree_v2_dataset_inner(
14981            name,
14982            datatype,
14983            dims,
14984            max_dims,
14985            chunk_dims,
14986            Some(pipeline),
14987        )
14988    }
14989
14990    fn create_btree_v2_dataset_inner(
14991        &self,
14992        name: &str,
14993        datatype: DatatypeMessage,
14994        dims: &[u64],
14995        max_dims: &[u64],
14996        chunk_dims: &[u64],
14997        pipeline: Option<FilterPipeline>,
14998    ) -> IoResult<usize> {
14999        use crate::format::chunk_index::btree_v2::Bt2Header;
15000
15001        let create = self.begin_create(name)?;
15002        let name = create.name.as_str();
15003        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
15004        let ndims = dims.len();
15005        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
15006        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
15007
15008        // The filtered record's size field is as wide as libhdf5 will
15009        // recompute it — from the uncompressed chunk size under layout v4,
15010        // the fixed `sizeof_size` under layout v5 — exactly as the
15011        // extensible- and fixed-array filtered paths size theirs.
15012        let bt2_index = match pipeline {
15013            Some(_) => {
15014                let len = self.chunk_size_len_for(layout_version, chunk_bytes);
15015                Bt2ChunkIndex::new_filtered(ndims, len)
15016            }
15017            None => Bt2ChunkIndex::new_unfiltered(ndims),
15018        };
15019
15020        // The bulk loader spreads a level's records evenly over its nodes, one
15021        // separator between adjacent siblings, which needs room for a few
15022        // records per node. HDF5's rank limit of 32 leaves room for seven; a
15023        // wider rank than that has no valid geometry, so reject it here rather
15024        // than emit a tree no reader can walk.
15025        let record_size = bt2_index.record_size(&self.ctx) as usize;
15026        let node_size = bt2_index.node_size as usize;
15027        if node_size < 10 + 3 * record_size {
15028            return Err(crate::io::IoError::InvalidState(format!(
15029                "a {ndims}-dimension v2 B-tree record is {record_size} bytes, too wide \
15030                 for a {node_size}-byte node"
15031            )));
15032        }
15033
15034        // Only the header gets a home now: it names an empty tree, whose root
15035        // is undefined until the first flush bulk-loads the index into nodes.
15036        let hdr = if bt2_index.filtered {
15037            Bt2Header::new_for_filtered_chunks(&self.ctx, ndims, bt2_index.chunk_size_len)
15038        } else {
15039            Bt2Header::new_for_chunks(&self.ctx, ndims)
15040        };
15041        let hdr_encoded = hdr.encode(&self.ctx);
15042        let bt2_header_addr = self
15043            .allocator
15044            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
15045        self.handle.write_at(bt2_header_addr, &hdr_encoded)?;
15046
15047        let dataspace = DataspaceMessage {
15048            // Chunked storage always requires at least one dimension, so
15049            // this is never Scalar or Null.
15050            class: DataspaceClass::Simple,
15051            dims: dims.to_vec(),
15052            max_dims: Some(max_dims.to_vec()),
15053        };
15054
15055        let idx = self.push_dataset(
15056            &create,
15057            DatasetInfo {
15058                name: name.to_string(),
15059                datatype,
15060                committed_type: None,
15061                external: None,
15062                virtual_storage: None,
15063                dataspace,
15064                read_format: None,
15065                obj_header_addr: 0,
15066                data_addr: UNDEF_ADDR,
15067                data_size: 0,
15068                compact: None,
15069                attributes: Vec::new(),
15070                obj_header_written_addr: None,
15071                obj_header_blocks: Vec::new(),
15072                filter_pipeline: pipeline,
15073                deleted: false,
15074                extent_dirty: false,
15075                header_dirty: false,
15076                nlink_written: 1,
15077                creation_seq: self.take_creation_seq(),
15078                track_attr_order: self.track_order.attrs,
15079                fill_value: None,
15080                fill_time: FILL_TIME_IFSET,
15081                layout_version,
15082                times: self.created_object_times(),
15083                chunked: None,
15084                fixed_array: None,
15085                implicit: None,
15086                single_chunk: None,
15087                btree_v1: None,
15088                btree_v2: Some(Bt2DatasetInfo {
15089                    chunk_dims: chunk_dims.to_vec(),
15090                    bt2_header_addr,
15091                    node_addrs: Vec::new(),
15092                    index: bt2_index,
15093                    chunks_written: 0,
15094                }),
15095                append: None,
15096            },
15097        );
15098
15099        Ok(idx)
15100    }
15101
15102    /// Create a chunked dataset with a custom filter pipeline.
15103    pub fn create_chunked_dataset_with_pipeline(
15104        &self,
15105        name: &str,
15106        datatype: DatatypeMessage,
15107        dims: &[u64],
15108        max_dims: &[u64],
15109        chunk_dims: &[u64],
15110        pipeline: FilterPipeline,
15111    ) -> IoResult<usize> {
15112        let create = self.begin_create(name)?;
15113        let name = create.name.as_str();
15114        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
15115        ensure_at_most_one_unlimited(max_dims)?;
15116        let element_size = datatype.element_size() as u64;
15117        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * element_size;
15118        let layout_version = self.chunk_layout_version(true, chunk_bytes);
15119        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
15120
15121        let earray_params = EarrayParams::default_params();
15122        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
15123        let nsblk_addrs = compute_nsblk_addrs(
15124            earray_params.idx_blk_elmts,
15125            earray_params.data_blk_min_elmts,
15126            earray_params.sup_blk_min_data_ptrs,
15127            earray_params.max_nelmts_bits,
15128        )?;
15129
15130        let mut ea_header =
15131            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
15132        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
15133        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
15134        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
15135        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
15136        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
15137
15138        let hdr_encoded = ea_header.encode(&self.ctx);
15139        let ea_header_addr = self
15140            .allocator
15141            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
15142
15143        let filt_iblk = FilteredIndexBlock::new(
15144            ea_header_addr,
15145            earray_params.idx_blk_elmts,
15146            ndblk_addrs,
15147            nsblk_addrs,
15148        );
15149        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
15150        let ea_iblk_addr = self
15151            .allocator
15152            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
15153
15154        ea_header.idx_blk_addr = ea_iblk_addr;
15155        let hdr_encoded = ea_header.encode(&self.ctx);
15156        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
15157        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
15158
15159        let dataspace = DataspaceMessage {
15160            // Chunked storage always requires at least one dimension, so
15161            // this is never Scalar or Null.
15162            class: DataspaceClass::Simple,
15163            dims: dims.to_vec(),
15164            max_dims: Some(max_dims.to_vec()),
15165        };
15166        let ea_iblk = ExtensibleArrayIndexBlock::new(
15167            ea_header_addr,
15168            earray_params.idx_blk_elmts,
15169            ndblk_addrs,
15170            nsblk_addrs,
15171        );
15172
15173        let idx = self.push_dataset(
15174            &create,
15175            DatasetInfo {
15176                name: name.to_string(),
15177                datatype,
15178                committed_type: None,
15179                external: None,
15180                virtual_storage: None,
15181                dataspace,
15182                read_format: None,
15183                obj_header_addr: 0,
15184                data_addr: UNDEF_ADDR,
15185                data_size: 0,
15186                compact: None,
15187                attributes: Vec::new(),
15188                obj_header_written_addr: None,
15189                obj_header_blocks: Vec::new(),
15190                filter_pipeline: Some(pipeline),
15191                deleted: false,
15192                extent_dirty: false,
15193                header_dirty: false,
15194                nlink_written: 1,
15195                creation_seq: self.take_creation_seq(),
15196                track_attr_order: self.track_order.attrs,
15197                fill_value: None,
15198                fill_time: FILL_TIME_IFSET,
15199                layout_version,
15200                times: self.created_object_times(),
15201                fixed_array: None,
15202                implicit: None,
15203                single_chunk: None,
15204                btree_v1: None,
15205                btree_v2: None,
15206                chunked: Some(ChunkedDatasetInfo {
15207                    chunk_dims: chunk_dims.to_vec(),
15208                    earray_params,
15209                    ea_header_addr,
15210                    ea_iblk_addr,
15211                    ea_header,
15212                    ea_iblk,
15213                    chunks_written: 0,
15214                    filt_iblk: Some(filt_iblk),
15215                    chunk_size_len,
15216                }),
15217                append: None,
15218            },
15219        );
15220        Ok(idx)
15221    }
15222
15223    /// Write a chunk to a fixed-array-indexed dataset.
15224    ///
15225    /// `chunk_coords` is the multidimensional chunk index (e.g., [row_chunk, col_chunk]).
15226    /// The uncompressed `data` must be exactly one chunk wide; the filter
15227    /// pipeline (if any) runs here before the bytes reach the index.
15228    pub fn write_chunk_fixed_array(
15229        &self,
15230        index: usize,
15231        chunk_coords: &[u64],
15232        data: &[u8],
15233    ) -> IoResult<()> {
15234        let ds = self.ds(index);
15235        let _op = ds.op.lock();
15236        self.write_chunk_fixed_array_inner(index, chunk_coords, data)
15237    }
15238
15239    /// [`Self::write_chunk_fixed_array`] body; the caller holds the dataset's
15240    /// op lock or the writer exclusively.
15241    pub(crate) fn write_chunk_fixed_array_inner(
15242        &self,
15243        index: usize,
15244        chunk_coords: &[u64],
15245        data: &[u8],
15246    ) -> IoResult<()> {
15247        // Read what we need under one brief slot guard, then compress
15248        // OUTSIDE the lock: `record_fixed_array_chunk` re-locks the same slot,
15249        // so the guard must be dropped before it (and before apply_filters).
15250        let ds = self.ds(index);
15251        let (chunk_bytes, pipeline) = {
15252            let m = ds.lock();
15253            let element_size = m.datatype.element_size() as u64;
15254            let fa = m.fixed_array.as_ref().ok_or_else(|| {
15255                crate::io::IoError::InvalidState("not a fixed-array dataset".into())
15256            })?;
15257            (
15258                fa.chunk_dims.iter().product::<u64>() * element_size,
15259                m.filter_pipeline.clone(),
15260            )
15261        };
15262
15263        if data.len() as u64 != chunk_bytes {
15264            return Err(crate::io::IoError::InvalidState(format!(
15265                "chunk data size mismatch: expected {} bytes, got {}",
15266                chunk_bytes,
15267                data.len()
15268            )));
15269        }
15270        let write_data;
15271        let data_to_write = if let Some(ref pipeline) = pipeline {
15272            write_data = filter::apply_filters(pipeline, data)?;
15273            &write_data[..]
15274        } else {
15275            data
15276        };
15277        // filter_mask = 0: the whole pipeline ran (or the dataset is
15278        // unfiltered), so no filter is skipped for this chunk.
15279        self.record_fixed_array_chunk(index, chunk_coords, data_to_write, 0)
15280    }
15281
15282    /// Write a pre-filtered chunk verbatim to a fixed-array dataset, recording
15283    /// the caller-supplied `filter_mask`.
15284    ///
15285    /// The bytes are stored exactly as given (no filter pipeline is run); this
15286    /// is the fixed-array half of the HDF5 "direct chunk write"
15287    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15288    /// means filter *i* of the pipeline was **not** applied to this chunk and
15289    /// must be skipped on read; pass 0 when the full pipeline was applied
15290    /// upstream.
15291    ///
15292    /// Requires a filtered dataset — only the filtered FA element carries the
15293    /// size+mask slot.
15294    ///
15295    /// The caller holds the dataset's op lock or the writer exclusively.
15296    pub(crate) fn write_compressed_chunk_fixed_array_inner(
15297        &self,
15298        index: usize,
15299        chunk_coords: &[u64],
15300        data: &[u8],
15301        filter_mask: u32,
15302    ) -> IoResult<()> {
15303        if self.ds(index).lock().filter_pipeline.is_none() {
15304            return Err(crate::io::IoError::InvalidState(
15305                "write_compressed_chunk_fixed_array requires a filtered dataset \
15306                 (no slot for a compressed size or filter mask on an unfiltered \
15307                 chunk index)"
15308                    .into(),
15309            ));
15310        }
15311        self.record_fixed_array_chunk(index, chunk_coords, data, filter_mask)
15312    }
15313
15314    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
15315    /// filtered if the dataset is filtered, raw otherwise) into a fixed-array
15316    /// dataset's data block, recording the caller-supplied `filter_mask`.
15317    /// Shared by [`write_chunk_fixed_array`](Self::write_chunk_fixed_array)
15318    /// and [`write_compressed_chunk_fixed_array`](Self::write_compressed_chunk_fixed_array).
15319    fn record_fixed_array_chunk(
15320        &self,
15321        index: usize,
15322        chunk_coords: &[u64],
15323        final_bytes: &[u8],
15324        filter_mask: u32,
15325    ) -> IoResult<()> {
15326        // Hold one slot guard for the whole method; `self.allocator`/`self.handle`/
15327        // `self.ctx` below touch disjoint fields safe to use with the guard held.
15328        let ds = self.ds(index);
15329        let mut m = ds.lock();
15330        let is_filtered = m.filter_pipeline.is_some();
15331        let fa = m
15332            .fixed_array
15333            .as_ref()
15334            .ok_or_else(|| crate::io::IoError::InvalidState("not a fixed-array dataset".into()))?;
15335
15336        // Linear chunk index in the maximum-extent grid — the slot the fixed
15337        // array (sized from that grid at create) records the chunk under.
15338        let linear_idx = crate::io::chunk_grid::linear_index(
15339            &m.dataspace.dims,
15340            m.dataspace.max_dims.as_deref(),
15341            &fa.chunk_dims,
15342            chunk_coords,
15343        )?;
15344
15345        // Update the fixed array data block. The slot is read before the bytes
15346        // are placed so a rewrite can stay where it is (see `place_chunk`).
15347        let fa = m.fixed_array.as_mut().unwrap();
15348        let lidx = linear_idx as usize;
15349        if is_filtered {
15350            // Filtered FA: store address + stored size + filter mask. A
15351            // non-zero mask bit means "filter i was skipped for this chunk".
15352            let stored_size = final_bytes.len();
15353            // The stored size is encoded in the FA header's `chunk_size_len`-byte
15354            // field; libhdf5 errors if it does not fit (H5D_CHUNK_ENCODE_SIZE_CHECK)
15355            // rather than truncating silently. element_size = sizeof_addr +
15356            // chunk_size_len + 4 by construction.
15357            let chunk_size_len = (fa.fa_header.element_size as usize)
15358                .checked_sub(self.ctx.sizeof_addr as usize + 4)
15359                .ok_or_else(|| {
15360                    crate::io::IoError::InvalidState(
15361                        "filtered fixed-array element size is too small".into(),
15362                    )
15363                })?;
15364            if chunk_size_len < 8 && stored_size >= (1usize << (chunk_size_len * 8)) {
15365                return Err(crate::io::IoError::InvalidState(format!(
15366                    "compressed chunk size {stored_size} does not fit in the \
15367                     {chunk_size_len}-byte fixed-array chunk-size field"
15368                )));
15369            }
15370            if lidx < fa.fa_dblk.filtered_elements.len() {
15371                let old = &fa.fa_dblk.filtered_elements[lidx];
15372                let chunk_addr =
15373                    self.place_chunk(Some((old.address, old.chunk_size)), stored_size as u64);
15374                self.handle.write_at(chunk_addr, final_bytes)?;
15375                fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
15376                    address: chunk_addr,
15377                    chunk_size: stored_size as u64,
15378                    filter_mask,
15379                };
15380                fa.chunks_written += 1;
15381            } else {
15382                return Err(crate::io::IoError::InvalidState(format!(
15383                    "chunk index {} out of range (max {})",
15384                    linear_idx,
15385                    fa.fa_dblk.filtered_elements.len()
15386                )));
15387            }
15388        } else {
15389            // An unfiltered fixed array stores only addresses — there is no
15390            // slot for a filter mask, so a non-zero mask cannot be honored.
15391            if filter_mask != 0 {
15392                return Err(crate::io::IoError::InvalidState(
15393                    "filter_mask is non-zero but the dataset is unfiltered".into(),
15394                ));
15395            }
15396            if lidx < fa.fa_dblk.elements.len() {
15397                // Unfiltered: the stored size is fixed by the chunk shape, so
15398                // a rewrite always fits its old block.
15399                let old = fa.fa_dblk.elements[lidx];
15400                let len = final_bytes.len() as u64;
15401                let chunk_addr = self.place_chunk(Some((old, len)), len);
15402                self.handle.write_at(chunk_addr, final_bytes)?;
15403                fa.fa_dblk.elements[lidx] = chunk_addr;
15404                fa.chunks_written += 1;
15405            } else {
15406                return Err(crate::io::IoError::InvalidState(format!(
15407                    "chunk index {} out of range (max {})",
15408                    linear_idx,
15409                    fa.fa_dblk.elements.len()
15410                )));
15411            }
15412        }
15413
15414        Ok(())
15415    }
15416
15417    /// Write the one chunk of a single-chunk indexed dataset.
15418    ///
15419    /// `chunk_coords` is validated against the grid the same way every other
15420    /// coordinate-addressed index does (`ChunkGeometry::linear_index`), even
15421    /// though the grid holds exactly one slot — this is what rejects an
15422    /// out-of-range coordinate instead of silently writing to that slot.
15423    /// `data` is the chunk's unfiltered bytes; the dataset's filter pipeline
15424    /// runs here if it has one.
15425    ///
15426    /// The caller holds the dataset's op lock or the writer exclusively.
15427    pub(crate) fn write_chunk_single_chunk_inner(
15428        &self,
15429        index: usize,
15430        chunk_coords: &[u64],
15431        data: &[u8],
15432    ) -> IoResult<()> {
15433        let geo = self.chunk_geometry(index)?;
15434        geo.linear_index(chunk_coords)?;
15435        let chunk_bytes = geo.chunk_bytes();
15436        if data.len() as u64 != chunk_bytes {
15437            return Err(crate::io::IoError::InvalidState(format!(
15438                "chunk data size mismatch: expected {} bytes, got {}",
15439                chunk_bytes,
15440                data.len()
15441            )));
15442        }
15443        let pipeline = self.ds(index).lock().filter_pipeline.clone();
15444        let write_data;
15445        let data_to_write = if let Some(ref pipeline) = pipeline {
15446            write_data = filter::apply_filters(pipeline, data)?;
15447            &write_data[..]
15448        } else {
15449            data
15450        };
15451        // filter_mask = 0: the whole pipeline ran (or the dataset is
15452        // unfiltered), so no filter is skipped for this chunk.
15453        self.record_single_chunk(index, data_to_write, 0)
15454    }
15455
15456    /// Write a pre-filtered chunk verbatim to a single-chunk dataset,
15457    /// recording the caller-supplied `filter_mask`.
15458    ///
15459    /// The bytes are stored exactly as given (no filter pipeline is run); this
15460    /// is the single-chunk half of the HDF5 "direct chunk write"
15461    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15462    /// means filter *i* of the pipeline was **not** applied to this chunk and
15463    /// must be skipped on read; pass 0 when the full pipeline was applied
15464    /// upstream.
15465    ///
15466    /// Requires a filtered dataset — only the filtered single-chunk layout
15467    /// carries a size+mask slot.
15468    ///
15469    /// The caller holds the dataset's op lock or the writer exclusively.
15470    pub(crate) fn write_compressed_chunk_single_chunk_inner(
15471        &self,
15472        index: usize,
15473        chunk_coords: &[u64],
15474        data: &[u8],
15475        filter_mask: u32,
15476    ) -> IoResult<()> {
15477        if self.ds(index).lock().filter_pipeline.is_none() {
15478            return Err(crate::io::IoError::InvalidState(
15479                "write_compressed_chunk_single_chunk requires a filtered dataset \
15480                 (no slot for a compressed size or filter mask on an unfiltered \
15481                 chunk index)"
15482                    .into(),
15483            ));
15484        }
15485        let geo = self.chunk_geometry(index)?;
15486        geo.linear_index(chunk_coords)?;
15487        self.record_single_chunk(index, data, filter_mask)
15488    }
15489
15490    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
15491    /// filtered if the dataset is filtered, raw otherwise) into a single-chunk
15492    /// dataset's layout message fields, recording the caller-supplied
15493    /// `filter_mask`. Shared by
15494    /// [`write_chunk_single_chunk_inner`](Self::write_chunk_single_chunk_inner)
15495    /// and
15496    /// [`write_compressed_chunk_single_chunk_inner`](Self::write_compressed_chunk_single_chunk_inner).
15497    ///
15498    /// Unlike the array indexes there is no per-chunk slot to look up — the
15499    /// dataset has exactly one chunk, and its address/size/mask live directly
15500    /// in the layout message (`H5Dsingle.c`) — so this only ever rewrites the
15501    /// one chunk in place, via [`place_chunk`](Self::place_chunk) the same as
15502    /// every other index's rewrite path.
15503    fn record_single_chunk(
15504        &self,
15505        index: usize,
15506        final_bytes: &[u8],
15507        filter_mask: u32,
15508    ) -> IoResult<()> {
15509        let ds = self.ds(index);
15510        let mut m = ds.lock();
15511        let is_filtered = m.filter_pipeline.is_some();
15512        if !is_filtered && filter_mask != 0 {
15513            return Err(crate::io::IoError::InvalidState(
15514                "filter_mask is non-zero but the dataset is unfiltered".into(),
15515            ));
15516        }
15517        let sc = m
15518            .single_chunk
15519            .as_ref()
15520            .ok_or_else(|| crate::io::IoError::InvalidState("not a single-chunk dataset".into()))?;
15521
15522        // A rewrite whose stored size is unchanged stays where it is (always
15523        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
15524        let old = if sc.data_addr == UNDEF_ADDR {
15525            None
15526        } else {
15527            Some((
15528                sc.data_addr,
15529                if is_filtered { sc.nbytes } else { sc.data_size },
15530            ))
15531        };
15532        let stored_size = final_bytes.len() as u64;
15533        let addr = self.place_chunk(old, stored_size);
15534        self.handle.write_at(addr, final_bytes)?;
15535
15536        let sc = m.single_chunk.as_mut().unwrap();
15537        sc.data_addr = addr;
15538        sc.nbytes = stored_size;
15539        sc.filter_mask = filter_mask;
15540        sc.chunks_written = 1;
15541        Ok(())
15542    }
15543
15544    /// Write a chunk to a B-tree v2 indexed dataset.
15545    ///
15546    /// `chunk_coords` is the scaled chunk coordinates (one per dimension).
15547    /// `data` is the chunk's unfiltered bytes; if the dataset has a filter
15548    /// pipeline it runs here and the index records the stored size and mask.
15549    ///
15550    /// Production writes call [`write_chunk_btree_v2_inner`](Self::write_chunk_btree_v2_inner)
15551    /// directly (they already hold the dataset's op lock); this self-locking
15552    /// form is kept as a direct entry point for this crate's own white-box
15553    /// tests.
15554    #[cfg(test)]
15555    pub fn write_chunk_btree_v2(
15556        &self,
15557        index: usize,
15558        chunk_coords: &[u64],
15559        data: &[u8],
15560    ) -> IoResult<()> {
15561        let ds = self.ds(index);
15562        let _op = ds.op.lock();
15563        self.write_chunk_btree_v2_inner(index, chunk_coords, data)
15564    }
15565
15566    /// [`Self::write_chunk_btree_v2`] body; the caller holds the dataset's op
15567    /// lock or the writer exclusively.
15568    pub(crate) fn write_chunk_btree_v2_inner(
15569        &self,
15570        index: usize,
15571        chunk_coords: &[u64],
15572        data: &[u8],
15573    ) -> IoResult<()> {
15574        // Read what the write needs under a brief guard, then compress OUTSIDE
15575        // the lock — filtering a chunk must not hold the dataset slot.
15576        let ds = self.ds(index);
15577        let (chunk_bytes, pipeline) = {
15578            let m = ds.lock();
15579            let element_size = m.datatype.element_size() as u64;
15580            let bt2 = m.btree_v2.as_ref().ok_or_else(|| {
15581                crate::io::IoError::InvalidState("not a B-tree v2 dataset".into())
15582            })?;
15583            (
15584                bt2.chunk_dims.iter().product::<u64>() * element_size,
15585                m.filter_pipeline.clone(),
15586            )
15587        };
15588
15589        if data.len() as u64 != chunk_bytes {
15590            return Err(crate::io::IoError::InvalidState(format!(
15591                "chunk data size mismatch: expected {} bytes, got {}",
15592                chunk_bytes,
15593                data.len()
15594            )));
15595        }
15596
15597        let filtered;
15598        let stored = match pipeline {
15599            Some(ref pl) => {
15600                filtered = filter::apply_filters(pl, data)?;
15601                &filtered[..]
15602            }
15603            None => data,
15604        };
15605
15606        // filter_mask = 0: the whole pipeline ran (or the dataset is
15607        // unfiltered), so no filter is skipped.
15608        self.record_btree_v2_chunk(index, chunk_coords, stored, 0)
15609    }
15610
15611    /// Write a pre-filtered chunk verbatim to a BT2-indexed dataset, recording
15612    /// the caller-supplied `filter_mask`.
15613    ///
15614    /// The v2-B-tree half of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
15615    /// The bytes are stored exactly as given; `filter_mask` bit *i* set means
15616    /// filter *i* of the pipeline was **not** applied and must be skipped on
15617    /// read. Requires a filtered dataset — only a type-11 record has a slot for
15618    /// a stored size and mask.
15619    ///
15620    /// The caller holds the dataset's op lock or the writer exclusively.
15621    pub(crate) fn write_compressed_chunk_btree_v2_inner(
15622        &self,
15623        index: usize,
15624        chunk_coords: &[u64],
15625        data: &[u8],
15626        filter_mask: u32,
15627    ) -> IoResult<()> {
15628        if self.ds(index).lock().filter_pipeline.is_none() {
15629            return Err(crate::io::IoError::InvalidState(
15630                "write_compressed_chunk_btree_v2 requires a filtered dataset (no \
15631                 slot for a compressed size or filter mask on an unfiltered chunk \
15632                 index)"
15633                    .into(),
15634            ));
15635        }
15636        self.record_btree_v2_chunk(index, chunk_coords, data, filter_mask)
15637    }
15638
15639    /// Place a chunk's already-final bytes (filtered if the dataset is
15640    /// filtered, raw otherwise) in the file and record them in the v2 B-tree,
15641    /// under the caller-supplied `filter_mask`.
15642    ///
15643    /// Shared by [`write_chunk_btree_v2`](Self::write_chunk_btree_v2) and
15644    /// [`write_compressed_chunk_btree_v2`](Self::write_compressed_chunk_btree_v2),
15645    /// so both reach the index through one placement rule.
15646    fn record_btree_v2_chunk(
15647        &self,
15648        index: usize,
15649        chunk_coords: &[u64],
15650        final_bytes: &[u8],
15651        filter_mask: u32,
15652    ) -> IoResult<()> {
15653        let stored_len = final_bytes.len() as u64;
15654        let ds = self.ds(index);
15655        let mut m = ds.lock();
15656        let element_size = m.datatype.element_size() as u64;
15657        let bt2 = m
15658            .btree_v2
15659            .as_ref()
15660            .ok_or_else(|| crate::io::IoError::InvalidState("not a B-tree v2 dataset".into()))?;
15661        let chunk_bytes = bt2.chunk_dims.iter().product::<u64>() * element_size;
15662        // A filtered record encodes the stored size in a `chunk_size_len`-byte
15663        // field that truncates silently. Reject a size that would not fit, as
15664        // the extensible-array path does — the compress path never exceeds it,
15665        // but a direct write with caller-supplied bytes can.
15666        if bt2.index.filtered {
15667            let chunk_size_len = bt2.index.chunk_size_len as usize;
15668            if chunk_size_len < 8 && stored_len >= (1u64 << (chunk_size_len * 8)) {
15669                return Err(crate::io::IoError::InvalidState(format!(
15670                    "filtered chunk size {stored_len} does not fit in the \
15671                     {chunk_size_len}-byte v2 B-tree chunk-size field"
15672                )));
15673            }
15674        }
15675        // Place the bytes: a rewrite whose stored size is unchanged stays
15676        // where it is (always so when unfiltered — the size is fixed by the
15677        // chunk shape), and one that no longer fits moves, releasing its old
15678        // block. See `place_chunk`.
15679        let old = if bt2.index.filtered {
15680            bt2.index
15681                .lookup_filtered(chunk_coords)
15682                .map(|r| (r.chunk_address, r.chunk_size))
15683        } else {
15684            bt2.index
15685                .lookup(chunk_coords)
15686                .map(|r| (r.chunk_address, chunk_bytes))
15687        };
15688        let chunk_addr = self.place_chunk(old, stored_len);
15689        self.handle.write_at(chunk_addr, final_bytes)?;
15690
15691        let bt2 = m.btree_v2.as_mut().unwrap();
15692        if bt2.index.filtered {
15693            bt2.index
15694                .insert_filtered(chunk_coords.to_vec(), chunk_addr, stored_len, filter_mask);
15695        } else {
15696            bt2.index.insert(chunk_coords.to_vec(), chunk_addr);
15697        }
15698        bt2.chunks_written += 1;
15699
15700        Ok(())
15701    }
15702
15703    /// Write multiple chunks in a batch, optionally compressing in parallel.
15704    ///
15705    /// `chunks` is a list of (chunk_idx, data) pairs for an EA-indexed dataset.
15706    pub fn write_chunks_batch(&self, ds_index: usize, chunks: &[(u64, &[u8])]) -> IoResult<()> {
15707        let ds = self.ds(ds_index);
15708        let _op = ds.op.lock();
15709        self.write_chunks_batch_inner(ds_index, chunks)
15710    }
15711
15712    /// [`Self::write_chunks_batch`] body; the caller holds the dataset's op
15713    /// lock or the writer exclusively.
15714    pub(crate) fn write_chunks_batch_inner(
15715        &self,
15716        ds_index: usize,
15717        chunks: &[(u64, &[u8])],
15718    ) -> IoResult<()> {
15719        #[cfg(feature = "parallel")]
15720        {
15721            // If filter pipeline is set, compress all chunks in parallel.
15722            // Clone the pipeline out under a brief slot guard so the parallel
15723            // compression below runs off the lock.
15724            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
15725            if let Some(ref pipeline) = pipeline {
15726                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
15727                // Propagate a filter error rather than storing raw bytes under a
15728                // filter_mask that claims the pipeline ran (see
15729                // apply_filters_parallel). Ok reaching here means every chunk
15730                // compressed fully, so filter_mask = 0 is truthful.
15731                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15732                for ((idx, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15733                    self.write_compressed_chunk_inner(ds_index, *idx, compressed_data, 0)?;
15734                }
15735                return Ok(());
15736            }
15737        }
15738        // Fallback: sequential
15739        for (idx, data) in chunks {
15740            self.write_chunk_inner(ds_index, *idx, data)?;
15741        }
15742        Ok(())
15743    }
15744
15745    /// Write multiple fixed-array chunks in a batch, compressing them in
15746    /// parallel when a filter pipeline is set and the `parallel` feature is on.
15747    ///
15748    /// The fixed-array analogue of [`write_chunks_batch`](Self::write_chunks_batch):
15749    /// chunks are addressed by grid coordinates rather than a linear index.
15750    /// `record_fixed_array_chunk` writes already-compressed bytes verbatim, so
15751    /// the parallel compressor is the only place a filter runs. Falls back to
15752    /// per-chunk [`write_chunk_fixed_array`](Self::write_chunk_fixed_array) when
15753    /// unfiltered or when `parallel` is off.
15754    ///
15755    /// The caller holds the dataset's op lock or the writer exclusively.
15756    pub(crate) fn write_chunks_fixed_array_batch_inner(
15757        &self,
15758        ds_index: usize,
15759        chunks: &[(&[u64], &[u8])],
15760    ) -> IoResult<()> {
15761        #[cfg(feature = "parallel")]
15762        {
15763            // Clone the pipeline out under a brief slot guard so the parallel
15764            // compression below runs off the lock.
15765            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
15766            if let Some(ref pipeline) = pipeline {
15767                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
15768                // Same single owner as the EA batch: apply_filters_parallel
15769                // propagates a filter error instead of storing raw bytes under a
15770                // filter_mask that claims the pipeline ran. Ok here means every
15771                // chunk compressed fully, so filter_mask = 0 is truthful.
15772                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15773                for ((coords, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15774                    self.record_fixed_array_chunk(ds_index, coords, compressed_data, 0)?;
15775                }
15776                return Ok(());
15777            }
15778        }
15779        // Fallback: sequential (write_chunk_fixed_array_inner compresses per
15780        // chunk).
15781        for (coords, data) in chunks {
15782            self.write_chunk_fixed_array_inner(ds_index, coords, data)?;
15783        }
15784        Ok(())
15785    }
15786
15787    /// Write a pre-filtered chunk verbatim to an EA-indexed dataset, recording
15788    /// the caller-supplied `filter_mask`.
15789    ///
15790    /// The bytes are stored exactly as given (no filter pipeline is run); this
15791    /// is the extensible-array half of the HDF5 "direct chunk write"
15792    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15793    /// means filter *i* of the pipeline was **not** applied to this chunk and
15794    /// must be skipped on read; pass 0 when the full pipeline was applied
15795    /// upstream.
15796    ///
15797    /// Requires a filtered dataset — only the filtered EA entry carries the
15798    /// size+mask slot. An unfiltered dataset has nowhere to record either.
15799    ///
15800    /// The caller holds the dataset's op lock or the writer exclusively.
15801    pub(crate) fn write_compressed_chunk_inner(
15802        &self,
15803        index: usize,
15804        chunk_idx: u64,
15805        compressed_data: &[u8],
15806        filter_mask: u32,
15807    ) -> IoResult<()> {
15808        if self.ds(index).lock().filter_pipeline.is_none() {
15809            return Err(crate::io::IoError::InvalidState(
15810                "write_compressed_chunk requires a filtered dataset (no slot for \
15811                 a compressed size or filter mask on an unfiltered chunk index)"
15812                    .into(),
15813            ));
15814        }
15815        self.record_ea_chunk(index, chunk_idx, compressed_data, filter_mask)
15816    }
15817
15818    /// Extend the dimensions of a chunked dataset.
15819    pub fn extend_dataset(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15820        let ds = self.ds(index);
15821        let _op = ds.op.lock();
15822        self.extend_dataset_inner(index, new_dims)
15823    }
15824
15825    /// [`Self::extend_dataset`] body; the caller holds the dataset's op lock
15826    /// or the writer exclusively.
15827    pub(crate) fn extend_dataset_inner(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15828        let ds = self.ds(index);
15829        let mut m = ds.lock();
15830        if !m.is_chunked() {
15831            return Err(crate::io::IoError::InvalidState(
15832                "can only extend chunked datasets".into(),
15833            ));
15834        }
15835        if new_dims.len() != m.dataspace.dims.len() {
15836            return Err(crate::io::IoError::InvalidState(format!(
15837                "extend_dataset rank mismatch: dataset has {} dimensions, got {}",
15838                m.dataspace.dims.len(),
15839                new_dims.len()
15840            )));
15841        }
15842        // The chunk index and append buffers assume the logical size only
15843        // grows; shrinking below already-written data desynchronizes them.
15844        for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15845            if new < cur {
15846                return Err(crate::io::IoError::InvalidState(format!(
15847                    "extend_dataset cannot shrink dimension {d} from {cur} to {new}"
15848                )));
15849            }
15850            // An absent maximum shape means the shape is fixed (libhdf5
15851            // defaults maxdims to dims at creation), so any growth exceeds it.
15852            match m.dataspace.max_dims {
15853                Some(ref max) if new > max[d] => {
15854                    return Err(crate::io::IoError::InvalidState(format!(
15855                        "extend_dataset dimension {d} ({new}) exceeds the maximum {}",
15856                        max[d]
15857                    )));
15858                }
15859                None if new > cur => {
15860                    return Err(crate::io::IoError::InvalidState(format!(
15861                        "extend_dataset dimension {d} ({new}) exceeds the maximum {cur}: \
15862                         a dataset without a stored maximum shape is fixed at its extent"
15863                    )));
15864                }
15865                _ => {}
15866            }
15867        }
15868        if m.dataspace.dims != new_dims {
15869            m.dataspace.dims = new_dims.to_vec();
15870            m.extent_dirty = true;
15871        }
15872        Ok(())
15873    }
15874
15875    /// Set the logical extent of a chunked dataset, growing **or shrinking**
15876    /// any dimension (unlike [`extend_dataset`](Self::extend_dataset), which
15877    /// only grows).
15878    ///
15879    /// A shrink prunes the stored chunks the way libhdf5's
15880    /// `H5D__chunk_prune_by_extent` (H5Dchunk.c) does: a chunk entirely
15881    /// beyond the new extent leaves the chunk index and its block is freed
15882    /// for reuse (kept under SWMR, where a live reader may still hold its
15883    /// address — the rule `H5Dearray.c` applies in `idx_remove`), and a
15884    /// chunk the new extent cuts through has its out-of-extent region
15885    /// overwritten with the fill value, so growing the extent back exposes
15886    /// fill values rather than the stale data.
15887    pub fn set_dataset_extent(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15888        let ds = self.ds(index);
15889        let _op = ds.op.lock();
15890        let old_dims = {
15891            let m = ds.lock();
15892            if !m.is_chunked() {
15893                return Err(crate::io::IoError::InvalidState(
15894                    "can only set the extent of chunked datasets".into(),
15895                ));
15896            }
15897            if new_dims.len() != m.dataspace.dims.len() {
15898                return Err(crate::io::IoError::InvalidState(format!(
15899                    "set_extent rank mismatch: dataset has {} dimensions, got {}",
15900                    m.dataspace.dims.len(),
15901                    new_dims.len()
15902                )));
15903            }
15904            // A shrink can cut into buffered rows, whose recorded base would
15905            // then point past the extent; refuse rather than reconcile.
15906            if m.append.is_some() {
15907                return Err(crate::io::IoError::InvalidState(
15908                    "set_extent cannot run while the dataset has buffered appends; \
15909                     flush them first"
15910                        .into(),
15911                ));
15912            }
15913            // An absent maximum shape means the shape is fixed (libhdf5
15914            // defaults maxdims to dims at creation), so growth is bounded by
15915            // the extent.
15916            match m.dataspace.max_dims {
15917                Some(ref max) => {
15918                    for (d, (&new, &mx)) in new_dims.iter().zip(max).enumerate() {
15919                        if new > mx {
15920                            return Err(crate::io::IoError::InvalidState(format!(
15921                                "set_extent dimension {d} ({new}) exceeds the maximum {mx}"
15922                            )));
15923                        }
15924                    }
15925                }
15926                None => {
15927                    for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15928                        if new > cur {
15929                            return Err(crate::io::IoError::InvalidState(format!(
15930                                "set_extent dimension {d} ({new}) exceeds the maximum {cur}: \
15931                                 a dataset without a stored maximum shape is fixed at its extent"
15932                            )));
15933                        }
15934                    }
15935                }
15936            }
15937            m.dataspace.dims.clone()
15938        };
15939        // A shrink strands chunks; prune them (and refill the straddlers)
15940        // *before* the dims update — chunk addressing uses the
15941        // maximum-extent grid, which the update does not change, and the
15942        // helpers re-lock the slot themselves.
15943        if new_dims.iter().zip(&old_dims).any(|(&n, &o)| n < o) {
15944            self.prune_chunks_beyond(index, new_dims)?;
15945        }
15946        let mut m = ds.lock();
15947        if m.dataspace.dims != new_dims {
15948            m.dataspace.dims = new_dims.to_vec();
15949            m.extent_dirty = true;
15950        }
15951        Ok(())
15952    }
15953
15954    /// Remove and refill the chunks a shrink to `new_dims` strands — the
15955    /// libhdf5 `H5D__chunk_prune_by_extent` behavior. A chunk entirely
15956    /// beyond the new extent leaves the index and its block is freed (kept
15957    /// under SWMR, where a live reader may still hold its address); a chunk
15958    /// the extent cuts through gets its out-of-extent region refilled with
15959    /// the fill value, so a later regrow reads fill, not stale elements.
15960    ///
15961    /// Runs *before* the dims update: the index grid chunks are addressed in
15962    /// comes from the maximum extent, which a shrink never changes, so every
15963    /// stored entry still resolves. The caller holds the dataset's op lock.
15964    fn prune_chunks_beyond(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15965        let geo = self.chunk_geometry(index)?;
15966        // A vlen dataset's elements are global-heap IDs: the pruned chunks
15967        // still reference live heap objects, so the walkers read each dead
15968        // chunk's bytes before freeing its block and the heap objects are
15969        // released here — otherwise every shrink strands its strings in the
15970        // file. `release_vlen_references` is a SWMR no-op, so the reads are
15971        // skipped under SWMR too.
15972        let collect_refs = !self.swmr_active && {
15973            let ds = self.ds(index);
15974            let m = ds.lock();
15975            matches!(
15976                m.datatype,
15977                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
15978            )
15979        };
15980        let (straddlers, dead_refs) = match geo.kind {
15981            ChunkIndexKind::ExtensibleArray => {
15982                self.prune_ea_chunks(index, &geo, new_dims, collect_refs)?
15983            }
15984            ChunkIndexKind::FixedArray => {
15985                self.prune_fa_chunks(index, &geo, new_dims, collect_refs)?
15986            }
15987            ChunkIndexKind::BtreeV2 => {
15988                self.prune_bt2_chunks(index, &geo, new_dims, collect_refs)?
15989            }
15990            // Removing a chunk from the implicit index is
15991            // `H5D__none_idx_remove`: a no-op, because the chunk's space is
15992            // the dataset's space and stays allocated either way. Only the
15993            // straddlers matter, and they are refilled by the caller.
15994            ChunkIndexKind::Implicit => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15995            // A single-chunk index has no per-chunk remove either — its one
15996            // chunk's address lives in the layout message, not an index
15997            // structure, and stays exactly where it is; a shrink only ever
15998            // straddles that one chunk (`H5D__single_idx_remove` is likewise
15999            // a no-op).
16000            ChunkIndexKind::SingleChunk => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
16001            ChunkIndexKind::BtreeV1 => {
16002                self.prune_btree_v1_chunks(index, &geo, new_dims, collect_refs)?
16003            }
16004        };
16005        if !dead_refs.is_empty() {
16006            self.release_vlen_references(&dead_refs)?;
16007        }
16008        // Whole-chunk read-modify-write per straddler: an unfiltered chunk
16009        // rewrites in place, a filtered one re-places through `place_chunk`.
16010        let chunk_bytes = geo.chunk_bytes() as usize;
16011        for coords in straddlers {
16012            let Some(mut data) = self.read_chunk_at_coords(index, &coords)? else {
16013                continue;
16014            };
16015            let fill = self.new_chunk_buffer(index, chunk_bytes);
16016            let replaced = refill_chunk_beyond_extent(
16017                &mut data,
16018                &fill,
16019                &coords,
16020                &geo.chunk_dims,
16021                new_dims,
16022                geo.element_size as usize,
16023            );
16024            // Release before the write-back: a filtered straddler re-places
16025            // its block, and freed heap space must be visible to that
16026            // allocation (free-before-alloc, as everywhere else).
16027            if collect_refs && !replaced.is_empty() {
16028                self.release_vlen_references(&replaced)?;
16029            }
16030            self.write_chunk_at_coords(index, &coords, &data)?;
16031        }
16032        Ok(())
16033    }
16034
16035    /// Extensible-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16036    /// walk every slot the array has ever set, free and clear the entries of
16037    /// chunks entirely beyond `new_dims`, and return the grid coordinates of
16038    /// the chunks that straddle it, plus — when `collect_refs` — the dead
16039    /// chunks' element bytes so the caller can release their heap objects.
16040    fn prune_ea_chunks(
16041        &self,
16042        index: usize,
16043        geo: &ChunkGeometry,
16044        new_dims: &[u64],
16045        collect_refs: bool,
16046    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16047        let ds = self.ds(index);
16048        // One slot guard for the whole walk, the `record_ea_chunk` pattern:
16049        // `self.handle`/`self.allocator`/`self.ctx` are disjoint fields.
16050        let mut m = ds.lock();
16051        let is_filtered = m.filter_pipeline.is_some();
16052        let pipeline = m.filter_pipeline.clone();
16053        let chunk_bytes = geo.chunk_bytes();
16054        let (ea_geo, max_nelmts_bits, chunk_size_len, max_idx) = {
16055            let c = m.chunked.as_ref().unwrap();
16056            let p = &c.earray_params;
16057            (
16058                EaGeometry::new(
16059                    p.idx_blk_elmts,
16060                    p.data_blk_min_elmts,
16061                    p.sup_blk_min_data_ptrs,
16062                    p.max_nelmts_bits,
16063                    p.max_dblk_page_nelmts_bits,
16064                )?,
16065                p.max_nelmts_bits,
16066                c.chunk_size_len,
16067                c.ea_header.max_idx_set,
16068            )
16069        };
16070
16071        let mut straddlers = Vec::new();
16072        let mut dead_refs = Vec::new();
16073
16074        // The decoded data block the walk is currently inside, written back
16075        // when the walk leaves it (or ends) having cleared an entry.
16076        enum Dblk {
16077            Unfiltered(ExtensibleArrayDataBlock),
16078            Filtered(FilteredDataBlock),
16079        }
16080        let mut cache: Option<(u64, Dblk, bool)> = None;
16081        let flush = |cache: &mut Option<(u64, Dblk, bool)>| -> IoResult<()> {
16082            if let Some((addr, blk, dirty)) = cache.take() {
16083                if dirty {
16084                    let enc = match &blk {
16085                        Dblk::Unfiltered(d) => d.encode(&self.ctx, max_nelmts_bits),
16086                        Dblk::Filtered(d) => d.encode(&self.ctx, max_nelmts_bits, chunk_size_len),
16087                    };
16088                    self.handle.write_at(addr, &enc)?;
16089                }
16090            }
16091            Ok(())
16092        };
16093        // Consecutive slots resolve through the same super block, so keep
16094        // the last decode. Super blocks are only read here — clearing a
16095        // data-block element never moves the block — so it never dirties.
16096        let mut sblk_cache: Option<(usize, ExtensibleArraySuperBlock)> = None;
16097
16098        let mut slot = 0u64;
16099        while slot < max_idx {
16100            let coords = crate::io::chunk_grid::coords_of(
16101                &geo.dims,
16102                geo.max_dims.as_deref(),
16103                &geo.chunk_dims,
16104                slot,
16105            )?;
16106            if !chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
16107                if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16108                    straddlers.push(coords);
16109                }
16110                slot += 1;
16111                continue;
16112            }
16113            match ea_geo.locate(slot)? {
16114                EaLoc::Index { elem } => {
16115                    let c = m.chunked.as_mut().unwrap();
16116                    if is_filtered {
16117                        let fiblk = c.filt_iblk.as_mut().unwrap();
16118                        let e = fiblk.elements[elem];
16119                        if e.addr != UNDEF_ADDR {
16120                            if collect_refs {
16121                                if let Some(bytes) = self.read_chunk_block(
16122                                    pipeline.as_ref(),
16123                                    e.addr,
16124                                    e.nbytes,
16125                                    e.filter_mask,
16126                                )? {
16127                                    dead_refs.extend_from_slice(&bytes);
16128                                }
16129                            }
16130                            if !self.swmr_active {
16131                                self.allocator
16132                                    .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
16133                            }
16134                            fiblk.elements[elem] = FilteredChunkEntry {
16135                                addr: UNDEF_ADDR,
16136                                nbytes: 0,
16137                                filter_mask: 0,
16138                            };
16139                        }
16140                    } else {
16141                        let a = c.ea_iblk.elements[elem];
16142                        if a != UNDEF_ADDR {
16143                            if collect_refs {
16144                                if let Some(bytes) =
16145                                    self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
16146                                {
16147                                    dead_refs.extend_from_slice(&bytes);
16148                                }
16149                            }
16150                            if !self.swmr_active {
16151                                self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
16152                            }
16153                            c.ea_iblk.elements[elem] = UNDEF_ADDR;
16154                        }
16155                    }
16156                    slot += 1;
16157                }
16158                EaLoc::Dblk(l) => {
16159                    if l.paged {
16160                        return Err(crate::io::IoError::InvalidState(format!(
16161                            "chunk index {slot} lives in a paged extensible-array \
16162                             data block, which is not yet supported"
16163                        )));
16164                    }
16165                    let dblk_start = slot - l.offset_in_dblk;
16166                    let dblk_end = dblk_start + l.dblk_nelmts;
16167                    // Resolve the data block's address; an undefined super or
16168                    // data block means nothing in its whole element range was
16169                    // ever written, so the walk skips the range.
16170                    let dblk_addr = {
16171                        let c = m.chunked.as_ref().unwrap();
16172                        match l.path {
16173                            EaDblkPath::Direct { idx } => {
16174                                if is_filtered {
16175                                    c.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
16176                                } else {
16177                                    c.ea_iblk.dblk_addrs[idx]
16178                                }
16179                            }
16180                            EaDblkPath::ViaSblk {
16181                                sblk_off,
16182                                local_dblk,
16183                                ndblks_in_sblk,
16184                                ..
16185                            } => {
16186                                let sblk_addr = if is_filtered {
16187                                    c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
16188                                } else {
16189                                    c.ea_iblk.sblk_addrs[sblk_off]
16190                                };
16191                                if sblk_addr == UNDEF_ADDR {
16192                                    UNDEF_ADDR
16193                                } else {
16194                                    if sblk_cache.as_ref().map(|&(o, _)| o) != Some(sblk_off) {
16195                                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
16196                                        let sb = ExtensibleArraySuperBlock::decode(
16197                                            &buf,
16198                                            &self.ctx,
16199                                            max_nelmts_bits,
16200                                            ndblks_in_sblk,
16201                                            0,
16202                                        )?;
16203                                        sblk_cache = Some((sblk_off, sb));
16204                                    }
16205                                    sblk_cache.as_ref().unwrap().1.dblk_addrs[local_dblk]
16206                                }
16207                            }
16208                        }
16209                    };
16210                    if dblk_addr == UNDEF_ADDR {
16211                        slot = dblk_end;
16212                        continue;
16213                    }
16214                    if cache.as_ref().map(|&(a, _, _)| a) != Some(dblk_addr) {
16215                        flush(&mut cache)?;
16216                        let buf = self.handle.read_at_most(dblk_addr, 65536)?;
16217                        let blk = if is_filtered {
16218                            Dblk::Filtered(FilteredDataBlock::decode(
16219                                &buf,
16220                                &self.ctx,
16221                                max_nelmts_bits,
16222                                l.dblk_nelmts as usize,
16223                                chunk_size_len,
16224                            )?)
16225                        } else {
16226                            Dblk::Unfiltered(ExtensibleArrayDataBlock::decode(
16227                                &buf,
16228                                &self.ctx,
16229                                max_nelmts_bits,
16230                                l.dblk_nelmts as usize,
16231                            )?)
16232                        };
16233                        cache = Some((dblk_addr, blk, false));
16234                    }
16235                    let (_, blk, dirty) = cache.as_mut().unwrap();
16236                    match blk {
16237                        Dblk::Filtered(d) => {
16238                            let e = d.elements[l.offset_in_dblk as usize];
16239                            if e.addr != UNDEF_ADDR {
16240                                if collect_refs {
16241                                    if let Some(bytes) = self.read_chunk_block(
16242                                        pipeline.as_ref(),
16243                                        e.addr,
16244                                        e.nbytes,
16245                                        e.filter_mask,
16246                                    )? {
16247                                        dead_refs.extend_from_slice(&bytes);
16248                                    }
16249                                }
16250                                if !self.swmr_active {
16251                                    self.allocator
16252                                        .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
16253                                }
16254                                d.elements[l.offset_in_dblk as usize] = FilteredChunkEntry {
16255                                    addr: UNDEF_ADDR,
16256                                    nbytes: 0,
16257                                    filter_mask: 0,
16258                                };
16259                                *dirty = true;
16260                            }
16261                        }
16262                        Dblk::Unfiltered(d) => {
16263                            let a = d.elements[l.offset_in_dblk as usize];
16264                            if a != UNDEF_ADDR {
16265                                if collect_refs {
16266                                    if let Some(bytes) =
16267                                        self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
16268                                    {
16269                                        dead_refs.extend_from_slice(&bytes);
16270                                    }
16271                                }
16272                                if !self.swmr_active {
16273                                    self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
16274                                }
16275                                d.elements[l.offset_in_dblk as usize] = UNDEF_ADDR;
16276                                *dirty = true;
16277                            }
16278                        }
16279                    }
16280                    slot += 1;
16281                }
16282            }
16283        }
16284        flush(&mut cache)?;
16285        Ok((straddlers, dead_refs))
16286    }
16287
16288    /// Fixed-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16289    /// the whole element array is in memory and flushed at close, so
16290    /// clearing an entry is pure bookkeeping.
16291    fn prune_fa_chunks(
16292        &self,
16293        index: usize,
16294        geo: &ChunkGeometry,
16295        new_dims: &[u64],
16296        collect_refs: bool,
16297    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16298        let ds = self.ds(index);
16299        let mut m = ds.lock();
16300        let is_filtered = m.filter_pipeline.is_some();
16301        let pipeline = m.filter_pipeline.clone();
16302        let chunk_bytes = geo.chunk_bytes();
16303        let mut straddlers = Vec::new();
16304        let mut dead_refs = Vec::new();
16305        let fa = m.fixed_array.as_mut().unwrap();
16306        let nslots = if is_filtered {
16307            fa.fa_dblk.filtered_elements.len()
16308        } else {
16309            fa.fa_dblk.elements.len()
16310        };
16311        for lidx in 0..nslots {
16312            let (addr, stored, mask) = if is_filtered {
16313                let e = &fa.fa_dblk.filtered_elements[lidx];
16314                (e.address, e.chunk_size, e.filter_mask)
16315            } else {
16316                (fa.fa_dblk.elements[lidx], chunk_bytes, 0)
16317            };
16318            if addr == UNDEF_ADDR {
16319                continue;
16320            }
16321            let coords = crate::io::chunk_grid::coords_of(
16322                &geo.dims,
16323                geo.max_dims.as_deref(),
16324                &geo.chunk_dims,
16325                lidx as u64,
16326            )?;
16327            if chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
16328                if collect_refs {
16329                    if let Some(bytes) =
16330                        self.read_chunk_block(pipeline.as_ref(), addr, stored, mask)?
16331                    {
16332                        dead_refs.extend_from_slice(&bytes);
16333                    }
16334                }
16335                if !self.swmr_active {
16336                    self.allocator.free(addr, stored, FreeSpaceClass::RawData);
16337                }
16338                if is_filtered {
16339                    fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
16340                        address: UNDEF_ADDR,
16341                        chunk_size: 0,
16342                        filter_mask: 0,
16343                    };
16344                } else {
16345                    fa.fa_dblk.elements[lidx] = UNDEF_ADDR;
16346                }
16347            } else if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16348                straddlers.push(coords);
16349            }
16350        }
16351        Ok((straddlers, dead_refs))
16352    }
16353
16354    /// Implicit half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16355    /// the grid coordinates of the chunks a shrink to `new_dims` cuts
16356    /// through. Nothing is freed or cleared — this index has no per-chunk
16357    /// state to clear and no per-chunk block to free — so the chunks wholly
16358    /// beyond the extent keep their bytes, exactly as `H5D__none_idx_remove`
16359    /// leaves them. That also means their elements stay reachable, so a
16360    /// variable-length dataset's heap objects must *not* be released here.
16361    fn implicit_straddlers(
16362        &self,
16363        geo: &ChunkGeometry,
16364        new_dims: &[u64],
16365    ) -> IoResult<Vec<Vec<u64>>> {
16366        let mut nchunks: u64 = 1;
16367        for g in
16368            crate::io::chunk_grid::index_grid(&geo.dims, geo.max_dims.as_deref(), &geo.chunk_dims)?
16369        {
16370            nchunks = nchunks.checked_mul(g).ok_or_else(|| {
16371                crate::io::IoError::InvalidState("chunk count overflows u64".into())
16372            })?;
16373        }
16374        let mut straddlers = Vec::new();
16375        for lidx in 0..nchunks {
16376            let coords = crate::io::chunk_grid::coords_of(
16377                &geo.dims,
16378                geo.max_dims.as_deref(),
16379                &geo.chunk_dims,
16380                lidx,
16381            )?;
16382            if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16383                straddlers.push(coords);
16384            }
16385        }
16386        Ok(straddlers)
16387    }
16388
16389    /// V2-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16390    /// drop the records of chunks beyond the extent — the next flush
16391    /// re-serializes the smaller tree over the node pool and releases the
16392    /// surplus node blocks.
16393    fn prune_bt2_chunks(
16394        &self,
16395        index: usize,
16396        geo: &ChunkGeometry,
16397        new_dims: &[u64],
16398        collect_refs: bool,
16399    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16400        let ds = self.ds(index);
16401        let mut m = ds.lock();
16402        let pipeline = m.filter_pipeline.clone();
16403        let chunk_bytes = geo.chunk_bytes();
16404        let swmr = self.swmr_active;
16405        let mut straddlers = Vec::new();
16406        let mut dead_refs = Vec::new();
16407        let bt2 = m.btree_v2.as_mut().unwrap();
16408        if bt2.index.filtered {
16409            let records = std::mem::take(&mut bt2.index.filtered_records);
16410            let mut kept = Vec::with_capacity(records.len());
16411            for r in records {
16412                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16413                    if collect_refs {
16414                        if let Some(bytes) = self.read_chunk_block(
16415                            pipeline.as_ref(),
16416                            r.chunk_address,
16417                            r.chunk_size,
16418                            r.filter_mask,
16419                        )? {
16420                            dead_refs.extend_from_slice(&bytes);
16421                        }
16422                    }
16423                    if !swmr {
16424                        self.allocator
16425                            .free(r.chunk_address, r.chunk_size, FreeSpaceClass::RawData);
16426                    }
16427                } else {
16428                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16429                        straddlers.push(r.scaled_offsets.clone());
16430                    }
16431                    kept.push(r);
16432                }
16433            }
16434            bt2.index.filtered_records = kept;
16435        } else {
16436            let records = std::mem::take(&mut bt2.index.records);
16437            let mut kept = Vec::with_capacity(records.len());
16438            for r in records {
16439                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16440                    if collect_refs {
16441                        if let Some(bytes) = self.read_chunk_block(
16442                            pipeline.as_ref(),
16443                            r.chunk_address,
16444                            chunk_bytes,
16445                            0,
16446                        )? {
16447                            dead_refs.extend_from_slice(&bytes);
16448                        }
16449                    }
16450                    if !swmr {
16451                        self.allocator
16452                            .free(r.chunk_address, chunk_bytes, FreeSpaceClass::RawData);
16453                    }
16454                } else {
16455                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16456                        straddlers.push(r.scaled_offsets.clone());
16457                    }
16458                    kept.push(r);
16459                }
16460            }
16461            bt2.index.records = kept;
16462        }
16463        Ok((straddlers, dead_refs))
16464    }
16465
16466    /// Version-1-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16467    /// drop the records of chunks beyond the extent — the next flush
16468    /// re-serializes the smaller tree over the node pool and releases the
16469    /// surplus node blocks.
16470    fn prune_btree_v1_chunks(
16471        &self,
16472        index: usize,
16473        geo: &ChunkGeometry,
16474        new_dims: &[u64],
16475        collect_refs: bool,
16476    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16477        let ds = self.ds(index);
16478        let mut m = ds.lock();
16479        let pipeline = m.filter_pipeline.clone();
16480        let swmr = self.swmr_active;
16481        let mut straddlers = Vec::new();
16482        let mut dead_refs = Vec::new();
16483        let bt1 = m.btree_v1.as_mut().unwrap();
16484        let records = std::mem::take(&mut bt1.records);
16485        let mut kept = Vec::with_capacity(records.len());
16486        for r in records {
16487            if chunk_outside_extent(&r.scaled, &geo.chunk_dims, new_dims) {
16488                if collect_refs {
16489                    if let Some(bytes) = self.read_chunk_block(
16490                        pipeline.as_ref(),
16491                        r.address,
16492                        r.nbytes as u64,
16493                        r.filter_mask,
16494                    )? {
16495                        dead_refs.extend_from_slice(&bytes);
16496                    }
16497                }
16498                if !swmr {
16499                    self.allocator
16500                        .free(r.address, r.nbytes as u64, FreeSpaceClass::RawData);
16501                }
16502            } else {
16503                if chunk_straddles_extent(&r.scaled, &geo.chunk_dims, new_dims) {
16504                    straddlers.push(r.scaled.clone());
16505                }
16506                kept.push(r);
16507            }
16508        }
16509        m.btree_v1.as_mut().unwrap().records = kept;
16510        Ok((straddlers, dead_refs))
16511    }
16512
16513    /// Flush a chunked dataset's index structures to disk (durable).
16514    ///
16515    /// Writes the index blocks and issues an `fdatasync` so the data is
16516    /// durable — the guarantee SWMR readers and standalone callers rely on.
16517    pub fn flush_dataset(&self, index: usize) -> IoResult<()> {
16518        let ds = self.ds(index);
16519        let _op = ds.op.lock();
16520        self.flush_dataset_synced(index, true)
16521    }
16522
16523    /// Flush a chunked dataset's index structures, syncing only if `sync`.
16524    ///
16525    /// `finalize` threads its own durability choice here so that a
16526    /// [`close_no_sync`](Self::close_no_sync) skips this per-dataset
16527    /// `sync_data` too — otherwise gating only the final `sync_all` would
16528    /// leave one `fdatasync` per indexed dataset and defeat the fast close.
16529    fn flush_dataset_synced(&self, index: usize, sync: bool) -> IoResult<()> {
16530        // Hold one slot guard for the whole method; `self.handle`/`self.ctx`/
16531        // `self.allocator` below touch disjoint fields.
16532        let ds = self.ds(index);
16533        let mut m = ds.lock();
16534
16535        // EA-indexed dataset
16536        if let Some(ref chunked) = m.chunked {
16537            if let Some(ref fiblk) = chunked.filt_iblk {
16538                // Filtered EA
16539                let iblk_encoded = fiblk.encode(&self.ctx, chunked.chunk_size_len);
16540                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
16541            } else {
16542                // Unfiltered EA
16543                let iblk_encoded = chunked.ea_iblk.encode(&self.ctx);
16544                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
16545            }
16546            let hdr_encoded = chunked.ea_header.encode(&self.ctx);
16547            self.handle.write_at(chunked.ea_header_addr, &hdr_encoded)?;
16548            if sync {
16549                self.handle.sync_data()?;
16550            }
16551            return Ok(());
16552        }
16553
16554        // Fixed-array-indexed dataset
16555        if let Some(ref fa) = m.fixed_array {
16556            let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa.fa_header, &fa.fa_dblk);
16557            self.handle.write_at(fa.fa_dblk_addr, &dblk_encoded)?;
16558            let hdr_encoded = fa.fa_header.encode(&self.ctx);
16559            self.handle.write_at(fa.fa_header_addr, &hdr_encoded)?;
16560            if sync {
16561                self.handle.sync_data()?;
16562            }
16563            return Ok(());
16564        }
16565
16566        // BT2-indexed dataset
16567        if let Some(ref bt2) = m.btree_v2 {
16568            // Bulk-load the index into fixed-size nodes and lay them over the
16569            // dataset's block pool. Because every node is the same size, the
16570            // blocks already on disk are reused in place and only the shortfall
16571            // is allocated — the pool is the single owner of these addresses,
16572            // so no flush leaves a block behind. The addresses a reader already
16573            // holds stay valid, which is also what SWMR needs.
16574            let tree = bt2.index.build_tree(&self.ctx);
16575            let mut node_addrs = bt2.node_addrs.clone();
16576            while node_addrs.len() < tree.nodes.len() {
16577                node_addrs.push(
16578                    self.allocator
16579                        .allocate(tree.node_size as u64, FreeSpaceClass::Metadata),
16580                );
16581            }
16582            // A tree with fewer nodes than last flush releases the surplus
16583            // rather than leaving it recorded and unreachable, so the pool is
16584            // exactly one block per node whichever way the count moved. Under
16585            // SWMR a reader may still hold a header naming those blocks, so
16586            // keep them out of the free list — the same rule `place_chunk`
16587            // applies to a relocated chunk.
16588            for addr in node_addrs.split_off(tree.nodes.len()) {
16589                if !self.swmr_active {
16590                    self.allocator
16591                        .free(addr, tree.node_size as u64, FreeSpaceClass::Metadata);
16592                }
16593            }
16594
16595            for (image, &addr) in tree.encode(&self.ctx, &node_addrs).iter().zip(&node_addrs) {
16596                self.handle.write_at(addr, image)?;
16597            }
16598
16599            // The root is the last node the bulk load emits.
16600            let root_addr = match tree.nodes.len() {
16601                0 => UNDEF_ADDR,
16602                n => node_addrs[n - 1],
16603            };
16604            let hdr_encoded = tree.header(root_addr).encode(&self.ctx);
16605            self.handle.write_at(bt2.bt2_header_addr, &hdr_encoded)?;
16606
16607            m.btree_v2.as_mut().unwrap().node_addrs = node_addrs;
16608
16609            if sync {
16610                self.handle.sync_data()?;
16611            }
16612            return Ok(());
16613        }
16614
16615        // Version-1-B-tree-indexed dataset
16616        if let Some(ref bt1) = m.btree_v1 {
16617            // Bulk-loaded over the same block pool the v2 B-tree above uses,
16618            // and for the same reason: every node of a v1 tree is the width
16619            // its "K" value gives, so a block stays usable however the tree
16620            // reshapes, and only the shortfall is ever allocated.
16621            let element_size = m.datatype.element_size() as u64;
16622            let tree = bt1.build_tree(element_size, self.ctx.sizeof_addr as usize);
16623            let node_size = tree.node_size() as u64;
16624            let mut node_addrs = bt1.node_addrs.clone();
16625            while node_addrs.len() < tree.node_count() {
16626                node_addrs.push(self.allocator.allocate(node_size, FreeSpaceClass::Metadata));
16627            }
16628            // A tree with fewer nodes than last flush releases the surplus
16629            // straight away, where the v2 B-tree has to keep it out of the
16630            // free list for a live SWMR reader: this index lives only in a
16631            // classic file, which `start_swmr` refuses outright (and upstream
16632            // says the same in `H5D_COPS_BTREE`).
16633            for addr in node_addrs.split_off(tree.node_count()) {
16634                self.allocator
16635                    .free(addr, node_size, FreeSpaceClass::Metadata);
16636            }
16637            for (image, &addr) in tree.encode(&node_addrs)?.iter().zip(&node_addrs) {
16638                self.handle.write_at(addr, image)?;
16639            }
16640            // The root is the last node the bulk load emits, and is undefined
16641            // while the dataset has no chunks — what the version-3 data
16642            // layout message then carries, exactly as libhdf5 leaves it.
16643            let root_addr = tree.root_address(&node_addrs);
16644            let bt1 = m.btree_v1.as_mut().unwrap();
16645            bt1.node_addrs = node_addrs;
16646            bt1.root_addr = root_addr;
16647
16648            if sync {
16649                self.handle.sync_data()?;
16650            }
16651            return Ok(());
16652        }
16653
16654        Ok(())
16655    }
16656
16657    /// Finalize and close the file.
16658    ///
16659    /// Writes the dataset object headers, root group object header, and
16660    /// superblock. After this call the file is a valid HDF5 file.
16661    pub fn close(mut self) -> IoResult<()> {
16662        self.close_in_place()
16663    }
16664
16665    /// [`close`](Self::close) for a holder that cannot give the writer up by
16666    /// value because it has a `Drop` of its own ([`SwmrWriter`]): the same
16667    /// one-shot commit, after which this writer's `Drop` is a no-op.
16668    ///
16669    /// [`SwmrWriter`]: crate::io::swmr::SwmrWriter
16670    pub(crate) fn close_in_place(&mut self) -> IoResult<()> {
16671        // Mark closed BEFORE finalizing: finalize writes external truth
16672        // (object headers + superblock) and must run exactly once. If we
16673        // finalized first and it failed, the `?` would return with `closed`
16674        // still false, and dropping `self` would re-run `finalize` a second
16675        // time over a half-written file (and print the "call close()" notice
16676        // the caller already heeded). Committing to the close path first makes
16677        // `Drop` (the only other finalize site) a no-op regardless of outcome,
16678        // so the error is reported exactly once via this `Result`.
16679        self.closed = true;
16680        self.finalize(true)
16681    }
16682
16683    /// Finalize and close the file without a final `fsync`.
16684    ///
16685    /// Identical to [`close`](Self::close) — the same object headers and
16686    /// superblock are written, so on return the file is a complete, valid HDF5
16687    /// file readable by any process — except that the trailing `sync_all`
16688    /// (fsync) is skipped. The bytes are handed to the OS but are not
16689    /// guaranteed durable against power loss or an OS crash until the OS
16690    /// flushes its page cache; a normal process exit or a same-machine reader
16691    /// sees the full file regardless.
16692    ///
16693    /// This trades durability for speed: `sync_all` typically dominates close
16694    /// latency, so bulk writers that do not need crash durability (the file can
16695    /// be regenerated) can use this to avoid that cost. Use [`close`](Self::close)
16696    /// when durability matters. `Drop` always finalizes durably, so a writer
16697    /// finalized this way must reach `close_no_sync` explicitly.
16698    pub fn close_no_sync(mut self) -> IoResult<()> {
16699        // Same close-once discipline as `close`: commit to the close path
16700        // before finalizing so `Drop` cannot re-run `finalize` on failure.
16701        self.closed = true;
16702        self.finalize(false)
16703    }
16704
16705    /// Provide mutable access to the underlying file handle.
16706    pub fn handle(&mut self) -> &mut FileHandle {
16707        &mut self.handle
16708    }
16709
16710    /// The superblock version this file will be written with.
16711    ///
16712    /// `H5F__super_init` takes the oldest version that can describe the file
16713    /// and raises it to the one the file's library-version low bound implies:
16714    /// `super_vers = MAX(super_vers, HDF5_superblock_ver_bounds[low_bound])`,
16715    /// with the bounds table reading 0, 2, 3, 3, 3, 3, 3 for EARLIEST, V18,
16716    /// V110, V112, V114, V200, LATEST (H5Fsuper.c:68, :1128-1154). A file
16717    /// created at `H5F_LIBVER_EARLIEST` takes that bound's entry directly
16718    /// ([`SuperblockVersion::Chosen`], and the classic branch below) — version
16719    /// 0, or version 2 when the file carries shared messages, whose master
16720    /// table needs the superblock extension only a version-2 superblock has
16721    /// (H5Fsuper.c:1135). For every other file the bound is read back from
16722    /// what this crate writes:
16723    ///
16724    /// * The floor is `H5F_LIBVER_V18`, hence version 2. Every group such a
16725    ///   file holds is a link-message group, which libhdf5 only writes at a
16726    ///   low bound of V18 or newer (`use_at_least_v18`, H5Gobj.c:179), and
16727    ///   every object header in it is version 2, which `H5O_obj_ver_bounds`
16728    ///   likewise puts at V18 (H5Oint.c:125). A version-0 superblock over
16729    ///   this content would claim a file libhdf5 1.6 can read, and no libhdf5
16730    ///   writes that combination.
16731    /// * A chunked dataset — extensible array, fixed array or version-2
16732    ///   B-tree, all reached through a version-4 or -5 data layout message —
16733    ///   reads back as V110 (`H5O_layout_ver_bounds`, H5Dlayout.c:44), hence
16734    ///   version 3.
16735    /// * SWMR writes version 3 outright (H5Fsuper.c:1129).
16736    ///
16737    /// A file whose caller *named* a bound skips the read-back and takes that
16738    /// bound's row directly, so `V18` stays at version 2 however its chunked
16739    /// datasets are indexed — which is what libhdf5 does, the layout version
16740    /// being no input to `H5F__super_init` at all.
16741    ///
16742    /// None of that applies to a reopened file. `H5F__super_read` validates
16743    /// the version it finds and never recomputes one, so the version written
16744    /// back is the version read — see [`SuperblockVersion`], which is also
16745    /// where the other half of that rule lives: the version floors the bound
16746    /// the appended structures are written at, which is why nothing this
16747    /// session adds can need a newer one.
16748    fn superblock_version_for(&self, flags: u8) -> u8 {
16749        let chosen = match self.superblock_version {
16750            SuperblockVersion::Existing(version) => return version,
16751            SuperblockVersion::Chosen(version) => version,
16752        };
16753        if self.is_legacy() {
16754            // A classic file keeps the version it was created at — 0, or 2
16755            // when its shared messages needed the extension. Nothing a session
16756            // can add reaches past that: its objects get symbol-table links,
16757            // its chunked datasets the version-1 B-tree behind a version-3
16758            // layout message, and the two features that would raise the bound
16759            // — SWMR and the 2.0 format — are refused where the caller asks
16760            // for them.
16761            return chosen;
16762        }
16763        let mut version = chosen
16764            .max(SUPERBLOCK_V2)
16765            .max(self.effective_libver().superblock_version());
16766        if self.swmr_active || flags & FLAG_SWMR_WRITE != 0 {
16767            version = version.max(SUPERBLOCK_V3);
16768        }
16769        version
16770    }
16771
16772    /// The low bound a modern file this writer *created* is effectively
16773    /// written at: the one the caller named, or — with none named — the one
16774    /// its content reads back as. A reopened file never reaches here; its
16775    /// superblock version is not derived from its content at all.
16776    ///
16777    /// The read-back is what `superblock_version_for` needs and the field
16778    /// alone cannot give: this crate's default file names no bound, and the
16779    /// generation it writes is not one bound but two rows (see the `libver`
16780    /// field). The floor is `V18`, the oldest bound under which libhdf5 writes
16781    /// link-message groups (`use_at_least_v18`, H5Gobj.c:179) and version-2
16782    /// object headers (`H5O_obj_ver_bounds`, H5Oint.c:125), which is all such
16783    /// a file holds; a v1.10 chunk index in it raises that to `V110`, the
16784    /// oldest bound whose `H5O_layout_ver_bounds` row reaches the version-4
16785    /// layout message that index is written behind.
16786    fn effective_libver(&self) -> LibverBound {
16787        self.libver.unwrap_or_else(|| {
16788            if self.has_v110_chunk_index() {
16789                LibverBound::V110
16790            } else {
16791                LibverBound::V18
16792            }
16793        })
16794    }
16795
16796    /// Whether any dataset still in the file is indexed by a v1.10 chunk
16797    /// index — the markers `build_dataset_header` turns into a version-4/5
16798    /// data layout message, and nothing else it can emit reaches that
16799    /// version.
16800    ///
16801    /// Not "is any dataset chunked": the version-1 B-tree is a chunk index
16802    /// that encodes as a *version-3* layout message, the version
16803    /// `H5O_layout_ver_bounds` gives the earliest bound, so a dataset using
16804    /// it asks nothing of the superblock.
16805    fn has_v110_chunk_index(&self) -> bool {
16806        self.dataset_refs().iter().any(|d| {
16807            let m = d.lock();
16808            !m.deleted
16809                && m.chunk_index_kind()
16810                    .is_some_and(|k| k != ChunkIndexKind::BtreeV1)
16811        })
16812    }
16813
16814    /// Write the superblock at offset 0 with the given flags.
16815    ///
16816    /// Requires that the root group has already been written (via `finalize`
16817    /// or `finalize_for_swmr`).
16818    pub fn write_superblock(&mut self, flags: u8) -> IoResult<()> {
16819        let root_addr = self
16820            .root_group_addr
16821            .ok_or_else(|| crate::io::IoError::InvalidState("root group not yet written".into()))?;
16822        // The userblock this file was opened with. `H5F__super_read` prefers
16823        // the located address over this field, but `H5Pget_userblock` reports
16824        // it, so a rewrite that zeroed it would hide the block from every
16825        // reader that asks for its size.
16826        let base = self.handle.base();
16827        // The end of file is the one address in the superblock measured from
16828        // the start of the *file* rather than from the base: `H5F__super_read`
16829        // sets the EOA to `stored_eof - base_addr` (H5Fsuper.c:635) and calls
16830        // the file truncated when `eof + base_addr < stored_eof` (:573). The
16831        // allocator counts in the based space, so the userblock is added back.
16832        let eof = self.allocator.eof() + base;
16833        let version = self.superblock_version_for(flags);
16834        // Which of the two images is written follows the version, not the
16835        // generation: a classic file carrying shared messages is a version-2
16836        // superblock over version-1 messages and symbol-table groups
16837        // (H5Fsuper.c:1135), and only the version-2/3 image has the extension
16838        // address that table is reached through. Below version 2 the file is
16839        // always a classic one — the other branch floors at 2.
16840        if let Some(legacy) = self.legacy.as_deref().filter(|_| version < SUPERBLOCK_V2) {
16841            // Re-emitted, not rebuilt: the "K" ranks, the userblock size and
16842            // the driver info address are recorded nowhere else in the file,
16843            // and every node width in it is derived from the ranks. Only the
16844            // three things this session can have changed are recomputed.
16845            let root_stab = self
16846                .symbol_tables
16847                .written
16848                .lock()
16849                .get(&LinkScope::Root)
16850                .copied();
16851            let mut sb = legacy.superblock.clone();
16852            sb.version = version;
16853            sb.file_consistency_flags = flags as u32;
16854            sb.end_of_file_address = eof;
16855            sb.root_symbol_table_entry.obj_header_addr = root_addr;
16856            // `H5G__stab_valid` (H5Groot.c) reads this pair back and compares
16857            // it against the root header's Symbol Table message, repairing the
16858            // superblock when they disagree. Writing the pair that message now
16859            // names is what keeps the file from needing that repair. A root
16860            // that keeps its links in messages has no such pair and no entry
16861            // in `written`, and gets `H5G_NOTHING_CACHED` — what libhdf5
16862            // writes for the same root.
16863            sb.root_symbol_table_entry.cache = match root_stab {
16864                Some(s) => SymbolTableCache::SymbolTable {
16865                    btree_addr: s.btree_addr,
16866                    heap_addr: s.heap_addr,
16867                },
16868                None => SymbolTableCache::Nothing,
16869            };
16870            self.handle.write_at(0, &sb.encode())?;
16871            return Ok(());
16872        }
16873        let sb = SuperblockV2V3 {
16874            version,
16875            sizeof_offsets: self.ctx.sizeof_addr,
16876            sizeof_lengths: self.ctx.sizeof_size,
16877            file_consistency_flags: flags,
16878            base_address: base,
16879            // Whatever `write_superblock_extension` put there, which is the
16880            // only place an extension is written.
16881            superblock_extension_address: self.extension.addr.lock().unwrap_or(UNDEF_ADDR),
16882            end_of_file_address: eof,
16883            root_group_object_header_address: root_addr,
16884        };
16885        self.handle.write_at(0, &sb.encode())?;
16886        Ok(())
16887    }
16888
16889    /// Re-write a dataset's object header in place (SWMR update).
16890    ///
16891    /// The header must have been written by `finalize_for_swmr`, and goes
16892    /// back over the same blocks: chunk 0 held to its block and the
16893    /// continuation chunk, when it has one, to its own. Only the dataspace
16894    /// dimensions are meant to change; a header that no longer fits is
16895    /// refused rather than moved, since a reader holds its address.
16896    pub fn write_dataset_header_inplace(&mut self, index: usize) -> IoResult<()> {
16897        // Scope the slot guard: `build_dataset_header` re-locks the same slot.
16898        let placement = {
16899            let ds = self.ds(index);
16900            let m = ds.lock();
16901            HeaderPlacement::over(&m.obj_header_blocks).ok_or_else(|| {
16902                crate::io::IoError::InvalidState("dataset header not yet written".into())
16903            })?
16904        };
16905
16906        let header = self.build_dataset_header(index)?;
16907        let nlink = self.object_link_count(HardLinkTarget::Dataset(index));
16908        let format = self.dataset_header_format(index);
16909        let images = self.encode_header_in(&header, nlink, format, &placement)?;
16910        let reserved = placement.blocks();
16911        let fits = images.len() == reserved.len()
16912            && images
16913                .iter()
16914                .zip(&reserved)
16915                .all(|((_, image), &(_, size))| image.len() as u64 == size);
16916        if !fits {
16917            return Err(crate::io::IoError::InvalidState(format!(
16918                "dataset header grew from {} to {} bytes; cannot rewrite in place",
16919                reserved.iter().map(|&(_, size)| size).sum::<u64>(),
16920                images.iter().map(|(_, image)| image.len()).sum::<usize>()
16921            )));
16922        }
16923        for (addr, image) in &images {
16924            self.handle.write_at(*addr, image)?;
16925        }
16926        // Only after the bytes are down: a failed write leaves the registry
16927        // describing the header the file still holds.
16928        self.ds(index).lock().header_written(nlink);
16929        Ok(())
16930    }
16931
16932    /// Perform a full finalize for SWMR mode.
16933    ///
16934    /// This writes all dataset object headers, the root group header, and the
16935    /// superblock with SWMR flags. After this call, the file is valid for
16936    /// SWMR readers. Subsequent writes use in-place updates.
16937    pub fn finalize_for_swmr(&mut self) -> IoResult<()> {
16938        self.reject_swmr()?;
16939        // 0. Flush all chunked dataset index structures.
16940        for i in 0..self.dataset_count() {
16941            let is_indexed = {
16942                let ds = self.ds(i);
16943                let m = ds.lock();
16944                !m.deleted && m.is_chunked()
16945            };
16946            if is_indexed {
16947                self.flush_dataset(i)?;
16948            }
16949        }
16950
16951        // 1. Allocate every object header (none for a dataset deleted before
16952        // start_swmr — its storage was freed at delete time). Same three
16953        // phases as the full finalize, and for the same reason: nothing a
16954        // header names can be laid out until every object has an address.
16955        let live: Vec<usize> = (0..self.dataset_count())
16956            .filter(|&i| !self.ds(i).lock().deleted)
16957            .collect();
16958        let kept = self.supersede_headers(&live);
16959        // Before any dataset header: a sharing dataset's header names the
16960        // committed type's address.
16961        self.write_committed_datatype_headers()?;
16962        let layout = self.allocate_object_headers(&live, &kept)?;
16963
16964        // 2. Build content against those addresses.
16965        self.prepare_dense_attributes(&live)?;
16966        self.prepare_link_storage()?;
16967        self.write_reference_values()?;
16968
16969        // 3. Write every object header.
16970        self.write_object_headers(&layout)?;
16971        // Where each header is published and how much room it has: what
16972        // `write_dataset_header_inplace` rewrites within, and what the
16973        // closing finalize writes over, since a reader may by then hold any
16974        // of these addresses.
16975        for &(i, placement) in &layout.datasets {
16976            let ds = self.ds(i);
16977            let mut m = ds.lock();
16978            m.obj_header_written_addr = Some(placement.addr);
16979            m.obj_header_blocks = placement.blocks();
16980        }
16981        for &(gi, placement) in &layout.groups {
16982            let grp = self.grp(gi);
16983            let mut g = grp.lock();
16984            g.obj_header_written_addr = Some(placement.addr);
16985            g.obj_header_blocks = placement.blocks();
16986        }
16987        self.superseded_root_header = layout.root.blocks();
16988
16989        // 4. Write superblock with SWMR flags.
16990        self.write_superblock(FLAG_WRITE_ACCESS | FLAG_SWMR_WRITE)?;
16991        self.handle.set_eof(self.allocator.eof())?;
16992
16993        self.handle.sync_all()?;
16994        // Readers can now be following this file, so a chunk that moves must
16995        // leave its old block intact for whoever is still holding the previous
16996        // index (see `swmr_active`).
16997        self.swmr_active = true;
16998        Ok(())
16999    }
17000
17001    // ------------------------------------------------------------------
17002    // Internal helpers
17003    // ------------------------------------------------------------------
17004
17005    /// Flush every dataset's append buffer into the chunks it belongs to,
17006    /// through [`flush_append_buffer`](Self::flush_append_buffer): frames
17007    /// already in the chunk survive, and the rest of it reads back as the
17008    /// dataset's fill value (zeros when none is defined).
17009    fn flush_append_buffers(&mut self) -> IoResult<()> {
17010        for i in 0..self.dataset_count() {
17011            if self.ds(i).lock().deleted {
17012                continue;
17013            }
17014            self.flush_append_buffer(i)?;
17015        }
17016        Ok(())
17017    }
17018
17019    /// Write all object headers and the superblock, producing a complete,
17020    /// valid HDF5 file.
17021    ///
17022    /// `sync == true` issues a final `sync_all` (fsync) so the bytes are
17023    /// durable against power loss / OS crash before returning. `sync == false`
17024    /// skips that fsync: the file is still fully written to the OS and readable
17025    /// by any process, but durability is left to the OS page-cache flush. This
17026    /// is the only difference between [`close`](Self::close) (durable) and
17027    /// [`close_no_sync`](Self::close_no_sync) (fast).
17028    fn finalize(&mut self, sync: bool) -> IoResult<()> {
17029        // Flush any partial append buffers before finalizing
17030        self.flush_append_buffers()?;
17031
17032        // A SWMR session (`finalize_for_swmr` already ran, so
17033        // `root_group_addr` is `Some`) is closed by the same full finalize as
17034        // a fresh write: every object header is rebuilt over its chunk 0 and
17035        // the superblock is written with clean-close flags. A full rebuild —
17036        // rather than the in-place header rewrite used by the live
17037        // `SwmrWriter::flush` path — is required so any structural change made
17038        // after `start_swmr` is committed to the final file. A hard link, in
17039        // particular, both grows its target's header with an object
17040        // reference-count message and adds a `MSG_LINK` record to a group
17041        // header; an in-place rewrite cannot accommodate the grown header and
17042        // never re-emits group/root headers. The fall-through below already
17043        // handles datasets whose header was written by `finalize_for_swmr`
17044        // (`obj_header_written_addr.is_some()`).
17045
17046        // 0. Flush chunked dataset index structures (only modified datasets).
17047        for i in 0..self.dataset_count() {
17048            let ds = self.ds(i);
17049            {
17050                let m = ds.lock();
17051                if m.deleted {
17052                    continue;
17053                }
17054                if m.obj_header_written_addr.is_some() && !m.storage_dirty() {
17055                    continue;
17056                }
17057                let is_indexed = m.is_chunked();
17058                if !is_indexed {
17059                    continue;
17060                }
17061            }
17062            self.flush_dataset_synced(i, sync)?;
17063        }
17064
17065        // 1. Plan. Which datasets get a header (deleted datasets get none —
17066        // their storage was already freed at delete time) is settled first,
17067        // because everything the next phases lay out is laid out only for the
17068        // headers this finalize actually rewrites; and every header those
17069        // phases supersede is taken here, before the first allocation, so
17070        // the rewrite lands over it.
17071        let mut rewritten: Vec<usize> = Vec::new();
17072        // A finalize that lays the shared-message table out afresh reassigns
17073        // every heap ID in the file, so no existing header can keep its bytes:
17074        // the pointers in them name heap objects the new table does not have.
17075        let table_replaced = self.rebuilds_shared_messages();
17076        for i in 0..self.dataset_count() {
17077            // Before the slot guard: `object_link_count` re-locks every
17078            // dataset and group slot, this one included.
17079            let nlink = self.object_link_count(HardLinkTarget::Dataset(i));
17080            let ds = self.ds(i);
17081            let mut m = ds.lock();
17082            if m.deleted {
17083                continue;
17084            }
17085            // An existing dataset from append mode keeps its header — and
17086            // everything that header names — unless this session changed
17087            // what the header says.
17088            if let Some(written) = m.obj_header_written_addr {
17089                if !table_replaced && !m.header_stale_with(nlink) {
17090                    // Keep the original object header address for the root group link.
17091                    m.obj_header_addr = written;
17092                    continue;
17093                }
17094            }
17095            rewritten.push(i);
17096        }
17097        let kept = self.supersede_headers(&rewritten);
17098
17099        // 2. Allocate. Committed datatype headers go down whole: a header of
17100        // theirs holds a datatype and a reference count, so it waits on
17101        // nothing, while a dataset sharing the type and the group naming it
17102        // both store its address. They are written before the shared-message
17103        // phase opens, so a committed type reaches the file as itself.
17104        self.write_committed_datatype_headers()?;
17105        self.begin_shared_message_layout();
17106        let layout = self.allocate_object_headers(&rewritten, &kept)?;
17107
17108        // 3. Build content, with every object header's address known. Dense
17109        // attribute storage holds the attribute messages themselves — an
17110        // object reference among them is a header address; dense links and
17111        // symbol tables name header addresses; a reference dataset's elements
17112        // are header addresses. Nothing here is a fixup: each is written once,
17113        // with the value the file keeps. The shared-message table comes last:
17114        // it counts the bodies the headers will hold, and the three above are
17115        // what settle them.
17116        self.prepare_dense_attributes(&rewritten)?;
17117        self.prepare_link_storage()?;
17118        self.write_reference_values()?;
17119        self.prepare_shared_messages(&rewritten)?;
17120        self.write_superblock_extension()?;
17121
17122        // 4. Write every object header over the block phase 2 reserved for it.
17123        self.write_object_headers(&layout)?;
17124
17125        // 5. Write superblock at offset 0.
17126        self.write_superblock(0)?;
17127
17128        // 6. End the file where its address space ends (`H5FD_truncate`, which
17129        // `H5F__dest` calls on every close). Allocated-but-unwritten space at
17130        // the end would otherwise leave the file shorter than the end-of-file
17131        // address the superblock just recorded, which libhdf5 reads as a
17132        // truncated file.
17133        self.handle.set_eof(self.allocator.eof())?;
17134
17135        // Durability is opt-in per call: `close` passes `true`, `close_no_sync`
17136        // passes `false`, and `Drop` passes `true` so an un-`close`d writer is
17137        // still finalized durably by default.
17138        if sync {
17139            self.handle.sync_all()?;
17140        }
17141        Ok(())
17142    }
17143
17144    /// Take the on-disk header of every object this finalize rewrites — the
17145    /// datasets in `datasets`, every group, and the root — and hand each
17146    /// chunk-0 block to [`allocate_object_headers`](Self::allocate_object_headers)
17147    /// to be written over.
17148    ///
17149    /// Chunk 0 stays where it is: its address is what every reference in the
17150    /// file holds. The continuation blocks behind it go back to the free
17151    /// list, so the rewrite reuses them instead of growing the file on every
17152    /// open/close cycle — nothing names one but its own header. Hard links
17153    /// can alias one header under several names; the set keeps an aliased
17154    /// chain from being taken twice. The registry forgets each header here,
17155    /// so a finalize that fails later describes none the file no longer holds.
17156    fn supersede_headers(&mut self, datasets: &[usize]) -> KeptChunks {
17157        let mut kept = KeptChunks::default();
17158        let mut taken = std::collections::HashSet::new();
17159        for &i in datasets {
17160            let ds = self.ds(i);
17161            let mut m = ds.lock();
17162            let Some(old) = m.obj_header_written_addr.take() else {
17163                continue;
17164            };
17165            let blocks = std::mem::take(&mut m.obj_header_blocks);
17166            if !blocks.is_empty() && taken.insert(old) {
17167                kept.datasets.insert(i, self.keep_chunk0(blocks));
17168            }
17169        }
17170        for gi in 0..self.group_count() {
17171            let grp = self.grp(gi);
17172            let mut g = grp.lock();
17173            let Some(old) = g.obj_header_written_addr.take() else {
17174                continue;
17175            };
17176            let blocks = std::mem::take(&mut g.obj_header_blocks);
17177            if !blocks.is_empty() && taken.insert(old) {
17178                kept.groups.insert(gi, self.keep_chunk0(blocks));
17179            }
17180        }
17181        let root_blocks = std::mem::take(&mut self.superseded_root_header);
17182        if root_blocks
17183            .first()
17184            .is_some_and(|&(addr, _)| taken.insert(addr))
17185        {
17186            kept.root = Some(self.keep_chunk0(root_blocks));
17187        }
17188        kept
17189    }
17190
17191    /// Keep `blocks`' chunk 0 for a rewrite and free the continuation blocks
17192    /// behind it — never under SWMR, where a live reader may be walking them,
17193    /// the same rule `release_vlen_references` and `place_chunk` follow.
17194    fn keep_chunk0(&self, blocks: crate::io::object_header_io::HeaderBlocks) -> (u64, u64) {
17195        let mut blocks = blocks.into_iter();
17196        let chunk0 = blocks.next().expect("a written header has a chunk 0");
17197        if !self.swmr_active {
17198            for (addr, len) in blocks {
17199                self.allocator.free(addr, len, FreeSpaceClass::Metadata);
17200            }
17201        }
17202        chunk0
17203    }
17204
17205    /// Give every object header this finalize writes an address, before
17206    /// anything that names one is built.
17207    ///
17208    /// INVARIANT: from the moment this returns until the file is closed, every
17209    /// object in it has the object header address it will be found at. That is
17210    /// what lets the phase after this one say an address wherever the format
17211    /// wants one — in a link message, in a symbol table entry, in a reference
17212    /// dataset's elements, and in an attribute's value, which is the one of the
17213    /// four that cannot be revisited after its header is written.
17214    ///
17215    /// An object in `kept` is placed over the chunk-0 block it already has, so
17216    /// its address is the one every reference in the file already holds. A
17217    /// header is measured before its content is final, which is sound because
17218    /// no address changes its length: every address is a fixed-width field,
17219    /// and an object that has none yet reads as zero, which is the same width.
17220    /// The storage a header names is laid out between the two passes for the
17221    /// same reason and answers the same way — `emit_attributes` and
17222    /// `emit_links` each fall back to a size-equal placeholder message. It is
17223    /// [`write_object_headers`](Self::write_object_headers) that checks this
17224    /// held, rather than either pass assuming it.
17225    fn allocate_object_headers(
17226        &mut self,
17227        datasets: &[usize],
17228        kept: &KeptChunks,
17229    ) -> IoResult<HeaderLayout> {
17230        let mut layout = HeaderLayout {
17231            datasets: Vec::with_capacity(datasets.len()),
17232            groups: Vec::new(),
17233            root: HeaderPlacement::fresh(0, 0),
17234        };
17235        for &i in datasets {
17236            let header = self.build_dataset_header(i)?;
17237            let format = self.dataset_header_format(i);
17238            let placement = self.place_header(&header, format, kept.datasets.get(&i).copied())?;
17239            self.ds(i).lock().obj_header_addr = placement.addr;
17240            layout.datasets.push((i, placement));
17241        }
17242        for gi in 0..self.group_count() {
17243            if self.grp(gi).lock().deleted {
17244                continue;
17245            }
17246            let header = self.build_group_header(gi)?;
17247            let format = self.group_header_format(gi);
17248            let placement = self.place_header(&header, format, kept.groups.get(&gi).copied())?;
17249            self.grp(gi).lock().obj_header_addr = placement.addr;
17250            layout.groups.push((gi, placement));
17251        }
17252        let header = self.build_root_group_header()?;
17253        let format = self.header_format(self.root_track_order);
17254        let placement = self.place_header(&header, format, kept.root)?;
17255        self.root_group_addr = Some(placement.addr);
17256        layout.root = placement;
17257        Ok(layout)
17258    }
17259
17260    /// Write every object header over the blocks
17261    /// [`allocate_object_headers`](Self::allocate_object_headers) reserved for
17262    /// it.
17263    ///
17264    /// The single owner of object header writing in both finalize paths, and
17265    /// the only place a header's body meets its block: a body that does not
17266    /// fill its measurement exactly fails the finalize here rather than
17267    /// overrunning the next object or leaving a tail of the previous one, which
17268    /// is how a message whose length turns out to depend on an address would
17269    /// show up.
17270    fn write_object_headers(&mut self, layout: &HeaderLayout) -> IoResult<()> {
17271        for &(i, placement) in &layout.datasets {
17272            let rc = self.object_link_count(HardLinkTarget::Dataset(i));
17273            let header = self.build_dataset_header(i)?;
17274            let format = self.dataset_header_format(i);
17275            let what = format!("dataset '{}'", self.ds(i).lock().name);
17276            self.write_header_in(&header, rc, format, &placement, &what)?;
17277            // Only after the bytes are down: a failed write leaves the registry
17278            // describing the header the file still holds.
17279            self.ds(i).lock().header_written(rc);
17280        }
17281        for &(gi, placement) in &layout.groups {
17282            let rc = self.object_link_count(HardLinkTarget::Group(gi));
17283            let header = self.build_group_header(gi)?;
17284            let format = self.group_header_format(gi);
17285            let what = format!("group '{}'", self.grp(gi).lock().name);
17286            self.write_header_in(&header, rc, format, &placement, &what)?;
17287        }
17288        let header = self.build_root_group_header()?;
17289        let format = self.header_format(self.root_track_order);
17290        self.write_header_in(&header, 1, format, &layout.root, "the root group")
17291    }
17292
17293    /// Encode `header` into `placement` and write it, after checking each
17294    /// image against the block reserved for it.
17295    fn write_header_in(
17296        &mut self,
17297        header: &ObjectHeader,
17298        rc: u32,
17299        format: ObjectFormat,
17300        placement: &HeaderPlacement,
17301        what: &str,
17302    ) -> IoResult<()> {
17303        let images = self.encode_header_in(header, rc, format, placement)?;
17304        let reserved =
17305            std::iter::once(placement.size).chain(placement.continuation.map(|(_, s)| s));
17306        for ((addr, image), size) in images.iter().zip(reserved) {
17307            check_header_size(image, size, || what.to_string())?;
17308            self.handle.write_at(*addr, image)?;
17309        }
17310        Ok(())
17311    }
17312
17313    fn build_dataset_header(&self, index: usize) -> IoResult<ObjectHeader> {
17314        // Compute the link count first: object_link_count re-locks dataset and
17315        // group slots (including this one), so it must run before we take this
17316        // dataset's slot guard — otherwise it would deadlock on the same slot.
17317        let rc = self.object_link_count(HardLinkTarget::Dataset(index));
17318        // Same reason: reading the committed type's address locks the
17319        // committed-datatype registry, which the slot guard below must not be
17320        // held across.
17321        let committed = self.ds(index).lock().committed_type;
17322        let committed_addr = committed.map(|r| match r {
17323            CommittedTypeRef::Session(ci) => self.committed_datatypes.lock()[ci].obj_header_addr,
17324            CommittedTypeRef::Preserved(addr) => addr,
17325        });
17326        // And again: an attribute holding an object reference is said in the
17327        // target's header address, which is read off that object's slot.
17328        let attributes = self.object_attributes(AttrScope::Dataset(index))?;
17329
17330        // Hold one slot guard for the whole header build.
17331        let ds = self.ds(index);
17332        let m = ds.lock();
17333        let mut header = ObjectHeader::new();
17334
17335        // Every message below is written in the format this dataset already
17336        // has, not the one this session would pick. libhdf5 grows a header in
17337        // place and never re-encodes a message it did not touch, so reopening
17338        // a superblock-v2 file — which raises the low bound to V18
17339        // (hdf5_1.14.6 H5Fsuper.c:460-462) — leaves the version-1 dataspaces
17340        // an EARLIEST-bound creating session wrote exactly as they are. This
17341        // writer has to lay the whole header out again whenever the
17342        // shared-message heap moves, so preserving the encoding is the only
17343        // way to land on the same bytes.
17344        let format = m.read_format.unwrap_or_else(|| self.message_format());
17345        let libver = match format {
17346            ObjectFormat::Legacy => LibverBound::Earliest,
17347            ObjectFormat::Modern => self.encoding_libver(),
17348        };
17349
17350        // Dataspace message (type 0x01)
17351        let ds_msg = m.dataspace.encode_for(&self.ctx, format);
17352        let owner = ShareOwner::Header(m.obj_header_addr);
17353        let (flags, ds_msg) = self.share_message(owner, MSG_DATASPACE, 0x00, ds_msg);
17354        header.add_message(MSG_DATASPACE, flags, ds_msg);
17355
17356        // Datatype message (type 0x03). A dataset built on a committed type
17357        // stores a pointer to that object header in place of the message, and
17358        // the shared flag is what says the body is a pointer — the two are one
17359        // statement, so they are written together.
17360        match committed_addr {
17361            Some(addr) => header.add_message(
17362                MSG_DATATYPE,
17363                MSG_FLAG_CONSTANT | MSG_FLAG_SHARED,
17364                SharedMessagePointer::encode_committed(addr, &self.ctx),
17365            ),
17366            None => {
17367                let body = m.datatype.encode_at(&self.ctx, libver);
17368                let (flags, body) = if self.dataset_datatype_shareable(&m.datatype, libver) {
17369                    self.share_message(owner, MSG_DATATYPE, MSG_FLAG_CONSTANT, body)
17370                } else {
17371                    (MSG_FLAG_CONSTANT, body)
17372                };
17373                header.add_message(MSG_DATATYPE, flags, body)
17374            }
17375        }
17376
17377        // Fill Value message (type 0x05)
17378        let is_chunked = m.is_chunked();
17379        // `H5P__init_def_layout` gives each storage class its own default
17380        // allocation time: incremental for chunked and for virtual (whose
17381        // source datasets are allocated as they are written), early for
17382        // compact (the space is the header, so it exists as soon as the
17383        // dataset does), late for contiguous. An implicitly indexed dataset is
17384        // the one chunked exception, and not by default but by definition:
17385        // early allocation is a *condition* of that index
17386        // (`H5D__layout_set_latest_indexing`), so a header claiming
17387        // incremental would describe a file libhdf5 would never have chosen
17388        // this index for. A single-chunk dataset can go either way — unlike
17389        // Implicit, early allocation is not one of its selection conditions
17390        // — so its `early_alloc` flag (set only for an unfiltered dataset
17391        // created that way) is what this checks instead.
17392        let alloc_time = if m.compact.is_some()
17393            || m.implicit.is_some()
17394            || m.single_chunk.as_ref().is_some_and(|s| s.early_alloc)
17395        {
17396            1 // early
17397        } else if is_chunked || m.virtual_storage.is_some() {
17398            3 // incremental
17399        } else {
17400            2 // late
17401        };
17402        // `H5D__update_oh_info` (H5Dint.c:927-943): a variable-length
17403        // datatype with no explicit fill value forces ALLOC regardless of
17404        // the declared policy — its heap-reference encoding has no safe
17405        // all-zero "no fill" representation, so libhdf5 always writes the
17406        // (empty) fill value at allocation for such a dataset. `IFSET` is
17407        // the only declared policy this touches: an explicit `ALLOC` is
17408        // already what it forces, and upstream rejects `NEVER` for a
17409        // VL-typed dataset at `H5Dcreate` outright — this crate's
17410        // VL-typed datasets have no builder path to declare `NEVER` in the
17411        // first place, so that branch cannot be reached here.
17412        let is_vlen = matches!(
17413            m.datatype,
17414            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
17415        );
17416        let fill_write_time = if is_vlen && m.fill_value.is_none() && m.fill_time == FILL_TIME_IFSET
17417        {
17418            FILL_TIME_ALLOC
17419        } else {
17420            m.fill_time
17421        };
17422        let fv = if let Some(ref bytes) = m.fill_value {
17423            // User-defined fill value (fill_defined = 2).
17424            FillValueMessage {
17425                alloc_time,
17426                fill_write_time,
17427                fill_defined: 2,
17428                fill_value: Some(bytes.clone()),
17429            }
17430        } else {
17431            // No fill value of the dataset's own (fill_defined = 1, the
17432            // implicit default zero fill) — `alloc_time` above already
17433            // carries the per-layout-class default (`H5P__set_layout`,
17434            // H5Pdcpl.c:1864-1877), so this branch must use it too instead
17435            // of `FillValueMessage::default()`'s hardcoded LATE: that was
17436            // wrong for a compact (EARLY) or virtual (INCR) dataset with no
17437            // fill value, only coincidentally right for contiguous.
17438            FillValueMessage {
17439                alloc_time,
17440                fill_write_time,
17441                fill_defined: 1, // default value (zeros)
17442                fill_value: None,
17443            }
17444        };
17445        // `H5O_MSG_FLAG_CONSTANT`, as `H5D__update_oh_info` appends it
17446        // (H5Dint.c:965) — the same flag the datatype message beside it
17447        // carries (H5Dint.c:961) and the old fill value below (H5Dint.c:981).
17448        // A dataset's fill value is fixed at creation: `H5Pset_fill_value` is
17449        // a creation property, so nothing can rewrite the message in place and
17450        // libhdf5 tells the header so.
17451        let fv_msg = fv.encode_for(format);
17452        let (flags, fv_msg) = self.share_message(owner, MSG_FILL_VALUE, MSG_FLAG_CONSTANT, fv_msg);
17453        header.add_message(MSG_FILL_VALUE, flags, fv_msg);
17454
17455        // The "fill value (old)" message (type 0x04) beside the new one, for a
17456        // user-defined fill value below the v1.8 bound. `H5D__update_oh_info`
17457        // (H5Dint.c:1024-1035) appends `H5O_FILL_ID` whenever `fill_prop->buf`
17458        // is set and `use_at_least_v18` — `H5F_LOW_BOUND(file) >= V18`, which
17459        // here is exactly a non-`Legacy` message format — is false, so that a
17460        // reader that predates the new message still finds the value. The body
17461        // is the size and the bytes and nothing else: no allocation time, no
17462        // write time, no defined flag (`H5O__fill_old_encode`, H5Ofill.c:512).
17463        if matches!(format, ObjectFormat::Legacy) {
17464            if let Some(ref bytes) = m.fill_value {
17465                let mut old = Vec::with_capacity(4 + bytes.len());
17466                old.extend_from_slice(&(bytes.len() as u32).to_le_bytes());
17467                old.extend_from_slice(bytes);
17468                let (flags, old) =
17469                    self.share_message(owner, MSG_FILL_VALUE_OLD, MSG_FLAG_CONSTANT, old);
17470                header.add_message(MSG_FILL_VALUE_OLD, flags, old);
17471            }
17472        }
17473
17474        // External Data Files message (type 0x07), before the layout message
17475        // and marked constant, exactly where `H5D__layout_oh_create` puts it.
17476        // It is what makes a reader route the dataset's I/O through the files
17477        // it names rather than through the undefined address the layout
17478        // message below still declares.
17479        if let Some(ref ext) = m.external {
17480            header.add_message(
17481                MSG_EXTERNAL_FILE_LIST,
17482                MSG_FLAG_CONSTANT,
17483                ext.message().encode(&self.ctx),
17484            );
17485        }
17486
17487        // Data Layout message (type 0x08)
17488        let layout = if let Some(ref chunked) = m.chunked {
17489            let mut layout_dims = chunked.chunk_dims.clone();
17490            layout_dims.push(m.datatype.element_size() as u64);
17491            DataLayoutMessage::chunked_v4_earray(
17492                m.layout_version,
17493                layout_dims,
17494                chunked.earray_params.clone(),
17495                chunked.ea_header_addr,
17496            )
17497        } else if let Some(ref fa) = m.fixed_array {
17498            let mut layout_dims = fa.chunk_dims.clone();
17499            layout_dims.push(m.datatype.element_size() as u64);
17500            DataLayoutMessage::chunked_v4_farray(
17501                m.layout_version,
17502                layout_dims,
17503                FixedArrayParams::default_params(),
17504                fa.fa_header_addr,
17505            )
17506        } else if let Some(ref bt2) = m.btree_v2 {
17507            let mut layout_dims = bt2.chunk_dims.clone();
17508            layout_dims.push(m.datatype.element_size() as u64);
17509            DataLayoutMessage::chunked_v4_btree_v2(
17510                m.layout_version,
17511                layout_dims,
17512                crate::format::messages::data_layout::Bt2Params {
17513                    node_size: bt2.index.node_size,
17514                    split_percent: bt2.index.split_percent,
17515                    merge_percent: bt2.index.merge_percent,
17516                },
17517                bt2.bt2_header_addr,
17518            )
17519        } else if let Some(ref imp) = m.implicit {
17520            let mut layout_dims = imp.chunk_dims.clone();
17521            layout_dims.push(m.datatype.element_size() as u64);
17522            DataLayoutMessage::chunked_v4_implicit(m.layout_version, layout_dims, imp.data_addr)
17523        } else if let Some(ref sc) = m.single_chunk {
17524            let mut layout_dims = sc.chunk_dims.clone();
17525            layout_dims.push(m.datatype.element_size() as u64);
17526            if m.filter_pipeline.is_some() {
17527                DataLayoutMessage::chunked_v4_single_filtered(
17528                    layout_dims,
17529                    sc.data_addr,
17530                    sc.nbytes,
17531                    sc.filter_mask,
17532                )
17533            } else {
17534                DataLayoutMessage::chunked_v4_single(layout_dims, sc.data_addr)
17535            }
17536        } else if let Some(ref bt1) = m.btree_v1 {
17537            // The classic index: a version-3 layout message carrying the
17538            // address of the tree's root node, which is undefined until a
17539            // chunk is written.
17540            let mut layout_dims = bt1.chunk_dims.clone();
17541            layout_dims.push(m.datatype.element_size() as u64);
17542            DataLayoutMessage::chunked_v3_btree_v1(layout_dims, bt1.root_addr)
17543        } else if let Some(ref image) = m.compact {
17544            DataLayoutMessage::compact(image.clone())
17545        } else if let Some(ref virt) = m.virtual_storage {
17546            // Version 4 always: the virtual layout class did not exist before
17547            // it, so the default virtual layout is created at version 4 and
17548            // `H5Pset_virtual` raises any lower one to it (H5Pdcpl.c),
17549            // whatever the file's library-version bounds say — which is why a
17550            // v0-superblock file can still hold one.
17551            DataLayoutMessage::virtual_layout(4, virt.heap_addr, virt.heap_index)
17552        } else {
17553            DataLayoutMessage::contiguous(m.data_addr, m.data_size)
17554        };
17555        // `H5D__layout_oh_create` (H5Dlayout.c:530-536) marks the layout
17556        // message constant only where the storage it names is certain to be
17557        // there already: allocation time is early, the class is not compact,
17558        // no filter can change a chunk's size, and the dataspace holds at
17559        // least one element. Anything else leaves the address undefined at
17560        // creation and rewrites the message when the space is allocated, so
17561        // the flag would be a lie. `H5S_GET_EXTENT_NPOINTS` is zero for a
17562        // NULL dataspace and for any extent with a zero-length dimension.
17563        let npoints: u64 = if m.dataspace.is_null() {
17564            0
17565        } else {
17566            m.dataspace.dims.iter().product()
17567        };
17568        let filtered = m
17569            .filter_pipeline
17570            .as_ref()
17571            .is_some_and(|p| !p.filters.is_empty());
17572        let layout_flags = if alloc_time == 1 && m.compact.is_none() && !filtered && npoints != 0 {
17573            MSG_FLAG_CONSTANT
17574        } else {
17575            0x00
17576        };
17577        let layout_msg = layout.encode(&self.ctx);
17578        header.add_message(MSG_DATA_LAYOUT, layout_flags, layout_msg);
17579
17580        // Filter Pipeline message (type 0x0B) -- only if filters are
17581        // configured. `H5D__layout_oh_create` appends it with
17582        // `H5O_MSG_FLAG_CONSTANT` (H5Dlayout.c:462), as does the group
17583        // pipeline for dense links (H5Gobj.c:264): the pipeline is a creation
17584        // property, and every chunk already written was filtered through it,
17585        // so it can never be rewritten in place.
17586        if let Some(ref pipeline) = m.filter_pipeline {
17587            if !pipeline.filters.is_empty() {
17588                let (flags, filter_msg) = self.share_message(
17589                    owner,
17590                    MSG_FILTER_PIPELINE,
17591                    MSG_FLAG_CONSTANT,
17592                    pipeline.encode_for(format),
17593                );
17594                header.add_message(MSG_FILTER_PIPELINE, flags, filter_msg);
17595            }
17596        }
17597
17598        // A dataset has no links, so only attribute creation order can raise
17599        // its header past version 1 (`H5O__set_version`).
17600        let format = self.header_format(TrackOrder {
17601            links: CreationOrder::default(),
17602            attrs: m.track_attr_order,
17603        });
17604
17605        // Modification time, here and not earlier: `H5D__update_oh_info` makes
17606        // this the last message it writes (H5Dint.c:1022-1026), and the
17607        // attributes below it are added by `H5A` calls that come after the
17608        // dataset exists.
17609        touch_oh(&mut header, format, m.times, true);
17610
17611        // Attribute Info (type 0x15) + attribute messages (type 0x0C).
17612        self.emit_attributes(
17613            &mut header,
17614            AttrScope::Dataset(index),
17615            &attributes,
17616            m.track_attr_order,
17617            format,
17618            owner,
17619        );
17620
17621        self.emit_refcount(&mut header, rc, format);
17622
17623        Ok(header)
17624    }
17625
17626    /// Write the object header of every committed datatype something still
17627    /// reaches, recording the address each one landed at.
17628    ///
17629    /// Runs before the dataset and group headers because both name these
17630    /// addresses — a sharing dataset in its datatype message, the parent
17631    /// group in the link. One pass is enough: the header holds a datatype
17632    /// message and at most a reference count, neither of which depends on an
17633    /// address.
17634    fn write_committed_datatype_headers(&mut self) -> IoResult<()> {
17635        // The count is bound first: a lock guard in the `for` iterator
17636        // expression would live for the whole loop body, which locks the same
17637        // registry again.
17638        let count = self.committed_datatypes.lock().len();
17639        for i in 0..count {
17640            let rc = self.committed_datatype_refcount(i);
17641            if rc == 0 {
17642                // Its name's group was deleted and no dataset shares it, so
17643                // nothing in the file could reach the header.
17644                continue;
17645            }
17646            let format = self.committed_datatype_header_format();
17647            let encoded = self
17648                .build_committed_datatype_header(i, rc, format)
17649                .encode_for(format, rc)?;
17650            let addr = self
17651                .allocator
17652                .allocate(encoded.len() as u64, FreeSpaceClass::Metadata);
17653            self.handle.write_at(addr, &encoded)?;
17654            self.committed_datatypes.lock()[i].obj_header_addr = addr;
17655        }
17656        Ok(())
17657    }
17658
17659    /// The header format a committed datatype gets.
17660    ///
17661    /// `H5T__commit` creates the header from the datatype creation property
17662    /// list (H5Tcommit.c:468), which carries no link order and, by default, no
17663    /// attribute order — so the version is the file's floor exactly as
17664    /// `H5O__set_version` computes it, and a committed datatype in a classic
17665    /// file is a version-1 header like every other object in it.
17666    fn committed_datatype_header_format(&self) -> ObjectFormat {
17667        self.header_format(TrackOrder::default())
17668    }
17669
17670    /// Build the object header for a committed datatype: the type, and the
17671    /// reference count when more than one name reaches it.
17672    fn build_committed_datatype_header(
17673        &self,
17674        index: usize,
17675        rc: u32,
17676        format: ObjectFormat,
17677    ) -> ObjectHeader {
17678        let (datatype, times) = {
17679            let reg = self.committed_datatypes.lock();
17680            (reg[index].datatype.clone(), reg[index].times)
17681        };
17682        let mut header = ObjectHeader::new();
17683        // No attributes to emit, so nothing else would apply the file-wide
17684        // floor to this header. `store_msg_crt_idx` is a property of the file,
17685        // not of the object: every header created under it records creation
17686        // indices, a committed datatype's included.
17687        header.set_attribute_creation_order(self.header_attr_order(CreationOrder::default()));
17688        // `H5T__commit` marks the message constant and unshareable: this
17689        // header is where shared datatype bodies are read *from*, so its own
17690        // message must never become a pointer into the shared-message heap.
17691        header.add_message(
17692            MSG_DATATYPE,
17693            MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
17694            datatype.encode_at(&self.ctx, self.encoding_libver()),
17695        );
17696        touch_oh(&mut header, format, times, false);
17697        // Through the same owner as every other object's count: a dataset
17698        // sharing this type raises it (`H5O__shared_link_adj`, H5Oshared.c:249)
17699        // just as a second name does, and where that count is recorded is the
17700        // header version's business, not the caller's.
17701        self.emit_refcount(&mut header, rc, format);
17702        header
17703    }
17704
17705    /// Build the object header for a subgroup.
17706    fn build_group_header(&self, group_idx: usize) -> IoResult<ObjectHeader> {
17707        let mut header = ObjectHeader::new();
17708
17709        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
17710        // themselves, compact or dense.
17711        // Snapshot what the header needs, then drop the slot guard: the calls
17712        // below re-lock group slots (including this one).
17713        let (track_order, times, owner) = {
17714            let grp = self.grp(group_idx);
17715            let g = grp.lock();
17716            (
17717                g.track_order,
17718                g.times,
17719                ShareOwner::Header(g.obj_header_addr),
17720            )
17721        };
17722        let attributes = self.object_attributes(AttrScope::Group(group_idx))?;
17723        touch_oh(&mut header, self.header_format(track_order), times, false);
17724
17725        let links = self.group_links(LinkScope::Group(group_idx), track_order.links);
17726        self.emit_links(
17727            &mut header,
17728            LinkScope::Group(group_idx),
17729            &links,
17730            track_order.links,
17731        );
17732
17733        // Attribute Info (type 0x15) + attributes (type 0x0C) -- e.g. NeXus
17734        // `NX_class`.
17735        let format = self.header_format(track_order);
17736        self.emit_attributes(
17737            &mut header,
17738            AttrScope::Group(group_idx),
17739            &attributes,
17740            track_order.attrs,
17741            format,
17742            owner,
17743        );
17744
17745        self.emit_refcount(
17746            &mut header,
17747            self.object_link_count(HardLinkTarget::Group(group_idx)),
17748            format,
17749        );
17750
17751        Ok(header)
17752    }
17753
17754    fn build_root_group_header(&self) -> IoResult<ObjectHeader> {
17755        let mut header = ObjectHeader::new();
17756        touch_oh(
17757            &mut header,
17758            self.header_format(self.root_track_order),
17759            self.root_times,
17760            false,
17761        );
17762
17763        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
17764        // themselves, compact or dense.
17765        let links = self.group_links(LinkScope::Root, self.root_track_order.links);
17766        self.emit_links(
17767            &mut header,
17768            LinkScope::Root,
17769            &links,
17770            self.root_track_order.links,
17771        );
17772
17773        // Root-level attributes
17774        let root_attributes = self.object_attributes(AttrScope::Root)?;
17775        self.emit_attributes(
17776            &mut header,
17777            AttrScope::Root,
17778            &root_attributes,
17779            self.root_track_order.attrs,
17780            self.header_format(self.root_track_order),
17781            ShareOwner::Header(self.root_group_addr.unwrap_or(0)),
17782        );
17783
17784        Ok(header)
17785    }
17786}
17787
17788impl Drop for Hdf5Writer {
17789    fn drop(&mut self) {
17790        if !self.closed {
17791            // Best-effort finalize on drop. Drop cannot return a Result, so a
17792            // failure here is otherwise invisible: it would leave a truncated
17793            // or unflushed file on disk while the caller believes the write
17794            // succeeded. Surface it on stderr instead of swallowing it.
17795            // Callers that need to handle the error must call
17796            // `H5File::close()` explicitly, which returns the Result.
17797            if let Err(e) = self.finalize(true) {
17798                eprintln!(
17799                    "rust-hdf5: failed to finalize HDF5 file on drop: {e}. \
17800                     The file may be incomplete or corrupt; call \
17801                     H5File::close() to handle this error explicitly."
17802                );
17803            }
17804        }
17805    }
17806}
17807
17808#[cfg(test)]
17809mod tests {
17810    use super::*;
17811    use crate::format::messages::datatype::DatatypeMessage;
17812    use crate::io::reader::Hdf5Reader;
17813
17814    fn fixture(name: &str) -> std::path::PathBuf {
17815        std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
17816            .join("tests/fixtures")
17817            .join(name)
17818    }
17819
17820    /// Copy a fixture so a test that appends does not edit the checked-in file.
17821    fn fixture_copy(name: &str, tag: &str) -> std::path::PathBuf {
17822        let path = temp_path(tag);
17823        std::fs::copy(fixture(name), &path).unwrap();
17824        path
17825    }
17826
17827    fn temp_path(tag: &str) -> std::path::PathBuf {
17828        use std::sync::atomic::{AtomicU64, Ordering};
17829        static COUNTER: AtomicU64 = AtomicU64::new(0);
17830        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
17831        std::env::temp_dir().join(format!(
17832            "rust_hdf5_w_{}_{}_{}.h5",
17833            std::process::id(),
17834            tag,
17835            n
17836        ))
17837    }
17838
17839    /// A group past the link phase change keeps its links in a fractal heap
17840    /// with a v2 B-tree name index. The reopen that rewrites that group's
17841    /// header lays a fresh pair out, so both blocks the old header named must
17842    /// come back to the allocator — every block of the heap, and the index
17843    /// header with its nodes.
17844    ///
17845    /// Asserted on the free list rather than on the file size: a reopen does
17846    /// not yet carry dense links forward, so the rewritten group's links (and
17847    /// the datasets they name) are dropped, and the file size that follows
17848    /// says more about that than about this.
17849    #[test]
17850    fn a_reopen_frees_the_dense_link_storage_its_rewrite_supersedes() {
17851        let path = temp_path("dense_link_reclaim");
17852
17853        let writer = Hdf5Writer::create(&path).unwrap();
17854        writer.create_group("/", "run").unwrap();
17855        for i in 0..12 {
17856            writer
17857                .create_dataset(&format!("run/d{i:02}"), DatatypeMessage::i32_type(), &[2])
17858                .unwrap();
17859        }
17860        writer.close().unwrap();
17861
17862        let writer = Hdf5Writer::open_append(&path).unwrap();
17863        let gidx = (0..writer.group_count())
17864            .find(|&g| writer.grp(g).lock().name == "/run")
17865            .expect("the reopen registered the group");
17866        let linfo = writer
17867            .superseded_dense
17868            .lock()
17869            .as_ref()
17870            .and_then(|s| s.links.get(&LinkScope::Group(gidx)).cloned())
17871            .expect("the reopen recorded the group's dense link storage");
17872        assert_ne!(linfo.fractal_heap_address, UNDEF_ADDR);
17873        assert_ne!(linfo.name_btree_address, UNDEF_ADDR);
17874
17875        writer
17876            .release_superseded_dense_links(LinkScope::Group(gidx))
17877            .unwrap();
17878        let freed = writer.allocator.free_blocks();
17879        let covers = |addr: u64| {
17880            freed
17881                .iter()
17882                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17883        };
17884        assert!(covers(linfo.fractal_heap_address), "heap header: {freed:?}");
17885        assert!(covers(linfo.name_btree_address), "name index: {freed:?}");
17886
17887        // And exactly once: the entry is gone, so the finalize that follows
17888        // cannot hand the same blocks back a second time.
17889        assert!(writer
17890            .superseded_dense
17891            .lock()
17892            .as_ref()
17893            .is_none_or(|s| s.links.is_empty()));
17894        writer
17895            .release_superseded_dense_links(LinkScope::Group(gidx))
17896            .unwrap();
17897        assert_eq!(writer.allocator.free_blocks(), freed);
17898
17899        writer.close().unwrap();
17900        std::fs::remove_file(&path).ok();
17901    }
17902
17903    /// The rewrite frees what it supersedes even when the replacement is not
17904    /// dense at all. An attribute set that drops back under `max_compact`
17905    /// goes into the object header, so nothing names the old heap any more —
17906    /// and a free driven by "the new set needs dense storage" would never
17907    /// reach this one.
17908    #[test]
17909    fn a_rewrite_that_drops_out_of_dense_storage_still_frees_it() {
17910        let path = temp_path("dense_attr_to_compact");
17911        let numeric = |name: &str| {
17912            AttributeMessage::scalar_numeric(
17913                name,
17914                DatatypeMessage::i32_type(),
17915                7i32.to_le_bytes().to_vec(),
17916            )
17917        };
17918
17919        let writer = Hdf5Writer::create(&path).unwrap();
17920        for i in 0..12 {
17921            writer
17922                .add_root_attribute(numeric(&format!("a{i:02}")))
17923                .unwrap();
17924        }
17925        writer.close().unwrap();
17926
17927        let writer = Hdf5Writer::open_append(&path).unwrap();
17928        let ainfo = writer
17929            .superseded_dense
17930            .lock()
17931            .as_ref()
17932            .and_then(|s| s.attrs.get(&AttrScope::Root).cloned())
17933            .expect("the reopen recorded the root's dense attribute storage");
17934        for i in 0..10 {
17935            writer
17936                .evict_attr(AttrTarget::Root, &format!("a{i:02}"))
17937                .unwrap();
17938        }
17939        assert!(!writer.attributes_need_dense(&writer.root_attributes.lock(), ObjectFormat::Modern));
17940
17941        writer.prepare_dense_attributes(&[]).unwrap();
17942        let freed = writer.allocator.free_blocks();
17943        let covers = |addr: u64| {
17944            freed
17945                .iter()
17946                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17947        };
17948        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17949        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17950        assert!(writer
17951            .superseded_dense
17952            .lock()
17953            .as_ref()
17954            .is_none_or(|s| s.attrs.is_empty()));
17955
17956        writer.close().unwrap();
17957        std::fs::remove_file(&path).ok();
17958    }
17959
17960    /// Deleting a reopened object supersedes its dense storage as surely as
17961    /// rewriting one does: nothing in the finalized file names the heap, so
17962    /// the delete owner frees it through the same entry.
17963    #[test]
17964    fn deleting_a_reopened_group_frees_its_dense_attribute_storage() {
17965        let path = temp_path("dense_attr_delete");
17966        let numeric = |name: &str| {
17967            AttributeMessage::scalar_numeric(
17968                name,
17969                DatatypeMessage::i32_type(),
17970                7i32.to_le_bytes().to_vec(),
17971            )
17972        };
17973
17974        let writer = Hdf5Writer::create(&path).unwrap();
17975        writer.create_group("/", "run").unwrap();
17976        for i in 0..12 {
17977            writer
17978                .set_attribute(AttrTarget::Group("/run"), numeric(&format!("a{i:02}")))
17979                .unwrap();
17980        }
17981        writer.close().unwrap();
17982
17983        let writer = Hdf5Writer::open_append(&path).unwrap();
17984        let gidx = (0..writer.group_count())
17985            .find(|&g| writer.grp(g).lock().name == "/run")
17986            .expect("the reopen registered the group");
17987        let ainfo = writer
17988            .superseded_dense
17989            .lock()
17990            .as_ref()
17991            .and_then(|s| s.attrs.get(&AttrScope::Group(gidx)).cloned())
17992            .expect("the reopen recorded the group's dense attribute storage");
17993
17994        writer.delete_group("/run").unwrap();
17995        let freed = writer.allocator.free_blocks();
17996        let covers = |addr: u64| {
17997            freed
17998                .iter()
17999                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
18000        };
18001        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
18002        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
18003        assert!(writer
18004            .superseded_dense
18005            .lock()
18006            .as_ref()
18007            .is_none_or(|s| s.attrs.is_empty()));
18008
18009        writer.close().unwrap();
18010        std::fs::remove_file(&path).ok();
18011    }
18012
18013    /// The charset rule is one owner shared by every vlen string writer:
18014    /// appends into an ASCII-declared dataset reject non-ASCII strings the
18015    /// same way the slice writer does, and a dataset whose elements are not
18016    /// vlen references at all is refused instead of overwritten with them.
18017    #[test]
18018    fn append_vlen_strings_checks_the_datatype_and_charset() {
18019        let path = temp_path("append_vlen_charset");
18020
18021        let writer = Hdf5Writer::create(&path).unwrap();
18022        let idx = writer
18023            .create_appendable_vlen_string_dataset("d", 4, None)
18024            .unwrap();
18025        writer.ds(idx).lock().datatype = DatatypeMessage::vlen_string_ascii();
18026        let err = writer
18027            .append_vlen_strings(idx, &["ok", "안녕"])
18028            .unwrap_err();
18029        assert!(
18030            err.to_string().contains("is not ASCII"),
18031            "unexpected error: {err}"
18032        );
18033        writer.append_vlen_strings(idx, &["ok", "fine"]).unwrap();
18034
18035        let nums = writer
18036            .create_chunked_dataset("n", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
18037            .unwrap();
18038        let err = writer.append_vlen_strings(nums, &["x"]).unwrap_err();
18039        assert!(
18040            err.to_string()
18041                .contains("only for variable-length string datasets"),
18042            "unexpected error: {err}"
18043        );
18044
18045        writer.close().unwrap();
18046        std::fs::remove_file(&path).ok();
18047    }
18048
18049    /// `create_chunked_dataset` builds an extensible-array index unconditionally
18050    /// (the caller — the high-level dataset API — is the one that decides when
18051    /// two-or-more unlimited dimensions should go to a v2 B-tree instead), so
18052    /// its own guard is the last line of defense against a shape that index
18053    /// can't represent at all.
18054    #[test]
18055    fn create_chunked_dataset_rejects_two_unlimited_dimensions() {
18056        let path = temp_path("earray_two_unlimited");
18057        let writer = Hdf5Writer::create(&path).unwrap();
18058        let err = writer
18059            .create_chunked_dataset(
18060                "d",
18061                DatatypeMessage::i32_type(),
18062                &[4, 4],
18063                &[u64::MAX, u64::MAX],
18064                &[2, 2],
18065            )
18066            .unwrap_err();
18067        assert!(err.to_string().contains("at most one unlimited"), "{err}");
18068        writer.close().unwrap();
18069        std::fs::remove_file(&path).ok();
18070    }
18071
18072    /// Every creator must enter through `begin_create`; the four that used
18073    /// to bypass it could push a second dataset under an existing name and
18074    /// emit an invalid file with two same-named links.
18075    #[test]
18076    fn every_creator_rejects_an_existing_dataset_name() {
18077        let path = temp_path("create_gate");
18078
18079        let writer = Hdf5Writer::create(&path).unwrap();
18080        writer
18081            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
18082            .unwrap();
18083
18084        let attempts: [(&str, IoResult<usize>); 4] = [
18085            (
18086                "vlen_string",
18087                writer.create_vlen_string_dataset("d", &["x"], 1),
18088            ),
18089            ("vlen_bytes", writer.create_vlen_bytes_dataset("d", &[b"x"])),
18090            (
18091                "vlen_string_compressed",
18092                writer.create_vlen_string_dataset_compressed(
18093                    "d",
18094                    &["x"],
18095                    1,
18096                    FilterPipeline::deflate(6),
18097                ),
18098            ),
18099            (
18100                "chunked_with_pipeline",
18101                writer.create_chunked_dataset_with_pipeline(
18102                    "d",
18103                    DatatypeMessage::i32_type(),
18104                    &[0],
18105                    &[u64::MAX],
18106                    &[4],
18107                    FilterPipeline::deflate(6),
18108                ),
18109            ),
18110        ];
18111        for (which, res) in attempts {
18112            match res {
18113                Ok(_) => panic!("{which} accepted a duplicate name"),
18114                Err(e) => assert!(
18115                    e.to_string().contains("already exists"),
18116                    "{which}: unexpected error: {e}"
18117                ),
18118            }
18119        }
18120
18121        writer.close().unwrap();
18122        std::fs::remove_file(&path).ok();
18123    }
18124
18125    /// Every creator and every kind of name meet at `ensure_name_free`.
18126    ///
18127    /// The gate's whole value is that it is one list: a creator must be
18128    /// blind neither to a name kind it does not itself make nor to one added
18129    /// after it. This crosses the two — six names, one of each kind the
18130    /// writer can put in a group, against every creator — so a creator that
18131    /// grows its own check, or a name kind that stops being on the list,
18132    /// fails here rather than in a file holding two links of one name.
18133    #[test]
18134    fn every_creator_refuses_every_kind_of_taken_name() {
18135        let path = temp_path("create_gate_matrix");
18136        let writer = Hdf5Writer::create(&path).unwrap();
18137
18138        let i32t = || DatatypeMessage::i32_type();
18139        writer.create_dataset("d", i32t(), &[2]).unwrap();
18140        writer.create_compact_dataset("c", i32t(), &[2]).unwrap();
18141        writer.create_group("/", "g").unwrap();
18142        writer.commit_datatype("t", i32t()).unwrap();
18143        writer.create_hard_link("/", "h", "d").unwrap();
18144        writer
18145            .create_symbolic_link(
18146                "/",
18147                "s",
18148                LinkTarget::Soft {
18149                    target: "/d".into(),
18150                },
18151            )
18152            .unwrap();
18153        writer
18154            .create_symbolic_link(
18155                "/",
18156                "e",
18157                LinkTarget::External {
18158                    file: "other.h5".into(),
18159                    path: "/x".into(),
18160                },
18161            )
18162            .unwrap();
18163
18164        for taken in ["d", "c", "g", "t", "h", "s", "e"] {
18165            let attempts: [(&str, IoResult<()>); 8] = [
18166                (
18167                    "dataset",
18168                    writer.create_dataset(taken, i32t(), &[2]).map(|_| ()),
18169                ),
18170                (
18171                    "compact",
18172                    writer
18173                        .create_compact_dataset(taken, i32t(), &[2])
18174                        .map(|_| ()),
18175                ),
18176                (
18177                    "chunked",
18178                    writer
18179                        .create_chunked_dataset(taken, i32t(), &[0], &[u64::MAX], &[4])
18180                        .map(|_| ()),
18181                ),
18182                (
18183                    "vlen_string",
18184                    writer
18185                        .create_vlen_string_dataset(taken, &["x"], 1)
18186                        .map(|_| ()),
18187                ),
18188                (
18189                    "committed datatype",
18190                    writer.commit_datatype(taken, i32t()).map(|_| ()),
18191                ),
18192                ("group", writer.create_group("/", taken).map(|_| ())),
18193                ("hard link", writer.create_hard_link("/", taken, "d")),
18194                (
18195                    "soft link",
18196                    writer.create_symbolic_link(
18197                        "/",
18198                        taken,
18199                        LinkTarget::Soft {
18200                            target: "/d".into(),
18201                        },
18202                    ),
18203                ),
18204            ];
18205            for (which, res) in attempts {
18206                match res {
18207                    Ok(()) => panic!("{which} accepted the taken name '{taken}'"),
18208                    Err(e) => assert!(
18209                        e.to_string().contains("already exists"),
18210                        "{which} on '{taken}': unexpected error: {e}"
18211                    ),
18212                }
18213            }
18214        }
18215
18216        writer.close().unwrap();
18217        std::fs::remove_file(&path).ok();
18218    }
18219
18220    /// The `H5T_VLEN` length field counts base elements, so an image that is
18221    /// not a whole number of them has no length that reads back as what was
18222    /// handed over; it is refused at the call rather than stored truncated.
18223    #[test]
18224    fn vlen_sequence_refuses_a_partial_element() {
18225        let path = temp_path("vlen_partial_element");
18226
18227        let writer = Hdf5Writer::create(&path).unwrap();
18228        let err = writer
18229            .create_vlen_sequence_dataset("d", DatatypeMessage::i32_type(), &[&[1u8, 2, 3, 4, 5]])
18230            .unwrap_err()
18231            .to_string();
18232        assert!(err.contains("5 bytes"), "unexpected error: {err}");
18233        assert!(err.contains("4-byte elements"), "unexpected error: {err}");
18234
18235        // The refusal is the length rule alone: the same base takes a whole
18236        // number of elements, and an empty sequence is a legal one.
18237        writer
18238            .create_vlen_sequence_dataset(
18239                "d",
18240                DatatypeMessage::i32_type(),
18241                &[&[1u8, 2, 3, 4], &[][..]],
18242            )
18243            .unwrap();
18244
18245        writer.close().unwrap();
18246        std::fs::remove_file(&path).ok();
18247    }
18248
18249    /// A corrupt file can declare a zero-length chunk dimension; the
18250    /// superseded-reference read must reject it the way `write_slice` does,
18251    /// not divide by it.
18252    #[test]
18253    fn vlen_slice_rejects_a_zero_chunk_dimension() {
18254        let path = temp_path("vlen_slice_zero_chunk");
18255
18256        let writer = Hdf5Writer::create(&path).unwrap();
18257        let idx = writer
18258            .create_appendable_vlen_string_dataset("d", 2, None)
18259            .unwrap();
18260        writer.append_vlen_strings(idx, &["a", "b"]).unwrap();
18261        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 0;
18262        let err = writer.write_vlen_strings_slice(idx, 0, &["x"]).unwrap_err();
18263        assert!(
18264            err.to_string().contains("zero-length dimension"),
18265            "unexpected error: {err}"
18266        );
18267
18268        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 2;
18269        writer.close().unwrap();
18270        std::fs::remove_file(&path).ok();
18271    }
18272
18273    /// A libhdf5-written collection can be 100% full — no free-space marker,
18274    /// content exactly the declared size. When a stale reference names an
18275    /// index that is not there, nothing is removed, and the collection must
18276    /// be left alone: re-encoding it at its declared size cannot fit the
18277    /// free-space marker and would fail the whole update.
18278    #[test]
18279    fn release_leaves_a_full_collection_it_removed_nothing_from() {
18280        use crate::format::global_heap::encode_vlen_reference;
18281
18282        let path = temp_path("release_full_gcol");
18283        let writer = Hdf5Writer::create(&path).unwrap();
18284
18285        // Hand-built full collection: 16-byte header + one 16+8-byte object,
18286        // declared size exactly 40, no free-space marker.
18287        let mut img = Vec::new();
18288        img.extend_from_slice(b"GCOL");
18289        img.push(1);
18290        img.extend_from_slice(&[0u8; 3]);
18291        img.extend_from_slice(&40u64.to_le_bytes());
18292        img.extend_from_slice(&1u16.to_le_bytes()); // object index 1
18293        img.extend_from_slice(&1u16.to_le_bytes()); // ref_count
18294        img.extend_from_slice(&0u32.to_le_bytes()); // reserved
18295        img.extend_from_slice(&8u64.to_le_bytes()); // data size
18296        img.extend_from_slice(b"deadbeef");
18297        assert_eq!(img.len(), 40);
18298        let addr = writer
18299            .allocator
18300            .allocate(img.len() as u64, FreeSpaceClass::RawData);
18301        writer.handle.write_at(addr, &img).unwrap();
18302
18303        // The superseded reference names index 2, which the collection does
18304        // not hold — a no-op removal.
18305        let refs = encode_vlen_reference(3, addr, 2, &writer.ctx);
18306        writer.release_vlen_references(&refs).unwrap();
18307        assert_eq!(writer.handle.read_at(addr, 40).unwrap(), img);
18308
18309        writer.close().unwrap();
18310        std::fs::remove_file(&path).ok();
18311    }
18312
18313    /// The CWFS second pass (`H5F_cwfs_find_free_heap`): an object too big
18314    /// for the listed collection's remaining free space extends the
18315    /// collection in place — the file allocation grows off the end of the
18316    /// file (`H5MF_try_extend`) and the collection's declared size and
18317    /// free-space marker grow with it (`H5HG_extend`) — instead of opening
18318    /// a second collection.
18319    #[test]
18320    fn an_oversized_vlen_insert_extends_the_listed_collection() {
18321        use crate::format::global_heap::GlobalHeapCollection;
18322
18323        let path = temp_path("cwfs_extend_tail");
18324        let writer = Hdf5Writer::create(&path).unwrap();
18325        // A small object opens a minimum-size (4096) listed collection —
18326        // the file's last allocation, so the extension grows the file end.
18327        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
18328        let big = vec![0x41u8; 5000]; // more than the ~4 KiB remaining
18329        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
18330        assert_eq!(
18331            p2[0].0, p1[0].0,
18332            "the big object opened a second collection"
18333        );
18334
18335        // The block on disk is one grown collection holding both objects.
18336        let img = writer.handle.read_at_most(p1[0].0, 65536).unwrap();
18337        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
18338        assert!(csize > 4096, "declared size did not grow: {csize}");
18339        assert_eq!(gcol.objects.len(), 2);
18340        assert_eq!(gcol.objects[1].data, big);
18341
18342        writer.close().unwrap();
18343        let bytes = std::fs::read(&path).unwrap();
18344        assert_eq!(
18345            bytes.windows(4).filter(|w| *w == b"GCOL").count(),
18346            1,
18347            "a second collection signature is in the file"
18348        );
18349        std::fs::remove_file(&path).ok();
18350    }
18351
18352    /// The non-tail counterpart: the collection is pinned away from the end
18353    /// of the file, but a released block starts right after it, so the
18354    /// extension consumes the front of that block (`H5MF_try_extend`'s
18355    /// free-section path) and the remainder stays reusable.
18356    #[test]
18357    fn extension_consumes_a_freed_block_after_the_collection() {
18358        use crate::format::global_heap::GlobalHeapCollection;
18359
18360        let path = temp_path("cwfs_extend_freed");
18361        let writer = Hdf5Writer::create(&path).unwrap();
18362        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
18363        let addr = p1[0].0;
18364        // Land a block right after the collection, pin the file end past
18365        // it, then release it: extension must use the released space.
18366        let spacer = writer.allocator.allocate(8192, FreeSpaceClass::RawData);
18367        assert_eq!(spacer, addr + 4096, "spacer not adjacent; layout changed");
18368        writer.allocator.allocate(8, FreeSpaceClass::RawData);
18369        writer.allocator.free(spacer, 8192, FreeSpaceClass::RawData);
18370
18371        let big = vec![0x42u8; 5000];
18372        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
18373        assert_eq!(p2[0].0, addr, "the big object opened a second collection");
18374
18375        let img = writer.handle.read_at_most(addr, 65536).unwrap();
18376        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
18377        assert_eq!(csize, 8192, "grew by max(size, shortfall) = 4096");
18378        assert_eq!(gcol.objects.len(), 2);
18379
18380        // The remainder of the released block is still allocatable.
18381        assert_eq!(
18382            writer.allocator.allocate(4096, FreeSpaceClass::RawData),
18383            addr + 8192,
18384            "the freed block's tail was lost"
18385        );
18386        writer.close().unwrap();
18387        std::fs::remove_file(&path).ok();
18388    }
18389
18390    /// Issue #10: a reopen-and-replace loop on a vlen string must not grow
18391    /// the file. The superseded heap objects are freed *before* the
18392    /// replacement is allocated, so each session reuses the block it just
18393    /// released even though the free list starts empty on reopen. The old
18394    /// free-after-alloc order failed this by one collection per session.
18395    #[test]
18396    fn vlen_replace_across_reopen_keeps_the_file_flat() {
18397        let path = temp_path("vlen_reopen_flat");
18398        let payload_a = "a".repeat(64 * 1024);
18399        let payload_b = "b".repeat(64 * 1024);
18400
18401        let writer = Hdf5Writer::create(&path).unwrap();
18402        writer
18403            .create_vlen_string_dataset("notes", &["initial"], 1)
18404            .unwrap();
18405        writer.close().unwrap();
18406
18407        let mut sizes = Vec::new();
18408        for i in 0..8 {
18409            let writer = Hdf5Writer::open_append(&path).unwrap();
18410            let payload = if i % 2 == 0 { &payload_a } else { &payload_b };
18411            writer
18412                .write_vlen_strings_slice(0, 0, &[payload.as_str()])
18413                .unwrap();
18414            writer.close().unwrap();
18415            sizes.push(std::fs::metadata(&path).unwrap().len());
18416        }
18417        // The first replacement grows the file once (the initial collection
18418        // cannot hold 64 KiB); every later equal-size replacement must land
18419        // in the block its own session just freed.
18420        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
18421
18422        // The reused blocks still form a valid file holding the last value.
18423        let mut reader = Hdf5Reader::open(&path).unwrap();
18424        assert_eq!(
18425            reader.read_vlen_strings("notes").unwrap(),
18426            vec![payload_b.clone()]
18427        );
18428
18429        std::fs::remove_file(&path).ok();
18430    }
18431
18432    /// Replacing a vlen string attribute must release the superseded
18433    /// global-heap collection *before* the replacement's collection is
18434    /// allocated, so a reopen-replace loop lands each new value in the block
18435    /// it just freed instead of growing the file by one collection per
18436    /// session — the attribute counterpart of
18437    /// [`vlen_replace_across_reopen_keeps_the_file_flat`].
18438    #[test]
18439    fn vlen_attr_replace_across_reopen_keeps_the_file_flat() {
18440        let path = temp_path("vlen_attr_reopen_flat");
18441        let payload_a = "a".repeat(8 * 1024);
18442        let payload_b = "b".repeat(8 * 1024);
18443
18444        let writer = Hdf5Writer::create(&path).unwrap();
18445        writer
18446            .set_vlen_string_attribute(AttrTarget::Root, "note", &payload_a)
18447            .unwrap();
18448        writer.close().unwrap();
18449
18450        let mut sizes = Vec::new();
18451        for i in 0..8 {
18452            let writer = Hdf5Writer::open_append(&path).unwrap();
18453            let payload = if i % 2 == 0 { &payload_b } else { &payload_a };
18454            writer
18455                .set_vlen_string_attribute(AttrTarget::Root, "note", payload)
18456                .unwrap();
18457            writer.close().unwrap();
18458            sizes.push(std::fs::metadata(&path).unwrap().len());
18459        }
18460        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
18461
18462        // The reused blocks still hold the last value.
18463        let reader = Hdf5Reader::open(&path).unwrap();
18464        let attr = reader.root_attr("note").unwrap().clone();
18465        let mut reader = reader;
18466        assert_eq!(reader.attr_string_value(&attr).unwrap(), payload_a);
18467
18468        std::fs::remove_file(&path).ok();
18469    }
18470
18471    /// A numeric attribute replacing a vlen one goes through the same list
18472    /// owner, so the superseded collection is released even though the new
18473    /// value holds no heap reference: a later same-size vlen attribute must
18474    /// land in the freed block, making the file exactly as large as one that
18475    /// never stored the replaced value.
18476    #[test]
18477    fn numeric_replacing_a_vlen_attr_releases_its_collection() {
18478        let payload = "x".repeat(8 * 1024);
18479        let numeric = || {
18480            AttributeMessage::scalar_numeric(
18481                "x",
18482                DatatypeMessage::i32_type(),
18483                7i32.to_le_bytes().to_vec(),
18484            )
18485        };
18486
18487        let path_a = temp_path("vlen_attr_cross_a");
18488        let writer = Hdf5Writer::create(&path_a).unwrap();
18489        writer
18490            .set_vlen_string_attribute(AttrTarget::Root, "x", &payload)
18491            .unwrap();
18492        writer.add_root_attribute(numeric()).unwrap();
18493        writer
18494            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
18495            .unwrap();
18496        writer.close().unwrap();
18497
18498        // The same end state written without the replaced vlen value.
18499        let path_b = temp_path("vlen_attr_cross_b");
18500        let writer = Hdf5Writer::create(&path_b).unwrap();
18501        writer.add_root_attribute(numeric()).unwrap();
18502        writer
18503            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
18504            .unwrap();
18505        writer.close().unwrap();
18506
18507        assert_eq!(
18508            std::fs::metadata(&path_a).unwrap().len(),
18509            std::fs::metadata(&path_b).unwrap().len()
18510        );
18511
18512        let reader = Hdf5Reader::open(&path_a).unwrap();
18513        let y = reader.root_attr("y").unwrap().clone();
18514        let mut reader = reader;
18515        assert_eq!(reader.attr_string_value(&y).unwrap(), payload);
18516
18517        std::fs::remove_file(&path_a).ok();
18518        std::fs::remove_file(&path_b).ok();
18519    }
18520
18521    /// Reopen/write/close cycles must not leak the object-header blocks
18522    /// finalize rewrites: the reopened root header, the reopened group
18523    /// header, and the modified chunked dataset's header are each freed
18524    /// before their replacements are allocated. The chunk rewrite itself is
18525    /// in place (unfiltered chunks never move), so a leak of any header
18526    /// block shows up as monotonic growth here.
18527    #[test]
18528    fn reopen_cycles_reuse_superseded_header_blocks() {
18529        let path = temp_path("header_reuse");
18530        {
18531            let writer = Hdf5Writer::create(&path).unwrap();
18532            writer.create_group("/", "g").unwrap();
18533            let idx = writer
18534                .create_chunked_dataset(
18535                    "g/data",
18536                    DatatypeMessage::i32_type(),
18537                    &[4],
18538                    &[u64::MAX],
18539                    &[4],
18540                )
18541                .unwrap();
18542            let seed: Vec<u8> = [1i32, 2, 3, 4]
18543                .iter()
18544                .flat_map(|v| v.to_le_bytes())
18545                .collect();
18546            writer.write_chunk(idx, 0, &seed).unwrap();
18547            writer.close().unwrap();
18548        }
18549
18550        let mut sizes = Vec::new();
18551        for i in 0..6i32 {
18552            let writer = Hdf5Writer::open_append(&path).unwrap();
18553            let data: Vec<u8> = [i; 4].iter().flat_map(|v| v.to_le_bytes()).collect();
18554            writer.write_chunk(0, 0, &data).unwrap();
18555            writer.close().unwrap();
18556            sizes.push(std::fs::metadata(&path).unwrap().len());
18557        }
18558        assert_eq!(&sizes[1..], &vec![sizes[0]; 5][..], "sizes: {sizes:?}");
18559
18560        // The reused header blocks still form a valid file.
18561        let mut reader = Hdf5Reader::open(&path).unwrap();
18562        let raw = reader.read_dataset_raw("g/data").unwrap();
18563        let values: Vec<i32> = raw
18564            .chunks(4)
18565            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18566            .collect();
18567        assert_eq!(values, vec![5, 5, 5, 5]);
18568
18569        std::fs::remove_file(&path).ok();
18570    }
18571
18572    #[test]
18573    fn create_empty_file() {
18574        let path = temp_path("empty");
18575
18576        let writer = Hdf5Writer::create(&path).unwrap();
18577        writer.close().unwrap();
18578
18579        // Verify we can read it back
18580        let reader = Hdf5Reader::open(&path).unwrap();
18581        assert!(reader.dataset_names().is_empty());
18582
18583        std::fs::remove_file(&path).ok();
18584    }
18585
18586    #[test]
18587    fn create_single_dataset() {
18588        let path = temp_path("single");
18589
18590        let writer = Hdf5Writer::create(&path).unwrap();
18591        let idx = writer
18592            .create_dataset("data", DatatypeMessage::f64_type(), &[4])
18593            .unwrap();
18594        let values: Vec<f64> = vec![1.0, 2.0, 3.0, 4.0];
18595        let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18596        writer.write_dataset_raw(idx, &raw).unwrap();
18597        writer.close().unwrap();
18598
18599        // Read back
18600        let mut reader = Hdf5Reader::open(&path).unwrap();
18601        assert_eq!(reader.dataset_names(), vec!["data"]);
18602        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4]);
18603        let readback = reader.read_dataset_raw("data").unwrap();
18604        assert_eq!(readback, raw);
18605
18606        std::fs::remove_file(&path).ok();
18607    }
18608
18609    #[test]
18610    fn create_multiple_datasets() {
18611        let path = temp_path("multi");
18612
18613        let writer = Hdf5Writer::create(&path).unwrap();
18614
18615        let idx0 = writer
18616            .create_dataset("ints", DatatypeMessage::i32_type(), &[3])
18617            .unwrap();
18618        let i_data: Vec<u8> = [10i32, 20, 30]
18619            .iter()
18620            .flat_map(|v| v.to_le_bytes())
18621            .collect();
18622        writer.write_dataset_raw(idx0, &i_data).unwrap();
18623
18624        let idx1 = writer
18625            .create_dataset("floats", DatatypeMessage::f32_type(), &[2, 2])
18626            .unwrap();
18627        let f_data: Vec<u8> = [1.0f32, 2.0, 3.0, 4.0]
18628            .iter()
18629            .flat_map(|v| v.to_le_bytes())
18630            .collect();
18631        writer.write_dataset_raw(idx1, &f_data).unwrap();
18632
18633        writer.close().unwrap();
18634
18635        let mut reader = Hdf5Reader::open(&path).unwrap();
18636        let names = reader.dataset_names();
18637        assert!(names.contains(&"ints"));
18638        assert!(names.contains(&"floats"));
18639        assert_eq!(reader.dataset_shape("ints").unwrap(), vec![3]);
18640        assert_eq!(reader.dataset_shape("floats").unwrap(), vec![2, 2]);
18641        assert_eq!(reader.read_dataset_raw("ints").unwrap(), i_data);
18642        assert_eq!(reader.read_dataset_raw("floats").unwrap(), f_data);
18643
18644        std::fs::remove_file(&path).ok();
18645    }
18646
18647    #[test]
18648    fn data_size_mismatch() {
18649        let path = temp_path("mismatch");
18650
18651        let writer = Hdf5Writer::create(&path).unwrap();
18652        let idx = writer
18653            .create_dataset("x", DatatypeMessage::u8_type(), &[4])
18654            .unwrap();
18655        let err = writer.write_dataset_raw(idx, &[1, 2, 3]); // 3 bytes instead of 4
18656        assert!(err.is_err());
18657
18658        std::fs::remove_file(&path).ok();
18659    }
18660
18661    #[test]
18662    fn create_chunked_dataset_simple() {
18663        let path = temp_path("chunked_simple");
18664
18665        let writer = Hdf5Writer::create(&path).unwrap();
18666        let idx = writer
18667            .create_chunked_dataset(
18668                "data",
18669                DatatypeMessage::f64_type(),
18670                &[0, 4],        // start empty
18671                &[u64::MAX, 4], // unlimited first dim
18672                &[1, 4],        // chunk = [1, 4]
18673            )
18674            .unwrap();
18675
18676        // Write 3 frames (chunks)
18677        for frame in 0..3u64 {
18678            let values: Vec<f64> = (0..4).map(|i| (frame * 4 + i) as f64).collect();
18679            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18680            writer.write_chunk(idx, frame, &raw).unwrap();
18681        }
18682
18683        // Extend dimensions
18684        writer.extend_dataset(idx, &[3, 4]).unwrap();
18685
18686        writer.close().unwrap();
18687
18688        // Read back
18689        let mut reader = Hdf5Reader::open(&path).unwrap();
18690        assert_eq!(reader.dataset_names(), vec!["data"]);
18691        assert_eq!(reader.dataset_shape("data").unwrap(), vec![3, 4]);
18692
18693        let raw = reader.read_dataset_raw("data").unwrap();
18694        let values: Vec<f64> = raw
18695            .chunks(8)
18696            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
18697            .collect();
18698        assert_eq!(values.len(), 12);
18699        for (i, val) in values.iter().enumerate() {
18700            assert_eq!(*val, i as f64);
18701        }
18702
18703        std::fs::remove_file(&path).ok();
18704    }
18705
18706    #[test]
18707    fn chunked_dataset_many_frames() {
18708        let path = temp_path("chunked_many");
18709
18710        let writer = Hdf5Writer::create(&path).unwrap();
18711        let idx = writer
18712            .create_chunked_dataset(
18713                "frames",
18714                DatatypeMessage::i32_type(),
18715                &[0, 2],
18716                &[u64::MAX, 2],
18717                &[1, 2],
18718            )
18719            .unwrap();
18720
18721        let n_frames = 10u64;
18722        for frame in 0..n_frames {
18723            let values = [(frame * 2) as i32, (frame * 2 + 1) as i32];
18724            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18725            writer.write_chunk(idx, frame, &raw).unwrap();
18726        }
18727
18728        writer.extend_dataset(idx, &[n_frames, 2]).unwrap();
18729        writer.close().unwrap();
18730
18731        // Read back
18732        let mut reader = Hdf5Reader::open(&path).unwrap();
18733        assert_eq!(reader.dataset_shape("frames").unwrap(), vec![10, 2]);
18734
18735        let raw = reader.read_dataset_raw("frames").unwrap();
18736        let values: Vec<i32> = raw
18737            .chunks(4)
18738            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18739            .collect();
18740        assert_eq!(values.len(), 20);
18741        for (i, val) in values.iter().enumerate() {
18742            assert_eq!(*val, i as i32);
18743        }
18744
18745        std::fs::remove_file(&path).ok();
18746    }
18747
18748    #[test]
18749    fn create_fixed_array_dataset_roundtrip() {
18750        let path = temp_path("fixed_array");
18751
18752        let writer = Hdf5Writer::create(&path).unwrap();
18753        let idx = writer
18754            .create_fixed_array_dataset(
18755                "grid",
18756                DatatypeMessage::i32_type(),
18757                &[4, 6], // 4x6 grid
18758                &[2, 3], // chunk = 2x3
18759            )
18760            .unwrap();
18761
18762        // Write all chunks: 2x2 = 4 chunks
18763        // chunk (0,0): rows 0-1, cols 0-2
18764        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18765            .iter()
18766            .flat_map(|v| v.to_le_bytes())
18767            .collect();
18768        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18769
18770        // chunk (0,1): rows 0-1, cols 3-5
18771        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18772            .iter()
18773            .flat_map(|v| v.to_le_bytes())
18774            .collect();
18775        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
18776
18777        // chunk (1,0): rows 2-3, cols 0-2
18778        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
18779            .iter()
18780            .flat_map(|v| v.to_le_bytes())
18781            .collect();
18782        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
18783
18784        // chunk (1,1): rows 2-3, cols 3-5
18785        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
18786            .iter()
18787            .flat_map(|v| v.to_le_bytes())
18788            .collect();
18789        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
18790
18791        writer.close().unwrap();
18792
18793        // Read back
18794        let mut reader = Hdf5Reader::open(&path).unwrap();
18795        assert_eq!(reader.dataset_names(), vec!["grid"]);
18796        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18797
18798        let raw = reader.read_dataset_raw("grid").unwrap();
18799        let values: Vec<i32> = raw
18800            .chunks(4)
18801            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18802            .collect();
18803        assert_eq!(values.len(), 24);
18804        for (i, val) in values.iter().enumerate() {
18805            assert_eq!(*val, i as i32);
18806        }
18807
18808        std::fs::remove_file(&path).ok();
18809    }
18810
18811    #[test]
18812    fn fixed_array_paged_dblk_disk_size() {
18813        let ctx = FormatContext {
18814            sizeof_addr: 8,
18815            sizeof_size: 8,
18816        };
18817        // 1024 elements per page (bits=10). 3000 chunks => 3 pages.
18818        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 3000);
18819        assert!(hdr.is_paged());
18820        assert_eq!(hdr.npages(), 3);
18821        // prefix: 4+1+1+8 + bitmap(1) + cksum(4) = 19
18822        // elements: 3000 * 8 = 24000 ; per-page cksum: 3 * 4 = 12
18823        assert_eq!(fixed_array_dblk_disk_size(&ctx, &hdr), 19 + 24000 + 12);
18824
18825        // Non-paged: 1000 elements. prefix(14) + 1000*8 + cksum(4).
18826        let small = FixedArrayHeader::new_for_chunks(&ctx, 1000);
18827        assert!(!small.is_paged());
18828        assert_eq!(fixed_array_dblk_disk_size(&ctx, &small), 14 + 8000 + 4);
18829    }
18830
18831    #[test]
18832    fn fixed_array_paged_encode_matches_reader_layout() {
18833        let ctx = FormatContext {
18834            sizeof_addr: 8,
18835            sizeof_size: 8,
18836        };
18837        let mut hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18838        hdr.data_blk_addr = 0x9000;
18839        let npages = hdr.npages() as usize; // ceil(2500/1024) = 3
18840
18841        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18842        for (i, e) in dblk.elements.iter_mut().enumerate() {
18843            *e = 0x10000 + (i as u64) * 0x100;
18844        }
18845
18846        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18847        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18848
18849        // Decode the prefix and pages exactly as the reader does.
18850        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18851        assert_eq!(prefix.header_addr, 0x1000);
18852        for p in 0..npages {
18853            assert!(prefix.page_initialized(p), "page {p} should be initialized");
18854        }
18855
18856        let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
18857        let page_stride = dblk_page_nelmts * 8 + 4;
18858        let mut recovered = Vec::new();
18859        for p in 0..npages {
18860            let page_nelmts = if p + 1 == npages {
18861                2500 - p * dblk_page_nelmts
18862            } else {
18863                dblk_page_nelmts
18864            };
18865            let off = prefix.prefix_size + p * page_stride;
18866            let page_buf = &encoded[off..];
18867            let addrs = crate::format::chunk_index::fixed_array::decode_unfiltered_page(
18868                page_buf,
18869                &ctx,
18870                page_nelmts,
18871            )
18872            .unwrap();
18873            recovered.extend(addrs);
18874        }
18875        assert_eq!(recovered, dblk.elements);
18876    }
18877
18878    #[test]
18879    fn fixed_array_paged_decode_roundtrip_with_uninitialized_page() {
18880        let ctx = FormatContext {
18881            sizeof_addr: 8,
18882            sizeof_size: 8,
18883        };
18884        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18885        let npages = hdr.npages() as usize; // 3
18886        let page = hdr.dblk_page_nelmts() as usize; // 1024
18887
18888        // Populate pages 0 and 2; leave page 1 entirely undefined so its
18889        // bitmap bit stays clear on encode.
18890        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18891        for i in (0..page).chain(2 * page..2500) {
18892            dblk.elements[i] = 0x10000 + (i as u64) * 0x100;
18893        }
18894
18895        let mut encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18896        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18897        assert!(prefix.page_initialized(0));
18898        assert!(!prefix.page_initialized(1));
18899        assert!(prefix.page_initialized(2));
18900
18901        // Corrupt the uninitialized page's bytes the way libhdf5 leaves
18902        // them: arbitrary, no valid checksum. Decode must not look at it.
18903        let page_stride = page * 8 + 4;
18904        let p1 = prefix.prefix_size + page_stride;
18905        for b in &mut encoded[p1..p1 + page_stride] {
18906            *b = 0x5A;
18907        }
18908
18909        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, 0).unwrap();
18910        assert_eq!(decoded.elements, dblk.elements);
18911        assert_eq!(decoded.header_addr, 0x1000);
18912    }
18913
18914    #[test]
18915    fn fixed_array_paged_decode_filtered_roundtrip() {
18916        let ctx = FormatContext {
18917            sizeof_addr: 8,
18918            sizeof_size: 8,
18919        };
18920        let chunk_size_len = 4usize;
18921        let hdr = FixedArrayHeader::new_for_filtered_chunks(&ctx, 1500, chunk_size_len as u8);
18922        assert!(hdr.is_paged());
18923
18924        let mut dblk = FixedArrayDataBlock::new_filtered(0x2000, 1500);
18925        for (i, e) in dblk.filtered_elements.iter_mut().enumerate() {
18926            e.address = 0x8000 + (i as u64) * 0x40;
18927            e.chunk_size = 100 + i as u64;
18928            e.filter_mask = (i % 3) as u32;
18929        }
18930
18931        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18932        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18933        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, chunk_size_len).unwrap();
18934        assert_eq!(decoded.filtered_elements, dblk.filtered_elements);
18935        assert_eq!(decoded.client_id, FA_CLIENT_FILT_CHUNK);
18936    }
18937
18938    #[test]
18939    fn create_fixed_array_paged_dataset_roundtrip() {
18940        let path = temp_path("fixed_array_paged");
18941
18942        // 1D dataset of 3000 elements, chunk size 1 => 3000 chunks.
18943        // 3000 > 1024 (one page) => the FA data block must be paged.
18944        let n: usize = 3000;
18945        let writer = Hdf5Writer::create(&path).unwrap();
18946        let idx = writer
18947            .create_fixed_array_dataset("paged", DatatypeMessage::i32_type(), &[n as u64], &[1])
18948            .unwrap();
18949
18950        for i in 0..n {
18951            let v = (i as i32).to_le_bytes();
18952            writer
18953                .write_chunk_fixed_array(idx, &[i as u64], &v)
18954                .unwrap();
18955        }
18956        writer.close().unwrap();
18957
18958        let mut reader = Hdf5Reader::open(&path).unwrap();
18959        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18960        let raw = reader.read_dataset_raw("paged").unwrap();
18961        let values: Vec<i32> = raw
18962            .chunks(4)
18963            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18964            .collect();
18965        assert_eq!(values.len(), n);
18966        for (i, v) in values.iter().enumerate() {
18967            assert_eq!(*v, i as i32, "element {i}");
18968        }
18969
18970        std::fs::remove_file(&path).ok();
18971    }
18972
18973    #[cfg(feature = "deflate")]
18974    #[test]
18975    fn create_filtered_fixed_array_dataset_roundtrip() {
18976        // Small compressed fixed-shape chunked dataset: flat filtered FA.
18977        let path = temp_path("fixed_array_filt");
18978
18979        let writer = Hdf5Writer::create(&path).unwrap();
18980        let idx = writer
18981            .create_fixed_array_dataset_with_pipeline(
18982                "grid",
18983                DatatypeMessage::i32_type(),
18984                &[4, 6], // 4x6 grid
18985                &[2, 3], // chunk = 2x3 => 2x2 = 4 chunks
18986                FilterPipeline::deflate(6),
18987            )
18988            .unwrap();
18989
18990        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18991            .iter()
18992            .flat_map(|v| v.to_le_bytes())
18993            .collect();
18994        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18995        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18996            .iter()
18997            .flat_map(|v| v.to_le_bytes())
18998            .collect();
18999        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
19000        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
19001            .iter()
19002            .flat_map(|v| v.to_le_bytes())
19003            .collect();
19004        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
19005        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
19006            .iter()
19007            .flat_map(|v| v.to_le_bytes())
19008            .collect();
19009        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
19010
19011        writer.close().unwrap();
19012
19013        let mut reader = Hdf5Reader::open(&path).unwrap();
19014        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
19015        let raw = reader.read_dataset_raw("grid").unwrap();
19016        let values: Vec<i32> = raw
19017            .chunks(4)
19018            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19019            .collect();
19020        assert_eq!(values.len(), 24);
19021        for (i, v) in values.iter().enumerate() {
19022            assert_eq!(*v, i as i32, "element {i}");
19023        }
19024
19025        std::fs::remove_file(&path).ok();
19026    }
19027
19028    #[cfg(feature = "deflate")]
19029    #[test]
19030    fn create_filtered_fixed_array_paged_dataset_roundtrip() {
19031        // Large compressed fixed-shape chunked dataset (>1024 chunks): the
19032        // filtered FA data block must be paged.
19033        let path = temp_path("fixed_array_filt_paged");
19034
19035        let n: usize = 3000;
19036        let writer = Hdf5Writer::create(&path).unwrap();
19037        let idx = writer
19038            .create_fixed_array_dataset_with_pipeline(
19039                "paged",
19040                DatatypeMessage::i32_type(),
19041                &[n as u64],
19042                &[1],
19043                FilterPipeline::deflate(6),
19044            )
19045            .unwrap();
19046
19047        for i in 0..n {
19048            let v = (i as i32).to_le_bytes();
19049            writer
19050                .write_chunk_fixed_array(idx, &[i as u64], &v)
19051                .unwrap();
19052        }
19053        writer.close().unwrap();
19054
19055        let mut reader = Hdf5Reader::open(&path).unwrap();
19056        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
19057        let raw = reader.read_dataset_raw("paged").unwrap();
19058        let values: Vec<i32> = raw
19059            .chunks(4)
19060            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19061            .collect();
19062        assert_eq!(values.len(), n);
19063        for (i, v) in values.iter().enumerate() {
19064            assert_eq!(*v, i as i32, "element {i}");
19065        }
19066
19067        std::fs::remove_file(&path).ok();
19068    }
19069
19070    #[test]
19071    fn filtered_fixed_array_dblk_disk_size_and_encode() {
19072        // Cross-check filtered FA data-block sizing against the encoded length,
19073        // for both flat and paged layouts.
19074        let ctx = FormatContext {
19075            sizeof_addr: 8,
19076            sizeof_size: 8,
19077        };
19078        let csl = 3u8; // chunk_size_len
19079        let elem_size = 8 + csl as usize + 4; // addr + size + filter_mask
19080
19081        // Flat: 100 chunks. prefix(14) + 100*elem_size + cksum(4).
19082        let mut flat = FixedArrayHeader::new_for_filtered_chunks(&ctx, 100, csl);
19083        flat.data_blk_addr = 0x4000;
19084        assert!(!flat.is_paged());
19085        assert_eq!(
19086            fixed_array_dblk_disk_size(&ctx, &flat),
19087            (14 + 100 * elem_size + 4) as u64
19088        );
19089        let flat_dblk = FixedArrayDataBlock::new_filtered(0x1000, 100);
19090        assert_eq!(
19091            encode_fixed_array_dblk(&ctx, &flat, &flat_dblk).len() as u64,
19092            fixed_array_dblk_disk_size(&ctx, &flat)
19093        );
19094
19095        // Paged: 2500 chunks => 3 pages. prefix(4+1+1+8+1+4=19)
19096        // + 2500*elem_size + 3*cksum(4).
19097        let mut paged = FixedArrayHeader::new_for_filtered_chunks(&ctx, 2500, csl);
19098        paged.data_blk_addr = 0x9000;
19099        assert!(paged.is_paged());
19100        assert_eq!(paged.npages(), 3);
19101        assert_eq!(
19102            fixed_array_dblk_disk_size(&ctx, &paged),
19103            (19 + 2500 * elem_size + 12) as u64
19104        );
19105        let mut paged_dblk = FixedArrayDataBlock::new_filtered(0x1000, 2500);
19106        for (i, e) in paged_dblk.filtered_elements.iter_mut().enumerate() {
19107            e.address = 0x10000 + (i as u64) * 0x100;
19108            e.chunk_size = (i % 200) as u64;
19109        }
19110        let encoded = encode_fixed_array_dblk(&ctx, &paged, &paged_dblk);
19111        assert_eq!(
19112            encoded.len() as u64,
19113            fixed_array_dblk_disk_size(&ctx, &paged)
19114        );
19115
19116        // Decode the paged prefix + pages as the reader does.
19117        let npages = paged.npages() as usize;
19118        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
19119        for p in 0..npages {
19120            assert!(prefix.page_initialized(p), "page {p}");
19121        }
19122        let dblk_page_nelmts = paged.dblk_page_nelmts() as usize;
19123        let page_stride = dblk_page_nelmts * elem_size + 4;
19124        let mut recovered = Vec::new();
19125        for p in 0..npages {
19126            let page_nelmts = if p + 1 == npages {
19127                2500 - p * dblk_page_nelmts
19128            } else {
19129                dblk_page_nelmts
19130            };
19131            let off = prefix.prefix_size + p * page_stride;
19132            let elems = crate::format::chunk_index::fixed_array::decode_filtered_page(
19133                &encoded[off..],
19134                &ctx,
19135                page_nelmts,
19136                csl as usize,
19137            )
19138            .unwrap();
19139            recovered.extend(elems);
19140        }
19141        assert_eq!(recovered, paged_dblk.filtered_elements);
19142    }
19143
19144    #[test]
19145    fn create_btree_v2_dataset_roundtrip() {
19146        let path = temp_path("btree_v2");
19147
19148        let writer = Hdf5Writer::create(&path).unwrap();
19149        let idx = writer
19150            .create_btree_v2_dataset(
19151                "data",
19152                DatatypeMessage::f64_type(),
19153                &[0, 0],               // start empty
19154                &[u64::MAX, u64::MAX], // both dims unlimited
19155                &[2, 3],               // chunk = 2x3
19156            )
19157            .unwrap();
19158
19159        // Write chunks for a 4x6 dataset
19160        // chunk (0,0)
19161        let c00: Vec<u8> = [0.0f64, 1.0, 2.0, 6.0, 7.0, 8.0]
19162            .iter()
19163            .flat_map(|v| v.to_le_bytes())
19164            .collect();
19165        writer.write_chunk_btree_v2(idx, &[0, 0], &c00).unwrap();
19166
19167        // chunk (0,1)
19168        let c01: Vec<u8> = [3.0f64, 4.0, 5.0, 9.0, 10.0, 11.0]
19169            .iter()
19170            .flat_map(|v| v.to_le_bytes())
19171            .collect();
19172        writer.write_chunk_btree_v2(idx, &[0, 1], &c01).unwrap();
19173
19174        // chunk (1,0)
19175        let c10: Vec<u8> = [12.0f64, 13.0, 14.0, 18.0, 19.0, 20.0]
19176            .iter()
19177            .flat_map(|v| v.to_le_bytes())
19178            .collect();
19179        writer.write_chunk_btree_v2(idx, &[1, 0], &c10).unwrap();
19180
19181        // chunk (1,1)
19182        let c11: Vec<u8> = [15.0f64, 16.0, 17.0, 21.0, 22.0, 23.0]
19183            .iter()
19184            .flat_map(|v| v.to_le_bytes())
19185            .collect();
19186        writer.write_chunk_btree_v2(idx, &[1, 1], &c11).unwrap();
19187
19188        writer.extend_dataset(idx, &[4, 6]).unwrap();
19189        writer.close().unwrap();
19190
19191        // Read back
19192        let mut reader = Hdf5Reader::open(&path).unwrap();
19193        assert_eq!(reader.dataset_names(), vec!["data"]);
19194        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4, 6]);
19195
19196        let raw = reader.read_dataset_raw("data").unwrap();
19197        let values: Vec<f64> = raw
19198            .chunks(8)
19199            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
19200            .collect();
19201        assert_eq!(values.len(), 24);
19202        for (i, val) in values.iter().enumerate() {
19203            assert_eq!(*val, i as f64);
19204        }
19205
19206        std::fs::remove_file(&path).ok();
19207    }
19208
19209    /// Bytes one chunk of [`btree_v2_flush_probe`]'s dataset occupies — an
19210    /// f64 element, so the allocator's alignment neither pads nor merges it and
19211    /// the file's growth is exactly the bytes asked for.
19212    const BT2_PROBE_CHUNK: u64 = 8;
19213
19214    /// Write chunks of a 1x1-chunked 2-D BT2 dataset, flushing at each batch
19215    /// boundary, and report `(node addresses, file length)` after every flush.
19216    /// Chunks are addressed down column 0 so the record count — and hence the
19217    /// tree's shape — grows one record at a time.
19218    fn btree_v2_flush_probe(path: &std::path::Path, batches: &[u64]) -> Vec<(Vec<u64>, u64)> {
19219        let writer = Hdf5Writer::create(path).unwrap();
19220        let idx = writer
19221            .create_btree_v2_dataset(
19222                "data",
19223                DatatypeMessage::f64_type(),
19224                &[0, 0],
19225                &[u64::MAX, u64::MAX],
19226                &[1, 1],
19227            )
19228            .unwrap();
19229        let mut written = 0u64;
19230        let mut out = Vec::new();
19231        for &upto in batches {
19232            while written < upto {
19233                writer
19234                    .write_chunk_btree_v2(idx, &[written, 0], &(written as f64).to_le_bytes())
19235                    .unwrap();
19236                written += 1;
19237            }
19238            writer.flush_dataset(idx).unwrap();
19239            let addrs = writer
19240                .ds(idx)
19241                .lock()
19242                .btree_v2
19243                .as_ref()
19244                .unwrap()
19245                .node_addrs
19246                .clone();
19247            out.push((addrs, std::fs::metadata(path).unwrap().len()));
19248        }
19249        writer.extend_dataset(idx, &[written.max(1), 1]).unwrap();
19250        writer.close().unwrap();
19251        out
19252    }
19253
19254    /// The node pool tracks the tree in both directions. Dropping records is
19255    /// what a removal path would do — [`Bt2ChunkIndex`] has none today, so the
19256    /// test drops them itself — and the flush that follows must hand the blocks
19257    /// its smaller tree no longer needs back to the allocator instead of
19258    /// leaving them recorded and unreachable.
19259    #[test]
19260    fn a_btree_v2_flush_frees_the_node_blocks_its_tree_gave_up() {
19261        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
19262
19263        let path = temp_path("bt2_node_shrink");
19264        let writer = Hdf5Writer::create(&path).unwrap();
19265        let idx = writer
19266            .create_btree_v2_dataset(
19267                "data",
19268                DatatypeMessage::f64_type(),
19269                &[0, 0],
19270                &[u64::MAX, u64::MAX],
19271                &[1, 1],
19272            )
19273            .unwrap();
19274        // 85 records is one past a leaf, so the tree is two leaves and a root.
19275        for i in 0..85u64 {
19276            writer
19277                .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19278                .unwrap();
19279        }
19280        writer.flush_dataset(idx).unwrap();
19281        let grown = writer
19282            .ds(idx)
19283            .lock()
19284            .btree_v2
19285            .as_ref()
19286            .unwrap()
19287            .node_addrs
19288            .clone();
19289        assert_eq!(grown.len(), 3, "expected two leaves and a root");
19290
19291        // Back to 84 records: one leaf, so two of the three blocks are surplus.
19292        writer
19293            .ds(idx)
19294            .lock()
19295            .btree_v2
19296            .as_mut()
19297            .unwrap()
19298            .index
19299            .records
19300            .truncate(84);
19301        writer.flush_dataset(idx).unwrap();
19302        let shrunk = writer
19303            .ds(idx)
19304            .lock()
19305            .btree_v2
19306            .as_ref()
19307            .unwrap()
19308            .node_addrs
19309            .clone();
19310        assert_eq!(
19311            shrunk,
19312            grown[..1],
19313            "the pool still records the surplus blocks"
19314        );
19315
19316        // The surplus went back to the allocator, not on the floor: the next
19317        // node-sized allocation lands inside the region the two blocks covered.
19318        let reused = writer
19319            .allocator
19320            .allocate(BT2_NODE_SIZE as u64, FreeSpaceClass::Metadata);
19321        assert!(
19322            (grown[1]..grown[1] + 2 * BT2_NODE_SIZE as u64).contains(&reused),
19323            "a node block allocated at {reused:#x}, outside the freed \
19324             [{:#x}, {:#x}) the flush gave up",
19325            grown[1],
19326            grown[1] + 2 * BT2_NODE_SIZE as u64
19327        );
19328
19329        writer.extend_dataset(idx, &[85, 1]).unwrap();
19330        writer.close().unwrap();
19331        std::fs::remove_file(&path).ok();
19332    }
19333
19334    /// A v2 B-tree whose header declares a non-default node size — libhdf5
19335    /// built with a different `H5D_BT2_NODE_SIZE`, or any other writer —
19336    /// reopens for append: the reconstruction adopts the header's node_size,
19337    /// split and merge instead of refusing everything but 2048, and the next
19338    /// flush re-serializes at that size (upstream allocates every node at
19339    /// `hdr->node_size`, H5B2leaf.c / H5B2internal.c).
19340    #[test]
19341    fn a_btree_v2_with_a_foreign_node_size_reopens_and_grows() {
19342        let path = temp_path("bt2_foreign_node_size");
19343        {
19344            let writer = Hdf5Writer::create(&path).unwrap();
19345            let idx = writer
19346                .create_btree_v2_dataset(
19347                    "data",
19348                    DatatypeMessage::f64_type(),
19349                    &[0, 0],
19350                    &[u64::MAX, u64::MAX],
19351                    &[1, 1],
19352                )
19353                .unwrap();
19354            // Act as a foreign writer: 512-byte nodes, non-default tuning.
19355            // record_size 24 => a 512-byte leaf holds 20 records, so 85
19356            // records make a depth-1 tree of 512-byte blocks.
19357            {
19358                let ds = writer.ds(idx);
19359                let mut m = ds.lock();
19360                let index = &mut m.btree_v2.as_mut().unwrap().index;
19361                index.node_size = 512;
19362                index.split_percent = 90;
19363                index.merge_percent = 30;
19364            }
19365            for i in 0..85u64 {
19366                writer
19367                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19368                    .unwrap();
19369            }
19370            writer.extend_dataset(idx, &[85, 1]).unwrap();
19371            writer.close().unwrap();
19372        }
19373        {
19374            let writer = Hdf5Writer::open_append(&path).unwrap();
19375            let idx = writer.dataset_index("data").unwrap();
19376            {
19377                let ds = writer.ds(idx);
19378                let m = ds.lock();
19379                let index = &m.btree_v2.as_ref().unwrap().index;
19380                assert_eq!(index.node_size, 512, "header node_size not adopted");
19381                assert_eq!(index.split_percent, 90);
19382                assert_eq!(index.merge_percent, 30);
19383                assert_eq!(index.records.len(), 85, "records not walked back");
19384            }
19385            for i in 85..115u64 {
19386                writer
19387                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19388                    .unwrap();
19389            }
19390            writer.extend_dataset(idx, &[115, 1]).unwrap();
19391            writer.close().unwrap();
19392        }
19393
19394        let mut reader = Hdf5Reader::open(&path).unwrap();
19395        let raw = reader.read_dataset_raw("data").unwrap();
19396        let values: Vec<f64> = raw
19397            .chunks(8)
19398            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19399            .collect();
19400        assert_eq!(values.len(), 115);
19401        for (i, v) in values.iter().enumerate() {
19402            assert_eq!(*v, i as f64, "element {i}");
19403        }
19404        std::fs::remove_file(&path).ok();
19405    }
19406
19407    /// A node's record count falls as well as rises: the tree's first leaf goes
19408    /// from a full 84 records to 42 when 85 records force it to split. The node
19409    /// image is padded to the whole block so re-serializing overwrites the
19410    /// block, not a prefix of it — otherwise that leaf keeps the tail of its
19411    /// 84-record self, stale records sitting in a live node block.
19412    #[test]
19413    fn a_shrinking_btree_v2_node_leaves_no_stale_records_behind() {
19414        use crate::format::chunk_index::btree_v2::{Bt2ChunkIndex, BT2_NODE_SIZE};
19415
19416        let path = temp_path("bt2_node_blocks");
19417        let probe = btree_v2_flush_probe(&path, &[84, 85]);
19418        let node0 = probe.last().unwrap().0[0];
19419
19420        // What the first leaf holds once the tree has split.
19421        let ctx = FormatContext {
19422            sizeof_addr: 8,
19423            sizeof_size: 8,
19424        };
19425        let mut index = Bt2ChunkIndex::new_unfiltered(2);
19426        for i in 0..85u64 {
19427            index.insert(vec![i, 0], 0);
19428        }
19429        let tree = index.build_tree(&ctx);
19430        assert!(
19431            tree.nodes[0].num_records < 84,
19432            "this test needs the first leaf to shrink, got {}",
19433            tree.nodes[0].num_records
19434        );
19435        // signature(4) + version(1) + type(1) + records + checksum(4)
19436        let used = 10 + tree.nodes[0].num_records as usize * tree.record_size as usize;
19437
19438        let bytes = std::fs::read(&path).unwrap();
19439        let block = &bytes[node0 as usize..node0 as usize + BT2_NODE_SIZE as usize];
19440        assert!(
19441            block[used..].iter().all(|&b| b == 0),
19442            "leaf block at {node0:#x} still holds {} bytes of its previous, larger image",
19443            block[used..].iter().rposition(|&b| b != 0).unwrap_or(0) + 1
19444        );
19445        std::fs::remove_file(&path).ok();
19446    }
19447
19448    /// The node pool is the single owner of the tree's block addresses: a flush
19449    /// reuses every block already in it and allocates only the shortfall. So
19450    /// re-flushing an unchanged index must cost nothing, and a flush that grows
19451    /// the tree must cost exactly the blocks it added — anything more means a
19452    /// block was stranded.
19453    #[test]
19454    fn a_btree_v2_flush_allocates_only_the_node_blocks_it_adds() {
19455        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
19456
19457        let path = temp_path("bt2_pool_growth");
19458        // Re-flush at 84 (still one leaf), then cross into a three-node depth-1
19459        // tree, then keep growing.
19460        let batches = [84u64, 84, 85, 200, 200];
19461        let probe = btree_v2_flush_probe(&path, &batches);
19462        for i in 1..probe.len() {
19463            let (prev_addrs, prev_len) = &probe[i - 1];
19464            let (addrs, len) = &probe[i];
19465            assert!(
19466                addrs.starts_with(prev_addrs),
19467                "flush {i} moved a node block instead of reusing it"
19468            );
19469            let new_blocks = (addrs.len() - prev_addrs.len()) as u64 * BT2_NODE_SIZE as u64;
19470            let new_chunks = (batches[i] - batches[i - 1]) * BT2_PROBE_CHUNK;
19471            assert_eq!(
19472                len - prev_len,
19473                new_blocks + new_chunks,
19474                "flush {i} grew the file by more than the blocks it added"
19475            );
19476        }
19477        // The unchanged re-flushes must be free.
19478        assert_eq!(probe[1].1, probe[0].1);
19479        assert_eq!(probe[4].1, probe[3].1);
19480        std::fs::remove_file(&path).ok();
19481    }
19482
19483    #[cfg(feature = "parallel")]
19484    #[test]
19485    fn parallel_batch_write_roundtrip() {
19486        let path = temp_path("parallel_batch");
19487
19488        let writer = Hdf5Writer::create(&path).unwrap();
19489        let idx = writer
19490            .create_chunked_dataset(
19491                "data",
19492                DatatypeMessage::i32_type(),
19493                &[0, 4],
19494                &[u64::MAX, 4],
19495                &[1, 4],
19496            )
19497            .unwrap();
19498
19499        // Prepare chunks
19500        let chunks_data: Vec<(u64, Vec<u8>)> = (0..8u64)
19501            .map(|frame| {
19502                let values: Vec<i32> = (0..4).map(|i| (frame * 4 + i) as i32).collect();
19503                let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
19504                (frame, raw)
19505            })
19506            .collect();
19507
19508        let batch: Vec<(u64, &[u8])> = chunks_data
19509            .iter()
19510            .map(|(idx, data)| (*idx, data.as_slice()))
19511            .collect();
19512
19513        writer.write_chunks_batch(idx, &batch).unwrap();
19514        writer.extend_dataset(idx, &[8, 4]).unwrap();
19515        writer.close().unwrap();
19516
19517        // Read back
19518        let mut reader = Hdf5Reader::open(&path).unwrap();
19519        assert_eq!(reader.dataset_shape("data").unwrap(), vec![8, 4]);
19520        let raw = reader.read_dataset_raw("data").unwrap();
19521        let values: Vec<i32> = raw
19522            .chunks(4)
19523            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
19524            .collect();
19525        assert_eq!(values.len(), 32);
19526        for (i, val) in values.iter().enumerate() {
19527            assert_eq!(*val, i as i32);
19528        }
19529
19530        std::fs::remove_file(&path).ok();
19531    }
19532
19533    #[test]
19534    fn swmr_writer_append_frames() {
19535        use crate::io::swmr::SwmrWriter;
19536
19537        // Per-call unique path so concurrent cargo invocations and
19538        // kernel-side flock release races cannot collide.
19539        use std::sync::atomic::{AtomicU64, Ordering};
19540        static COUNTER: AtomicU64 = AtomicU64::new(0);
19541        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19542        let path = std::env::temp_dir().join(format!(
19543            "rust_hdf5_swmr_append_{}_{}.h5",
19544            std::process::id(),
19545            n
19546        ));
19547
19548        let mut swmr = SwmrWriter::create(&path).unwrap();
19549        let idx = swmr
19550            .create_streaming_dataset("detector", DatatypeMessage::u16_type(), &[4, 4])
19551            .unwrap();
19552
19553        swmr.start_swmr().unwrap();
19554
19555        // Append 5 frames
19556        for frame in 0..5u16 {
19557            let data: Vec<u16> = (0..16).map(|i| frame * 16 + i).collect();
19558            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19559            swmr.append_frame(idx, &raw).unwrap();
19560        }
19561
19562        swmr.flush().unwrap();
19563        swmr.close().unwrap();
19564
19565        // Read back
19566        let mut reader = Hdf5Reader::open(&path).unwrap();
19567        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![5, 4, 4]);
19568
19569        let raw = reader.read_dataset_raw("detector").unwrap();
19570        let values: Vec<u16> = raw
19571            .chunks(2)
19572            .map(|chunk| u16::from_le_bytes(chunk.try_into().unwrap()))
19573            .collect();
19574        assert_eq!(values.len(), 80); // 5 * 4 * 4
19575                                      // Verify first frame
19576        for (i, val) in values.iter().enumerate().take(16) {
19577            assert_eq!(*val, i as u16);
19578        }
19579        // Verify last frame
19580        for (i, val) in values[64..80].iter().enumerate() {
19581            assert_eq!(*val, 4 * 16 + i as u16);
19582        }
19583
19584        std::fs::remove_file(&path).ok();
19585    }
19586
19587    #[test]
19588    fn swmr_writer_tiled_frames() {
19589        use crate::io::swmr::SwmrWriter;
19590        use std::sync::atomic::{AtomicU64, Ordering};
19591        static COUNTER: AtomicU64 = AtomicU64::new(0);
19592        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19593        let path = std::env::temp_dir().join(format!(
19594            "rust_hdf5_swmr_tiled_{}_{}.h5",
19595            std::process::id(),
19596            n
19597        ));
19598
19599        let mut swmr = SwmrWriter::create(&path).unwrap();
19600        // 4x4 frames, tiled into 2x2 chunks -> 4 chunks per frame.
19601        let idx = swmr
19602            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[4, 4], &[2, 2])
19603            .unwrap();
19604        swmr.start_swmr().unwrap();
19605
19606        for frame in 0..3u16 {
19607            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
19608            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19609            swmr.append_frame(idx, &raw).unwrap();
19610        }
19611        swmr.flush().unwrap();
19612        swmr.close().unwrap();
19613
19614        let mut reader = Hdf5Reader::open(&path).unwrap();
19615        assert_eq!(reader.dataset_shape("det").unwrap(), vec![3, 4, 4]);
19616        let raw = reader.read_dataset_raw("det").unwrap();
19617        let values: Vec<u16> = raw
19618            .chunks(2)
19619            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19620            .collect();
19621        assert_eq!(values.len(), 48);
19622        // Every element must survive the frame -> tile split and the
19623        // tile -> frame reassembly on read.
19624        for frame in 0..3u16 {
19625            for i in 0..16usize {
19626                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
19627            }
19628        }
19629        std::fs::remove_file(&path).ok();
19630    }
19631
19632    /// A chunk tile larger than the frame is geometry libhdf5 refuses to
19633    /// create (`H5D__chunk_construct`: chunk must not exceed a fixed maximum
19634    /// dimension), so no libhdf5-based writer — including the NDFileHDF5
19635    /// tiling controls this API mirrors — can produce such a file. Until
19636    /// 0.4.1 we accepted it and zero-padded the frame up to the tile; now
19637    /// the create is rejected like every other creator's.
19638    #[test]
19639    fn swmr_writer_tiled_chunk_larger_than_frame_is_rejected() {
19640        use crate::io::swmr::SwmrWriter;
19641        use std::sync::atomic::{AtomicU64, Ordering};
19642        static COUNTER: AtomicU64 = AtomicU64::new(0);
19643        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19644        let path = std::env::temp_dir().join(format!(
19645            "rust_hdf5_swmr_bigchunk_{}_{}.h5",
19646            std::process::id(),
19647            n
19648        ));
19649
19650        let mut swmr = SwmrWriter::create(&path).unwrap();
19651        let err = swmr
19652            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[3, 3], &[8, 8])
19653            .unwrap_err();
19654        assert!(
19655            err.to_string().contains("maximum dimension size"),
19656            "unexpected error: {err}"
19657        );
19658        swmr.close().unwrap();
19659        std::fs::remove_file(&path).ok();
19660    }
19661
19662    #[test]
19663    fn swmr_writer_multi_frame_chunks() {
19664        use crate::io::swmr::SwmrWriter;
19665        use std::sync::atomic::{AtomicU64, Ordering};
19666        static COUNTER: AtomicU64 = AtomicU64::new(0);
19667        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19668        let path = std::env::temp_dir().join(format!(
19669            "rust_hdf5_swmr_mfc_{}_{}.h5",
19670            std::process::id(),
19671            n
19672        ));
19673
19674        // 3x3 frames, chunk = 4 frames x full frame. 10 frames -> 3 bands
19675        // of 4, 4, 2 (the last band partial).
19676        let mut swmr = SwmrWriter::create(&path).unwrap();
19677        let idx = swmr
19678            .create_streaming_dataset_chunked(
19679                "det",
19680                DatatypeMessage::u16_type(),
19681                &[3, 3],
19682                &[4, 3, 3],
19683            )
19684            .unwrap();
19685        swmr.start_swmr().unwrap();
19686        for frame in 0..10u16 {
19687            let data: Vec<u16> = (0..9).map(|i| frame * 100 + i).collect();
19688            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19689            swmr.append_frame(idx, &raw).unwrap();
19690        }
19691        swmr.flush().unwrap();
19692        swmr.close().unwrap();
19693
19694        let mut reader = Hdf5Reader::open(&path).unwrap();
19695        // The partial last band must not over-extend the frame count.
19696        assert_eq!(reader.dataset_shape("det").unwrap(), vec![10, 3, 3]);
19697        let raw = reader.read_dataset_raw("det").unwrap();
19698        let values: Vec<u16> = raw
19699            .chunks(2)
19700            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19701            .collect();
19702        assert_eq!(values.len(), 90);
19703        for frame in 0..10u16 {
19704            for i in 0..9usize {
19705                assert_eq!(values[frame as usize * 9 + i], frame * 100 + i as u16);
19706            }
19707        }
19708        std::fs::remove_file(&path).ok();
19709    }
19710
19711    #[test]
19712    fn swmr_writer_multi_frame_tiled_chunks() {
19713        use crate::io::swmr::SwmrWriter;
19714        use std::sync::atomic::{AtomicU64, Ordering};
19715        static COUNTER: AtomicU64 = AtomicU64::new(0);
19716        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19717        let path = std::env::temp_dir().join(format!(
19718            "rust_hdf5_swmr_mftc_{}_{}.h5",
19719            std::process::id(),
19720            n
19721        ));
19722
19723        // 4x4 frames, chunk = 2 frames x 2x2 tiles. 5 frames -> bands of
19724        // 2, 2, 1; every frame is also split into a 2x2 tile grid.
19725        let mut swmr = SwmrWriter::create(&path).unwrap();
19726        let idx = swmr
19727            .create_streaming_dataset_chunked(
19728                "det",
19729                DatatypeMessage::u16_type(),
19730                &[4, 4],
19731                &[2, 2, 2],
19732            )
19733            .unwrap();
19734        swmr.start_swmr().unwrap();
19735        for frame in 0..5u16 {
19736            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
19737            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19738            swmr.append_frame(idx, &raw).unwrap();
19739        }
19740        swmr.flush().unwrap();
19741        swmr.close().unwrap();
19742
19743        let mut reader = Hdf5Reader::open(&path).unwrap();
19744        assert_eq!(reader.dataset_shape("det").unwrap(), vec![5, 4, 4]);
19745        let raw = reader.read_dataset_raw("det").unwrap();
19746        let values: Vec<u16> = raw
19747            .chunks(2)
19748            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19749            .collect();
19750        assert_eq!(values.len(), 80);
19751        for frame in 0..5u16 {
19752            for i in 0..16usize {
19753                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
19754            }
19755        }
19756        std::fs::remove_file(&path).ok();
19757    }
19758
19759    #[cfg(feature = "deflate")]
19760    #[test]
19761    fn swmr_writer_compressed_frames() {
19762        use crate::io::swmr::SwmrWriter;
19763        use std::sync::atomic::{AtomicU64, Ordering};
19764        static COUNTER: AtomicU64 = AtomicU64::new(0);
19765        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19766        let path = std::env::temp_dir().join(format!(
19767            "rust_hdf5_swmr_comp_{}_{}.h5",
19768            std::process::id(),
19769            n
19770        ));
19771
19772        let mut swmr = SwmrWriter::create(&path).unwrap();
19773        let pipeline = crate::format::messages::filter::FilterPipeline::deflate(4);
19774        let idx = swmr
19775            .create_streaming_dataset_compressed(
19776                "detector",
19777                DatatypeMessage::i32_type(),
19778                &[8],
19779                pipeline,
19780            )
19781            .unwrap();
19782        swmr.start_swmr().unwrap();
19783
19784        for frame in 0..40i32 {
19785            let raw: Vec<u8> = (0..8).flat_map(|i| (frame * 8 + i).to_le_bytes()).collect();
19786            swmr.append_frame(idx, &raw).unwrap();
19787            if frame % 7 == 0 {
19788                swmr.flush().unwrap();
19789            }
19790        }
19791        swmr.flush().unwrap();
19792        swmr.close().unwrap();
19793
19794        let mut reader = Hdf5Reader::open(&path).unwrap();
19795        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![40, 8]);
19796        let raw = reader.read_dataset_raw("detector").unwrap();
19797        let values: Vec<i32> = raw
19798            .chunks(4)
19799            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19800            .collect();
19801        assert_eq!(values, (0..320).collect::<Vec<i32>>());
19802
19803        std::fs::remove_file(&path).ok();
19804    }
19805
19806    #[test]
19807    fn group_hierarchy_writer_reader() {
19808        let path = temp_path("group_hierarchy");
19809
19810        let writer = Hdf5Writer::create(&path).unwrap();
19811
19812        // Create groups
19813        let g0 = writer.create_group("/", "group1").unwrap();
19814        let g1 = writer.create_group("/group1", "sub").unwrap();
19815        assert_eq!(g0, 0);
19816        assert_eq!(g1, 1);
19817
19818        // Create datasets
19819        let ds_root = writer
19820            .create_dataset("root_data", DatatypeMessage::f64_type(), &[2])
19821            .unwrap();
19822        let raw_root: Vec<u8> = [1.0f64, 2.0].iter().flat_map(|v| v.to_le_bytes()).collect();
19823        writer.write_dataset_raw(ds_root, &raw_root).unwrap();
19824
19825        let ds_g0 = writer
19826            .create_dataset("group1/data", DatatypeMessage::i32_type(), &[3])
19827            .unwrap();
19828        let raw_g0: Vec<u8> = [10i32, 20, 30]
19829            .iter()
19830            .flat_map(|v| v.to_le_bytes())
19831            .collect();
19832        writer.write_dataset_raw(ds_g0, &raw_g0).unwrap();
19833
19834        let ds_g1 = writer
19835            .create_dataset("group1/sub/values", DatatypeMessage::u8_type(), &[4])
19836            .unwrap();
19837        writer.write_dataset_raw(ds_g1, &[1u8, 2, 3, 4]).unwrap();
19838
19839        writer.close().unwrap();
19840
19841        // Read back
19842        let mut reader = Hdf5Reader::open(&path).unwrap();
19843        let names = reader.dataset_names();
19844        assert!(names.contains(&"root_data"), "names: {:?}", names);
19845        assert!(names.contains(&"group1/data"), "names: {:?}", names);
19846        assert!(names.contains(&"group1/sub/values"), "names: {:?}", names);
19847
19848        let raw = reader.read_dataset_raw("root_data").unwrap();
19849        let vals: Vec<f64> = raw
19850            .chunks(8)
19851            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19852            .collect();
19853        assert_eq!(vals, vec![1.0, 2.0]);
19854
19855        let raw = reader.read_dataset_raw("group1/data").unwrap();
19856        let vals: Vec<i32> = raw
19857            .chunks(4)
19858            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19859            .collect();
19860        assert_eq!(vals, vec![10, 20, 30]);
19861
19862        let raw = reader.read_dataset_raw("group1/sub/values").unwrap();
19863        assert_eq!(raw, vec![1, 2, 3, 4]);
19864
19865        std::fs::remove_file(&path).ok();
19866    }
19867
19868    /// libhdf5 (`H5D__chunk_construct`) rejects a chunk dimension that
19869    /// exceeds a fixed maximum dimension. Before this check, such a dataset
19870    /// was created and appends landed rows at the chunk stride instead of
19871    /// the row stride, reading back [1, 2, 0, 0] for [1, 2, 3, 4].
19872    #[test]
19873    fn create_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
19874        let path = temp_path("chunk_wider_than_max");
19875
19876        let writer = Hdf5Writer::create(&path).unwrap();
19877        let err = writer
19878            .create_chunked_dataset(
19879                "data",
19880                DatatypeMessage::f64_type(),
19881                &[0, 2],
19882                &[u64::MAX, 2],
19883                &[2, 4],
19884            )
19885            .unwrap_err();
19886        assert!(
19887            err.to_string().contains("maximum dimension size"),
19888            "unexpected error: {err}"
19889        );
19890
19891        // The fixed-array creators derive the maximum from the fixed dims.
19892        let err = writer
19893            .create_fixed_array_dataset("fa", DatatypeMessage::f64_type(), &[3], &[5])
19894            .unwrap_err();
19895        assert!(
19896            err.to_string().contains("maximum dimension size"),
19897            "unexpected error: {err}"
19898        );
19899
19900        writer.close().unwrap();
19901        std::fs::remove_file(&path).ok();
19902    }
19903
19904    /// libhdf5 exempts a dimension whose *current* size is zero from the
19905    /// chunk-vs-maximum check (`curr_dims[u] &&` in `H5D__chunk_construct`),
19906    /// and rejects a zero chunk dimension on every path.
19907    #[test]
19908    fn create_mirrors_the_libhdf5_chunk_geometry_exemptions() {
19909        let path = temp_path("chunk_geometry_exemptions");
19910
19911        let writer = Hdf5Writer::create(&path).unwrap();
19912        // dims[1] == 0: chunk 4 > max 2 is allowed, as libhdf5 allows it.
19913        writer
19914            .create_chunked_dataset(
19915                "exempt",
19916                DatatypeMessage::f64_type(),
19917                &[0, 0],
19918                &[u64::MAX, 2],
19919                &[2, 4],
19920            )
19921            .unwrap();
19922
19923        let err = writer
19924            .create_chunked_dataset("zero", DatatypeMessage::f64_type(), &[0], &[u64::MAX], &[0])
19925            .unwrap_err();
19926        assert!(
19927            err.to_string().contains("chunk dimension 0 is zero"),
19928            "unexpected error: {err}"
19929        );
19930
19931        writer.close().unwrap();
19932        std::fs::remove_file(&path).ok();
19933    }
19934
19935    /// A file written by 0.4.0 can carry a chunk row wider than the frame
19936    /// row — create now rejects that geometry, but reopened files keep it.
19937    /// Appends must scatter frames at the chunk stride, not pack them at
19938    /// the frame stride (which read back `[1, 2, 0, 0]` for `[1, 2, 3, 4]`).
19939    /// The wide shape is simulated by widening the registered chunk dims
19940    /// after create, which also lands in the layout message at close.
19941    #[test]
19942    fn append_scatters_into_a_legacy_wider_than_row_chunk() {
19943        let path = temp_path("legacy_wide_chunk_append");
19944
19945        let writer = Hdf5Writer::create(&path).unwrap();
19946        let idx = writer
19947            .create_chunked_dataset(
19948                "data",
19949                DatatypeMessage::i32_type(),
19950                &[0, 2],
19951                &[u64::MAX, 2],
19952                &[2, 2],
19953            )
19954            .unwrap();
19955        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims = vec![2, 4];
19956
19957        let frames: Vec<u8> = [1i32, 2, 3, 4]
19958            .iter()
19959            .flat_map(|v| v.to_le_bytes())
19960            .collect();
19961        writer.write_append_frames(idx, 0, 2, &frames).unwrap();
19962        writer.extend_dataset(idx, &[2, 2]).unwrap();
19963        writer.close().unwrap();
19964
19965        let mut reader = Hdf5Reader::open(&path).unwrap();
19966        assert_eq!(reader.dataset_shape("data").unwrap(), vec![2, 2]);
19967        let raw = reader.read_dataset_raw("data").unwrap();
19968        let values: Vec<i32> = raw
19969            .chunks(4)
19970            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19971            .collect();
19972        assert_eq!(values, vec![1, 2, 3, 4]);
19973        std::fs::remove_file(&path).ok();
19974    }
19975
19976    /// The compressed vlen creator sizes its chunked layout from a
19977    /// caller-supplied chunk size; it goes through the same geometry
19978    /// validation as every other creator (empty inputs are exempt because
19979    /// their current size is zero).
19980    #[test]
19981    #[cfg(feature = "deflate")]
19982    fn compressed_vlen_create_validates_its_chunk_size() {
19983        use crate::format::messages::filter::FilterPipeline;
19984        let path = temp_path("vlen_compressed_chunk");
19985
19986        let writer = Hdf5Writer::create(&path).unwrap();
19987        let err = writer
19988            .create_vlen_string_dataset_compressed(
19989                "texts",
19990                &["a", "b", "c"],
19991                100,
19992                FilterPipeline::deflate(6),
19993            )
19994            .unwrap_err();
19995        assert!(
19996            err.to_string().contains("maximum dimension size"),
19997            "unexpected error: {err}"
19998        );
19999
20000        writer
20001            .create_vlen_string_dataset_compressed("empty", &[], 16, FilterPipeline::deflate(6))
20002            .unwrap();
20003
20004        writer.close().unwrap();
20005        std::fs::remove_file(&path).ok();
20006    }
20007
20008    /// `set_libver_latest` moves *filtered* chunked datasets to layout v5 with
20009    /// fixed 8-byte chunk-size fields; unfiltered chunked and pre-opt-in
20010    /// datasets keep v4 with the derived width, matching libhdf5's
20011    /// `version_perf` rule (only the filtered index arms bump to 5).
20012    #[cfg(feature = "deflate")]
20013    #[test]
20014    fn libver_latest_selects_v5_for_filtered_chunks_only() {
20015        let path = temp_path("libver_v5_select");
20016
20017        let mut writer = Hdf5Writer::create(&path).unwrap();
20018        let before = writer
20019            .create_chunked_dataset_with_pipeline(
20020                "d4",
20021                DatatypeMessage::i32_type(),
20022                &[0],
20023                &[u64::MAX],
20024                &[16],
20025                FilterPipeline::deflate(4),
20026            )
20027            .unwrap();
20028        writer.set_libver_latest(true).unwrap();
20029        let ea5 = writer
20030            .create_chunked_dataset_with_pipeline(
20031                "ea5",
20032                DatatypeMessage::i32_type(),
20033                &[0],
20034                &[u64::MAX],
20035                &[16],
20036                FilterPipeline::deflate(4),
20037            )
20038            .unwrap();
20039        let plain = writer
20040            .create_chunked_dataset(
20041                "plain",
20042                DatatypeMessage::i32_type(),
20043                &[0],
20044                &[u64::MAX],
20045                &[16],
20046            )
20047            .unwrap();
20048        let fa5 = writer
20049            .create_fixed_array_dataset_with_pipeline(
20050                "fa5",
20051                DatatypeMessage::i32_type(),
20052                &[4, 6],
20053                &[2, 3],
20054                FilterPipeline::deflate(6),
20055            )
20056            .unwrap();
20057        let bt5 = writer
20058            .create_btree_v2_dataset_with_pipeline(
20059                "bt5",
20060                DatatypeMessage::i32_type(),
20061                &[0, 0],
20062                &[u64::MAX, u64::MAX],
20063                &[2, 3],
20064                FilterPipeline::deflate(6),
20065            )
20066            .unwrap();
20067
20068        {
20069            let d4 = writer.ds(before);
20070            let d4 = d4.lock();
20071            assert_eq!(d4.layout_version, 4);
20072            assert_eq!(
20073                d4.chunked.as_ref().unwrap().chunk_size_len,
20074                compute_chunk_size_len(16 * 4)
20075            );
20076            let e5 = writer.ds(ea5);
20077            let e5 = e5.lock();
20078            assert_eq!(e5.layout_version, 5);
20079            assert_eq!(e5.chunked.as_ref().unwrap().chunk_size_len, 8);
20080            assert_eq!(writer.ds(plain).lock().layout_version, 4);
20081            assert_eq!(writer.ds(fa5).lock().layout_version, 5);
20082            assert_eq!(writer.ds(bt5).lock().layout_version, 5);
20083        }
20084
20085        // Write through the FA and BT2 v5 indexes so their 8-byte chunk-size
20086        // fields are exercised end to end, not just selected.
20087        for (coords, vals) in [
20088            ([0u64, 0], [0i32, 1, 2, 6, 7, 8]),
20089            ([0, 1], [3, 4, 5, 9, 10, 11]),
20090            ([1, 0], [12, 13, 14, 18, 19, 20]),
20091            ([1, 1], [15, 16, 17, 21, 22, 23]),
20092        ] {
20093            let bytes: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
20094            writer
20095                .write_chunk_fixed_array(fa5, &coords, &bytes)
20096                .unwrap();
20097            writer.write_chunk_btree_v2(bt5, &coords, &bytes).unwrap();
20098        }
20099        writer.extend_dataset(bt5, &[4, 6]).unwrap();
20100        writer.close().unwrap();
20101
20102        let mut reader = Hdf5Reader::open(&path).unwrap();
20103        for name in ["fa5", "bt5"] {
20104            let raw = reader.read_dataset_raw(name).unwrap();
20105            let values: Vec<i32> = raw
20106                .chunks(4)
20107                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
20108                .collect();
20109            assert_eq!(values, (0..24).collect::<Vec<i32>>(), "dataset {name}");
20110        }
20111
20112        std::fs::remove_file(&path).ok();
20113    }
20114
20115    /// A v5 file reopened for append must stay v5: the decode → `DatasetInfo`
20116    /// → finalize path carries the version through, so the re-encoded layout
20117    /// message matches the 8-byte size fields the filtered index was built
20118    /// with. A silent v4 downgrade here would make libhdf5 derive a narrower
20119    /// field width than the index uses.
20120    #[cfg(feature = "deflate")]
20121    #[test]
20122    fn v5_layout_survives_reopen_and_append() {
20123        let path = temp_path("libver_v5_reopen");
20124        let chunk: usize = 8;
20125
20126        let mut writer = Hdf5Writer::create(&path).unwrap();
20127        writer.set_libver_latest(true).unwrap();
20128        let idx = writer
20129            .create_chunked_dataset_with_pipeline(
20130                "d",
20131                DatatypeMessage::i32_type(),
20132                &[0],
20133                &[u64::MAX],
20134                &[chunk as u64],
20135                FilterPipeline::deflate(4),
20136            )
20137            .unwrap();
20138        for c in 0..2u64 {
20139            let data: Vec<u8> = (0..chunk as i32)
20140                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
20141                .collect();
20142            writer.write_chunk(idx, c, &data).unwrap();
20143        }
20144        writer.extend_dataset(idx, &[2 * chunk as u64]).unwrap();
20145        writer.close().unwrap();
20146
20147        // Reopen: the decoded layout version must be preserved, and appends
20148        // must keep working against the 8-byte-size-field index.
20149        let writer = Hdf5Writer::open_append(&path).unwrap();
20150        assert_eq!(writer.ds(0).lock().layout_version, 5);
20151        for c in 2..4u64 {
20152            let data: Vec<u8> = (0..chunk as i32)
20153                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
20154                .collect();
20155            writer.write_chunk(0, c, &data).unwrap();
20156        }
20157        writer.extend_dataset(0, &[4 * chunk as u64]).unwrap();
20158        writer.close().unwrap();
20159
20160        // Still v5 after the second finalize, and fully readable.
20161        let writer = Hdf5Writer::open_append(&path).unwrap();
20162        assert_eq!(writer.ds(0).lock().layout_version, 5);
20163        writer.close().unwrap();
20164
20165        let mut reader = Hdf5Reader::open(&path).unwrap();
20166        let raw = reader.read_dataset_raw("d").unwrap();
20167        let values: Vec<i32> = raw
20168            .chunks(4)
20169            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
20170            .collect();
20171        assert_eq!(values, (0..4 * chunk as i32).collect::<Vec<i32>>());
20172
20173        std::fs::remove_file(&path).ok();
20174    }
20175
20176    /// A chunk strictly larger than `u32::MAX` bytes forces layout v5 with no
20177    /// opt-in — v4's size field cannot represent it — while a chunk of exactly
20178    /// `u32::MAX` bytes stays v4, matching libhdf5's `version_req` boundary
20179    /// (`> 0xffffffff`, filtered or not).
20180    #[test]
20181    fn oversized_chunk_forces_v5_without_opt_in() {
20182        let path = temp_path("libver_4gib_force");
20183
20184        let writer = Hdf5Writer::create(&path).unwrap();
20185        let at_limit = writer
20186            .create_chunked_dataset_with_pipeline(
20187                "at_limit",
20188                DatatypeMessage::u8_type(),
20189                &[0],
20190                &[u64::MAX],
20191                &[u32::MAX as u64],
20192                FilterPipeline::deflate(4),
20193            )
20194            .unwrap();
20195        let over = writer
20196            .create_chunked_dataset_with_pipeline(
20197                "over",
20198                DatatypeMessage::u8_type(),
20199                &[0],
20200                &[u64::MAX],
20201                &[u32::MAX as u64 + 1],
20202                FilterPipeline::deflate(4),
20203            )
20204            .unwrap();
20205        let over_unfiltered = writer
20206            .create_chunked_dataset(
20207                "over_plain",
20208                DatatypeMessage::u8_type(),
20209                &[0],
20210                &[u64::MAX],
20211                &[u32::MAX as u64 + 1],
20212            )
20213            .unwrap();
20214
20215        assert_eq!(writer.ds(at_limit).lock().layout_version, 4);
20216        {
20217            let ds = writer.ds(over);
20218            let ds = ds.lock();
20219            assert_eq!(ds.layout_version, 5);
20220            assert_eq!(ds.chunked.as_ref().unwrap().chunk_size_len, 8);
20221        }
20222        assert_eq!(writer.ds(over_unfiltered).lock().layout_version, 5);
20223        writer.close().unwrap();
20224        std::fs::remove_file(&path).ok();
20225    }
20226
20227    /// SWMR reaches version 3 on its own, without a chunked dataset to raise
20228    /// the bound — through the flags `finalize_for_swmr` passes, and then
20229    /// through `swmr_active` for every superblock written after it. Only a
20230    /// file with nothing else newer in it can tell the two arms apart, and
20231    /// the public SWMR API always creates a chunked streaming dataset.
20232    #[test]
20233    fn swmr_reaches_version_3_with_no_chunked_dataset_in_the_file() {
20234        let path = temp_path("swmr_superblock");
20235
20236        let mut writer = Hdf5Writer::create(&path).unwrap();
20237        writer
20238            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
20239            .unwrap();
20240        assert_eq!(writer.superblock_version_for(0), SUPERBLOCK_V2);
20241
20242        writer.finalize_for_swmr().unwrap();
20243        // What `start_swmr` does after finalizing, and what lets a second
20244        // handle read the file while this writer lives — the writer's
20245        // exclusive lock is mandatory on Windows.
20246        writer.handle().release_lock().unwrap();
20247        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
20248
20249        // The close-time finalize carries no SWMR flag; the file is still an
20250        // SWMR file and must not be handed back a version older than the one
20251        // its readers attached to.
20252        writer.close().unwrap();
20253        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
20254        std::fs::remove_file(&path).ok();
20255    }
20256
20257    /// A named bound below `H5F_LIBVER_V110` refuses the session instead —
20258    /// the two checks `H5F__start_swmr_write` opens with, a version-3
20259    /// superblock (H5Fint.c:3814) and a low bound of at least V110
20260    /// (H5Fint.c:3818). Naming no bound at all is what the test above does,
20261    /// and that file is free to become version 3.
20262    #[test]
20263    fn a_named_bound_below_v110_refuses_an_swmr_session() {
20264        for bound in [LibverBound::Earliest, LibverBound::V18] {
20265            let path = temp_path(&format!("swmr_refused_{bound:?}"));
20266            let mut writer = Hdf5Writer::create_with_options(
20267                &path,
20268                FileCreateOptions {
20269                    libver: Some(bound),
20270                    ..Default::default()
20271                },
20272            )
20273            .unwrap();
20274            writer
20275                .create_dataset("d", DatatypeMessage::i32_type(), &[2])
20276                .unwrap();
20277
20278            let err = writer.finalize_for_swmr().unwrap_err().to_string();
20279            assert!(err.contains("SWMR"), "{bound:?}: {err}");
20280            assert!(err.contains("H5F_LIBVER_V110"), "{bound:?}: {err}");
20281
20282            // Refused, not half-done: nothing was published, and the close
20283            // writes the file the bound asked for.
20284            writer.close().unwrap();
20285            let version = std::fs::read(&path).unwrap()[8];
20286            assert_eq!(version, bound.superblock_version(), "{bound:?}");
20287            std::fs::remove_file(&path).ok();
20288        }
20289    }
20290
20291    /// A dataset header the SWMR publish could not fit into the chunk 0 it
20292    /// already had chains into a continuation block, and the in-place rewrite
20293    /// goes back over both: chunk 0 stays at the address the file's readers
20294    /// hold, and the continuation chunk at the one chunk 0 names.
20295    #[test]
20296    fn inplace_rewrite_goes_over_a_chained_header() {
20297        let path = temp_path("inplace_rewrite_chained");
20298        let writer = Hdf5Writer::create_with_options(
20299            &path,
20300            FileCreateOptions {
20301                libver: Some(LibverBound::V110),
20302                ..Default::default()
20303            },
20304        )
20305        .unwrap();
20306        writer
20307            .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
20308            .unwrap();
20309        writer.close().unwrap();
20310
20311        let mut writer = Hdf5Writer::open_append(&path).unwrap();
20312        let idx = 0;
20313        let published = writer.ds(idx).lock().obj_header_written_addr.unwrap();
20314        for i in 0..4 {
20315            writer
20316                .add_dataset_attribute(
20317                    idx,
20318                    AttributeMessage::array_numeric(
20319                        &format!("wide{i}"),
20320                        DatatypeMessage::f64_type(),
20321                        &[32],
20322                        vec![0u8; 256],
20323                    ),
20324                )
20325                .unwrap();
20326        }
20327        writer.finalize_for_swmr().unwrap();
20328        let blocks = writer.ds(idx).lock().obj_header_blocks.clone();
20329        assert_eq!(blocks.len(), 2, "chunk 0 and a continuation: {blocks:?}");
20330        assert_eq!(blocks[0].0, published, "chunk 0 stayed where it was");
20331
20332        writer.write_dataset_header_inplace(idx).unwrap();
20333        assert_eq!(writer.ds(idx).lock().obj_header_blocks, blocks);
20334        writer.close().unwrap();
20335
20336        // The closing finalize wrote over the same chunk 0, and the chained
20337        // header reads back whole.
20338        let writer = Hdf5Writer::open_append(&path).unwrap();
20339        assert_eq!(writer.ds(0).lock().obj_header_written_addr, Some(published));
20340        assert_eq!(writer.ds(0).lock().attributes.len(), 4);
20341        std::fs::remove_file(&path).ok();
20342    }
20343
20344    /// After every writer of a dataset object header, `nlink_written` is the
20345    /// count that writer encoded.
20346    ///
20347    /// `header_stale_with` is the one authority for "does the on-disk header
20348    /// still describe this dataset?", and it reads `nlink_written`; the three
20349    /// writers — `finalize`, `finalize_for_swmr` and
20350    /// `write_dataset_header_inplace` — therefore all record through
20351    /// `DatasetInfo::header_written`. This walks the SWMR sequence, where the
20352    /// in-place writer is the one that could drift, and pins why it does not:
20353    /// a name added after the publish grows the header past the block it was
20354    /// published into, so the rewrite is refused rather than half-applied and
20355    /// the count on disk stays the one the registry names.
20356    #[test]
20357    fn every_dataset_header_write_records_its_link_count() {
20358        let path = temp_path("header_write_records_nlink");
20359        let writer = Hdf5Writer::create(&path).unwrap();
20360        let idx = writer
20361            .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
20362            .unwrap();
20363        let mut writer = writer;
20364        writer.finalize_for_swmr().unwrap();
20365        assert_eq!(
20366            writer.ds(idx).lock().nlink_written,
20367            1,
20368            "the SWMR publish put one name in the header"
20369        );
20370        writer.write_dataset_header_inplace(idx).unwrap();
20371        assert_eq!(writer.ds(idx).lock().nlink_written, 1);
20372
20373        // A second name after the publish: the reference-count message it
20374        // adds does not fit the published block.
20375        writer.create_hard_link("/", "alias", "d").unwrap();
20376        assert_eq!(writer.object_link_count(HardLinkTarget::Dataset(idx)), 2);
20377        let grew = writer
20378            .write_dataset_header_inplace(idx)
20379            .unwrap_err()
20380            .to_string();
20381        assert!(
20382            grew.contains("cannot rewrite in place"),
20383            "a header that outgrew its block must be refused: {grew}"
20384        );
20385        assert_eq!(
20386            writer.ds(idx).lock().nlink_written,
20387            1,
20388            "a refused rewrite leaves the registry describing the header the file holds"
20389        );
20390
20391        // The close-time finalize is the writer that commits the second name,
20392        // and a reopen reads the same count back off the link graph.
20393        writer.close().unwrap();
20394        let writer = Hdf5Writer::open_append(&path).unwrap();
20395        assert_eq!(
20396            writer.ds(0).lock().nlink_written,
20397            2,
20398            "finalize wrote two names and the reopen reads two"
20399        );
20400        writer.close().unwrap();
20401        std::fs::remove_file(&path).ok();
20402    }
20403
20404    /// `H5D__chunk_set_info`'s `version_req` (H5Dchunk.c:909, :936): version 5
20405    /// is required for a chunk over 4 GiB — the version-4 layout message's
20406    /// stored-size field is 32 bits and cannot record one — and
20407    /// `LAYOUT_VERSION_DEFAULT` (3, `H5O_LAYOUT_VERSION_DEFAULT`) is the floor
20408    /// for everything at or under that limit. Pure arithmetic on the byte
20409    /// count: no chunk is ever allocated.
20410    #[test]
20411    fn required_chunk_layout_version_pins_5_past_4_gib() {
20412        assert_eq!(
20413            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64),
20414            LAYOUT_VERSION_DEFAULT
20415        );
20416        assert_eq!(
20417            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64 + 1),
20418            5
20419        );
20420    }
20421
20422    /// `H5D__chunk_set_info`'s index-selection gate (H5Dchunk.c:936): a chunk
20423    /// over 4 GiB reaches the v1.10 chunk indexes even under a bound whose
20424    /// `H5O_layout_ver_bounds` row (`LibverBound::layout_version`) is below
20425    /// 4 — `V18` (row 3) and `Earliest` (row 1) both normally keep an
20426    /// ordinary chunk on the version-1 B-tree, but
20427    /// `required_chunk_layout_version`'s own escape to 5 overrides that row
20428    /// for this one chunk. The default bound (`V110`, row 4) already crosses
20429    /// the threshold on its own, so it is asserted only as the baseline, not
20430    /// as a distinguishing case for the escape.
20431    #[test]
20432    fn uses_v110_chunk_indexing_escapes_past_4_gib_at_every_bound() {
20433        let over_4gib = u32::MAX as u64 + 1;
20434        let small = 1024u64;
20435
20436        let path = temp_path("uses_v110_default");
20437        let writer = Hdf5Writer::create(&path).unwrap();
20438        assert!(writer.uses_v110_chunk_indexing(small));
20439        assert!(writer.uses_v110_chunk_indexing(over_4gib));
20440        writer.close().unwrap();
20441        std::fs::remove_file(&path).ok();
20442
20443        let path = temp_path("uses_v110_v18");
20444        let mut writer = Hdf5Writer::create(&path).unwrap();
20445        writer.set_libver_bound(LibverBound::V18).unwrap();
20446        assert!(
20447            !writer.uses_v110_chunk_indexing(small),
20448            "V18's layout row (3) stays below the v1.10 gate for an ordinary chunk"
20449        );
20450        assert!(
20451            writer.uses_v110_chunk_indexing(over_4gib),
20452            "the >4 GiB escape reaches v1.10 indexing despite V18's row"
20453        );
20454        writer.close().unwrap();
20455        std::fs::remove_file(&path).ok();
20456
20457        let path = temp_path("uses_v110_earliest");
20458        let mut writer = Hdf5Writer::create(&path).unwrap();
20459        writer.set_libver_bound(LibverBound::Earliest).unwrap();
20460        assert!(
20461            !writer.uses_v110_chunk_indexing(small),
20462            "Earliest's layout row (1) stays below the v1.10 gate for an ordinary chunk"
20463        );
20464        assert!(
20465            writer.uses_v110_chunk_indexing(over_4gib),
20466            "the >4 GiB escape reaches v1.10 indexing despite Earliest's row"
20467        );
20468        writer.close().unwrap();
20469        std::fs::remove_file(&path).ok();
20470    }
20471
20472    /// `H5D__chunk_set_info`'s closing `MAX3` (H5Dchunk.c:1046): the same
20473    /// escape pins the layout message itself at version 5 for a chunk over
20474    /// 4 GiB regardless of bound — `required_chunk_layout_version` dominates
20475    /// the max chain ahead of both the bound-derived preference and
20476    /// `LAYOUT_VERSION_DEFAULT`.
20477    #[test]
20478    fn chunk_layout_version_pins_5_past_4_gib_at_every_bound() {
20479        let over_4gib = u32::MAX as u64 + 1;
20480        let small = 1024u64;
20481
20482        let path = temp_path("chunk_ver_default");
20483        let writer = Hdf5Writer::create(&path).unwrap();
20484        assert_eq!(writer.chunk_layout_version(false, small), 4);
20485        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20486        writer.close().unwrap();
20487        std::fs::remove_file(&path).ok();
20488
20489        let path = temp_path("chunk_ver_v18");
20490        let mut writer = Hdf5Writer::create(&path).unwrap();
20491        writer.set_libver_bound(LibverBound::V18).unwrap();
20492        assert_eq!(writer.chunk_layout_version(false, small), 3);
20493        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20494        writer.close().unwrap();
20495        std::fs::remove_file(&path).ok();
20496
20497        let path = temp_path("chunk_ver_earliest");
20498        let mut writer = Hdf5Writer::create(&path).unwrap();
20499        writer.set_libver_bound(LibverBound::Earliest).unwrap();
20500        assert_eq!(
20501            writer.chunk_layout_version(false, small),
20502            LAYOUT_VERSION_DEFAULT
20503        );
20504        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20505        writer.close().unwrap();
20506        std::fs::remove_file(&path).ok();
20507    }
20508    /// `fsm_persist.h5` persists two managers — metadata and raw data. The
20509    /// reopen reads both, hands their merged sections to the allocator, and
20510    /// claims the four blocks the managers themselves occupy.
20511    #[test]
20512    fn a_persisting_file_reopens_with_its_free_sections() {
20513        let path = fixture_copy("fsm_persist.h5", "fsm_read");
20514        let writer = Hdf5Writer::open_append(&path).unwrap();
20515        let fs = writer.free_space.as_deref().expect("managers were read");
20516
20517        assert!(fs.info.persist);
20518        assert_eq!(fs.info.strategy, FileSpaceStrategy::FsmAggr);
20519        assert_eq!(fs.info.threshold, 1);
20520
20521        let sections = writer.allocator.free_blocks();
20522        // h5stat -S reports 1910 bytes of tracked free space for this file.
20523        assert_eq!(sections.iter().map(|s| s.1).sum::<u64>(), 1910);
20524        // Address-ordered, and no two sections touch: what the two managers
20525        // held separately came out coalesced.
20526        for w in sections.windows(2) {
20527            assert!(w[0].0 + w[0].1 < w[1].0, "{sections:?}");
20528        }
20529        // Two headers plus the two sections blocks they name.
20530        assert_eq!(fs.superseded.len(), 4);
20531        for &(addr, len) in &fs.superseded {
20532            assert!(len > 0);
20533            assert!(
20534                !sections
20535                    .iter()
20536                    .any(|&(a, l)| addr < a + l && a < addr + len),
20537                "manager block {addr:#x}+{len} sits in a free section"
20538            );
20539        }
20540        drop(writer);
20541        let _ = std::fs::remove_file(&path);
20542    }
20543
20544    /// A file created with non-default file-space properties carries the
20545    /// message that declares them, and one created to persist gets real
20546    /// managers as soon as anything is freed.
20547    #[test]
20548    fn a_created_file_declares_the_strategy_it_was_made_with() {
20549        let path = temp_path("fsm_create");
20550        {
20551            let w = Hdf5Writer::create_with_options(
20552                &path,
20553                FileCreateOptions {
20554                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
20555                    ..Default::default()
20556                },
20557            )
20558            .unwrap();
20559            let i = w
20560                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20561                .unwrap();
20562            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20563            w.close().unwrap();
20564        }
20565
20566        let info = read_only_append(&path)
20567            .free_space
20568            .as_deref()
20569            .expect("the created file declares a strategy")
20570            .info
20571            .clone();
20572        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
20573        assert!(info.persist);
20574        assert_eq!(info.threshold, 1);
20575        assert_eq!(info.page_size, 4096);
20576        // The alignment fragments the creation left behind are the file's
20577        // first free space, so the metadata manager already has an address
20578        // and the raw-data one, which nothing freed into, does not.
20579        assert_ne!(info.fs_addr[0], UNDEF_ADDR);
20580        assert!(info.fs_addr.iter().skip(1).all(|&a| a == UNDEF_ADDR));
20581
20582        // An append supersedes the root header and the extension, and that
20583        // freed space is what the managers now record.
20584        append_one(&path, "added", false);
20585        assert!(
20586            tracked_free_space(&path) > 0,
20587            "the append recorded no free space"
20588        );
20589        let _ = std::fs::remove_file(&path);
20590    }
20591
20592    /// The two strategies without managers, and the default. All three are
20593    /// `H5Pset_file_space_strategy` settings; only the default leaves the file
20594    /// without the message.
20595    #[test]
20596    fn a_strategy_without_managers_still_declares_itself() {
20597        for (strategy, persist) in [
20598            (FileSpaceStrategy::Aggr, true),
20599            (FileSpaceStrategy::None, false),
20600        ] {
20601            let path = temp_path("fsm_nomgr");
20602            {
20603                let w = Hdf5Writer::create_with_options(
20604                    &path,
20605                    FileCreateOptions {
20606                        file_space: FileSpaceConfig::new(strategy, persist, 7),
20607                        ..Default::default()
20608                    },
20609                )
20610                .unwrap();
20611                w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20612                    .unwrap();
20613                w.close().unwrap();
20614            }
20615            // Read through the reader, not the writer: a reopen only builds
20616            // free-space state for a file it will rewrite managers for, and
20617            // these two have none.
20618            let info = declared_file_space(&path).expect("the strategy is declared");
20619            assert_eq!(info.strategy, strategy);
20620            // `H5P__set_file_space_strategy` stores neither for a strategy
20621            // that has no managers, so both keep the library defaults.
20622            assert!(!info.persist);
20623            assert_eq!(info.threshold, 1);
20624            let _ = std::fs::remove_file(&path);
20625        }
20626    }
20627
20628    /// The library defaults are what a file says by saying nothing.
20629    #[test]
20630    fn the_default_strategy_writes_no_message() {
20631        let path = temp_path("fsm_default");
20632        {
20633            let w = Hdf5Writer::create_with_options(
20634                &path,
20635                FileCreateOptions {
20636                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, false, 1),
20637                    ..Default::default()
20638                },
20639            )
20640            .unwrap();
20641            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20642                .unwrap();
20643            w.close().unwrap();
20644        }
20645        assert!(declared_file_space(&path).is_none());
20646        let _ = std::fs::remove_file(&path);
20647    }
20648
20649    /// The file-space info message a file carries, read back the way any
20650    /// reader sees it.
20651    fn declared_file_space(path: &std::path::Path) -> Option<FileSpaceInfoMessage> {
20652        crate::io::reader::Hdf5Reader::open(path)
20653            .unwrap()
20654            .superblock_extension()
20655            .file_space_info
20656            .clone()
20657    }
20658
20659    /// A created paged file is laid out on its page grid: the superblock takes
20660    /// the whole of page zero and the rest of that page is the metadata
20661    /// manager's first section, which is what `H5MF__alloc_pagefs` gives
20662    /// `H5F__super_init`'s `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`.
20663    #[test]
20664    fn a_created_paged_file_lays_its_pages_out() {
20665        let path = temp_path("fsm_paged_created");
20666        {
20667            let w = Hdf5Writer::create_with_options(
20668                &path,
20669                FileCreateOptions {
20670                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
20671                    ..Default::default()
20672                },
20673            )
20674            .unwrap();
20675            let i = w
20676                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20677                .unwrap();
20678            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20679            w.close().unwrap();
20680        }
20681        let info = read_only_append(&path)
20682            .free_space
20683            .as_deref()
20684            .expect("the created file declares a strategy")
20685            .info
20686            .clone();
20687        assert_eq!(info.strategy, FileSpaceStrategy::Page);
20688        assert!(info.persist);
20689        assert_eq!(info.page_size, 4096);
20690        assert_eq!(
20691            std::fs::metadata(&path).unwrap().len() % info.page_size,
20692            0,
20693            "a paged file ends on a page boundary"
20694        );
20695        let _ = std::fs::remove_file(&path);
20696    }
20697
20698    /// A userblock has to be a whole number of pages, or every page boundary
20699    /// after it is off the file's own grid — `H5F__super_init` refuses one
20700    /// that is not (H5Fsuper.c:1182-1192).
20701    #[test]
20702    fn a_paged_file_refuses_a_userblock_smaller_than_its_page() {
20703        let path = temp_path("fsm_paged_userblock");
20704        let Err(err) = Hdf5Writer::create_with_options(
20705            &path,
20706            FileCreateOptions {
20707                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
20708                userblock: 512,
20709                ..Default::default()
20710            },
20711        ) else {
20712            panic!("a 512-byte userblock was accepted on a 4096-byte page");
20713        };
20714        assert!(
20715            format!("{err}").contains("multiple of its 4096-byte"),
20716            "{err}"
20717        );
20718        let _ = std::fs::remove_file(&path);
20719    }
20720
20721    /// A page size the builder names is the page the file is actually laid
20722    /// out in, not just a number the message repeats: every allocation is
20723    /// shaped by it and the file ends on one of its boundaries.
20724    #[test]
20725    fn a_file_created_at_a_non_default_page_size_allocates_by_it() {
20726        let path = temp_path("fsm_page_size_8k");
20727        {
20728            let w = Hdf5Writer::create_with_options(
20729                &path,
20730                FileCreateOptions {
20731                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20732                        .with_page_size(8192),
20733                    ..Default::default()
20734                },
20735            )
20736            .unwrap();
20737            let i = w
20738                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20739                .unwrap();
20740            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20741            w.close().unwrap();
20742        }
20743        let info = read_only_append(&path)
20744            .free_space
20745            .as_deref()
20746            .expect("the created file declares a strategy")
20747            .info
20748            .clone();
20749        assert_eq!(info.page_size, 8192);
20750        assert_eq!(
20751            std::fs::metadata(&path).unwrap().len() % 8192,
20752            0,
20753            "the file ends on one of the pages it was created with"
20754        );
20755        let _ = std::fs::remove_file(&path);
20756    }
20757
20758    /// The page size is the fourth of the four properties `H5F__super_init`
20759    /// compares against the library defaults (H5Fsuper.c:1092-1097), so
20760    /// naming it is on its own enough to give a file the message — under the
20761    /// default strategy, which allocates without it.
20762    #[test]
20763    fn a_non_default_page_size_alone_gives_the_file_a_message() {
20764        let path = temp_path("fsm_page_size_only");
20765        {
20766            let w = Hdf5Writer::create_with_options(
20767                &path,
20768                FileCreateOptions {
20769                    file_space: FileSpaceConfig::default().with_page_size(1024),
20770                    ..Default::default()
20771                },
20772            )
20773            .unwrap();
20774            w.close().unwrap();
20775        }
20776        let info = declared_file_space(&path)
20777            .expect("a file naming only a page size still carries the message");
20778        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
20779        assert!(!info.persist);
20780        assert_eq!(info.page_size, 1024);
20781        let _ = std::fs::remove_file(&path);
20782    }
20783
20784    /// `H5Pset_file_space_page_size` refuses anything below 512 or above
20785    /// 1 GiB (H5Pfcpl.c:1389-1393), and nothing between: no power of two is
20786    /// required, so a size the bounds admit is one the file may carry.
20787    #[test]
20788    fn a_page_size_outside_the_library_bounds_is_refused() {
20789        for size in [0, 1, 511, PAGE_SIZE_MAX + 1] {
20790            let path = temp_path(&format!("fsm_page_size_bad_{size}"));
20791            let Err(err) = Hdf5Writer::create_with_options(
20792                &path,
20793                FileCreateOptions {
20794                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20795                        .with_page_size(size),
20796                    ..Default::default()
20797                },
20798            ) else {
20799                panic!("a {size}-byte file-space page was accepted");
20800            };
20801            assert!(
20802                format!("{err}").contains("between 512 bytes and 1073741824"),
20803                "{err}"
20804            );
20805            let _ = std::fs::remove_file(&path);
20806        }
20807        let path = temp_path("fsm_page_size_odd");
20808        let w = Hdf5Writer::create_with_options(
20809            &path,
20810            FileCreateOptions {
20811                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20812                    .with_page_size(513),
20813                ..Default::default()
20814            },
20815        )
20816        .expect("513 is inside the bounds, and no power of two is required");
20817        w.close().unwrap();
20818        let _ = std::fs::remove_file(&path);
20819    }
20820
20821    /// A paged file's managers are read on reopen, the same as any other
20822    /// file's: paged aggregation changes which manager a request maps to, not
20823    /// whether the file has managers to rewrite.
20824    #[test]
20825    fn a_paged_file_reports_the_managers_it_persists() {
20826        let path = fixture_copy("fsm_persist_page.h5", "fsm_read_paged");
20827        let writer = Hdf5Writer::open_append(&path).unwrap();
20828        let fs = writer.free_space.as_deref().expect("no managers read");
20829        assert_eq!(fs.info.strategy, FileSpaceStrategy::Page);
20830        assert!(
20831            !writer.allocator.free_extents().is_empty(),
20832            "the sections the file records were not put back in circulation"
20833        );
20834        drop(writer);
20835        let _ = std::fs::remove_file(&path);
20836    }
20837
20838    /// A file with no file-space info message at all — every file this crate
20839    /// creates — has nothing to read and nothing to write back.
20840    #[test]
20841    fn a_file_without_a_strategy_has_no_managers() {
20842        let path = temp_path("fsm_none");
20843        {
20844            let w = Hdf5Writer::create(&path).unwrap();
20845            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20846                .unwrap();
20847            w.close().unwrap();
20848        }
20849        let writer = Hdf5Writer::open_append(&path).unwrap();
20850        assert!(writer.free_space.is_none());
20851        drop(writer);
20852        let _ = std::fs::remove_file(&path);
20853    }
20854    /// Sum of the sections the managers a file names actually hold — what
20855    /// `h5stat -S` prints as "Amount of tracked free space", read back through
20856    /// this crate's own decoder so a test can assert on it. A reopen seeds the
20857    /// allocator with exactly those sections, so its free list is the number.
20858    fn tracked_free_space(path: &std::path::Path) -> u64 {
20859        read_only_append(path)
20860            .allocator
20861            .free_blocks()
20862            .iter()
20863            .map(|b| b.1)
20864            .sum()
20865    }
20866
20867    /// Open for append and mark the writer closed, so dropping it releases the
20868    /// file lock instead of finalizing and rewriting what is being inspected.
20869    fn read_only_append(path: &std::path::Path) -> Hdf5Writer {
20870        let mut w = Hdf5Writer::open_append(path).unwrap();
20871        w.closed = true;
20872        w
20873    }
20874
20875    /// Add one small dataset, the smallest append that still rewrites the root
20876    /// header, the superblock extension and — on a persisting file — the
20877    /// free-space manager.
20878    fn append_one(path: &std::path::Path, name: &str, disable_managers: bool) {
20879        let mut w = Hdf5Writer::open_append(path).unwrap();
20880        if disable_managers {
20881            // Both halves of the change, so the control is the file as this
20882            // crate wrote it before: the session neither allocates from the
20883            // recorded sections nor writes any back.
20884            w.free_space = None;
20885            w.allocator.reset_free_list(&[]);
20886        }
20887        let i = w
20888            .create_dataset(name, DatatypeMessage::i32_type(), &[8])
20889            .unwrap();
20890        w.write_dataset_raw(
20891            i,
20892            &(0..8i32).flat_map(|v| v.to_le_bytes()).collect::<Vec<u8>>(),
20893        )
20894        .unwrap();
20895        w.close().unwrap();
20896    }
20897
20898    /// The block list a reopen carries for the superblock extension covers
20899    /// every chunk of the header, not just the first. The fixture's extension
20900    /// is a two-chunk header — libhdf5 put the file-space info message in a
20901    /// continuation — and freeing chunk zero alone left the continuation
20902    /// allocated with nothing naming it.
20903    #[test]
20904    fn a_reopen_carries_every_chunk_of_the_superblock_extension() {
20905        let path = fixture_copy("fsm_persist.h5", "fsm_ext_chunks");
20906        let blocks = read_only_append(&path).extension.superseded.clone();
20907        assert!(
20908            blocks.len() > 1,
20909            "the fixture's extension is one chunk, so this proves nothing: {blocks:?}"
20910        );
20911        let _ = std::fs::remove_file(&path);
20912    }
20913
20914    /// An append on a persisting file both spends and records the space its
20915    /// managers track: the new dataset comes out of the sections the file
20916    /// already had, and what the rewrite frees goes back into them.
20917    #[test]
20918    fn an_append_reuses_and_records_the_space_the_managers_track() {
20919        let path = fixture_copy("fsm_persist.h5", "fsm_write");
20920        let original = std::fs::metadata(&path).unwrap().len();
20921        let before = tracked_free_space(&path);
20922        assert_eq!(before, 1910, "the fixture's own managers");
20923
20924        append_one(&path, "added", false);
20925        let size = std::fs::metadata(&path).unwrap().len();
20926        let tracked = tracked_free_space(&path);
20927
20928        // Negative control: the same append with both halves of this off — no
20929        // allocating out of the recorded sections and no writing any back —
20930        // which is what this crate did before it read free space at all.
20931        let control = fixture_copy("fsm_persist.h5", "fsm_write_control");
20932        append_one(&control, "added", true);
20933        let control_size = std::fs::metadata(&control).unwrap().len();
20934        assert_eq!(
20935            tracked_free_space(&control),
20936            before,
20937            "with the manager rewrite disabled the number must not move"
20938        );
20939
20940        // The new dataset's raw data comes out of the raw-data sections the
20941        // file already recorded, so the append grows the file by less than the
20942        // same append with the reuse off. It does not stop the growth:
20943        // `H5MF_alloc` asks one manager and no other, and of this fixture's
20944        // 1910 free bytes 1848 are raw-data ones, so the metadata the append
20945        // writes still comes from the end of the file.
20946        assert!(
20947            size < control_size,
20948            "the append took nothing from the {before} bytes free: \
20949             {original} grew to {size}, the control to {control_size}"
20950        );
20951        assert!(
20952            control_size > original,
20953            "the control has to grow or it proves nothing"
20954        );
20955        // Space no manager and no object claims — `h5stat -S`'s "unaccounted
20956        // space" — is what the leak was, and it is smaller now.
20957        assert!(
20958            size - tracked < control_size - before,
20959            "unaccounted space went from {} to {}",
20960            control_size - before,
20961            size - tracked
20962        );
20963
20964        for p in [&path, &control] {
20965            let _ = std::fs::remove_file(p);
20966        }
20967    }
20968
20969    /// The set the writer holds free when it finishes is exactly the set the
20970    /// manager it just wrote records — the invariant that makes the on-disk
20971    /// managers a faithful account of the file's free space.
20972    #[test]
20973    fn the_manager_records_the_free_list_the_close_ends_with() {
20974        let path = fixture_copy("fsm_persist.h5", "fsm_roundtrip");
20975        let internal = {
20976            let mut w = Hdf5Writer::open_append(&path).unwrap();
20977            let i = w
20978                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20979                .unwrap();
20980            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20981            w.finalize(true).unwrap();
20982            let blocks = w.allocator.free_extents();
20983            w.closed = true;
20984            blocks
20985        };
20986        assert!(!internal.is_empty(), "the append freed nothing");
20987
20988        // Classes included: a section read back out of the wrong manager is a
20989        // section libhdf5 would offer to the wrong kind of allocation.
20990        let reread = {
20991            let w = read_only_append(&path);
20992            assert!(w.free_space.is_some(), "managers were written");
20993            w.allocator.free_extents()
20994        };
20995        assert_eq!(internal, reread);
20996        let _ = std::fs::remove_file(&path);
20997    }
20998
20999    /// The paged half of
21000    /// [`the_manager_records_the_free_list_the_close_ends_with`]: a paged
21001    /// file's sections carry a page and a class as well as an address, and a
21002    /// section written into the wrong manager or split across a page boundary
21003    /// would come back different.
21004    #[test]
21005    fn the_manager_records_the_free_list_a_paged_close_ends_with() {
21006        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_roundtrip");
21007        let internal = {
21008            let mut w = Hdf5Writer::open_append(&path).unwrap();
21009            let i = w
21010                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
21011                .unwrap();
21012            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
21013            w.finalize(true).unwrap();
21014            let blocks = w.allocator.free_extents();
21015            w.closed = true;
21016            blocks
21017        };
21018        assert!(!internal.is_empty(), "the append freed nothing");
21019
21020        let reread = {
21021            let w = read_only_append(&path);
21022            assert!(w.free_space.is_some(), "managers were written");
21023            w.allocator.free_extents()
21024        };
21025        assert_eq!(internal, reread);
21026        let _ = std::fs::remove_file(&path);
21027    }
21028
21029    /// Negative control for the paged managers: with the read and the rewrite
21030    /// both off — the file as this crate handled a paged file before — the
21031    /// space the append frees is recorded nowhere, and the number this crate
21032    /// reads back is the fixture's own.
21033    #[test]
21034    fn a_paged_append_records_nothing_without_the_manager_rewrite() {
21035        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_measured");
21036        let control = fixture_copy("fsm_persist_page.h5", "fsm_paged_control");
21037        let before = tracked_free_space(&path);
21038        let original = std::fs::metadata(&path).unwrap().len();
21039
21040        append_one(&path, "added", false);
21041        append_one(&control, "added", true);
21042
21043        assert_eq!(
21044            tracked_free_space(&control),
21045            before,
21046            "the control moved the number it is there to hold still"
21047        );
21048        assert_eq!(
21049            std::fs::metadata(&path).unwrap().len(),
21050            original,
21051            "the append grew a paged file with {before} bytes recorded free"
21052        );
21053        assert!(
21054            std::fs::metadata(&control).unwrap().len() > original,
21055            "the control has to grow or it proves nothing"
21056        );
21057        assert_ne!(
21058            tracked_free_space(&path),
21059            before,
21060            "the managers came back holding what the fixture wrote"
21061        );
21062        for p in [&path, &control] {
21063            let _ = std::fs::remove_file(p);
21064        }
21065    }
21066
21067    /// A block released from a dataset's raw data is recorded by the manager
21068    /// `H5MF_ALLOC_TO_FS_AGGR_TYPE` maps `H5FD_MEM_DRAW` to, and nothing else
21069    /// is: the dichotomy the sec2 driver installs is what decides, and the two
21070    /// managers it collapses to are the file-space info message's slots 0 and
21071    /// 2.
21072    #[test]
21073    fn a_released_raw_block_lands_in_the_raw_data_manager() {
21074        let path = temp_path("fsm_dichotomy");
21075        {
21076            let w = Hdf5Writer::create_with_options(
21077                &path,
21078                FileCreateOptions {
21079                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
21080                    ..Default::default()
21081                },
21082            )
21083            .unwrap();
21084            let i = w
21085                .create_dataset("bulk", DatatypeMessage::i32_type(), &[256])
21086                .unwrap();
21087            w.write_dataset_raw(i, &vec![0u8; 1024]).unwrap();
21088            w.create_dataset("keep", DatatypeMessage::i32_type(), &[8])
21089                .unwrap();
21090            w.close().unwrap();
21091        }
21092        let (raw_addr, raw_len) = {
21093            let w = read_only_append(&path);
21094            let i = w.dataset_index("bulk").unwrap();
21095            let ds = w.ds(i);
21096            let m = ds.lock();
21097            (m.data_addr, m.data_size)
21098        };
21099        assert!(raw_len >= 1024, "the raw block is {raw_len} bytes");
21100        {
21101            let w = Hdf5Writer::open_append(&path).unwrap();
21102            w.delete_dataset("bulk").unwrap();
21103            w.close().unwrap();
21104        }
21105
21106        let mut w = read_only_append(&path);
21107        let info = w
21108            .free_space
21109            .as_deref()
21110            .expect("the file persists managers")
21111            .info
21112            .clone();
21113        assert_ne!(info.fs_addr[0], UNDEF_ADDR, "no metadata manager");
21114        assert_ne!(info.fs_addr[2], UNDEF_ADDR, "no raw-data manager");
21115        for (slot, &addr) in info.fs_addr.iter().enumerate() {
21116            if slot != 0 && slot != 2 {
21117                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
21118            }
21119        }
21120
21121        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21122        let inside = |b: &FreeBlock| b.addr >= raw_addr && b.addr + b.len <= raw_addr + raw_len;
21123        let raw: Vec<&FreeBlock> = found
21124            .sections
21125            .iter()
21126            .filter(|b| b.manager == FreeSpaceManager::RawData)
21127            .collect();
21128        assert!(
21129            !raw.is_empty(),
21130            "the deleted dataset's bytes were not recorded"
21131        );
21132        assert!(
21133            raw.iter().all(|b| inside(b)),
21134            "a raw-data section is outside the deleted dataset's block: {raw:?}"
21135        );
21136        assert!(
21137            found
21138                .sections
21139                .iter()
21140                .filter(|b| b.manager == FreeSpaceManager::Metadata)
21141                .all(|b| !inside(b)),
21142            "raw-data bytes were recorded by the metadata manager"
21143        );
21144        drop(w);
21145        let _ = std::fs::remove_file(&path);
21146    }
21147
21148    /// A reopened paged file's managers are this writer's to rewrite, and the
21149    /// three the sec2 driver can reach are the only ones it names.
21150    ///
21151    /// `H5MF__alloc_to_fs_type` (H5MF.c:265) sends a request of at least one
21152    /// page to `H5F_MEM_PAGE_GENERIC` unless the driver declares
21153    /// `H5FD_FEAT_PAGED_AGGR`, which only the multi and split drivers do, so a
21154    /// sec2 file has the dichotomy's two small managers and that one large
21155    /// one: message slots 0, 2 and 6.
21156    #[test]
21157    fn a_paged_file_names_only_the_managers_sec2_can_reach() {
21158        let path = fixture_copy("fsm_persist_page.h5", "fsm_write_paged");
21159        assert!(
21160            read_only_append(&path).free_space.is_some(),
21161            "the paged fixture's managers were not read"
21162        );
21163        append_one(&path, "added", false);
21164
21165        let mut w = read_only_append(&path);
21166        let info = w
21167            .free_space
21168            .as_deref()
21169            .expect("the file persists managers")
21170            .info
21171            .clone();
21172        assert_eq!(info.strategy, FileSpaceStrategy::Page);
21173        for (slot, &addr) in info.fs_addr.iter().enumerate() {
21174            if !matches!(slot, 0 | 2 | 6) {
21175                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
21176            }
21177        }
21178        assert!(
21179            info.fs_addr.iter().any(|&a| a != UNDEF_ADDR),
21180            "the rewritten file records nothing free"
21181        );
21182        crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21183        drop(w);
21184        let _ = std::fs::remove_file(&path);
21185    }
21186
21187    /// Every section a paged file records sits inside one page, and the pages
21188    /// its small managers use are pages of their own kind — the invariant
21189    /// `H5MF__alloc_pagefs` maintains by giving each small request a whole
21190    /// page of its class and recording the rest of it in that class's manager.
21191    #[test]
21192    fn a_paged_files_small_sections_stay_inside_one_page_of_one_kind() {
21193        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_pages");
21194        append_one(&path, "added", false);
21195
21196        let mut w = read_only_append(&path);
21197        let info = w
21198            .free_space
21199            .as_deref()
21200            .expect("the file persists managers")
21201            .info
21202            .clone();
21203        let page = info.page_size;
21204        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21205        let mut kind_of_page: std::collections::HashMap<u64, FreeSpaceManager> =
21206            std::collections::HashMap::new();
21207        for section in &found.sections {
21208            if section.manager == FreeSpaceManager::Large {
21209                continue;
21210            }
21211            assert_eq!(
21212                section.addr / page,
21213                (section.addr + section.len - 1) / page,
21214                "the section at {:#x} crosses a page boundary",
21215                section.addr
21216            );
21217            let owner = kind_of_page
21218                .entry(section.addr / page)
21219                .or_insert(section.manager);
21220            assert_eq!(
21221                *owner,
21222                section.manager,
21223                "page {} holds sections of two kinds",
21224                section.addr / page
21225            );
21226        }
21227        drop(w);
21228        let _ = std::fs::remove_file(&path);
21229    }
21230}