Skip to main content

rust_hdf5/io/
writer.rs

1//! HDF5 file writer.
2//!
3//! Produces a valid HDF5 file with superblock v3, a root group object header,
4//! and datasets with contiguous or chunked storage. The output is readable by `h5dump`.
5
6use std::collections::{HashMap, HashSet};
7use std::path::{Path, PathBuf};
8
9use crate::dataset::DatasetAccess;
10use crate::format::btree_v1::{BTreeV1Config, ChunkBTreeV1Node, ChunkBTreeV1Tree, ChunkKey};
11use crate::format::chunk_index::btree_v2::Bt2ChunkIndex;
12use crate::format::chunk_index::extensible_array::{
13    compute_chunk_size_len, compute_ndblk_addrs, compute_nsblk_addrs, EaDblkPath, EaGeometry,
14    EaLoc, ExtensibleArrayDataBlock, ExtensibleArrayHeader, ExtensibleArrayIndexBlock,
15    ExtensibleArraySuperBlock, FilteredChunkEntry, FilteredDataBlock, FilteredIndexBlock,
16    EA_CLS_CHUNK, EA_CLS_FILT_CHUNK,
17};
18use crate::format::chunk_index::fixed_array::{
19    decode_filtered_page, decode_unfiltered_page, encode_filtered_page, encode_unfiltered_page,
20    FixedArrayDataBlock, FixedArrayFilteredChunkElement, FixedArrayHeader, FixedArrayPagedPrefix,
21    FA_CLIENT_FILT_CHUNK,
22};
23use crate::format::creation_order::CreationOrder;
24use crate::format::dense_attr::build_dense_attributes;
25use crate::format::dense_link::build_dense_links;
26use crate::format::free_space::{
27    self, FreeSection, FreeSpaceClass, FreeSpaceHeader, FreeSpaceManager,
28};
29use crate::format::local_heap::{
30    local_heap_header_size, LocalHeapHeader, LocalHeapImage, LOCAL_HEAP_FREE_NULL,
31};
32use crate::format::messages::attr_info::{next_creation_index, AttributeInfoMessage};
33use crate::format::messages::attribute::{
34    AttributeEntry, AttributeMessage, ATTR_FLAG_SPACE_SHARED, ATTR_FLAG_TYPE_SHARED,
35};
36use crate::format::messages::data_layout::{
37    DataLayoutMessage, EarrayParams, FixedArrayParams, LAYOUT_VERSION_DEFAULT,
38};
39use crate::format::messages::dataspace::{DataspaceClass, DataspaceMessage};
40use crate::format::messages::datatype::{DatatypeMessage, ReferenceKind};
41use crate::format::messages::external_file_list::{ExternalFileListMessage, UNLIMITED};
42use crate::format::messages::fill_value::{
43    FillValueMessage, FILL_TIME_ALLOC, FILL_TIME_IFSET, FILL_TIME_NEVER,
44};
45use crate::format::messages::filter::{self, FilterPipeline};
46use crate::format::messages::group_info::GroupInfoMessage;
47use crate::format::messages::link::{CharacterSet, LinkMessage, LinkTarget};
48use crate::format::messages::link_info::LinkInfoMessage;
49use crate::format::messages::mod_time::ModificationTime;
50use crate::format::messages::superblock_ext::{
51    FileSpaceInfoMessage, FileSpaceStrategy, SharedMessageTableMessage,
52    DEFAULT_FILE_SPACE_PAGE_SIZE, FS_ADDR_COUNT_V1, PAGE_SIZE_MAX, PAGE_SIZE_MIN,
53};
54use crate::format::messages::virtual_mapping::{
55    parse_source_name, VirtualMapping, VirtualMappingList,
56};
57use crate::format::messages::*;
58use crate::format::object_header::{ObjectHeader, ObjectTimes, MAX_MESSAGE_SIZE};
59use crate::format::reference::{
60    encode_reference_element, encode_revised_blob, ReferenceElementImage, ReferenceTarget,
61    REVISED_BLOB_TOKEN_OFFSET,
62};
63use crate::format::selection::Selection;
64use crate::format::sohm::{
65    type_flag, SharedMessagePointer, MAX_SOHM_INDEXES, SOHM_HEAP_ID_LEN, SOHM_POINTER_HEAP_ID_AT,
66};
67use crate::format::sohm_write::{
68    build_shared_messages, NestedShare, SharedMessage, SohmIndexContent, SohmIndexSpec,
69};
70use crate::format::superblock::*;
71use crate::format::{FormatContext, LibverBound, ObjectFormat, UNDEF_ADDR};
72
73use crate::format::selection::check_hyperslab;
74use crate::io::allocator::{FileAllocator, FreeBlock};
75use crate::io::file_handle::FileHandle;
76use crate::io::hyperslab::{for_each_contiguous_run, for_each_dual_run};
77use crate::io::symbol_table_io::{free_stab, write_stab, Stab, StabExtents, StabLink, StabTarget};
78use crate::io::{FileMeta, IoResult};
79
80/// On-disk size in bytes of a fixed-array data block, for the layout (paged or
81/// flat) implied by `hdr`.
82///
83/// Mirrors `H5FA_DBLOCK_SIZE` (`H5FApkg.h`):
84///   - non-paged: `prefix + nelmts * raw_elmt_size + checksum`
85///   - paged: `prefix + page_init_bitmap + nelmts * raw_elmt_size
86///     + npages * checksum`, where the prefix checksum covers the bitmap.
87///
88/// `raw_elmt_size` is `sizeof_addr` for an unfiltered array, and
89/// `sizeof_addr + chunk_size_len + 4` (the filtered element: address +
90/// compressed size + filter mask) for a filtered array. libhdf5 carries this
91/// value as `hdr->cparam.raw_elmt_size`, i.e. exactly `hdr.element_size`.
92fn fixed_array_dblk_disk_size(ctx: &FormatContext, hdr: &FixedArrayHeader) -> u64 {
93    let elem_size = hdr.element_size as u64;
94    let sa = ctx.sizeof_addr as u64;
95    let nelmts = hdr.num_elmts;
96    // Common metadata prefix: signature(4) + version(1) + client_id(1) + header_addr(sa).
97    let meta_prefix = 4 + 1 + 1 + sa;
98    if hdr.is_paged() {
99        let npages = hdr.npages();
100        let bitmap_size = npages.div_ceil(8);
101        // prefix (incl. its own 4-byte checksum) + elements + per-page checksums.
102        (meta_prefix + bitmap_size + 4) + nelmts * elem_size + npages * 4
103    } else {
104        // prefix + elements + single 4-byte checksum.
105        meta_prefix + nelmts * elem_size + 4
106    }
107}
108
109/// A walk of a v2 B-tree: the file and node geometry the descent reads
110/// through, and the two collections it fills — every node's raw record
111/// bytes and every node block's address, the latter because `open_append`
112/// needs it so the reconstructed [`Bt2DatasetInfo::node_addrs`] pool owns
113/// the on-disk nodes (the next flush re-serializes the tree over them, and
114/// a delete frees them).
115///
116/// `record_size`, `node_size` and `geo` are constant for the whole walk, so
117/// [`descend`](Self::descend) takes only what changes per level: the node's
118/// address, its depth, and how many records it holds.
119struct Bt2Walk<'a> {
120    handle: &'a FileHandle,
121    ctx: &'a FormatContext,
122    record_size: u16,
123    node_size: u32,
124    geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
125    records: Vec<u8>,
126    node_addrs: Vec<u64>,
127}
128
129impl<'a> Bt2Walk<'a> {
130    fn new(
131        handle: &'a FileHandle,
132        ctx: &'a FormatContext,
133        record_size: u16,
134        node_size: u32,
135        geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
136    ) -> Self {
137        Self {
138            handle,
139            ctx,
140            record_size,
141            node_size,
142            geo,
143            records: Vec::new(),
144            node_addrs: Vec::new(),
145        }
146    }
147
148    /// Walk the subtree rooted at `addr`, at depth `depth` with `nrec`
149    /// records, collecting every node's raw record bytes and every node
150    /// block's address.
151    fn descend(&mut self, addr: u64, depth: u16, nrec: u16) -> IoResult<()> {
152        use crate::format::chunk_index::btree_v2::{Bt2InternalNode, Bt2LeafNode};
153
154        self.node_addrs.push(addr);
155        let buf = self.handle.read_at_most(addr, self.node_size as usize)?;
156        if depth == 0 {
157            let leaf = Bt2LeafNode::decode(&buf, nrec, self.record_size)?;
158            self.records.extend_from_slice(&leaf.record_data);
159        } else {
160            let node = Bt2InternalNode::decode(
161                &buf,
162                self.ctx,
163                depth,
164                nrec,
165                self.record_size,
166                self.geo.max_nrec_size,
167                self.geo.child_total_size(depth),
168            )?;
169            // In-order: an internal node's records separate its children, so each
170            // one belongs between the subtrees on either side of it.
171            let children: Vec<(u64, u16)> = node
172                .child_addrs
173                .iter()
174                .zip(node.child_nrecords.iter())
175                .map(|(&a, &n)| (a, n))
176                .collect();
177            let rec = self.record_size as usize;
178            for (i, (child_addr, child_nrec)) in children.into_iter().enumerate() {
179                self.descend(child_addr, depth - 1, child_nrec)?;
180                if let Some(record) = node.record_data.get(i * rec..(i + 1) * rec) {
181                    self.records.extend_from_slice(record);
182                }
183            }
184        }
185        Ok(())
186    }
187}
188
189/// A walk of a version-1 raw-data-chunk B-tree: the file and geometry the
190/// descent reads through, and the two collections it fills.
191///
192/// The v1 counterpart of [`Bt2Walk`], and for the same reason: the
193/// records are what [`BtreeV1DatasetInfo::build_tree`] bulk-loads on the next
194/// flush, and the addresses are the block pool that flush re-serializes over,
195/// so a reopened tree owns the nodes it found instead of leaking them and
196/// allocating a second set beside them.
197///
198/// One value rather than nine parameters threaded through the recursion: only
199/// `addr` and `depth` change between one level and the next, so they are what
200/// [`descend`](Self::descend) takes and everything else lives here.
201struct BtreeV1Walk<'a> {
202    handle: &'a FileHandle,
203    ctx: &'a FormatContext,
204    config: &'a BTreeV1Config,
205    /// The chunk edge lengths, *without* the trailing element-size dimension,
206    /// so `chunk_dims.len()` is the rank the node keys are decoded at.
207    chunk_dims: &'a [u64],
208    file_size: u64,
209    records: Vec<BtreeV1ChunkRecord>,
210    node_addrs: Vec<u64>,
211}
212
213impl<'a> BtreeV1Walk<'a> {
214    fn new(
215        handle: &'a FileHandle,
216        ctx: &'a FormatContext,
217        config: &'a BTreeV1Config,
218        chunk_dims: &'a [u64],
219        file_size: u64,
220    ) -> Self {
221        Self {
222            handle,
223            ctx,
224            config,
225            chunk_dims,
226            file_size,
227            records: Vec::new(),
228            node_addrs: Vec::new(),
229        }
230    }
231
232    /// Walk the subtree rooted at `addr`, collecting every leaf entry as a
233    /// [`BtreeV1ChunkRecord`] and every node block's address.
234    ///
235    /// Records come out in key order because a v1 B-tree's leaves are in key
236    /// order and this descends left to right, which is what
237    /// [`BtreeV1DatasetInfo::position`]'s binary search needs. The keys store
238    /// element offsets (`scaled * chunk_dim`, `H5D__btree_encode_key`), so the
239    /// grid position this records is the quotient.
240    fn descend(&mut self, addr: u64, depth: u32) -> IoResult<()> {
241        // The same bound the reader's walk uses: a node's level is one byte, so
242        // no honest tree is deeper than that, and a cyclic index stops here.
243        if depth > 256 {
244            return Err(crate::io::IoError::InvalidState(
245                "chunk B-tree v1 exceeds maximum depth".into(),
246            ));
247        }
248        if addr == UNDEF_ADDR || addr >= self.file_size {
249            return Ok(());
250        }
251        let rank = self.chunk_dims.len();
252        let sa = self.ctx.sizeof_addr as usize;
253        let node_size = self.config.chunk_btree_node_size(sa, rank);
254        let buf = self.handle.read_at_most(addr, node_size)?;
255        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.config.chunk_max_entries())?;
256        self.node_addrs.push(addr);
257
258        if node.level == 0 {
259            for (i, &child_addr) in node.children.iter().enumerate() {
260                let key = &node.keys[i];
261                let scaled: Vec<u64> = key.offsets[..rank]
262                    .iter()
263                    .zip(self.chunk_dims)
264                    .map(|(&offset, &dim)| offset.checked_div(dim).unwrap_or(0))
265                    .collect();
266                self.records.push(BtreeV1ChunkRecord {
267                    scaled,
268                    address: child_addr,
269                    nbytes: key.chunk_size,
270                    filter_mask: key.filter_mask,
271                });
272            }
273        } else {
274            for &child_addr in &node.children {
275                self.descend(child_addr, depth + 1)?;
276            }
277        }
278        Ok(())
279    }
280}
281
282/// Encode a fixed-array data block for the layout implied by `hdr`, using the
283/// chunk addresses held in `dblk.elements` (unfiltered) or the filtered chunk
284/// entries in `dblk.filtered_elements` (filtered, `client_id == 1`).
285///
286/// For the paged layout (`hdr.is_paged()`), emits the `FADB` prefix with a
287/// page-init bitmap followed by `npages` checksummed element pages. A page is
288/// marked initialized iff at least one of its chunk addresses is defined,
289/// mirroring libhdf5's lazy `H5FA__dblk_page_create`. Uninitialized pages are
290/// still written (all `UNDEF_ADDR`, valid checksum) so the file contains no
291/// uninitialized bytes; the reader skips them via the bitmap.
292fn encode_fixed_array_dblk(
293    ctx: &FormatContext,
294    hdr: &FixedArrayHeader,
295    dblk: &FixedArrayDataBlock,
296) -> Vec<u8> {
297    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
298    let sa = ctx.sizeof_addr as usize;
299    // chunk_size_len for filtered entries = element_size - sizeof_addr - 4.
300    // libhdf5 carries element_size = sizeof_addr + chunk_size_len + 4.
301    let chunk_size_len = (hdr.element_size as usize).saturating_sub(sa + 4);
302
303    if !hdr.is_paged() {
304        return if is_filtered {
305            dblk.encode_filtered(ctx, chunk_size_len)
306        } else {
307            dblk.encode_unfiltered(ctx)
308        };
309    }
310
311    let npages = hdr.npages() as usize;
312    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
313
314    // Build the page-init bitmap (MSB-first): a page is initialized iff any of
315    // its elements points at a defined address.
316    let mut bitmap = vec![0u8; npages.div_ceil(8)];
317    let nelmts = if is_filtered {
318        dblk.filtered_elements.len()
319    } else {
320        dblk.elements.len()
321    };
322    for p in 0..npages {
323        let start = p * dblk_page_nelmts;
324        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
325        let initialized = if is_filtered {
326            dblk.filtered_elements[start..end]
327                .iter()
328                .any(|e| e.address != UNDEF_ADDR)
329        } else {
330            dblk.elements[start..end].iter().any(|&a| a != UNDEF_ADDR)
331        };
332        if initialized {
333            bitmap[p / 8] |= 0x80u8 >> (p % 8);
334        }
335    }
336
337    let prefix = FixedArrayPagedPrefix {
338        client_id: hdr.client_id,
339        header_addr: dblk.header_addr,
340        page_init_bitmap: bitmap,
341        prefix_size: 4 + 1 + 1 + sa + npages.div_ceil(8) + 4,
342    };
343
344    let mut buf = prefix.encode(ctx);
345    debug_assert_eq!(buf.len(), prefix.prefix_size);
346
347    // Append each page: all pages use the full `dblk_page_nelmts` stride;
348    // only the last page holds fewer elements (libhdf5 H5FA.c).
349    for p in 0..npages {
350        let start = p * dblk_page_nelmts;
351        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
352        if is_filtered {
353            buf.extend_from_slice(&encode_filtered_page(
354                &dblk.filtered_elements[start..end],
355                ctx,
356                chunk_size_len,
357            ));
358        } else {
359            buf.extend_from_slice(&encode_unfiltered_page(&dblk.elements[start..end], ctx));
360        }
361    }
362    buf
363}
364
365/// Decode a fixed-array data block for the layout implied by `hdr` — the
366/// inverse of [`encode_fixed_array_dblk`], and the single decode dispatch
367/// over non-paged/paged × unfiltered/filtered.
368///
369/// For the paged layout, pages whose bitmap bit is clear are skipped, not
370/// decoded: libhdf5 never writes an uninitialized page, so its bytes are
371/// arbitrary and carry no valid checksum. Their elements stay at the
372/// undefined-address defaults, which is exactly what the bitmap means.
373fn decode_fixed_array_dblk(
374    ctx: &FormatContext,
375    hdr: &FixedArrayHeader,
376    buf: &[u8],
377    chunk_size_len: usize,
378) -> crate::format::FormatResult<FixedArrayDataBlock> {
379    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
380    let num_elmts = hdr.num_elmts as usize;
381
382    if !hdr.is_paged() {
383        return if is_filtered {
384            FixedArrayDataBlock::decode_filtered(buf, ctx, num_elmts, chunk_size_len)
385        } else {
386            FixedArrayDataBlock::decode_unfiltered(buf, ctx, num_elmts)
387        };
388    }
389
390    let npages = hdr.npages() as usize;
391    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
392    let prefix = FixedArrayPagedPrefix::decode(buf, ctx, npages as u64)?;
393
394    let mut dblk = if is_filtered {
395        FixedArrayDataBlock::new_filtered(prefix.header_addr, num_elmts)
396    } else {
397        FixedArrayDataBlock::new_unfiltered(prefix.header_addr, num_elmts)
398    };
399    dblk.client_id = hdr.client_id;
400
401    // Pages follow the prefix back to back; every page spans the full
402    // `dblk_page_nelmts` stride except the last, which holds the remainder.
403    let mut pos = prefix.prefix_size;
404    for p in 0..npages {
405        let start = p * dblk_page_nelmts;
406        let end = ((p + 1) * dblk_page_nelmts).min(num_elmts);
407        let nelmts = end - start;
408        if prefix.page_initialized(p) {
409            let page_buf = buf.get(pos..).unwrap_or(&[]);
410            if is_filtered {
411                let elems = decode_filtered_page(page_buf, ctx, nelmts, chunk_size_len)?;
412                dblk.filtered_elements[start..end].clone_from_slice(&elems);
413            } else {
414                let addrs = decode_unfiltered_page(page_buf, ctx, nelmts)?;
415                dblk.elements[start..end].copy_from_slice(&addrs);
416            }
417        }
418        pos += nelmts * hdr.element_size as usize + 4;
419    }
420    Ok(dblk)
421}
422
423/// Interior-mutability cell for per-dataset write state, selected by feature.
424///
425/// This is the §5-B "cfg-selected interior types" from
426/// `docs/threadsafe-fine-grained-locking.md`: the single-threaded build uses a
427/// `RefCell` (zero overhead, no atomics), while the `threadsafe` build uses a
428/// `Mutex` so two threads can write *different* datasets concurrently while the
429/// same dataset's writes serialize. Call sites are identical across both via
430/// [`Slot::lock`].
431#[cfg(not(feature = "threadsafe"))]
432pub(crate) struct Slot<T>(std::cell::RefCell<T>);
433
434#[cfg(not(feature = "threadsafe"))]
435impl<T> Slot<T> {
436    pub(crate) fn new(value: T) -> Self {
437        Slot(std::cell::RefCell::new(value))
438    }
439    /// Borrow the contents mutably (an uncontended `RefCell` borrow).
440    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, T> {
441        self.0.borrow_mut()
442    }
443}
444
445#[cfg(feature = "threadsafe")]
446pub(crate) struct Slot<T>(std::sync::Mutex<T>);
447
448#[cfg(feature = "threadsafe")]
449impl<T> Slot<T> {
450    pub(crate) fn new(value: T) -> Self {
451        Slot(std::sync::Mutex::new(value))
452    }
453    /// Lock the contents. Different datasets hold different slots, so this
454    /// only contends when two threads write the *same* dataset.
455    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, T> {
456        self.0.lock().unwrap()
457    }
458}
459
460/// Proof that the create gate (`create_lock`) is held and the new dataset's
461/// name passed the uniqueness check. Only [`Hdf5Writer::begin_create`]
462/// constructs one and [`Hdf5Writer::push_dataset`] demands one, so a creator
463/// cannot reach the dataset registry while skipping either step. Carries
464/// the canonical (link-resolved) name the creator must store, so the
465/// registry only ever holds tree paths.
466pub(crate) struct CreateGuard<'a> {
467    #[cfg(not(feature = "threadsafe"))]
468    _gate: std::cell::RefMut<'a, ()>,
469    #[cfg(feature = "threadsafe")]
470    _gate: std::sync::MutexGuard<'a, ()>,
471    /// The dataset name with every group hard link in it resolved.
472    pub(crate) name: String,
473    /// The group that will hold the new dataset's link, resolved from the
474    /// path components of `name`; `None` is the root group. Carried here so
475    /// [`Hdf5Writer::push_dataset`] registers the child itself and no creator
476    /// can leave a dataset whose name says one thing and whose parent group
477    /// says another.
478    pub(crate) parent: Option<usize>,
479}
480
481/// Reference-counted shared pointer, feature-selected. The single-thread
482/// build uses `Rc` (no atomics); the `threadsafe` build uses `Arc` so a
483/// dataset/group slot can be cloned out of the registry and locked on its
484/// own — letting writes to *different* datasets proceed concurrently without
485/// holding the registry lock. See `docs/threadsafe-fine-grained-locking.md`
486/// (Stage 3).
487#[cfg(not(feature = "threadsafe"))]
488pub(crate) type Shared<T> = std::rc::Rc<T>;
489#[cfg(feature = "threadsafe")]
490pub(crate) type Shared<T> = std::sync::Arc<T>;
491
492/// One dataset's cell in the registry: its metadata slot plus the operation
493/// lock that serializes whole logical operations on it. Both live in one
494/// allocation so they cannot fall out of step — every dataset has its op
495/// lock by construction.
496pub(crate) struct DatasetCell {
497    /// Serializes one *whole* logical operation on this dataset.
498    ///
499    /// The metadata slot below serializes each individual acquisition, but a
500    /// multi-acquisition operation — take the append buffer → write chunks →
501    /// re-buffer the tail → extend, or flush-then-overwrite in a slice write
502    /// — would interleave with a concurrent same-dataset operation *between*
503    /// its acquisitions under `threadsafe`. Public write entries take this
504    /// lock and delegate to `_inner` variants; `_inner` variants and the
505    /// `pub(crate)` write helpers require the caller to hold it (or to hold
506    /// the writer exclusively via `&mut`, as close and the SWMR wrapper do).
507    ///
508    /// Not reentrant: the single-thread build's `RefCell` panics instantly
509    /// on a nested acquisition, so a missed entry/inner split fails loudly
510    /// in every test run rather than deadlocking only under `threadsafe`.
511    ///
512    /// Lock order: `create_lock → op → registry spine → metadata slot`. An
513    /// op lock is never held across another dataset's op lock, and no
514    /// op-lock holder takes `create_lock`, so the order is acyclic.
515    pub(crate) op: Slot<()>,
516    info: Slot<DatasetInfo>,
517}
518
519impl DatasetCell {
520    pub(crate) fn new(info: DatasetInfo) -> Self {
521        DatasetCell {
522            op: Slot::new(()),
523            info: Slot::new(info),
524        }
525    }
526
527    /// Borrow the metadata slot (a single acquisition; see [`Self::op`] for
528    /// whole-operation serialization).
529    #[cfg(not(feature = "threadsafe"))]
530    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, DatasetInfo> {
531        self.info.lock()
532    }
533
534    /// Lock the metadata slot (a single acquisition; see [`Self::op`] for
535    /// whole-operation serialization).
536    #[cfg(feature = "threadsafe")]
537    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, DatasetInfo> {
538        self.info.lock()
539    }
540}
541
542/// A single dataset's [`DatasetCell`], reference-counted so a writer can
543/// clone it out of the registry (releasing the registry lock) and then lock
544/// just this one dataset. Two threads writing different datasets take
545/// different `DatasetRef` locks and never contend; the same dataset's writes
546/// serialize, which is required because one chunk index is not concurrently
547/// mutable.
548pub(crate) type DatasetRef = Shared<DatasetCell>;
549
550/// A single group's metadata behind its own [`Slot`], reference-counted like
551/// [`DatasetRef`].
552pub(crate) type GroupRef = Shared<Slot<GroupInfo>>;
553
554/// Appended frames held back until they complete a chunk.
555///
556/// The buffer is the sole authority for rows `base .. base + frames`: the
557/// file's chunks do not hold them yet, and any operation that writes those
558/// rows must go through [`Hdf5Writer::flush_append_buffer`] first. `base` is
559/// recorded when the frames are buffered — never derived from the current
560/// extent, which an `extend_dataset` can move independently.
561pub struct AppendBuffer {
562    /// Absolute row of the first buffered frame.
563    pub base: u64,
564    /// Number of buffered frames.
565    pub frames: u64,
566    /// The frames' bytes, `frames` whole rows, row-major.
567    pub bytes: Vec<u8>,
568}
569
570/// One file a dataset's raw data lives in, as the writer holds it: the name
571/// the I/O path opens, together with the local-heap offset the External File
572/// List message stores that name as.
573///
574/// The two halves are one entry rather than two parallel lists because they
575/// describe one slot — the message encodes `name_offset`, and every read or
576/// write of the slot's bytes opens `name`; splitting them is what lets a
577/// rewrite pair a name with another slot's offset.
578#[derive(Debug, Clone, PartialEq, Eq)]
579pub struct ExternalFile {
580    /// The file name exactly as the heap stores it. Resolved against
581    /// `HDF5_EXTFILE_PREFIX` at I/O time, never here — the same rule the read
582    /// side follows.
583    pub name: String,
584    /// Where `name` sits in the local heap at [`ExternalStorage::heap_addr`].
585    pub name_offset: u64,
586    /// Byte offset within `name` where this slot's region begins.
587    pub offset: u64,
588    /// Bytes of the dataset's raw data this slot holds.
589    pub size: u64,
590}
591
592/// A dataset whose contiguous raw data lives outside this file — the External
593/// File List message (`H5O_EFL_ID`) and the local heap its names are in.
594///
595/// The data layout message of such a dataset still says `Contiguous`, with
596/// its address left undefined: it is this message's presence that makes
597/// libhdf5 route the dataset's I/O through `H5D_LOPS_EFL` (H5Dlayout.c).
598#[derive(Debug, Clone)]
599pub struct ExternalStorage {
600    /// Address of the local heap header holding every slot's name.
601    pub heap_addr: u64,
602    /// The files, in the order their regions concatenate into the dataset's
603    /// logical byte range.
604    pub files: Vec<ExternalFile>,
605    /// The prefix every one of those names is joined against, and the open
606    /// that settled it. Lives here rather than on [`DatasetInfo`] so a
607    /// dataset with no external storage cannot carry a prefix and a dataset
608    /// with external storage cannot lack one.
609    prefix: EfilePrefix,
610}
611
612/// The expanded external file prefix in force for one dataset, and the open
613/// that decided it — libhdf5's `dset->shared->extfile_prefix`.
614///
615/// `H5D__build_file_prefix` runs it once per open of the shared info, from
616/// the dapl of `H5D__create` (H5Dint.c:1318) or of the `H5D__open` that
617/// found no shared info yet (:1537), and both `H5D__efl_read` and
618/// `H5D__efl_write` then join against that one answer (H5Defl.c:315-317,
619/// :429-431). Measured under libhdf5 1.14.6 and 2.0.0: `H5Dcreate2` with a
620/// dapl naming a directory creates the raw data file there at `H5Dwrite`,
621/// and `HDF5_EXTFILE_PREFIX` shadows that property on the write path exactly
622/// as it does on the read path.
623#[derive(Debug, Clone, Default)]
624struct EfilePrefix {
625    /// The expansion itself; `None` is "no prefix", which leaves a stored
626    /// name to resolve against the process's current directory.
627    expanded: Option<PathBuf>,
628    /// The open that decided [`expanded`](Self::expanded). An expired handle
629    /// means no open is holding the answer any more, so the next one settles
630    /// it afresh — which is the state a dataset this session reopened starts
631    /// in, `H5Fopen` opening no dataset of its own.
632    open: std::sync::Weak<()>,
633}
634
635impl ExternalStorage {
636    /// The message this storage encodes to (`H5O_efl_t`).
637    fn message(&self) -> ExternalFileListMessage {
638        ExternalFileListMessage {
639            heap_addr: self.heap_addr,
640            slots: self
641                .files
642                .iter()
643                .map(
644                    |f| crate::format::messages::external_file_list::ExternalFileSlot {
645                        name_offset: f.name_offset,
646                        offset: f.offset,
647                        size: f.size,
648                    },
649                )
650                .collect(),
651        }
652    }
653
654    /// Bytes the slots reserve in total (`H5O_efl_total_size`), saturating
655    /// rather than wrapping so an overflowing list reads as "as large as it
656    /// gets" and passes any size check instead of failing one.
657    fn total_size(&self) -> u64 {
658        self.files
659            .iter()
660            .fold(0u64, |acc, f| acc.saturating_add(f.size))
661    }
662}
663
664/// A dataset whose elements are read out of other datasets — the virtual
665/// layout message (`H5D_VIRTUAL`) and the mapping list it points at.
666///
667/// The mappings live in one global heap object rather than in the header
668/// (`H5D__virtual_store_layout`), so the layout message carries only its
669/// address and index; the list itself is kept here so a rewrite of the header
670/// can re-emit the message pointing at the same object.
671#[derive(Debug, Clone, PartialEq, Eq)]
672pub struct VirtualStorage {
673    /// Address of the global heap collection holding the mapping list.
674    pub heap_addr: u64,
675    /// Index of the mapping-list object within that collection.
676    pub heap_index: u32,
677    /// The mappings themselves, in the order they were declared — which is
678    /// the order libhdf5 resolves overlapping ones in.
679    pub mappings: Vec<VirtualMapping>,
680}
681
682/// Where a contiguous dataset's raw bytes live, read off its registry entry
683/// so the write itself can run with the slot unlocked.
684///
685/// The one place the local-versus-external-versus-nowhere choice is made; see
686/// [`DatasetInfo::contiguous_target`].
687enum ContiguousTarget {
688    /// A block in this file, starting at this address.
689    Local(u64),
690    /// The files an External File List names, in dataset order, and the
691    /// prefix in force for the open doing the writing — carried together
692    /// because a slot name means nothing without it.
693    External {
694        files: Vec<ExternalFile>,
695        prefix: Option<PathBuf>,
696    },
697    /// Nowhere: the dataset is virtual, and every element of it is stored in
698    /// whichever source dataset its mappings send that element to.
699    Virtual,
700}
701
702/// What a writer-mode `H5Dataset` handle is built from — the shape and
703/// element width it answers questions with, the chunk index it writes
704/// through, and the open it holds.
705pub(crate) struct DatasetHandleParts {
706    pub(crate) shape: Vec<usize>,
707    pub(crate) element_size: usize,
708    /// `None` for storage that is not chunked.
709    pub(crate) chunk_index: Option<ChunkIndexKind>,
710    /// Keeps this open alive; see [`Hdf5Writer::bind_efile_prefix`].
711    pub(crate) open: Option<crate::io::reader::DatasetOpenToken>,
712}
713
714impl ContiguousTarget {
715    /// Whether this target is storage bytes can be written into at all —
716    /// false only for [`ContiguousTarget::Virtual`], which names sources
717    /// rather than storage.
718    fn is_storage(&self) -> bool {
719        !matches!(self, Self::Virtual)
720    }
721}
722
723/// The one refusal of a write into a virtual dataset, so the two paths that
724/// can reach one — [`Hdf5Writer::write_contiguous_bytes`] and the pre-insert
725/// gate of [`Hdf5Writer::write_vlen_strings_slice`] — say the same thing.
726///
727/// libhdf5 does take this write, pushing each element through the mapping
728/// that covers it into the source dataset holding it (`H5D__virtual_write`);
729/// this writer never opens a source file, so it refuses rather than dropping
730/// the bytes somewhere they cannot be read back from.
731/// The legality checks `H5Pset_virtual` runs over one mapping —
732/// `H5D_virtual_check_mapping_pre` and `H5D_virtual_check_mapping_post`
733/// (H5Dvirtual.c).
734///
735/// The two upstream checks that need the *source dataset's* own extent (the
736/// limited/limited element-count match, and a printf mapping's single-block
737/// match) are not run here for the same reason upstream skips them when the
738/// source space status is `H5O_VIRTUAL_STATUS_INVALID`: a mapping may name a
739/// source that does not exist yet, and nothing here opens one.
740fn check_virtual_mapping(dataset: &str, m: &VirtualMapping) -> IoResult<()> {
741    for (which, sel) in [
742        ("virtual", &m.virtual_selection),
743        ("source", &m.source_selection),
744    ] {
745        if matches!(sel, Selection::Points(_)) {
746            return Err(crate::io::IoError::Unsupported(format!(
747                "virtual dataset '{dataset}' has a point {which} selection, which \
748                 H5D_virtual_check_mapping_pre refuses for every virtual dataset mapping \
749                 (\"point selections not currently supported with virtual datasets\")"
750            )));
751        }
752    }
753
754    let unlim_virtual = m.virtual_selection.unlim_dim().is_some();
755    let unlim_source = m.source_selection.unlim_dim().is_some();
756
757    // Both sides unbounded: the mapping grows with its source, so the slices
758    // they exchange must be the same shape whatever either extent becomes.
759    if unlim_virtual && unlim_source {
760        if let (Some(v), Some(sr)) = (
761            regular_hyperslab(&m.virtual_selection),
762            regular_hyperslab(&m.source_selection),
763        ) {
764            let (nv, ns) = (v.num_elem_non_unlim(), sr.num_elem_non_unlim());
765            if nv != ns {
766                return Err(crate::io::IoError::InvalidState(format!(
767                    "virtual dataset '{dataset}' maps an unlimited source selection onto an \
768                     unlimited virtual selection, but a slice of the non-unlimited \
769                     dimensions holds {ns:?} source elements and {nv:?} virtual ones"
770                )));
771            }
772        }
773    }
774
775    // `H5D_virtual_check_mapping_post`: an unlimited virtual selection over a
776    // limited source selection is the printf shape, where each block of the
777    // virtual selection is filled by a *different* source dataset named by
778    // substituting that block's index. It needs a `%b` to name them, and a
779    // hyperslab virtual selection to have blocks at all; every other shape
780    // needs the opposite, since a substitution with only one block to fill
781    // has nothing to vary over.
782    let nsubs = parse_source_name(&m.source_file_name)
783        .and_then(|f| Ok(f.nsubs() + parse_source_name(&m.source_dset_name)?.nsubs()))
784        .map_err(|e| {
785            crate::io::IoError::InvalidState(format!(
786                "virtual dataset '{dataset}' source name: {e}"
787            ))
788        })?;
789    if unlim_virtual && !unlim_source {
790        if nsubs == 0 {
791            return Err(crate::io::IoError::InvalidState(format!(
792                "virtual dataset '{dataset}' has an unlimited virtual selection, a limited \
793                 source selection, and no printf specifiers in source names"
794            )));
795        }
796        if !matches!(m.virtual_selection, Selection::Hyperslab { .. }) {
797            return Err(crate::io::IoError::InvalidState(format!(
798                "virtual dataset '{dataset}' has a printf mapping whose virtual selection is \
799                 not a hyperslab; the substitution runs over the blocks of that hyperslab"
800            )));
801        }
802    } else if nsubs > 0 {
803        return Err(crate::io::IoError::InvalidState(format!(
804            "virtual dataset '{dataset}' has printf specifier(s) in source name(s) without \
805             an unlimited virtual selection and limited source selection"
806        )));
807    }
808    Ok(())
809}
810
811/// The regular (start, stride, count, block) form behind a selection, or
812/// `None` — the only form that can carry `H5S_UNLIMITED`, so every unlimited
813/// check goes through it.
814fn regular_hyperslab(sel: &Selection) -> Option<&crate::format::selection::RegularHyperslab> {
815    match sel {
816        Selection::Hyperslab {
817            form: crate::format::selection::Hyperslab::Regular(r),
818            ..
819        } => Some(r),
820        _ => None,
821    }
822}
823
824fn virtual_write_refused() -> crate::io::IoError {
825    crate::io::IoError::Unsupported(
826        "cannot write into a virtual dataset: its elements live in the source datasets \
827         its mappings name, and this writer does not write through to them — write the \
828         source datasets themselves"
829            .into(),
830    )
831}
832
833/// Metadata for a dataset being written.
834///
835/// The whole struct lives behind a per-dataset [`Slot`] (via [`DatasetRef`]).
836/// The streaming write path locks it only briefly — compression runs *outside*
837/// the lock — so writes to different datasets do not contend, and a structural
838/// op (create/delete) that scans names only momentarily touches a sibling
839/// slot.
840pub struct DatasetInfo {
841    /// Link name within the root group.
842    pub name: String,
843    /// Element datatype.
844    pub datatype: DatatypeMessage,
845    /// The committed datatype this dataset shares, when it was created from
846    /// one. The type itself stays in [`datatype`](Self::datatype) — the
847    /// dataspace, the element width and every payload check need it — and
848    /// this says the header must store a pointer to that object instead of a
849    /// datatype message of its own.
850    pub committed_type: Option<CommittedTypeRef>,
851    /// Dataspace (dimensionality).
852    pub dataspace: DataspaceMessage,
853    /// The object format the reopen found this dataset's messages written in,
854    /// `None` for a dataset this session created.
855    ///
856    /// A rewrite re-encodes the whole header — the shared-message table is
857    /// laid out whole, so every heap ID moves and every header naming one has
858    /// to be written again. Re-deriving the message format from the reopened
859    /// session's bounds would upgrade messages the file already has, which
860    /// libhdf5 never does: it grows a header in place and leaves every
861    /// message it did not touch alone. The same rule the reopen already
862    /// applies to a group it found in a symbol table
863    /// ([`uses_symbol_table`](Hdf5Writer::uses_symbol_table)) — what the file
864    /// says governs, not what this session's bound would have chosen.
865    pub read_format: Option<ObjectFormat>,
866    /// File offset of the dataset's object header (set during finalize).
867    pub obj_header_addr: u64,
868    /// File offset of the raw data block (contiguous only).
869    pub data_addr: u64,
870    /// Size of the raw data in bytes (contiguous only).
871    pub data_size: u64,
872    /// The raw data itself, for a compact dataset — the whole image, which
873    /// [`build_dataset_header`](Hdf5Writer::build_dataset_header) puts inside
874    /// the data layout message rather than in a block of its own. `Some` is
875    /// what makes a dataset compact, and the buffer is created at its final
876    /// length (filled, as `H5D__compact_fill` does, before any write), so it
877    /// is also the dataset's byte count; `data_addr`/`data_size` stay at the
878    /// "no block in the file" values a compact dataset shares with a NULL one.
879    pub compact: Option<Vec<u8>>,
880    /// The files this dataset's contiguous raw data lives in, when it lives
881    /// outside this HDF5 file. `Some` is what makes a contiguous dataset
882    /// externally stored: its `data_addr` stays [`UNDEF_ADDR`] and every byte
883    /// goes to the files named here instead of to a block of this file's own.
884    pub external: Option<ExternalStorage>,
885    /// The source datasets this dataset's elements are read from, when it is
886    /// virtual. `Some` is what makes it virtual, and it stores nothing of its
887    /// own: `data_addr`/`data_size` keep the "no block in this file" values a
888    /// compact dataset also has.
889    pub virtual_storage: Option<VirtualStorage>,
890    /// Chunked storage info (None for contiguous).
891    pub chunked: Option<ChunkedDatasetInfo>,
892    /// Fixed array chunked storage info.
893    pub fixed_array: Option<FixedArrayDatasetInfo>,
894    /// B-tree v2 chunked storage info.
895    pub btree_v2: Option<Bt2DatasetInfo>,
896    /// Implicit (no structure) chunked storage info.
897    pub implicit: Option<ImplicitDatasetInfo>,
898    /// Single-chunk chunked storage info: the whole (fixed) dataspace is
899    /// exactly one chunk.
900    pub single_chunk: Option<SingleChunkDatasetInfo>,
901    /// Version-1 B-tree chunked storage info — the classic-format index.
902    pub btree_v1: Option<BtreeV1DatasetInfo>,
903    /// Appended frames not yet written to chunks, `None` when empty.
904    pub append: Option<AppendBuffer>,
905    /// Attributes attached to this dataset.
906    pub attributes: Vec<AttributeEntry>,
907    /// File offset where the dataset object header was written (for SWMR in-place rewrites).
908    pub obj_header_written_addr: Option<u64>,
909    /// Encoded size of the dataset object header (for verifying in-place rewrites fit).
910    /// Every block the object's on-disk header occupies, chunk 0 first, or
911    /// empty when it has none yet. All of them are freed together: a rewrite
912    /// re-encodes the whole chain into one fresh chunk, so a continuation
913    /// block left behind is space no free-space manager records.
914    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
915    /// Filter pipeline for compressed chunks.
916    pub filter_pipeline: Option<FilterPipeline>,
917    /// Soft-deleted: excluded from finalize output.
918    pub deleted: bool,
919    /// The dataspace extent changed this session (`extend_dataset` /
920    /// `set_dataset_extent`). On a reopened dataset the finalize gate
921    /// otherwise infers "modified" from `chunks_written` alone, and a
922    /// session that only changed the extent would keep the old on-disk
923    /// header — silently dropping the new shape.
924    pub extent_dirty: bool,
925    /// Something the object header encodes changed this session without
926    /// touching the dataset's storage — an attribute set or removed, a fill
927    /// value defined. See [`header_stale`](DatasetInfo::header_stale).
928    pub header_dirty: bool,
929    /// The hard link count the on-disk header was written with, so finalize
930    /// can tell that this session changed it.
931    ///
932    /// A count, not a flag, because the count is what the header records and
933    /// the ways to change it are many: creating a link, unlinking one,
934    /// deleting a link's parent group, promoting a link to a primary name.
935    /// Comparing the value closes all of them at once, where a dirty flag
936    /// would have to be set at each and would be forgotten at the next one
937    /// added.
938    pub nlink_written: u32,
939    /// When the link naming this dataset was created; see
940    /// [`GroupInfo::creation_seq`].
941    pub creation_seq: u64,
942    /// How this dataset records creation order for its attributes — the
943    /// file's creation-order policy captured when the dataset was created,
944    /// the way libhdf5 captures the DCPL. A dataset holds no links, so only
945    /// the attribute half of [`TrackOrder`] applies to it.
946    pub track_attr_order: CreationOrder,
947    /// User-defined fill value bytes (exactly one element wide). `None`
948    /// means default zero-fill; `Some` is emitted as a `fill_defined = 2`
949    /// fill-value message in the dataset object header.
950    pub fill_value: Option<Vec<u8>>,
951    /// Fill value write time (`H5Pset_fill_time`'s `H5D_fill_time_t`, one of
952    /// [`FILL_TIME_ALLOC`], [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]),
953    /// emitted verbatim into the fill-value message's write-time field.
954    /// Defaults to `FILL_TIME_IFSET`, `H5D_CRT_FILL_TIME_DEF` — what a fresh
955    /// dataset creation property list carries until `set_dataset_fill_time`
956    /// says otherwise.
957    pub fill_time: u8,
958    /// Layout message version for chunked storage: 4, or 5 when the chunk
959    /// index encodes stored chunk sizes in a fixed `sizeof_size` field
960    /// (libhdf5 2.0). Chosen at create by `Hdf5Writer::chunk_layout_version`,
961    /// preserved from the file on reopen, and emitted verbatim at finalize.
962    /// Contiguous datasets ignore it.
963    pub layout_version: u8,
964    /// The times this object tracks: `Some` exactly when it was created with
965    /// `H5Pset_obj_track_times(true)`, `None` when it was not.
966    ///
967    /// One meaning on both header versions, which store them differently and
968    /// store different amounts of them: a version-2 header keeps all four in
969    /// its prefix, and a version-1 dataset keeps one, in an `H5O_MTIME_NEW`
970    /// message. [`touch_oh`] is the single place that turns this into either
971    /// of those, so the four fields are here whichever version the object
972    /// has, exactly as `H5O_t` carries `atime`/`mtime`/`ctime`/`btime` for a
973    /// version-1 header it never serialises them from.
974    pub times: Option<ObjectTimes>,
975}
976
977impl DatasetInfo {
978    /// Which chunk index this dataset uses, `None` for storage that is not
979    /// chunked — the one place the index-carrying fields are turned into an
980    /// answer.
981    ///
982    /// INVARIANT: a chunk index added to this struct is added here. A site
983    /// that spells the disjunction out itself is what classifies a new index
984    /// as contiguous storage, and contiguous storage is read and written at
985    /// [`data_addr`](Self::data_addr) — which a chunked dataset leaves
986    /// undefined, so the misclassification is a read or a write at
987    /// `UNDEF_ADDR` rather than an error.
988    pub(crate) fn chunk_index_kind(&self) -> Option<ChunkIndexKind> {
989        if self.chunked.is_some() {
990            Some(ChunkIndexKind::ExtensibleArray)
991        } else if self.fixed_array.is_some() {
992            Some(ChunkIndexKind::FixedArray)
993        } else if self.btree_v2.is_some() {
994            Some(ChunkIndexKind::BtreeV2)
995        } else if self.implicit.is_some() {
996            Some(ChunkIndexKind::Implicit)
997        } else if self.single_chunk.is_some() {
998            Some(ChunkIndexKind::SingleChunk)
999        } else if self.btree_v1.is_some() {
1000            Some(ChunkIndexKind::BtreeV1)
1001        } else {
1002            None
1003        }
1004    }
1005
1006    /// Whether this dataset's raw data is stored in chunks — the question
1007    /// every storage-form test asks, asked in one place.
1008    pub(crate) fn is_chunked(&self) -> bool {
1009        self.chunk_index_kind().is_some()
1010    }
1011
1012    /// Where this dataset's contiguous raw bytes live, or `None` when it has
1013    /// no contiguous storage to write into at all — a chunked dataset, a
1014    /// compact one (whose bytes *are* the layout message), or one whose block
1015    /// was never allocated.
1016    ///
1017    /// INVARIANT: every write of a contiguous dataset's raw bytes picks its
1018    /// destination here and reaches it through
1019    /// [`Hdf5Writer::write_contiguous_bytes`]. A site that read `data_addr`
1020    /// itself would write an externally-stored dataset's data into this file
1021    /// — at [`UNDEF_ADDR`], the far end of the address space — instead of into
1022    /// the files its header names, and would do the same to a virtual one,
1023    /// whose bytes are not this file's to write at all.
1024    ///
1025    /// Chunked storage is excluded through
1026    /// [`chunk_index_kind`](Self::chunk_index_kind) rather than by naming the
1027    /// index-carrying fields, so an index added to this struct cannot arrive
1028    /// here as contiguous storage: an implicit-indexed dataset reads
1029    /// `data_addr` as the base of its chunk grid, which as a contiguous
1030    /// destination would take a raw write meant for one chunk and lay it over
1031    /// the whole grid.
1032    fn contiguous_target(&self) -> Option<ContiguousTarget> {
1033        if self.is_chunked() || self.compact.is_some() {
1034            return None;
1035        }
1036        if self.virtual_storage.is_some() {
1037            return Some(ContiguousTarget::Virtual);
1038        }
1039        match &self.external {
1040            Some(ext) => Some(ContiguousTarget::External {
1041                files: ext.files.clone(),
1042                prefix: ext.prefix.expanded.clone(),
1043            }),
1044            None => {
1045                (self.data_addr != UNDEF_ADDR).then_some(ContiguousTarget::Local(self.data_addr))
1046            }
1047        }
1048    }
1049
1050    /// The one run of file bytes an implicitly indexed dataset's chunk grid
1051    /// is — its start and its length — or `None` when the dataset is indexed
1052    /// some other way or its space is not allocated yet.
1053    ///
1054    /// That index has no per-chunk structure to hold an address in: every
1055    /// chunk sits at `data_addr + linear_index * chunk_bytes` and the grid is
1056    /// allocated whole at create (`H5D__none_idx_get_addr`, H5Dnone.c). So the
1057    /// run is file space this writer allocated, and it is the *only* storage a
1058    /// chunk of such a dataset can occupy — the builder refuses external and
1059    /// virtual storage together with chunked storage, which is why
1060    /// [`allocated_storage_run`](Self::allocated_storage_run) can name it
1061    /// [`ContiguousTarget::Local`] and no chunk write can reach the other two.
1062    fn implicit_grid(&self) -> Option<(u64, u64)> {
1063        let imp = self.implicit.as_ref()?;
1064        (imp.data_addr != UNDEF_ADDR).then_some((imp.data_addr, imp.data_size))
1065    }
1066
1067    /// The run of raw storage this writer *allocated* for the dataset — the
1068    /// target to initialise it through and its size — or `None` when it
1069    /// allocated none.
1070    ///
1071    /// The two storage forms that are one run of bytes: a contiguous
1072    /// dataset's data block, and an implicitly indexed dataset's chunk grid.
1073    /// A compact dataset is excluded (its bytes are its layout message) and so
1074    /// is every other chunk index, whose chunks are placed one at a time.
1075    ///
1076    /// External storage is excluded because this writer does not allocate it:
1077    /// `H5D__alloc_storage` skips its whole body — the space reservation and
1078    /// the `H5D__init_storage` that would tile the fill value into it — for a
1079    /// dataset with an external file list or an empty extent, "we assume that
1080    /// external storage is already allocated by the caller, or at least will
1081    /// be before I/O is performed" (H5Dint.c:2270-2274). Measured under
1082    /// libhdf5 1.14.6 and 2.0.0: a user fill value, `H5D_FILL_TIME_ALLOC` and
1083    /// `H5D_ALLOC_TIME_EARLY` together leave the raw data file uncreated at
1084    /// `H5Dcreate2`, and a read before any write fails with "unable to open
1085    /// external raw data file" rather than reporting the fill.
1086    ///
1087    /// INVARIANT: only storage whose bytes this file owns is initialised as
1088    /// one run, so the allocate-time fill cannot reach the files an external
1089    /// file list names or the sources a virtual dataset maps.
1090    fn allocated_storage_run(&self) -> Option<(ContiguousTarget, u64)> {
1091        match self.implicit_grid() {
1092            Some((addr, size)) => Some((ContiguousTarget::Local(addr), size)),
1093            // Not a fallthrough for an unallocated implicit grid:
1094            // `contiguous_target` answers `None` for every chunked dataset.
1095            None => match self.contiguous_target() {
1096                Some(t @ ContiguousTarget::Local(_)) => Some((t, self.data_size)),
1097                _ => None,
1098            },
1099        }
1100    }
1101
1102    /// Whether this session wrote chunk data or changed the extent, so the
1103    /// dataset's index structures have to be re-flushed.
1104    fn storage_dirty(&self) -> bool {
1105        self.chunked.as_ref().is_some_and(|c| c.chunks_written > 0)
1106            || self
1107                .fixed_array
1108                .as_ref()
1109                .is_some_and(|f| f.chunks_written > 0)
1110            || self.btree_v2.as_ref().is_some_and(|b| b.chunks_written > 0)
1111            || self.btree_v1.as_ref().is_some_and(|b| b.chunks_written > 0)
1112            || self
1113                .single_chunk
1114                .as_ref()
1115                .is_some_and(|s| s.chunks_written > 0)
1116            || self.extent_dirty
1117    }
1118
1119    /// Whether a reopened dataset's on-disk object header no longer describes
1120    /// it.
1121    ///
1122    /// INVARIANT: every mutation of something `build_dataset_header` encodes
1123    /// must show up here. Finalize keeps the original header when this is
1124    /// false, so a change this misses is not deferred — it is discarded, with
1125    /// no error to say so. Attributes were the case that proved it: they are
1126    /// invisible to the chunk-write counters, so an attribute set on a
1127    /// reopened dataset vanished at close.
1128    fn header_stale(&self) -> bool {
1129        self.storage_dirty() || self.header_dirty
1130    }
1131
1132    /// The same question for the one thing the dataset itself cannot see: how
1133    /// many hard links resolve to it. That count lives in the header — an
1134    /// Object Reference Count message in a version-2 header, the `nlink`
1135    /// prefix field of a version-1 one — but it is a property of the file's
1136    /// link graph, so the caller supplies today's value.
1137    fn header_stale_with(&self, nlink: u32) -> bool {
1138        self.header_stale() || nlink != self.nlink_written
1139    }
1140
1141    /// Record that this dataset's on-disk object header was just written with
1142    /// `nlink` in it.
1143    ///
1144    /// INVARIANT: every write of a dataset object header passes through here.
1145    /// [`header_stale_with`](Self::header_stale_with) is the one authority for
1146    /// "does what is on disk still describe this dataset?", and it answers by
1147    /// comparing against [`nlink_written`](Self::nlink_written) — so a site
1148    /// that writes a header without saying so leaves that answer describing an
1149    /// older write. There are three writers: `finalize`, `finalize_for_swmr`
1150    /// and `write_dataset_header_inplace`. The last recorded nothing; it could
1151    /// not drift today only because a count it could write is a count that
1152    /// makes the header outgrow its block, which it refuses. That is a
1153    /// property of the reference-count message's size, not a rule anything
1154    /// states, and it is not what the field's definition rests on.
1155    fn header_written(&mut self, nlink: u32) {
1156        self.nlink_written = nlink;
1157    }
1158}
1159
1160/// Runtime metadata for a chunked dataset.
1161pub struct ChunkedDatasetInfo {
1162    /// Chunk dimension sizes.
1163    pub chunk_dims: Vec<u64>,
1164    /// Extensible array parameters.
1165    pub earray_params: EarrayParams,
1166    /// File offset of the EA header.
1167    pub ea_header_addr: u64,
1168    /// File offset of the EA index block.
1169    pub ea_iblk_addr: u64,
1170    /// In-memory copy of the EA header (for updating statistics).
1171    pub ea_header: ExtensibleArrayHeader,
1172    /// In-memory copy of the EA index block (for unfiltered datasets).
1173    pub ea_iblk: ExtensibleArrayIndexBlock,
1174    /// Number of chunks written so far.
1175    pub chunks_written: u64,
1176    /// Filtered index block (for compressed datasets).
1177    pub filt_iblk: Option<FilteredIndexBlock>,
1178    /// chunk_size_len for filtered entries.
1179    pub chunk_size_len: u8,
1180}
1181
1182/// Where a newly-created EA data block's address must be recorded.
1183enum DblkParent {
1184    /// Slot `index_block.dblk_addrs[idx]`.
1185    IndexBlock(usize),
1186    /// Slot `super_block.dblk_addrs[local_dblk]` of the super block at `sblk_addr`.
1187    SuperBlock {
1188        sblk_addr: u64,
1189        ndblks_in_sblk: usize,
1190        local_dblk: usize,
1191    },
1192}
1193
1194/// Which attribute list an attribute operation targets: the root group's,
1195/// a group's (by full path), or a dataset's (by writer index).
1196#[derive(Clone, Copy)]
1197pub enum AttrTarget<'a> {
1198    /// The root group's (file-level) attributes.
1199    Root,
1200    /// A group's attributes, by full path.
1201    Group(&'a str),
1202    /// A dataset's attributes, by writer index.
1203    Dataset(usize),
1204}
1205
1206/// Which chunk index a dataset uses.
1207///
1208/// The five above the line are what `H5D__layout_set_latest_indexing`
1209/// (H5Dlayout.c) picks between once the file format allows a version-4 data
1210/// layout message, in this precedence: a v2 B-tree for two or more unlimited
1211/// dimensions, an extensible array for exactly one, and — for a fixed shape —
1212/// the single-chunk index whenever exactly one chunk covers the whole
1213/// dataspace (`dims == max_dims == chunk_dims`, checked before either
1214/// alternative below and taken regardless of filter or allocation-time), else
1215/// the implicit index when nothing has to be recorded per chunk (no filter,
1216/// early allocation), else a fixed array. [`BtreeV1`](Self::BtreeV1) is not
1217/// one of them: it belongs to the version-3 layout message, and a file whose
1218/// superblock is older than version 2 can carry no other.
1219#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1220pub(crate) enum ChunkIndexKind {
1221    ExtensibleArray,
1222    FixedArray,
1223    BtreeV2,
1224    Implicit,
1225    SingleChunk,
1226    BtreeV1,
1227}
1228
1229/// A chunked dataset's grid geometry, snapshotted out of its slot.
1230///
1231/// The single owner of chunk-grid arithmetic: how many chunks span each
1232/// dimension, where a coordinate sits in the row-major order the array
1233/// indices record, and how many bytes one chunk holds.
1234struct ChunkGeometry {
1235    kind: ChunkIndexKind,
1236    dims: Vec<u64>,
1237    max_dims: Option<Vec<u64>>,
1238    chunk_dims: Vec<u64>,
1239    element_size: u64,
1240}
1241
1242impl ChunkGeometry {
1243    /// Unfiltered byte size of one whole chunk.
1244    fn chunk_bytes(&self) -> u64 {
1245        self.chunk_dims.iter().product::<u64>() * self.element_size
1246    }
1247
1248    /// Row-major position of `coords` in the chunk grid — the linear index an
1249    /// extensible or fixed array records the chunk under, computed against
1250    /// the maximum-extent grid by [`crate::io::chunk_grid::linear_index`].
1251    fn linear_index(&self, coords: &[u64]) -> IoResult<u64> {
1252        crate::io::chunk_grid::linear_index(
1253            &self.dims,
1254            self.max_dims.as_deref(),
1255            &self.chunk_dims,
1256            coords,
1257        )
1258    }
1259}
1260
1261/// The refusal every attribute mutation gets while SWMR streaming is
1262/// active, from the two owners of attribute-list change
1263/// ([`Hdf5Writer::set_attribute`] and `evict_attr`).
1264fn swmr_attr_error(name: &str) -> crate::io::IoError {
1265    crate::io::IoError::InvalidState(format!(
1266        "cannot add or modify attribute '{name}' during SWMR streaming: object \
1267         headers are frozen while readers stream, and a superseded variable-length \
1268         value's heap storage could never be reclaimed; set attributes before \
1269         start_swmr (libhdf5 forbids attribute changes during SWMR writes too)"
1270    ))
1271}
1272
1273/// Where an attribute arriving at [`Hdf5Writer::insert_attribute`] came from.
1274///
1275/// The variable-length setters have to evict before they allocate — the
1276/// free-before-alloc order — so by the time the replacement is inserted the
1277/// list no longer holds the entry it replaces, and the ordinary "already
1278/// present, so keep its index" test cannot see it. `H5A__attr_write` does not
1279/// create the attribute again, so the index travels with the eviction rather
1280/// than being stamped afresh; without it a rewritten attribute takes the set's
1281/// running maximum and moves to the end of the creation order.
1282#[derive(Debug, Clone, Copy)]
1283enum AttrOrigin {
1284    /// A new attribute, which takes the set's next creation index.
1285    Created,
1286    /// A value written over an attribute this writer has just evicted, which
1287    /// keeps that attribute's creation index — `None` when the object tracks
1288    /// no order, and so records none. An eviction that found nothing to remove
1289    /// answers `Created`: what follows it is a create like any other.
1290    Rewritten(Option<u16>),
1291}
1292use AttrOrigin::{Created, Rewritten};
1293
1294/// Take an object's attributes into the append session, or refuse the reopen.
1295///
1296/// Append mode rebuilds every object header it touches out of the attributes
1297/// read from it, so what this returns is what the object will still have when
1298/// the session finalizes. An attribute set that could not be read whole —
1299/// `ObjectAttributes::into_complete` refuses it — would come back as the part
1300/// that did read, silently deleting the rest.
1301///
1302/// Left to surface at `finalize`, that failure would land after this session's
1303/// chunk data and indices had already been written past the allocation point
1304/// the superblock still records, leaving a file libhdf5 reads as truncated.
1305/// Refusing the open leaves it untouched.
1306///
1307/// Size is no longer a reason to refuse: an attribute too large for a header
1308/// message goes back out through dense storage, the form libhdf5 read it from.
1309///
1310/// The set comes back in creation-index order, which is the order the registry
1311/// holds attributes in for an object made in this session too. A dense set is
1312/// read through the name index, so the order it arrives in is the order a hash
1313/// walk took; sorting here is what makes "the list is in creation order" true
1314/// of a reopened object as well, without any later stage having to know which
1315/// storage form the attributes came out of. Attributes of an untracked object
1316/// carry no index and keep the order they were read in.
1317fn take_reopened_attributes(
1318    attrs: crate::io::reader::ObjectAttributes,
1319    owner: &str,
1320) -> IoResult<Vec<AttributeEntry>> {
1321    let mut attrs = attrs.into_complete(owner)?;
1322    attrs.sort_by_key(|a| a.creation_index());
1323    Ok(attrs)
1324}
1325
1326/// The creation-order policy an on-disk object header declares — the single
1327/// owner of the recovery rule, used for the root group, every reopened group
1328/// and (through its attribute half) every reopened dataset.
1329///
1330/// The two halves come from two different places, and reading one for both is
1331/// how a file that sets only one of them came back with both or neither:
1332///
1333///   * links — the `Link Info` message's flag bits, which is what
1334///     `H5Pget_link_creation_order` reads (`H5G__get_create_plist`). A group
1335///     with no such message (or one this crate cannot decode) tracks nothing;
1336///     so does every dataset, which has no links to order.
1337///   * attributes — the object header's own flag bits, which is what
1338///     `H5Pget_attr_creation_order` reads (`H5Pocpl.c`). The `Attribute Info`
1339///     message carries the same two bits, but the header is the authority
1340///     libhdf5 consults, and it is present even when the object has no
1341///     attributes yet.
1342fn recover_track_order(
1343    header: &crate::format::object_header::ObjectHeader,
1344    ctx: &FormatContext,
1345) -> TrackOrder {
1346    let links = header
1347        .messages
1348        .iter()
1349        .find(|m| m.msg_type == crate::format::messages::MSG_LINK_INFO)
1350        .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
1351        .map(|(info, _)| info.creation_order())
1352        .unwrap_or_default();
1353    TrackOrder {
1354        links,
1355        attrs: header.attribute_creation_order(),
1356    }
1357}
1358
1359/// `H5O_touch_oh` (H5Oint.c:1273): put an object's tracked times where its
1360/// header version keeps them.
1361///
1362/// INVARIANT: every object header this writer builds passes its times through
1363/// here. The version decides the storage and nothing else does — a caller that
1364/// set `ObjectHeader::times` itself would hand a version-1 encode a prefix
1365/// field that version has no room for, and one that added the message itself
1366/// would put a second copy in a version-2 header.
1367///
1368/// `force` is upstream's own parameter, and it is what splits datasets from
1369/// everything else: it creates the version-1 `H5O_MTIME_NEW` message when the
1370/// header has none, and only `H5D__update_oh_info` passes it true
1371/// (H5Dint.c:1022-1026). Every other caller passes false and so creates no
1372/// message at all, which is why a version-1 group or committed datatype
1373/// records no time even when it is tracking them. A version-2 header keeps all
1374/// four times in its prefix whatever `force` says.
1375fn touch_oh(
1376    header: &mut ObjectHeader,
1377    format: ObjectFormat,
1378    times: Option<ObjectTimes>,
1379    force: bool,
1380) {
1381    let Some(times) = touched_times(times) else {
1382        return;
1383    };
1384    match format {
1385        ObjectFormat::Modern => header.times = Some(times),
1386        ObjectFormat::Legacy if force => header.add_message(
1387            crate::format::messages::MSG_MOD_TIME,
1388            0x00,
1389            ModificationTime(times.change).encode(),
1390        ),
1391        ObjectFormat::Legacy => {}
1392    }
1393}
1394
1395/// The times a header being (re)written carries, given what the object had.
1396///
1397/// Every object header this writer emits is one it is writing *now*, which is
1398/// what `H5O_touch_oh` is called for: an object that stores times gets its
1399/// access and change time moved to now, and one that does not store them stays
1400/// that way — the flag belongs to the object's creation property list, and a
1401/// rewrite is not a creation.
1402fn touched_times(times: Option<ObjectTimes>) -> Option<ObjectTimes> {
1403    times.map(|t| t.touched(now_seconds()))
1404}
1405
1406/// Seconds since the epoch, as an object header stores them (`H5_now`).
1407///
1408/// Saturates rather than wrapping: the field is a 32-bit count, and a clock
1409/// past 2106 is better reported as the largest time the format can express
1410/// than as a time in 1970. A clock before the epoch yields 0, which is what
1411/// libhdf5 writes for "no time recorded".
1412fn now_seconds() -> u32 {
1413    std::time::SystemTime::now()
1414        .duration_since(std::time::UNIX_EPOCH)
1415        .map_or(0, |d| u32::try_from(d.as_secs()).unwrap_or(u32::MAX))
1416}
1417
1418/// The dense storage an on-disk object header names: the fractal heap and the
1419/// indices its `Attribute Info` and `Link Info` messages point at.
1420///
1421/// A rewrite of that header lays fresh storage out and stops naming this, so
1422/// what this returns is exactly what the rewrite supersedes and must free.
1423/// Compact storage names no heap and yields `None` — there is nothing to free
1424/// and nothing that could be freed twice.
1425fn superseded_dense(
1426    header: &crate::format::object_header::ObjectHeader,
1427    ctx: &FormatContext,
1428) -> (Option<AttributeInfoMessage>, Option<LinkInfoMessage>) {
1429    let decode = |msg_type: u8| {
1430        header
1431            .messages
1432            .iter()
1433            .find(|m| m.msg_type == msg_type)
1434            .map(|m| m.data.as_slice())
1435    };
1436    let attrs = decode(crate::format::messages::MSG_ATTR_INFO)
1437        .and_then(|d| AttributeInfoMessage::decode(d, ctx).ok())
1438        .map(|(info, _)| info)
1439        .filter(|info| info.is_dense());
1440    let links = decode(crate::format::messages::MSG_LINK_INFO)
1441        .and_then(|d| LinkInfoMessage::decode(d, ctx).ok())
1442        .map(|(info, _)| info)
1443        .filter(|info| info.is_dense());
1444    (attrs, links)
1445}
1446
1447/// One collection block with free space that a later vlen insert may
1448/// fill — an entry in the writer's CWFS list (libhdf5 `f->shared->cwfs`).
1449struct CwfsEntry {
1450    /// Block address of the collection.
1451    addr: u64,
1452    /// Declared block size; never changes after allocation.
1453    size: usize,
1454    /// Bytes its free-space marker owns, per
1455    /// [`GlobalHeapCollection::free_space_at`](crate::format::global_heap::GlobalHeapCollection::free_space_at).
1456    free: usize,
1457}
1458
1459/// Maximum CWFS entries tracked — libhdf5's `H5HG_NCWFS` (H5HGpkg.h).
1460const H5HG_NCWFS: usize = 16;
1461
1462/// Record a collection with `free` bytes in the CWFS list: update its
1463/// entry if present, append while the list is short, and otherwise
1464/// replace the entry with the least free space when this one has more —
1465/// the retention rule of libhdf5's `H5HG_insert`.
1466fn cwfs_note(cwfs: &mut Vec<CwfsEntry>, addr: u64, size: usize, free: usize) {
1467    if let Some(p) = cwfs.iter().position(|e| e.addr == addr) {
1468        cwfs[p].free = free;
1469        return;
1470    }
1471    if cwfs.len() < H5HG_NCWFS {
1472        cwfs.insert(0, CwfsEntry { addr, size, free });
1473        return;
1474    }
1475    if let Some(p) = (0..cwfs.len()).min_by_key(|&p| cwfs[p].free) {
1476        if free > cwfs[p].free {
1477            cwfs[p] = CwfsEntry { addr, size, free };
1478        }
1479    }
1480}
1481
1482/// The uniform rejection for `delete_dataset` / `delete_group` while SWMR
1483/// streaming is active: deleting frees the object's blocks, and a live
1484/// reader may hold any of their addresses.
1485fn swmr_delete_error(name: &str) -> crate::io::IoError {
1486    crate::io::IoError::InvalidState(format!(
1487        "cannot delete '{name}' during SWMR streaming: a reader may hold the \
1488         object's header and storage addresses (libhdf5 forbids link deletion \
1489         during SWMR writes too)"
1490    ))
1491}
1492
1493/// Whether the chunk at grid `coords` lies entirely at or beyond `extent` in
1494/// some dimension — no element of it would survive a shrink to that extent.
1495fn chunk_outside_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1496    coords
1497        .iter()
1498        .zip(chunk_dims)
1499        .zip(extent)
1500        .any(|((&c, &cd), &e)| c.saturating_mul(cd) >= e)
1501}
1502
1503/// Whether the chunk at grid `coords` keeps elements under `extent` but
1504/// extends past it in some dimension — a shrink must refill its
1505/// out-of-extent region with the fill value.
1506fn chunk_straddles_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1507    !chunk_outside_extent(coords, chunk_dims, extent)
1508        && coords
1509            .iter()
1510            .zip(chunk_dims)
1511            .zip(extent)
1512            .any(|((&c, &cd), &e)| (c + 1).saturating_mul(cd) > e)
1513}
1514
1515/// Overwrite, in `data` (one whole chunk, unfiltered, row-major), every
1516/// element at or beyond `extent` with the matching bytes of `fill` — a
1517/// same-sized buffer tiled with the fill value. The caller guarantees the
1518/// chunk at `coords` straddles `extent`, so every dimension keeps at least
1519/// one element. Returns the replaced bytes, so a vlen dataset's dead
1520/// heap references can be released rather than stranded.
1521fn refill_chunk_beyond_extent(
1522    data: &mut [u8],
1523    fill: &[u8],
1524    coords: &[u64],
1525    chunk_dims: &[u64],
1526    extent: &[u64],
1527    element_size: usize,
1528) -> Vec<u8> {
1529    let ndims = chunk_dims.len();
1530    let keep: Vec<usize> = (0..ndims)
1531        .map(|d| {
1532            let origin = coords[d] * chunk_dims[d];
1533            chunk_dims[d].min(extent[d].saturating_sub(origin)) as usize
1534        })
1535        .collect();
1536    // Row-major walk: for every row (all dimensions but the last),
1537    // overwrite the whole row when its prefix is outside the keep box,
1538    // else only the row's out-of-extent tail.
1539    let row_elems = chunk_dims[ndims - 1] as usize;
1540    let keep_last = keep[ndims - 1];
1541    let nrows: u64 = chunk_dims[..ndims - 1].iter().product();
1542    let mut replaced = Vec::new();
1543    for r in 0..nrows {
1544        let mut rem = r;
1545        let mut in_keep = true;
1546        for d in (0..ndims - 1).rev() {
1547            let c = rem % chunk_dims[d];
1548            rem /= chunk_dims[d];
1549            if c as usize >= keep[d] {
1550                in_keep = false;
1551            }
1552        }
1553        let start = if in_keep { keep_last } else { 0 };
1554        if start == row_elems {
1555            continue;
1556        }
1557        let a = (r as usize * row_elems + start) * element_size;
1558        let b = (r as usize + 1) * row_elems * element_size;
1559        replaced.extend_from_slice(&data[a..b]);
1560        data[a..b].copy_from_slice(&fill[a..b]);
1561    }
1562    replaced
1563}
1564
1565/// Validate caller-supplied chunk geometry at dataset definition, the rule
1566/// libhdf5 applies in `H5D__chunk_construct` (H5Dchunk.c): the chunk rank
1567/// must match the dataspace rank, no chunk dimension may be zero, and a
1568/// chunk dimension may not exceed a fixed maximum dimension — except in a
1569/// dimension whose current size is zero, which libhdf5 exempts.
1570fn validate_chunk_geometry(dims: &[u64], max_dims: &[u64], chunk_dims: &[u64]) -> IoResult<()> {
1571    let ndims = dims.len();
1572    if chunk_dims.len() != ndims {
1573        return Err(crate::io::IoError::InvalidState(format!(
1574            "chunk shape has {} dimensions but the dataspace has {}",
1575            chunk_dims.len(),
1576            ndims
1577        )));
1578    }
1579    if max_dims.len() != ndims {
1580        return Err(crate::io::IoError::InvalidState(format!(
1581            "maximum shape has {} dimensions but the dataspace has {}",
1582            max_dims.len(),
1583            ndims
1584        )));
1585    }
1586    for d in 0..ndims {
1587        if chunk_dims[d] == 0 {
1588            return Err(crate::io::IoError::InvalidState(format!(
1589                "chunk dimension {d} is zero"
1590            )));
1591        }
1592        if dims[d] != 0 && max_dims[d] != u64::MAX && max_dims[d] < chunk_dims[d] {
1593            return Err(crate::io::IoError::InvalidState(format!(
1594                "chunk dimension {} is {} but the maximum dimension size is {}",
1595                d, chunk_dims[d], max_dims[d]
1596            )));
1597        }
1598    }
1599    Ok(())
1600}
1601
1602/// An extensible-array index requires at most one unlimited dimension —
1603/// `H5D__chunk_construct` (H5Dchunk.c) only selects this index for exactly
1604/// one — at any position: `chunk_grid::linear_index` seeds the unlimited
1605/// dimension into the slot no down-chunks multiplier touches, the same
1606/// address libhdf5 reaches by swizzling it to the slowest position
1607/// (`H5VM_swizzle_coords`, H5Dearray.c). Two or more unlimited dimensions
1608/// have no finite grid at all; that shape needs a v2 B-tree index instead.
1609fn ensure_at_most_one_unlimited(max_dims: &[u64]) -> IoResult<()> {
1610    let unlimited: Vec<usize> = max_dims
1611        .iter()
1612        .enumerate()
1613        .filter(|&(_, &m)| m == u64::MAX)
1614        .map(|(d, _)| d)
1615        .collect();
1616    if unlimited.len() > 1 {
1617        return Err(crate::io::IoError::InvalidState(format!(
1618            "an extensible-array index supports at most one unlimited dimension, \
1619             but dimensions {unlimited:?} are all unlimited; a v2 B-tree index \
1620             handles two or more"
1621        )));
1622    }
1623    Ok(())
1624}
1625
1626/// Reject strings the dataset's declared character set cannot label.
1627///
1628/// A Rust `&str` is always UTF-8, so only an ASCII declaration (charset 0)
1629/// can be violated. libhdf5 stores the bytes unvalidated — its vlen write
1630/// path has no cset check anywhere — which mislabels them for every reader
1631/// that trusts the declaration (h5py raises on the same mismatch).
1632fn ensure_vlen_charset(charset: u8, strings: &[&str]) -> IoResult<()> {
1633    if charset == 0 {
1634        if let Some((i, s)) = strings.iter().enumerate().find(|(_, s)| !s.is_ascii()) {
1635            return Err(crate::io::IoError::InvalidState(format!(
1636                "string {i} ({s:?}) is not ASCII, but the dataset's character set is"
1637            )));
1638        }
1639    }
1640    Ok(())
1641}
1642
1643/// Runtime metadata for a fixed-array-indexed chunked dataset.
1644pub struct FixedArrayDatasetInfo {
1645    /// Chunk dimension sizes.
1646    pub chunk_dims: Vec<u64>,
1647    /// File offset of the FA header.
1648    pub fa_header_addr: u64,
1649    /// File offset of the FA data block.
1650    pub fa_dblk_addr: u64,
1651    /// In-memory copy of the FA header.
1652    pub fa_header: FixedArrayHeader,
1653    /// In-memory copy of the FA data block.
1654    pub fa_dblk: FixedArrayDataBlock,
1655    /// Number of chunks written so far.
1656    pub chunks_written: u64,
1657}
1658
1659/// Runtime metadata for an implicitly indexed chunked dataset — the index
1660/// that is no structure at all (`H5Dnone.c`).
1661///
1662/// Every chunk of the maximum-extent grid is allocated at create in one
1663/// contiguous run, in the row-major order [`crate::io::chunk_grid`] defines,
1664/// so a chunk's address is `data_addr + linear_index * chunk_bytes` and
1665/// nothing has to be recorded when one is written. libhdf5 picks this index
1666/// only when that arithmetic is total: no filter (every chunk is exactly
1667/// `chunk_bytes` long), no unlimited dimension (the run has a finite length),
1668/// and early allocation (the run exists before any write).
1669pub struct ImplicitDatasetInfo {
1670    /// Chunk dimension sizes.
1671    pub chunk_dims: Vec<u64>,
1672    /// File offset of the first chunk — the layout message's index address.
1673    pub data_addr: u64,
1674    /// Byte length of the whole chunk run: `nchunks * chunk_bytes`.
1675    pub data_size: u64,
1676}
1677
1678/// Runtime metadata for a single-chunk indexed dataset (`H5Dsingle.c`): a
1679/// fixed dataspace exactly one chunk wide in every dimension
1680/// (`dims == max_dims == chunk_dims`), so there is exactly one chunk and its
1681/// address — and, when filtered, its stored size and filter mask — are held
1682/// directly in the layout message rather than in any index structure.
1683///
1684/// libhdf5 selects this index ahead of the implicit and fixed-array indexes
1685/// whenever the shape qualifies, whether or not the dataset is filtered or
1686/// early-allocated (`H5D__layout_set_latest_indexing`, H5Dlayout.c).
1687pub struct SingleChunkDatasetInfo {
1688    /// Chunk dimension sizes (equal to the dataspace's `dims`).
1689    pub chunk_dims: Vec<u64>,
1690    /// File offset of the chunk, [`UNDEF_ADDR`] until the chunk is written
1691    /// (or immediately, for an unfiltered dataset created with early
1692    /// allocation).
1693    pub data_addr: u64,
1694    /// The chunk's full unfiltered byte length — `chunk_dims.product() *
1695    /// element_size`, fixed for the dataset's lifetime.
1696    pub data_size: u64,
1697    /// Stored (on-disk) byte length: equal to `data_size` when the dataset
1698    /// carries no filter pipeline; the filtered length once the chunk has
1699    /// been written, 0 before then.
1700    pub nbytes: u64,
1701    /// Filter mask recorded for the stored chunk (bit *i* set means filter
1702    /// *i* was skipped); meaningful only when the dataset is filtered.
1703    pub filter_mask: u32,
1704    /// Chunks written this session (0 or 1) — `storage_dirty`'s signal that
1705    /// the layout message's address/size/mask fields must be re-flushed.
1706    pub chunks_written: u64,
1707    /// Whether this dataset was created with early allocation
1708    /// (`H5D_ALLOC_TIME_EARLY`) — distinct from `data_addr` being defined,
1709    /// which also becomes true the moment an incrementally allocated
1710    /// dataset's one chunk is written; `build_dataset_header` needs this to
1711    /// tell the two apart when it reports the fill-value message's
1712    /// allocation time. Only ever set for an unfiltered dataset: a filtered
1713    /// chunk's stored length is not known until it is compressed, so there
1714    /// is nothing to allocate ahead of that write regardless of alloc time
1715    /// (the same gap `create_fixed_array_dataset_with_pipeline` has).
1716    pub early_alloc: bool,
1717}
1718
1719/// One chunk as the version-1 B-tree records it — the key libhdf5 stores
1720/// (`H5D_btree_key_t`) plus the address it keys.
1721pub struct BtreeV1ChunkRecord {
1722    /// Grid position of the chunk. The key's element offsets are derived from
1723    /// it at encode time (`scaled * chunk_dim`), so this is the one place the
1724    /// position is stored and the sort order is over these coordinates.
1725    pub scaled: Vec<u64>,
1726    /// File offset of the chunk's bytes.
1727    pub address: u64,
1728    /// Stored byte length — the filtered length when the dataset is filtered,
1729    /// the full chunk otherwise. `u32` because the key's field is.
1730    pub nbytes: u32,
1731    /// Filter mask: bit `i` set means filter `i` was skipped for this chunk.
1732    pub filter_mask: u32,
1733}
1734
1735/// Runtime metadata for a chunked dataset indexed by a version-1 B-tree —
1736/// the classic-format chunk index (`H5Dbtree.c`), and the only one a
1737/// version-0/1 superblock file can carry.
1738pub struct BtreeV1DatasetInfo {
1739    /// Chunk dimension sizes.
1740    pub chunk_dims: Vec<u64>,
1741    /// Maximum dimensions (u64::MAX = unlimited).
1742    pub max_dims: Vec<u64>,
1743    /// The file's v1-B-tree "K" ranks. Every node's width is derived from
1744    /// them, and they are recorded only in the superblock this file was
1745    /// opened with — so they are carried rather than re-derived.
1746    pub config: BTreeV1Config,
1747    /// The chunks, in key order (`scaled` ascending, lexicographically).
1748    pub records: Vec<BtreeV1ChunkRecord>,
1749    /// Pool of node-size blocks holding the tree's nodes, on the same terms
1750    /// as [`Bt2DatasetInfo::node_addrs`]: a flush re-serializes the whole
1751    /// bulk-loaded tree over them and allocates only the shortfall, so no
1752    /// flush can orphan a block it replaced.
1753    pub node_addrs: Vec<u64>,
1754    /// Address of the tree's root node — what the version-3 data layout
1755    /// message carries. `UNDEF_ADDR` until a flush puts a node in the file,
1756    /// which is the state libhdf5 leaves a chunked dataset in until its first
1757    /// chunk is written.
1758    pub root_addr: u64,
1759    /// Number of chunks written so far.
1760    pub chunks_written: u64,
1761}
1762
1763impl BtreeV1DatasetInfo {
1764    /// The chunk shape a key's offsets are scaled by: the chunk dimensions
1765    /// with the element size appended, which is also what the layout message
1766    /// stores.
1767    fn key_dims(&self, element_size: u64) -> Vec<u64> {
1768        let mut dims = self.chunk_dims.clone();
1769        dims.push(element_size);
1770        dims
1771    }
1772
1773    /// Bulk-load the tree this index's records describe.
1774    fn build_tree(&self, element_size: u64, sizeof_addr: usize) -> ChunkBTreeV1Tree {
1775        let dims = self.key_dims(element_size);
1776        let entries: Vec<(ChunkKey, u64)> = self
1777            .records
1778            .iter()
1779            .map(|r| {
1780                (
1781                    ChunkKey::for_chunk(&r.scaled, &dims, r.nbytes, r.filter_mask),
1782                    r.address,
1783                )
1784            })
1785            .collect();
1786        // The right boundary closes the tree past its greatest key, which is
1787        // the last record's — the records are kept in key order.
1788        let last = self
1789            .records
1790            .last()
1791            .map_or_else(|| vec![0; self.chunk_dims.len()], |r| r.scaled.clone());
1792        ChunkBTreeV1Tree::build(
1793            &entries,
1794            ChunkKey::right_bound(&last, &dims),
1795            &self.config,
1796            sizeof_addr,
1797        )
1798    }
1799
1800    /// Where `scaled` sits in [`records`](Self::records): `Ok` at its record,
1801    /// `Err` at the position one would be inserted at.
1802    fn position(&self, scaled: &[u64]) -> Result<usize, usize> {
1803        self.records
1804            .binary_search_by(|r| r.scaled.as_slice().cmp(scaled))
1805    }
1806}
1807
1808/// Runtime metadata for a B-tree v2 indexed chunked dataset.
1809pub struct Bt2DatasetInfo {
1810    /// Chunk dimension sizes.
1811    pub chunk_dims: Vec<u64>,
1812    /// File offset of the BT2 header.
1813    pub bt2_header_addr: u64,
1814    /// Pool of node-size blocks (the index's
1815    /// [`node_size`](Bt2ChunkIndex::node_size) bytes each) holding the tree's
1816    /// nodes, in the order [`Bt2Tree::encode`] emits them.
1817    ///
1818    /// The single owner of the tree's node addresses: a flush re-serializes the
1819    /// whole tree over these blocks and allocates only the shortfall, so no
1820    /// flush can orphan a block it replaced. Every node is the same size, so a
1821    /// block stays usable however the tree reshapes.
1822    ///
1823    /// The pool holds exactly one block per node after every flush, in both
1824    /// directions: a taller tree allocates the shortfall, a smaller one frees
1825    /// the surplus. Nothing here depends on the record count only ever rising,
1826    /// so a record-removal path can be added to [`Bt2ChunkIndex`] without the
1827    /// blocks it drops going unreachable.
1828    pub node_addrs: Vec<u64>,
1829    /// In-memory chunk index.
1830    pub index: Bt2ChunkIndex,
1831    /// Number of chunks written so far.
1832    pub chunks_written: u64,
1833}
1834
1835/// Metadata for a group being written.
1836pub struct GroupInfo {
1837    /// Full path of this group (e.g. "/detector" or "/detector/raw").
1838    pub name: String,
1839    /// Index of the parent group in the groups vec, or None for root-level groups.
1840    pub parent: Option<usize>,
1841    /// Indices of child datasets (into `datasets` vec).
1842    pub child_datasets: Vec<usize>,
1843    /// Indices of child groups (into `groups` vec).
1844    pub child_groups: Vec<usize>,
1845    /// File offset of this group's object header (set during finalize).
1846    pub obj_header_addr: u64,
1847    /// File offset of the on-disk header a reopen found for this group, so
1848    /// finalize can free the block it supersedes.
1849    pub obj_header_written_addr: Option<u64>,
1850    /// Encoded size of that on-disk header (first block).
1851    /// Every block the object's on-disk header occupies, chunk 0 first, or
1852    /// empty when it has none yet. All of them are freed together: a rewrite
1853    /// re-encodes the whole chain into one fresh chunk, so a continuation
1854    /// block left behind is space no free-space manager records.
1855    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
1856    /// Soft-deleted: excluded from finalize output.
1857    pub deleted: bool,
1858    /// Attributes attached to this group (e.g. NeXus `NX_class`).
1859    pub attributes: Vec<AttributeEntry>,
1860    /// When the link naming this group was created, on the writer's single
1861    /// monotonic sequence. Groups, datasets and hard links share it, so a
1862    /// parent can order its links the way they were actually made.
1863    pub creation_seq: u64,
1864    /// How this group records creation order for its links and, separately,
1865    /// for its attributes. Creation-order tracking is a property of the
1866    /// object's creation property list in libhdf5, so it is captured here
1867    /// when the group is created rather than read from the writer at
1868    /// finalize: a later change of policy must not rewrite an object already
1869    /// made.
1870    pub track_order: TrackOrder,
1871    /// The times this group tracks, on the same terms as
1872    /// [`DatasetInfo::times`]. A version-1 group header records none of them:
1873    /// nothing calls `H5O_touch_oh` with `force` for a group, so the message a
1874    /// version-1 dataset gets is never created for one.
1875    pub times: Option<ObjectTimes>,
1876}
1877
1878/// One object's creation-order policy, with the two subsystems libhdf5 keeps
1879/// apart kept apart here too.
1880///
1881/// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` are separate
1882/// calls reading back out of separate places on disk — the Link Info message
1883/// and the object header's own flag bits — and a file may set either alone.
1884/// Carrying them as one flag made a reopen give a one-of-two file both or
1885/// neither.
1886#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
1887pub struct TrackOrder {
1888    /// Creation order of the links this group holds. Meaningless for a
1889    /// dataset, which is why `DatasetInfo` keeps only the attribute half.
1890    pub links: CreationOrder,
1891    /// Creation order of the attributes attached to this object.
1892    pub attrs: CreationOrder,
1893}
1894
1895impl TrackOrder {
1896    /// The policy the crate's single `track_order` knob selects: both
1897    /// subsystems tracked *and* indexed, or neither — the pair h5py's
1898    /// `File(track_order=True)` writes.
1899    pub fn uniform(track: bool) -> Self {
1900        let order = if track {
1901            CreationOrder::Indexed
1902        } else {
1903            CreationOrder::Untracked
1904        };
1905        Self {
1906            links: order,
1907            attrs: order,
1908        }
1909    }
1910}
1911
1912/// The object a [`HardLink`] resolves to.
1913#[derive(Clone, Copy)]
1914pub enum HardLinkTarget {
1915    /// Index into the writer's `datasets` vec.
1916    Dataset(usize),
1917    /// Index into the writer's `groups` vec.
1918    Group(usize),
1919}
1920
1921/// A user-created hard link: an additional name, in some group, for an
1922/// object that already exists under its own name.
1923///
1924/// The HDF5 file format makes every group entry a `name -> object header
1925/// address` mapping, so a hard link is just a second such entry pointing at
1926/// an already-written object. No data is copied.
1927#[derive(Clone)]
1928pub struct HardLink {
1929    /// Parent group index (`None` = the root group).
1930    pub parent: Option<usize>,
1931    /// Leaf name of the link within the parent group.
1932    pub name: String,
1933    /// Object this link resolves to.
1934    pub target: HardLinkTarget,
1935    /// When this link was created; see [`GroupInfo::creation_seq`].
1936    pub creation_seq: u64,
1937}
1938
1939/// A user-created symbolic link: a name in a group whose value is a path
1940/// rather than an object header address.
1941///
1942/// A soft link holds a path within this file; an external link holds a file
1943/// name and a path within that file. Neither names an object this writer
1944/// owns, so — unlike [`HardLink`] — nothing about it is resolved: the link is
1945/// stored as written and answered at traversal time, exactly as `H5Lcreate_soft`
1946/// and `H5Lcreate_external` store theirs.
1947#[derive(Clone)]
1948pub struct SymbolicLink {
1949    /// Parent group index (`None` = the root group).
1950    pub parent: Option<usize>,
1951    /// Leaf name of the link within the parent group.
1952    pub name: String,
1953    /// The path (and, for an external link, the file) this link names.
1954    pub target: LinkTarget,
1955    /// When this link was created; see [`GroupInfo::creation_seq`].
1956    pub creation_seq: u64,
1957}
1958
1959/// A committed (named) datatype: an object header holding one datatype
1960/// message and nothing else, reached by a link like any other object.
1961///
1962/// `H5Tcommit2` makes the type an object in its own right so several datasets
1963/// can declare they share it; each of those datasets then stores a pointer to
1964/// this object header in place of its own datatype message. The object's
1965/// reference count is therefore the links naming it *plus* the datasets
1966/// sharing it — `H5O__shared_link_adj` counts a share as a link — and an
1967/// object no link and no dataset reaches is not written at all.
1968#[derive(Clone)]
1969pub struct CommittedDatatype {
1970    /// Full path with no leading `/`, the form dataset names take.
1971    pub name: String,
1972    /// Parent group index (`None` = the root group).
1973    pub parent: Option<usize>,
1974    /// The committed type.
1975    pub datatype: DatatypeMessage,
1976    /// When the link naming it was created; see [`GroupInfo::creation_seq`].
1977    pub creation_seq: u64,
1978    /// The times it tracks, on the same terms as [`DatasetInfo::times`]. A
1979    /// version-1 committed datatype header records none of them, for the same
1980    /// reason a version-1 group's does not.
1981    pub times: Option<ObjectTimes>,
1982    /// File offset of its object header (set during finalize).
1983    pub obj_header_addr: u64,
1984}
1985
1986/// Where the object header a dataset's shared datatype pointer must name
1987/// comes from.
1988///
1989/// A dataset built on a committed type stores no datatype message: it stores
1990/// the address of the type's object header. Only the address matters at
1991/// encode time, but it is knowable at two different moments — a type this
1992/// session commits has no address until finalize lays the file out, while one
1993/// a reopen found is already at an address this session will not move. Naming
1994/// both here keeps [`build_dataset_header`](Hdf5Writer::build_dataset_header)
1995/// the one place that turns a share into a pointer, whichever way the share
1996/// arrived.
1997#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1998pub enum CommittedTypeRef {
1999    /// A type committed in this session, by its index in
2000    /// [`committed_datatypes`](Hdf5Writer::committed_datatypes); its address
2001    /// is read from that registry once finalize has stamped one.
2002    Session(usize),
2003    /// A committed datatype a reopen kept by its bytes, at the object header
2004    /// address it already occupies.
2005    Preserved(u64),
2006}
2007
2008/// A link a reopened file already held that this writer cannot express.
2009///
2010/// Soft, external and user-defined links have no creation, retarget or delete
2011/// operation here — only hard links do — so a header rewrite that emits what
2012/// the registry models would erase them. Their encoded `Link` message rides
2013/// along instead and is written back byte for byte, which preserves every
2014/// field (name character set, creation order, the link value) without this
2015/// writer having to model any of them.
2016///
2017/// A *hard* link is preserved the same way when the object it names is one
2018/// the reopen could not model: writing the link back unchanged leaves that
2019/// object's header exactly where it is, which is the only way the rewrite can
2020/// keep what it cannot rebuild.
2021#[derive(Clone)]
2022pub struct PreservedLink {
2023    /// Parent group index (`None` = the root group).
2024    pub parent: Option<usize>,
2025    /// Leaf name of the link within the parent group.
2026    pub name: String,
2027    /// The link's class, decoded once at collection so listings can report
2028    /// it. Never the source of what gets written — `encoded` is.
2029    pub class: crate::io::reader::LinkClass,
2030    /// The encoded `Link` message body, exactly as read from the file.
2031    pub encoded: Vec<u8>,
2032    /// Why the object this link names could not be modelled, for the callers
2033    /// that ask for it by name. `None` when the link's own class — not its
2034    /// target — is what this writer cannot express.
2035    pub reason: Option<String>,
2036    /// What the object this link names is, when the walk could tell. A
2037    /// listing asks this; `reason` is prose for the caller that asks why.
2038    pub kind: PreservedKind,
2039}
2040
2041/// Every link a reopen walk met, split by what the writer can do with it.
2042/// A header rewrite emits both halves, so a link in neither half is a link
2043/// the close would destroy.
2044#[derive(Default)]
2045struct CollectedLinks {
2046    /// Hard links whose target the reopen modelled, with the plan that says
2047    /// how to rebuild it.
2048    hard: Vec<(HardEntry, CollectedObject)>,
2049    /// Links written back unchanged: the class this writer cannot express,
2050    /// and the hard links whose object it cannot model.
2051    preserved: Vec<PreservedEntry>,
2052}
2053
2054/// One hard link the reopen walk met: what it names, and the exact message
2055/// that names it.
2056#[derive(Clone)]
2057struct HardEntry {
2058    /// Full link path, in the no-leading-`/` form the registry uses.
2059    path: String,
2060    /// Object header address the link names.
2061    address: u64,
2062    /// The encoded `Link` message body, exactly as read from the file.
2063    encoded: Vec<u8>,
2064}
2065
2066/// A link the rewrite writes back exactly as it read it.
2067struct PreservedEntry {
2068    path: String,
2069    class: crate::io::reader::LinkClass,
2070    encoded: Vec<u8>,
2071    /// Why the object it names could not be modelled; `None` when the link's
2072    /// own class is what this writer cannot express.
2073    reason: Option<String>,
2074    /// What the object is, when the walk could tell.
2075    kind: PreservedKind,
2076}
2077
2078/// What a reopen can do with one object it reached.
2079///
2080/// A header rewrite emits a modelled object out of the registry, so the
2081/// registry may hold an object only when *every* message the model consumes
2082/// decoded. A partial read is not a smaller object, it is a different one:
2083/// before this rule a dataset whose datatype message did not decode was
2084/// registered as a group, and the close rewrote its header as one.
2085enum ObjectPlan {
2086    /// A dataset the rewrite can rebuild.
2087    Dataset(Box<DatasetParts>),
2088    /// A group the rewrite can rebuild, and the links it holds.
2089    Group(GroupParts),
2090    /// An object this writer cannot model, and why. Its header is never
2091    /// rewritten and never freed; the link naming it is written back byte for
2092    /// byte, so the object stays exactly as the file already had it — what
2093    /// libhdf5 does with the parts of a file it does not understand.
2094    ///
2095    /// `kind` is what the walk could still tell about the object it is
2096    /// keeping. Not modelling an object is not the same as not knowing what
2097    /// it is, and answering the second question with the first is what made
2098    /// `named_datatype_names` deny, in write mode, a datatype the same file
2099    /// reports in read mode.
2100    Preserve { why: String, kind: PreservedKind },
2101}
2102
2103/// What a preserved object is, as far as the reopen walk could tell.
2104///
2105/// Deliberately not a copy of the reader's `ObjectKind`: that one carries the
2106/// decoded object, and a preserved object is precisely the one whose contents
2107/// the writer does not decode. This says only what a listing needs.
2108#[derive(Clone, Copy, PartialEq, Eq, Debug)]
2109pub enum PreservedKind {
2110    /// The walk did not classify it — or the link's own class, not its
2111    /// target, is what could not be expressed.
2112    Unclassified,
2113    /// A committed (named) datatype, by
2114    /// [`header_is_committed_datatype`](crate::io::reader::header_is_committed_datatype).
2115    NamedDatatype,
2116}
2117
2118impl ObjectPlan {
2119    /// An object kept by its bytes, of a kind the walk did not classify.
2120    ///
2121    /// Every reason that is a *failure* to read reaches this: a message that
2122    /// did not decode says nothing about what the object was.
2123    fn preserve(why: impl Into<String>) -> Self {
2124        ObjectPlan::Preserve {
2125            why: why.into(),
2126            kind: PreservedKind::Unclassified,
2127        }
2128    }
2129}
2130
2131/// The messages a dataset's rewrite is built from, all decoded.
2132struct DatasetParts {
2133    /// Every block the header chain occupies, chunk 0 first. All of them are
2134    /// superseded: the rewrite re-encodes the whole chain into one fresh
2135    /// chunk, so a continuation left unfreed is space nothing claims.
2136    header_blocks: crate::io::object_header_io::HeaderBlocks,
2137    datatype: DatatypeMessage,
2138    /// The committed datatype object header `datatype` was read *through*,
2139    /// when the header stores a pointer instead of a message of its own.
2140    ///
2141    /// The literal type is in `datatype` either way, because the read resolves
2142    /// the pointer before anything decodes it; this is what a rewrite needs to
2143    /// put the pointer back rather than inline a copy of the named type and
2144    /// leave `H5Tcommitted` false.
2145    committed_type: Option<u64>,
2146    dataspace: crate::format::messages::dataspace::DataspaceMessage,
2147    /// The object format the reopen found this dataset's messages written in,
2148    /// read from the dataspace message's own version byte.
2149    ///
2150    /// A version-2 superblock does not settle it: `H5F__super_init` raises the
2151    /// superblock for a shared-message table or non-default file-space
2152    /// properties without touching `H5F_LOW_BOUND` (H5Fsuper.c:1135, :1144), so
2153    /// a file created at the earliest bound with either can hold version-1
2154    /// messages under a version-2 superblock — which is what
2155    /// `tests/fixtures/sohm_*.h5` are.
2156    read_format: ObjectFormat,
2157    layout: crate::format::messages::data_layout::DataLayoutMessage,
2158    filter_pipeline: Option<FilterPipeline>,
2159    fill_value: Option<Vec<u8>>,
2160    /// The fill-value message's write-time byte, preserved across a
2161    /// rewrite the same way `fill_value` is — an appended-to dataset must
2162    /// keep the policy libhdf5 (or this writer) declared for it, not fall
2163    /// back to the `H5D_CRT_FILL_TIME_DEF` a fresh dataset gets.
2164    fill_write_time: u8,
2165    attributes: Vec<AttributeEntry>,
2166    /// The creation-order policy the on-disk header declares; a rewrite that
2167    /// read it from the writer instead would stamp this session's policy onto
2168    /// an object libhdf5 created under another.
2169    track_order: TrackOrder,
2170    /// The times the on-disk header records, for the same reason: whether an
2171    /// object tracks them is settled when it is created, not when it is
2172    /// rewritten. Recovered by [`ObjectHeader::recorded_times`].
2173    times: Option<ObjectTimes>,
2174    /// The dense storage the rewrite supersedes and must free.
2175    dense: DenseCarry,
2176    /// The External File List the header carries, with each slot's name
2177    /// already read back out of the local heap the message points at. `None`
2178    /// for a dataset whose raw data is in this file.
2179    ///
2180    /// Carried rather than re-derived because the rewrite has to re-emit the
2181    /// message: a contiguous layout with an undefined address and no EFL
2182    /// beside it is a dataset with no data at all, so dropping this on a
2183    /// header rewrite would silently unlink every external byte.
2184    external: Option<ExternalStorage>,
2185}
2186
2187/// The same for a group, plus the links it holds — decoded once, with the
2188/// bytes they came from, so the walk and the rewrite agree on its contents.
2189struct GroupParts {
2190    header_blocks: crate::io::object_header_io::HeaderBlocks,
2191    attributes: Vec<AttributeEntry>,
2192    links: Vec<(crate::format::messages::link::LinkMessage, Vec<u8>)>,
2193    track_order: TrackOrder,
2194    times: Option<ObjectTimes>,
2195    dense: DenseCarry,
2196    /// The symbol-table storage a classic group's header names — the blocks
2197    /// the rewrite supersedes. `None` for a link-message group, which has
2198    /// none. Its links are already in `links`: the walk turns each symbol
2199    /// table entry into the link message it stands for, so nothing downstream
2200    /// has to know which of the two forms the group was in.
2201    stab: Option<StabExtents>,
2202}
2203
2204/// The dense storage one reopened object's header names, which the rewrite of
2205/// that header stops naming and therefore has to free. Both halves are read
2206/// back before this is built — a heap that could not be read makes the object
2207/// [`ObjectPlan::Preserve`], so nothing here describes storage whose contents
2208/// were lost.
2209#[derive(Default)]
2210struct DenseCarry {
2211    attrs: Option<AttributeInfoMessage>,
2212    links: Option<LinkInfoMessage>,
2213}
2214
2215/// A modelled object, as the walk hands it to the registry rebuild. A group's
2216/// links are not here: the walk followed them, and each child is an entry of
2217/// its own.
2218enum CollectedObject {
2219    Dataset(Box<DatasetParts>),
2220    Group {
2221        header_blocks: crate::io::object_header_io::HeaderBlocks,
2222        attributes: Vec<AttributeEntry>,
2223        track_order: TrackOrder,
2224        times: Option<ObjectTimes>,
2225        dense: DenseCarry,
2226        stab: Option<StabExtents>,
2227    },
2228}
2229
2230/// The reopen's discovery pass: one walk that classifies every object it
2231/// reaches and descends into the groups among them.
2232///
2233/// Every object the close will touch is decided here and nowhere else, so
2234/// "modelled or preserved" is a property of the walk rather than of whatever
2235/// each later stage happened to be able to decode.
2236struct ReopenWalk<'a> {
2237    handle: &'a mut FileHandle,
2238    meta: &'a crate::io::FileMeta,
2239    out: CollectedLinks,
2240    /// Object headers already descended into, so hard-link cycles end.
2241    visited: std::collections::HashSet<u64>,
2242}
2243
2244impl<'a> ReopenWalk<'a> {
2245    fn new(handle: &'a mut FileHandle, meta: &'a crate::io::FileMeta) -> Self {
2246        Self {
2247            handle,
2248            meta,
2249            out: CollectedLinks::default(),
2250            visited: std::collections::HashSet::new(),
2251        }
2252    }
2253
2254    /// Everything the walk found.
2255    fn finish(self) -> CollectedLinks {
2256        self.out
2257    }
2258
2259    /// Decide what the reopen can do with the object at `addr`.
2260    ///
2261    /// The single gate: every object the rewrite touches is classified here,
2262    /// and an object is modelled only when each message the model consumes
2263    /// decoded. See [`ObjectPlan`] for why anything else must keep its bytes.
2264    fn plan(&mut self, addr: u64) -> IoResult<ObjectPlan> {
2265        let (handle, meta) = (&mut *self.handle, self.meta);
2266        let ctx = &meta.ctx;
2267        use crate::format::messages::data_layout::DataLayoutMessage;
2268        use crate::format::messages::dataspace::DataspaceMessage;
2269        use crate::format::messages::link::{CharacterSet, LinkMessage};
2270        use crate::format::messages::link_info::LinkInfoMessage;
2271        use crate::format::messages::shared::MSG_FLAG_SHARED;
2272        use crate::format::messages::{
2273            MSG_ATTRIBUTE, MSG_DATASPACE, MSG_DATATYPE, MSG_DATA_LAYOUT, MSG_EXTERNAL_FILE_LIST,
2274            MSG_FILL_VALUE, MSG_FILTER_PIPELINE, MSG_LINK, MSG_LINK_INFO, MSG_SYMBOL_TABLE,
2275        };
2276
2277        // The whole chain, messages and blocks alike: a filter pipeline or an
2278        // attribute that spilled into a continuation is one the rewrite would
2279        // otherwise drop, and a continuation block it does not know about is
2280        // one the rewrite would orphan.
2281        let (header, header_blocks) =
2282            match crate::io::object_header_io::read_object_header_with_blocks(handle, meta, addr) {
2283                Ok(h) => h,
2284                Err(e) => {
2285                    return Ok(ObjectPlan::preserve(format!(
2286                        "its object header chain does not read: {e}"
2287                    )))
2288                }
2289            };
2290
2291        // The policy, the times and the storage the header declares, read once
2292        // from the whole chain: all three are properties of the object, not of
2293        // any one message the loop below happens to reach.
2294        let track_order = recover_track_order(&header, ctx);
2295        let times = header.recorded_times();
2296        let (dense_attrs, dense_links) = superseded_dense(&header, ctx);
2297
2298        // Attributes come from the reader's collector rather than from the
2299        // loop below, so compact, dense and shared attributes all reach the
2300        // rewrite by the one path that knows how to read each of them. An
2301        // object whose set did not read whole is preserved: a short set here
2302        // would be a rewrite deleting the attributes it could not read.
2303        let attributes = match take_reopened_attributes(
2304            crate::io::reader::collect_object_attributes(handle, ctx, &header),
2305            &format!("the object at {addr:#x}"),
2306        ) {
2307            Ok(a) => a,
2308            Err(e) => {
2309                return Ok(ObjectPlan::preserve(format!(
2310                    "its attributes do not read back whole: {e}"
2311                )))
2312            }
2313        };
2314
2315        let mut datatype = None;
2316        let mut dataspace = None;
2317        let mut layout = None;
2318        let mut filter_pipeline = None;
2319        let mut fill_value = None;
2320        // No fill-value message at all is the library default, the same
2321        // convention the reader-side decode (`Hdf5Reader::dataset_info`)
2322        // uses for `fill_defined`.
2323        let mut fill_write_time: u8 = FILL_TIME_IFSET;
2324        let mut external = None;
2325        let mut links = Vec::new();
2326        let mut stab = None;
2327        // A datatype, dataspace or layout message says the object is not a
2328        // group, whether or not the three a dataset needs are all there.
2329        let mut dataset_shaped = false;
2330
2331        for msg in &header.messages {
2332            let consumed = matches!(
2333                msg.msg_type,
2334                MSG_DATATYPE
2335                    | MSG_DATASPACE
2336                    | MSG_DATA_LAYOUT
2337                    | MSG_FILTER_PIPELINE
2338                    | MSG_FILL_VALUE
2339                    | MSG_EXTERNAL_FILE_LIST
2340                    | MSG_ATTRIBUTE
2341                    | MSG_LINK
2342                    | MSG_LINK_INFO
2343                    | MSG_SYMBOL_TABLE
2344            );
2345            // A shared message holds a reference to where its body lives, not
2346            // the body. Decoding those bytes as one does not fail loudly — the
2347            // reference's version byte reads as a version and a class of its
2348            // own — so the guard is the only thing between a shared datatype
2349            // and a rewrite that invents a type for it.
2350            if consumed && msg.flags & MSG_FLAG_SHARED != 0 {
2351                return Ok(ObjectPlan::preserve(format!(
2352                    "its message of type {:#04x} is a shared-message reference, which this \
2353                     writer does not resolve",
2354                    msg.msg_type
2355                )));
2356            }
2357            macro_rules! consume {
2358                ($decode:expr, $what:literal) => {
2359                    match $decode {
2360                        Ok(v) => v,
2361                        Err(e) => {
2362                            return Ok(ObjectPlan::preserve(format!(
2363                                "its {} message does not decode: {e}",
2364                                $what
2365                            )))
2366                        }
2367                    }
2368                };
2369            }
2370            match msg.msg_type {
2371                // The pre-1.6 modification time, a formatted date string
2372                // (`H5O_MTIME`, type 0x0E). `recorded_times` reads only the
2373                // modern form, and a rewrite emits only that, so an object
2374                // carrying this one would come back out with the time it
2375                // recorded gone. Keeping its bytes is the same answer an
2376                // undecodable message already gets.
2377                crate::format::messages::MSG_MOD_TIME_OLD => {
2378                    return Ok(ObjectPlan::preserve(
2379                        "it carries a pre-1.6 modification time message, which this writer \
2380                         reads but does not write",
2381                    ))
2382                }
2383                MSG_DATATYPE => {
2384                    dataset_shaped = true;
2385                    let (dt, _) = consume!(DatatypeMessage::decode(&msg.data, ctx), "datatype");
2386                    datatype = Some(dt);
2387                }
2388                MSG_DATASPACE => {
2389                    dataset_shaped = true;
2390                    let version = msg.data.first().copied().unwrap_or(1);
2391                    let (ds, _) = consume!(DataspaceMessage::decode(&msg.data, ctx), "dataspace");
2392                    dataspace = Some((ds, version));
2393                }
2394                MSG_DATA_LAYOUT => {
2395                    dataset_shaped = true;
2396                    let (dl, _) =
2397                        consume!(DataLayoutMessage::decode(&msg.data, ctx), "data layout");
2398                    layout = Some(dl);
2399                }
2400                MSG_FILTER_PIPELINE => {
2401                    let (p, _) = consume!(FilterPipeline::decode(&msg.data), "filter pipeline");
2402                    if !p.filters.is_empty() {
2403                        filter_pipeline = Some(p);
2404                    }
2405                }
2406                MSG_FILL_VALUE => {
2407                    let (fv, _) = consume!(FillValueMessage::decode(&msg.data), "fill value");
2408                    if fv.fill_defined == 2 {
2409                        fill_value = fv.fill_value;
2410                    }
2411                    fill_write_time = fv.fill_write_time;
2412                }
2413                MSG_EXTERNAL_FILE_LIST => {
2414                    dataset_shaped = true;
2415                    let (efl, _) = consume!(
2416                        ExternalFileListMessage::decode(&msg.data, ctx),
2417                        "external file list"
2418                    );
2419                    // The names live in a local heap of their own, so the
2420                    // rewrite cannot re-emit the message from its bytes alone
2421                    // — it has to be able to point at the same strings. A heap
2422                    // that does not read back leaves the object preserved,
2423                    // which is what keeps its data reachable.
2424                    let resolved = match crate::io::reader::Hdf5Reader::resolve_external_file_slots(
2425                        handle, ctx, &efl,
2426                    ) {
2427                        Ok(r) => r,
2428                        Err(e) => {
2429                            return Ok(ObjectPlan::preserve(format!(
2430                                "its external file list names do not read back: {e}"
2431                            )))
2432                        }
2433                    };
2434                    external = Some(ExternalStorage {
2435                        heap_addr: efl.heap_addr,
2436                        // `H5Fopen` opens no dataset, so nothing has read a
2437                        // dapl for this one yet; the first handle it hands
2438                        // out settles the prefix.
2439                        prefix: EfilePrefix::default(),
2440                        files: efl
2441                            .slots
2442                            .iter()
2443                            .zip(resolved)
2444                            .map(|(slot, seg)| ExternalFile {
2445                                name: seg.name,
2446                                name_offset: slot.name_offset,
2447                                offset: slot.offset,
2448                                size: slot.size,
2449                            })
2450                            .collect(),
2451                    });
2452                }
2453                MSG_LINK => {
2454                    let (l, _) = consume!(LinkMessage::decode(&msg.data, ctx), "link");
2455                    links.push((l, msg.data.clone()));
2456                }
2457                MSG_LINK_INFO => {
2458                    let (li, _) = consume!(LinkInfoMessage::decode(&msg.data, ctx), "link info");
2459                    // Once a group holds enough links libhdf5 moves them into
2460                    // the fractal heap this message names and writes no `Link`
2461                    // messages at all. Reading them back is what makes the
2462                    // rewrite emit the group with its children; a rewrite from
2463                    // the header messages alone emitted it empty, orphaning
2464                    // every object below it.
2465                    if li.fractal_heap_address != UNDEF_ADDR {
2466                        let dense = match crate::io::reader::Hdf5Reader::read_dense_links(
2467                            handle,
2468                            ctx,
2469                            li.fractal_heap_address,
2470                        ) {
2471                            Ok(l) => l,
2472                            Err(e) => {
2473                                return Ok(ObjectPlan::preserve(format!(
2474                                    "its dense link storage does not read: {e}"
2475                                )))
2476                            }
2477                        };
2478                        // Re-encoded rather than carried as bytes: a heap
2479                        // object is not a header message, so there are no
2480                        // message bytes to carry. The encoding round-trips
2481                        // through the same decoder that just read it.
2482                        links.extend(dense.into_iter().map(|l| {
2483                            let bytes = l.encode(ctx);
2484                            (l, bytes)
2485                        }));
2486                    }
2487                }
2488                MSG_SYMBOL_TABLE => {
2489                    // A classic group keeps no link message at all: its links
2490                    // are symbol table entries in the B-tree this message
2491                    // names. Turning each into the link message it stands for
2492                    // is what lets the rest of the reopen — the walk, the
2493                    // registry, the preserve path — work on one link model
2494                    // whichever form the group is in.
2495                    let Some(s) = Stab::decode(&msg.data, ctx) else {
2496                        return Ok(ObjectPlan::preserve(
2497                            "its symbol table message is shorter than the two addresses it \
2498                             must carry",
2499                        ));
2500                    };
2501                    let contents = match crate::io::symbol_table_io::read_stab(handle, meta, s) {
2502                        Ok(c) => c,
2503                        Err(e) => {
2504                            return Ok(ObjectPlan::preserve(format!(
2505                                "its symbol table does not read: {e}"
2506                            )))
2507                        }
2508                    };
2509                    stab = Some(contents.extents);
2510                    links.extend(contents.links.into_iter().map(|l| {
2511                        let msg = match l.target {
2512                            StabTarget::Hard { addr, .. } => LinkMessage::hard(&l.name, addr),
2513                            StabTarget::Soft { value } => LinkMessage::soft(&l.name, &value),
2514                        };
2515                        // An entry carries no character set field, so the link
2516                        // it stands for has the file default whatever its name
2517                        // looks like (`H5G__ent_to_link`, H5Gent.c:372).
2518                        // Deriving one from the name would take a group
2519                        // libhdf5 wrote with a high-byte ASCII name out of its
2520                        // symbol table on the rewrite.
2521                        let msg = msg.with_cset(CharacterSet::Ascii);
2522                        let bytes = msg.encode(ctx);
2523                        (msg, bytes)
2524                    }));
2525                }
2526                _ => {}
2527            }
2528        }
2529
2530        // libhdf5 refuses a layout that disagrees with its sibling dataspace
2531        // and datatype as the dataset opens (`H5O__layout_decode` for the
2532        // chunk rank, `H5D__compact_init` for the compact size); modelled
2533        // anyway, the disagreement would be read at the wrong rank or past
2534        // the compact payload, so the dataset keeps its bytes, exactly as
2535        // unreadable as the file already had it.
2536        if let (Some((ds, _)), Some(dt), Some(dl)) = (&dataspace, &datatype, &layout) {
2537            if let Err(e) = dl.check_against_dataset(ds, dt, ctx) {
2538                return Ok(ObjectPlan::preserve(format!(
2539                    "its layout doesn't fit its dataspace and datatype: {e}"
2540                )));
2541            }
2542        }
2543
2544        match (datatype, dataspace, layout) {
2545            // A layout `rebuild_dataset` has no arm for leaves the registry
2546            // entry with an undefined data address, and the close then rewrites
2547            // the header as a contiguous, unallocated dataset — every element
2548            // gone, silently. Only the layouts that rebuild are modelled; the
2549            // rest keep their bytes, as an undecodable message already does.
2550            // The virtual layout is this.
2551            (Some(_), Some(_), Some(layout)) if !layout_rebuilds(&layout) => {
2552                Ok(ObjectPlan::preserve(format!(
2553                    "its data layout is {}, which this writer reads but does not build",
2554                    layout.describe()
2555                )))
2556            }
2557            (Some(datatype), Some((dataspace, dataspace_version)), Some(layout)) => {
2558                // Asked of the raw chain, not of `header`: the read above has
2559                // already put the named type's message in place of the pointer.
2560                let committed_type = match crate::io::object_header_io::committed_datatype_address(
2561                    handle, meta, addr,
2562                ) {
2563                    Ok(c) => c,
2564                    Err(e) => {
2565                        return Ok(ObjectPlan::preserve(format!(
2566                            "its shared datatype pointer does not decode: {e}"
2567                        )))
2568                    }
2569                };
2570                Ok(ObjectPlan::Dataset(Box::new(DatasetParts {
2571                    header_blocks,
2572                    datatype,
2573                    committed_type,
2574                    dataspace,
2575                    read_format: if dataspace_version <= 1 {
2576                        ObjectFormat::Legacy
2577                    } else {
2578                        ObjectFormat::Modern
2579                    },
2580                    layout,
2581                    filter_pipeline,
2582                    fill_value,
2583                    fill_write_time,
2584                    attributes,
2585                    track_order,
2586                    times,
2587                    dense: DenseCarry {
2588                        attrs: dense_attrs,
2589                        links: dense_links,
2590                    },
2591                    external,
2592                })))
2593            }
2594            // A committed (named) datatype has a datatype message and neither
2595            // of the other two; so does a dataset whose header this crate only
2596            // half understands. Neither is a group, and modelling either as
2597            // one is what rewrote them into empty groups. They part company
2598            // here and nowhere else: the datatype is kept by its bytes like
2599            // the other, but a listing can still name it.
2600            _ if crate::io::reader::header_is_committed_datatype(&header) => {
2601                Ok(ObjectPlan::Preserve {
2602                    why: "it is a committed (named) datatype, which this writer carries by \
2603                          its bytes rather than re-encoding"
2604                        .into(),
2605                    kind: PreservedKind::NamedDatatype,
2606                })
2607            }
2608            _ if dataset_shaped => Ok(ObjectPlan::preserve(
2609                "it carries a datatype, dataspace or layout message but not the three a \
2610                 dataset is built from; this writer models only groups and datasets",
2611            )),
2612            _ => Ok(ObjectPlan::Group(GroupParts {
2613                header_blocks,
2614                attributes,
2615                links,
2616                track_order,
2617                times,
2618                dense: DenseCarry {
2619                    attrs: dense_attrs,
2620                    links: dense_links,
2621                },
2622                stab,
2623            })),
2624        }
2625    }
2626
2627    /// Walk `links` (one group's, already decoded), classifying every object
2628    /// they name and descending into the groups among them.
2629    fn group(
2630        &mut self,
2631        links: &[(crate::format::messages::link::LinkMessage, Vec<u8>)],
2632        prefix: &str,
2633        depth: usize,
2634    ) -> IoResult<()> {
2635        // Bound nesting depth so a pathologically deep group chain cannot
2636        // overflow the stack (the `visited` set bounds total work but not
2637        // recursion depth).
2638        if depth > 256 {
2639            return Ok(());
2640        }
2641        use crate::format::messages::link::LinkTarget;
2642        for (link, encoded) in links {
2643            let full_name = if prefix.is_empty() {
2644                link.name.clone()
2645            } else {
2646                format!("{}/{}", prefix, link.name)
2647            };
2648
2649            // Only a hard link names an object this writer can rebuild. Every
2650            // other class is kept by its bytes, because a close that emitted
2651            // only what the registry models would drop it from the file.
2652            let LinkTarget::Hard { address } = &link.target else {
2653                self.out.preserved.push(PreservedEntry {
2654                    path: full_name,
2655                    class: crate::io::reader::LinkClass::from_target(&link.target),
2656                    encoded: encoded.clone(),
2657                    reason: None,
2658                    kind: PreservedKind::Unclassified,
2659                });
2660                continue;
2661            };
2662            let entry = HardEntry {
2663                path: full_name.clone(),
2664                address: *address,
2665                encoded: encoded.clone(),
2666            };
2667
2668            match self.plan(*address)? {
2669                // Kept by its bytes, exactly as a link class this writer
2670                // cannot express is: writing the link back unchanged is what
2671                // leaves the object's header where the file already has it.
2672                ObjectPlan::Preserve { why, kind } => self.out.preserved.push(PreservedEntry {
2673                    path: full_name,
2674                    class: crate::io::reader::LinkClass::Hard,
2675                    encoded: entry.encoded,
2676                    reason: Some(why),
2677                    kind,
2678                }),
2679                ObjectPlan::Dataset(parts) => {
2680                    self.out.hard.push((entry, CollectedObject::Dataset(parts)));
2681                }
2682                ObjectPlan::Group(parts) => {
2683                    self.out.hard.push((
2684                        entry,
2685                        CollectedObject::Group {
2686                            header_blocks: parts.header_blocks,
2687                            attributes: parts.attributes,
2688                            track_order: parts.track_order,
2689                            times: parts.times,
2690                            dense: parts.dense,
2691                            stab: parts.stab,
2692                        },
2693                    ));
2694                    // Recurse only into a group's header we have not entered
2695                    // before — breaks hard-link cycles.
2696                    if self.visited.insert(*address) {
2697                        self.group(&parts.links, &full_name, depth + 1)?;
2698                    }
2699                }
2700            }
2701        }
2702        Ok(())
2703    }
2704}
2705
2706/// Rebuild one reopened dataset's in-memory registry entry, storage and
2707/// all, from the header messages the walk decoded.
2708///
2709/// Fails when the chunk index the file names does not read back. The
2710/// caller answers that by preserving the object rather than registering
2711/// a dataset whose index has forgotten where its chunks are: the close
2712/// rewrites what the registry holds, so an index rebuilt from the part of
2713/// it that decoded would strand every chunk it could not read.
2714fn rebuild_dataset(
2715    handle: &mut FileHandle,
2716    meta: &FileMeta,
2717    file_size: u64,
2718    name: String,
2719    obj_addr: u64,
2720    parts: DatasetParts,
2721) -> IoResult<DatasetInfo> {
2722    let ctx = &meta.ctx;
2723    let DatasetParts {
2724        header_blocks,
2725        datatype: dt,
2726        committed_type,
2727        dataspace: ds,
2728        read_format,
2729        layout: dl,
2730        filter_pipeline: fp,
2731        fill_value,
2732        fill_write_time,
2733        attributes: attrs,
2734        track_order,
2735        times,
2736        dense: _,
2737        external,
2738    } = parts;
2739
2740    let mut info = DatasetInfo {
2741        name,
2742        datatype: dt,
2743        // The named type's own object is preserved by its bytes, so the
2744        // address the walk read the pointer from is the address it will still
2745        // be at when this header is written back.
2746        committed_type: committed_type.map(CommittedTypeRef::Preserved),
2747        read_format: Some(read_format),
2748        external,
2749        virtual_storage: None,
2750        dataspace: ds,
2751        obj_header_addr: obj_addr,
2752        data_addr: UNDEF_ADDR,
2753        data_size: 0,
2754        compact: None,
2755        chunked: None,
2756        fixed_array: None,
2757        implicit: None,
2758        single_chunk: None,
2759        btree_v1: None,
2760        btree_v2: None,
2761        append: None,
2762        attributes: attrs,
2763        obj_header_written_addr: Some(obj_addr),
2764        obj_header_blocks: header_blocks,
2765        filter_pipeline: fp,
2766        deleted: false,
2767        extent_dirty: false,
2768        header_dirty: false,
2769        // Stamped by the caller once the whole link graph is registered: it
2770        // is the count of links reaching this object, which one dataset's
2771        // parts cannot see.
2772        nlink_written: 1,
2773        // Stamped by the caller, which knows the order the walk met each
2774        // object; the rebuild sees one dataset at a time.
2775        creation_seq: 0,
2776        track_attr_order: track_order.attrs,
2777        fill_value,
2778        fill_time: fill_write_time,
2779        // Preserve the on-disk layout version so finalize re-encodes
2780        // what it read: a v5 file reopened and appended to must not be
2781        // silently downgraded to v4 (the filtered indexes keep their
2782        // 8-byte size fields, which v4 readers would mis-derive).
2783        layout_version: match &dl {
2784            DataLayoutMessage::ChunkedV4 { version, .. } => *version,
2785            // The classic index has no version above its own: a version-3
2786            // message is the whole of `H5D__chunk_set_info`'s MAX below the
2787            // version-4 gate, and re-encoding it any higher would name an
2788            // index the message cannot carry.
2789            DataLayoutMessage::ChunkedV3 { .. } => LAYOUT_VERSION_DEFAULT,
2790            _ => 4,
2791        },
2792        times,
2793    };
2794
2795    // Reconstruct storage-specific metadata
2796    debug_assert!(
2797        layout_rebuilds(&dl),
2798        "ReopenWalk::plan must preserve a layout this has no arm for"
2799    );
2800    match &dl {
2801        DataLayoutMessage::Contiguous { address, size } => {
2802            info.data_addr = *address;
2803            info.data_size = *size;
2804        }
2805        // The image is the layout message, so the rebuild carries it out of
2806        // the header it came from: anything that makes this dataset's header
2807        // stale rewrites the layout message from `compact`, and a rebuild
2808        // that left it empty would rewrite the dataset as an unallocated
2809        // contiguous one — dropping every byte.
2810        DataLayoutMessage::Compact { data } => {
2811            info.compact = Some(data.clone());
2812        }
2813        // The classic chunk index, reconstructed into the same
2814        // `BtreeV1DatasetInfo` a chunked dataset *created* in this format
2815        // gets, so the one set of machinery — `build_tree`, the flush's block
2816        // pool, `write_chunk`, `extend_dataset`, the prune a delete runs —
2817        // drives a reopened dataset and a fresh one alike. `root_addr` is what
2818        // the layout message carries and stays undefined for a dataset whose
2819        // chunks were never written, exactly as libhdf5 leaves it.
2820        DataLayoutMessage::ChunkedV3 {
2821            chunk_dims,
2822            b_tree_address,
2823        } => {
2824            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2825            let mut walk = BtreeV1Walk::new(handle, ctx, &meta.btree, &real_chunk_dims, file_size);
2826            walk.descend(*b_tree_address, 0)?;
2827            let BtreeV1Walk {
2828                records,
2829                node_addrs,
2830                ..
2831            } = walk;
2832            let max_dims = info
2833                .dataspace
2834                .max_dims
2835                .clone()
2836                .unwrap_or_else(|| info.dataspace.dims.clone());
2837            info.btree_v1 = Some(BtreeV1DatasetInfo {
2838                chunk_dims: real_chunk_dims,
2839                max_dims,
2840                // The file's own "K" ranks, not this session's defaults: they
2841                // set every node's width, so a tree bulk-loaded under the
2842                // wrong ones would re-serialize over blocks of the wrong size.
2843                config: meta.btree,
2844                records,
2845                node_addrs,
2846                root_addr: *b_tree_address,
2847                chunks_written: 0,
2848            });
2849        }
2850        DataLayoutMessage::ChunkedV4 {
2851            chunk_dims,
2852            index_address,
2853            index_type,
2854            earray_params,
2855            single_chunk_filter,
2856            ..
2857        } => {
2858            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2859
2860            if *index_type == crate::format::messages::data_layout::ChunkIndexType::ExtensibleArray
2861            {
2862                if let Some(params) = earray_params {
2863                    let ep = EarrayParams {
2864                        max_nelmts_bits: params.max_nelmts_bits,
2865                        idx_blk_elmts: params.idx_blk_elmts,
2866                        sup_blk_min_data_ptrs: params.sup_blk_min_data_ptrs,
2867                        data_blk_min_elmts: params.data_blk_min_elmts,
2868                        max_dblk_page_nelmts_bits: params.max_dblk_page_nelmts_bits,
2869                    };
2870                    let ndblk_addrs = compute_ndblk_addrs(ep.sup_blk_min_data_ptrs)?;
2871                    let nsblk_addrs = compute_nsblk_addrs(
2872                        ep.idx_blk_elmts,
2873                        ep.data_blk_min_elmts,
2874                        ep.sup_blk_min_data_ptrs,
2875                        ep.max_nelmts_bits,
2876                    )?;
2877
2878                    // Read EA header
2879                    let hdr_buf = handle.read_at_most(*index_address, 256)?;
2880                    let ea_header = ExtensibleArrayHeader::decode(&hdr_buf, ctx)?;
2881
2882                    let is_filtered = ea_header.class_id
2883                        == crate::format::chunk_index::extensible_array::EA_CLS_FILT_CHUNK;
2884                    let chunk_size_len = if is_filtered {
2885                        ea_header.raw_elmt_size - ctx.sizeof_addr - 4
2886                    } else {
2887                        0
2888                    };
2889
2890                    // Read the EA index block. Filtered datasets
2891                    // store a `FilteredIndexBlock`; unfiltered ones a
2892                    // plain `ExtensibleArrayIndexBlock`. Both must be
2893                    // reconstructed so a reopened dataset can append
2894                    // (write_chunk consults whichever applies).
2895                    let ea_iblk_addr = ea_header.idx_blk_addr;
2896                    let (ea_iblk, filt_iblk) = if is_filtered {
2897                        let placeholder = ExtensibleArrayIndexBlock::new(
2898                            *index_address,
2899                            ep.idx_blk_elmts,
2900                            ndblk_addrs,
2901                            nsblk_addrs,
2902                        );
2903                        let fib = if ea_iblk_addr != UNDEF_ADDR {
2904                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2905                            FilteredIndexBlock::decode(
2906                                &iblk_buf,
2907                                ctx,
2908                                ep.idx_blk_elmts as usize,
2909                                ndblk_addrs,
2910                                nsblk_addrs,
2911                                chunk_size_len,
2912                            )?
2913                        } else {
2914                            FilteredIndexBlock::new(
2915                                *index_address,
2916                                ep.idx_blk_elmts,
2917                                ndblk_addrs,
2918                                nsblk_addrs,
2919                            )
2920                        };
2921                        (placeholder, Some(fib))
2922                    } else {
2923                        let eib = if ea_iblk_addr != UNDEF_ADDR {
2924                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2925                            ExtensibleArrayIndexBlock::decode(
2926                                &iblk_buf,
2927                                ctx,
2928                                ep.idx_blk_elmts as usize,
2929                                ndblk_addrs,
2930                                nsblk_addrs,
2931                            )?
2932                        } else {
2933                            ExtensibleArrayIndexBlock::new(
2934                                *index_address,
2935                                ep.idx_blk_elmts,
2936                                ndblk_addrs,
2937                                nsblk_addrs,
2938                            )
2939                        };
2940                        (eib, None)
2941                    };
2942
2943                    info.chunked = Some(ChunkedDatasetInfo {
2944                        chunk_dims: real_chunk_dims,
2945                        earray_params: ep,
2946                        ea_header_addr: *index_address,
2947                        ea_iblk_addr,
2948                        ea_header,
2949                        ea_iblk,
2950                        chunks_written: 0,
2951                        filt_iblk,
2952                        chunk_size_len,
2953                    });
2954                }
2955            } else if *index_type
2956                == crate::format::messages::data_layout::ChunkIndexType::FixedArray
2957            {
2958                // Read the FA header and data block back so a
2959                // reopened dataset is writable and deletable, not
2960                // re-link only — a placeholder made a delete free
2961                // just the header and leak every chunk plus the
2962                // index. Paged data blocks (any FA with more than
2963                // dblk_page_nelmts chunks, libhdf5 default 1024)
2964                // reconstruct through the same decode owner; only
2965                // pages the bitmap marks initialized are decoded.
2966                let hdr_buf = handle.read_at_most(*index_address, 256)?;
2967                let fa_header = FixedArrayHeader::decode(&hdr_buf, ctx)?;
2968                let is_filtered = fa_header.client_id == FA_CLIENT_FILT_CHUNK;
2969                let chunk_size_len = if is_filtered {
2970                    (fa_header.element_size as usize)
2971                        .checked_sub(ctx.sizeof_addr as usize + 4)
2972                        .ok_or_else(|| {
2973                            crate::io::IoError::InvalidState(
2974                                "fixed array filtered element_size too small".into(),
2975                            )
2976                        })?
2977                } else {
2978                    0
2979                };
2980                if fa_header.data_blk_addr != UNDEF_ADDR && chunk_size_len <= 8 {
2981                    let dblk_size = fixed_array_dblk_disk_size(ctx, &fa_header) as usize;
2982                    let dblk_buf = handle.read_at_most(fa_header.data_blk_addr, dblk_size)?;
2983                    let fa_dblk =
2984                        decode_fixed_array_dblk(ctx, &fa_header, &dblk_buf, chunk_size_len)?;
2985                    info.fixed_array = Some(FixedArrayDatasetInfo {
2986                        chunk_dims: real_chunk_dims,
2987                        fa_header_addr: *index_address,
2988                        fa_dblk_addr: fa_header.data_blk_addr,
2989                        fa_header,
2990                        fa_dblk,
2991                        // Chunks written this session, matching the
2992                        // EA reconstruction above.
2993                        chunks_written: 0,
2994                    });
2995                }
2996            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::BTreeV2 {
2997                use crate::format::chunk_index::btree_v2::{
2998                    Bt2Geometry, Bt2Header, BT2_TYPE_CHUNK_FILT, BT2_TYPE_CHUNK_UNFILT,
2999                };
3000
3001                // Walk the tree back into the in-memory index and
3002                // adopt its node blocks as the flush pool. The pool
3003                // re-serializes at the header's node_size, whatever
3004                // it is — libhdf5 sizes every node from
3005                // hdr->node_size (H5B2leaf.c, H5B2internal.c) — so
3006                // a foreign size reopens too. Only a record type
3007                // that is not a chunk record, or a node size below
3008                // the bulk loader's few-records-per-node floor
3009                // (the same bound creation enforces), stays
3010                // re-link only.
3011                let hdr_buf = handle.read_at_most(*index_address, 256)?;
3012                let bt2_hdr = Bt2Header::decode(&hdr_buf, ctx)?;
3013                let ndims = real_chunk_dims.len();
3014                let is_filt = match bt2_hdr.record_type {
3015                    BT2_TYPE_CHUNK_UNFILT => Some(false),
3016                    BT2_TYPE_CHUNK_FILT => Some(true),
3017                    _ => None,
3018                };
3019                if let (Some(is_filt), true) = (
3020                    is_filt,
3021                    bt2_hdr.node_size as usize >= 10 + 3 * bt2_hdr.record_size as usize,
3022                ) {
3023                    let mut index = if is_filt {
3024                        let csl = (bt2_hdr.record_size as usize)
3025                            .checked_sub(ctx.sizeof_addr as usize + 4 + ndims * 8)
3026                            .filter(|&c| c <= 8)
3027                            .ok_or_else(|| {
3028                                crate::io::IoError::InvalidState(
3029                                    "v2 B-tree filtered record size does not fit \
3030                                     its rank and address width"
3031                                        .into(),
3032                                )
3033                            })?;
3034                        Bt2ChunkIndex::new_filtered(ndims, csl as u8)
3035                    } else {
3036                        Bt2ChunkIndex::new_unfiltered(ndims)
3037                    };
3038                    // Re-serialize with the creator's parameters:
3039                    // node blocks keep their size and the rewritten
3040                    // header keeps its declared split/merge.
3041                    index.node_size = bt2_hdr.node_size;
3042                    index.split_percent = bt2_hdr.split_percent;
3043                    index.merge_percent = bt2_hdr.merge_percent;
3044                    let mut node_addrs = Vec::new();
3045                    if bt2_hdr.root_node_addr != UNDEF_ADDR && bt2_hdr.total_num_records > 0 {
3046                        let geo = Bt2Geometry::new(
3047                            bt2_hdr.node_size,
3048                            bt2_hdr.record_size,
3049                            bt2_hdr.depth,
3050                            ctx.sizeof_addr,
3051                        );
3052                        let mut walk =
3053                            Bt2Walk::new(handle, ctx, bt2_hdr.record_size, bt2_hdr.node_size, &geo);
3054                        walk.descend(
3055                            bt2_hdr.root_node_addr,
3056                            bt2_hdr.depth,
3057                            bt2_hdr.num_records_in_root,
3058                        )?;
3059                        node_addrs = walk.node_addrs;
3060                        let record_bytes = walk.records;
3061                        let total = if bt2_hdr.record_size > 0 {
3062                            record_bytes.len() / bt2_hdr.record_size as usize
3063                        } else {
3064                            0
3065                        };
3066                        if is_filt {
3067                            for r in Bt2ChunkIndex::decode_filtered_records(
3068                                &record_bytes,
3069                                total,
3070                                ndims,
3071                                bt2_hdr.record_size,
3072                                ctx,
3073                            )? {
3074                                index.insert_filtered(
3075                                    r.scaled_offsets,
3076                                    r.chunk_address,
3077                                    r.chunk_size,
3078                                    r.filter_mask,
3079                                );
3080                            }
3081                        } else {
3082                            for r in Bt2ChunkIndex::decode_unfiltered_records(
3083                                &record_bytes,
3084                                total,
3085                                ndims,
3086                                ctx,
3087                            )? {
3088                                index.insert(r.scaled_offsets, r.chunk_address);
3089                            }
3090                        }
3091                    }
3092                    info.btree_v2 = Some(Bt2DatasetInfo {
3093                        chunk_dims: real_chunk_dims,
3094                        bt2_header_addr: *index_address,
3095                        node_addrs,
3096                        index,
3097                        chunks_written: 0,
3098                    });
3099                }
3100            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::Implicit
3101            {
3102                // Nothing to read back: the index *is* the run of chunk space
3103                // at `index_address`, and its length is the chunk grid times
3104                // the chunk size. Reconstructing that length is what lets a
3105                // delete free the storage and a write address it — a rebuild
3106                // that left this empty would rewrite the dataset as an
3107                // unallocated contiguous one, dropping every byte.
3108                let mut nchunks: u64 = 1;
3109                for g in crate::io::chunk_grid::index_grid(
3110                    &info.dataspace.dims,
3111                    info.dataspace.max_dims.as_deref(),
3112                    &real_chunk_dims,
3113                )? {
3114                    nchunks = nchunks.checked_mul(g).ok_or_else(|| {
3115                        crate::io::IoError::InvalidState("chunk count overflows u64".into())
3116                    })?;
3117                }
3118                let data_size = nchunks
3119                    .checked_mul(chunk_dims.iter().product::<u64>())
3120                    .ok_or_else(|| {
3121                        crate::io::IoError::InvalidState(
3122                            "implicit chunk storage overflows u64".into(),
3123                        )
3124                    })?;
3125                info.implicit = Some(ImplicitDatasetInfo {
3126                    chunk_dims: real_chunk_dims,
3127                    data_addr: *index_address,
3128                    data_size,
3129                });
3130            } else if *index_type
3131                == crate::format::messages::data_layout::ChunkIndexType::SingleChunk
3132            {
3133                // No index structure to read back either: the one chunk's
3134                // address, and its stored size and filter mask if the
3135                // layout's filtered flag is set, are the whole of the
3136                // layout message. `chunk_dims` already includes the
3137                // trailing element-size dimension, so its product is the
3138                // chunk's unfiltered byte length directly (see `data_size`
3139                // in the Implicit arm above).
3140                let data_size = chunk_dims.iter().product::<u64>();
3141                let (nbytes, filter_mask) = match single_chunk_filter {
3142                    Some(scf) => (scf.nbytes, scf.filter_mask),
3143                    None => (data_size, 0),
3144                };
3145                info.single_chunk = Some(SingleChunkDatasetInfo {
3146                    chunk_dims: real_chunk_dims,
3147                    data_addr: *index_address,
3148                    data_size,
3149                    nbytes,
3150                    filter_mask,
3151                    chunks_written: 0,
3152                    // Whether this was created with early allocation isn't
3153                    // recoverable here: `fill_value` above is only the
3154                    // decoded fill bytes, not the fill-value message's
3155                    // `alloc_time` byte the layout was chosen under. A
3156                    // reopened dataset that later gets a header rewrite
3157                    // therefore reports incremental allocation regardless
3158                    // of how it was actually created — the same
3159                    // imprecision a reopened `fixed_array`/`btree_v2`
3160                    // dataset already has, for the same reason.
3161                    early_alloc: false,
3162                });
3163            }
3164        }
3165        // Unreachable by `layout_rebuilds`, which is the gate
3166        // `ReopenWalk::plan` consults before it ever calls this.
3167        _ => {}
3168    }
3169
3170    Ok(info)
3171}
3172
3173/// Write `data` at *dataset-relative* byte offset `skip` into an external file
3174/// list, walking slots by cumulative declared size exactly like
3175/// `H5D__efl_write` (H5Defl.c).
3176///
3177/// Each slot's file is opened create-if-missing and never truncated, so a
3178/// write touches only the byte range that slot owns. A write past the *total*
3179/// declared size of the list is an error, matching upstream's "write past
3180/// logical end of file" check.
3181fn write_external_file_bytes(
3182    files: &[ExternalFile],
3183    extfile_prefix: Option<&Path>,
3184    mut skip: u64,
3185    data: &[u8],
3186) -> IoResult<()> {
3187    // `H5D__efl_write`'s slot walk: an `H5O_EFL_UNLIMITED` slot matches every
3188    // remaining offset (`skip >= u64::MAX` is never true), so the search stops
3189    // there and the write below takes the whole rest of the data.
3190    let mut slot_idx = 0usize;
3191    while slot_idx < files.len() && skip >= files[slot_idx].size {
3192        skip -= files[slot_idx].size;
3193        slot_idx += 1;
3194    }
3195
3196    let mut written = 0usize;
3197    while written < data.len() {
3198        let Some(slot) = files.get(slot_idx) else {
3199            return Err(crate::io::IoError::InvalidState(
3200                "write past the logical end of the external file list".into(),
3201            ));
3202        };
3203        let full_path = crate::io::reader::combine_prefixed_path(extfile_prefix, &slot.name);
3204        let ext_handle = FileHandle::open_or_create_readwrite_with_locking(
3205            &full_path,
3206            crate::io::locking::FileLocking::Disabled,
3207        )
3208        .map_err(|e| {
3209            crate::io::IoError::InvalidState(format!(
3210                "unable to open external raw data file {} for writing: {e}",
3211                full_path.display()
3212            ))
3213        })?;
3214        let this_write = (slot.size - skip).min((data.len() - written) as u64) as usize;
3215        let at = slot.offset.checked_add(skip).ok_or_else(|| {
3216            crate::io::IoError::InvalidState(format!(
3217                "external file '{}' slot offset {} overflows {skip} bytes into the slot",
3218                slot.name, slot.offset
3219            ))
3220        })?;
3221        ext_handle.write_at(at, &data[written..written + this_write])?;
3222        // This handle is dropped at the end of the iteration, and `Drop` can
3223        // only print a flush failure. Empty the accumulator here instead, so a
3224        // full disk on an external raw-data file reaches the caller.
3225        ext_handle.flush()?;
3226
3227        written += this_write;
3228        skip = 0;
3229        slot_idx += 1;
3230    }
3231    Ok(())
3232}
3233
3234/// The directory the HDF5 file at `path` sits in — libhdf5's `H5F_t::extpath`,
3235/// which `H5D__build_file_prefix` expands `${ORIGIN}` to.
3236///
3237/// Canonicalized, so the value survives the process changing directory and so
3238/// a writer and a reader of the same file agree on it. Called once per open,
3239/// never per I/O, for exactly that reason.
3240fn source_dir_of(path: &Path) -> IoResult<PathBuf> {
3241    let canonical = std::fs::canonicalize(path)?;
3242    Ok(canonical
3243        .parent()
3244        .map(Path::to_path_buf)
3245        .unwrap_or_default())
3246}
3247
3248/// Whether [`rebuild_dataset`] has an arm that reconstructs this layout.
3249///
3250/// The single list: `ReopenWalk::plan` preserves an object whose layout this
3251/// says no to, so a layout added to one side and not the other cannot happen.
3252/// Keeping two lists is what would rewrite a modelled dataset as unallocated
3253/// contiguous storage, or preserve one the writer can now build.
3254fn layout_rebuilds(layout: &DataLayoutMessage) -> bool {
3255    matches!(
3256        layout,
3257        DataLayoutMessage::Contiguous { .. }
3258            | DataLayoutMessage::Compact { .. }
3259            | DataLayoutMessage::ChunkedV3 { .. }
3260            | DataLayoutMessage::ChunkedV4 { .. }
3261    )
3262}
3263
3264/// Encode an Object Reference Count message (type 0x16) body: a version
3265/// byte (`H5O_REFCOUNT_VERSION` = 0) followed by the little-endian u32
3266/// count. Emitted on objects reached by more than one hard link.
3267fn encode_refcount(refcount: u32) -> Vec<u8> {
3268    let mut v = Vec::with_capacity(5);
3269    v.push(0u8);
3270    v.extend_from_slice(&refcount.to_le_bytes());
3271    v
3272}
3273
3274/// The symbol-table storage of every group that has one, and the single owner
3275/// of which groups those are.
3276///
3277/// A group stores its links in a symbol table because the file was *made* that
3278/// way — `H5F_LIBVER_EARLIEST` is the one bound `H5G__obj_create_real`
3279/// (H5Gobj.c:179) writes them at — or because it already had one when the file
3280/// was reopened. The second is not the first: `H5G_obj_insert` inserts into
3281/// whatever storage the group is in and converts only when a link will not fit
3282/// an entry (H5Gobj.c:512), so a symbol table survives a reopen at any bound.
3283/// A file with shared messages is where the two come apart, because its
3284/// superblock extension forces a version-2 superblock over symbol-table groups
3285/// (H5Fsuper.c:1135) and `H5F__super_read` then raises the low bound to
3286/// `H5F_LIBVER_V18` on reopen — new objects are the modern generation while the
3287/// groups already there stay symbol tables.
3288struct SymbolTables {
3289    /// The scopes the reopen found a Symbol Table message on. Fixed for the
3290    /// session: a group already in that storage stays in it, whatever bound
3291    /// the objects added beside it are written at.
3292    found: HashSet<LinkScope>,
3293    /// The symbol-table storage each group's header already names, by the
3294    /// scope whose rewrite supersedes it.
3295    ///
3296    /// INVARIANT: every entry is freed exactly once, by
3297    /// [`Hdf5Writer::prepare_symbol_tables`], which removes it as it frees.
3298    superseded: Slot<HashMap<LinkScope, StabExtents>>,
3299    /// The storage that same pass laid out, read by the header builders.
3300    ///
3301    /// INVARIANT: an entry exists here only after every block of that group's
3302    /// heap and B-tree is on disk. `build_group_header` reads it and never
3303    /// builds — a header is sized and then written by two separate calls, so a
3304    /// build that allocated would allocate twice.
3305    written: Slot<HashMap<LinkScope, Stab>>,
3306}
3307
3308impl SymbolTables {
3309    /// What a file being created starts from: no group found in a symbol table
3310    /// because none was read, and nothing on disk to free.
3311    fn none_found() -> Self {
3312        Self {
3313            found: HashSet::new(),
3314            superseded: Slot::new(HashMap::new()),
3315            written: Slot::new(HashMap::new()),
3316        }
3317    }
3318}
3319
3320/// Everything a version-0/1 (symbol-table) file carries that a version-2/3 one
3321/// does not.
3322///
3323/// Its presence *is* the generation switch — [`Hdf5Writer::message_format`]
3324/// reads nothing else: libhdf5 at `H5F_LIBVER_EARLIEST` writes a version-0/1
3325/// superblock over version-1 object headers over symbol-table groups. Which
3326/// groups are symbol tables is the separate question [`SymbolTables`] answers,
3327/// because a reopen at a newer bound keeps the ones it finds.
3328///
3329/// Two things put one here, and only two: reopening a file that already is in
3330/// that format, and creating one at that bound
3331/// ([`LegacyFile::created`]). Neither is distinguished afterwards — a file is
3332/// classic or it is not, and every encoder asks only that.
3333struct LegacyFile {
3334    /// The superblock as it was read, or as [`LegacyFile::created`] built it.
3335    /// The close re-emits it with only the end of file and the root symbol
3336    /// table entry recomputed: the "K" ranks in particular are recorded
3337    /// nowhere else, and every node width in the file is derived from them.
3338    superblock: SuperblockV0V1,
3339}
3340
3341impl LegacyFile {
3342    /// The classic-format state a file created at `H5F_LIBVER_EARLIEST`
3343    /// starts from.
3344    ///
3345    /// A new file has no symbol table on disk to free and none laid out, so
3346    /// its [`SymbolTables`] starts empty and every group it makes takes that
3347    /// storage from the bound rather than from what was found.
3348    ///
3349    /// The superblock is the one `H5F__super_init` writes at that bound: the
3350    /// library-default "K" ranks (`H5F_CRT_SYM_LEAF_DEF`,
3351    /// `HDF5_BTREE_SNODE_IK_DEF`), no free-space info and no driver info. The
3352    /// root entry's object header address and cached symbol table are stamped
3353    /// in by [`Hdf5Writer::write_superblock`] once the root group has one;
3354    /// its name offset is the empty string at the front of every local heap.
3355    ///
3356    /// Version 0, not 1: a version-1 superblock exists only to carry a
3357    /// non-default chunked-storage "K" value (H5Fsuper.c:1150), and this
3358    /// writer has no property to set one.
3359    fn created(ctx: FormatContext, base_address: u64) -> Self {
3360        let btree = BTreeV1Config::default();
3361        Self {
3362            superblock: SuperblockV0V1 {
3363                version: SUPERBLOCK_V0,
3364                sizeof_offsets: ctx.sizeof_addr,
3365                sizeof_lengths: ctx.sizeof_size,
3366                file_consistency_flags: 0,
3367                sym_leaf_k: btree.sym_leaf_k,
3368                btree_internal_k: btree.snode_internal_k,
3369                indexed_storage_k: None,
3370                base_address,
3371                superblock_extension_address: UNDEF_ADDR,
3372                end_of_file_address: 0,
3373                driver_info_address: UNDEF_ADDR,
3374                root_symbol_table_entry: SymbolTableEntry {
3375                    name_offset: 0,
3376                    obj_header_addr: UNDEF_ADDR,
3377                    cache: SymbolTableCache::Nothing,
3378                },
3379            },
3380        }
3381    }
3382}
3383
3384/// The superblock extension a reopen found, and the single owner of the one
3385/// this file's close writes back.
3386///
3387/// The extension is external truth: it is where a file records the things its
3388/// superblock has no field for — non-default v1 B-tree "K" ranks, a driver's
3389/// settings, the file space strategy and its persisted free-space managers,
3390/// and the shared object header message table. `H5F__super_ext_write_msg`
3391/// modifies one message of it and leaves the rest alone, so a close that lays
3392/// a fresh extension out from what *this writer* models drops everything it
3393/// does not — and the K ranks are not decoration: a chunked dataset's version-1
3394/// B-tree nodes are sized from `chunk_internal_k`, so a reader that has lost
3395/// the message reads the tree at the default rank and fails outright.
3396///
3397/// INVARIANT: every message of the extension read is re-emitted by
3398/// [`Hdf5Writer::write_superblock_extension`], byte for byte, except the
3399/// shared-message table — the one message naming storage this session lays out
3400/// afresh, which [`SohmState`] recomputes. Nothing else here is interpreted,
3401/// so a message this crate does not model survives exactly as a modelled one
3402/// does.
3403struct CarriedExtension {
3404    /// Every block the extension header occupied — chunk 0 and each
3405    /// continuation it named — freed once the replacement is laid out. Empty
3406    /// for a file with no extension, and for one whose extension this session
3407    /// is the first to write. A rewrite re-encodes the whole chain into one
3408    /// chunk, so freeing only the first would leave the rest as space no
3409    /// free-space manager records and no object claims.
3410    superseded: crate::io::object_header_io::HeaderBlocks,
3411    /// Every message that header held — the shared-message table,
3412    /// continuations and null padding excepted. The first two are structure
3413    /// rather than content; the third is free space.
3414    carried: Vec<crate::io::object_header_io::ExtensionMessage>,
3415    /// Where [`Hdf5Writer::write_superblock_extension`] put the replacement,
3416    /// and the only value the superblock's extension address is read from.
3417    /// `None` until that pass runs, and for a file that needs no extension.
3418    addr: Slot<Option<u64>>,
3419}
3420
3421/// What a reopen learns from a file's free-space managers, split by who owns
3422/// it: the sections go to the allocator and the rest stays with the writer.
3423struct ReopenedFreeSpace {
3424    /// `None` for a file this writer records no free space for.
3425    state: Option<Box<FileSpaceState>>,
3426    /// Every section the managers held, each tagged with the manager it came
3427    /// out of and merged only within it, address-ordered. Empty whenever
3428    /// `state` is `None`.
3429    sections: Vec<FreeBlock>,
3430}
3431
3432/// The file-space info message this session is responsible for, and the
3433/// manager blocks it supersedes.
3434///
3435/// A file whose message says `persist` records the space its own edits
3436/// released in one free-space manager per allocation type: a header block
3437/// (`FSHD`) naming a sections block (`FSSE`) that lists every free region.
3438/// Nothing else in the file says those regions are free, so a session that
3439/// rewrites the file without reading them either leaks the space it frees or
3440/// hands out space a manager still claims.
3441///
3442/// Present for a file this writer *created* with non-default file-space
3443/// properties as well, where there is nothing to read and the message is this
3444/// session's to write. `None` — the field, not this struct — is the third
3445/// case: a reopened file whose message this session must not touch, which the
3446/// carried extension re-emits byte for byte.
3447///
3448/// INVARIANT: the sections read are handed to [`FileAllocator`] and tracked
3449/// there alone, so there is one account of the file's free space and not two.
3450/// What stays here is only what the allocator has no place for: the message to
3451/// write, and the managers' own blocks, which are not free space until the
3452/// close that replaces them frees them.
3453struct FileSpaceState {
3454    /// The message, as read or as the creation options declared it. It is the
3455    /// only place the manager addresses are recorded, so the close that moves
3456    /// them rewrites this message.
3457    info: FileSpaceInfoMessage,
3458    /// The manager blocks themselves — one header, and one sections block per
3459    /// manager that had any sections. Freed by the close that lays their
3460    /// replacements out, the rule every other superseded structure follows.
3461    /// Empty for a created file, which supersedes nothing.
3462    superseded: Vec<(u64, u64)>,
3463}
3464
3465impl FileSpaceState {
3466    /// Whether this file keeps free-space managers on disk. Both strategies
3467    /// that have managers do — paged aggregation has the same managers plus a
3468    /// large one — while the two aggregator-only strategies and
3469    /// `persist: false` still carry the message with nothing to write into it.
3470    fn records_free_space(&self) -> bool {
3471        self.info.persist
3472            && matches!(
3473                self.info.strategy,
3474                FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
3475            )
3476    }
3477}
3478
3479/// One free-space manager that has been given its own two blocks, and the
3480/// sections it will write into them.
3481///
3482/// Produced by
3483/// [`settle_free_space_managers`](Hdf5Writer::settle_free_space_managers).
3484/// Both blocks are ordinary allocations out of the same [`FileAllocator`] the
3485/// rest of the file uses, because upstream's are too:
3486/// `H5FS_vfd_alloc_hdr_and_section_info_if_needed` calls `H5MF_alloc`
3487/// (H5FSsection.c:2352, 2406).
3488struct PlacedManager {
3489    /// Which of the file's managers this is; its message slot names it in the
3490    /// file-space info message.
3491    manager: FreeSpaceManager,
3492    /// Header block address.
3493    hdr_addr: u64,
3494    /// Sections block address.
3495    sect_addr: u64,
3496    /// Bytes the sections block occupies. What the header records as both
3497    /// `sect_size` and `alloc_sect_size`, so an image shorter than the block
3498    /// is padded rather than reported short.
3499    sect_size: u64,
3500    /// The sections this manager records, in serialization order. Filled on
3501    /// the settling round, once no allocation can change them.
3502    sections: Vec<FreeSection>,
3503}
3504
3505/// The manager header for `sections`, before its own blocks have addresses.
3506///
3507/// Every width the section encoding uses comes from here, and the only one
3508/// that varies with the content is `serial_sections` — it decides how many
3509/// bytes a per-size run count takes — so sizing a layout and encoding it must
3510/// go through this one function or the two disagree.
3511fn manager_header(sections: &[FreeSection]) -> FreeSpaceHeader {
3512    FreeSpaceHeader {
3513        client: free_space::CLIENT_FILE,
3514        total_space: sections.iter().map(|s| s.len).sum(),
3515        total_sections: sections.len() as u64,
3516        // Every class the file client registers is serializable; only a
3517        // fractal heap's manager has ghost sections.
3518        serial_sections: sections.len() as u64,
3519        ghost_sections: 0,
3520        nclasses: free_space::FILE_SECT_CLASSES,
3521        shrink_percent: free_space::SHRINK_PERCENT,
3522        expand_percent: free_space::EXPAND_PERCENT,
3523        max_sect_addr: free_space::SEC2_MAX_SECT_ADDR,
3524        max_sect_size: free_space::SEC2_MAXADDR,
3525        sect_addr: UNDEF_ADDR,
3526        sect_size: 0,
3527        alloc_sect_size: 0,
3528    }
3529}
3530
3531impl Default for CarriedExtension {
3532    /// What a file with no extension carries: nothing to free, nothing to
3533    /// re-emit, and no address until a shared-message table gives it one.
3534    fn default() -> Self {
3535        Self {
3536            superseded: Vec::new(),
3537            carried: Vec::new(),
3538            addr: Slot::new(None),
3539        }
3540    }
3541}
3542
3543/// Where a file's superblock version comes from — the two cases libhdf5 keeps
3544/// strictly apart, and this writer's single source for both the version it
3545/// writes back and the generation it writes new structures in.
3546///
3547/// INVARIANT: reopening a file never changes its superblock version, and every
3548/// structure appended to it is written at a library-version bound of at least
3549/// the row that version belongs to.
3550///
3551/// libhdf5 splits the same way. `H5F__super_init` is the only place a version
3552/// is *decided* — content first, then `MAX(super_vers,
3553/// HDF5_superblock_ver_bounds[low_bound])` (H5Fsuper.c:1128-1154).
3554/// `H5F__super_read` never recomputes one; it validates what it read and
3555/// raises the file's low bound to match, version 2 to at least
3556/// `H5F_LIBVER_V18` and version 3 to at least `H5F_LIBVER_V110`
3557/// (hdf5_1.14.6 H5Fsuper.c:460-466). One direction only: the version bounds
3558/// the structures, the structures never bound the version back.
3559///
3560/// Two variants rather than one number with a rule attached, because the
3561/// number means different things on the two paths — a floor to raise on the
3562/// create path, a fixed value on the reopen path — and a single field would
3563/// have every reader re-derive which.
3564#[derive(Debug, Clone, Copy)]
3565enum SuperblockVersion {
3566    /// A file this writer created. The version its creation options start
3567    /// from, which [`superblock_version_for`](Hdf5Writer::superblock_version_for)
3568    /// raises to what the content and the named bound need. Nothing is on
3569    /// disk yet, so nothing floors the bound.
3570    Chosen(u8),
3571    /// A file this writer reopened: the version already in the file. Written
3572    /// back unchanged, and the floor under every bound this session writes at.
3573    Existing(u8),
3574}
3575
3576impl SuperblockVersion {
3577    /// The oldest library-version bound this file may be written at.
3578    ///
3579    /// `H5F__super_read`'s upgrade, as a table rather than two `if`s: the
3580    /// oldest row of `HDF5_superblock_ver_bounds` (H5Fsuper.c:68) whose entry
3581    /// is the version on disk. A created file has no superblock on disk, so
3582    /// its floor is the oldest bound there is.
3583    ///
3584    /// `Existing(0..=1)` and `Hdf5Writer::legacy` say the same thing from two
3585    /// directions and cannot disagree: `open_append_with_locking` builds the
3586    /// `LegacyFile` from exactly the versions this arm covers.
3587    fn libver_floor(self) -> LibverBound {
3588        match self {
3589            Self::Chosen(_) => LibverBound::Earliest,
3590            Self::Existing(0..=1) => LibverBound::Earliest,
3591            Self::Existing(2) => LibverBound::V18,
3592            Self::Existing(_) => LibverBound::V110,
3593        }
3594    }
3595}
3596
3597/// A registry entry that has held some name.
3598///
3599/// Datasets, groups and committed datatypes keep stable indices — their
3600/// registries only grow, deletion being a flag — so the index can name the
3601/// exact entry. The link registries shrink as links are unlinked, and a
3602/// link's path is derived from its parent group's current name, so for those
3603/// the index records only that the kind once claimed the name and the (short)
3604/// list itself answers.
3605#[derive(Clone, Copy, PartialEq, Eq)]
3606enum NameHit {
3607    Dataset(usize),
3608    Group(usize),
3609    Datatype(usize),
3610    HardLink,
3611    SymbolicLink,
3612    PreservedLink,
3613}
3614
3615/// Which names the file model already holds, so creating an object does not
3616/// have to walk every registry to find out.
3617///
3618/// INVARIANT: while `map` is `Some`, every name a registry entry currently
3619/// holds has an entry in `map` covering that entry. The converse is not
3620/// required: a hit whose object was since deleted, or whose name has since
3621/// changed, stays in the map and is filtered out by
3622/// [`Hdf5Writer::name_holder`], which re-runs the very predicates the linear
3623/// scan used. The index may therefore answer "maybe", never "free" for a name
3624/// that is taken.
3625///
3626/// MUST NOT: no code may give a registry entry a name, or move the path a
3627/// link is emitted under, without either registering the new name through
3628/// [`Hdf5Writer::register_name`] or dropping the index through
3629/// [`Hdf5Writer::forget_name_index`]. State a constructor puts straight into
3630/// the registries needs neither — `map` starts `None`, and the first query
3631/// builds it from the registries as they then stand.
3632struct NameIndex {
3633    map: Option<HashMap<String, Vec<NameHit>>>,
3634    /// Bumped whenever the registries move under a build in flight, so that
3635    /// build's result is discarded instead of being installed stale.
3636    epoch: u64,
3637}
3638
3639impl NameIndex {
3640    fn new() -> Self {
3641        NameIndex {
3642            map: None,
3643            epoch: 0,
3644        }
3645    }
3646
3647    /// Record that `hit` holds `name`. With no map built there is nothing to
3648    /// record, but the registries have moved, so any build in flight is
3649    /// invalidated rather than trusted.
3650    fn insert(&mut self, name: &str, hit: NameHit) {
3651        match self.map.as_mut() {
3652            None => self.epoch += 1,
3653            Some(map) => {
3654                let hits = map.entry(name.to_string()).or_default();
3655                if !hits.contains(&hit) {
3656                    hits.push(hit);
3657                }
3658            }
3659        }
3660    }
3661
3662    /// Throw the index away: the next query rebuilds it from the registries.
3663    fn forget(&mut self) {
3664        self.map = None;
3665        self.epoch += 1;
3666    }
3667}
3668
3669/// HDF5 file writer.
3670///
3671/// Usage:
3672/// 1. `Hdf5Writer::create(path)` to create a new file.
3673/// 2. `create_dataset(name, datatype, dims)` to define datasets.
3674/// 3. `write_dataset_raw(index, data)` to write raw data.
3675/// 4. `close()` to finalize the file (writes superblock, headers, etc.).
3676pub struct Hdf5Writer {
3677    handle: FileHandle,
3678    allocator: FileAllocator,
3679    ctx: FormatContext,
3680    /// Dataset registry. The outer [`Slot`] guards the spine (push on create,
3681    /// index/clone on access) and is held only briefly; each [`DatasetRef`]
3682    /// carries one dataset's metadata behind its own lock. A writer clones
3683    /// the `DatasetRef` out (releasing this lock) before doing the long
3684    /// per-dataset work, so a create never blocks an in-flight write.
3685    pub(crate) datasets: Slot<Vec<DatasetRef>>,
3686    /// Group registry, same shape as [`Self::datasets`].
3687    pub(crate) groups: Slot<Vec<GroupRef>>,
3688    /// User-created hard links (additional names for existing objects),
3689    /// resolved and emitted during finalize.
3690    pub(crate) hard_links: Slot<Vec<HardLink>>,
3691    /// User-created soft and external links. Held apart from
3692    /// [`Self::hard_links`] because they name a path rather than an object:
3693    /// nothing resolves them, and no object's reference count counts them.
3694    pub(crate) symbolic_links: Slot<Vec<SymbolicLink>>,
3695    /// Datatypes committed this session, each an object of its own; see
3696    /// [`CommittedDatatype`].
3697    pub(crate) committed_datatypes: Slot<Vec<CommittedDatatype>>,
3698    /// Links a reopened file held that this writer cannot express, carried
3699    /// through every header rewrite by their encoded bytes. Always empty for
3700    /// a freshly created file; see [`PreservedLink`].
3701    pub(crate) preserved_links: Slot<Vec<PreservedLink>>,
3702    /// Which names the registries above already hold; see [`NameIndex`].
3703    /// Boxed so this side table costs the writer one pointer: inline, its
3704    /// map shifted every field after it and cost the attribute path ~5%.
3705    name_index: Slot<Box<NameIndex>>,
3706    /// Attributes attached to the root group (file-level attributes).
3707    pub(crate) root_attributes: Slot<Vec<crate::format::messages::attribute::AttributeEntry>>,
3708    /// Serializes object creation so name-uniqueness check and registry insert
3709    /// happen atomically.
3710    ///
3711    /// INVARIANT: no two emitted links share a full-path name. Under
3712    /// `threadsafe`, create methods run on the shared read guard, so without
3713    /// this gate two threads could both pass the duplicate-name check (which
3714    /// snapshots a registry and drops its lock) and both push, writing an
3715    /// invalid HDF5 file with two same-named links. A create holds this lock
3716    /// across its check *and* its push; the streaming write path never takes
3717    /// it, so writes to existing datasets stay fully concurrent. It is the
3718    /// outermost lock a create acquires (create_lock → spine → slot), and no
3719    /// write path takes it, so it cannot deadlock with the registry locks.
3720    pub(crate) create_lock: Slot<()>,
3721    /// The low `H5Pset_libver_bounds` bound the *caller named*, or `None`
3722    /// when none was: the oldest libhdf5 a file this writer creates must stay
3723    /// readable by. It is the one switch the version-bearing messages read —
3724    /// the datatype message version (`H5O_dtype_ver_bounds`), the data layout
3725    /// message version (`H5O_layout_ver_bounds`) and with it the chunk index,
3726    /// and the superblock floor (`HDF5_superblock_ver_bounds`).
3727    ///
3728    /// `None` is not `Some(Earliest)`. No single libhdf5 bound describes this
3729    /// crate's default file: it takes the earliest row of the datatype and
3730    /// superblock tables (version-1 datatypes, a version-2 superblock raised
3731    /// to 3 only by what the content needs) over the v1.10 chunk indexes,
3732    /// which is the `H5F_LIBVER_V110` row of the layout table. Naming a bound
3733    /// asks for one whole libhdf5 generation instead, so the two cannot share
3734    /// a field.
3735    ///
3736    /// Nothing reads this directly:
3737    /// [`session_libver`](Hdf5Writer::session_libver) is the only reader, and
3738    /// it is where the default meets the floor the file's own superblock puts
3739    /// under it (see [`SuperblockVersion`]). A default is a bound the *writer*
3740    /// picks, and on a reopened file the writer has no say — which is exactly
3741    /// the difference this field cannot express on its own.
3742    libver: Option<LibverBound>,
3743    closed: bool,
3744    /// Set once `finalize_for_swmr` has published a readable file.
3745    ///
3746    /// A SWMR reader may hold a chunk index that still points at a block this
3747    /// writer has since replaced, so from that point on a relocated chunk's
3748    /// old block is kept rather than released for reuse — the same rule as
3749    /// libhdf5's `H5D__chunk_file_alloc`, which skips `H5MF_xfree` under
3750    /// `H5F_ACC_SWMR_WRITE`.
3751    swmr_active: bool,
3752    /// Collections with free space — libhdf5's `f->shared->cwfs` list. A
3753    /// vlen insert fills these partially-filled collection blocks before
3754    /// creating a new one, so many small writes share 4096-byte blocks
3755    /// instead of each taking their own. Entries hold `(addr, block size,
3756    /// free bytes)` hints; the block on disk stays the single truth for
3757    /// contents, and only the two functions that rewrite collection blocks
3758    /// ([`insert_vlen_objects`](Self::insert_vlen_objects) and
3759    /// [`release_vlen_references`](Self::release_vlen_references)) may
3760    /// update this list. In-memory only, like the allocator's free list:
3761    /// a reopened file's free space is rediscovered as releases touch its
3762    /// collections. Capped at [`H5HG_NCWFS`] entries.
3763    cwfs: Slot<Vec<CwfsEntry>>,
3764    /// Address of the root group object header (set after first finalize).
3765    root_group_addr: Option<u64>,
3766    /// Size of the encoded root group object header (for in-place rewrites).
3767    root_group_encoded_size: usize,
3768    /// The on-disk root header block a reopen found, `(addr, len)`, so
3769    /// finalize can free the block its rewrite supersedes.
3770    superseded_root_header: crate::io::object_header_io::HeaderBlocks,
3771    /// Where this file's superblock version comes from. The single owner of
3772    /// both halves of the reopen invariant — see [`SuperblockVersion`],
3773    /// [`superblock_version_for`](Self::superblock_version_for) and
3774    /// [`libver_floor`](Self::libver_floor).
3775    superblock_version: SuperblockVersion,
3776    /// Objects whose attributes this finalize spilled to dense storage, and
3777    /// the `Attribute Info` message naming what was written for each.
3778    ///
3779    /// INVARIANT: an entry exists here only after every block of that
3780    /// object's heap and name index is on disk, and only
3781    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes) may add
3782    /// one. `emit_attributes` reads it and never builds — a header is sized
3783    /// and then written by two separate `build_*_header` calls, so a build
3784    /// that allocated would allocate twice and leave the sized-for blocks
3785    /// stranded.
3786    dense_attributes: Slot<HashMap<AttrScope, AttributeInfoMessage>>,
3787    /// Groups whose links this finalize spilled to dense storage, and the
3788    /// `Link Info` message naming what was written for each.
3789    ///
3790    /// INVARIANT: an entry exists here only after every block of that group's
3791    /// heap and name index is on disk, and only
3792    /// [`prepare_dense_links`](Self::prepare_dense_links) may add one.
3793    dense_links: Slot<HashMap<LinkScope, LinkInfoMessage>>,
3794    /// The dense storage the reopened object headers already name — the heaps
3795    /// and indices this session's rewrites and deletes supersede.
3796    ///
3797    /// `None` for a file this session created: every block such a file will
3798    /// hold was allocated here, so there is nothing on disk to supersede and
3799    /// nothing to allocate for the bookkeeping either.
3800    ///
3801    /// INVARIANT: every entry is freed exactly once, by
3802    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs)
3803    /// or [`release_superseded_dense_links`](Self::release_superseded_dense_links),
3804    /// which remove it as they free. Nothing else may remove one: an entry
3805    /// that leaves without reaching the allocator is a leaked heap, and one
3806    /// that reaches it twice hands the same blocks to two objects.
3807    superseded_dense: Slot<Option<Box<SupersededDense>>>,
3808    /// The creation-order policy in force: whether an object created from
3809    /// now on records creation order for its links and its attributes. The
3810    /// h5py `track_order` analogue; see
3811    /// [`set_track_order`](Self::set_track_order). Each object captures this
3812    /// at creation, so changing it never rewrites an object already made.
3813    track_order: TrackOrder,
3814    /// Whether an object created from now on records the times its header can
3815    /// hold — `H5Pset_obj_track_times`, whose default is on
3816    /// (`H5O_CRT_OHDR_FLAGS_DEF` is `H5O_HDR_STORE_TIMES`, H5Opkg.h:74).
3817    /// Captured by each object at creation for the same reason
3818    /// [`track_order`](Self::track_order) is: it belongs to the creation
3819    /// property list, so a later change must not rewrite an object already
3820    /// made.
3821    track_times: bool,
3822    /// The root group's own captured policy. The root is created with the
3823    /// file, so its value comes from
3824    /// [`create_with_options`](Self::create_with_options) — or, on reopen,
3825    /// from the header already on disk.
3826    root_track_order: TrackOrder,
3827    /// The root group's stored times, on the same terms as
3828    /// [`GroupInfo::times`]: whatever a reopened file's root header had, and
3829    /// `None` for a file this writer created.
3830    root_times: Option<ObjectTimes>,
3831    /// Hands out the creation sequence numbers that order a group's links.
3832    next_creation_seq: Slot<u64>,
3833    /// Object-reference elements waiting for their target's object header
3834    /// address, which only exists once finalize has placed every header.
3835    pending_object_references: Slot<Vec<PendingObjectReference>>,
3836    /// Heap-backed reference objects waiting for the same address — the
3837    /// pre-1.12 region form and every 1.12 form whose element is a blob id.
3838    pending_heap_references: Slot<Vec<PendingHeapReference>>,
3839    /// What each object-reference attribute's value *means*, so
3840    /// [`object_attributes`](Hdf5Writer::object_attributes) can say it in
3841    /// addresses every time an object header is built.
3842    attribute_references: Slot<Vec<AttributeReferenceValue>>,
3843    /// Set when this file is in the classic (version-0/1 superblock) format,
3844    /// whether it was reopened in it or created at `H5F_LIBVER_EARLIEST`.
3845    /// See [`LegacyFile`]; [`is_legacy`](Self::is_legacy) is the only reader
3846    /// of whether it is there.
3847    legacy: Option<Box<LegacyFile>>,
3848    /// Which groups keep their links in a symbol table, and the storage each
3849    /// of them has. Empty for a file whose groups all store links in messages;
3850    /// see [`SymbolTables`], which owns the question.
3851    symbol_tables: SymbolTables,
3852    /// The v1 B-tree "K" ranks every node width in this file is derived from,
3853    /// after the superblock extension has had its say. A property of the file
3854    /// rather than of its generation: a version-2 superblock records no ranks
3855    /// of its own but its extension may, and a rewrite that used the library
3856    /// defaults there would write nodes of the wrong width.
3857    /// [`btree_v1_config`](Hdf5Writer::btree_v1_config) is the only reader.
3858    btree: BTreeV1Config,
3859    /// The superblock extension this file carries, and where the replacement
3860    /// went; see [`CarriedExtension`].
3861    extension: Box<CarriedExtension>,
3862    /// The free-space managers a reopened `persist: true` file carries; see
3863    /// [`FileSpaceState`]. `None` for every other file — one with no
3864    /// file-space info message, one that does not persist, one under paged
3865    /// aggregation, and every file this session created — and those files get
3866    /// no free-space manager written either.
3867    free_space: Option<Box<FileSpaceState>>,
3868    /// The file's shared-message indexes, when it was created with any.
3869    /// `None` — the default — is a file with no shared-message table, where
3870    /// [`share_message`](Self::share_message) is the identity.
3871    sohm: Option<Box<SohmState>>,
3872    /// The directory holding this HDF5 file, resolved once when it was opened
3873    /// — libhdf5's `H5F_t::extpath`, and the same value the read side keeps.
3874    /// External raw-data file names are joined against it when
3875    /// `HDF5_EXTFILE_PREFIX` names `${ORIGIN}`, so a write and a later read of
3876    /// the same dataset must resolve a relative name identically; capturing it
3877    /// at open time rather than reading the process's current directory per
3878    /// write is what makes that hold.
3879    source_dir: PathBuf,
3880}
3881
3882/// A file's shared object header messages, from creation to the table on disk.
3883///
3884/// INVARIANT: a message body reaches the file either literally or as a pointer
3885/// to exactly one heap object, never both, and the reference count of that
3886/// object is the number of headers that hold the pointer.
3887/// [`share_message`](Hdf5Writer::share_message) is the only place a body is
3888/// offered to an index, and
3889/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) is the
3890/// only place the phase changes — so counting and substituting are two passes
3891/// over the same call site rather than two pieces of logic that must agree.
3892struct SohmState {
3893    /// The indexes the file was created with, in table order.
3894    indexes: Vec<SohmIndexSpec>,
3895    /// What `share_message` does to an eligible message right now.
3896    phase: Slot<SohmPhase>,
3897    /// Address of the master table this session laid out, once it has one.
3898    /// Also the once-only latch on the layout: a second finalize keeps the
3899    /// table the first one published, and
3900    /// [`Hdf5Writer::write_superblock_extension`] reads it to name that table
3901    /// in the extension.
3902    table_addr: Slot<Option<u64>>,
3903    /// The blocks the table a reopen found occupies — the master table and
3904    /// each index's heap and index structure — taken by the finalize that
3905    /// replaces them. Empty for a file this session created.
3906    ///
3907    /// The table is laid out whole from the whole message set, so a reopen
3908    /// replaces it rather than inserting into it, and every header holding a
3909    /// pointer into the old one is rewritten in the same finalize
3910    /// ([`Hdf5Writer::rebuilds_shared_messages`]).
3911    superseded: Slot<Vec<(u64, u64)>>,
3912}
3913
3914/// The passes `share_message` runs in, and the state between them.
3915enum SohmPhase {
3916    /// Outside a finalize: every message stays literal.
3917    Idle,
3918    /// Measuring headers, before the bodies they will hold are final. A
3919    /// shareable message answers at the width of a heap pointer over a heap
3920    /// object that does not exist yet, which is the width the one it ends up
3921    /// pointing at has: a `H5O_shared_t` in heap form is the same size
3922    /// whatever it names. Nothing this pass produces is written — it exists so
3923    /// [`allocate_object_headers`](Hdf5Writer::allocate_object_headers) can
3924    /// reserve a block for a header whose messages are shared before the
3925    /// content phase has decided which heap object each one shares.
3926    ///
3927    /// The set is [`FirstCopies`], and it is why this pass has state at all:
3928    /// a message left literal is *wider* than a pointer, so a header can only
3929    /// be measured by making the same first-copy decision the substituting
3930    /// pass will make.
3931    Predict(FirstCopies),
3932    /// Counting the bodies the file will share. Messages still go in
3933    /// literally, so nothing this pass builds is written.
3934    Collect(SohmCollector),
3935    /// Substituting. A body the collect pass never saw stays literal, which
3936    /// is a valid file: the record it would have shared simply keeps a
3937    /// reference count one higher than the pointers that reach it.
3938    Resolve {
3939        /// Heap ID per body, from the table this finalize laid out.
3940        ids: HashMap<(u8, Vec<u8>), [u8; SOHM_HEAP_ID_LEN]>,
3941        /// The first copies this pass has already handed out; see
3942        /// [`FirstCopies`].
3943        first: FirstCopies,
3944    },
3945}
3946
3947/// The bodies a pass has already left literal in the header that offered them
3948/// first (`H5SM_IN_OH`, H5SM.c:1400-1417).
3949///
3950/// INVARIANT: the three passes walk the same object headers in the same order
3951/// — [`allocate_object_headers`](Hdf5Writer::allocate_object_headers),
3952/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) and
3953/// [`write_object_headers`](Hdf5Writer::write_object_headers) each build every
3954/// dataset in `datasets` order, then every group, then the root — so "the
3955/// header that offered this body first" is the same header in all three. Each
3956/// pass keeps its own set rather than sharing one, so a pass that does not run
3957/// cannot leave a stale decision behind for the next one. A divergence would
3958/// make a header wider than the block reserved for it, which
3959/// [`check_header_size`] refuses rather than writing.
3960type FirstCopies = std::collections::HashSet<(u8, Vec<u8>)>;
3961
3962/// The object header a message is being written into — `H5SM_try_share`'s
3963/// `open_oh` argument, which is what decides whether a first copy has a header
3964/// to stay literal in at all.
3965#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3966enum ShareOwner {
3967    /// `H5SM_try_share(f, NULL, ...)`: the message belongs to no object header
3968    /// of its own. An attribute's datatype and dataspace are offered this way
3969    /// (H5Aint.c:375-377) — they live inside the attribute's body, so there is
3970    /// no header message for a record to name and the body goes to the heap on
3971    /// first use however shareable its class is.
3972    Detached,
3973    /// `H5SM_try_share(f, oh, ...)`: the message is a message of the object
3974    /// header at this address (`H5O__msg_alloc`, H5Omessage.c:1735).
3975    Header(u64),
3976}
3977
3978impl SohmState {
3979    /// A file's indexes, plus the blocks of the table they were read out of
3980    /// when the file was reopened (empty when it was created this session).
3981    fn new(indexes: Vec<SohmIndexSpec>, superseded: Vec<(u64, u64)>) -> Self {
3982        Self {
3983            indexes,
3984            phase: Slot::new(SohmPhase::Idle),
3985            table_addr: Slot::new(None),
3986            superseded: Slot::new(superseded),
3987        }
3988    }
3989
3990    /// The index that would take a `msg_type` message of `body_len` bytes,
3991    /// as `H5SM_try_share` resolves one: the first index whose type mask
3992    /// covers the class, and then only if the message reaches that index's
3993    /// minimum. A message too small for its index is not offered to another —
3994    /// `H5SM__get_index` picks by type alone and the size check comes after.
3995    fn index_for(&self, msg_type: u8, body_len: usize) -> Option<usize> {
3996        let flag = type_flag(msg_type)?;
3997        let (at, spec) = self
3998            .indexes
3999            .iter()
4000            .enumerate()
4001            .find(|(_, spec)| spec.mesg_types & flag != 0)?;
4002        (body_len as u64 >= u64::from(spec.min_mesg_size)).then_some(at)
4003    }
4004
4005    /// Whether any index takes attribute messages, which is what makes the
4006    /// file record message creation indices — `H5SM_init` sets
4007    /// `store_msg_crt_idx` on exactly this condition (H5SM.c:220).
4008    fn shares_attributes(&self) -> bool {
4009        let Some(flag) = type_flag(MSG_ATTRIBUTE) else {
4010            return false;
4011        };
4012        self.indexes.iter().any(|spec| spec.mesg_types & flag != 0)
4013    }
4014}
4015
4016/// What decides whether two offers are the same shared message: the class,
4017/// the bytes, and the messages the bytes will end up pointing at.
4018type CollectedKey = (u8, Vec<u8>, Vec<NestedShare>);
4019
4020/// The shareable message bodies of one collect pass, in first-seen order.
4021struct SohmCollector {
4022    /// Per index, its bodies with the number of headers holding each.
4023    messages: Vec<Vec<SharedMessage>>,
4024    /// Where a body sits: `(index, position in that index's messages)`, keyed
4025    /// by everything that decides what will be stored — the class, the bytes,
4026    /// and the messages the bytes will end up pointing at.
4027    seen: HashMap<CollectedKey, (usize, usize)>,
4028}
4029
4030impl SohmCollector {
4031    fn new(nindexes: usize) -> Self {
4032        Self {
4033            messages: vec![Vec::new(); nindexes],
4034            seen: HashMap::new(),
4035        }
4036    }
4037
4038    /// Count one message against `index`, adding the body the first time it
4039    /// is seen, and say whether that body is new.
4040    ///
4041    /// `ohdr` is the header this offer would leave the body literal in when it
4042    /// is the first — `None` when the class cannot be shared in an object
4043    /// header or the offer names none. It is recorded only for a first copy:
4044    /// once a body is in the heap, later offers of it are pointers whatever
4045    /// header they come from.
4046    ///
4047    /// Two bodies are the same message only if their nesting agrees as well:
4048    /// the heap IDs a nesting body will hold are still zero here, so two
4049    /// attributes that differ only in their datatype are the same bytes at
4050    /// this point and different bytes on disk.
4051    fn record(
4052        &mut self,
4053        index: usize,
4054        msg_type: u8,
4055        body: &[u8],
4056        nested: &[NestedShare],
4057        ohdr: Option<u64>,
4058    ) -> bool {
4059        let key = (msg_type, body.to_vec(), nested.to_vec());
4060        match self.seen.get(&key) {
4061            Some(&(at, pos)) => {
4062                self.messages[at][pos].ref_count += 1;
4063                false
4064            }
4065            None => {
4066                let pos = self.messages[index].len();
4067                self.messages[index].push(SharedMessage {
4068                    msg_type,
4069                    body: body.to_vec(),
4070                    nested: nested.to_vec(),
4071                    ref_count: 1,
4072                    ohdr_addr: ohdr,
4073                });
4074                self.seen.insert(key, (index, pos));
4075                true
4076            }
4077        }
4078    }
4079
4080    /// Give back the reference [`record`](Self::record) took for a body whose
4081    /// container turned out to be a copy of one already here.
4082    ///
4083    /// A body reached only through a shared container is referenced once per
4084    /// container *record*, not once per object that has one: the pointer to
4085    /// it lives in the container's heap object, which exists once however
4086    /// many headers name it. `H5O__attr_create` reaches the same count from
4087    /// the other side, by building each attribute's components shared and
4088    /// then calling `H5O__attr_delete` — which decrements exactly the
4089    /// datatype and dataspace (H5Oattr.c:568-585) — whenever the attribute it
4090    /// built was not the first copy (H5Oattribute.c:331-366).
4091    fn release(&mut self, msg_type: u8, body: &[u8]) {
4092        if let Some(&(at, pos)) = self.seen.get(&(msg_type, body.to_vec(), Vec::new())) {
4093            let count = &mut self.messages[at][pos].ref_count;
4094            *count = count.saturating_sub(1);
4095        }
4096    }
4097}
4098
4099/// The file-creation properties a brand-new file is made with.
4100///
4101/// libhdf5 splits these across the file creation and file access property
4102/// lists (`H5Pset_userblock`, `H5Pset_link_creation_order`,
4103/// `H5Pset_libver_bounds`, the locking property); what they have in common is
4104/// that they are read once, when the file is created, and cannot be changed
4105/// afterwards without rewriting it. Options that *can* change mid-session —
4106/// the bound for objects created later, the creation-order policy for later
4107/// objects — have their own setters.
4108#[derive(Debug, Clone, Copy, Default)]
4109pub struct FileCreateOptions {
4110    /// OS-level locking policy for the new file.
4111    pub locking: crate::io::locking::FileLocking,
4112    /// Creation-order policy for the root group, and the default for every
4113    /// object created afterwards; see [`Hdf5Writer::set_track_order`].
4114    pub track_order: bool,
4115    /// Time-tracking policy for the root group, and the default for every
4116    /// object created afterwards; see [`Hdf5Writer::set_track_times`].
4117    pub track_times: bool,
4118    /// The file's low library-version bound (`H5Pset_libver_bounds`'s `low`),
4119    /// or `None` when the caller named none.
4120    ///
4121    /// The distinction is not decoration. `Some(LibverBound::Earliest)` is a
4122    /// request for the format libhdf5 writes at `H5F_LIBVER_EARLIEST` — a
4123    /// version-0 superblock over symbol-table groups and version-1 object
4124    /// headers, which is what [`ObjectFormat::Legacy`] encodes. `None` keeps
4125    /// what this crate has always written for a file whose creator said
4126    /// nothing: the version-2 superblock and link-message groups of the v1.8
4127    /// format, with the earliest bound's message versions where they can
4128    /// express the content. That combination is this crate's own, not one
4129    /// libhdf5 writes, so it cannot be spelled as a bound.
4130    pub libver: Option<LibverBound>,
4131    /// Bytes reserved in front of the superblock for the application's own
4132    /// use (`H5Pset_userblock`). Zero, the default, places the superblock at
4133    /// offset 0; otherwise a power of two of at least
4134    /// [`MIN_USERBLOCK`] bytes, since a reader finds the
4135    /// superblock by doubling its search offset from there.
4136    pub userblock: u64,
4137    /// Shared object header message indexes; see [`SharedMessageConfig`].
4138    pub shared_messages: SharedMessageConfig,
4139    /// How the file manages its own space; see [`FileSpaceConfig`].
4140    pub file_space: FileSpaceConfig,
4141}
4142
4143/// The file-space handling properties a new file is created with — the three
4144/// arguments of `H5Pset_file_space_strategy` and the one of
4145/// `H5Pset_file_space_page_size`.
4146///
4147/// The four together are what `H5F__super_init` compares against the library
4148/// defaults to decide whether the file needs a file-space info message at all
4149/// (H5Fsuper.c:1092-1097), which is why the page size belongs here even though
4150/// only paged aggregation allocates by it: a file that names a page size and
4151/// nothing else still carries the message.
4152#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4153pub struct FileSpaceConfig {
4154    /// `H5F_fspace_strategy_t`.
4155    pub strategy: FileSpaceStrategy,
4156    /// Whether the free-space managers are written to the file on close.
4157    pub persist: bool,
4158    /// The smallest section a manager records; a block freed below it is
4159    /// space the file leaks rather than tracks.
4160    pub threshold: u64,
4161    /// `H5Pset_file_space_page_size`: the file-space page every allocation of
4162    /// a paged file is shaped by, and the value the message carries whatever
4163    /// the strategy.
4164    pub page_size: u64,
4165}
4166
4167impl Default for FileSpaceConfig {
4168    /// `H5F_FILE_SPACE_STRATEGY_DEF`, `H5F_FREE_SPACE_PERSIST_DEF`,
4169    /// `H5F_FREE_SPACE_THRESHOLD_DEF` and `H5F_FILE_SPACE_PAGE_SIZE_DEF`
4170    /// (H5Fprivate.h:326-336).
4171    fn default() -> Self {
4172        Self {
4173            strategy: FileSpaceStrategy::FsmAggr,
4174            persist: false,
4175            threshold: 1,
4176            page_size: DEFAULT_FILE_SPACE_PAGE_SIZE,
4177        }
4178    }
4179}
4180
4181impl FileSpaceConfig {
4182    /// The properties as `H5P__set_file_space_strategy` (H5Pfcpl.c:1176)
4183    /// stores them: `persist` and `threshold` are set only for the two
4184    /// strategies that have free-space managers to persist, and keep their
4185    /// defaults for the two that do not.
4186    pub fn new(strategy: FileSpaceStrategy, persist: bool, threshold: u64) -> Self {
4187        let uses_managers = matches!(
4188            strategy,
4189            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
4190        );
4191        Self {
4192            strategy,
4193            persist: uses_managers && persist,
4194            threshold: if uses_managers {
4195                threshold
4196            } else {
4197                Self::default().threshold
4198            },
4199            ..Self::default()
4200        }
4201    }
4202
4203    /// `H5Pset_file_space_page_size`, the fourth file-space property and the
4204    /// one libhdf5 sets on its own call.
4205    ///
4206    /// Independent of the strategy, as upstream is: the value reaches the
4207    /// file-space info message whatever the strategy is, and only paged
4208    /// aggregation allocates by it. Out-of-range sizes are refused where the
4209    /// file is created ([`validate`](Self::validate)) rather than here, so a
4210    /// builder chain stays a builder chain.
4211    pub fn with_page_size(mut self, page_size: u64) -> Self {
4212        self.page_size = page_size;
4213        self
4214    }
4215
4216    /// Whether the file has to say any of this on disk. `H5F__super_init`
4217    /// writes the file-space info message only for a file that differs from
4218    /// the library defaults in one of the four properties (H5Fsuper.c:1092),
4219    /// and raises such a file's superblock to version 2 so it has an
4220    /// extension to write it into (H5Fsuper.c:1144).
4221    pub fn is_default(&self) -> bool {
4222        *self == Self::default()
4223    }
4224
4225    /// Refuse what this writer cannot make. `H5Pset_file_space_strategy`
4226    /// itself only refuses a strategy outside the enum (H5Pfcpl.c:1223), and
4227    /// `H5Pset_file_space_page_size` a page size outside `[512, 1 GiB]`
4228    /// (H5Pfcpl.c:1389-1393) — no power of two required, only the bounds.
4229    fn validate(&self) -> IoResult<()> {
4230        if !(PAGE_SIZE_MIN..=PAGE_SIZE_MAX).contains(&self.page_size) {
4231            return Err(crate::io::IoError::InvalidState(format!(
4232                "a file-space page size is between {PAGE_SIZE_MIN} bytes and \
4233                 {PAGE_SIZE_MAX}, not {}",
4234                self.page_size
4235            )));
4236        }
4237        match self.strategy {
4238            FileSpaceStrategy::FsmAggr
4239            | FileSpaceStrategy::Aggr
4240            | FileSpaceStrategy::None
4241            | FileSpaceStrategy::Page => Ok(()),
4242            FileSpaceStrategy::Unknown(b) => Err(crate::io::IoError::InvalidState(format!(
4243                "invalid file-space strategy {b}"
4244            ))),
4245        }
4246    }
4247
4248    /// The message a created file carries, before anything is allocated:
4249    /// every manager address undefined and no end-of-allocation recorded,
4250    /// which is what `H5F__super_init` writes (H5Fsuper.c:1369-1382).
4251    fn message(&self) -> FileSpaceInfoMessage {
4252        FileSpaceInfoMessage {
4253            // `H5O_fsinfo_set_version` starts at version 1 and only ever
4254            // raises it, so a created file never carries the version-0 form
4255            // however low its version bounds are.
4256            version: 1,
4257            strategy: self.strategy,
4258            persist: self.persist,
4259            threshold: self.threshold,
4260            page_size: self.page_size,
4261            pgend_meta_thres: 0,
4262            eoa_pre_fsm_fsalloc: UNDEF_ADDR,
4263            fs_addr: vec![UNDEF_ADDR; FS_ADDR_COUNT_V1],
4264        }
4265    }
4266}
4267
4268/// The shared object header message indexes a new file is created with.
4269///
4270/// libhdf5 sets these with three calls on the file creation property list:
4271/// `H5Pset_shared_mesg_nindexes` fixes how many indexes there are,
4272/// `H5Pset_shared_mesg_index` gives each one the message types it covers and
4273/// the smallest message it will take, and `H5Pset_shared_mesg_phase_change`
4274/// sets the list/B-tree thresholds for all of them at once. The default —
4275/// no indexes — is a file with no shared-message table, which is what every
4276/// file this crate wrote before the option existed.
4277#[derive(Debug, Clone, Copy, PartialEq)]
4278pub struct SharedMessageConfig {
4279    /// Indexes in table order; only the first `count` are in use.
4280    indexes: [SohmIndexSpec; MAX_SOHM_INDEXES],
4281    /// How many indexes the caller asked for. Kept even when it is more than
4282    /// the array holds, so file creation can refuse the count the way
4283    /// `H5Pset_shared_mesg_nindexes` does rather than silently drop indexes.
4284    count: usize,
4285}
4286
4287impl Default for SharedMessageConfig {
4288    fn default() -> Self {
4289        Self {
4290            indexes: [SohmIndexSpec {
4291                mesg_types: 0,
4292                min_mesg_size: 0,
4293                list_max: DEFAULT_SOHM_LIST_MAX,
4294                btree_min: DEFAULT_SOHM_BTREE_MIN,
4295            }; MAX_SOHM_INDEXES],
4296            count: 0,
4297        }
4298    }
4299}
4300
4301impl SharedMessageConfig {
4302    /// One index per `(mesg_types, min_mesg_size)` pair — the arguments
4303    /// `H5Pset_shared_mesg_index` takes, where `mesg_types` is the bit mask
4304    /// [`type_flag`](crate::format::sohm::type_flag) builds — with the
4305    /// file-wide phase change `H5Pset_shared_mesg_phase_change` sets: above
4306    /// `list_max` an index is a v2 B-tree, below `btree_min` it is a list
4307    /// again, and `list_max == 0` makes it a B-tree from its first message.
4308    ///
4309    /// Nothing is validated here; [`Hdf5Writer::create_with_options`] refuses
4310    /// a configuration libhdf5 would refuse, so an invalid one is reported
4311    /// where the file is made rather than where the value is typed.
4312    pub fn new(indexes: &[(u16, u32)], list_max: u16, btree_min: u16) -> Self {
4313        let mut config = Self {
4314            count: indexes.len(),
4315            ..Self::default()
4316        };
4317        for (slot, &(mesg_types, min_mesg_size)) in config.indexes.iter_mut().zip(indexes) {
4318            *slot = SohmIndexSpec {
4319                mesg_types,
4320                min_mesg_size,
4321                list_max,
4322                btree_min,
4323            };
4324        }
4325        config
4326    }
4327
4328    /// The indexes in use, in table order.
4329    pub(crate) fn specs(&self) -> &[SohmIndexSpec] {
4330        &self.indexes[..self.count.min(MAX_SOHM_INDEXES)]
4331    }
4332
4333    /// Refuse a configuration `H5Pset_shared_mesg_nindexes` or
4334    /// `H5Pset_shared_mesg_phase_change` would refuse.
4335    fn validate(&self) -> IoResult<()> {
4336        if self.count > MAX_SOHM_INDEXES {
4337            return Err(crate::io::IoError::InvalidState(format!(
4338                "a file may declare at most {MAX_SOHM_INDEXES} shared-message \
4339                 indexes, not {}",
4340                self.count
4341            )));
4342        }
4343        for spec in self.specs() {
4344            // The two thresholds must not overlap, or an index would convert
4345            // back and forth on every insert.
4346            if u32::from(spec.btree_min) > u32::from(spec.list_max) + 1 {
4347                return Err(crate::io::IoError::InvalidState(format!(
4348                    "shared-message phase change needs btree_min ({}) at most one \
4349                     past list_max ({}), or an index converts on every insert",
4350                    spec.btree_min, spec.list_max
4351                )));
4352            }
4353            if spec.mesg_types == 0 {
4354                return Err(crate::io::IoError::InvalidState(
4355                    "a shared-message index covering no message type would never \
4356                     be used; give it a type mask or drop it"
4357                        .into(),
4358                ));
4359            }
4360        }
4361        Ok(())
4362    }
4363}
4364
4365/// One object-reference element written before its value could be known.
4366///
4367/// An `H5R_OBJECT1` element is the target's object header address, and
4368/// addresses are assigned during finalize, so a write records the target by
4369/// path here and [`Hdf5Writer::write_object_reference_values`] puts the address
4370/// down once every header has one.
4371pub(crate) struct PendingObjectReference {
4372    /// Dataset holding the element.
4373    dataset: usize,
4374    /// Element index within that dataset.
4375    element: u64,
4376    /// Path of the object the element names; `/` is the root group.
4377    target: String,
4378}
4379
4380/// One heap-backed reference object written before its target's address could
4381/// be known.
4382///
4383/// The *element* of a `H5R_DATASET_REGION1`, and of every 1.12 reference whose
4384/// encoding does not fit inline, is final at write time — it is the global-heap
4385/// id of the object the write inserted. What waits is the `sizeof_addr` bytes
4386/// of that heap object holding the target's object header address, which
4387/// [`Hdf5Writer::write_heap_reference_values`] stamps in.
4388pub(crate) struct PendingHeapReference {
4389    /// Address of the global-heap collection holding the object.
4390    collection: u64,
4391    /// The object's index within that collection.
4392    index: u16,
4393    /// Where the target's token sits inside that object. The pre-1.12 region
4394    /// form leads with it (`H5R__encode_token_region_compat`); every 1.12 form
4395    /// puts the token's length byte first (`H5R__encode_obj_token`).
4396    token_offset: usize,
4397    /// What the reference names, and how strictly its path must resolve.
4398    target: PendingHeapTarget,
4399}
4400
4401/// What the path of a heap-backed reference must resolve to.
4402///
4403/// The two rules `H5R` applies: a region reference names a *dataset*, since
4404/// `H5Rcreate_region` takes one dataset's dataspace and every reader
4405/// dereferences it as one, while an attribute reference names the attribute's
4406/// owner, which `H5Rcreate_attr` lets be any object.
4407#[derive(Debug, Clone)]
4408pub(crate) enum PendingHeapTarget {
4409    Dataset(String),
4410    Object(String),
4411}
4412
4413/// The value of an attribute whose elements are object references, kept as
4414/// what it means rather than as what it encodes to.
4415///
4416/// An attribute's value is part of its object header message, so it cannot be
4417/// stamped after the fact the way a dataset element can — the header is one
4418/// block, written once. What is stored instead is the paths, and
4419/// [`Hdf5Writer::object_attributes`] turns them into addresses every time the
4420/// attribute set is built: the measuring pass reads the zeros of objects that
4421/// have no address yet, the content pass reads the addresses the file will
4422/// have, and the two agree in length because an address is a fixed-width
4423/// field. The entry in the object's attribute list carries a zero image of
4424/// exactly that length and is never itself written.
4425pub(crate) struct AttributeReferenceValue {
4426    /// The object the attribute hangs on.
4427    scope: AttrScope,
4428    /// The attribute's name within that object.
4429    name: String,
4430    /// Paths of the objects the elements name, in element order; `/` is the
4431    /// root group.
4432    targets: Vec<String>,
4433}
4434
4435/// Refuse an object header body that is not the length its block was reserved
4436/// at.
4437///
4438/// The one check standing behind
4439/// [`HeaderLayout`]'s premise that measuring a header before its content is
4440/// final gives the same length as encoding it after. `what` names the object
4441/// only when the check fails, so the caller pays for the lookup only then.
4442fn check_header_size(
4443    encoded: &[u8],
4444    reserved: usize,
4445    what: impl FnOnce() -> String,
4446) -> IoResult<()> {
4447    if encoded.len() == reserved {
4448        return Ok(());
4449    }
4450    Err(crate::io::IoError::InvalidState(format!(
4451        "the object header of {} encodes to {} bytes but was measured at {}; \
4452         a message in it changed length once the addresses it names were known",
4453        what(),
4454        encoded.len(),
4455        reserved
4456    )))
4457}
4458
4459/// Where every object header a finalize writes will sit, and how long the pass
4460/// that measured it said it is.
4461///
4462/// Produced by [`Hdf5Writer::allocate_object_headers`] and consumed by
4463/// [`Hdf5Writer::write_object_headers`]; between the two, everything a header
4464/// names is built against the addresses it records. The size travels with the
4465/// address because it is what the block was reserved at: the writing pass
4466/// checks its body against it rather than trusting that the two passes agreed.
4467struct HeaderLayout {
4468    /// `(dataset index, address, measured size)`, in write order.
4469    datasets: Vec<(usize, u64, usize)>,
4470    /// `(group index, address, measured size)`, in write order.
4471    groups: Vec<(usize, u64, usize)>,
4472    /// The root group's `(address, measured size)`.
4473    root: (u64, usize),
4474}
4475
4476/// Refuse a region-reference selection the target dataset's extent does not
4477/// admit — libhdf5's `H5S_select_valid`, which `H5Rcreate` applies before it
4478/// serializes anything.
4479///
4480/// The rank check comes from [`Selection::to_boxes`], which also refuses a
4481/// regular hyperslab with an unlimited count or block; a region reference has
4482/// no growable extent to resolve one against.
4483fn validate_region_selection(selection: &Selection, dims: &[u64], path: &str) -> IoResult<()> {
4484    let boxes = selection.to_boxes(dims).map_err(|e| {
4485        crate::io::IoError::InvalidState(format!("region reference over '{path}': {e}"))
4486    })?;
4487    for (start, count) in boxes {
4488        for (d, (&s, &c)) in start.iter().zip(&count).enumerate() {
4489            if s.checked_add(c).is_none_or(|end| end > dims[d]) {
4490                return Err(crate::io::IoError::InvalidState(format!(
4491                    "region reference over '{path}' selects {s}..{} in dimension {d}, \
4492                     outside the dataset's extent of {}",
4493                    s.saturating_add(c),
4494                    dims[d]
4495                )));
4496            }
4497        }
4498    }
4499    Ok(())
4500}
4501
4502/// What a reopen found already on disk in dense form, by the scope whose
4503/// header names it.
4504///
4505/// Both halves together because they are found together — one walk of the
4506/// reopened headers fills both — and released together only in the delete
4507/// path; a finalize supersedes attribute storage before it lays object
4508/// headers out and link storage after, so each half has its own owner.
4509#[derive(Debug, Default)]
4510struct SupersededDense {
4511    attrs: HashMap<AttrScope, AttributeInfoMessage>,
4512    links: HashMap<LinkScope, LinkInfoMessage>,
4513}
4514
4515/// Which object's attribute list a prepared dense layout belongs to.
4516#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4517pub(crate) enum AttrScope {
4518    Root,
4519    Group(usize),
4520    Dataset(usize),
4521}
4522
4523/// Which group's link list a prepared dense layout belongs to.
4524#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4525pub(crate) enum LinkScope {
4526    Root,
4527    Group(usize),
4528}
4529
4530/// Attributes an object header keeps before libhdf5 spills the whole set to
4531/// dense storage (`H5O_CRT_ATTR_MAX_COMPACT_DEF`).
4532const MAX_COMPACT_ATTRS: usize = 8;
4533
4534/// Links `H5G__obj_create_real` sizes a new group's object header for
4535/// (`H5G_CRT_GINFO_EST_NUM_ENTRIES`), and the name length it assumes for each
4536/// (`H5G_CRT_GINFO_EST_NAME_LEN`). Together with the link info and group info
4537/// messages they are the whole of chunk 0 — see
4538/// [`chunk0_capacity`](Hdf5Writer::chunk0_capacity).
4539const EST_LINK_COUNT: usize = 4;
4540/// See [`EST_LINK_COUNT`].
4541const EST_LINK_NAME_LEN: usize = 8;
4542
4543/// Messages a shared-message index keeps in list form before it becomes a v2
4544/// B-tree (`H5F_CRT_SHMSG_LIST_MAX_DEF`).
4545const DEFAULT_SOHM_LIST_MAX: u16 = 50;
4546
4547/// Messages a shared-message B-tree index drops to before it reverts to a
4548/// list (`H5F_CRT_SHMSG_BTREE_MIN_DEF`).
4549const DEFAULT_SOHM_BTREE_MIN: u16 = 40;
4550
4551/// Links a group header keeps before libhdf5 spills the whole set to dense
4552/// storage (`H5G_CRT_GINFO_MAX_COMPACT`). This writer emits no phase-change
4553/// values in the Group Info message, so the default is what applies.
4554const MAX_COMPACT_LINKS: usize = 8;
4555
4556/// Bytes a compact dataset's raw image may occupy.
4557///
4558/// `H5D__compact_construct` bounds it by `H5O_MESG_MAX_SIZE` less the layout
4559/// message's own four bytes (version, class, and the 2-byte data length).
4560/// The constant it subtracts from is 65536, one past what the object header's
4561/// 2-byte message size field can express, so the ceiling here is taken from
4562/// [`MAX_MESSAGE_SIZE`] — the largest message that actually encodes — and is
4563/// one byte below libhdf5's.
4564pub const MAX_COMPACT_DATA: usize = MAX_MESSAGE_SIZE - 4;
4565
4566/// Smallest userblock a file can be created with, and the granularity of
4567/// every larger one: `H5Pset_userblock` takes 0 or a power of two from here
4568/// up, because `H5FD_locate_signature` looks for the superblock at 0 and then
4569/// at this offset doubled repeatedly.
4570pub const MIN_USERBLOCK: u64 = 512;
4571
4572impl Hdf5Writer {
4573    /// Create a new HDF5 file at `path` using the env-var-derived locking
4574    /// policy (controlled by `HDF5_USE_FILE_LOCKING`).
4575    ///
4576    /// The superblock (48 bytes for v3 with 8-byte offsets) is reserved at
4577    /// offset 0 and written during `close()`.
4578    pub fn create(path: &Path) -> IoResult<Self> {
4579        Self::create_with_locking(
4580            path,
4581            crate::io::locking::FileLocking::from_env_or(Default::default()),
4582        )
4583    }
4584
4585    /// Create a new HDF5 file at `path` with an explicit locking policy.
4586    pub fn create_with_locking(
4587        path: &Path,
4588        locking: crate::io::locking::FileLocking,
4589    ) -> IoResult<Self> {
4590        Self::create_with_options(
4591            path,
4592            FileCreateOptions {
4593                locking,
4594                ..Default::default()
4595            },
4596        )
4597    }
4598
4599    /// Create a new HDF5 file at `path` with explicit file-creation options.
4600    pub fn create_with_options(path: &Path, options: FileCreateOptions) -> IoResult<Self> {
4601        let FileCreateOptions {
4602            locking,
4603            track_order,
4604            track_times,
4605            libver,
4606            userblock,
4607            shared_messages,
4608            file_space,
4609        } = options;
4610        shared_messages.validate()?;
4611        file_space.validate()?;
4612        if userblock != 0 && (userblock < MIN_USERBLOCK || !userblock.is_power_of_two()) {
4613            return Err(crate::io::IoError::InvalidState(format!(
4614                "a userblock is {MIN_USERBLOCK} bytes or a power of two above it, \
4615                 not {userblock}: a reader locates the superblock by doubling its \
4616                 search offset from {MIN_USERBLOCK}, so no other size can hold one"
4617            )));
4618        }
4619        let policy = free_space::SpacePolicy::for_message(&file_space.message());
4620        // `H5F__super_init` (H5Fsuper.c:1182-1192) refuses a userblock that is
4621        // not a whole number of allocation units, which for a paged file is
4622        // the file-space page: everything after the userblock is addressed
4623        // from its end, so a userblock that is not a page multiple would put
4624        // every page boundary off the file's own grid.
4625        if let Some(page) = policy.page() {
4626            if userblock != 0 && userblock % page != 0 {
4627                return Err(crate::io::IoError::InvalidState(format!(
4628                    "a paged file's userblock is a multiple of its {page}-byte \
4629                     file-space page, not {userblock}"
4630                )));
4631            }
4632        }
4633        let mut handle = FileHandle::create_with_locking(path, locking)?;
4634        if userblock != 0 {
4635            // Written while the handle is still unbased, so offset 0 is the
4636            // start of the file: the block belongs to the application, not to
4637            // the HDF5 address space that begins where it ends. libhdf5 zeroes
4638            // it the same way (`H5F__super_init`), leaving a file whose first
4639            // `userblock` bytes are the application's to overwrite.
4640            handle.write_at(0, &vec![0u8; userblock as usize])?;
4641            handle.set_base(userblock);
4642        }
4643        let ctx = FormatContext::default_v3();
4644
4645        // `H5F_LIBVER_EARLIEST` is the one bound under which libhdf5 writes
4646        // the classic generation — the version-1 rows of every
4647        // message-version table, the symbol-table group form
4648        // (`H5G__obj_create_real`, H5Gobj.c:179) and the version-0 superblock
4649        // row of `HDF5_superblock_ver_bounds`.
4650        //
4651        // Shared object header messages move the last of those three and
4652        // nothing else. Their master table lives in a superblock extension,
4653        // which only a version-2 superblock has, so `H5F__super_init` raises
4654        // the superblock to version 2 whatever the low bound says
4655        // (H5Fsuper.c:1135) — but it does not touch `H5F_LOW_BOUND`, which is
4656        // what every other rule reads. So such a file is a version-2
4657        // superblock over symbol-table groups and version-1 messages, which
4658        // is what the `tests/fixtures/sohm_*.h5` files libhdf5 itself wrote
4659        // are.
4660        let classic = libver == Some(LibverBound::Earliest);
4661        let legacy = classic.then(|| Box::new(LegacyFile::created(ctx, userblock)));
4662        // Non-default file-space properties raise the superblock the same way
4663        // a shared-message table does, and for the same reason: the message
4664        // that declares them lives in an extension, and only a version-2
4665        // superblock has one (H5Fsuper.c:1144).
4666        let superblock_version = SuperblockVersion::Chosen(
4667            if classic && shared_messages.specs().is_empty() && file_space.is_default() {
4668                SUPERBLOCK_V0
4669            } else {
4670                SUPERBLOCK_V2
4671            },
4672        );
4673
4674        // Reserve the superblock at offset 0. Which version it gets is only
4675        // known once the file's content is (see `superblock_version_for`),
4676        // but the two a version-2 file can reach — 2 and 3 — encode to the
4677        // same size, so the reservation follows the base version alone.
4678        let superblock_size = match legacy.as_deref() {
4679            Some(l) if matches!(superblock_version, SuperblockVersion::Chosen(v) if v < SUPERBLOCK_V2) => {
4680                l.superblock.encoded_size()
4681            }
4682            _ => SuperblockV2V3::size_for(ctx.sizeof_addr),
4683        };
4684        // The superblock is an ordinary allocation, not a reservation: under
4685        // paged aggregation it takes the whole of page zero and leaves the
4686        // rest of that page as a section of the metadata manager, which is
4687        // what `H5F__super_init` gets from `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`
4688        // going through `H5MF__alloc_pagefs`. Unpaged it returns offset zero
4689        // and moves the end of the file to `superblock_size`, which is what
4690        // reserving it did.
4691        let allocator = FileAllocator::with_policy(0, policy);
4692        allocator.allocate(superblock_size as u64, FreeSpaceClass::Metadata);
4693
4694        Ok(Self {
4695            handle,
4696            allocator,
4697            ctx,
4698            datasets: Slot::new(Vec::new()),
4699            groups: Slot::new(Vec::new()),
4700            hard_links: Slot::new(Vec::new()),
4701            symbolic_links: Slot::new(Vec::new()),
4702            committed_datatypes: Slot::new(Vec::new()),
4703            preserved_links: Slot::new(Vec::new()),
4704            name_index: Slot::new(Box::new(NameIndex::new())),
4705            root_attributes: Slot::new(Vec::new()),
4706            create_lock: Slot::new(()),
4707            libver,
4708            closed: false,
4709            swmr_active: false,
4710            cwfs: Slot::new(Vec::new()),
4711            root_group_addr: None,
4712            root_group_encoded_size: 0,
4713            superseded_root_header: Vec::new(),
4714            // A new file starts at the oldest superblock the generation it was
4715            // created in allows, and finalize raises it if the content needs a
4716            // newer one.
4717            superblock_version,
4718            dense_attributes: Slot::new(HashMap::new()),
4719            dense_links: Slot::new(HashMap::new()),
4720            superseded_dense: Slot::new(None),
4721            track_order: TrackOrder::uniform(track_order),
4722            track_times,
4723            root_track_order: TrackOrder::uniform(track_order),
4724            // The root group is created with the file, so it captures the
4725            // policy the same instant every other field of it is settled.
4726            root_times: track_times.then(|| ObjectTimes::created_at(now_seconds())),
4727            next_creation_seq: Slot::new(0),
4728            pending_object_references: Slot::new(Vec::new()),
4729            pending_heap_references: Slot::new(Vec::new()),
4730            attribute_references: Slot::new(Vec::new()),
4731            legacy,
4732            symbol_tables: SymbolTables::none_found(),
4733            // A created file has no extension to carry and no ranks but the
4734            // library defaults: `H5Pset_sym_k`/`H5Pset_istore_k` have no
4735            // equivalent on this writer's creation path.
4736            btree: BTreeV1Config::default(),
4737            extension: Box::default(),
4738            // A file created at the library defaults declares no file-space
4739            // strategy, so it has no message to write and no manager to keep;
4740            // one created with any other properties owns both.
4741            free_space: (!file_space.is_default()).then(|| {
4742                Box::new(FileSpaceState {
4743                    info: file_space.message(),
4744                    superseded: Vec::new(),
4745                })
4746            }),
4747            sohm: (!shared_messages.specs().is_empty())
4748                .then(|| Box::new(SohmState::new(shared_messages.specs().to_vec(), Vec::new()))),
4749            source_dir: source_dir_of(path)?,
4750        })
4751    }
4752
4753    /// Target the libhdf5 2.0 file format for datasets created after this
4754    /// call: filtered chunked datasets get layout message version 5, whose
4755    /// chunk indexes store chunk sizes in a fixed `sizeof_size`-byte field
4756    /// with no overflow limit (see [`Self::chunk_layout_version`]). Off by
4757    /// default, because readers older than libhdf5 2.0 — including the
4758    /// 1.14-based h5py wheels — reject version 5.
4759    ///
4760    /// `false` names `H5F_LIBVER_EARLIEST`, the far end of the same table,
4761    /// rather than un-naming the bound: it is `set_libver_bound`'s contract
4762    /// that applies, chunk index included.
4763    pub fn set_libver_latest(&mut self, latest: bool) -> IoResult<()> {
4764        self.set_libver_bound(if latest {
4765            LibverBound::V200
4766        } else {
4767            LibverBound::Earliest
4768        })
4769    }
4770
4771    /// Bytes this file reserves in front of its superblock
4772    /// (`H5Pget_userblock`).
4773    ///
4774    /// The same value for a file created with one and for a file reopened
4775    /// through [`open_append_with_locking`](Self::open_append_with_locking),
4776    /// which takes it from where the signature turned up: it is the base of
4777    /// the handle's address space either way.
4778    pub fn userblock_size(&self) -> u64 {
4779        self.handle.base()
4780    }
4781
4782    /// Set the file's low libver bound, the equivalent of
4783    /// `H5Pset_libver_bounds`'s `low` argument. Objects created after this
4784    /// call encode their messages at the versions that bound calls for.
4785    ///
4786    /// On a reopened file the bound is raised to the row the file's superblock
4787    /// version belongs to if it names an older one, exactly as
4788    /// `H5F__super_read` raises the fapl's value — see
4789    /// [`libver_floor`](Self::libver_floor). Only a bound the file's format
4790    /// cannot express at all is refused.
4791    pub fn set_libver_bound(&mut self, libver: LibverBound) -> IoResult<()> {
4792        // A classic file cannot honour a newer bound: every encoder in it
4793        // reads `H5F_LOW_BOUND`, and raising that is what makes libhdf5 write
4794        // the version-2/3 superblock this file does not have. Refused rather
4795        // than pinned silently, so the caller learns the bound did not take.
4796        if libver != LibverBound::Earliest && self.is_legacy() {
4797            return Err(crate::io::IoError::Unsupported(format!(
4798                "cannot set the library-version bound to {libver:?} on this file: it is                  in the classic (version-0/1 superblock) format, which libhdf5 writes                  only at H5F_LIBVER_EARLIEST"
4799            )));
4800        }
4801        self.libver = Some(libver);
4802        Ok(())
4803    }
4804
4805    /// The generation the *message* encoders follow — dataspace, datatype,
4806    /// fill value, attribute.
4807    ///
4808    /// A property of the file, not of the object: `H5S__set_version`,
4809    /// `H5O__fill_set_version`, `H5A__set_version` and `H5T_set_version` all
4810    /// read `H5F_LOW_BOUND(f)` and nothing about the object they are encoding
4811    /// for. So a creation-order-tracking group in a classic file still gets
4812    /// version-1 dataspaces and version-1 attribute messages, even though its
4813    /// own header is version 2.
4814    fn message_format(&self) -> ObjectFormat {
4815        match self.legacy {
4816            Some(_) => ObjectFormat::Legacy,
4817            None => ObjectFormat::Modern,
4818        }
4819    }
4820
4821    /// The object header version an object with this creation-order policy
4822    /// gets — `H5O__set_version` (H5Oint.c:251).
4823    ///
4824    /// Version 1 is the floor a classic file's low bound sets, but tracking
4825    /// creation order of *either* kind raises the object past it: the link
4826    /// creation index lives in the message envelope and the attribute tracking
4827    /// flags live in the header prefix, and version 1 has neither. This is a
4828    /// per-object question in a classic file, which is why the format is not
4829    /// one switch for the whole file — libhdf5 writes version-2 headers inside
4830    /// a version-0 superblock whenever the creation property list asks for
4831    /// creation order.
4832    fn header_format(&self, track: TrackOrder) -> ObjectFormat {
4833        let attrs = self.header_attr_order(track.attrs);
4834        if self.legacy.is_some() && !track.links.is_tracked() && !attrs.is_tracked() {
4835            ObjectFormat::Legacy
4836        } else {
4837            ObjectFormat::Modern
4838        }
4839    }
4840
4841    /// The attribute creation-order policy an object header records, given
4842    /// what the object's creation property list asked for.
4843    ///
4844    /// A file whose shared-message configuration covers attributes records a
4845    /// creation index on every object header message: a shared attribute is
4846    /// found again through it, so `H5SM_init` sets `store_msg_crt_idx`
4847    /// (H5SM.c:220) and `H5O__create_ohdr` then raises every header it creates
4848    /// to version 2 and ORs `H5O_HDR_ATTR_CRT_ORDER_TRACKED` into its flags
4849    /// (H5Oint.c:364, H5Oint.c:442) whatever the property list says. So on
4850    /// such a file the floor is `Tracked` — this is the only place that floor
4851    /// is applied, and both the header version and the header flags come
4852    /// through here.
4853    fn header_attr_order(&self, requested: CreationOrder) -> CreationOrder {
4854        if requested.is_tracked() || !self.tracks_message_creation_index() {
4855            return requested;
4856        }
4857        CreationOrder::Tracked
4858    }
4859
4860    /// Whether this finalize replaces the file's shared-message table.
4861    ///
4862    /// It does whenever the file has indexes and no table has been published
4863    /// this session — every finalize of a file created with them, and the
4864    /// first finalize after a reopen. `build_shared_messages` lays a table out
4865    /// whole from the whole message set rather than inserting into an existing
4866    /// one, so a reopen's table is a *replacement*: every heap ID in the file
4867    /// is reassigned, which makes every object header that holds one stale
4868    /// however little else about it changed. A second finalize (a SWMR close)
4869    /// keeps the table the first published and answers `false`.
4870    fn rebuilds_shared_messages(&self) -> bool {
4871        self.sohm
4872            .as_deref()
4873            .is_some_and(|s| s.table_addr.lock().is_none())
4874    }
4875
4876    /// Whether every object header this writer emits records message creation
4877    /// indices.
4878    fn tracks_message_creation_index(&self) -> bool {
4879        self.sohm
4880            .as_deref()
4881            .is_some_and(SohmState::shares_attributes)
4882    }
4883
4884    /// Whether the group at `scope` stores its links in a symbol table —
4885    /// `H5G__obj_create_real` (H5Gobj.c:129) and the conversion
4886    /// `H5G_obj_insert` performs (H5Gobj.c:512).
4887    ///
4888    /// The new group format is used unconditionally from `H5F_LIBVER_V18` up
4889    /// *for a group being created*, and below it only when the group tracks
4890    /// link creation order: a symbol table entry has no room for a creation
4891    /// index. The two axes are independent — a group that tracks only
4892    /// *attribute* creation order gets a version-2 header over a symbol table,
4893    /// which is what libhdf5 writes for it.
4894    ///
4895    /// A group the reopen found in a symbol table is not being created, and
4896    /// `H5G_obj_insert` never moves an existing group to the new format for
4897    /// the bound's sake. So [`SymbolTables::found`] answers for it whatever
4898    /// generation the rest of this session writes at.
4899    ///
4900    /// The content of the group is the third axis. A symbol table entry has
4901    /// three cache types and no room for a fourth, so an external or
4902    /// user-defined link cannot go in one; libhdf5 answers by converting that
4903    /// one group to link messages the moment such a link is inserted, leaving
4904    /// the superblock version, the object header version and every other group
4905    /// in the file alone. This writer builds each group's storage once at
4906    /// finalize rather than link by link, so the same rule reads as a question
4907    /// about the finished set.
4908    fn uses_symbol_table(&self, scope: LinkScope, links: CreationOrder) -> bool {
4909        (self.legacy.is_some() || self.symbol_tables.found.contains(&scope))
4910            && !links.is_tracked()
4911            && self.links_fit_symbol_table(scope, links)
4912    }
4913
4914    /// Whether every link `scope` holds is one a symbol table entry can
4915    /// express — `H5G_obj_insert`'s `obj_lnk->cset != H5T_CSET_ASCII ||
4916    /// obj_lnk->type > H5L_TYPE_BUILTIN_MAX` test (H5Gobj.c:514), asked of the
4917    /// whole set.
4918    ///
4919    /// A link a reopen carried through verbatim counts too, and one this
4920    /// writer cannot even decode counts as not fitting: the entry would have
4921    /// to be built from the decoded form, while a link message is re-emitted
4922    /// byte for byte.
4923    fn links_fit_symbol_table(&self, scope: LinkScope, order: CreationOrder) -> bool {
4924        self.group_links(scope, order)
4925            .iter()
4926            .all(LinkMessage::fits_symbol_table)
4927            && self.preserved_links_for(scope).iter().all(|encoded| {
4928                LinkMessage::decode(encoded, &self.ctx)
4929                    .is_ok_and(|(link, _)| link.fits_symbol_table())
4930            })
4931    }
4932
4933    /// The header format of the registered dataset at `index`.
4934    ///
4935    /// A dataset has no links, so only the attribute half of the policy can
4936    /// raise it past version 1.
4937    fn dataset_header_format(&self, index: usize) -> ObjectFormat {
4938        let ds = self.ds(index);
4939        let attrs = ds.lock().track_attr_order;
4940        self.header_format(TrackOrder {
4941            links: CreationOrder::default(),
4942            attrs,
4943        })
4944    }
4945
4946    /// The header format of the registered group at `index`.
4947    fn group_header_format(&self, index: usize) -> ObjectFormat {
4948        let grp = self.grp(index);
4949        let track = grp.lock().track_order;
4950        self.header_format(track)
4951    }
4952
4953    /// The oldest bound this file may be written at, and the single owner of
4954    /// the reopen half of the [`SuperblockVersion`] invariant.
4955    ///
4956    /// A reopened file's superblock version is the only thing on disk that
4957    /// says which generation the file is, and `H5F__super_read` reads it as
4958    /// exactly that: it raises `H5F_LOW_BOUND` to the row that version belongs
4959    /// to (hdf5_1.14.6 H5Fsuper.c:460-466). Every version-selecting site below
4960    /// goes through [`session_libver`](Self::session_libver) rather than
4961    /// reading the `libver` field, so none of them can hand a reopened file a
4962    /// structure older than the file already claims to hold.
4963    ///
4964    /// A floor, not a ceiling. `H5Fopen` takes a fapl like `H5Fcreate` does,
4965    /// and a bound named above this one applies: libhdf5 1.14.6 writes a
4966    /// version-4 layout message into a version-2 superblock when asked at
4967    /// `H5F_LIBVER_V110`, leaving the superblock version alone. The ceiling is
4968    /// the separate question [`set_libver_bound`](Self::set_libver_bound)
4969    /// answers — no bound but `Earliest` may be named on a classic file.
4970    fn libver_floor(&self) -> LibverBound {
4971        self.superblock_version.libver_floor()
4972    }
4973
4974    /// The bound one family of encoders is written at, and the single reader
4975    /// of the `libver` field.
4976    ///
4977    /// Three inputs, in the order libhdf5 applies them. A bound the caller
4978    /// named is the fapl's `low`, raised to the floor exactly as
4979    /// `H5F__super_read` raises it. With no bound named the answer depends on
4980    /// which superblock this file has:
4981    ///
4982    /// * A file this writer created has none yet, so the writer picks —
4983    ///   `create_default`, which differs per family because this crate's
4984    ///   default file is two rows rather than one bound (see the `libver`
4985    ///   field, [`encoding_libver`] and [`layout_version_bound`]). The
4986    ///   superblock is then written to match what was picked.
4987    /// * A reopened file has already said which generation it is, and its
4988    ///   superblock cannot be rewritten to match a newer pick. So the floor is
4989    ///   the whole answer — the same value `H5F_LOW_BOUND` has after
4990    ///   `H5F__super_read` under a default fapl.
4991    ///
4992    /// [`encoding_libver`]: Self::encoding_libver
4993    /// [`layout_version_bound`]: Self::layout_version_bound
4994    fn session_libver(&self, create_default: LibverBound) -> LibverBound {
4995        let floor = self.libver_floor();
4996        let bound = match (self.libver, self.superblock_version) {
4997            (Some(named), _) => named.max(floor),
4998            (None, SuperblockVersion::Existing(_)) => floor,
4999            (None, SuperblockVersion::Chosen(_)) => create_default,
5000        };
5001        match self.message_format() {
5002            // `H5F_LIBVER_EARLIEST` is the only low bound under which libhdf5
5003            // writes a version-0/1 superblock at all, so a newer structure
5004            // inside one is a combination no libhdf5 produces. Refused where
5005            // the caller asks for it (`set_libver_bound`) rather than silently
5006            // dropped; capping here is what keeps the encoders honest if a
5007            // path ever misses that gate.
5008            ObjectFormat::Legacy => bound.min(LibverBound::Earliest),
5009            ObjectFormat::Modern => bound,
5010        }
5011    }
5012
5013    /// The bound the message encoders see — dataspace, datatype, fill value,
5014    /// attribute.
5015    fn encoding_libver(&self) -> LibverBound {
5016        self.session_libver(LibverBound::Earliest)
5017    }
5018
5019    /// The data layout message version this file's bound calls for —
5020    /// `H5O_layout_ver_bounds[H5F_LOW_BOUND(f)]` (H5Dlayout.c:44), the term
5021    /// `H5D__chunk_set_info` weighs against the version a chunk *requires*
5022    /// (H5Dchunk.c:936, :1046).
5023    ///
5024    /// With no bound named the row is `H5F_LIBVER_V110`'s: this crate's
5025    /// default file uses the v1.10 chunk indexes, which is exactly what that
5026    /// row says and what no other row does (see the `libver` field for why the
5027    /// default is not `Earliest` here even though the datatype and superblock
5028    /// tables read it that way). A file whose superblock already places it on
5029    /// an older row takes that row instead — a reopened version-2 superblock
5030    /// is the `V18` row, whose layout version of 3 has no index-type field at
5031    /// all, so its appended chunked datasets go on the version-1 B-tree.
5032    fn layout_version_bound(&self) -> u8 {
5033        self.session_libver(LibverBound::V110).layout_version()
5034    }
5035
5036    /// The data layout version a chunk of `chunk_bytes` *requires* whatever
5037    /// the bound says — `version_req` in `H5D__chunk_set_info` (H5Dchunk.c:909).
5038    ///
5039    /// Only one thing raises it: a chunk over 4 GiB does not fit the version-4
5040    /// message's 32-bit stored-size field. The floor is the default the
5041    /// creation property list carries, `H5O_LAYOUT_VERSION_DEFAULT`
5042    /// (H5Oprivate.h:451), which is why a classic file's chunked dataset is a
5043    /// version-3 message rather than the version-1 its bound's row names.
5044    fn required_chunk_layout_version(chunk_bytes: u64) -> u8 {
5045        if chunk_bytes > u32::MAX as u64 {
5046            5
5047        } else {
5048            LAYOUT_VERSION_DEFAULT
5049        }
5050    }
5051
5052    /// Whether a new chunked dataset of this chunk size is indexed by one of
5053    /// the v1.10 indexes — extensible array, fixed array, v2 B-tree, single
5054    /// chunk or implicit — rather than by the version-1 B-tree.
5055    ///
5056    /// The gate `H5D__chunk_set_info` puts in front of the whole
5057    /// index-selection block (H5Dchunk.c:936): the bound's layout version
5058    /// reaches 4, or the chunk requires a version that does. Only inside it
5059    /// does the dataspace get to pick between the five; below it the layout
5060    /// message has no index-type field and the chunks go on the version-1
5061    /// B-tree. So the format decides before the shape does — a fixed shape
5062    /// covered by exactly one chunk takes the single-chunk index only on the
5063    /// near side of this gate.
5064    pub(crate) fn uses_v110_chunk_indexing(&self, chunk_bytes: u64) -> bool {
5065        self.layout_version_bound() >= 4 || Self::required_chunk_layout_version(chunk_bytes) >= 4
5066    }
5067
5068    /// Refuse an SWMR session this file's format cannot record.
5069    ///
5070    /// The two checks `H5F__start_swmr_write` opens with: the superblock must
5071    /// be at least version 3 (H5Fint.c:3814, hdf5_1.14.6 H5Fint.c:3751) — the
5072    /// only version with the status-flags field that says a writer is attached
5073    /// — and the low bound must be at least `H5F_LIBVER_V110` (H5Fint.c:3818),
5074    /// the oldest bound whose `HDF5_superblock_ver_bounds` row reaches version
5075    /// 3.
5076    ///
5077    /// Which of the two applies is the [`SuperblockVersion`] question. A
5078    /// reopened file already has its version and reopening never rewrites one,
5079    /// so the first check decides and the second cannot fail after it: the
5080    /// version-3 floor is `V110`. A file this writer created has no version on
5081    /// disk yet, so only the second is askable — and a caller who named no
5082    /// bound at all passes it, because nothing in such a file says the
5083    /// superblock may not be version 3 and SWMR is what makes it one.
5084    ///
5085    /// Named, not silently upgraded. libhdf5 upgrades in the one case where
5086    /// SWMR is asked for at *create* time (`H5F_ACC_SWMR_WRITE` raises the
5087    /// bound to V110 in `H5F__super_init`, H5Fsuper.c:1131); on the reopen
5088    /// path it refuses instead, and so does this.
5089    fn reject_swmr(&self) -> IoResult<()> {
5090        let why = match self.superblock_version {
5091            SuperblockVersion::Existing(version) if version >= SUPERBLOCK_V3 => return Ok(()),
5092            SuperblockVersion::Existing(version) => format!(
5093                "its superblock is version {version}, and reopening a file never \
5094                 rewrites that"
5095            ),
5096            SuperblockVersion::Chosen(_) if self.is_legacy() => {
5097                "it is in the classic (version-0/1 superblock) format that \
5098                 H5F_LIBVER_EARLIEST selects"
5099                    .to_string()
5100            }
5101            SuperblockVersion::Chosen(_) if self.libver.is_some_and(|b| b < LibverBound::V110) => {
5102                "it was asked for at a library-version bound below H5F_LIBVER_V110, \
5103                 whose superblock row is version 2"
5104                    .to_string()
5105            }
5106            SuperblockVersion::Chosen(_) => return Ok(()),
5107        };
5108        Err(crate::io::IoError::Unsupported(format!(
5109            "cannot start an SWMR session on this file: {why}, and SWMR needs a \
5110             version-3 superblock to record that a writer is attached; create the \
5111             file at H5F_LIBVER_V110 or newer"
5112        )))
5113    }
5114
5115    /// Whether this file is in the classic (version-0/1 superblock) format,
5116    /// whose groups store their links in symbol tables — either because it
5117    /// was reopened in it or because it was created at
5118    /// `H5F_LIBVER_EARLIEST`.
5119    pub(crate) fn is_legacy(&self) -> bool {
5120        self.legacy.is_some()
5121    }
5122
5123    /// The v1-B-tree "K" ranks in force for this file, from which every v1
5124    /// node's width is derived.
5125    ///
5126    /// A version-0/1 superblock records them in a field of its own and a
5127    /// version-2/3 one in a B-tree-K message in its superblock extension, so
5128    /// the file's generation says nothing about whether they are the defaults
5129    /// — `H5F__super_read` reads both into the same `H5F_shared_t`, and so
5130    /// does the reopen, into `btree`.
5131    fn btree_v1_config(&self) -> BTreeV1Config {
5132        self.btree
5133    }
5134
5135    /// Track and index creation order for the links and the attributes of
5136    /// every object created after this call — the equivalent of setting
5137    /// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` to
5138    /// `H5P_CRT_ORDER_TRACKED | H5P_CRT_ORDER_INDEXED` on the creation
5139    /// property lists those objects are made with.
5140    ///
5141    /// Objects already created keep the policy they were made under, exactly
5142    /// as libhdf5 keeps what their creation property list said. The root
5143    /// group is created with the file, so its policy comes from
5144    /// [`create_with_options`](Self::create_with_options) instead.
5145    pub fn set_track_order(&mut self, track: bool) {
5146        self.track_order = TrackOrder::uniform(track);
5147    }
5148
5149    /// Record the times of every object created after this call —
5150    /// `H5Pset_obj_track_times` on the creation property lists those objects
5151    /// are made with.
5152    ///
5153    /// Off by default, which is h5py's default and not libhdf5's: h5py's
5154    /// high-level API sets `track_times=False` on every object it makes
5155    /// (`_hl/files.py:189`, `_hl/dataset.py:39`, `_hl/group.py:42`), while a
5156    /// bare creation property list leaves it on (`H5O_CRT_OHDR_FLAGS_DEF` is
5157    /// `H5O_HDR_STORE_TIMES`, H5Opkg.h:74). A caller after libhdf5's own
5158    /// bytes turns it on here.
5159    ///
5160    /// Objects already created keep the policy they were made under, and the
5161    /// root group takes its own from
5162    /// [`create_with_options`](Self::create_with_options) — the same split
5163    /// [`set_track_order`](Self::set_track_order) has, and for the same
5164    /// reason: this is a creation property, not a file-wide setting.
5165    pub fn set_track_times(&mut self, track: bool) {
5166        self.track_times = track;
5167    }
5168
5169    /// The times an object created right now records — all four set to the
5170    /// current time, as `H5O_apply_ohdr` initialises them (H5Oint.c:411-414),
5171    /// or `None` when this session is not tracking times.
5172    ///
5173    /// INVARIANT: every object this writer registers takes its `times` from
5174    /// here. The policy belongs to the creation property list, so reading
5175    /// [`track_times`](Self::track_times) at any later moment — a finalize, a
5176    /// header rewrite — would stamp a policy the object was not made under.
5177    fn created_object_times(&self) -> Option<ObjectTimes> {
5178        self.track_times
5179            .then(|| ObjectTimes::created_at(now_seconds()))
5180    }
5181
5182    /// Layout message version for a new chunked dataset on one of the v1.10
5183    /// indexes — `H5D__chunk_set_info`'s closing
5184    /// `MAX3(layout->version, version_req, MIN(bound, version_perf))`
5185    /// (H5Dchunk.c:1046).
5186    ///
5187    /// Version 5 is *required* for a chunk over 4 GiB (pre-2.0 readers cannot
5188    /// handle one even though the v4 wire format could express it) and
5189    /// *preferred* for filtered chunks, which is why it takes the file's
5190    /// bound to get there: the preference is capped by the bound's own row,
5191    /// so only the 2.0 format lets it through. Everything else stays at
5192    /// version 4, which every 1.10+ reader accepts.
5193    fn chunk_layout_version(&self, filtered: bool, chunk_bytes: u64) -> u8 {
5194        // `version_perf`: 4 for the v1.10 indexes as such, 5 when a filter
5195        // can make a chunk expand past what version 4 can record.
5196        let preferred = if filtered { 5 } else { 4 };
5197        Self::required_chunk_layout_version(chunk_bytes)
5198            .max(self.layout_version_bound().min(preferred))
5199            .max(LAYOUT_VERSION_DEFAULT)
5200    }
5201
5202    /// Width of the stored-chunk-size field in a filtered chunk index:
5203    /// version 5 uses the fixed `sizeof_size`; version 4 derives it from the
5204    /// uncompressed chunk byte count (one spare byte included), the
5205    /// `H5D_*_COMPUTE_CHUNK_SIZE_LEN` rule shared by the extensible-array,
5206    /// fixed-array and v2-B-tree indexes.
5207    fn chunk_size_len_for(&self, layout_version: u8, chunk_bytes: u64) -> u8 {
5208        if layout_version >= 5 {
5209            self.ctx.sizeof_size
5210        } else {
5211            compute_chunk_size_len(chunk_bytes)
5212        }
5213    }
5214
5215    /// Provide public access to the format context.
5216    pub fn ctx(&self) -> &FormatContext {
5217        &self.ctx
5218    }
5219
5220    /// Number of dataset slots in the registry (including soft-deleted ones).
5221    pub(crate) fn dataset_count(&self) -> usize {
5222        self.datasets.lock().len()
5223    }
5224
5225    /// Clone out the [`DatasetRef`] for `index`, releasing the registry lock
5226    /// immediately. Lock the returned ref to read or mutate that one dataset.
5227    ///
5228    /// Panics on an out-of-range index, exactly like the `Vec` indexing it
5229    /// replaces; bounds-checking callers consult [`Self::dataset_count`] first.
5230    ///
5231    /// MUST NOT be called while the registry [`Slot`] is already locked (it
5232    /// would deadlock the `threadsafe` mutex / panic the single-thread
5233    /// `RefCell`): collect the refs you need, drop the registry guard, then work.
5234    pub(crate) fn ds(&self, index: usize) -> DatasetRef {
5235        Shared::clone(&self.datasets.lock()[index])
5236    }
5237
5238    /// Number of group slots in the registry (including soft-deleted ones).
5239    pub(crate) fn group_count(&self) -> usize {
5240        self.groups.lock().len()
5241    }
5242
5243    /// Clone out the [`GroupRef`] for `index`. Same contract as [`Self::ds`].
5244    pub(crate) fn grp(&self, index: usize) -> GroupRef {
5245        Shared::clone(&self.groups.lock()[index])
5246    }
5247
5248    /// Enter the create gate: take `create_lock` and check that `name` is not
5249    /// already taken. The returned witness is what [`Self::push_dataset`]
5250    /// requires, so the uniqueness check and the registry push are atomic
5251    /// (see `create_lock`) at every creator by construction.
5252    pub(crate) fn begin_create(&self, name: &str) -> IoResult<CreateGuard<'_>> {
5253        let gate = self.create_lock.lock();
5254        // A creation path through hard links lands in the link's target
5255        // group, as HDF5 traversal does. Canonicalizing here — the one
5256        // entry every creator passes — keeps alias forms out of the
5257        // registry.
5258        let name = self.canonical_dataset_path(name);
5259        // A path that leaves this file, or that runs into an object the
5260        // reopen kept verbatim, is refused here rather than at each creator:
5261        // this is the one gate every creation passes, so a creator added
5262        // later cannot forget the check. Both run before the parent lookup,
5263        // which would otherwise report the group such a path names as absent
5264        // instead of naming what stops the path. Uniqueness comes first among
5265        // them: a name already in the file is taken whatever holds it.
5266        self.reject_external_traversal(&name)?;
5267        self.ensure_name_free(&name)?;
5268        self.reject_preserved_object(&name)?;
5269        let (parent, _leaf) = self.split_parent(&name)?;
5270        Ok(CreateGuard {
5271            _gate: gate,
5272            name,
5273            parent,
5274        })
5275    }
5276
5277    /// Split an object path into the group that will hold its link and the
5278    /// leaf link name, resolving every component through the group registry.
5279    ///
5280    /// `path` is the registry form — no leading `/`, e.g. `"grp/sub/late"`.
5281    /// This is what keeps a `/` out of a link name: HDF5 link names are
5282    /// single path components (`H5G_traverse` splits on `/` before it ever
5283    /// reaches `H5L_link`), so a name that carries a path must name a group
5284    /// that exists, or be refused.
5285    ///
5286    /// A missing component is an error rather than an implicit group: the
5287    /// default link creation property list has `H5Pset_create_intermediate_group`
5288    /// off, and this writer exposes no property list to turn it on with.
5289    fn split_parent(&self, path: &str) -> IoResult<(Option<usize>, String)> {
5290        let (parent_path, leaf) = path.rsplit_once('/').unwrap_or(("", path));
5291        if leaf.is_empty() {
5292            return Err(crate::io::IoError::InvalidState(format!(
5293                "'{path}' does not end in a link name"
5294            )));
5295        }
5296        if parent_path.is_empty() {
5297            return Ok((None, leaf.to_string()));
5298        }
5299        let abs = format!("/{parent_path}");
5300        let groups = self.group_refs();
5301        let idx = groups
5302            .iter()
5303            .position(|g| {
5304                let gg = g.lock();
5305                gg.name == abs && !gg.deleted
5306            })
5307            .ok_or_else(|| {
5308                crate::io::IoError::NotFound(format!(
5309                    "cannot create '{path}': group '{abs}' does not exist"
5310                ))
5311            })?;
5312        Ok((Some(idx), leaf.to_string()))
5313    }
5314
5315    /// Push a freshly-built dataset into the registry and return its index.
5316    /// Takes the registry lock only for the push, so it does not block an
5317    /// in-flight write that already cloned its own [`DatasetRef`] out.
5318    /// The [`CreateGuard`] proves the caller entered through
5319    /// [`Self::begin_create`] and still holds the gate.
5320    pub(crate) fn push_dataset(&self, create: &CreateGuard<'_>, info: DatasetInfo) -> usize {
5321        let name = info.name.clone();
5322        let idx = {
5323            let mut reg = self.datasets.lock();
5324            let idx = reg.len();
5325            reg.push(Shared::new(DatasetCell::new(info)));
5326            idx
5327        };
5328        self.register_name(&name, NameHit::Dataset(idx));
5329        // The spine guard is dropped before the group slot is taken: the lock
5330        // order is spine -> slot and never the reverse.
5331        if let Some(pidx) = create.parent {
5332            self.grp(pidx).lock().child_datasets.push(idx);
5333        }
5334        idx
5335    }
5336
5337    /// Push a freshly-built group into the registry and return its index.
5338    pub(crate) fn push_group(&self, info: GroupInfo) -> usize {
5339        let name = info.name.trim_start_matches('/').to_string();
5340        let idx = {
5341            let mut reg = self.groups.lock();
5342            let idx = reg.len();
5343            reg.push(Shared::new(Slot::new(info)));
5344            idx
5345        };
5346        self.register_name(&name, NameHit::Group(idx));
5347        idx
5348    }
5349
5350    /// Snapshot every [`DatasetRef`] (spine lock held only for the clone).
5351    /// Iterate the snapshot to lock each dataset one at a time — this keeps
5352    /// the lock order *spine → slot* and never reacquires the spine while a
5353    /// slot is held, which is what makes the registry deadlock-free.
5354    pub(crate) fn dataset_refs(&self) -> Vec<DatasetRef> {
5355        self.datasets.lock().iter().map(Shared::clone).collect()
5356    }
5357
5358    /// Snapshot every [`GroupRef`]; see [`Self::dataset_refs`].
5359    pub(crate) fn group_refs(&self) -> Vec<GroupRef> {
5360        self.groups.lock().iter().map(Shared::clone).collect()
5361    }
5362
5363    /// Snapshot the hard-link list (the lock is held only for the clone), so
5364    /// callers can resolve each link's target/parent — which locks dataset and
5365    /// group slots — without holding the hard-link lock.
5366    /// The next creation sequence number.
5367    ///
5368    /// One monotonic counter for datasets, groups and hard links alike: a
5369    /// group orders its links by it, so an interleaved run of `create_group`
5370    /// and `create_dataset` comes back out in the order it was made rather
5371    /// than grouped by kind.
5372    fn take_creation_seq(&self) -> u64 {
5373        let mut next = self.next_creation_seq.lock();
5374        let seq = *next;
5375        *next += 1;
5376        seq
5377    }
5378
5379    pub(crate) fn hard_links_vec(&self) -> Vec<HardLink> {
5380        self.hard_links.lock().clone()
5381    }
5382
5383    /// Snapshot the symbolic-link list; see [`Self::hard_links_vec`].
5384    pub(crate) fn symbolic_links_vec(&self) -> Vec<SymbolicLink> {
5385        self.symbolic_links.lock().clone()
5386    }
5387
5388    /// Open an existing HDF5 file for appending new datasets, using the
5389    /// env-var-derived locking policy.
5390    ///
5391    /// Reads existing dataset object headers fully, reconstructing metadata
5392    /// for chunked datasets so that `write_chunk` and `extend_dataset` work
5393    /// on reopened datasets.
5394    pub fn open_append(path: &Path) -> IoResult<Self> {
5395        Self::open_append_with_locking(
5396            path,
5397            crate::io::locking::FileLocking::from_env_or(Default::default()),
5398        )
5399    }
5400
5401    /// Carry a reopened file's shared-message table into the writer's model:
5402    /// the index specifications the file was created with, and every block the
5403    /// table occupies so the finalize that replaces it can give them back.
5404    ///
5405    /// `H5SM_init` fixes the index count, each index's type mask, its minimum
5406    /// message size and the file-wide phase-change pair when the file is
5407    /// created, and nothing afterwards changes any of them — they are file
5408    /// creation properties. So the master table on disk *is* the
5409    /// [`SharedMessageConfig`] the file was made with, read back.
5410    ///
5411    /// Returns `None` for a file with no shared-message table, which is every
5412    /// file libhdf5 writes without `H5Pset_shared_mesg_nindexes`.
5413    /// Read the free-space managers a reopened file persists, if it does.
5414    ///
5415    /// `H5F__super_read` copies the file-space info message's addresses into
5416    /// `f->shared->fs_addr[]` and the library opens each manager lazily; this
5417    /// reads them all at once, because the writer needs the whole section set
5418    /// before it allocates anything.
5419    ///
5420    /// Returns `None` — nothing read, nothing to write back — for a file with
5421    /// no file-space info message, one that does not persist, and one whose
5422    /// strategy keeps no managers at all.
5423    fn reopen_free_space(
5424        handle: &mut FileHandle,
5425        meta: &crate::io::FileMeta,
5426        ext: &crate::io::reader::SuperblockExtension,
5427    ) -> IoResult<ReopenedFreeSpace> {
5428        let none = || ReopenedFreeSpace {
5429            state: None,
5430            sections: Vec::new(),
5431        };
5432        let Some(info) = ext.file_space_info.as_ref().filter(|i| i.persist) else {
5433            return Ok(none());
5434        };
5435        if !matches!(
5436            info.strategy,
5437            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
5438        ) {
5439            return Ok(none());
5440        }
5441        let found = crate::io::free_space_io::read_managers(handle, &meta.ctx, info)?;
5442        Ok(ReopenedFreeSpace {
5443            state: Some(Box::new(FileSpaceState {
5444                info: info.clone(),
5445                superseded: found.blocks,
5446            })),
5447            sections: found.sections,
5448        })
5449    }
5450
5451    fn reopen_shared_messages(
5452        handle: &mut FileHandle,
5453        meta: &crate::io::FileMeta,
5454        ext: &crate::io::reader::SuperblockExtension,
5455    ) -> IoResult<Option<Box<SohmState>>> {
5456        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
5457        use crate::format::fractal_heap::collect_heap_extents;
5458        use crate::format::sohm::{list_size, SohmMasterTable, SOHM_INDEX_LIST};
5459
5460        let (Some(table), Some(smt)) = (
5461            meta.sohm.as_ref().filter(|t| !t.indexes.is_empty()),
5462            ext.shared_message_table.as_ref(),
5463        ) else {
5464            return Ok(None);
5465        };
5466        let ctx = &meta.ctx;
5467
5468        // The extension header itself is superseded by `CarriedExtension`,
5469        // which owns it whether or not the file has shared messages; what is
5470        // superseded here is only the storage the table message names.
5471        let mut superseded = Vec::new();
5472        superseded.push((
5473            smt.table_address,
5474            SohmMasterTable::encoded_size(ctx, smt.nindexes) as u64,
5475        ));
5476
5477        let mut specs = Vec::with_capacity(table.indexes.len());
5478        for index in &table.indexes {
5479            specs.push(SohmIndexSpec {
5480                mesg_types: index.mesg_types,
5481                min_mesg_size: index.min_mesg_size,
5482                list_max: index.list_max,
5483                btree_min: index.btree_min,
5484            });
5485            let mut reader = crate::io::reader::HandleBlockReader { handle };
5486            if index.heap_addr != UNDEF_ADDR {
5487                superseded.extend(collect_heap_extents(index.heap_addr, ctx, &mut reader)?);
5488            }
5489            if index.index_addr != UNDEF_ADDR {
5490                if index.index_type == SOHM_INDEX_LIST {
5491                    // `H5SM_LIST_SIZE`: the block is sized for `list_max`
5492                    // records however few are in it.
5493                    superseded.push((index.index_addr, list_size(ctx, index.list_max) as u64));
5494                } else {
5495                    superseded.extend(collect_btree_v2_extents(
5496                        index.index_addr,
5497                        ctx,
5498                        &mut reader,
5499                    )?);
5500                }
5501            }
5502        }
5503        Ok(Some(Box::new(SohmState::new(specs, superseded))))
5504    }
5505
5506    /// Open an existing HDF5 file for appending with an explicit locking
5507    /// policy.
5508    pub fn open_append_with_locking(
5509        path: &Path,
5510        locking: crate::io::locking::FileLocking,
5511    ) -> IoResult<Self> {
5512        let mut handle = FileHandle::open_readwrite_with_locking(path, locking)?;
5513        // The same `H5FD_locate_signature` search the read path makes, through
5514        // the same handle mechanism: the offset it finds is the file's base
5515        // address, so the allocator's end-of-file, every write and the
5516        // superblock rewrite all work in the HDF5 address space, and the
5517        // userblock in `[0, base)` is not addressable from this writer at all.
5518        let super_addr = handle
5519            .locate_signature()?
5520            .ok_or(crate::format::FormatError::InvalidSignature)?;
5521        handle.set_base(super_addr);
5522        let file_size = handle.file_size()?;
5523
5524        let sb_buf = handle.read_at_most(0, 256)?;
5525        // Which generation the file is decides everything the close then
5526        // writes back: version-1 object headers and symbol-table groups over a
5527        // version-0/1 superblock, or version-2 headers and link-message groups
5528        // over a version-2/3 one. libhdf5 writes those two combinations and no
5529        // mixture of them, so the branch is taken once, here, and carried as
5530        // `legacy`.
5531        let version = crate::format::superblock::detect_superblock_version(&sb_buf)?;
5532        let (ctx, sb_btree, root_addr, ext_addr, legacy) = if version <= 1 {
5533            let sb = SuperblockV0V1::decode(&sb_buf)?;
5534            let ctx = FormatContext {
5535                sizeof_addr: sb.sizeof_offsets,
5536                sizeof_size: sb.sizeof_lengths,
5537            };
5538            // Unlike a v2/v3 superblock, a classic one carries the "K" ranks
5539            // itself; every v1-B-tree and symbol-table node width in the file
5540            // comes from them.
5541            let btree = crate::format::btree_v1::BTreeV1Config {
5542                sym_leaf_k: sb.sym_leaf_k,
5543                snode_internal_k: sb.btree_internal_k,
5544                chunk_internal_k: sb.indexed_storage_k.unwrap_or(32),
5545            };
5546            let root = sb.root_symbol_table_entry.obj_header_addr;
5547            let ext = sb.superblock_extension_address;
5548            (ctx, btree, root, ext, Some(sb))
5549        } else {
5550            let sb = SuperblockV2V3::decode(&sb_buf)?;
5551            let ctx = FormatContext {
5552                sizeof_addr: sb.sizeof_offsets,
5553                sizeof_size: sb.sizeof_lengths,
5554            };
5555            (
5556                ctx,
5557                crate::format::btree_v1::BTreeV1Config::default(),
5558                sb.root_group_object_header_address,
5559                sb.superblock_extension_address,
5560                None,
5561            )
5562        };
5563
5564        // The reopen reads object headers exactly as the reader does, so it
5565        // needs the same file-level parameters: a v2/v3 superblock carries no
5566        // B-tree K values, and only the extension can override the defaults.
5567        let (meta, ext) = crate::io::reader::Hdf5Reader::read_extension_and_meta(
5568            &mut handle,
5569            ctx,
5570            sb_btree,
5571            ext_addr,
5572        )?;
5573
5574        // A file with shared object header messages keeps datatypes,
5575        // dataspaces and attributes in a fractal heap per index, and each
5576        // object header holds a heap ID pointing at one. The table is laid out
5577        // whole from the whole message set (`build_shared_messages`), never
5578        // grown insert by insert, so a reopen carries the indexes and the
5579        // bodies forward and the next finalize lays a new table out over the
5580        // old one's blocks — which is sound exactly while no header keeping
5581        // its bytes still points into the old heap. The walk below is what
5582        // settles that.
5583        let sohm = Self::reopen_shared_messages(&mut handle, &meta, &ext)?;
5584
5585        // The extension is external truth this close rewrites, so what it held
5586        // is captured whole here — before anything else reads the file — and
5587        // re-emitted by `write_superblock_extension`. Read from the raw chain
5588        // rather than from `ext`, which keeps only the messages this crate
5589        // models.
5590        let extension = if ext_addr == UNDEF_ADDR || ext_addr == 0 {
5591            Box::<CarriedExtension>::default()
5592        } else {
5593            let (carried, blocks) = crate::io::object_header_io::superblock_extension_messages(
5594                &mut handle,
5595                &meta,
5596                ext_addr,
5597            )?;
5598            Box::new(CarriedExtension {
5599                superseded: blocks,
5600                carried,
5601                addr: Slot::new(None),
5602            })
5603        };
5604
5605        // The managers that extension's file-space info message names, read
5606        // before anything allocates: the sections they hold are file space
5607        // this session may hand out, and the close rewrites them.
5608        let reopened_free_space = Self::reopen_free_space(&mut handle, &meta, &ext)?;
5609
5610        // Discover links from root group (and subgroups recursively). Every
5611        // object is classified before it is registered, and the root is the
5612        // one object with no alternative: its header must be rewritten to
5613        // hold anything new, so an unmodellable root is refused here rather
5614        // than rewritten into whatever this writer could read of it.
5615        let mut walk = ReopenWalk::new(&mut handle, &meta);
5616        let root = match walk.plan(root_addr)? {
5617            ObjectPlan::Group(parts) => parts,
5618            ObjectPlan::Dataset(_) => {
5619                return Err(crate::io::IoError::InvalidState(
5620                    "cannot open this file for appending: its root object is a dataset, \
5621                     not a group"
5622                        .into(),
5623                ))
5624            }
5625            ObjectPlan::Preserve { why, .. } => {
5626                return Err(crate::io::IoError::Unsupported(format!(
5627                    "cannot open this file for appending: {why}. Every append rewrites the \
5628                     root group's header, and this writer will not rewrite it from the part \
5629                     of it that it can read"
5630                )));
5631            }
5632        };
5633        let root_header_blocks = root.header_blocks;
5634        let root_attributes = root.attributes;
5635        let root_track_order = root.track_order;
5636        let root_times = root.times;
5637        let root_dense = root.dense;
5638        let root_stab = root.stab;
5639
5640        walk.group(&root.links, "", 0)?;
5641        let collected = walk.finish();
5642        let mut link_entries = collected.hard;
5643        let mut preserved = collected.preserved;
5644        // Objects the loop below could not rebuild, by header address, so the
5645        // other links to one are preserved with it rather than left pointing
5646        // at a registry entry that is no longer there.
5647        let mut unrebuilt: std::collections::HashMap<u64, String> = Default::default();
5648
5649        // Two link entries can share one object header — hard links. Only
5650        // the first-walked path becomes the object; the rest are rebuilt
5651        // as hard-link registry entries further down. Without this split
5652        // every alias came back as its own DatasetInfo carrying the same
5653        // storage addresses, so deleting (or finalizing) one freed blocks
5654        // the others still referenced.
5655        let mut seen_header_addrs = std::collections::HashSet::new();
5656        let mut alias_entries: Vec<HardEntry> = Vec::new();
5657        link_entries.retain(|(entry, _)| {
5658            if seen_header_addrs.insert(entry.address) {
5659                true
5660            } else {
5661                alias_entries.push(entry.clone());
5662                false
5663            }
5664        });
5665
5666        // The order the walk met each object, kept before the loop below
5667        // consumes the entries: `ensure_groups_for` needs parents to precede
5668        // children.
5669        let walk_order: Vec<String> = link_entries.iter().map(|(e, _)| e.path.clone()).collect();
5670
5671        let mut existing_datasets = Vec::new();
5672        // Non-dataset link targets (groups): the header's chunk-0 address and
5673        // every block its chain occupies, by link path — so finalize can free
5674        // the blocks its rewrite supersedes — plus the attributes the header
5675        // carries, which the group registry below must keep or finalize
5676        // rewrites the group without them.
5677        type GroupHeaderInfo = (
5678            u64,
5679            crate::io::object_header_io::HeaderBlocks,
5680            Vec<AttributeEntry>,
5681            TrackOrder,
5682            Option<ObjectTimes>,
5683        );
5684        let mut group_headers: std::collections::HashMap<String, GroupHeaderInfo> =
5685            Default::default();
5686        // The dense storage each rebuilt dataset's header named, by registry
5687        // index, so finalize frees exactly what its rewrite supersedes. Keyed
5688        // after the rebuild succeeded: a preserved dataset keeps its header,
5689        // and freeing the heap that header still names would strand it.
5690        let mut dataset_dense: Vec<(usize, AttributeInfoMessage)> = Vec::new();
5691        let mut group_dense: Vec<(String, DenseCarry)> = Vec::new();
5692        // The same, for the symbol-table storage a classic group's header
5693        // names: keyed by path here, by registry index once every group has
5694        // one.
5695        let mut group_stabs: Vec<(String, StabExtents)> = Vec::new();
5696        for (entry, object) in link_entries {
5697            let HardEntry {
5698                path: name,
5699                address: obj_addr,
5700                encoded,
5701            } = entry;
5702            let parts = match object {
5703                CollectedObject::Group {
5704                    header_blocks,
5705                    attributes,
5706                    track_order,
5707                    times,
5708                    dense,
5709                    stab,
5710                } => {
5711                    group_dense.push((name.clone(), dense));
5712                    if let Some(stab) = stab {
5713                        group_stabs.push((name.clone(), stab));
5714                    }
5715                    group_headers.insert(
5716                        name,
5717                        (obj_addr, header_blocks, attributes, track_order, times),
5718                    );
5719                    continue;
5720                }
5721                CollectedObject::Dataset(parts) => *parts,
5722            };
5723            let dense_attrs = parts.dense.attrs.clone();
5724            match rebuild_dataset(&mut handle, &meta, file_size, name.clone(), obj_addr, parts) {
5725                Ok(info) => {
5726                    if let Some(ainfo) = dense_attrs {
5727                        dataset_dense.push((existing_datasets.len(), ainfo));
5728                    }
5729                    existing_datasets.push(info);
5730                }
5731                // Kept by its bytes for the same reason a header this walk
5732                // could not decode is: the rewrite would otherwise emit an
5733                // object whose chunk index no longer names its chunks.
5734                Err(e) => {
5735                    let why = format!("this writer could not rebuild its chunk index: {e}");
5736                    unrebuilt.insert(obj_addr, why.clone());
5737                    preserved.push(PreservedEntry {
5738                        path: name,
5739                        class: crate::io::reader::LinkClass::Hard,
5740                        encoded,
5741                        reason: Some(why),
5742                        // A dataset whose chunk index would not rebuild: the
5743                        // walk classified it, and it is not a datatype.
5744                        kind: PreservedKind::Unclassified,
5745                    });
5746                }
5747            }
5748        }
5749
5750        // Reconstruct the group registry. Every group is a link entry of its
5751        // own, whether or not a dataset lives under it, so the registry is
5752        // built from the discovered links — rebuilding it from dataset paths
5753        // alone made attribute-only and empty groups vanish at close, and
5754        // dropped the attributes of the groups that survived.
5755        let mut groups: Vec<GroupInfo> = Vec::new();
5756        let mut group_index_map: std::collections::HashMap<String, usize> =
5757            std::collections::HashMap::new();
5758
5759        // Register the chain of groups "/a", "/a/b", … for the link-style
5760        // path `link_path` ("a/b"), taking each one's on-disk header block
5761        // and attributes out of `group_headers` when the link walk saw it.
5762        fn ensure_groups_for(
5763            link_path: &str,
5764            groups: &mut Vec<GroupInfo>,
5765            group_index_map: &mut std::collections::HashMap<String, usize>,
5766            group_headers: &mut std::collections::HashMap<String, GroupHeaderInfo>,
5767        ) {
5768            let mut path = String::new();
5769            for part in link_path.split('/') {
5770                let parent_path = if path.is_empty() {
5771                    "/".to_string()
5772                } else {
5773                    path.clone()
5774                };
5775                if path.is_empty() {
5776                    path = format!("/{}", part);
5777                } else {
5778                    path = format!("{}/{}", path, part);
5779                }
5780                if group_index_map.contains_key(&path) {
5781                    continue;
5782                }
5783                let parent = if parent_path == "/" {
5784                    None
5785                } else {
5786                    group_index_map.get(&parent_path).copied()
5787                };
5788                let gidx = groups.len();
5789                let (obj_header_written_addr, obj_header_blocks, attributes, track_order, times) =
5790                    group_headers.remove(path.trim_start_matches('/')).map_or(
5791                        (None, Vec::new(), Vec::new(), TrackOrder::default(), None),
5792                        |(addr, blocks, attrs, track, times)| {
5793                            (Some(addr), blocks, attrs, track, times)
5794                        },
5795                    );
5796                groups.push(GroupInfo {
5797                    name: path.clone(),
5798                    parent,
5799                    creation_seq: 0,
5800                    track_order,
5801                    times,
5802                    child_datasets: Vec::new(),
5803                    child_groups: Vec::new(),
5804                    obj_header_addr: 0,
5805                    obj_header_written_addr,
5806                    obj_header_blocks,
5807                    deleted: false,
5808                    attributes,
5809                });
5810                if let Some(pidx) = parent {
5811                    groups[pidx].child_groups.push(gidx);
5812                }
5813                group_index_map.insert(path.clone(), gidx);
5814            }
5815        }
5816
5817        // Every linked group, in link-walk order (parents precede children).
5818        for name in &walk_order {
5819            if group_headers.contains_key(name.as_str()) {
5820                ensure_groups_for(name, &mut groups, &mut group_index_map, &mut group_headers);
5821            }
5822        }
5823
5824        // Assign each dataset to its immediate parent group, creating any
5825        // group the link walk could not decode (its chain stays placeholder).
5826        for (di, ds) in existing_datasets.iter().enumerate() {
5827            let parts: Vec<&str> = ds.name.split('/').collect();
5828            if parts.len() <= 1 {
5829                continue; // root-level dataset, no group
5830            }
5831            let parent_link_path = parts[..parts.len() - 1].join("/");
5832            ensure_groups_for(
5833                &parent_link_path,
5834                &mut groups,
5835                &mut group_index_map,
5836                &mut group_headers,
5837            );
5838            let gidx = group_index_map[&format!("/{}", parent_link_path)];
5839            groups[gidx].child_datasets.push(di);
5840        }
5841
5842        // An object the rebuild above gave up on is preserved by its bytes,
5843        // so the other links to it are preserved too: there is no registry
5844        // entry for them to name.
5845        alias_entries.retain(|entry| match unrebuilt.get(&entry.address) {
5846            None => true,
5847            Some(why) => {
5848                preserved.push(PreservedEntry {
5849                    path: entry.path.clone(),
5850                    class: crate::io::reader::LinkClass::Hard,
5851                    encoded: entry.encoded.clone(),
5852                    reason: Some(why.clone()),
5853                    kind: PreservedKind::Unclassified,
5854                });
5855                false
5856            }
5857        });
5858
5859        // The one thing a rebuilt shared-message table can break: an object
5860        // kept by its bytes keeps the heap IDs its header holds, and the
5861        // finalize gives the heap those IDs name back to the allocator. Every
5862        // object the registry holds is rewritten instead
5863        // ([`rebuilds_shared_messages`](Self::rebuilds_shared_messages)), so
5864        // this asks only the preserved ones, and names the object rather than
5865        // the feature — the file is appendable the moment nothing preserved
5866        // holds a heap ID or hides a subtree that might.
5867        if sohm.is_some() {
5868            for entry in &preserved {
5869                if !matches!(entry.class, crate::io::reader::LinkClass::Hard) {
5870                    continue;
5871                }
5872                let Ok((link, _)) = LinkMessage::decode(&entry.encoded, &meta.ctx) else {
5873                    continue;
5874                };
5875                let LinkTarget::Hard { address } = link.target else {
5876                    continue;
5877                };
5878                if let Some(blocks) = crate::io::object_header_io::blocks_shared_message_rebuild(
5879                    &mut handle,
5880                    &meta,
5881                    address,
5882                )? {
5883                    let why = entry
5884                        .reason
5885                        .as_deref()
5886                        .unwrap_or("this writer cannot model it");
5887                    return Err(crate::io::IoError::Unsupported(format!(
5888                        "cannot open this file for appending: '{}' {blocks}, but {why}, so \
5889                         its header keeps the bytes it has while the append lays the \
5890                         shared-message table out afresh",
5891                        entry.path
5892                    )));
5893                }
5894            }
5895        }
5896
5897        // Rebuild the hard-link registry from the alias entries set aside
5898        // above, so the H5Ldelete semantics survive a reopen. An alias whose
5899        // target the walk could not model is not here at all: it was
5900        // preserved by its own bytes, exactly as the first link to that
5901        // object was.
5902        let mut hard_links: Vec<HardLink> = Vec::new();
5903        for HardEntry {
5904            path,
5905            address: addr,
5906            ..
5907        } in alias_entries
5908        {
5909            let target = if let Some(di) = existing_datasets
5910                .iter()
5911                .position(|d| d.obj_header_addr == addr)
5912            {
5913                HardLinkTarget::Dataset(di)
5914            } else if let Some(gi) = groups
5915                .iter()
5916                .position(|g| g.obj_header_written_addr == Some(addr))
5917            {
5918                HardLinkTarget::Group(gi)
5919            } else {
5920                continue;
5921            };
5922            let (parent, link_name) = match path.rsplit_once('/') {
5923                None => (None, path),
5924                Some((dir, leaf)) => {
5925                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
5926                    (
5927                        group_index_map.get(&format!("/{dir}")).copied(),
5928                        leaf.to_string(),
5929                    )
5930                }
5931            };
5932            hard_links.push(HardLink {
5933                parent,
5934                name: link_name,
5935                target,
5936                creation_seq: 0,
5937            });
5938        }
5939
5940        // Attach every link the writer cannot express to the group that
5941        // holds it, so the rewrite of that group's header emits it again.
5942        // `ensure_groups_for` registers the parent chain, which matters for
5943        // a group whose only content is such a link: nothing else would put
5944        // it in the registry, and the close would drop group and link alike.
5945        let mut preserved_links: Vec<PreservedLink> = Vec::new();
5946        for PreservedEntry {
5947            path,
5948            class,
5949            encoded,
5950            reason,
5951            kind,
5952        } in preserved
5953        {
5954            let (parent, link_name) = match path.rsplit_once('/') {
5955                None => (None, path),
5956                Some((dir, leaf)) => {
5957                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
5958                    (
5959                        group_index_map.get(&format!("/{dir}")).copied(),
5960                        leaf.to_string(),
5961                    )
5962                }
5963            };
5964            preserved_links.push(PreservedLink {
5965                parent,
5966                name: link_name,
5967                class,
5968                encoded,
5969                reason,
5970                kind,
5971            });
5972        }
5973
5974        // Stamp the creation sequence a reopened file cannot supply. Nothing
5975        // on disk says which link was made first unless the group tracked
5976        // creation order, and this reader does not carry that back out, so
5977        // discovery order is what there is: datasets, then groups, then the
5978        // hard links found beside them — the order the writer emitted links
5979        // in before it ordered them at all.
5980        let mut creation_seq = 0u64;
5981        for d in &mut existing_datasets {
5982            d.creation_seq = creation_seq;
5983            creation_seq += 1;
5984        }
5985        for g in &mut groups {
5986            g.creation_seq = creation_seq;
5987            creation_seq += 1;
5988        }
5989        for l in &mut hard_links {
5990            l.creation_seq = creation_seq;
5991            creation_seq += 1;
5992        }
5993
5994        // The strategy is the file's, not this session's: a paged file
5995        // allocates on its own page grid however it was opened, `persist`
5996        // deciding only whether the managers survive the close.
5997        let allocator = FileAllocator::with_policy(
5998            file_size,
5999            ext.file_space_info
6000                .as_ref()
6001                .map_or(free_space::SpacePolicy::Aggr, |info| {
6002                    free_space::SpacePolicy::for_message(info)
6003                }),
6004        );
6005        // The sections the file's own managers recorded are free space, so
6006        // they are what this session allocates from first — `H5MF_alloc` asks
6007        // the free-space manager before it bumps the end of the file, and a
6008        // reopen that skipped this would grow a file that had room.
6009        allocator.reset_free_list(&reopened_free_space.sections);
6010
6011        // Now that every object has its registry index, key the dense storage
6012        // found on disk by the scope that will supersede it. A group the link
6013        // walk saw but never registered is not rewritten either, so leaving it
6014        // out is what keeps its storage referenced.
6015        let mut superseded = SupersededDense {
6016            attrs: dataset_dense
6017                .into_iter()
6018                .map(|(di, ainfo)| (AttrScope::Dataset(di), ainfo))
6019                .collect(),
6020            links: HashMap::new(),
6021        };
6022        superseded
6023            .attrs
6024            .extend(root_dense.attrs.map(|a| (AttrScope::Root, a)));
6025        superseded
6026            .links
6027            .extend(root_dense.links.map(|l| (LinkScope::Root, l)));
6028        for (name, dense) in group_dense {
6029            let Some(&gidx) = group_index_map.get(&format!("/{name}")) else {
6030                continue;
6031            };
6032            superseded
6033                .attrs
6034                .extend(dense.attrs.map(|a| (AttrScope::Group(gidx), a)));
6035            superseded
6036                .links
6037                .extend(dense.links.map(|l| (LinkScope::Group(gidx), l)));
6038        }
6039        let superseded = (!superseded.attrs.is_empty() || !superseded.links.is_empty())
6040            .then(|| Box::new(superseded));
6041
6042        // The same keying for the symbol-table storage. Built from the headers
6043        // alone, not from the superblock version: a group whose header carried
6044        // no Symbol Table message contributes nothing — what happens to a group
6045        // libhdf5 wrote at a newer bound inside an otherwise classic file — and
6046        // one that carried it keeps its storage even where the superblock is
6047        // version 2, which is what a file with shared messages is.
6048        let mut stabs: HashMap<LinkScope, StabExtents> = HashMap::new();
6049        stabs.extend(root_stab.map(|s| (LinkScope::Root, s)));
6050        for (name, extents) in group_stabs {
6051            if let Some(&gidx) = group_index_map.get(&format!("/{name}")) {
6052                stabs.insert(LinkScope::Group(gidx), extents);
6053            }
6054        }
6055        let symbol_tables = SymbolTables {
6056            found: stabs.keys().copied().collect(),
6057            superseded: Slot::new(stabs),
6058            written: Slot::new(HashMap::new()),
6059        };
6060
6061        // The superblock the close re-emits, and the generation every message
6062        // this session encodes belongs to.
6063        let legacy = legacy.map(|superblock| Box::new(LegacyFile { superblock }));
6064
6065        // Wrap the reconstructed plain vecs into the per-slot registry. The
6066        // reconstruction logic above runs single-threaded on local `Vec`s;
6067        // only the final hand-off needs the `Shared<Slot<_>>` shape.
6068        let datasets = existing_datasets
6069            .into_iter()
6070            .map(|i| Shared::new(DatasetCell::new(i)))
6071            .collect();
6072        let groups = groups
6073            .into_iter()
6074            .map(|g| Shared::new(Slot::new(g)))
6075            .collect();
6076
6077        let writer = Self {
6078            handle,
6079            allocator,
6080            ctx,
6081            datasets: Slot::new(datasets),
6082            groups: Slot::new(groups),
6083            hard_links: Slot::new(hard_links),
6084            // A reopen carries the soft and external links it found as
6085            // `preserved_links`, byte for byte; this list holds only the ones
6086            // created in this session.
6087            symbolic_links: Slot::new(Vec::new()),
6088            committed_datatypes: Slot::new(Vec::new()),
6089            preserved_links: Slot::new(preserved_links),
6090            name_index: Slot::new(Box::new(NameIndex::new())),
6091            root_attributes: Slot::new(root_attributes),
6092            create_lock: Slot::new(()),
6093            // A reopen names no bound: the file already is whichever
6094            // generation it is, and the version in its superblock is what
6095            // says so — see `libver_floor`. `set_libver_bound` is where a
6096            // caller asks for a newer one, exactly as `H5Fopen` takes a fapl.
6097            libver: None,
6098            closed: false,
6099            swmr_active: false,
6100            cwfs: Slot::new(Vec::new()),
6101            root_group_addr: None,
6102            root_group_encoded_size: 0,
6103            superseded_root_header: root_header_blocks,
6104            // The version the file already has. It is written back unchanged
6105            // and it floors every bound this session writes at, so the append
6106            // hands the file back in the generation it found it in.
6107            superblock_version: SuperblockVersion::Existing(version),
6108            // The reopened file's own policy, so objects added in this
6109            // session are made the way the file already declares.
6110            root_track_order,
6111            root_times,
6112            dense_attributes: Slot::new(HashMap::new()),
6113            dense_links: Slot::new(HashMap::new()),
6114            superseded_dense: Slot::new(superseded),
6115            track_order: root_track_order,
6116            // Not recovered from the file the way the creation-order policy
6117            // is: a version-1 header leaves no trace of whether the object was
6118            // tracking times, so there is nothing on disk to read the policy
6119            // back from. An object added to a reopened file gets this writer's
6120            // own default, the same one a created file starts at.
6121            track_times: false,
6122            next_creation_seq: Slot::new(creation_seq),
6123            pending_object_references: Slot::new(Vec::new()),
6124            pending_heap_references: Slot::new(Vec::new()),
6125            attribute_references: Slot::new(Vec::new()),
6126            legacy,
6127            symbol_tables,
6128            // The ranks the superblock or its extension declared, which every
6129            // v1-B-tree and symbol-table node this session writes is sized by.
6130            btree: meta.btree,
6131            extension,
6132            free_space: reopened_free_space.state,
6133            // The indexes the file was created with, and the blocks its
6134            // current table occupies; the next finalize lays a new table out
6135            // over them from the whole message set.
6136            sohm,
6137            source_dir: source_dir_of(path)?,
6138        };
6139        // The link graph is complete only now, so this is the first point the
6140        // count each on-disk header was written with can be read off it: in a
6141        // well-formed file the links the walk found reaching an object *are*
6142        // that count, so nothing has to be decoded out of the headers.
6143        for i in 0..writer.dataset_count() {
6144            let nlink = writer.object_link_count(HardLinkTarget::Dataset(i));
6145            writer.ds(i).lock().nlink_written = nlink;
6146        }
6147        Ok(writer)
6148    }
6149
6150    /// Return the names of all datasets created so far.
6151    pub fn dataset_names(&self) -> Vec<String> {
6152        self.dataset_refs()
6153            .iter()
6154            .filter_map(|d| {
6155                let g = d.lock();
6156                (!g.deleted).then(|| g.name.clone())
6157            })
6158            .collect()
6159    }
6160
6161    /// Find a dataset index by name. Like `H5Dopen`, the name may be any
6162    /// link path to the dataset: a user hard link's path — or a path
6163    /// whose group components pass through such links — resolves to its
6164    /// target.
6165    pub fn dataset_index(&self, name: &str) -> Option<usize> {
6166        let name = self.canonical_dataset_path(name);
6167        self.dataset_refs()
6168            .iter()
6169            .position(|d| {
6170                let g = d.lock();
6171                g.name == name && !g.deleted
6172            })
6173            .or_else(|| {
6174                self.hard_links_vec().iter().find_map(|l| match l.target {
6175                    HardLinkTarget::Dataset(i)
6176                        if self.hard_link_emitted(l) && self.hard_link_full_path(l) == name =>
6177                    {
6178                        Some(i)
6179                    }
6180                    _ => None,
6181                })
6182            })
6183    }
6184
6185    /// Reconstruct the fields a writer-mode `H5Dataset` handle needs for the
6186    /// dataset at `index`, and open it under `access`. Single owner of this
6187    /// mapping so `H5File::dataset_writer`, `H5Group::dataset_writer`, and
6188    /// the vlen-string helpers all agree — including on
6189    /// [`bind_efile_prefix`](Self::bind_efile_prefix), which no handle site
6190    /// can then forget to run.
6191    pub(crate) fn dataset_handle_parts(
6192        &self,
6193        index: usize,
6194        access: &DatasetAccess,
6195    ) -> IoResult<DatasetHandleParts> {
6196        let open = self.bind_efile_prefix(index, access)?;
6197        let ds = self.ds(index);
6198        let g = ds.lock();
6199        Ok(DatasetHandleParts {
6200            shape: g.dataspace.dims.iter().map(|&d| d as usize).collect(),
6201            element_size: g.datatype.element_size() as usize,
6202            chunk_index: g.chunk_index_kind(),
6203            open,
6204        })
6205    }
6206
6207    /// Put `access`'s external file prefix in force for the dataset at
6208    /// `index`, or join the open that already settled one.
6209    ///
6210    /// INVARIANT: every write of an externally stored dataset's raw bytes
6211    /// joins its slot names against the prefix an *open* settled, and this is
6212    /// the only place that settles one. `write_contiguous_bytes` reads it and
6213    /// nothing else writes it, so a write cannot resolve a prefix of its own
6214    /// and land bytes where a read under the same properties would not look
6215    /// for them.
6216    ///
6217    /// First open wins, and a joining open may not disagree: `H5D__open_name`
6218    /// compares its own expanded prefix against the open dataset's and fails
6219    /// when they differ (H5Dint.c:1533-1545). Measured under libhdf5 1.14.6
6220    /// and 2.0.0, with a dataset created through a dapl naming a directory
6221    /// and its handle still alive: a second open naming another directory is
6222    /// refused, one naming the same directory joins, one naming none is
6223    /// refused too, and with `HDF5_EXTFILE_PREFIX` set — which shadows every
6224    /// property, so all three expand alike — none of them is. Dropping every
6225    /// handle releases the answer and the next open settles it afresh, which
6226    /// the same measurement confirms.
6227    ///
6228    /// Returns the token that keeps the open alive, `None` for a dataset
6229    /// whose raw data is in this file and which therefore has no prefix to
6230    /// agree about.
6231    pub(crate) fn bind_efile_prefix(
6232        &self,
6233        index: usize,
6234        access: &DatasetAccess,
6235    ) -> IoResult<Option<crate::io::reader::DatasetOpenToken>> {
6236        let ds = self.ds(index);
6237        let mut g = ds.lock();
6238        let source_dir = &self.source_dir;
6239        let Some(ext) = g.external.as_mut() else {
6240            return Ok(None);
6241        };
6242        let want =
6243            crate::io::reader::resolve_extfile_prefix(access.efile_prefix_value(), source_dir);
6244        if let Some(open) = ext.prefix.open.upgrade() {
6245            if ext.prefix.expanded != want {
6246                let name = g.name.clone();
6247                return Err(crate::io::IoError::InvalidState(format!(
6248                    "dataset {name:?} is already open under a different external file                      prefix, and libhdf5 refuses to join an open that disagrees about one"
6249                )));
6250            }
6251            return Ok(Some(open));
6252        }
6253        let token: crate::io::reader::DatasetOpenToken = std::sync::Arc::new(());
6254        ext.prefix = EfilePrefix {
6255            expanded: want,
6256            open: std::sync::Arc::downgrade(&token),
6257        };
6258        Ok(Some(token))
6259    }
6260
6261    /// Reject a name some other link in the file already occupies.
6262    ///
6263    /// `name` is the registry's full-path form, with no leading `/`. HDF5
6264    /// requires link names to be unique within their group, and every kind of
6265    /// link this writer can emit competes for the same name: a dataset's own
6266    /// link, a group's, a user hard link, a soft or external link, and a link
6267    /// a reopen is carrying through verbatim. This is the one place that list
6268    /// is written down, so a creator cannot be blind to a kind it does not
6269    /// itself make — nor a kind added after it.
6270    fn ensure_name_free(&self, name: &str) -> IoResult<()> {
6271        let holder = self.name_holder(name);
6272        // The index is a filter over the registries, not a second copy of
6273        // them, so a debug build re-derives the answer on every create: a
6274        // name it failed to record surfaces as a failing assertion in the
6275        // suite rather than as two links of one name in somebody's file.
6276        #[cfg(debug_assertions)]
6277        assert_eq!(
6278            holder,
6279            self.scan_name_holder(name),
6280            "the name index disagrees with the registries for '{name}'"
6281        );
6282        match holder {
6283            None => Ok(()),
6284            Some(kind) => Err(crate::io::IoError::InvalidState(format!(
6285                "a {kind} named '{name}' already exists"
6286            ))),
6287        }
6288    }
6289
6290    /// What already holds `name`, or `None` if it is free.
6291    ///
6292    /// The kinds answer in a fixed order — dataset, group, committed
6293    /// datatype, hard link, symbolic link, preserved link — because the
6294    /// refusal names the first one that holds it. [`NameIndex`] narrows each
6295    /// kind to the entries that ever took this name; every candidate is then
6296    /// put through the same predicate the full scan used, so a hit left
6297    /// behind by a delete or a rename answers exactly as an absent one does.
6298    fn name_holder(&self, name: &str) -> Option<&'static str> {
6299        self.build_name_index();
6300        let hits: Vec<NameHit> = {
6301            let index = self.name_index.lock();
6302            index.map.as_ref().and_then(|m| m.get(name))?.clone()
6303        };
6304        for hit in &hits {
6305            if let NameHit::Dataset(i) = *hit {
6306                let ds = self.ds(i);
6307                let d = ds.lock();
6308                if !d.deleted && d.name == name {
6309                    return Some("dataset");
6310                }
6311            }
6312        }
6313        for hit in &hits {
6314            if let NameHit::Group(i) = *hit {
6315                let grp = self.grp(i);
6316                let g = grp.lock();
6317                if !g.deleted && g.name.trim_start_matches('/') == name {
6318                    return Some("group");
6319                }
6320            }
6321        }
6322        for hit in &hits {
6323            if let NameHit::Datatype(i) = *hit {
6324                // The registry lock goes before `parent_alive` takes a group
6325                // slot, never across it.
6326                let (parent, held) = {
6327                    let reg = self.committed_datatypes.lock();
6328                    (reg[i].parent, reg[i].name == name)
6329                };
6330                if held && self.parent_alive(parent) {
6331                    return Some("committed datatype");
6332                }
6333            }
6334        }
6335        if hits.contains(&NameHit::HardLink)
6336            && self
6337                .hard_links_vec()
6338                .iter()
6339                .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6340        {
6341            return Some("hard link");
6342        }
6343        if hits.contains(&NameHit::SymbolicLink)
6344            && self
6345                .symbolic_links_vec()
6346                .iter()
6347                .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6348        {
6349            return Some("link");
6350        }
6351        // A preserved link occupies its name in the group just as a modelled
6352        // one does; both are emitted, and two link messages of one name in a
6353        // group is an invalid file.
6354        if hits.contains(&NameHit::PreservedLink)
6355            && self.preserved_link_paths().iter().any(|(p, _)| *p == name)
6356        {
6357            return Some("link");
6358        }
6359        None
6360    }
6361
6362    /// The same answer read straight off the registries, which is what the
6363    /// index is checked against in a debug build.
6364    #[cfg(debug_assertions)]
6365    fn scan_name_holder(&self, name: &str) -> Option<&'static str> {
6366        if self.dataset_refs().iter().any(|d| {
6367            let g = d.lock();
6368            !g.deleted && g.name == name
6369        }) {
6370            return Some("dataset");
6371        }
6372        if self.group_refs().iter().any(|g| {
6373            let gg = g.lock();
6374            !gg.deleted && gg.name.trim_start_matches('/') == name
6375        }) {
6376            return Some("group");
6377        }
6378        if self
6379            .committed_datatypes_vec()
6380            .iter()
6381            .any(|c| self.parent_alive(c.parent) && c.name == name)
6382        {
6383            return Some("committed datatype");
6384        }
6385        if self
6386            .hard_links_vec()
6387            .iter()
6388            .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6389        {
6390            return Some("hard link");
6391        }
6392        if self
6393            .symbolic_links_vec()
6394            .iter()
6395            .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6396        {
6397            return Some("link");
6398        }
6399        if self.preserved_link_paths().iter().any(|(p, _)| *p == name) {
6400            return Some("link");
6401        }
6402        None
6403    }
6404
6405    /// Build the name index unless it is already built.
6406    ///
6407    /// The walk takes the registry spines and their slots, so it runs with no
6408    /// index lock held — the writer never holds one lock across another — and
6409    /// the result is kept only if nothing renamed, created or unlinked
6410    /// anything while it ran.
6411    fn build_name_index(&self) {
6412        let epoch = {
6413            let index = self.name_index.lock();
6414            if index.map.is_some() {
6415                return;
6416            }
6417            index.epoch
6418        };
6419        let mut map: HashMap<String, Vec<NameHit>> = HashMap::new();
6420        for (i, ds) in self.dataset_refs().iter().enumerate() {
6421            let d = ds.lock();
6422            if !d.deleted {
6423                map.entry(d.name.clone())
6424                    .or_default()
6425                    .push(NameHit::Dataset(i));
6426            }
6427        }
6428        for (i, grp) in self.group_refs().iter().enumerate() {
6429            let g = grp.lock();
6430            if !g.deleted {
6431                map.entry(g.name.trim_start_matches('/').to_string())
6432                    .or_default()
6433                    .push(NameHit::Group(i));
6434            }
6435        }
6436        for (i, c) in self.committed_datatypes_vec().iter().enumerate() {
6437            map.entry(c.name.clone())
6438                .or_default()
6439                .push(NameHit::Datatype(i));
6440        }
6441        for l in self.hard_links_vec().iter() {
6442            map.entry(self.hard_link_full_path(l))
6443                .or_default()
6444                .push(NameHit::HardLink);
6445        }
6446        for l in self.symbolic_links_vec().iter() {
6447            map.entry(self.symbolic_link_full_path(l))
6448                .or_default()
6449                .push(NameHit::SymbolicLink);
6450        }
6451        for (path, _) in self.preserved_link_paths() {
6452            map.entry(path).or_default().push(NameHit::PreservedLink);
6453        }
6454        let mut index = self.name_index.lock();
6455        if index.map.is_none() && index.epoch == epoch {
6456            index.map = Some(map);
6457        }
6458    }
6459
6460    /// Record that `hit` now holds `name` — the one way a new name enters the
6461    /// index, called from every push that gives a registry entry a name.
6462    fn register_name(&self, name: &str, hit: NameHit) {
6463        self.name_index.lock().insert(name, hit);
6464    }
6465
6466    /// Drop the index because something moved names wholesale (a group
6467    /// rename carries its subtree and every link path under it).
6468    fn forget_name_index(&self) {
6469        self.name_index.lock().forget();
6470    }
6471
6472    /// Delete a dataset name, with libhdf5's `H5Ldelete` semantics: a name
6473    /// is only a link. If `name` is a user hard link's path, just that
6474    /// link is removed and the object is untouched. If it is the tree name
6475    /// and a user hard link still names the object, the object survives
6476    /// under it — the link becomes the primary name and nothing is freed.
6477    /// Only deleting the *last* name soft-deletes the object and frees the
6478    /// file space it owned: its chunk blocks and chunk-index structures
6479    /// (or contiguous data block), the global-heap objects of its
6480    /// variable-length data and attributes, and — on a reopened file — the
6481    /// on-disk object header block. The freed space is reused by later
6482    /// allocations in this session; the file does not shrink.
6483    ///
6484    /// Refused while SWMR streaming is active: a live reader may hold any
6485    /// of those addresses (libhdf5 forbids link deletion during SWMR
6486    /// writes too).
6487    pub fn delete_dataset(&self, name: &str) -> IoResult<()> {
6488        if self.swmr_active {
6489            return Err(swmr_delete_error(name));
6490        }
6491        self.reject_external_traversal(name)?;
6492        // The gate keeps the link list and child lists still while this
6493        // delete reads and rewrites them (create_lock → op → slot order,
6494        // the same as every creator).
6495        let _create = self.create_lock.lock();
6496        // `H5Ldelete` resolves the path through links only *up to* the
6497        // leaf — the leaf is what gets deleted, so a leaf naming a user
6498        // link must stay literal and be unlinked, not its target.
6499        let name = match name.rsplit_once('/') {
6500            None => name.to_string(),
6501            Some((dir, leaf)) => format!(
6502                "{}/{leaf}",
6503                self.canonical_group_path(&format!("/{dir}"))
6504                    .trim_start_matches('/')
6505            ),
6506        };
6507        let refs = self.dataset_refs();
6508        let idx = match refs.iter().position(|d| {
6509            let g = d.lock();
6510            g.name == name && !g.deleted
6511        }) {
6512            Some(i) => i,
6513            None => {
6514                // Not a tree name — the path may name a user hard link,
6515                // and deleting a link path unlinks just that link (the
6516                // creation collision checks keep the two namespaces
6517                // disjoint, so the order of the lookups cannot matter).
6518                let link = self.hard_links_vec().iter().position(|l| {
6519                    self.hard_link_emitted(l)
6520                        && matches!(l.target, HardLinkTarget::Dataset(_))
6521                        && self.hard_link_full_path(l) == name
6522                });
6523                let Some(pos) = link else {
6524                    return Err(crate::io::IoError::NotFound(name));
6525                };
6526                self.hard_links.lock().remove(pos);
6527                return Ok(());
6528            }
6529        };
6530        // A surviving hard link keeps the object: promote the first one to
6531        // the primary name and delete nothing.
6532        let promote = self.hard_links_vec().iter().position(|l| {
6533            self.hard_link_emitted(l) && matches!(l.target, HardLinkTarget::Dataset(i) if i == idx)
6534        });
6535        if let Some(pos) = promote {
6536            self.promote_dataset_to_link(idx, pos);
6537            return Ok(());
6538        }
6539        refs[idx].lock().deleted = true;
6540        // Remove from parent group's child_datasets
6541        for grp in self.group_refs() {
6542            grp.lock().child_datasets.retain(|&di| di != idx);
6543        }
6544        self.purge_dead_links();
6545        let ds = self.ds(idx);
6546        let _op = ds.op.lock();
6547        self.release_dataset_storage(idx)
6548    }
6549
6550    /// Soft-delete a group and all its child datasets and sub-groups,
6551    /// freeing every deleted object's file space the way
6552    /// [`delete_dataset`](Self::delete_dataset) does — with the same
6553    /// `H5Ldelete` semantics: a `name` that is a user hard link's path
6554    /// unlinks just that link, and hard links from *outside* the subtree
6555    /// keep their targets. A dataset or group such a link names survives,
6556    /// re-homed under the link (a group brings its whole subtree with
6557    /// it); a link naming the deleted group itself turns the call into a
6558    /// pure rename and nothing is freed. Refused while SWMR streaming is
6559    /// active, same rule as `delete_dataset`.
6560    pub fn delete_group(&self, name: &str) -> IoResult<()> {
6561        if self.swmr_active {
6562            return Err(swmr_delete_error(name));
6563        }
6564        self.reject_external_traversal(name)?;
6565        // Same gate as `delete_dataset`: the pre-scan below and the
6566        // promotions must see a still link list and child lists.
6567        let _create = self.create_lock.lock();
6568        let name = if name.starts_with('/') {
6569            name.to_string()
6570        } else {
6571            format!("/{}", name)
6572        };
6573        // Leaf stays literal, directory resolves through links — the
6574        // same `H5Ldelete` rule as `delete_dataset`.
6575        let name = match name.rsplit_once('/') {
6576            Some((dir, leaf)) if !dir.is_empty() => {
6577                format!("{}/{leaf}", self.canonical_group_path(dir))
6578            }
6579            _ => name,
6580        };
6581        let groups = self.group_refs();
6582        let gidx = match groups.iter().position(|g| {
6583            let gg = g.lock();
6584            gg.name == name && !gg.deleted
6585        }) {
6586            Some(i) => i,
6587            None => {
6588                // Same `H5Ldelete` rule as `delete_dataset`: a path naming
6589                // a user hard link to a group unlinks just that link.
6590                let trimmed = name.trim_start_matches('/');
6591                let link = self.hard_links_vec().iter().position(|l| {
6592                    self.hard_link_emitted(l)
6593                        && matches!(l.target, HardLinkTarget::Group(_))
6594                        && self.hard_link_full_path(l) == trimmed
6595                });
6596                let Some(pos) = link else {
6597                    return Err(crate::io::IoError::NotFound(name.clone()));
6598                };
6599                self.hard_links.lock().remove(pos);
6600                return Ok(());
6601            }
6602        };
6603
6604        // A link is "outside" when its parent group does not die with the
6605        // subtree; only outside links can keep their targets alive.
6606        fn outside(parent: Option<usize>, doomed_gs: &[usize]) -> bool {
6607            match parent {
6608                None => true,
6609                Some(pi) => !doomed_gs.contains(&pi),
6610            }
6611        }
6612        // A group an outside link names survives, re-homed with its whole
6613        // subtree under the link. Each promotion moves that subtree out of
6614        // the doomed set — and can turn a link inside it into an outside
6615        // one — so rescan from scratch until no promotable group is left.
6616        // Promoting `gidx` itself makes the delete a pure rename: return.
6617        let mut doomed_ds = Vec::new();
6618        let mut doomed_gs = Vec::new();
6619        loop {
6620            doomed_ds.clear();
6621            doomed_gs.clear();
6622            self.collect_live_subtree(gidx, &mut doomed_ds, &mut doomed_gs);
6623            let promote = self
6624                .hard_links_vec()
6625                .iter()
6626                .enumerate()
6627                .find_map(|(pos, l)| match l.target {
6628                    HardLinkTarget::Group(gi)
6629                        if self.hard_link_emitted(l)
6630                            && outside(l.parent, &doomed_gs)
6631                            && doomed_gs.contains(&gi) =>
6632                    {
6633                        Some((pos, gi))
6634                    }
6635                    _ => None,
6636                });
6637            let Some((pos, gi)) = promote else { break };
6638            self.promote_group_to_link(gi, pos);
6639            if gi == gidx {
6640                return Ok(());
6641            }
6642        }
6643        // A dataset an outside link names survives its container: re-home
6644        // it under the link now, so the marking pass below never sees it.
6645        for di in doomed_ds {
6646            let promote = self.hard_links_vec().iter().position(|l| {
6647                self.hard_link_emitted(l)
6648                    && outside(l.parent, &doomed_gs)
6649                    && matches!(l.target, HardLinkTarget::Dataset(i) if i == di)
6650            });
6651            if let Some(pos) = promote {
6652                self.promote_dataset_to_link(di, pos);
6653            }
6654        }
6655
6656        let mut ds_deleted = Vec::new();
6657        let mut gs_deleted = Vec::new();
6658        self.delete_group_recursive(gidx, &mut ds_deleted, &mut gs_deleted);
6659        // Remove from parent's child_groups
6660        let parent = groups[gidx].lock().parent;
6661        if let Some(pidx) = parent {
6662            groups[pidx].lock().child_groups.retain(|&gi| gi != gidx);
6663        }
6664        self.purge_dead_links();
6665        // Free storage only after the whole subtree is marked: the lists
6666        // hold each object exactly once (the marking pass skips anything
6667        // already deleted), so nothing is freed twice.
6668        for di in ds_deleted {
6669            let ds = self.ds(di);
6670            let _op = ds.op.lock();
6671            self.release_dataset_storage(di)?;
6672        }
6673        for gi in gs_deleted {
6674            self.release_group_storage(gi)?;
6675        }
6676        Ok(())
6677    }
6678
6679    /// Collect the live (not soft-deleted) members of `gidx`'s subtree,
6680    /// each exactly once, without changing anything — the read-only twin
6681    /// of [`delete_group_recursive`](Self::delete_group_recursive), for
6682    /// the pre-scan that must run before any marking.
6683    fn collect_live_subtree(&self, gidx: usize, ds_out: &mut Vec<usize>, gs_out: &mut Vec<usize>) {
6684        if gs_out.contains(&gidx) {
6685            return;
6686        }
6687        let (child_ds, child_gs) = {
6688            let grp = self.grp(gidx);
6689            let g = grp.lock();
6690            if g.deleted {
6691                return;
6692            }
6693            (g.child_datasets.clone(), g.child_groups.clone())
6694        };
6695        gs_out.push(gidx);
6696        for di in child_ds {
6697            if !self.ds(di).lock().deleted && !ds_out.contains(&di) {
6698                ds_out.push(di);
6699            }
6700        }
6701        for gi in child_gs {
6702            self.collect_live_subtree(gi, ds_out, gs_out);
6703        }
6704    }
6705
6706    /// Re-home dataset `idx` under the hard link at `pos` in the link
6707    /// list — the surviving half of `H5Ldelete`: the link leaves the user
6708    /// list and becomes the dataset's primary (tree) name, in the link's
6709    /// parent group. Storage is untouched; any further links to the
6710    /// dataset stay in the list and keep resolving.
6711    fn promote_dataset_to_link(&self, idx: usize, pos: usize) {
6712        let link = self.hard_links.lock().remove(pos);
6713        let new_name = self.hard_link_full_path(&link);
6714        for grp in self.group_refs() {
6715            grp.lock().child_datasets.retain(|&di| di != idx);
6716        }
6717        if let Some(pi) = link.parent {
6718            self.grp(pi).lock().child_datasets.push(idx);
6719        }
6720        self.ds(idx).lock().name = new_name.clone();
6721        self.register_name(&new_name, NameHit::Dataset(idx));
6722    }
6723
6724    /// The group counterpart of
6725    /// [`promote_dataset_to_link`](Self::promote_dataset_to_link): re-home
6726    /// group `gidx` under the hard link at `pos`, bringing its whole
6727    /// subtree with it. Names are stored as full paths, so every live
6728    /// descendant is renamed by prefix.
6729    fn promote_group_to_link(&self, gidx: usize, pos: usize) {
6730        let link = self.hard_links.lock().remove(pos);
6731        let new_name = format!("/{}", self.hard_link_full_path(&link));
6732        let old_name = self.grp(gidx).lock().name.clone();
6733        for grp in self.group_refs() {
6734            grp.lock().child_groups.retain(|&g| g != gidx);
6735        }
6736        {
6737            let grp = self.grp(gidx);
6738            let mut g = grp.lock();
6739            g.parent = link.parent;
6740            g.name = new_name.clone();
6741        }
6742        if let Some(pi) = link.parent {
6743            self.grp(pi).lock().child_groups.push(gidx);
6744        }
6745
6746        let mut ds_in = Vec::new();
6747        let mut gs_in = Vec::new();
6748        self.collect_live_subtree(gidx, &mut ds_in, &mut gs_in);
6749        // Group names carry a leading '/' ("/a/b"), dataset names none
6750        // ("a/b/ds") — two prefix forms of the same rename.
6751        let old_grp_prefix = format!("{old_name}/");
6752        let new_grp_prefix = format!("{new_name}/");
6753        let old_ds_prefix = old_grp_prefix.trim_start_matches('/').to_string();
6754        let new_ds_prefix = new_grp_prefix.trim_start_matches('/').to_string();
6755        for gi in gs_in {
6756            if gi == gidx {
6757                continue;
6758            }
6759            let grp = self.grp(gi);
6760            let mut g = grp.lock();
6761            let renamed = g
6762                .name
6763                .strip_prefix(&old_grp_prefix)
6764                .map(|rest| format!("{new_grp_prefix}{rest}"));
6765            if let Some(n) = renamed {
6766                g.name = n;
6767            }
6768        }
6769        for di in ds_in {
6770            let ds = self.ds(di);
6771            let mut d = ds.lock();
6772            let renamed = d
6773                .name
6774                .strip_prefix(&old_ds_prefix)
6775                .map(|rest| format!("{new_ds_prefix}{rest}"));
6776            if let Some(n) = renamed {
6777                d.name = n;
6778            }
6779        }
6780        // A group carries its subtree and every link path under it, so far
6781        // more names moved than this function can enumerate: start over.
6782        self.forget_name_index();
6783    }
6784
6785    /// Drop link entries that can no longer be emitted — their parent group
6786    /// or, for a hard link, their target object was just deleted — so the
6787    /// lists mirror what the file will hold instead of carrying suppressed
6788    /// zombies. Both kinds are purged here so a delete cannot clear one list
6789    /// and leave the other holding a name in a group that is gone.
6790    fn purge_dead_links(&self) {
6791        let dead: Vec<usize> = self
6792            .hard_links_vec()
6793            .iter()
6794            .enumerate()
6795            .filter(|(_, l)| !self.hard_link_emitted(l))
6796            .map(|(p, _)| p)
6797            .collect();
6798        let mut links = self.hard_links.lock();
6799        for p in dead.into_iter().rev() {
6800            links.remove(p);
6801        }
6802        drop(links);
6803
6804        let dead: Vec<usize> = self
6805            .symbolic_links_vec()
6806            .iter()
6807            .enumerate()
6808            .filter(|(_, l)| !self.symbolic_link_emitted(l))
6809            .map(|(p, _)| p)
6810            .collect();
6811        let mut links = self.symbolic_links.lock();
6812        for p in dead.into_iter().rev() {
6813            links.remove(p);
6814        }
6815    }
6816
6817    /// Mark `gidx` and its subtree deleted, appending each newly-deleted
6818    /// object's index to `ds_out` / `gs_out` exactly once — the caller
6819    /// frees their storage, and an object reachable twice (or a subtree
6820    /// already deleted) must not be freed twice.
6821    fn delete_group_recursive(
6822        &self,
6823        gidx: usize,
6824        ds_out: &mut Vec<usize>,
6825        gs_out: &mut Vec<usize>,
6826    ) {
6827        // Mark deleted and snapshot the child lists, releasing the group lock
6828        // before locking any dataset/child-group slot (spine → slot order).
6829        let (child_ds, child_gs) = {
6830            let grp = self.grp(gidx);
6831            let mut g = grp.lock();
6832            if g.deleted {
6833                return;
6834            }
6835            g.deleted = true;
6836            (g.child_datasets.clone(), g.child_groups.clone())
6837        };
6838        gs_out.push(gidx);
6839        for di in child_ds {
6840            let ds = self.ds(di);
6841            let mut d = ds.lock();
6842            if !d.deleted {
6843                d.deleted = true;
6844                ds_out.push(di);
6845            }
6846        }
6847        for gi in child_gs {
6848            self.delete_group_recursive(gi, ds_out, gs_out);
6849        }
6850    }
6851
6852    /// Free everything a soft-deleted dataset owned. The single owner of
6853    /// delete-time reclamation, called only from the two delete paths with
6854    /// the dataset already marked deleted and its op lock held.
6855    ///
6856    /// A deleted dataset contributes nothing to finalize (the header,
6857    /// index-flush and append-flush loops all skip it), so nothing in the
6858    /// finalized file can reference the blocks freed here. Never runs under
6859    /// SWMR — the delete entry points refuse first.
6860    fn release_dataset_storage(&self, index: usize) -> IoResult<()> {
6861        use crate::format::messages::datatype::DatatypeMessage;
6862        let (indexed, ndims, contiguous, is_vlen, attrs, header_blocks, mapping_list) = {
6863            let ds = self.ds(index);
6864            let mut m = ds.lock();
6865            // Buffered rows were never written to a chunk; they die with
6866            // the dataset instead of being flushed at close.
6867            m.append = None;
6868            let indexed = m.is_chunked();
6869            let contiguous = (!indexed && m.data_addr != UNDEF_ADDR && m.data_size > 0)
6870                .then_some((m.data_addr, m.data_size));
6871            m.data_addr = UNDEF_ADDR;
6872            m.data_size = 0;
6873            // The external files themselves are the application's, not this
6874            // file's, and neither is the name heap freed: `H5O_MSG_EFL`
6875            // installs no file-delete method, so libhdf5 leaves the heap block
6876            // behind too. Dropping the list is what stops a deleted dataset
6877            // still claiming storage.
6878            m.external = None;
6879            // The mapping list is this file's own metadata, so unlike the
6880            // external files above it *is* freed — `H5D__virtual_delete`
6881            // removes the heap object. The source datasets it named are
6882            // another file's and are left alone.
6883            let mapping_list = m
6884                .virtual_storage
6885                .take()
6886                .and_then(|v| u16::try_from(v.heap_index).ok().map(|i| (v.heap_addr, i)));
6887            let is_vlen = matches!(
6888                m.datatype,
6889                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
6890            );
6891            let attrs = std::mem::take(&mut m.attributes);
6892            m.obj_header_written_addr = None;
6893            let header_blocks = std::mem::take(&mut m.obj_header_blocks);
6894            (
6895                indexed,
6896                m.dataspace.dims.len(),
6897                contiguous,
6898                is_vlen,
6899                attrs,
6900                header_blocks,
6901                mapping_list,
6902            )
6903        };
6904        if let Some((addr, idx)) = mapping_list {
6905            self.remove_heap_objects([(addr, vec![idx])].into_iter().collect())?;
6906        }
6907        if indexed {
6908            // Prune to a zero extent: every stored chunk is entirely beyond
6909            // it, so the walk frees each chunk block and collects the vlen
6910            // references its bytes held (released inside).
6911            self.prune_chunks_beyond(index, &vec![0; ndims])?;
6912            self.free_chunk_index(index)?;
6913        } else if let Some((addr, size)) = contiguous {
6914            if is_vlen {
6915                let data = self.handle.read_at(addr, size as usize)?;
6916                self.release_vlen_references(&data)?;
6917            }
6918            self.allocator.free(addr, size, FreeSpaceClass::RawData);
6919        }
6920        for attr in &attrs {
6921            self.release_attr_vlen(attr)?;
6922        }
6923        self.release_superseded_dense_attrs(AttrScope::Dataset(index))?;
6924        for (addr, size) in header_blocks {
6925            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
6926        }
6927        Ok(())
6928    }
6929
6930    /// Free a deleted group's file space: its attributes' global-heap
6931    /// objects and, on a reopened file, the on-disk header block. The
6932    /// group counterpart of
6933    /// [`release_dataset_storage`](Self::release_dataset_storage).
6934    fn release_group_storage(&self, gidx: usize) -> IoResult<()> {
6935        let (attrs, header_blocks) = {
6936            let grp = self.grp(gidx);
6937            let mut g = grp.lock();
6938            let attrs = std::mem::take(&mut g.attributes);
6939            g.obj_header_written_addr = None;
6940            (attrs, std::mem::take(&mut g.obj_header_blocks))
6941        };
6942        for attr in &attrs {
6943            self.release_attr_vlen(attr)?;
6944        }
6945        self.release_superseded_dense_attrs(AttrScope::Group(gidx))?;
6946        self.release_superseded_dense_links(LinkScope::Group(gidx))?;
6947        for (addr, size) in header_blocks {
6948            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
6949        }
6950        Ok(())
6951    }
6952
6953    /// Free the dense attribute storage a reopened header names, once, when
6954    /// this session stops naming it — because the header is being rewritten
6955    /// around fresh storage, or because the object was deleted.
6956    ///
6957    /// The single owner of that transition: nothing else removes an attribute
6958    /// entry from [`superseded_dense`](Self::superseded_dense), and this
6959    /// removes it as it frees, so no heap is freed twice or left half freed.
6960    /// An object whose storage was compact, or whose header this session
6961    /// keeps, has no entry and nothing happens.
6962    ///
6963    /// Never under SWMR: a live reader may still be walking the storage the
6964    /// published headers name, the same rule the superseded-header and
6965    /// relocated-chunk paths follow. The entry stays in place, unfreed.
6966    fn release_superseded_dense_attrs(&self, scope: AttrScope) -> IoResult<()> {
6967        if self.swmr_active {
6968            return Ok(());
6969        }
6970        let taken = self
6971            .superseded_dense
6972            .lock()
6973            .as_mut()
6974            .and_then(|s| s.attrs.remove(&scope));
6975        let Some(ainfo) = taken else {
6976            return Ok(());
6977        };
6978        self.release_dense_storage(
6979            ainfo.fractal_heap_address,
6980            ainfo.name_btree_address,
6981            ainfo.creation_order_btree_address,
6982        )
6983    }
6984
6985    /// The link counterpart of
6986    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs),
6987    /// under the same invariant and the same SWMR rule. Split from it because
6988    /// the two are superseded at different points of a finalize: attribute
6989    /// storage before the object headers are laid out, link storage after
6990    /// every one of them has an address.
6991    fn release_superseded_dense_links(&self, scope: LinkScope) -> IoResult<()> {
6992        if self.swmr_active {
6993            return Ok(());
6994        }
6995        let taken = self
6996            .superseded_dense
6997            .lock()
6998            .as_mut()
6999            .and_then(|s| s.links.remove(&scope));
7000        let Some(linfo) = taken else {
7001            return Ok(());
7002        };
7003        self.release_dense_storage(
7004            linfo.fractal_heap_address,
7005            linfo.name_btree_address,
7006            linfo.creation_order_btree_address,
7007        )
7008    }
7009
7010    /// Return one dense storage's file space to the allocator: the fractal
7011    /// heap in full, its name index, and the creation-order index when the
7012    /// object had one.
7013    ///
7014    /// The extents come from walking the structures themselves rather than
7015    /// from re-deriving what a writer would have allocated, so storage
7016    /// libhdf5 laid out is freed as accurately as storage this crate wrote.
7017    /// Every walk here already ran once this session — the reopen read every
7018    /// attribute out of this heap through the same index — so a failure means
7019    /// the file changed underneath us, and surfacing it beats freeing a
7020    /// partial extent list.
7021    fn release_dense_storage(
7022        &self,
7023        heap_addr: u64,
7024        name_bt2_addr: u64,
7025        corder_bt2_addr: Option<u64>,
7026    ) -> IoResult<()> {
7027        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
7028        use crate::format::fractal_heap::collect_heap_extents;
7029
7030        let mut reader = crate::io::reader::HandleBlockReader {
7031            handle: &self.handle,
7032        };
7033        let mut extents = Vec::new();
7034        if heap_addr != UNDEF_ADDR {
7035            extents.extend(collect_heap_extents(heap_addr, &self.ctx, &mut reader)?);
7036        }
7037        for addr in [Some(name_bt2_addr), corder_bt2_addr]
7038            .into_iter()
7039            .flatten()
7040            .filter(|&a| a != UNDEF_ADDR)
7041        {
7042            extents.extend(collect_btree_v2_extents(addr, &self.ctx, &mut reader)?);
7043        }
7044        for (addr, len) in extents {
7045            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
7046        }
7047        Ok(())
7048    }
7049
7050    /// Free a deleted dataset's chunk-index structures, after the chunks
7051    /// themselves were freed by a zero-extent prune. Takes the index info
7052    /// out of the slot, so the dataset no longer claims chunked storage.
7053    ///
7054    /// Every block's size is recovered the way its allocation computed it:
7055    /// re-encoding the in-memory copy (EA header and index block, FA
7056    /// header and data block, BT2 header) or sizing a same-shape dummy
7057    /// from the array geometry (EA data blocks, whose element counts come
7058    /// from [`EaGeometry`]; BT2 nodes are all `node_size`).
7059    fn free_chunk_index(&self, index: usize) -> IoResult<()> {
7060        let ds = self.ds(index);
7061        let mut m = ds.lock();
7062        let is_filtered = m.filter_pipeline.is_some();
7063        if let Some(c) = m.chunked.take() {
7064            let p = &c.earray_params;
7065            let bits = p.max_nelmts_bits;
7066            let csl = c.chunk_size_len;
7067            let geo = EaGeometry::new(
7068                p.idx_blk_elmts,
7069                p.data_blk_min_elmts,
7070                p.sup_blk_min_data_ptrs,
7071                bits,
7072                p.max_dblk_page_nelmts_bits,
7073            )?;
7074            let dblk_size = |nelmts: u64| -> u64 {
7075                if is_filtered {
7076                    FilteredDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7077                        .encode(&self.ctx, bits, csl)
7078                        .len() as u64
7079                } else {
7080                    ExtensibleArrayDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7081                        .encoded_size(&self.ctx, bits) as u64
7082                }
7083            };
7084            let (dblk_addrs, sblk_addrs, iblk_size) = if is_filtered {
7085                let f = c.filt_iblk.as_ref().unwrap();
7086                (
7087                    f.dblk_addrs.clone(),
7088                    f.sblk_addrs.clone(),
7089                    f.encode(&self.ctx, csl).len() as u64,
7090                )
7091            } else {
7092                (
7093                    c.ea_iblk.dblk_addrs.clone(),
7094                    c.ea_iblk.sblk_addrs.clone(),
7095                    c.ea_iblk.encoded_size(&self.ctx) as u64,
7096                )
7097            };
7098            // Data blocks addressed from the index block belong to the
7099            // first `iblock_nsblks` super blocks; each of those defines the
7100            // element count (and so the disk size) of its data blocks.
7101            let mut g = 0usize;
7102            'direct: for s in geo.sblk.iter().take(geo.iblock_nsblks) {
7103                for _ in 0..s.ndblks {
7104                    let Some(&a) = dblk_addrs.get(g) else {
7105                        break 'direct;
7106                    };
7107                    g += 1;
7108                    if a == UNDEF_ADDR {
7109                        continue;
7110                    }
7111                    if s.dblk_nelmts > geo.dblk_page_nelmts {
7112                        return Err(crate::io::IoError::InvalidState(
7113                            "cannot free a paged extensible-array data block, \
7114                             which is not yet supported"
7115                                .into(),
7116                        ));
7117                    }
7118                    self.allocator
7119                        .free(a, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7120                }
7121            }
7122            for (off, &sa) in sblk_addrs.iter().enumerate() {
7123                if sa == UNDEF_ADDR {
7124                    continue;
7125                }
7126                let s = geo.sblk[geo.iblock_nsblks + off];
7127                if s.dblk_nelmts > geo.dblk_page_nelmts {
7128                    return Err(crate::io::IoError::InvalidState(
7129                        "cannot free a paged extensible-array data block, \
7130                         which is not yet supported"
7131                            .into(),
7132                    ));
7133                }
7134                let buf = self.handle.read_at_most(sa, 65536)?;
7135                let sb =
7136                    ExtensibleArraySuperBlock::decode(&buf, &self.ctx, bits, s.ndblks as usize, 0)?;
7137                for &da in &sb.dblk_addrs {
7138                    if da != UNDEF_ADDR {
7139                        self.allocator
7140                            .free(da, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7141                    }
7142                }
7143                self.allocator.free(
7144                    sa,
7145                    sb.encode(&self.ctx, bits).len() as u64,
7146                    FreeSpaceClass::Metadata,
7147                );
7148            }
7149            self.allocator
7150                .free(c.ea_iblk_addr, iblk_size, FreeSpaceClass::Metadata);
7151            self.allocator.free(
7152                c.ea_header_addr,
7153                c.ea_header.encoded_size(&self.ctx) as u64,
7154                FreeSpaceClass::Metadata,
7155            );
7156            return Ok(());
7157        }
7158        if let Some(fa) = m.fixed_array.take() {
7159            self.allocator.free(
7160                fa.fa_dblk_addr,
7161                fixed_array_dblk_disk_size(&self.ctx, &fa.fa_header),
7162                FreeSpaceClass::Metadata,
7163            );
7164            self.allocator.free(
7165                fa.fa_header_addr,
7166                fa.fa_header.encode(&self.ctx).len() as u64,
7167                FreeSpaceClass::Metadata,
7168            );
7169            return Ok(());
7170        }
7171        // The implicit index has no structure to free, only the one run of
7172        // chunk space it was given at create — which is the whole of its
7173        // storage, so nothing else can be leaked or double-freed here.
7174        if let Some(imp) = m.implicit.take() {
7175            self.allocator
7176                .free(imp.data_addr, imp.data_size, FreeSpaceClass::RawData);
7177            return Ok(());
7178        }
7179        // The single-chunk index has no structure of its own either: its one
7180        // chunk is the whole of its storage, addressed directly from the
7181        // layout message rather than any index this function's doc comment's
7182        // "chunks already freed by a zero-extent prune" applies to — so
7183        // freeing it here, if it was ever allocated, is the only place it
7184        // happens.
7185        if let Some(sc) = m.single_chunk.take() {
7186            if sc.data_addr != UNDEF_ADDR {
7187                let len = if is_filtered { sc.nbytes } else { sc.data_size };
7188                self.allocator
7189                    .free(sc.data_addr, len, FreeSpaceClass::RawData);
7190            }
7191            return Ok(());
7192        }
7193        // The version-1 B-tree owns nothing but its node blocks: the header
7194        // every other index has is, here, the root pointer inside the layout
7195        // message.
7196        if let Some(bt1) = m.btree_v1.take() {
7197            let element_size = m.datatype.element_size() as u64;
7198            let node_size = bt1
7199                .build_tree(element_size, self.ctx.sizeof_addr as usize)
7200                .node_size();
7201            for &a in &bt1.node_addrs {
7202                self.allocator
7203                    .free(a, node_size as u64, FreeSpaceClass::Metadata);
7204            }
7205            return Ok(());
7206        }
7207        if let Some(bt2) = m.btree_v2.take() {
7208            let tree = bt2.index.build_tree(&self.ctx);
7209            for &a in &bt2.node_addrs {
7210                self.allocator
7211                    .free(a, tree.node_size as u64, FreeSpaceClass::Metadata);
7212            }
7213            self.allocator.free(
7214                bt2.bt2_header_addr,
7215                tree.header(UNDEF_ADDR).encode(&self.ctx).len() as u64,
7216                FreeSpaceClass::Metadata,
7217            );
7218        }
7219        Ok(())
7220    }
7221
7222    /// Return the chunk dimensions for a dataset, if chunked.
7223    ///
7224    /// Returns an owned `Vec` because the chunk geometry now lives behind the
7225    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7226    pub fn dataset_chunk_dims(&self, index: usize) -> Option<Vec<u64>> {
7227        let ds = self.ds(index);
7228        let m = ds.lock();
7229        m.chunk_index_kind().map(|kind| match kind {
7230            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
7231            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
7232            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
7233            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
7234            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
7235            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
7236        })
7237    }
7238
7239    /// Return the current dimensions of a dataset.
7240    ///
7241    /// Returns an owned `Vec` because the dataspace now lives behind the
7242    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7243    pub fn dataset_dims(&self, index: usize) -> Vec<u64> {
7244        self.ds(index).lock().dataspace.dims.clone()
7245    }
7246
7247    /// Return the maximum extent a dataset declares, per dimension.
7248    ///
7249    /// An absent maximum shape means the shape is fixed at its current extent
7250    /// (libhdf5 defaults maxdims to dims at creation), so the current
7251    /// dimensions are returned; `H5S_UNLIMITED` is `u64::MAX`.
7252    pub fn dataset_max_dims(&self, index: usize) -> Vec<u64> {
7253        let ds = self.ds(index);
7254        let m = ds.lock();
7255        m.dataspace
7256            .max_dims
7257            .clone()
7258            .unwrap_or_else(|| m.dataspace.dims.clone())
7259    }
7260
7261    /// Whether a dataset stores its raw data through a filter pipeline.
7262    ///
7263    /// The write paths ask before choosing how to hand a chunk over: an
7264    /// unfiltered chunk's bytes go to the file exactly as the caller holds
7265    /// them, while a filtered one has to be compressed first.
7266    pub(crate) fn dataset_is_filtered(&self, index: usize) -> bool {
7267        self.ds(index).lock().filter_pipeline.is_some()
7268    }
7269
7270    /// Return the datatype a dataset declares on disk.
7271    ///
7272    /// The typed write paths need it to store bytes in the declared byte
7273    /// order; a reopened dataset handle has no copy of its own, and a cached
7274    /// one could disagree with what the header will say.
7275    pub fn dataset_datatype(&self, index: usize) -> DatatypeMessage {
7276        self.ds(index).lock().datatype.clone()
7277    }
7278
7279    /// Create a group in the file hierarchy.
7280    ///
7281    /// `parent_path` is the full path of the parent group (e.g., "/" for root).
7282    /// `name` is the name of the new group (e.g., "detector").
7283    ///
7284    /// Returns the group index in the writer's group list.
7285    pub fn create_group(&self, parent_path: &str, name: &str) -> IoResult<usize> {
7286        // Hold the create gate across the uniqueness check and the registry
7287        // push so the two are atomic (see `create_lock`).
7288        let _create = self.create_lock.lock();
7289        // A parent path through hard links creates in the link's target,
7290        // as HDF5 traversal does.
7291        let parent_path = self.canonical_group_path(parent_path);
7292        let parent_path = parent_path.as_str();
7293        let full_name = if parent_path == "/" {
7294            format!("/{}", name)
7295        } else {
7296            format!("{}/{}", parent_path, name)
7297        };
7298        // Same rule as dataset creation: a path through a carried external
7299        // link names a group in the other file, which this writer cannot make.
7300        self.reject_external_traversal(&full_name)?;
7301        // `name` may itself carry path components; resolving the whole thing
7302        // is what keeps a '/' out of the link this group will be reached by.
7303        let (parent_idx, _leaf) = self.split_parent(full_name.trim_start_matches('/'))?;
7304
7305        self.ensure_name_free(full_name.trim_start_matches('/'))?;
7306
7307        let group_idx = self.push_group(GroupInfo {
7308            name: full_name,
7309            parent: parent_idx,
7310            creation_seq: self.take_creation_seq(),
7311            track_order: self.track_order,
7312            times: self.created_object_times(),
7313            child_datasets: Vec::new(),
7314            child_groups: Vec::new(),
7315            obj_header_addr: 0,
7316            obj_header_written_addr: None,
7317            obj_header_blocks: Vec::new(),
7318            deleted: false,
7319            attributes: Vec::new(),
7320        });
7321
7322        // Register this group as a child of its parent
7323        if let Some(pidx) = parent_idx {
7324            self.grp(pidx).lock().child_groups.push(group_idx);
7325        }
7326
7327        Ok(group_idx)
7328    }
7329
7330    /// Register a dataset as belonging to a group.
7331    ///
7332    /// `group_path` is the full path of the group (e.g., "/detector").
7333    /// `ds_index` is the dataset index returned by `create_dataset`.
7334    pub fn assign_dataset_to_group(&self, group_path: &str, ds_index: usize) -> IoResult<()> {
7335        let group_path = self.canonical_group_path(group_path);
7336        let group_path = group_path.as_str();
7337        let groups = self.group_refs();
7338        let group_idx = groups
7339            .iter()
7340            .position(|g| {
7341                let gg = g.lock();
7342                gg.name == group_path && !gg.deleted
7343            })
7344            .ok_or_else(|| {
7345                crate::io::IoError::NotFound(format!("group '{}' not found", group_path))
7346            })?;
7347        // A move, not an addition: the create gate has already placed every
7348        // dataset from the path components of its name, so appending here
7349        // would leave one dataset linked from two groups at once.
7350        for g in &groups {
7351            g.lock().child_datasets.retain(|&d| d != ds_index);
7352        }
7353        groups[group_idx].lock().child_datasets.push(ds_index);
7354        Ok(())
7355    }
7356
7357    /// Create a hard link: an additional name for an object that already
7358    /// exists in the file.
7359    ///
7360    /// No data is copied — the link and its target share one object header,
7361    /// exactly as `h5py` / libhdf5 hard links do.
7362    ///
7363    /// * `parent_group_path` — full path of the group that will hold the
7364    ///   link (`"/"` for the root group).
7365    /// * `link_name` — leaf name of the new link within that group.
7366    /// * `target_path` — full path of an existing dataset or group, with or
7367    ///   without a leading `/`.
7368    pub fn create_hard_link(
7369        &self,
7370        parent_group_path: &str,
7371        link_name: &str,
7372        target_path: &str,
7373    ) -> IoResult<()> {
7374        if link_name.is_empty() || link_name.contains('/') {
7375            return Err(crate::io::IoError::InvalidState(format!(
7376                "hard link name '{link_name}' must be a non-empty leaf name"
7377            )));
7378        }
7379
7380        // Neither end may sit across a carried external link: the target
7381        // would be an object in the other file, and the link itself would be
7382        // a name in a group this writer does not own.
7383        self.reject_external_traversal(target_path)?;
7384        self.reject_external_traversal(&format!(
7385            "{}/{link_name}",
7386            parent_group_path.trim_end_matches('/')
7387        ))?;
7388
7389        // Hold the create gate across the collision check and the hard-link
7390        // push so the two are atomic (see `create_lock`).
7391        let _create = self.create_lock.lock();
7392        // Both paths resolve through hard links, as HDF5 traversal does.
7393        let parent_group_path = self.canonical_group_path(parent_group_path);
7394        let parent_group_path = parent_group_path.as_str();
7395
7396        // Resolve the parent group (None == root).
7397        let parent = if parent_group_path == "/" {
7398            None
7399        } else {
7400            Some(
7401                self.group_refs()
7402                    .iter()
7403                    .position(|g| {
7404                        let gg = g.lock();
7405                        gg.name == parent_group_path && !gg.deleted
7406                    })
7407                    .ok_or_else(|| {
7408                        crate::io::IoError::NotFound(format!(
7409                            "parent group '{parent_group_path}' not found"
7410                        ))
7411                    })?,
7412            )
7413        };
7414
7415        // Resolve the target. Dataset names are stored without a leading
7416        // '/', group names with one — compare on the trimmed form. A
7417        // trailing '/' is tolerated too.
7418        let target_rel = self.canonical_dataset_path(target_path.trim_matches('/'));
7419        let target_rel = target_rel.as_str();
7420        if target_rel.is_empty() {
7421            return Err(crate::io::IoError::InvalidState(
7422                "cannot hard-link the root group".into(),
7423            ));
7424        }
7425        let target = self.resolve_object(target_rel).ok_or_else(|| {
7426            crate::io::IoError::NotFound(format!("hard link target '{target_path}' not found"))
7427        })?;
7428
7429        // Reject a name already taken in the parent group.
7430        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7431
7432        self.hard_links.lock().push(HardLink {
7433            parent,
7434            name: link_name.to_string(),
7435            target,
7436            creation_seq: self.take_creation_seq(),
7437        });
7438        self.register_name(&self.link_full_path(parent, link_name), NameHit::HardLink);
7439        Ok(())
7440    }
7441
7442    /// Whether a hard link will actually be emitted: both its parent group
7443    /// and its target object must still be present (not soft-deleted).
7444    fn hard_link_emitted(&self, link: &HardLink) -> bool {
7445        let parent_ok = self.parent_alive(link.parent);
7446        let target_ok = match link.target {
7447            HardLinkTarget::Dataset(i) => !self.ds(i).lock().deleted,
7448            HardLinkTarget::Group(i) => !self.grp(i).lock().deleted,
7449        };
7450        parent_ok && target_ok
7451    }
7452
7453    /// The full path a link occupies, with no leading `/` — the same form
7454    /// dataset names are stored in. The one place a parent index and a leaf
7455    /// name become a path, so every link kind answers the collision check in
7456    /// the same spelling.
7457    fn link_full_path(&self, parent: Option<usize>, name: &str) -> String {
7458        match parent {
7459            None => name.to_string(),
7460            Some(pi) => format!(
7461                "{}/{name}",
7462                self.grp(pi).lock().name.trim_start_matches('/')
7463            ),
7464        }
7465    }
7466
7467    /// The full path a hard link occupies; see [`Self::link_full_path`].
7468    fn hard_link_full_path(&self, link: &HardLink) -> String {
7469        self.link_full_path(link.parent, &link.name)
7470    }
7471
7472    /// Whether a symbolic link will actually be emitted: its parent group
7473    /// must still be present. There is no target to check — a soft or
7474    /// external link is allowed to dangle, and `H5Lcreate_soft` does not look
7475    /// at the path it stores.
7476    fn symbolic_link_emitted(&self, link: &SymbolicLink) -> bool {
7477        self.parent_alive(link.parent)
7478    }
7479
7480    /// Whether the group that would hold a link still exists; `None` is the
7481    /// root group, which cannot be deleted.
7482    ///
7483    /// A deleted group's header is never written, so nothing it would have
7484    /// held is in the file — and the name is free again. Every registry
7485    /// decides that the same way, through here.
7486    fn parent_alive(&self, parent: Option<usize>) -> bool {
7487        match parent {
7488            None => true,
7489            Some(pi) => !self.grp(pi).lock().deleted,
7490        }
7491    }
7492
7493    /// The full path a symbolic link occupies; see [`Self::link_full_path`].
7494    fn symbolic_link_full_path(&self, link: &SymbolicLink) -> String {
7495        self.link_full_path(link.parent, &link.name)
7496    }
7497
7498    /// Create a soft or external link: a name in a group whose value is a
7499    /// path rather than an object.
7500    ///
7501    /// The single owner of symbolic-link creation — `H5Lcreate_soft` and
7502    /// `H5Lcreate_external` differ only in the value they store, and the
7503    /// name, parent and collision rules they share are all here.
7504    ///
7505    /// * `parent_group_path` — full path of the group that will hold the
7506    ///   link (`"/"` for the root group).
7507    /// * `link_name` — leaf name of the new link within that group.
7508    /// * `target` — the path this link names, and for an external link the
7509    ///   file holding it. Neither is resolved or required to exist: HDF5
7510    ///   answers a symbolic link at traversal time, so a dangling one is a
7511    ///   legal file.
7512    pub fn create_symbolic_link(
7513        &self,
7514        parent_group_path: &str,
7515        link_name: &str,
7516        target: LinkTarget,
7517    ) -> IoResult<()> {
7518        if link_name.is_empty() || link_name.contains('/') {
7519            return Err(crate::io::IoError::InvalidState(format!(
7520                "link name '{link_name}' must be a non-empty leaf name"
7521            )));
7522        }
7523        // `H5Lcreate_external` refuses an empty file or object name, and
7524        // stores the object path normalized; a link written here and one
7525        // libhdf5 writes from the same arguments then hold the same bytes.
7526        let target = match target {
7527            LinkTarget::External { file, path } => {
7528                if file.is_empty() || path.is_empty() {
7529                    return Err(crate::io::IoError::InvalidState(
7530                        "an external link needs both a file name and an object path".into(),
7531                    ));
7532                }
7533                LinkTarget::External {
7534                    file,
7535                    path: crate::format::messages::link::normalize_object_path(&path),
7536                }
7537            }
7538            other => other,
7539        };
7540        // The link itself would be a name in a group that lives in another
7541        // file; its *value* may name anything, including a path this writer
7542        // cannot follow, because nothing follows it here.
7543        self.reject_external_traversal(&format!(
7544            "{}/{link_name}",
7545            parent_group_path.trim_end_matches('/')
7546        ))?;
7547
7548        let _create = self.create_lock.lock();
7549        let parent_group_path = self.canonical_group_path(parent_group_path);
7550        let parent_group_path = parent_group_path.as_str();
7551        let parent = if parent_group_path == "/" {
7552            None
7553        } else {
7554            Some(
7555                self.group_refs()
7556                    .iter()
7557                    .position(|g| {
7558                        let gg = g.lock();
7559                        gg.name == parent_group_path && !gg.deleted
7560                    })
7561                    .ok_or_else(|| {
7562                        crate::io::IoError::NotFound(format!(
7563                            "parent group '{parent_group_path}' not found"
7564                        ))
7565                    })?,
7566            )
7567        };
7568
7569        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7570        self.symbolic_links.lock().push(SymbolicLink {
7571            parent,
7572            name: link_name.to_string(),
7573            target,
7574            creation_seq: self.take_creation_seq(),
7575        });
7576        self.register_name(
7577            &self.link_full_path(parent, link_name),
7578            NameHit::SymbolicLink,
7579        );
7580        Ok(())
7581    }
7582
7583    // ---------------------------------------------------------- committed types
7584
7585    /// Snapshot the committed-datatype list; see [`Self::hard_links_vec`].
7586    pub(crate) fn committed_datatypes_vec(&self) -> Vec<CommittedDatatype> {
7587        self.committed_datatypes.lock().clone()
7588    }
7589
7590    /// The paths of every committed datatype a name still reaches, in
7591    /// creation order. One inside a deleted group is not among them: no link
7592    /// to it is emitted, so the file will not hold that name.
7593    ///
7594    /// Both halves of the file answer. A datatype an earlier session
7595    /// committed is carried by its bytes, not re-encoded, so it lives in the
7596    /// preserved-link list rather than the registry — and listing only the
7597    /// registry is what made this answer `[]` for a file whose every named
7598    /// type was committed before it was opened, while a reader of the same
7599    /// file named them all.
7600    pub(crate) fn committed_datatype_names(&self) -> Vec<String> {
7601        let mut out: Vec<String> = self
7602            .committed_datatypes_vec()
7603            .iter()
7604            .filter(|c| self.parent_alive(c.parent))
7605            .map(|c| c.name.clone())
7606            .collect();
7607        out.extend(
7608            self.preserved_links
7609                .lock()
7610                .iter()
7611                .filter(|l| l.kind == PreservedKind::NamedDatatype)
7612                .map(|l| self.preserved_link_full_path(l)),
7613        );
7614        out
7615    }
7616
7617    /// Commit `datatype` as an object of its own under `name` —
7618    /// `H5Tcommit2`. Returns its index in the committed-datatype registry.
7619    ///
7620    /// The object holds one datatype message and nothing else. It goes
7621    /// through [`begin_create`](Self::begin_create) like a dataset, so its
7622    /// name is resolved to a real parent group, refused if taken, and refused
7623    /// if it would cross a carried external link.
7624    pub fn commit_datatype(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
7625        let create = self.begin_create(name.trim_start_matches('/'))?;
7626        let entry = CommittedDatatype {
7627            name: create.name.clone(),
7628            parent: create.parent,
7629            datatype,
7630            creation_seq: self.take_creation_seq(),
7631            times: self.created_object_times(),
7632            obj_header_addr: 0,
7633        };
7634        let name = entry.name.clone();
7635        let idx = {
7636            let mut reg = self.committed_datatypes.lock();
7637            let idx = reg.len();
7638            reg.push(entry);
7639            idx
7640        };
7641        self.register_name(&name, NameHit::Datatype(idx));
7642        Ok(idx)
7643    }
7644
7645    /// Resolve a committed datatype's path to its registry index and the type
7646    /// it holds — the pair a dataset needs to be built on it.
7647    ///
7648    /// Returned together so the caller cannot pair one committed type's index
7649    /// with another's datatype: the dataset's element width, dataspace and
7650    /// payload checks all come from the type, and its header names the index.
7651    pub(crate) fn committed_datatype_for_share(
7652        &self,
7653        name: &str,
7654    ) -> IoResult<(usize, DatatypeMessage)> {
7655        let name = self.canonical_dataset_path(name.trim_start_matches('/'));
7656        let all = self.committed_datatypes_vec();
7657        all.iter()
7658            .position(|c| self.parent_alive(c.parent) && c.name == name)
7659            .map(|i| (i, all[i].datatype.clone()))
7660            .ok_or_else(|| {
7661                crate::io::IoError::NotFound(format!("no committed datatype named '{name}'"))
7662            })
7663    }
7664
7665    /// Record that dataset `dataset` stores its datatype as a pointer to the
7666    /// committed datatype `committed`.
7667    ///
7668    /// Takes an index [`committed_datatype_for_share`](Self::committed_datatype_for_share)
7669    /// produced, alongside the datatype from the same call, so the two cannot
7670    /// disagree and there is nothing here that can fail after the dataset
7671    /// exists.
7672    pub(crate) fn share_committed_type(&self, dataset: usize, committed: usize) {
7673        debug_assert!(committed < self.committed_datatypes.lock().len());
7674        self.ds(dataset).lock().committed_type = Some(CommittedTypeRef::Session(committed));
7675    }
7676
7677    /// How many names reach the committed datatype `index`: the link that
7678    /// gave it its name, plus every live dataset that shares it.
7679    ///
7680    /// `H5O__shared_link_adj` counts a share as a link, which is why a type
7681    /// h5py commits and then builds one dataset on reports `rc == 2`. Zero
7682    /// means nothing reaches it at all — the group holding its name was
7683    /// deleted and no dataset shares it — and then it is not written.
7684    fn committed_datatype_refcount(&self, index: usize) -> u32 {
7685        let linked = {
7686            let parent = self.committed_datatypes.lock()[index].parent;
7687            u32::from(self.parent_alive(parent))
7688        };
7689        let shares = self
7690            .dataset_refs()
7691            .iter()
7692            .filter(|d| {
7693                let m = d.lock();
7694                !m.deleted && m.committed_type == Some(CommittedTypeRef::Session(index))
7695            })
7696            .count() as u32;
7697        linked + shares
7698    }
7699
7700    /// Append the link naming each committed datatype whose parent group is
7701    /// `parent`. A committed datatype is reached by an ordinary hard link —
7702    /// what makes it a datatype rather than a group or a dataset is the one
7703    /// message in the header it points at.
7704    ///
7705    /// Only a live group's links are collected, and a live parent is itself a
7706    /// reference, so every address named here belongs to a header
7707    /// `write_committed_datatype_headers` wrote.
7708    fn push_committed_datatypes(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
7709        for cd in self.committed_datatypes_vec() {
7710            if cd.parent != parent {
7711                continue;
7712            }
7713            let leaf = cd.name.rsplit('/').next().unwrap_or(&cd.name);
7714            links.push((cd.creation_seq, LinkMessage::hard(leaf, cd.obj_header_addr)));
7715        }
7716    }
7717
7718    /// Rewrite a group path that passes through hard links into the tree
7719    /// path of the group it reaches — HDF5 traversal, where any link in a
7720    /// path component resolves to its target. Group-name form (leading
7721    /// `/`). Repeats because a substituted target's subtree can hold
7722    /// further links; bounded like libhdf5's link-traversal limit, so a
7723    /// link cycle cannot loop forever. A path with no link components
7724    /// (including one naming nothing at all) comes back unchanged.
7725    pub(crate) fn canonical_group_path(&self, path: &str) -> String {
7726        let mut path = path.to_string();
7727        for _ in 0..64 {
7728            // The longest emitted group-link path that is the whole of
7729            // `path` or a '/'-boundary prefix of it.
7730            let mut best: Option<(usize, usize)> = None; // (prefix len, target)
7731            for l in self.hard_links_vec() {
7732                let HardLinkTarget::Group(gi) = l.target else {
7733                    continue;
7734                };
7735                if !self.hard_link_emitted(&l) {
7736                    continue;
7737                }
7738                let lp = format!("/{}", self.hard_link_full_path(&l));
7739                let covers = path == lp || path.starts_with(&format!("{lp}/"));
7740                if covers && best.is_none_or(|(len, _)| lp.len() > len) {
7741                    best = Some((lp.len(), gi));
7742                }
7743            }
7744            let Some((len, gi)) = best else { break };
7745            let target_name = self.grp(gi).lock().name.clone();
7746            path = format!("{}{}", target_name, &path[len..]);
7747        }
7748        path
7749    }
7750
7751    /// [`canonical_group_path`](Self::canonical_group_path) in the
7752    /// dataset-name form (no leading `/`): the leaf is a dataset, so only
7753    /// group links can appear as components and the whole path can go
7754    /// through the group rewrite unchanged.
7755    fn canonical_dataset_path(&self, name: &str) -> String {
7756        self.canonical_group_path(&format!("/{name}"))
7757            .trim_start_matches('/')
7758            .to_string()
7759    }
7760
7761    /// Total number of hard links resolving to an object: its own tree link
7762    /// plus every emitted user-created hard link pointing at it.
7763    fn object_link_count(&self, target: HardLinkTarget) -> u32 {
7764        let same = |a: HardLinkTarget, b: HardLinkTarget| -> bool {
7765            matches!(
7766                (a, b),
7767                (HardLinkTarget::Dataset(x), HardLinkTarget::Dataset(y))
7768                    | (HardLinkTarget::Group(x), HardLinkTarget::Group(y))
7769                if x == y
7770            )
7771        };
7772        1 + self
7773            .hard_links_vec()
7774            .iter()
7775            .filter(|l| self.hard_link_emitted(l) && same(l.target, target))
7776            .count() as u32
7777    }
7778
7779    /// The object a path names, or `None` when nothing in the file does.
7780    ///
7781    /// `path` is the trimmed, hard-link-canonical form (no leading or
7782    /// trailing `/`) that dataset and group names compare against. The single
7783    /// owner of path→object resolution on the write side: hard links and
7784    /// object references must agree on what a path means, including that a
7785    /// path may itself be a user hard link — links have no chain (each points
7786    /// straight at the object header, as in libhdf5), so the existing link's
7787    /// target is the answer.
7788    pub(crate) fn resolve_object(&self, path: &str) -> Option<HardLinkTarget> {
7789        if let Some(idx) = self.dataset_refs().iter().position(|d| {
7790            let g = d.lock();
7791            !g.deleted && g.name.trim_start_matches('/') == path
7792        }) {
7793            return Some(HardLinkTarget::Dataset(idx));
7794        }
7795        if let Some(idx) = self.group_refs().iter().position(|g| {
7796            let gg = g.lock();
7797            !gg.deleted && gg.name.trim_start_matches('/') == path
7798        }) {
7799            return Some(HardLinkTarget::Group(idx));
7800        }
7801        self.hard_links_vec().iter().find_map(|l| {
7802            (self.hard_link_emitted(l) && self.hard_link_full_path(l) == path).then_some(l.target)
7803        })
7804    }
7805
7806    /// The address of dataset `index`'s own contiguous block, for the two
7807    /// writers that stamp single elements into it by file offset — object and
7808    /// region references, whose values are only known once finalize has placed
7809    /// every object header.
7810    ///
7811    /// Refuses, rather than handing back an address that is not one, every
7812    /// dataset that has no such block: chunked, compact, unallocated, or with
7813    /// its raw data in files outside this one.
7814    fn local_element_block(&self, index: usize, what: &str) -> IoResult<u64> {
7815        let ds = self.ds(index);
7816        let m = ds.lock();
7817        match m.contiguous_target() {
7818            Some(ContiguousTarget::Local(addr)) => Ok(addr),
7819            Some(ContiguousTarget::External { .. }) => {
7820                Err(crate::io::IoError::InvalidState(format!(
7821                    "{what} are stamped into the dataset's own contiguous block, and \
7822                 dataset '{}' has none: its raw data lives in external files",
7823                    m.name
7824                )))
7825            }
7826            Some(ContiguousTarget::Virtual) => Err(crate::io::IoError::InvalidState(format!(
7827                "{what} are stamped into the dataset's own contiguous block, and \
7828                 dataset '{}' has none: it is virtual, and its elements come from \
7829                 the source datasets its mappings name",
7830                m.name
7831            ))),
7832            None => Err(crate::io::IoError::InvalidState(format!(
7833                "{what} are stamped into contiguous storage; create the dataset \
7834                 without chunking"
7835            ))),
7836        }
7837    }
7838
7839    /// Store object references naming `paths` into the elements of dataset
7840    /// `index` starting at `start`.
7841    ///
7842    /// The value of an `H5R_OBJECT1` element is its target's object header
7843    /// address, which finalize assigns, so what lands here is the target path;
7844    /// [`Self::write_object_reference_values`] writes the addresses. Elements
7845    /// never written keep the zero image libhdf5 reads back as a null
7846    /// reference.
7847    pub fn write_object_references(
7848        &self,
7849        index: usize,
7850        start: u64,
7851        paths: &[&str],
7852    ) -> IoResult<()> {
7853        let elements = {
7854            let ds = self.ds(index);
7855            let m = ds.lock();
7856            match &m.datatype {
7857                // Both generations of object reference: `H5T_STD_REF_OBJ` and
7858                // the 1.12 `H5T_STD_REF`. They differ only in the element
7859                // image, which `encode_reference_element` owns.
7860                DatatypeMessage::Reference {
7861                    kind: ReferenceKind::Object1 | ReferenceKind::Object2,
7862                    ..
7863                } => {}
7864                other => {
7865                    return Err(crate::io::IoError::InvalidState(format!(
7866                        "dataset '{}' has datatype {other}, not an object reference",
7867                        m.name
7868                    )))
7869                }
7870            }
7871            m.dataspace
7872                .dims
7873                .iter()
7874                .fold(1u64, |a, &d| a.saturating_mul(d))
7875        };
7876        // Refused here as well as at fixup time, so a dataset whose storage
7877        // cannot hold stamped elements is reported at the call that chose it.
7878        self.local_element_block(index, "object references")?;
7879        let end = start.saturating_add(paths.len() as u64);
7880        if end > elements {
7881            return Err(crate::io::IoError::InvalidState(format!(
7882                "elements {start}..{end} are outside the dataset's {elements}"
7883            )));
7884        }
7885        // Resolve now as well as at fixup time, so a path that names nothing
7886        // is reported at the call that got it wrong.
7887        for path in paths {
7888            self.object_reference_target(path)?;
7889        }
7890        let mut pending = self.pending_object_references.lock();
7891        for (i, path) in paths.iter().enumerate() {
7892            pending.push(PendingObjectReference {
7893                dataset: index,
7894                element: start + i as u64,
7895                target: (*path).to_string(),
7896            });
7897        }
7898        Ok(())
7899    }
7900
7901    /// Record a hard link count of `rc` in `header`, if this file's format
7902    /// needs a message to carry it.
7903    ///
7904    /// A version-2 header carries the count in an Object Reference Count
7905    /// message, and only when more than one link reaches the object. A
7906    /// version-1 header carries it in its prefix and gets no message at all —
7907    /// `H5O_link_oh` gates every refcount-message operation on
7908    /// `oh->version > H5O_VERSION_1` (H5Oint.c:851), so a version-1 header
7909    /// holding one is a shape libhdf5 never writes.
7910    ///
7911    /// The message carries `H5O_MSG_FLAG_DONTSHARE`, which both refcount
7912    /// operations pass (H5Oint.c:874 append, H5Oint.c:864 write): the count is
7913    /// a property of this one object header, so a shared-message index that
7914    /// pointed several headers at one copy would make every object with the
7915    /// same link count share a single number.
7916    fn emit_refcount(&self, header: &mut ObjectHeader, rc: u32, format: ObjectFormat) {
7917        if rc > 1 && format == ObjectFormat::Modern {
7918            header.add_message(MSG_OBJ_REF_COUNT, MSG_FLAG_DONTSHARE, encode_refcount(rc));
7919        }
7920    }
7921
7922    /// Encode an object header for the block at `addr`, at the version this
7923    /// file's format calls for and with `rc` as the object's hard link count.
7924    ///
7925    /// The count is passed rather than read off the header because the two
7926    /// versions carry it in different places — the version-1 prefix's `nlink`
7927    /// field, the version-2 Reference Count message
7928    /// [`emit_refcount`](Self::emit_refcount) already added — and only the
7929    /// caller knows it.
7930    ///
7931    /// INVARIANT: every chunk of an object header lives in the one block its
7932    /// address and encoded size describe. A header whose messages overflow
7933    /// chunk 0 gets a continuation chunk immediately behind it in that same
7934    /// block, so the address is enough to free, relocate or supersede the
7935    /// whole header — which is what every caller already assumes. libhdf5
7936    /// would have grown chunk 0 into space that free rather than chaining
7937    /// onto it, but it reads a continuation chunk by the address and length
7938    /// its message states and cares nothing for where that lands.
7939    fn encode_header_at(
7940        &self,
7941        header: &ObjectHeader,
7942        rc: u32,
7943        format: ObjectFormat,
7944        addr: u64,
7945    ) -> IoResult<Vec<u8>> {
7946        if format == ObjectFormat::Legacy {
7947            return Ok(header.encode_for(format, rc)?);
7948        }
7949        let plan = header.plan_chunks(self.chunk0_capacity(header), &self.ctx)?;
7950        let (mut image, continuation) =
7951            header.encode_chunked(&plan, &self.ctx, addr + plan.chunk0_size as u64)?;
7952        if let Some(chunk) = continuation {
7953            image.extend_from_slice(&chunk);
7954        }
7955        Ok(image)
7956    }
7957
7958    /// The bytes [`encode_header_at`](Self::encode_header_at) will produce for
7959    /// `header`, without an address and without producing them.
7960    ///
7961    /// A header's encoded size does not depend on the addresses it carries,
7962    /// which is what lets the group pass hand every group header an address
7963    /// before it writes any of their content.
7964    fn header_encoded_size(
7965        &self,
7966        header: &ObjectHeader,
7967        rc: u32,
7968        format: ObjectFormat,
7969    ) -> IoResult<usize> {
7970        if format == ObjectFormat::Legacy {
7971            return Ok(header.encode_for(format, rc)?.len());
7972        }
7973        let plan = header.plan_chunks(self.chunk0_capacity(header), &self.ctx)?;
7974        Ok(plan.chunk0_size + plan.continuation_size)
7975    }
7976
7977    /// How many bytes of messages `header`'s chunk 0 holds before the rest
7978    /// spill into a continuation chunk.
7979    ///
7980    /// libhdf5 sizes chunk 0 once, when the object header is created, and can
7981    /// only grow it while the space behind it is still free — so an object
7982    /// whose creation-time estimate covered every message it would ever hold
7983    /// keeps one chunk, and one whose estimate was a guess does not. A dataset
7984    /// or a committed datatype is created from messages already in hand
7985    /// (`H5D__update_oh_info`, `H5T__commit`), so its estimate is exact and
7986    /// this writer's exact fit is the same answer.
7987    ///
7988    /// A group is the exception: `H5G__obj_create_real` (H5Gobj.c:219) sizes
7989    /// its header for the link info and group info messages plus
7990    /// `H5G_CRT_GINFO_EST_NUM_ENTRIES` links of `H5G_CRT_GINFO_EST_NAME_LEN`
7991    /// characters, and nothing else — attributes above all — is in that
7992    /// estimate. The Link Info message is what identifies one: it is the
7993    /// message that makes an object a new-format group, and
7994    /// `H5G__obj_get_linfo` uses it for exactly this question.
7995    fn chunk0_capacity(&self, header: &ObjectHeader) -> usize {
7996        let envelope = header.message_envelope_size();
7997        let sized = |msg_type: u8| {
7998            header
7999                .messages
8000                .iter()
8001                .find(|m| m.msg_type == msg_type)
8002                .map(|m| envelope + m.data.len())
8003        };
8004        let Some(link_info) = sized(MSG_LINK_INFO) else {
8005            return usize::MAX;
8006        };
8007        // One estimated hard link: version, flags, a one-byte name length for
8008        // a name this short, the name, and the object header address.
8009        let link = envelope + 1 + 1 + 1 + EST_LINK_NAME_LEN + self.ctx.sizeof_addr as usize;
8010        link_info + sized(MSG_GROUP_INFO).unwrap_or(0) + EST_LINK_COUNT * link
8011    }
8012
8013    /// The object an object reference's path names, as a hard-link target;
8014    /// `None` for the root group, which has no registry slot.
8015    fn object_reference_target(&self, path: &str) -> IoResult<Option<HardLinkTarget>> {
8016        let rel = self.canonical_dataset_path(path.trim_matches('/'));
8017        if rel.is_empty() {
8018            return Ok(None);
8019        }
8020        self.resolve_object(&rel)
8021            .map(Some)
8022            .ok_or_else(|| crate::io::IoError::NotFound(format!("reference target '{path}'")))
8023    }
8024
8025    /// The object header address an object reference's `path` names, or zero
8026    /// when that object has not been given one yet.
8027    ///
8028    /// Zero is where the superblock sits, so it is never an object header's
8029    /// address. It is what every object reads as before
8030    /// [`allocate_object_headers`](Self::allocate_object_headers) runs, which
8031    /// is what lets the pass that measures a header stand in for the pass that
8032    /// writes it: an address is a fixed-width field, so the placeholder is the
8033    /// same size as the answer.
8034    fn object_reference_address(&self, path: &str) -> IoResult<u64> {
8035        Ok(match self.object_reference_target(path)? {
8036            Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8037            Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8038            None => self.root_group_addr.unwrap_or(0),
8039        })
8040    }
8041
8042    /// `scope`'s attributes as this finalize will write them: the stored set,
8043    /// with every object-reference attribute's value said in the object header
8044    /// addresses assigned so far.
8045    ///
8046    /// The single owner of a reference attribute's value, and the only source
8047    /// an object header build may take an attribute set from. Nothing stored
8048    /// is mutated, so the pass that measures a header and the pass that writes
8049    /// it cannot disagree about anything but the addresses — which they cannot
8050    /// disagree about in length.
8051    ///
8052    /// INVARIANT: the stored attribute list is what says which attributes
8053    /// exist; a recorded reference value can only give a value to one already
8054    /// in it. So a value left behind by an object whose list was emptied — a
8055    /// deleted group or dataset — cannot put the attribute back, and a value
8056    /// whose attribute was replaced by one of another type is dropped at the
8057    /// replacement instead of reaching it (see
8058    /// [`forget_attribute_reference`](Self::forget_attribute_reference)).
8059    fn object_attributes(&self, scope: AttrScope) -> IoResult<Vec<AttributeEntry>> {
8060        let mut attrs = match scope {
8061            AttrScope::Root => self.root_attributes.lock().clone(),
8062            AttrScope::Group(gi) => self.grp(gi).lock().attributes.clone(),
8063            AttrScope::Dataset(i) => self.ds(i).lock().attributes.clone(),
8064        };
8065        // Snapshot first: resolving a path locks group and dataset slots.
8066        let values: Vec<(String, Vec<String>)> = self
8067            .attribute_references
8068            .lock()
8069            .iter()
8070            .filter(|r| r.scope == scope)
8071            .map(|r| (r.name.clone(), r.targets.clone()))
8072            .collect();
8073        let width = self.ctx.sizeof_addr as usize;
8074        for (name, targets) in values {
8075            let Some(pos) = attrs.iter().position(|a| a.name() == name) else {
8076                continue;
8077            };
8078            let Some(msg) = attrs[pos].readable() else {
8079                continue;
8080            };
8081            let mut msg = msg.clone();
8082            let mut data = Vec::with_capacity(targets.len() * width);
8083            for target in &targets {
8084                data.extend_from_slice(
8085                    &self.object_reference_address(target)?.to_le_bytes()[..width],
8086                );
8087            }
8088            msg.data = data;
8089            attrs[pos] = AttributeEntry::from(msg).with_creation_index(attrs[pos].creation_index());
8090        }
8091        Ok(attrs)
8092    }
8093
8094    /// The registry scope `target` names — the same object
8095    /// [`with_attr_list`](Self::with_attr_list) reaches, as the key the
8096    /// reference-value registry is indexed by. Refuses what that accessor
8097    /// refuses, and for the same reasons.
8098    fn attr_scope(&self, target: AttrTarget<'_>) -> IoResult<AttrScope> {
8099        match target {
8100            AttrTarget::Root => Ok(AttrScope::Root),
8101            AttrTarget::Group(path) => {
8102                let path = self.canonical_group_path(path);
8103                self.group_refs()
8104                    .iter()
8105                    .position(|g| {
8106                        let gg = g.lock();
8107                        gg.name == path && !gg.deleted
8108                    })
8109                    .map(AttrScope::Group)
8110                    .ok_or_else(|| {
8111                        crate::io::IoError::NotFound(format!("group '{path}' not found"))
8112                    })
8113            }
8114            AttrTarget::Dataset(index) => {
8115                let count = self.dataset_count();
8116                if index >= count {
8117                    return Err(crate::io::IoError::InvalidState(format!(
8118                        "dataset index {index} out of range (have {count})"
8119                    )));
8120                }
8121                Ok(AttrScope::Dataset(index))
8122            }
8123        }
8124    }
8125
8126    /// Drop the reference value recorded for `scope`'s attribute `name`.
8127    ///
8128    /// Called by both owners of attribute-list mutation —
8129    /// [`insert_attribute`](Self::insert_attribute) and
8130    /// [`evict_attr`](Self::evict_attr) — so an attribute that is replaced or
8131    /// removed cannot leave its value behind for whatever takes its name next.
8132    /// A string attribute written over a reference attribute is the case that
8133    /// needs it: without this the string's bytes would be overwritten with
8134    /// addresses at finalize.
8135    fn forget_attribute_reference(&self, scope: AttrScope, name: &str) {
8136        self.attribute_references
8137            .lock()
8138            .retain(|r| !(r.scope == scope && r.name == name));
8139    }
8140
8141    /// Write every pending object reference element as its target's object
8142    /// header address.
8143    ///
8144    /// INVARIANT: a reference element on disk holds its target's header
8145    /// address. Reached through [`write_reference_values`](Self::write_reference_values),
8146    /// which places it after every header has an address; a target that no
8147    /// longer resolves fails the finalize rather than leaving a placeholder
8148    /// behind.
8149    fn write_object_reference_values(&mut self) -> IoResult<()> {
8150        // Snapshot rather than drain: a SWMR session finalizes twice, and the
8151        // close-time finalize rebuilds every header at a fresh address, so the
8152        // elements must be stamped again with the addresses that survive.
8153        let pending: Vec<(usize, u64, String)> = self
8154            .pending_object_references
8155            .lock()
8156            .iter()
8157            .map(|p| (p.dataset, p.element, p.target.clone()))
8158            .collect();
8159        for (dataset, element, target) in &pending {
8160            let addr = match self.object_reference_target(target)? {
8161                Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8162                Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8163                None => self.root_group_addr.ok_or_else(|| {
8164                    crate::io::IoError::InvalidState(
8165                        "root group header address is not assigned yet".into(),
8166                    )
8167                })?,
8168            };
8169            // The element image is the dataset's own datatype's business: the
8170            // pre-1.12 and 1.12 forms differ in width and in layout, and the
8171            // dataset says which it holds.
8172            let (kind, width) = {
8173                let ds = self.ds(*dataset);
8174                let m = ds.lock();
8175                let DatatypeMessage::Reference { kind, size } = &m.datatype else {
8176                    return Err(crate::io::IoError::InvalidState(format!(
8177                        "dataset '{}' is no longer a reference dataset",
8178                        m.name
8179                    )));
8180                };
8181                (*kind, *size as usize)
8182            };
8183            let image = match kind {
8184                ReferenceKind::Object1 => ReferenceElementImage::Legacy(addr),
8185                ReferenceKind::Object2 => ReferenceElementImage::Inline(addr),
8186                other => {
8187                    return Err(crate::io::IoError::InvalidState(format!(
8188                        "dataset {dataset} now holds {other:?} elements, not object references"
8189                    )))
8190                }
8191            };
8192            let image = encode_reference_element(&image, width, &self.ctx)?;
8193            let data_addr = self.local_element_block(*dataset, "object references")?;
8194            let at = data_addr + element * width as u64;
8195            self.handle.write_at(at, &image)?;
8196        }
8197        Ok(())
8198    }
8199
8200    /// Store region references over `targets` into the elements of dataset
8201    /// `index` starting at `start`.
8202    ///
8203    /// Each target is the path of a dataset and a selection over it. What the
8204    /// element holds is a global-heap id — collection address then object index
8205    /// (`H5R__encode_heap`) — and the heap object it names is the target's
8206    /// object header address followed by the serialized selection
8207    /// (`H5R__encode_token_region_compat`). Both the object and the element are
8208    /// written here; only the address inside the object waits for
8209    /// [`Self::write_heap_reference_values`]. Elements never written keep the
8210    /// zero image libhdf5 reads back as a null reference.
8211    pub fn write_region_references(
8212        &self,
8213        index: usize,
8214        start: u64,
8215        targets: &[(&str, Selection)],
8216    ) -> IoResult<()> {
8217        let elements = {
8218            let ds = self.ds(index);
8219            let m = ds.lock();
8220            match &m.datatype {
8221                DatatypeMessage::Reference {
8222                    kind: ReferenceKind::DatasetRegion1,
8223                    ..
8224                } => {}
8225                other => {
8226                    return Err(crate::io::IoError::InvalidState(format!(
8227                        "dataset '{}' has datatype {other}, not a region reference",
8228                        m.name
8229                    )))
8230                }
8231            }
8232            m.dataspace
8233                .dims
8234                .iter()
8235                .fold(1u64, |a, &d| a.saturating_mul(d))
8236        };
8237        let data_addr = self.local_element_block(index, "region references")?;
8238        let end = start.saturating_add(targets.len() as u64);
8239        if end > elements {
8240            return Err(crate::io::IoError::InvalidState(format!(
8241                "elements {start}..{end} are outside the dataset's {elements}"
8242            )));
8243        }
8244
8245        // Build every heap object before inserting any: a path that names no
8246        // dataset, or a selection its extent does not admit, is reported at the
8247        // call that got it wrong rather than after half the batch is on disk.
8248        let sa = self.ctx.sizeof_addr as usize;
8249        let mut blobs = Vec::with_capacity(targets.len());
8250        for (path, selection) in targets {
8251            let target = self.region_reference_target(path)?;
8252            let dims = self.ds(target).lock().dataspace.dims.clone();
8253            validate_region_selection(selection, &dims, path)?;
8254            let mut blob = vec![0u8; sa];
8255            blob.extend_from_slice(&selection.encode()?);
8256            blobs.push(blob);
8257        }
8258        let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
8259        let placements = self.insert_vlen_objects(&items)?;
8260
8261        let width = (sa + 4) as u64;
8262        let mut pending = self.pending_heap_references.lock();
8263        for (i, &(collection, obj_index)) in placements.iter().enumerate() {
8264            let mut elem = Vec::with_capacity(width as usize);
8265            elem.extend_from_slice(&collection.to_le_bytes()[..sa]);
8266            elem.extend_from_slice(&u32::from(obj_index).to_le_bytes());
8267            self.handle
8268                .write_at(data_addr + (start + i as u64) * width, &elem)?;
8269            pending.push(PendingHeapReference {
8270                collection,
8271                index: obj_index,
8272                token_offset: 0,
8273                target: PendingHeapTarget::Dataset(targets[i].0.to_string()),
8274            });
8275        }
8276        Ok(())
8277    }
8278
8279    /// Store 1.12 references over `targets` into the elements of dataset
8280    /// `index` starting at `start` — the `H5T_STD_REF` trio.
8281    ///
8282    /// One datatype holds all three kinds, because a 1.12 element leads with
8283    /// the kind it holds; which is why this takes a [`ReferenceTarget`] per
8284    /// element rather than a fixed kind. `H5R_OBJECT2` needs nothing but the
8285    /// target's address, so its element is written inline by the same finalize
8286    /// pass every object reference goes through. The other two encode a
8287    /// selection or an attribute name alongside the token, which does not fit
8288    /// an element, so what is stored is a global-heap blob and the element is
8289    /// its id (`H5T__ref_disk_write`). Elements never written keep the zero
8290    /// image `H5T__ref_disk_isnull` reads back as a null reference.
8291    pub fn write_revised_references(
8292        &self,
8293        index: usize,
8294        start: u64,
8295        targets: &[(&str, ReferenceTarget)],
8296    ) -> IoResult<()> {
8297        let (width, elements) = {
8298            let ds = self.ds(index);
8299            let m = ds.lock();
8300            match &m.datatype {
8301                DatatypeMessage::Reference {
8302                    kind: ReferenceKind::Object2,
8303                    size,
8304                } => (
8305                    *size as u64,
8306                    m.dataspace
8307                        .dims
8308                        .iter()
8309                        .fold(1u64, |a, &d| a.saturating_mul(d)),
8310                ),
8311                other => {
8312                    return Err(crate::io::IoError::InvalidState(format!(
8313                        "dataset '{}' has datatype {other}, not the 1.12 H5T_STD_REF",
8314                        m.name
8315                    )))
8316                }
8317            }
8318        };
8319        let data_addr = self.local_element_block(index, "references")?;
8320        let end = start.saturating_add(targets.len() as u64);
8321        if end > elements {
8322            return Err(crate::io::IoError::InvalidState(format!(
8323                "elements {start}..{end} are outside the dataset's {elements}"
8324            )));
8325        }
8326
8327        // Build every blob before inserting any, so a path that names nothing,
8328        // a selection an extent does not admit or an attribute that does not
8329        // exist is reported at the call that got it wrong rather than after
8330        // half the batch is on disk.
8331        let mut blobs: Vec<(u64, ReferenceKind, PendingHeapTarget, Vec<u8>)> = Vec::new();
8332        let mut inline: Vec<(u64, String)> = Vec::new();
8333        for (i, (path, target)) in targets.iter().enumerate() {
8334            let element = start + i as u64;
8335            // The rank of the extent the selection is over, which only a region
8336            // reference encodes and takes from the target's dataspace.
8337            let mut extent_rank = 0;
8338            let (kind, pending) = match target {
8339                ReferenceTarget::Object => {
8340                    self.object_reference_target(path)?;
8341                    inline.push((element, (*path).to_string()));
8342                    continue;
8343                }
8344                ReferenceTarget::Region(selection) => {
8345                    let ds = self.region_reference_target(path)?;
8346                    let dims = self.ds(ds).lock().dataspace.dims.clone();
8347                    validate_region_selection(selection, &dims, path)?;
8348                    extent_rank = dims.len();
8349                    (
8350                        ReferenceKind::DatasetRegion2,
8351                        PendingHeapTarget::Dataset((*path).to_string()),
8352                    )
8353                }
8354                ReferenceTarget::Attribute(name) => {
8355                    let scope = match self.object_reference_target(path)? {
8356                        Some(HardLinkTarget::Dataset(i)) => AttrScope::Dataset(i),
8357                        Some(HardLinkTarget::Group(i)) => AttrScope::Group(i),
8358                        None => AttrScope::Root,
8359                    };
8360                    if !self
8361                        .object_attributes(scope)?
8362                        .iter()
8363                        .any(|a| a.name() == name)
8364                    {
8365                        return Err(crate::io::IoError::NotFound(format!(
8366                            "attribute '{name}' of reference target '{path}'"
8367                        )));
8368                    }
8369                    (
8370                        ReferenceKind::Attr,
8371                        PendingHeapTarget::Object((*path).to_string()),
8372                    )
8373                }
8374            };
8375            blobs.push((
8376                element,
8377                kind,
8378                pending,
8379                encode_revised_blob(0, target, extent_rank, &self.ctx)?,
8380            ));
8381        }
8382
8383        let items: Vec<&[u8]> = blobs.iter().map(|(_, _, _, b)| b.as_slice()).collect();
8384        let placements = self.insert_vlen_objects(&items)?;
8385
8386        let mut pending = self.pending_heap_references.lock();
8387        for ((element, kind, target, blob), &(collection, obj_index)) in
8388            blobs.iter().zip(&placements)
8389        {
8390            // The size the element declares is the heap object's own byte
8391            // count: `H5VL__native_blob_get` refuses to read one whose size
8392            // does not match what the element says.
8393            let image = encode_reference_element(
8394                &ReferenceElementImage::Blob {
8395                    kind: *kind,
8396                    size: blob.len() as u32,
8397                    collection,
8398                    index: u32::from(obj_index),
8399                },
8400                width as usize,
8401                &self.ctx,
8402            )?;
8403            self.handle.write_at(data_addr + element * width, &image)?;
8404            pending.push(PendingHeapReference {
8405                collection,
8406                index: obj_index,
8407                token_offset: REVISED_BLOB_TOKEN_OFFSET,
8408                target: target.clone(),
8409            });
8410        }
8411        drop(pending);
8412
8413        let mut pending = self.pending_object_references.lock();
8414        for (element, path) in inline {
8415            pending.push(PendingObjectReference {
8416                dataset: index,
8417                element,
8418                target: path,
8419            });
8420        }
8421        Ok(())
8422    }
8423
8424    /// The dataset a region reference's path names.
8425    ///
8426    /// A region reference names a *dataset*: `H5Rcreate` with
8427    /// `H5R_DATASET_REGION` takes the dataspace of one, and every reader
8428    /// dereferences it as one. A path that resolves to a group — or to the root
8429    /// group, which has no registry slot — is refused here rather than stored
8430    /// as a reference nothing can dereference.
8431    fn region_reference_target(&self, path: &str) -> IoResult<usize> {
8432        match self.object_reference_target(path)? {
8433            Some(HardLinkTarget::Dataset(i)) => Ok(i),
8434            _ => Err(crate::io::IoError::InvalidState(format!(
8435                "region reference target '{path}' is not a dataset"
8436            ))),
8437        }
8438    }
8439
8440    /// Stamp every pending heap-backed reference's object with its target's
8441    /// object header address.
8442    ///
8443    /// The references that are still stamped rather than written once: the
8444    /// *element* is a global-heap id, so the heap object has to exist at the
8445    /// call that stores the reference, long before any address does. The object
8446    /// was inserted with its token zeroed, so its size does not change here:
8447    /// each collection is read once, patched, and rewritten at its own declared
8448    /// size, which leaves every element's heap id valid — and leaves the
8449    /// object's byte count equal to the size the 1.12 element declares, which
8450    /// `H5VL__native_blob_get` refuses to read past.
8451    fn write_heap_reference_values(&mut self) -> IoResult<()> {
8452        use crate::format::global_heap::GlobalHeapCollection;
8453
8454        // Snapshot rather than drain, for the same reason the object-reference
8455        // pass does: a SWMR session finalizes twice and the close-time finalize
8456        // rebuilds every header at a fresh address.
8457        let pending: Vec<(u64, u16, usize, PendingHeapTarget)> = self
8458            .pending_heap_references
8459            .lock()
8460            .iter()
8461            .map(|p| (p.collection, p.index, p.token_offset, p.target.clone()))
8462            .collect();
8463        if pending.is_empty() {
8464            return Ok(());
8465        }
8466        let sa = self.ctx.sizeof_addr as usize;
8467        // Group by collection so one holding several references is read and
8468        // rewritten once.
8469        let mut per_collection: std::collections::BTreeMap<u64, Vec<(u16, usize, u64)>> =
8470            Default::default();
8471        for (collection, index, token_offset, target) in &pending {
8472            let addr = match target {
8473                PendingHeapTarget::Dataset(path) => {
8474                    let ds = self.region_reference_target(path)?;
8475                    self.ds(ds).lock().obj_header_addr
8476                }
8477                PendingHeapTarget::Object(path) => self.object_reference_address(path)?,
8478            };
8479            per_collection
8480                .entry(*collection)
8481                .or_default()
8482                .push((*index, *token_offset, addr));
8483        }
8484        for (collection, patches) in per_collection {
8485            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
8486            // exactly that, so one read usually covers the whole image.
8487            let mut image = self.handle.read_at_most(collection, 4096)?;
8488            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
8489            if declared > image.len() {
8490                image = self.handle.read_at(collection, declared)?;
8491            }
8492            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
8493            for (index, token_offset, addr) in patches {
8494                let token = gcol
8495                    .objects
8496                    .iter_mut()
8497                    .find(|o| o.index == index)
8498                    .and_then(|o| o.data.get_mut(token_offset..token_offset + sa))
8499                    .ok_or_else(|| {
8500                        crate::io::IoError::InvalidState(format!(
8501                            "object {index} of global heap collection {collection:#x} is no \
8502                             longer the reference written into it"
8503                        ))
8504                    })?;
8505                token.copy_from_slice(&addr.to_le_bytes()[..sa]);
8506            }
8507            let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
8508            self.handle.write_at(collection, &rewritten)?;
8509        }
8510        Ok(())
8511    }
8512
8513    /// Give every reference written this session its target's object header
8514    /// address.
8515    ///
8516    /// INVARIANT: no file is closed holding a reference whose target address is
8517    /// still the placeholder its write left. Both finalize paths call this in
8518    /// the content phase — after
8519    /// [`allocate_object_headers`](Self::allocate_object_headers), so every
8520    /// address exists, and before any object header is written — and this is
8521    /// the only caller of the per-kind passes, so a reference kind added later
8522    /// is written at both finalize sites or at neither. A target that no longer
8523    /// resolves fails the finalize rather than leaving a placeholder behind.
8524    ///
8525    /// This covers the two reference kinds whose value lives outside an object
8526    /// header. An attribute's value lives *inside* one, so it has no pass here:
8527    /// [`object_attributes`](Self::object_attributes) says it in addresses as
8528    /// the header is built.
8529    fn write_reference_values(&mut self) -> IoResult<()> {
8530        self.write_object_reference_values()?;
8531        self.write_heap_reference_values()
8532    }
8533
8534    /// Append every user-created hard link whose parent group is `parent`
8535    /// (`None` == the root group). Called while collecting a group's links,
8536    /// once every object's header address has been assigned.
8537    fn push_hard_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8538        for link in self.hard_links_vec() {
8539            if link.parent != parent || !self.hard_link_emitted(&link) {
8540                continue;
8541            }
8542            let addr = match link.target {
8543                HardLinkTarget::Dataset(i) => self.ds(i).lock().obj_header_addr,
8544                HardLinkTarget::Group(i) => self.grp(i).lock().obj_header_addr,
8545            };
8546            links.push((link.creation_seq, LinkMessage::hard(&link.name, addr)));
8547        }
8548    }
8549
8550    /// Append every user-created symbolic link whose parent group is `parent`
8551    /// (`None` == the root group).
8552    ///
8553    /// Nothing here waits on the layout pass — the link's value is a path, not
8554    /// an address — but it is collected with the rest so it takes its place in
8555    /// creation order and counts toward the phase change.
8556    fn push_symbolic_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8557        for link in self.symbolic_links_vec() {
8558            if link.parent != parent || !self.symbolic_link_emitted(&link) {
8559                continue;
8560            }
8561            links.push((
8562                link.creation_seq,
8563                LinkMessage {
8564                    name: link.name.clone(),
8565                    target: link.target.clone(),
8566                    creation_order: None,
8567                    cset: CharacterSet::for_name(&link.name),
8568                },
8569            ));
8570        }
8571    }
8572
8573    /// Refuse a caller path that would have to leave this file through one of
8574    /// the external links a reopened file brought in.
8575    ///
8576    /// The reader follows such a path into the file the link names; the writer
8577    /// cannot, because it models one file and would have to write into
8578    /// another. Saying which link stops the path — rather than reporting the
8579    /// name as absent, or worse, creating a second link of that name beside
8580    /// it — is the whole of what write mode does here.
8581    pub(crate) fn reject_external_traversal(&self, path: &str) -> IoResult<()> {
8582        let path = path.trim_start_matches('/');
8583        let crossing = self.preserved_link_paths().into_iter().find(|(p, class)| {
8584            matches!(class, crate::io::reader::LinkClass::External { .. })
8585                && (path == p || path.starts_with(&format!("{p}/")))
8586        });
8587        match crossing {
8588            None => Ok(()),
8589            Some((link, crate::io::reader::LinkClass::External { file, path: target })) => {
8590                Err(crate::io::IoError::Unsupported(format!(
8591                    "'{path}' resolves through the external link '{link}' to '{target}' in \
8592                     '{file}'; this writer carries external links through a rewrite but does \
8593                     not open the file they name"
8594                )))
8595            }
8596            // `find` matched on the External arm, so no other class reaches here.
8597            Some(_) => Ok(()),
8598        }
8599    }
8600
8601    /// Resolve `name` to a live dataset index, reporting *why* it does not
8602    /// resolve rather than collapsing every cause into absence.
8603    ///
8604    /// The write-mode counterpart of [`Hdf5Reader::open_dataset`]: the single
8605    /// gate every by-name dataset lookup in write mode goes through.
8606    ///
8607    /// [`Hdf5Reader::open_dataset`]: crate::io::reader::Hdf5Reader::open_dataset
8608    pub(crate) fn open_dataset_index(&self, name: &str) -> IoResult<usize> {
8609        self.reject_external_traversal(name)?;
8610        self.reject_preserved_object(name)?;
8611        self.dataset_index(name)
8612            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
8613    }
8614
8615    /// Refuse a caller path that names an object the reopen kept by its bytes
8616    /// rather than modelling.
8617    ///
8618    /// Such an object is in the file and stays in it, but this writer holds
8619    /// none of what it would need to read or rewrite it. Saying so — with the
8620    /// reason the classification recorded — is the difference between an
8621    /// object the writer will not touch and a name the file does not have.
8622    pub(crate) fn reject_preserved_object(&self, path: &str) -> IoResult<()> {
8623        let path = path.trim_start_matches('/');
8624        let objects: Vec<(String, String)> = {
8625            let preserved = self.preserved_links.lock();
8626            preserved
8627                .iter()
8628                .filter_map(|l| {
8629                    l.reason
8630                        .as_ref()
8631                        .map(|why| (self.preserved_link_full_path(l), why.clone()))
8632                })
8633                .collect()
8634        };
8635        match objects
8636            .into_iter()
8637            .find(|(full, _)| path == full || path.starts_with(&format!("{full}/")))
8638        {
8639            None => Ok(()),
8640            Some((link, why)) => Err(crate::io::IoError::Unsupported(format!(
8641                "'{path}' is, or is inside, the object '{link}', which this file's reopen \
8642                 kept exactly as it found it because {why}"
8643            ))),
8644        }
8645    }
8646
8647    /// Every link this writer will emit that names a *path* rather than an
8648    /// object, with the class a listing reports for it: the soft and external
8649    /// links created this session, and the ones a reopen is carrying through.
8650    ///
8651    /// The object listings answer for hard links, so a write-mode link
8652    /// listing is this plus those; keeping both sources in one place is what
8653    /// stops a listing from seeing a kind the class lookup does not, or the
8654    /// reverse.
8655    pub(crate) fn path_link_classes(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8656        let mut out: Vec<(String, crate::io::reader::LinkClass)> = self
8657            .symbolic_links_vec()
8658            .iter()
8659            .filter(|l| self.symbolic_link_emitted(l))
8660            .map(|l| {
8661                (
8662                    self.symbolic_link_full_path(l),
8663                    crate::io::reader::LinkClass::from_target(&l.target),
8664                )
8665            })
8666            .collect();
8667        out.extend(self.preserved_link_paths());
8668        out
8669    }
8670
8671    /// Every link this writer is carrying but cannot express, by full path.
8672    pub(crate) fn preserved_link_paths(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8673        self.preserved_links
8674            .lock()
8675            .iter()
8676            .map(|l| (self.preserved_link_full_path(l), l.class.clone()))
8677            .collect()
8678    }
8679
8680    /// The full path of a preserved link: its parent group's path plus its
8681    /// leaf name, in the no-leading-`/` form the registry uses.
8682    fn preserved_link_full_path(&self, link: &PreservedLink) -> String {
8683        match link.parent {
8684            None => link.name.clone(),
8685            Some(gi) => {
8686                let group = self.grp(gi).lock().name.clone();
8687                format!("{}/{}", group.trim_start_matches('/'), link.name)
8688            }
8689        }
8690    }
8691
8692    /// The single owner of "which links does this group hold", in the order
8693    /// they were created and, when the file tracks creation order, stamped
8694    /// with it.
8695    ///
8696    /// Both the compact form (one `MSG_LINK` per link) and the dense form (the
8697    /// same messages inside a fractal heap) are built from this one list, so
8698    /// the phase-change decision, the storage it selects and the creation
8699    /// order recorded in either can never disagree about what the group
8700    /// contains.
8701    fn group_links(&self, scope: LinkScope, order: CreationOrder) -> Vec<LinkMessage> {
8702        let mut links: Vec<(u64, LinkMessage)> = Vec::new();
8703        match scope {
8704            LinkScope::Root => {
8705                // Datasets that belong to a subgroup are that group's links,
8706                // not the root's. Each group slot is locked one at a time.
8707                let mut datasets_in_subgroups: std::collections::HashSet<usize> =
8708                    std::collections::HashSet::new();
8709                for grp in self.group_refs() {
8710                    let g = grp.lock();
8711                    if g.deleted {
8712                        continue;
8713                    }
8714                    datasets_in_subgroups.extend(g.child_datasets.iter().copied());
8715                }
8716                // `dataset_refs` preserves registry order, so `enumerate`
8717                // yields each dataset's true index.
8718                for (i, ds) in self.dataset_refs().into_iter().enumerate() {
8719                    let m = ds.lock();
8720                    if m.deleted || datasets_in_subgroups.contains(&i) {
8721                        continue;
8722                    }
8723                    // The leaf, never the registry path: a link name is one
8724                    // path component, and `H5G_traverse` would split a '/'
8725                    // in it before `H5L_link` ever saw the name.
8726                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8727                    links.push((
8728                        m.creation_seq,
8729                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8730                    ));
8731                }
8732                for grp in self.group_refs() {
8733                    let g = grp.lock();
8734                    if g.deleted || g.parent.is_some() {
8735                        continue;
8736                    }
8737                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8738                    links.push((
8739                        g.creation_seq,
8740                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8741                    ));
8742                }
8743                self.push_hard_links(&mut links, None);
8744                self.push_symbolic_links(&mut links, None);
8745                self.push_committed_datatypes(&mut links, None);
8746            }
8747            LinkScope::Group(group_idx) => {
8748                // Snapshot the child lists, then drop the slot guard: the
8749                // per-child reads below re-lock dataset and group slots
8750                // (including this one).
8751                let (child_datasets, child_groups) = {
8752                    let grp = self.grp(group_idx);
8753                    let g = grp.lock();
8754                    (g.child_datasets.clone(), g.child_groups.clone())
8755                };
8756                for ds_idx in child_datasets {
8757                    let ds = self.ds(ds_idx);
8758                    let m = ds.lock();
8759                    if m.deleted {
8760                        continue;
8761                    }
8762                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8763                    links.push((
8764                        m.creation_seq,
8765                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8766                    ));
8767                }
8768                for child_idx in child_groups {
8769                    let child_grp = self.grp(child_idx);
8770                    let g = child_grp.lock();
8771                    if g.deleted {
8772                        continue;
8773                    }
8774                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8775                    links.push((
8776                        g.creation_seq,
8777                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8778                    ));
8779                }
8780                self.push_hard_links(&mut links, Some(group_idx));
8781                self.push_symbolic_links(&mut links, Some(group_idx));
8782                self.push_committed_datatypes(&mut links, Some(group_idx));
8783            }
8784        }
8785        // Creation order, not order by kind: a run of create_group and
8786        // create_dataset draws from one counter, so this is the order the
8787        // caller made them in. `H5G_obj_insert` numbers from zero within the
8788        // group, so the rank here is the link's creation order.
8789        links.sort_by_key(|(seq, _)| *seq);
8790        links
8791            .into_iter()
8792            .enumerate()
8793            .map(|(rank, (_, link))| {
8794                if order.is_tracked() {
8795                    link.with_creation_order(rank as i64)
8796                } else {
8797                    link
8798                }
8799            })
8800            .collect()
8801    }
8802
8803    /// Whether `links` must live in dense storage rather than in the group's
8804    /// object header — the `H5G_obj_insert` phase-change rule, applied to the
8805    /// whole set at once because this writer builds each header from scratch
8806    /// rather than inserting one link at a time.
8807    ///
8808    /// libhdf5 converts when the count *reaches* `max_compact` and another
8809    /// link arrives, so a set of exactly `max_compact` is still compact; and
8810    /// separately when one message would not fit the 16-bit size field an
8811    /// object header message has.
8812    ///
8813    /// The answer depends only on the link names and kinds, never on the
8814    /// addresses they point at, which is what lets a group header be sized
8815    /// before [`prepare_dense_links`](Self::prepare_dense_links) has run.
8816    fn links_need_dense(&self, links: &[LinkMessage]) -> bool {
8817        links.len() > MAX_COMPACT_LINKS
8818            || links
8819                .iter()
8820                .any(|l| l.encode(&self.ctx).len() > MAX_MESSAGE_SIZE)
8821    }
8822
8823    /// The single owner of link emission into a group object header: the Link
8824    /// Info and Group Info messages, and then either one `MSG_LINK` per link
8825    /// or nothing at all when the set has spilled to dense storage.
8826    ///
8827    /// The two storage forms are exclusive (`H5G_obj_insert` moves the whole
8828    /// set at once), and a header carrying both would report every link twice.
8829    ///
8830    /// A group whose links are dense but not yet laid out gets a compact Link
8831    /// Info message here. That is deliberate: the message encodes to the same
8832    /// length either way — two addresses, defined or not — so the sizing pass
8833    /// that runs before `prepare_dense_links` still reserves the right number
8834    /// of bytes, and the write pass that runs after it emits the real heap and
8835    /// index addresses. It is the same two-pass rule the child link addresses
8836    /// already follow.
8837    fn emit_links(
8838        &self,
8839        header: &mut ObjectHeader,
8840        scope: LinkScope,
8841        links: &[LinkMessage],
8842        order: CreationOrder,
8843    ) {
8844        // A symbol-table group holds no link messages at all: its links are the
8845        // entries of the symbol table `prepare_symbol_tables` laid out, and
8846        // the header carries only the two addresses naming it. Link Info and
8847        // Group Info are version-1.8 messages and have no business in a
8848        // version-1 header — `H5G__stab_valid` reads the Symbol Table message
8849        // and nothing else.
8850        if self.uses_symbol_table(scope, order) {
8851            // Sizing runs before the tables are laid out; the message is the
8852            // same two addresses wide either way, so the placeholder reserves
8853            // exactly what the real one needs. Same two-pass rule the child
8854            // link addresses already follow.
8855            let stab = self
8856                .symbol_tables
8857                .written
8858                .lock()
8859                .get(&scope)
8860                .copied()
8861                .unwrap_or(Stab {
8862                    btree_addr: UNDEF_ADDR,
8863                    heap_addr: UNDEF_ADDR,
8864                });
8865            header.add_message(MSG_SYMBOL_TABLE, 0x00, stab.encode(&self.ctx));
8866            return;
8867        }
8868        // Links a reopen carried through verbatim because this writer cannot
8869        // express them. They are emitted here rather than by a second caller
8870        // so that no header-rewrite path can drop them, and their presence
8871        // pins the group to compact storage: dense storage would have to
8872        // re-encode each link into the heap, which is exactly the byte
8873        // fidelity preserving them is for.
8874        let preserved = self.preserved_links_for(scope);
8875        let dense = preserved.is_empty() && self.links_need_dense(links);
8876        let link_info = self.dense_links.lock().get(&scope).cloned();
8877        let link_info = link_info.unwrap_or_else(|| {
8878            let mut info = LinkInfoMessage::compact();
8879            if order.is_tracked() {
8880                // `H5G__obj_insert` post-increments `max_corder`, so a group
8881                // holding n links reports n.
8882                info.max_creation_order = Some(links.len() as u64);
8883            }
8884            if order.is_indexed() {
8885                // The index address stays undefined while the links live in
8886                // the header, but the message must still carry the field:
8887                // `H5Pget_link_creation_order` reads INDEXED off this flag,
8888                // not off the address.
8889                info.creation_order_btree_address = Some(UNDEF_ADDR);
8890            }
8891            info
8892        });
8893        header.add_message(MSG_LINK_INFO, 0x00, link_info.encode(&self.ctx));
8894        // The link info message takes no flags and the group info message
8895        // takes `H5O_MSG_FLAG_CONSTANT`, exactly as `H5G__obj_create_real`
8896        // creates the pair (H5Gobj.c:255, :259) and as
8897        // `H5G__obj_insert`'s phase change re-creates it (H5Gobj.c:526). The
8898        // asymmetry is real: the link info message records the group's
8899        // storage and its creation-order counter, both of which change as
8900        // links come and go, while the group info message holds the phase
8901        // change and estimated-name-length constants of the creation property
8902        // list, which nothing after creation rewrites.
8903        header.add_message(
8904            MSG_GROUP_INFO,
8905            MSG_FLAG_CONSTANT,
8906            GroupInfoMessage::default().encode(),
8907        );
8908        if dense {
8909            return;
8910        }
8911        for link in links {
8912            header.add_message(MSG_LINK, 0x00, link.encode(&self.ctx));
8913        }
8914        for encoded in preserved {
8915            header.add_message(MSG_LINK, 0x00, encoded);
8916        }
8917    }
8918
8919    /// The verbatim link bodies a reopen carried into `scope`.
8920    fn preserved_links_for(&self, scope: LinkScope) -> Vec<Vec<u8>> {
8921        let parent = match scope {
8922            LinkScope::Root => None,
8923            LinkScope::Group(i) => Some(i),
8924        };
8925        self.preserved_links
8926            .lock()
8927            .iter()
8928            .filter(|l| l.parent == parent)
8929            .map(|l| l.encoded.clone())
8930            .collect()
8931    }
8932
8933    /// Lay out and write dense link storage for every group that needs it,
8934    /// recording the resulting `Link Info` message per group.
8935    ///
8936    /// The sole owner of that transition. It must run after every object
8937    /// header address is assigned — the heap holds encoded link messages, and
8938    /// those name their targets — and before any group header is written.
8939    ///
8940    /// Every group whose header this finalize rewrites passes through here,
8941    /// dense or not: the storage a reopened header named is superseded by the
8942    /// rewrite whichever form the new link set takes, and freeing it first is
8943    /// what lets the replacement reuse those blocks.
8944    fn prepare_dense_links(&self) -> IoResult<()> {
8945        let mut scopes: Vec<(LinkScope, Vec<LinkMessage>, CreationOrder)> = Vec::new();
8946        for gi in 0..self.group_count() {
8947            let (deleted, order) = {
8948                let grp = self.grp(gi);
8949                let g = grp.lock();
8950                (g.deleted, g.track_order.links)
8951            };
8952            // A symbol-table group is `prepare_symbol_tables`' business; it
8953            // has no Link Info message to hold a fractal heap address, and it
8954            // never had dense storage to release.
8955            if deleted || self.uses_symbol_table(LinkScope::Group(gi), order) {
8956                continue;
8957            }
8958            self.release_superseded_dense_links(LinkScope::Group(gi))?;
8959            let links = self.group_links(LinkScope::Group(gi), order);
8960            if self.links_need_dense(&links) {
8961                scopes.push((LinkScope::Group(gi), links, order));
8962            }
8963        }
8964        let root_order = self.root_track_order.links;
8965        if !self.uses_symbol_table(LinkScope::Root, root_order) {
8966            self.release_superseded_dense_links(LinkScope::Root)?;
8967            let root_links = self.group_links(LinkScope::Root, root_order);
8968            if self.links_need_dense(&root_links) {
8969                scopes.push((LinkScope::Root, root_links, root_order));
8970            }
8971        }
8972
8973        for (scope, links, order) in scopes {
8974            // `close` after `start_swmr` finalizes a second time over the same
8975            // groups, so rebuilding here would allocate a whole second heap
8976            // and strand the one the published headers already name.
8977            if self.dense_links.lock().contains_key(&scope) {
8978                continue;
8979            }
8980            let dense = build_dense_links(&links, &self.ctx, order, &mut |len| {
8981                self.allocator.allocate(len, FreeSpaceClass::Metadata)
8982            })?;
8983            for block in &dense.blocks {
8984                self.handle.write_at(block.addr, &block.image)?;
8985            }
8986            self.dense_links.lock().insert(scope, dense.linfo);
8987        }
8988        Ok(())
8989    }
8990
8991    /// Lay out whichever of the two forms of link storage this file uses,
8992    /// before any group header is written.
8993    ///
8994    /// The two are exclusive because the formats are: a classic group has no
8995    /// Link Info message to put a fractal heap address in, and a link-message
8996    /// group has no symbol table.
8997    fn prepare_link_storage(&self) -> IoResult<()> {
8998        self.prepare_dense_links()?;
8999        self.prepare_symbol_tables()
9000    }
9001
9002    /// Lay out and write the symbol table of every classic group, and free the
9003    /// storage each rewrite supersedes. A no-op on a link-message file.
9004    ///
9005    /// The classic counterpart of [`prepare_dense_links`](Self::prepare_dense_links),
9006    /// and the sole owner of that transition. The same two placement rules
9007    /// apply for the same two reasons: it runs after every object header has
9008    /// an address, because a symbol table entry names its target's header, and
9009    /// before any group header is written, because the header carries the
9010    /// Symbol Table message naming what this laid out.
9011    ///
9012    /// Deepest group first, root last. A hard link to a group caches that
9013    /// group's own B-tree and heap in the entry's scratch pad
9014    /// (`H5G__link_to_ent`), so the child's table must exist before the
9015    /// parent's is built; `H5G__stab_valid` checks the root entry's cache
9016    /// against the root header's Symbol Table message, so a stale pair there
9017    /// is not a slow lookup but a file `H5Fopen` rejects.
9018    ///
9019    /// Every classic group is rebuilt on every pass — there is no "already
9020    /// done" short-circuit like the dense one, because the only way this runs
9021    /// twice is a `Drop` retry after a failed `close`, and the entries of the
9022    /// first pass name header addresses the second pass has moved. (A SWMR
9023    /// session, the other double-finalize, cannot reach here: SWMR needs a
9024    /// version-3 superblock, so `start_swmr` refuses a classic file.)
9025    fn prepare_symbol_tables(&self) -> IoResult<()> {
9026        // Depth by parent chain, not by counting separators in the registry
9027        // path: the chain is what actually says which table has to exist first.
9028        let mut scopes: Vec<(usize, LinkScope, CreationOrder)> = Vec::new();
9029        for gi in 0..self.group_count() {
9030            let (deleted, order, mut parent) = {
9031                let grp = self.grp(gi);
9032                let g = grp.lock();
9033                (g.deleted, g.track_order.links, g.parent)
9034            };
9035            if deleted || !self.uses_symbol_table(LinkScope::Group(gi), order) {
9036                continue;
9037            }
9038            let mut depth = 1usize;
9039            while let Some(p) = parent {
9040                depth += 1;
9041                parent = self.grp(p).lock().parent;
9042            }
9043            scopes.push((depth, LinkScope::Group(gi), order));
9044        }
9045        scopes.sort_by_key(|&(depth, ..)| std::cmp::Reverse(depth));
9046        let root_order = self.root_track_order.links;
9047        if self.uses_symbol_table(LinkScope::Root, root_order) {
9048            scopes.push((0, LinkScope::Root, root_order));
9049        }
9050
9051        let meta = self.stab_meta();
9052        for (_, scope, order) in scopes {
9053            // Freed before the replacement is laid out, so a rewrite reuses
9054            // the same blocks instead of growing the file on every open/close
9055            // cycle — the rule `prepare_dense_links` and the header rewrite
9056            // already follow. Removed as it is freed, so no second pass can
9057            // free it twice.
9058            let superseded = self.symbol_tables.superseded.lock().remove(&scope);
9059            if let Some(extents) = superseded {
9060                free_stab(&self.allocator, &extents);
9061            }
9062            let links = self.stab_links_for(scope, order)?;
9063            let stab = write_stab(&self.handle, &self.allocator, &meta, &links)?;
9064            self.symbol_tables.written.lock().insert(scope, stab);
9065        }
9066        Ok(())
9067    }
9068
9069    /// The file-level parameters every symbol-table node width is derived from
9070    /// — the address/length widths and the B-tree "K" ranks. Only a version-0/1
9071    /// superblock records ranks of its own; [`btree_v1_config`] is the one
9072    /// place that decides whether this file has any.
9073    ///
9074    /// [`btree_v1_config`]: Self::btree_v1_config
9075    fn stab_meta(&self) -> FileMeta {
9076        FileMeta {
9077            ctx: self.ctx,
9078            btree: self.btree_v1_config(),
9079            sohm: None,
9080        }
9081    }
9082
9083    /// `scope`'s links as symbol table entries.
9084    ///
9085    /// A link a reopen carried through verbatim is decoded back out of its
9086    /// encoded Link message here, because a classic group has no link message
9087    /// to preserve it into. Nothing is lost in the round trip: the walk built
9088    /// that message from a symbol table entry in the first place, and the two
9089    /// forms carry the same three facts.
9090    fn stab_links_for(&self, scope: LinkScope, order: CreationOrder) -> IoResult<Vec<StabLink>> {
9091        let groups = self.group_header_scopes();
9092        let mut out = Vec::new();
9093        for link in self.group_links(scope, order) {
9094            out.push(self.stab_link(&link, &groups)?);
9095        }
9096        for encoded in self.preserved_links_for(scope) {
9097            let (link, _) = LinkMessage::decode(&encoded, &self.ctx)?;
9098            out.push(self.stab_link(&link, &groups)?);
9099        }
9100        Ok(out)
9101    }
9102
9103    /// Where each group's object header now sits, so a hard link that lands on
9104    /// one can cache that group's symbol table in its scratch pad.
9105    fn group_header_scopes(&self) -> HashMap<u64, LinkScope> {
9106        let mut map = HashMap::new();
9107        for gi in 0..self.group_count() {
9108            let grp = self.grp(gi);
9109            let g = grp.lock();
9110            if !g.deleted {
9111                map.insert(g.obj_header_addr, LinkScope::Group(gi));
9112            }
9113        }
9114        map
9115    }
9116
9117    /// One link as a symbol table entry.
9118    ///
9119    /// The scratch pad caches the target group's B-tree and heap when the
9120    /// target is a group this pass has already laid out — what
9121    /// `H5G__link_to_ent` does, and what lets `H5G__stab_lookup` walk a path
9122    /// without opening each header on the way. For anything else the pad stays
9123    /// `H5G_NOTHING_CACHED`, the value libhdf5 itself writes whenever the
9124    /// target has no Symbol Table message to read.
9125    fn stab_link(
9126        &self,
9127        link: &LinkMessage,
9128        groups: &HashMap<u64, LinkScope>,
9129    ) -> IoResult<StabLink> {
9130        let target = match &link.target {
9131            LinkTarget::Hard { address } => {
9132                let cached = groups
9133                    .get(address)
9134                    .and_then(|scope| self.symbol_tables.written.lock().get(scope).copied());
9135                StabTarget::Hard {
9136                    addr: *address,
9137                    cached,
9138                }
9139            }
9140            LinkTarget::Soft { target } => StabTarget::Soft {
9141                value: target.clone(),
9142            },
9143            // Unreachable by construction: a group holding one of these is
9144            // not a symbol-table group at all
9145            // ([`LinkMessage::fits_symbol_table`] is what
9146            // [`Hdf5Writer::uses_symbol_table`] asks), so this pass never
9147            // visits it. Reported rather than panicked so a future caller
9148            // that skips that gate learns which link it lost.
9149            LinkTarget::External { .. } | LinkTarget::UserDefined { .. } => {
9150                return Err(crate::io::IoError::InvalidState(format!(
9151                    "cannot store the link {:?} in a symbol table: it holds only \
9152                     hard and soft links, and this group was not converted to link \
9153                     messages the way `H5G_obj_insert` converts it",
9154                    link.name
9155                )))
9156            }
9157        };
9158        Ok(StabLink {
9159            name: link.name.clone(),
9160            target,
9161        })
9162    }
9163
9164    /// The single owner of attribute emission into an object header: appends
9165    /// the Attribute Info message and then one `MSG_ATTRIBUTE` per attribute.
9166    ///
9167    /// On a version-2 object header the two are inseparable.
9168    /// `H5O__attr_count_real` derives `H5Oget_info().num_attrs` from the
9169    /// Attribute Info message alone — with no such message the count reads as
9170    /// zero however many attribute messages follow, which is what made every
9171    /// rust-written file report `num_attrs == 0` to libhdf5 while
9172    /// `H5Aiterate2` still yielded the attributes. The message carries no
9173    /// count of its own: `H5A__get_ainfo` fills `nattrs` from the attribute
9174    /// messages the header loader actually saw, so compact storage needs
9175    /// nothing but the message's presence.
9176    ///
9177    /// When [`prepare_dense_attributes`](Self::prepare_dense_attributes) has
9178    /// spilled `scope`'s attributes to a fractal heap, the same message names
9179    /// that heap instead and *no* attribute message follows: the two storage
9180    /// forms are exclusive (`H5O__attr_create` moves the whole set at once),
9181    /// and a header carrying both would report every attribute twice.
9182    fn emit_attributes(
9183        &self,
9184        header: &mut ObjectHeader,
9185        scope: AttrScope,
9186        attributes: &[AttributeEntry],
9187        order: CreationOrder,
9188        format: ObjectFormat,
9189        owner: ShareOwner,
9190    ) {
9191        // `H5Pget_attr_creation_order` reads the object header's own flags,
9192        // not the Attribute Info message, so this is what makes the object
9193        // report creation-ordered attributes — and tracking widens every
9194        // message envelope by the creation index below.
9195        let order = self.header_attr_order(order);
9196        header.set_attribute_creation_order(order);
9197        if attributes.is_empty() {
9198            return;
9199        }
9200        // A version-1 object header gets the attribute messages alone.
9201        // `H5O__attr_create` gates every mention of the Attribute Info message
9202        // on `oh->version > H5O_VERSION_1` (H5Oattribute.c:218), and so does
9203        // `H5O__attr_count_real`, which is why the count still reads correctly
9204        // without it: on a version-1 header libhdf5 counts the messages.
9205        if format == ObjectFormat::Legacy {
9206            for attr in attributes {
9207                header.add_message(MSG_ATTRIBUTE, 0x00, self.encode_attribute(attr));
9208            }
9209            return;
9210        }
9211        // Whether the set spills is a property of the set alone, so it is the
9212        // same answer in the pass that measures this header and in the pass
9213        // that writes it — even though the storage itself is laid out between
9214        // the two, because it can only be laid out once every object header
9215        // has an address. Sizing therefore falls back to a placeholder message
9216        // of the same width: only the creation-order flags change the
9217        // Attribute Info message's length, so the header measured here holds
9218        // the header written against the storage that replaces it. Same
9219        // two-pass rule `emit_links` follows for dense links and symbol
9220        // tables.
9221        let dense = self.attributes_need_dense(attributes, format);
9222        let stored = self.dense_attributes.lock().get(&scope).cloned();
9223        let ainfo = stored.unwrap_or_else(|| {
9224            let mut ainfo = AttributeInfoMessage::compact();
9225            if order.is_tracked() {
9226                ainfo.max_creation_index = Some(next_creation_index(attributes));
9227            }
9228            if order.is_indexed() {
9229                // Compact storage has no index B-tree, but the message still
9230                // announces one so that its flags match the header's
9231                // (`H5O__attr_create` asserts they agree).
9232                ainfo.creation_order_btree_address = Some(UNDEF_ADDR);
9233            }
9234            ainfo
9235        });
9236        header.add_message(MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, ainfo.encode(&self.ctx));
9237        if dense {
9238            return;
9239        }
9240        // Each attribute states its own creation index — the one it was
9241        // created with here, or the one the file it was read from records. An
9242        // attribute with none belongs to an object that tracks no order, where
9243        // the field is not encoded at all.
9244        for attr in attributes {
9245            let (flags, body) = self.share_attribute(attr, format, owner);
9246            header.add_message_indexed(
9247                MSG_ATTRIBUTE,
9248                flags,
9249                body,
9250                attr.creation_index().unwrap_or(0),
9251            );
9252        }
9253    }
9254
9255    /// One attribute message body, at the version this file's low library
9256    /// bound calls for (`H5A__set_version`, which reads the bound and nothing
9257    /// about the object the attribute hangs on).
9258    fn encode_attribute(&self, attr: &AttributeEntry) -> Vec<u8> {
9259        attr.encode_for(&self.ctx, self.encoding_libver(), self.message_format())
9260    }
9261
9262    /// What a header stores for one attribute: the message flags and the body,
9263    /// with the attribute's own datatype and dataspace shared wherever an
9264    /// index covers them.
9265    ///
9266    /// `H5A__create` offers both to `H5SM_try_share` (H5Aint.c:375-377) before
9267    /// `H5O__attr_create` offers the attribute itself (H5Oattribute.c:726), so
9268    /// the attribute body that reaches the heap already holds their pointers
9269    /// and says which fields they are in its own flags byte
9270    /// (`H5O_ATTR_FLAG_TYPE_SHARED` / `H5O_ATTR_FLAG_SPACE_SHARED`,
9271    /// H5Oattr.c:358-359). Both offers go through
9272    /// [`share_message`](Self::share_message) like any other, so the pass that
9273    /// counts references and the pass that substitutes see the same three
9274    /// messages.
9275    fn share_attribute(
9276        &self,
9277        attr: &AttributeEntry,
9278        format: ObjectFormat,
9279        owner: ShareOwner,
9280    ) -> (u8, Vec<u8>) {
9281        let libver = self.encoding_libver();
9282        // Only a readable attribute has pieces to offer: an unreadable one is
9283        // the bytes it was read from, put back as they were. Version 1 has no
9284        // flags byte to record a shared field in — `H5O__attr_encode` writes a
9285        // reserved zero there — so a classic file shares the attribute whole
9286        // or not at all.
9287        let Some(message) = attr.readable().filter(|_| format.attribute_version() >= 2) else {
9288            return self.share_message(
9289                owner,
9290                MSG_ATTRIBUTE,
9291                0x00,
9292                attr.encode_for(&self.ctx, libver, format),
9293            );
9294        };
9295
9296        let datatype = message.datatype.encode_at(&self.ctx, libver);
9297        let dataspace = message.dataspace.encode_for(&self.ctx, format);
9298        // `H5A__create` passes no open header for either (H5Aint.c:375-377):
9299        // both live inside the attribute's body, so neither has a header
9300        // message a `H5SM_IN_OH` record could name and both reach the heap on
9301        // first use.
9302        let (dt_flags, dt_field) =
9303            self.share_message(ShareOwner::Detached, MSG_DATATYPE, 0x00, datatype.clone());
9304        let (ds_flags, ds_field) =
9305            self.share_message(ShareOwner::Detached, MSG_DATASPACE, 0x00, dataspace.clone());
9306
9307        let mut attr_flags = 0u8;
9308        if dt_flags & MSG_FLAG_SHARED != 0 {
9309            attr_flags |= ATTR_FLAG_TYPE_SHARED;
9310        }
9311        if ds_flags & MSG_FLAG_SHARED != 0 {
9312            attr_flags |= ATTR_FLAG_SPACE_SHARED;
9313        }
9314        let encoded = message.encode_with_fields(attr_flags, &dt_field, &ds_field);
9315
9316        // Each shared field's heap ID sits two bytes into the pointer that
9317        // replaced it; the body offered below carries whatever
9318        // `share_message` just produced, which is a zeroed ID in the pass that
9319        // counts and the real one in the pass that substitutes.
9320        let mut nested = Vec::new();
9321        if attr_flags & ATTR_FLAG_TYPE_SHARED != 0 {
9322            nested.push(NestedShare {
9323                heap_id_at: encoded.datatype_at + SOHM_POINTER_HEAP_ID_AT,
9324                target: (MSG_DATATYPE, datatype),
9325            });
9326        }
9327        if attr_flags & ATTR_FLAG_SPACE_SHARED != 0 {
9328            nested.push(NestedShare {
9329                heap_id_at: encoded.dataspace_at + SOHM_POINTER_HEAP_ID_AT,
9330                target: (MSG_DATASPACE, dataspace),
9331            });
9332        }
9333        self.share_nesting_message(owner, MSG_ATTRIBUTE, 0x00, encoded.body, nested)
9334    }
9335
9336    /// Whether `attributes` must live in dense storage rather than in the
9337    /// object header — the `H5O__attr_create` phase-change rule, applied to
9338    /// the whole set at once because this writer builds each header from
9339    /// scratch rather than inserting one attribute at a time.
9340    ///
9341    /// libhdf5 converts when the count *reaches* `max_compact` and another
9342    /// attribute arrives, so a set of exactly `max_compact` is still compact;
9343    /// and separately when one message would not fit the 16-bit size field an
9344    /// object header message has.
9345    ///
9346    /// Never in a classic file. Dense attribute storage is a fractal heap
9347    /// reached through an Attribute Info message, both introduced in the 1.8
9348    /// format; at `H5F_LIBVER_EARLIEST` libhdf5 keeps every attribute in the
9349    /// header however many there are (`H5O__attr_create` reaches the phase
9350    /// change only when the object header version allows it). An attribute
9351    /// too large for the 16-bit size field is then an error, which
9352    /// `ObjectHeader::encode_v1` raises, rather than a reason to spill.
9353    fn attributes_need_dense(&self, attributes: &[AttributeEntry], format: ObjectFormat) -> bool {
9354        if format == ObjectFormat::Legacy {
9355            return false;
9356        }
9357        attributes.len() > MAX_COMPACT_ATTRS
9358            || attributes
9359                .iter()
9360                .any(|a| self.encode_attribute(a).len() > MAX_MESSAGE_SIZE)
9361    }
9362
9363    /// Every object whose attributes this finalize re-lays-out, with the
9364    /// creation-order policy each one's storage must follow.
9365    ///
9366    /// `datasets` lists the datasets whose headers this finalize will
9367    /// actually write. A reopened dataset that took no writes keeps its
9368    /// original header — and with it whatever storage that header already
9369    /// names — so touching its attribute storage would strand every block of
9370    /// it.
9371    ///
9372    /// The policy is the one the *header* records, not the one the object's
9373    /// creation property list asked for: those differ on a file whose
9374    /// shared-message configuration covers attributes, where
9375    /// [`header_attr_order`](Self::header_attr_order) raises every object to
9376    /// tracked. Storage laid out against the property list would then omit the
9377    /// creation indices the header says are there — and, since the Attribute
9378    /// Info message carries a maximum creation index only when tracked, would
9379    /// be two bytes shorter than the message the sizing pass measured.
9380    fn attribute_scopes(&self, datasets: &[usize]) -> Vec<(AttrScope, CreationOrder)> {
9381        let order_of = |requested| self.header_attr_order(requested);
9382        let mut scopes = vec![(AttrScope::Root, order_of(self.root_track_order.attrs))];
9383        for gi in 0..self.group_count() {
9384            if self.grp(gi).lock().deleted {
9385                continue;
9386            }
9387            let order = self.grp(gi).lock().track_order.attrs;
9388            scopes.push((AttrScope::Group(gi), order_of(order)));
9389        }
9390        for &i in datasets {
9391            let order = self.ds(i).lock().track_attr_order;
9392            scopes.push((AttrScope::Dataset(i), order_of(order)));
9393        }
9394        scopes
9395    }
9396
9397    /// Lay out and write dense attribute storage for every object that needs
9398    /// it, recording the resulting `Attribute Info` message per object.
9399    ///
9400    /// The sole owner of that transition. It runs after every object header
9401    /// has an address — an attribute may hold an object reference, and the
9402    /// heap holds the encoded attribute messages — and before any object
9403    /// header is written, because the header carries the Attribute Info
9404    /// message naming what this laid out. Every block is on disk before the
9405    /// map naming it is populated, so a header written from that map can only
9406    /// point at bytes that exist. The same placement rule, for the same two
9407    /// reasons, as [`prepare_dense_links`](Self::prepare_dense_links).
9408    ///
9409    /// Which objects spill is not decided here: `emit_attributes` asks
9410    /// [`attributes_need_dense`](Self::attributes_need_dense) itself, so the
9411    /// header measured before this ran and the header written after it agree
9412    /// without either consulting the other.
9413    fn prepare_dense_attributes(&self, datasets: &[usize]) -> IoResult<()> {
9414        for (scope, order) in self.attribute_scopes(datasets) {
9415            // Every scope here has its header rewritten, so the storage a
9416            // reopen found on it is superseded whether or not the new set is
9417            // dense again — a free driven by "the new set needs a heap" would
9418            // never reach an object that dropped back to compact. Freed
9419            // immediately before its replacement is laid out, so the rewrite
9420            // lands in the blocks it just gave back instead of growing the
9421            // file on every open/close cycle.
9422            self.release_superseded_dense_attrs(scope)?;
9423            // `close` after `start_swmr` finalizes a second time over the same
9424            // attribute sets — SWMR refuses every attribute mutation — so
9425            // rebuilding here would allocate a whole second heap and strand
9426            // the one the published headers already name.
9427            if self.dense_attributes.lock().contains_key(&scope) {
9428                continue;
9429            }
9430            let attributes = self.object_attributes(scope)?;
9431            if !self.attributes_need_dense(&attributes, self.attr_scope_format(scope)) {
9432                continue;
9433            }
9434            let dense = build_dense_attributes(&attributes, &self.ctx, order, &mut |len| {
9435                self.allocator.allocate(len, FreeSpaceClass::Metadata)
9436            })?;
9437            for block in &dense.blocks {
9438                self.handle.write_at(block.addr, &block.image)?;
9439            }
9440            self.dense_attributes.lock().insert(scope, dense.ainfo);
9441        }
9442        Ok(())
9443    }
9444
9445    /// The object header format `scope`'s owner is written at, which is what
9446    /// decides whether its attributes may spill at all.
9447    fn attr_scope_format(&self, scope: AttrScope) -> ObjectFormat {
9448        match scope {
9449            AttrScope::Root => self.header_format(self.root_track_order),
9450            AttrScope::Group(gi) => self.group_header_format(gi),
9451            AttrScope::Dataset(i) => self.dataset_header_format(i),
9452        }
9453    }
9454
9455    /// Whether a dataset's datatype message may be offered to a
9456    /// shared-message index at all.
9457    ///
9458    /// The datatype is the one message class carrying a `can_share` callback
9459    /// (`H5O__dtype_can_share`, H5Odtype.c:99), and `H5SM__can_share_common`
9460    /// asks it before any index is consulted (H5SM.c:895-899). It refuses an
9461    /// immutable type and a committed one (H5Odtype.c:1893-1901); the
9462    /// committed half is already answered by address at the call site.
9463    ///
9464    /// A dataset's type reaches that predicate still immutable only when
9465    /// `H5D__init_type` kept the caller's own `H5T_t` rather than copying it,
9466    /// which it does exactly when the type is immutable, is not relocatable,
9467    /// and the low bound this dataset's messages are written at is below
9468    /// `H5F_LIBVER_V18` (H5Dint.c:569-572) — the bound the dataset was
9469    /// *created* under, which for a dataset a reopen found is not this
9470    /// session's.
9471    /// Any of the three failing produces an `H5T_COPY_ALL` copy, which is
9472    /// `H5T_STATE_RDONLY` rather than immutable (H5T.c:4461-4462) and so is
9473    /// shareable — which is why `H5Tcopy(H5T_STD_I32LE)` shares where
9474    /// `H5T_STD_I32LE` itself does not (tests/fixtures/gen_sohm.c).
9475    ///
9476    /// An attribute has no such branch: `H5A__create` copies unconditionally
9477    /// (H5Aint.c:341), so its datatype is always eligible and
9478    /// [`share_attribute`](Self::share_attribute) offers it without asking.
9479    fn dataset_datatype_shareable(&self, datatype: &DatatypeMessage, libver: LibverBound) -> bool {
9480        !datatype.is_predefined() || datatype.is_relocatable() || libver >= LibverBound::V18
9481    }
9482
9483    /// Whether the first copy of a `msg_type` message may stay literal in the
9484    /// object header that writes it.
9485    ///
9486    /// `H5O_msg_can_share_in_ohdr` reads the class's `H5O_SHARE_IN_OHDR` flag
9487    /// (H5Omessage.c:1426); the five classes that carry it are datatype
9488    /// (H5Odtype.c:89), dataspace (H5Osdspace.c:61), both fill value messages
9489    /// (H5Ofill.c:106 and :130) and the filter pipeline (H5Opline.c:65). The
9490    /// attribute class does not, which is why an attribute reaches the heap on
9491    /// its first use.
9492    const fn shares_in_ohdr(msg_type: u8) -> bool {
9493        matches!(
9494            msg_type,
9495            MSG_DATASPACE
9496                | MSG_DATATYPE
9497                | MSG_FILL_VALUE
9498                | MSG_FILL_VALUE_OLD
9499                | MSG_FILTER_PIPELINE
9500        )
9501    }
9502
9503    /// What a header stores for a message a shared-message index may cover:
9504    /// the body itself, or a pointer into the shared-message heap.
9505    ///
9506    /// The single point at which a message is offered to an index. Every
9507    /// header builder routes its shareable messages through here, so the pass
9508    /// that counts references and the pass that substitutes pointers walk
9509    /// exactly the same set — the counting and the substituting cannot drift
9510    /// apart, because they are one call site in two phases.
9511    ///
9512    /// `owner` is `H5SM_try_share`'s `open_oh`: the header this message
9513    /// belongs to, or [`ShareOwner::Detached`] for a body that is part of
9514    /// another message rather than a message of a header.
9515    ///
9516    /// Outside a finalize, and in any file created without indexes, this is
9517    /// the identity.
9518    fn share_message(
9519        &self,
9520        owner: ShareOwner,
9521        msg_type: u8,
9522        flags: u8,
9523        body: Vec<u8>,
9524    ) -> (u8, Vec<u8>) {
9525        self.share_nesting_message(owner, msg_type, flags, body, Vec::new())
9526    }
9527
9528    /// [`share_message`](Self::share_message) for a body that itself holds
9529    /// shared-message pointers.
9530    ///
9531    /// `nested` names each heap ID inside `body`, which is zero until the
9532    /// table is laid out. Two bodies that differ only in what they point at
9533    /// are the same bytes here and different bytes on disk, so the count and
9534    /// the substitute are keyed on the pair.
9535    fn share_nesting_message(
9536        &self,
9537        owner: ShareOwner,
9538        msg_type: u8,
9539        flags: u8,
9540        body: Vec<u8>,
9541        nested: Vec<NestedShare>,
9542    ) -> (u8, Vec<u8>) {
9543        let Some(sohm) = self.sohm.as_deref() else {
9544            return (flags, body);
9545        };
9546        // A message already carrying a pointer — a committed datatype — is
9547        // shared by address and must not be shared again, and the message
9548        // classes libhdf5 marks `H5O_MSG_FLAG_DONTSHARE` never reach an index.
9549        if flags & (MSG_FLAG_SHARED | MSG_FLAG_DONTSHARE) != 0 {
9550            return (flags, body);
9551        }
9552        let Some(index) = sohm.index_for(msg_type, body.len()) else {
9553            return (flags, body);
9554        };
9555        // `share_in_ohdr && open_oh` (H5SM.c:1400): the first copy of one of
9556        // these classes stays where it was written, marked shareable, and only
9557        // a second use moves the body to the heap.
9558        let ohdr = match owner {
9559            ShareOwner::Header(addr) if Self::shares_in_ohdr(msg_type) => Some(addr),
9560            _ => None,
9561        };
9562        // What a pointer to this body looks like: a zeroed heap ID until the
9563        // table exists, which is the width the real one has.
9564        let pointer = |id| {
9565            (
9566                flags | MSG_FLAG_SHARED,
9567                SharedMessagePointer::encode_sohm(id),
9568            )
9569        };
9570        match &mut *sohm.phase.lock() {
9571            SohmPhase::Idle => (flags, body),
9572            SohmPhase::Predict(first) => {
9573                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9574                    return (flags | MSG_FLAG_SHAREABLE, body);
9575                }
9576                pointer([0u8; SOHM_HEAP_ID_LEN])
9577            }
9578            // The same substitution `Predict` makes, so that what the collect
9579            // pass builds around a shared message is the width the resolve
9580            // pass will build — which is what lets an attribute body assembled
9581            // in this pass be the body assembled in that one, bar the heap IDs
9582            // it is here recording a need for.
9583            SohmPhase::Collect(collector) => {
9584                let first = collector.record(index, msg_type, &body, &nested, ohdr);
9585                if !first && !nested.is_empty() {
9586                    // This body is already here, so the pointers it holds
9587                    // already exist in the heap and the offers that built
9588                    // this copy of it must not count a second time.
9589                    for share in &nested {
9590                        collector.release(share.target.0, &share.target.1);
9591                    }
9592                }
9593                if ohdr.is_some() && first {
9594                    return (flags | MSG_FLAG_SHAREABLE, body);
9595                }
9596                pointer([0u8; SOHM_HEAP_ID_LEN])
9597            }
9598            SohmPhase::Resolve { ids, first } => {
9599                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9600                    return (flags | MSG_FLAG_SHAREABLE, body);
9601                }
9602                let key = (msg_type, body);
9603                match ids.get(&key) {
9604                    Some(&id) => pointer(id),
9605                    // The collect pass never saw this body — a dataspace a
9606                    // SWMR extend changed after the table was laid out, say.
9607                    // Left literal, which leaves a heap object counted for one
9608                    // reference more than reaches it and nothing else.
9609                    None => (flags, key.1),
9610                }
9611            }
9612        }
9613    }
9614
9615    /// Answer every shareable message at a heap pointer's width for the rest
9616    /// of this finalize's allocation phase.
9617    ///
9618    /// Half of the bracket [`prepare_shared_messages`](Self::prepare_shared_messages)
9619    /// closes, and the reason the two can sit on opposite sides of the
9620    /// allocation: a header cannot be measured until it is known which of its
9621    /// messages are pointers, and a body cannot be counted until every address
9622    /// it names exists. Only the width is knowable in the first phase, and the
9623    /// width is all the measurement needs.
9624    ///
9625    /// A finalize that will not lay a table out — a `finalize_for_swmr`, a
9626    /// second finalize over a table already published — leaves the phase where
9627    /// it found it, so what that pass measures is what it writes.
9628    fn begin_shared_message_layout(&self) {
9629        let Some(sohm) = self.sohm.as_deref() else {
9630            return;
9631        };
9632        let mut phase = sohm.phase.lock();
9633        if matches!(*phase, SohmPhase::Idle) && sohm.table_addr.lock().is_none() {
9634            *phase = SohmPhase::Predict(FirstCopies::default());
9635        }
9636    }
9637
9638    /// Lay out the file's shared-message table: count the bodies every header
9639    /// this finalize writes would share, put them in their index's heap, and
9640    /// arm the substitution the header builders then apply.
9641    ///
9642    /// The sole owner of the transition to `Resolve`. It runs last in the
9643    /// content phase, after
9644    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes),
9645    /// [`prepare_link_storage`](Self::prepare_link_storage) and
9646    /// [`write_reference_values`](Self::write_reference_values), because a
9647    /// body is only counted once it is the body the file will hold: an
9648    /// attribute that spilled into dense storage is not in a header to be
9649    /// shared at all, and one holding an object reference says an object
9650    /// header address that exists only after the allocation phase. Counting
9651    /// either of them earlier would count a body no header ends up carrying,
9652    /// and leave the header that carries the real one literal — which
9653    /// [`check_header_size`] would then refuse, the block having been
9654    /// reserved at a pointer's width.
9655    ///
9656    /// Once per file: a second finalize (a SWMR session's close) keeps the
9657    /// table the first one published rather than allocating a second one and
9658    /// stranding the first.
9659    fn prepare_shared_messages(&self, datasets: &[usize]) -> IoResult<()> {
9660        let Some(sohm) = self.sohm.as_deref() else {
9661            return Ok(());
9662        };
9663        if sohm.table_addr.lock().is_some() {
9664            return Ok(());
9665        }
9666
9667        // Collect: build every header this finalize will write and throw it
9668        // away, keeping only what its shareable messages were.
9669        *sohm.phase.lock() = SohmPhase::Collect(SohmCollector::new(sohm.indexes.len()));
9670        for &i in datasets {
9671            self.build_dataset_header(i)?;
9672        }
9673        for gi in 0..self.group_count() {
9674            if self.grp(gi).lock().deleted {
9675                continue;
9676            }
9677            self.build_group_header(gi)?;
9678        }
9679        self.build_root_group_header()?;
9680        let SohmPhase::Collect(collector) =
9681            std::mem::replace(&mut *sohm.phase.lock(), SohmPhase::Idle)
9682        else {
9683            return Err(crate::io::IoError::InvalidState(
9684                "the shared-message collect pass did not finish in the collect phase".into(),
9685            ));
9686        };
9687
9688        let indexes: Vec<SohmIndexContent> = sohm
9689            .indexes
9690            .iter()
9691            .zip(collector.messages)
9692            .map(|(&spec, messages)| SohmIndexContent { spec, messages })
9693            .collect();
9694        // The table a reopen found is superseded whole by the one below, and
9695        // every header that pointed into it is in this finalize's rewrite set
9696        // — so its blocks go back immediately before the replacement is laid
9697        // out, and the new table lands in them instead of growing the file on
9698        // every open/close cycle. Taken, not read: a second finalize must not
9699        // free the same blocks twice.
9700        for (addr, len) in std::mem::take(&mut *sohm.superseded.lock()) {
9701            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9702        }
9703        let built = build_shared_messages(&indexes, &self.ctx, &mut |len| {
9704            self.allocator.allocate(len, FreeSpaceClass::Metadata)
9705        })?;
9706        for block in &built.blocks {
9707            self.handle.write_at(block.addr, &block.image)?;
9708        }
9709
9710        // Only now, with every block on disk: from here the header builders
9711        // substitute pointers, and `write_superblock_extension` names the
9712        // table this laid out.
9713        *sohm.phase.lock() = SohmPhase::Resolve {
9714            ids: built.heap_ids,
9715            first: FirstCopies::default(),
9716        };
9717        *sohm.table_addr.lock() = Some(built.table_addr);
9718        Ok(())
9719    }
9720
9721    /// Write the file's free-space managers over the space this close leaves
9722    /// free, and return the file-space info message body naming them.
9723    ///
9724    /// Called from [`write_superblock_extension`](Self::write_superblock_extension)
9725    /// once every other block of the file has an address, which is what makes
9726    /// the allocator's free list the file's *final* free space: a block
9727    /// allocated after this point would land in space a manager still claims.
9728    ///
9729    /// INVARIANT: from the moment this returns, every byte the allocator holds
9730    /// free is a byte some sections block records, and the two blocks each
9731    /// manager itself occupies are held by neither. Nothing may allocate
9732    /// between here and the superblock write; `write_object_headers` writes
9733    /// over blocks reserved in an earlier phase and is the only thing that
9734    /// runs in between.
9735    ///
9736    /// Returns `None` for a file with no message of its own to write — a
9737    /// reopen whose carried message this session must not touch, and a file
9738    /// created at the library defaults — which leaves both byte-identical to
9739    /// what the same close wrote before free space was recorded at all. A file
9740    /// that carries the message but keeps no managers (either non-manager
9741    /// strategy, or `persist: false`) gets the message back with every address
9742    /// undefined, which is what `H5F__super_init` writes for it.
9743    fn write_free_space_managers(&self) -> IoResult<Option<Vec<u8>>> {
9744        let Some(fs) = self.free_space.as_deref() else {
9745            return Ok(None);
9746        };
9747        if !fs.records_free_space() {
9748            return Ok(Some(fs.info.encode(&self.ctx)?));
9749        }
9750        // The managers a reopen found are superseded whole by the ones below,
9751        // so their blocks go back before anything is laid out: the space the
9752        // old manager occupied is free space the new one records, and the new
9753        // one may be laid out in it.
9754        for &(addr, len) in &fs.superseded {
9755            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9756        }
9757
9758        let hdr_size = FreeSpaceHeader::encoded_size(&self.ctx) as u64;
9759        let settled = self.settle_free_space_managers(hdr_size, fs.info.threshold)?;
9760
9761        let mut info = fs.info.clone();
9762        info.fs_addr = vec![UNDEF_ADDR; info.fs_addr.len()];
9763        for placed in &settled {
9764            let mut header = manager_header(&placed.sections);
9765            // The settle loop sized the block; that the encode agrees is the
9766            // invariant that makes `sect_size` a length a reader can trust.
9767            let needed = free_space::sinfo_encoded_size(&header, &placed.sections, &self.ctx);
9768            if needed > placed.sect_size {
9769                return Err(crate::io::IoError::InvalidState(format!(
9770                    "the free-space sections need {needed} bytes, not the {} laid out",
9771                    placed.sect_size
9772                )));
9773            }
9774            header.sect_addr = placed.sect_addr;
9775            header.sect_size = placed.sect_size;
9776            header.alloc_sect_size = placed.sect_size;
9777            self.handle.write_at(
9778                placed.sect_addr,
9779                &free_space::encode_sections(
9780                    &header,
9781                    placed.hdr_addr,
9782                    &placed.sections,
9783                    placed.sect_size as usize,
9784                    &self.ctx,
9785                ),
9786            )?;
9787            self.handle
9788                .write_at(placed.hdr_addr, &header.encode(&self.ctx))?;
9789            // `H5MF__close_delete_fstype` leaves a manager with no sections
9790            // without an address, so only the ones written name themselves.
9791            info.fs_addr[placed.manager.message_slot()] = placed.hdr_addr;
9792        }
9793        // The end of the file *after* the settle above, not before it, which
9794        // the field's name denies: it is 1.10 vintage, where two EOAs were
9795        // kept — one taken before the self-referential managers were placed
9796        // and one after (H5MF.c:3305 and 3382 in 1.10.11) — and the message
9797        // carried the first (1.10.11 H5MF.c:1833, 1999). 1.14 keeps one,
9798        // `f->shared->eoa_fsm_fsalloc`, read once the allocation loop has run
9799        // (H5MF.c:3234-3240) and encoded into this field by both close paths
9800        // (H5MF.c:1759, 1923); H5Fsuper.c:826 names it "the final eoa". A
9801        // 1.10 reader wants that value and not the older one: equal EOAs are
9802        // the case `H5MF_tidy_self_referential_fsm_hack` returns on
9803        // (1.10.11 H5MF.c:3620-3622), which is what leaves the managers this
9804        // close wrote in place.
9805        info.eoa_pre_fsm_fsalloc = self.allocator.eof();
9806        Ok(Some(info.encode(&self.ctx)?))
9807    }
9808
9809    /// The file's free space as each manager will record it: address-ordered
9810    /// per manager, tagged with the section class that manager writes, and
9811    /// with everything below `threshold` left out.
9812    ///
9813    /// The allocator is the single owner of merging — `H5FS__sect_merge`'s
9814    /// rules, per manager and, on a paged file, per page — so nothing merges
9815    /// here; overlap is checked because two overlapping sections would be a
9816    /// manager claiming space another structure holds.
9817    fn free_sections(&self, threshold: u64) -> IoResult<Vec<(FreeSpaceManager, Vec<FreeSection>)>> {
9818        let policy = self.allocator.policy();
9819        let extents = self.allocator.free_extents();
9820        let mut sets = Vec::new();
9821        for manager in FreeSpaceManager::ALL {
9822            let mut sections: Vec<FreeSection> = extents
9823                .iter()
9824                .filter(|b| b.manager == manager)
9825                // `H5FS_sect_add` refuses a section below the file's
9826                // threshold, so a block smaller than it is space the file
9827                // leaks rather than records — the same trade the threshold is
9828                // there to make.
9829                .filter(|b| b.len >= threshold)
9830                .map(|b| FreeSection {
9831                    addr: b.addr,
9832                    len: b.len,
9833                    class: policy.section_class(manager),
9834                })
9835                .collect();
9836            sections.sort_unstable_by_key(|s| s.addr);
9837            if let Some(bad) = sections
9838                .windows(2)
9839                .find(|w| w[0].addr + w[0].len > w[1].addr)
9840            {
9841                return Err(crate::io::IoError::InvalidState(format!(
9842                    "this session freed overlapping blocks: {:#x}+{} overlaps {:#x}",
9843                    bad[0].addr, bad[0].len, bad[1].addr
9844                )));
9845            }
9846            sets.push((manager, sections));
9847        }
9848        Ok(sets)
9849    }
9850
9851    /// Give every manager that records anything its own header and sections
9852    /// blocks, and return what each will write.
9853    ///
9854    /// Self-referential, which is the whole difficulty: a manager's two blocks
9855    /// come out of the free space the managers record, and taking them changes
9856    /// that space, which changes how many bytes the sections block needs.
9857    /// Upstream reruns the allocation pass until no manager allocates anything
9858    /// further — the `do { ... } while (continue_alloc_fsm)` loop in
9859    /// `H5MF_settle_meta_data_fsm` (H5MF.c:3213-3247) around
9860    /// `H5FS_vfd_alloc_hdr_and_section_info_if_needed`, which allocates
9861    /// through `H5MF_alloc` like everything else. So does this: the blocks
9862    /// come out of the same [`FileAllocator`], under the same strategy, so a
9863    /// paged file's manager blocks land in pages and their page remainders are
9864    /// recorded like any others.
9865    ///
9866    /// Two rules make it terminate. A manager, once placed, stays placed: were
9867    /// its blocks released because its sections had been consumed, freeing
9868    /// them would put those sections back and the next round would place it
9869    /// again. And a sections block only ever grows: upstream frees a block
9870    /// that turned out too small and reallocates it next round
9871    /// (H5FSsection.c:2418-2423), and a size that only rises reaches its
9872    /// bound.
9873    fn settle_free_space_managers(
9874        &self,
9875        hdr_size: u64,
9876        threshold: u64,
9877    ) -> IoResult<Vec<PlacedManager>> {
9878        /// Rounds before the layout is called divergent. A round either places
9879        /// a manager or grows one sections block, and there are three
9880        /// managers, so a file that needs more than this is not converging.
9881        const ROUNDS: usize = 16;
9882
9883        // Raw data first and metadata last, in `H5MF_settle_raw_data_fsm`'s
9884        // order (H5C.c:689-696): every manager's own blocks are metadata
9885        // allocations, so the metadata manager funds all of them and is the
9886        // one whose section set the others change.
9887        const ORDER: [FreeSpaceManager; 3] = [
9888            FreeSpaceManager::RawData,
9889            FreeSpaceManager::Large,
9890            FreeSpaceManager::Metadata,
9891        ];
9892
9893        let size_of = |sections: &[FreeSection]| {
9894            let ordered = free_space::serialization_order(sections);
9895            free_space::sinfo_encoded_size(&manager_header(&ordered), &ordered, &self.ctx)
9896        };
9897        let mut placed: Vec<PlacedManager> = Vec::new();
9898        for _ in 0..ROUNDS {
9899            let sets = self.free_sections(threshold)?;
9900            let sections_of = |manager: FreeSpaceManager| {
9901                sets.iter()
9902                    .find(|(m, _)| *m == manager)
9903                    .map(|(_, s)| s.as_slice())
9904                    .unwrap_or_default()
9905            };
9906
9907            let mut changed = false;
9908            for manager in ORDER {
9909                let sections = sections_of(manager);
9910                if sections.is_empty() || placed.iter().any(|p| p.manager == manager) {
9911                    continue;
9912                }
9913                let sect_size = size_of(sections);
9914                let hdr_addr = self.allocator.allocate(hdr_size, FreeSpaceClass::Metadata);
9915                let sect_addr = self.allocator.allocate(sect_size, FreeSpaceClass::Metadata);
9916                placed.push(PlacedManager {
9917                    manager,
9918                    hdr_addr,
9919                    sect_addr,
9920                    sect_size,
9921                    sections: Vec::new(),
9922                });
9923                changed = true;
9924            }
9925            if !changed {
9926                for p in &mut placed {
9927                    let needed = size_of(sections_of(p.manager));
9928                    if needed > p.sect_size {
9929                        self.allocator
9930                            .free(p.sect_addr, p.sect_size, FreeSpaceClass::Metadata);
9931                        p.sect_size = needed;
9932                        p.sect_addr = self.allocator.allocate(needed, FreeSpaceClass::Metadata);
9933                        changed = true;
9934                    }
9935                }
9936            }
9937            if !changed {
9938                for p in &mut placed {
9939                    p.sections = free_space::serialization_order(sections_of(p.manager));
9940                }
9941                return Ok(placed);
9942            }
9943        }
9944        Err(crate::io::IoError::InvalidState(format!(
9945            "the free-space managers did not settle in {ROUNDS} rounds"
9946        )))
9947    }
9948
9949    /// Write the file's superblock extension, and the sole owner of that
9950    /// object header.
9951    ///
9952    /// Runs after [`prepare_shared_messages`](Self::prepare_shared_messages),
9953    /// whose table it names, and before the superblock that names it. What it
9954    /// writes is [`CarriedExtension`] — every message the reopened file's
9955    /// extension held — plus the shared-message table message, which is the
9956    /// one message whose content this session owns: the table moved, so the
9957    /// message read is stale and the message written names the new address.
9958    ///
9959    /// A file with neither carried messages nor shared messages gets no
9960    /// extension, which is what libhdf5 writes for it: `H5F__super_ext_create`
9961    /// is called only when there is a message to put in one.
9962    ///
9963    /// Version 1, holding its messages in one chunk: the extension is created
9964    /// before anything raises the file's object header version
9965    /// (`H5F__super_ext_create` passes `H5O_HDR_STORE_TIMES` off and takes the
9966    /// version-1 path), so an extension of any generation of file looks the
9967    /// same.
9968    fn write_superblock_extension(&self) -> IoResult<()> {
9969        if self.extension.addr.lock().is_some() {
9970            return Ok(());
9971        }
9972        let table = self.sohm.as_deref().and_then(|sohm| {
9973            sohm.table_addr
9974                .lock()
9975                .map(|addr| (sohm.indexes.len(), addr))
9976        });
9977        // A file with file-space properties of its own needs an extension
9978        // too: the message that declares them is the only place they are
9979        // recorded, and a file created with them carries nothing else.
9980        if self.extension.carried.is_empty() && table.is_none() && self.free_space.is_none() {
9981            return Ok(());
9982        }
9983
9984        let mut messages: Vec<crate::io::object_header_io::ExtensionMessage> =
9985            self.extension.carried.clone();
9986        if let Some(fs) = self.free_space.as_deref() {
9987            // The declared message, at exactly the length the one written
9988            // below will have — every field of it is fixed-width, and only
9989            // `persist` and the message version change the count of
9990            // addresses, neither of which the close alters. The image is sized
9991            // and its block allocated before the managers can be laid out, so
9992            // the message has to reach its final *length* here even though its
9993            // content is settled later.
9994            let declared = fs.info.encode(&self.ctx)?;
9995            match messages
9996                .iter_mut()
9997                .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
9998            {
9999                Some(msg) => msg.body = declared,
10000                None => messages.push(crate::io::object_header_io::ExtensionMessage {
10001                    msg_type: MSG_FILE_SPACE_INFO,
10002                    flags: MSG_FLAG_DONTSHARE | MSG_FLAG_MARK_IF_UNKNOWN,
10003                    body: declared,
10004                }),
10005            }
10006        }
10007        if let Some((nindexes, table_addr)) = table {
10008            let nindexes = u8::try_from(nindexes).map_err(|_| {
10009                crate::io::IoError::InvalidState(format!("{nindexes} shared-message indexes"))
10010            })?;
10011            messages.push(crate::io::object_header_io::ExtensionMessage {
10012                msg_type: MSG_SHARED_MESSAGE_TABLE,
10013                flags: MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
10014                body: SharedMessageTableMessage {
10015                    version: 0,
10016                    table_address: table_addr,
10017                    nindexes,
10018                }
10019                .encode(&self.ctx),
10020            });
10021        }
10022        let encode = |messages: &[crate::io::object_header_io::ExtensionMessage]| {
10023            let mut extension = ObjectHeader::new();
10024            for msg in messages {
10025                extension.add_message(msg.msg_type, msg.flags, msg.body.clone());
10026            }
10027            extension.encode_v1(1)
10028        };
10029        let image = encode(&messages)?;
10030        // Freed before the replacement is placed, so a reopen reuses the block
10031        // instead of stranding one per open/close cycle — the rule every other
10032        // superseded structure follows.
10033        for &(addr, len) in &self.extension.superseded {
10034            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
10035        }
10036        let addr = self
10037            .allocator
10038            .allocate(image.len() as u64, FreeSpaceClass::Metadata);
10039
10040        // Every block of this file now has an address, so the allocator holds
10041        // exactly the file's free space: settle the free-space managers over
10042        // it and say in this extension where they went.
10043        let image = match self.write_free_space_managers()? {
10044            None => image,
10045            Some(body) => {
10046                let msg = messages
10047                    .iter_mut()
10048                    .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10049                    .ok_or_else(|| {
10050                        crate::io::IoError::InvalidState(
10051                            "a persisting file lost its file-space info message".into(),
10052                        )
10053                    })?;
10054                // Same length as the declared body put in above, so the
10055                // image measured before the block was allocated still fits.
10056                if body.len() != msg.body.len() {
10057                    return Err(crate::io::IoError::InvalidState(format!(
10058                        "the file-space info message was laid out at {} bytes and \
10059                         written back at {}",
10060                        msg.body.len(),
10061                        body.len()
10062                    )));
10063                }
10064                msg.body = body;
10065                encode(&messages)?
10066            }
10067        };
10068        self.handle.write_at(addr, &image)?;
10069        *self.extension.addr.lock() = Some(addr);
10070        Ok(())
10071    }
10072
10073    /// Define a new contiguous dataset. Returns the dataset index (used with
10074    /// `write_dataset_raw`).
10075    ///
10076    /// The raw-data region is allocated immediately so that
10077    /// `write_dataset_raw` can be called at any time before `close()`.
10078    pub fn create_dataset(
10079        &self,
10080        name: &str,
10081        datatype: DatatypeMessage,
10082        dims: &[u64],
10083    ) -> IoResult<usize> {
10084        let create = self.begin_create(name)?;
10085        let name = create.name.as_str();
10086        let total_elements: u64 = if dims.is_empty() {
10087            1
10088        } else {
10089            dims.iter().product()
10090        };
10091        let element_size = datatype.element_size() as u64;
10092        let data_size = total_elements * element_size;
10093
10094        // Allocate space for the raw data.
10095        let data_addr = if data_size > 0 {
10096            self.allocator.allocate(data_size, FreeSpaceClass::RawData)
10097        } else {
10098            UNDEF_ADDR
10099        };
10100
10101        let dataspace = if dims.is_empty() {
10102            DataspaceMessage::scalar()
10103        } else {
10104            DataspaceMessage::simple(dims)
10105        };
10106
10107        let idx = self.push_dataset(
10108            &create,
10109            DatasetInfo {
10110                name: name.to_string(),
10111                datatype,
10112                committed_type: None,
10113                external: None,
10114                virtual_storage: None,
10115                dataspace,
10116                read_format: None,
10117                obj_header_addr: 0, // set during finalize
10118                data_addr,
10119                data_size,
10120                compact: None,
10121                chunked: None,
10122                fixed_array: None,
10123                implicit: None,
10124                single_chunk: None,
10125                btree_v1: None,
10126                btree_v2: None,
10127                append: None,
10128                attributes: Vec::new(),
10129                obj_header_written_addr: None,
10130                obj_header_blocks: Vec::new(),
10131                filter_pipeline: None,
10132                deleted: false,
10133                extent_dirty: false,
10134                header_dirty: false,
10135                nlink_written: 1,
10136                creation_seq: self.take_creation_seq(),
10137                track_attr_order: self.track_order.attrs,
10138                fill_value: None,
10139                fill_time: FILL_TIME_IFSET,
10140                layout_version: 4,
10141                times: self.created_object_times(),
10142            },
10143        );
10144
10145        Ok(idx)
10146    }
10147
10148    /// Define a new dataset whose raw data lives in files outside this one —
10149    /// `H5Pset_external`, h5py's `external=[(name, offset, size)]`.
10150    ///
10151    /// Each entry names a file, the byte offset in it where that entry's
10152    /// region starts, and how many bytes of the dataset the region holds; the
10153    /// entries concatenate, in order, into the dataset's logical byte range,
10154    /// and together must cover it. Nothing is allocated in this file: the data
10155    /// layout message says contiguous storage at an undefined address, and it
10156    /// is the External File List beside it that says where the bytes are
10157    /// (`H5D__layout_oh_create`).
10158    ///
10159    /// A named file is created on first write and never truncated, so several
10160    /// slots — or several datasets — may own disjoint ranges of one file, the
10161    /// way `H5D__efl_write` opens them.
10162    ///
10163    /// The last slot may take the unlimited size `H5O_EFL_UNLIMITED`, which
10164    /// makes it absorb however many bytes the dataset comes to hold; a
10165    /// dataset whose dataspace is unlimited must have one, since nothing
10166    /// finite could cover it (`H5D__efl_construct`: "unlimited dataspace but
10167    /// finite storage"). Only the first dimension may be extendible, which is
10168    /// the same function's other rule.
10169    pub fn create_external_dataset(
10170        &self,
10171        name: &str,
10172        datatype: DatatypeMessage,
10173        dims: &[u64],
10174        max_dims: Option<&[u64]>,
10175        files: &[(&str, u64, u64)],
10176    ) -> IoResult<usize> {
10177        if files.is_empty() {
10178            return Err(crate::io::IoError::InvalidState(format!(
10179                "external dataset '{name}' names no files; external storage is defined by \
10180                 the files it lives in, so at least one is required"
10181            )));
10182        }
10183        let create = self.begin_create(name)?;
10184        let name = create.name.as_str();
10185        let total_elements: u64 = if dims.is_empty() {
10186            1
10187        } else {
10188            dims.iter().product()
10189        };
10190        let data_size = total_elements * datatype.element_size() as u64;
10191
10192        let mut heap = LocalHeapImage::with_empty_string();
10193        let mut entries = Vec::with_capacity(files.len());
10194        for (i, &(file_name, offset, size)) in files.iter().enumerate() {
10195            if file_name.is_empty() {
10196                return Err(crate::io::IoError::InvalidState(format!(
10197                    "external dataset '{name}' has a slot with an empty file name"
10198                )));
10199            }
10200            // `H5Pset_external` refuses to add a slot behind an unlimited one
10201            // ("previous file size is unlimited"): the unlimited slot already
10202            // owns every byte from its own start onwards, so nothing after it
10203            // could ever be reached.
10204            if size == UNLIMITED && i + 1 != files.len() {
10205                return Err(crate::io::IoError::InvalidState(format!(
10206                    "external dataset '{name}' gives slot {i} ('{file_name}') the unlimited \
10207                     size H5O_EFL_UNLIMITED with {} slot(s) behind it; an unlimited slot \
10208                     absorbs the rest of the dataset, so it can only be the last",
10209                    files.len() - i - 1
10210                )));
10211            }
10212            if offset.checked_add(size).is_none() {
10213                return Err(crate::io::IoError::InvalidState(format!(
10214                    "external dataset '{name}' slot '{file_name}' spans offset {offset} \
10215                     plus {size} bytes, past the end of the 64-bit address space"
10216                )));
10217            }
10218            entries.push(ExternalFile {
10219                name: file_name.to_string(),
10220                name_offset: heap.insert_str(file_name),
10221                offset,
10222                size,
10223            });
10224        }
10225        let external = ExternalStorage {
10226            // Filled in below, once the heap the names went into has an
10227            // address; the names' offsets within it are already final.
10228            heap_addr: UNDEF_ADDR,
10229            files: entries,
10230            // Settled by the open this create hands a handle out for, which
10231            // is `H5D__create` reading the dapl at H5Dint.c:1318.
10232            prefix: EfilePrefix::default(),
10233        };
10234        // `H5D__efl_construct`, over the dataset's *maximum* extent: the
10235        // slots must reserve at least every byte the dataset could come to
10236        // hold, and an unlimited extent can only be covered by an unlimited
10237        // last slot ("unlimited dataspace but finite storage").
10238        let max_dims = max_dims.unwrap_or(dims);
10239        if max_dims.len() != dims.len() {
10240            return Err(crate::io::IoError::InvalidState(format!(
10241                "external dataset '{name}' has {} dimensions but {} maximum ones",
10242                dims.len(),
10243                max_dims.len()
10244            )));
10245        }
10246        for (d, (&max, &cur)) in max_dims.iter().zip(dims).enumerate().skip(1) {
10247            if max > cur {
10248                return Err(crate::io::IoError::InvalidState(format!(
10249                    "external dataset '{name}' makes dimension {d} extendible ({cur} of \
10250                     {max}); only the first dimension can be extendible for external storage"
10251                )));
10252            }
10253        }
10254        let reserved = external.total_size();
10255        if max_dims.contains(&u64::MAX) {
10256            if reserved != UNLIMITED {
10257                return Err(crate::io::IoError::InvalidState(format!(
10258                    "external dataset '{name}' has an unlimited dataspace but its files \
10259                     reserve only {reserved} bytes; the last slot must take the unlimited \
10260                     size H5O_EFL_UNLIMITED"
10261                )));
10262            }
10263        } else {
10264            let max_bytes = max_dims
10265                .iter()
10266                .try_fold(datatype.element_size() as u64, |acc, &d| acc.checked_mul(d))
10267                .ok_or_else(|| {
10268                    crate::io::IoError::InvalidState(format!(
10269                        "external dataset '{name}' maximum extent times its element size \
10270                         overflows 64 bits"
10271                    ))
10272                })?;
10273            if reserved < max_bytes {
10274                return Err(crate::io::IoError::InvalidState(format!(
10275                    "external dataset '{name}' needs {max_bytes} bytes but its files reserve \
10276                     only {reserved}"
10277                )));
10278            }
10279        }
10280
10281        // The names' heap, written now: it is ordinary metadata of this file,
10282        // and the message the header carries is only an address into it.
10283        let sa = self.ctx.sizeof_addr as usize;
10284        let ss = self.ctx.sizeof_size as usize;
10285        let heap_bytes = heap.as_bytes().to_vec();
10286        let heap_addr = self.allocator.allocate(
10287            local_heap_header_size(sa, ss) as u64,
10288            FreeSpaceClass::Metadata,
10289        );
10290        let heap_data_addr = self
10291            .allocator
10292            .allocate(heap_bytes.len() as u64, FreeSpaceClass::Metadata);
10293        let heap_hdr = LocalHeapHeader {
10294            data_size: heap_bytes.len() as u64,
10295            // Sized to hold exactly these names, so no block of it is free.
10296            free_list_offset: LOCAL_HEAP_FREE_NULL,
10297            data_addr: heap_data_addr,
10298        };
10299        self.handle.write_at(heap_addr, &heap_hdr.encode(sa, ss))?;
10300        self.handle.write_at(heap_data_addr, &heap_bytes)?;
10301        let external = ExternalStorage {
10302            heap_addr,
10303            ..external
10304        };
10305
10306        let dataspace = if dims.is_empty() {
10307            DataspaceMessage::scalar()
10308        } else {
10309            let mut ds = DataspaceMessage::simple(dims);
10310            if max_dims != dims {
10311                ds.max_dims = Some(max_dims.to_vec());
10312            }
10313            ds
10314        };
10315
10316        let idx = self.push_dataset(
10317            &create,
10318            DatasetInfo {
10319                name: name.to_string(),
10320                datatype,
10321                committed_type: None,
10322                external: Some(external),
10323                virtual_storage: None,
10324                dataspace,
10325                read_format: None,
10326                obj_header_addr: 0, // set during finalize
10327                // No block of this file's own: the layout message declares
10328                // contiguous storage at an undefined address, which is what
10329                // sends a reader to the external file list instead.
10330                data_addr: UNDEF_ADDR,
10331                data_size,
10332                compact: None,
10333                chunked: None,
10334                fixed_array: None,
10335                btree_v2: None,
10336                implicit: None,
10337                single_chunk: None,
10338                btree_v1: None,
10339                append: None,
10340                attributes: Vec::new(),
10341                obj_header_written_addr: None,
10342                obj_header_blocks: Vec::new(),
10343                filter_pipeline: None,
10344                deleted: false,
10345                extent_dirty: false,
10346                header_dirty: false,
10347                nlink_written: 1,
10348                creation_seq: self.take_creation_seq(),
10349                track_attr_order: self.track_order.attrs,
10350                fill_value: None,
10351                fill_time: FILL_TIME_IFSET,
10352                layout_version: 4,
10353                times: self.created_object_times(),
10354            },
10355        );
10356
10357        Ok(idx)
10358    }
10359
10360    /// Define a new virtual dataset — `H5Pset_virtual`, h5py's
10361    /// `create_virtual_dataset(name, VirtualLayout)`.
10362    ///
10363    /// Each mapping says which elements of this dataset (`virtual_selection`)
10364    /// are read from which elements (`source_selection`) of a dataset in
10365    /// another file; the sources are never opened here, and a mapping naming
10366    /// one that does not exist yet is perfectly legal — libhdf5 resolves each
10367    /// at read time, filling from the fill value where nothing maps.
10368    ///
10369    /// The mappings do not live in the object header: they are serialized
10370    /// into one global heap object and the layout message carries only its
10371    /// address and index (`H5D__virtual_store_layout`), which is why this
10372    /// allocates a heap object and nothing else.
10373    ///
10374    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: the
10375    /// mapping grows with its source, and the virtual dataset's extent in
10376    /// that dimension is whatever the sources reachable at read time supply
10377    /// (`H5D__virtual_set_extent_unlim`). A `printf`-style source name is
10378    /// written as one too: `%b` substitutes the block index, so one mapping
10379    /// stands for the family of source datasets that fill the successive
10380    /// blocks of an unlimited virtual selection.
10381    pub fn create_virtual_dataset(
10382        &self,
10383        name: &str,
10384        datatype: DatatypeMessage,
10385        dims: &[u64],
10386        max_dims: Option<&[u64]>,
10387        mappings: &[VirtualMapping],
10388    ) -> IoResult<usize> {
10389        if mappings.is_empty() {
10390            return Err(crate::io::IoError::InvalidState(format!(
10391                "virtual dataset '{name}' names no mappings; a virtual dataset is defined \
10392                 by the source datasets it maps, so at least one is required"
10393            )));
10394        }
10395        for m in mappings {
10396            check_virtual_mapping(name, m)?;
10397        }
10398
10399        let create = self.begin_create(name)?;
10400        let name = create.name.as_str();
10401
10402        // The mapping list is ordinary file metadata, written now: the header
10403        // built at finalize carries only the heap address and object index it
10404        // lands at.
10405        let block = VirtualMappingList {
10406            mappings: mappings.to_vec(),
10407        }
10408        .encode(&self.ctx)?;
10409        let (heap_addr, heap_index) = self.insert_vlen_objects(&[&block])?[0];
10410
10411        let dataspace = if dims.is_empty() {
10412            DataspaceMessage::scalar()
10413        } else {
10414            let mut ds = DataspaceMessage::simple(dims);
10415            // A caller that named no maximum gets the current dimensions, the
10416            // maximum `simple` already filled in: `H5Screate_simple(rank,
10417            // dims, NULL)` reaches the encoder with `extent.max` set
10418            // (H5S.c:1293-1299), so leaving it absent here would write a
10419            // message no upstream API call can produce.
10420            if let Some(max) = max_dims {
10421                ds.max_dims = Some(max.to_vec());
10422            }
10423            ds
10424        };
10425
10426        let idx = self.push_dataset(
10427            &create,
10428            DatasetInfo {
10429                name: name.to_string(),
10430                datatype,
10431                committed_type: None,
10432                external: None,
10433                virtual_storage: Some(VirtualStorage {
10434                    heap_addr,
10435                    heap_index: heap_index as u32,
10436                    mappings: mappings.to_vec(),
10437                }),
10438                dataspace,
10439                read_format: None,
10440                obj_header_addr: 0, // set during finalize
10441                // Not a block of this file at all: every element is read out
10442                // of a source dataset, so there is nothing here to allocate
10443                // and nothing to free when the dataset is deleted.
10444                data_addr: UNDEF_ADDR,
10445                data_size: 0,
10446                compact: None,
10447                chunked: None,
10448                fixed_array: None,
10449                btree_v2: None,
10450                implicit: None,
10451                single_chunk: None,
10452                btree_v1: None,
10453                append: None,
10454                attributes: Vec::new(),
10455                obj_header_written_addr: None,
10456                obj_header_blocks: Vec::new(),
10457                filter_pipeline: None,
10458                deleted: false,
10459                extent_dirty: false,
10460                header_dirty: false,
10461                nlink_written: 1,
10462                creation_seq: self.take_creation_seq(),
10463                track_attr_order: self.track_order.attrs,
10464                fill_value: None,
10465                fill_time: FILL_TIME_IFSET,
10466                layout_version: 4,
10467                times: self.created_object_times(),
10468            },
10469        );
10470
10471        Ok(idx)
10472    }
10473
10474    /// Define a new compact dataset — `H5Pset_layout(dcpl, H5D_COMPACT)`.
10475    ///
10476    /// The raw data lives inside the data layout message in the dataset's own
10477    /// object header, so it costs no block of its own and no extra seek to
10478    /// read; the price is the ceiling, and that the whole image is rewritten
10479    /// whenever the header is. The buffer is created at its final length and
10480    /// zero-filled, which is what `H5D__compact_fill` does at create time, so
10481    /// a dataset never written still reads back as its fill value.
10482    ///
10483    /// Errors when the image exceeds [`MAX_COMPACT_DATA`].
10484    pub fn create_compact_dataset(
10485        &self,
10486        name: &str,
10487        datatype: DatatypeMessage,
10488        dims: &[u64],
10489    ) -> IoResult<usize> {
10490        let total_elements: u64 = if dims.is_empty() {
10491            1
10492        } else {
10493            dims.iter().product()
10494        };
10495        let data_size = total_elements * datatype.element_size() as u64;
10496        if data_size > MAX_COMPACT_DATA as u64 {
10497            return Err(crate::io::IoError::InvalidState(format!(
10498                "compact dataset '{name}' needs {data_size} bytes, above the \
10499                 {MAX_COMPACT_DATA}-byte ceiling a data layout message can hold; \
10500                 use contiguous or chunked storage"
10501            )));
10502        }
10503
10504        let create = self.begin_create(name)?;
10505        let name = create.name.as_str();
10506        let dataspace = if dims.is_empty() {
10507            DataspaceMessage::scalar()
10508        } else {
10509            DataspaceMessage::simple(dims)
10510        };
10511
10512        let idx = self.push_dataset(
10513            &create,
10514            DatasetInfo {
10515                name: name.to_string(),
10516                datatype,
10517                committed_type: None,
10518                external: None,
10519                virtual_storage: None,
10520                dataspace,
10521                read_format: None,
10522                obj_header_addr: 0, // set during finalize
10523                data_addr: UNDEF_ADDR,
10524                data_size: 0,
10525                compact: Some(vec![0u8; data_size as usize]),
10526                chunked: None,
10527                fixed_array: None,
10528                implicit: None,
10529                single_chunk: None,
10530                btree_v1: None,
10531                btree_v2: None,
10532                append: None,
10533                attributes: Vec::new(),
10534                obj_header_written_addr: None,
10535                obj_header_blocks: Vec::new(),
10536                filter_pipeline: None,
10537                deleted: false,
10538                extent_dirty: false,
10539                header_dirty: false,
10540                nlink_written: 1,
10541                creation_seq: self.take_creation_seq(),
10542                track_attr_order: self.track_order.attrs,
10543                fill_value: None,
10544                fill_time: FILL_TIME_IFSET,
10545                layout_version: 4,
10546                times: self.created_object_times(),
10547            },
10548        );
10549
10550        Ok(idx)
10551    }
10552
10553    /// Define a new dataset with the NULL dataspace: no elements at all.
10554    ///
10555    /// Distinct from a scalar dataset (`create_dataset` with `dims == []`),
10556    /// which holds exactly one element — a NULL dataspace holds zero, so
10557    /// there is no raw image to allocate: `data_addr` stays `UNDEF_ADDR` and
10558    /// `data_size` stays 0 permanently, the same terminal state
10559    /// `create_dataset` already reaches for a zero-length dimension.
10560    pub fn create_null_dataset(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
10561        let create = self.begin_create(name)?;
10562        let name = create.name.as_str();
10563
10564        let idx = self.push_dataset(
10565            &create,
10566            DatasetInfo {
10567                name: name.to_string(),
10568                datatype,
10569                committed_type: None,
10570                external: None,
10571                virtual_storage: None,
10572                dataspace: DataspaceMessage::null(),
10573                read_format: None,
10574                obj_header_addr: 0, // set during finalize
10575                data_addr: UNDEF_ADDR,
10576                data_size: 0,
10577                compact: None,
10578                chunked: None,
10579                fixed_array: None,
10580                implicit: None,
10581                single_chunk: None,
10582                btree_v1: None,
10583                btree_v2: None,
10584                append: None,
10585                attributes: Vec::new(),
10586                obj_header_written_addr: None,
10587                obj_header_blocks: Vec::new(),
10588                filter_pipeline: None,
10589                deleted: false,
10590                extent_dirty: false,
10591                header_dirty: false,
10592                nlink_written: 1,
10593                creation_seq: self.take_creation_seq(),
10594                track_attr_order: self.track_order.attrs,
10595                fill_value: None,
10596                fill_time: FILL_TIME_IFSET,
10597                layout_version: 4,
10598                times: self.created_object_times(),
10599            },
10600        );
10601
10602        Ok(idx)
10603    }
10604
10605    /// Define a new chunked dataset with an extensible array index.
10606    ///
10607    /// Returns the dataset index. The dataset starts empty (dims[0] = 0 if
10608    /// the first dimension is unlimited). Use `write_chunk` and
10609    /// `extend_dataset` to add data.
10610    pub fn create_chunked_dataset(
10611        &self,
10612        name: &str,
10613        datatype: DatatypeMessage,
10614        dims: &[u64],
10615        max_dims: &[u64],
10616        chunk_dims: &[u64],
10617    ) -> IoResult<usize> {
10618        let create = self.begin_create(name)?;
10619        let name = create.name.as_str();
10620        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
10621        ensure_at_most_one_unlimited(max_dims)?;
10622        let chunk_bytes = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
10623        let layout_version = self.chunk_layout_version(false, chunk_bytes);
10624        let earray_params = EarrayParams::default_params();
10625        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
10626        let nsblk_addrs = compute_nsblk_addrs(
10627            earray_params.idx_blk_elmts,
10628            earray_params.data_blk_min_elmts,
10629            earray_params.sup_blk_min_data_ptrs,
10630            earray_params.max_nelmts_bits,
10631        )?;
10632
10633        // Create EA header
10634        let mut ea_header = ExtensibleArrayHeader::new_for_chunks(&self.ctx);
10635        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
10636        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
10637        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
10638        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
10639        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
10640
10641        // Allocate and write EA header (placeholder, will be updated)
10642        let hdr_encoded = ea_header.encode(&self.ctx);
10643        let ea_header_addr = self
10644            .allocator
10645            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
10646
10647        // Create EA index block with pre-allocated super block address slots
10648        let ea_iblk = ExtensibleArrayIndexBlock::new(
10649            ea_header_addr,
10650            earray_params.idx_blk_elmts,
10651            ndblk_addrs,
10652            nsblk_addrs,
10653        );
10654
10655        // Allocate and write EA index block
10656        let iblk_encoded = ea_iblk.encode(&self.ctx);
10657        let ea_iblk_addr = self
10658            .allocator
10659            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
10660
10661        // Update header with index block address
10662        ea_header.idx_blk_addr = ea_iblk_addr;
10663
10664        // Write both to disk
10665        let hdr_encoded = ea_header.encode(&self.ctx);
10666        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
10667        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
10668
10669        // Build dataspace with max dims
10670        let dataspace = DataspaceMessage {
10671            // Chunked storage always requires at least one dimension, so
10672            // this is never Scalar or Null.
10673            class: DataspaceClass::Simple,
10674            dims: dims.to_vec(),
10675            max_dims: Some(max_dims.to_vec()),
10676        };
10677
10678        let idx = self.push_dataset(
10679            &create,
10680            DatasetInfo {
10681                name: name.to_string(),
10682                datatype,
10683                committed_type: None,
10684                external: None,
10685                virtual_storage: None,
10686                dataspace,
10687                read_format: None,
10688                obj_header_addr: 0,
10689                data_addr: UNDEF_ADDR,
10690                data_size: 0,
10691                compact: None,
10692                attributes: Vec::new(),
10693                obj_header_written_addr: None,
10694                obj_header_blocks: Vec::new(),
10695                filter_pipeline: None,
10696                deleted: false,
10697                extent_dirty: false,
10698                header_dirty: false,
10699                nlink_written: 1,
10700                creation_seq: self.take_creation_seq(),
10701                track_attr_order: self.track_order.attrs,
10702                fill_value: None,
10703                fill_time: FILL_TIME_IFSET,
10704                layout_version,
10705                times: self.created_object_times(),
10706                fixed_array: None,
10707                implicit: None,
10708                single_chunk: None,
10709                btree_v1: None,
10710                btree_v2: None,
10711                chunked: Some(ChunkedDatasetInfo {
10712                    chunk_dims: chunk_dims.to_vec(),
10713                    earray_params,
10714                    ea_header_addr,
10715                    ea_iblk_addr,
10716                    ea_header,
10717                    ea_iblk,
10718                    chunks_written: 0,
10719                    filt_iblk: None,
10720                    chunk_size_len: 0,
10721                }),
10722                append: None,
10723            },
10724        );
10725
10726        Ok(idx)
10727    }
10728
10729    /// Write `data` into a contiguous dataset's raw storage at *dataset-
10730    /// relative* byte offset `off`.
10731    ///
10732    /// The single owner of a contiguous raw-data write. Which storage that is
10733    /// — a block of this file, or the files an External File List names — is
10734    /// decided once, by [`DatasetInfo::contiguous_target`], and never at a
10735    /// call site.
10736    fn write_contiguous_bytes(
10737        &self,
10738        target: &ContiguousTarget,
10739        off: u64,
10740        data: &[u8],
10741    ) -> IoResult<()> {
10742        match target {
10743            ContiguousTarget::Local(addr) => Ok(self.handle.write_at(addr + off, data)?),
10744            ContiguousTarget::External { files, prefix } => {
10745                // The prefix the open settled, not one resolved here:
10746                // `H5D__efl_write` joins against `dset->shared->extfile_prefix`
10747                // (H5Defl.c:429-431), the same field `H5D__efl_read` joins
10748                // against, so a relative name lands where a later read looks.
10749                write_external_file_bytes(files, prefix.as_deref(), off, data)
10750            }
10751            ContiguousTarget::Virtual => Err(virtual_write_refused()),
10752        }
10753    }
10754
10755    /// Write raw bytes to a contiguous dataset identified by `index`.
10756    ///
10757    /// The caller is responsible for providing data in the correct byte order
10758    /// and layout. The length must match the total data size declared at
10759    /// creation time.
10760    pub fn write_dataset_raw(&self, index: usize, data: &[u8]) -> IoResult<()> {
10761        let ds = self.ds(index);
10762        let _op = ds.op.lock();
10763        let target = {
10764            let mut g = ds.lock();
10765            if g.is_chunked() {
10766                return Err(crate::io::IoError::InvalidState(
10767                    "use write_chunk for chunked datasets".into(),
10768                ));
10769            }
10770            // A compact dataset's raw image is its layout message, so the
10771            // write lands in the buffer the header is built from rather than
10772            // at a file offset, and the header it is built into is now stale.
10773            if let Some(image) = g.compact.as_mut() {
10774                if data.len() != image.len() {
10775                    return Err(crate::io::IoError::InvalidState(format!(
10776                        "data size mismatch: expected {} bytes, got {}",
10777                        image.len(),
10778                        data.len()
10779                    )));
10780                }
10781                image.copy_from_slice(data);
10782                g.header_dirty = true;
10783                return Ok(());
10784            }
10785            let Some(target) = g.contiguous_target() else {
10786                return Err(crate::io::IoError::InvalidState(
10787                    "dataset has no data allocated".into(),
10788                ));
10789            };
10790            // A dataset that stores nothing of its own has no byte count to
10791            // check a write against — `write_contiguous_bytes` refuses it by
10792            // name below, which is the answer the caller needs.
10793            if target.is_storage() && data.len() as u64 != g.data_size {
10794                return Err(crate::io::IoError::InvalidState(format!(
10795                    "data size mismatch: expected {} bytes, got {}",
10796                    g.data_size,
10797                    data.len()
10798                )));
10799            }
10800            target
10801        };
10802        self.write_contiguous_bytes(&target, 0, data)
10803    }
10804
10805    /// Write a chunk of data to a chunked dataset.
10806    ///
10807    /// `chunk_offset` is the chunk coordinates (e.g., [frame_idx] for a 1D-chunked
10808    /// streaming dataset where chunk_dims = [1, H, W]).
10809    /// Only the first (unlimited) dimension index is used for EA indexing.
10810    ///
10811    /// `data` must be exactly chunk_size bytes (product of chunk_dims * element_size).
10812    pub fn write_chunk(&self, index: usize, chunk_idx: u64, data: &[u8]) -> IoResult<()> {
10813        let ds = self.ds(index);
10814        let _op = ds.op.lock();
10815        self.write_chunk_inner(index, chunk_idx, data)
10816    }
10817
10818    /// [`Self::write_chunk`] body; the caller holds the dataset's op lock or
10819    /// the writer exclusively.
10820    pub(crate) fn write_chunk_inner(
10821        &self,
10822        index: usize,
10823        chunk_idx: u64,
10824        data: &[u8],
10825    ) -> IoResult<()> {
10826        let ds = self.ds(index);
10827        // Read the chunk geometry and filter pipeline under one brief lock,
10828        // then drop it: compression runs *outside* the lock, and
10829        // `record_ea_chunk` re-locks the same slot, so the guard must not be
10830        // held across either.
10831        let (chunk_bytes, pipeline) = {
10832            let g = ds.lock();
10833            let element_size = g.datatype.element_size() as u64;
10834            let chunked = g
10835                .chunked
10836                .as_ref()
10837                .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?;
10838            (
10839                chunked.chunk_dims.iter().product::<u64>() * element_size,
10840                g.filter_pipeline.clone(),
10841            )
10842        };
10843
10844        if data.len() as u64 != chunk_bytes {
10845            return Err(crate::io::IoError::InvalidState(format!(
10846                "chunk data size mismatch: expected {} bytes, got {}",
10847                chunk_bytes,
10848                data.len()
10849            )));
10850        }
10851
10852        // Apply compression if filter pipeline is set
10853        let compressed;
10854        let write_data = if let Some(ref pipeline) = pipeline {
10855            compressed = filter::apply_filters(pipeline, data)?;
10856            &compressed
10857        } else {
10858            data
10859        };
10860        // filter_mask = 0: this path runs the whole pipeline, so no filter is
10861        // skipped for the chunk.
10862        self.record_ea_chunk(index, chunk_idx, write_data, 0)
10863    }
10864
10865    /// Decide where a chunk's bytes belong and put them there, returning the
10866    /// address to record in the index.
10867    ///
10868    /// `old` is the chunk's current `(address, stored length)` if the index
10869    /// already holds an entry for it. This is the single owner of the
10870    /// rewrite-placement rule, mirroring libhdf5's `H5D__chunk_file_alloc`
10871    /// (`H5Dchunk.c`): a chunk whose stored size is unchanged is overwritten
10872    /// where it already lives, and only a chunk that no longer fits moves,
10873    /// releasing its old block. Without this every rewrite would abandon the
10874    /// old block and grow the file.
10875    fn place_chunk(&self, old: Option<(u64, u64)>, new_len: u64) -> u64 {
10876        match old {
10877            // Same stored size: overwrite in place. This is every unfiltered
10878            // rewrite (the stored size is fixed by the chunk shape) and every
10879            // filtered rewrite that compressed to the same length.
10880            Some((addr, len)) if addr != UNDEF_ADDR && len == new_len => addr,
10881            Some((addr, len)) if addr != UNDEF_ADDR => {
10882                // The chunk has to move. Under SWMR a reader may still hold an
10883                // index that points at the old block, so libhdf5 keeps it
10884                // (H5D__chunk_file_alloc skips H5MF_xfree when the file is
10885                // open for SWMR writing); do the same.
10886                if !self.swmr_active {
10887                    self.allocator.free(addr, len, FreeSpaceClass::RawData);
10888                }
10889                self.allocator.allocate(new_len, FreeSpaceClass::RawData)
10890            }
10891            _ => self.allocator.allocate(new_len, FreeSpaceClass::RawData),
10892        }
10893    }
10894
10895    /// Place a chunk's already-final bytes (filtered if the dataset is
10896    /// filtered) in the file and record them in the extensible-array index —
10897    /// in the index block, a data block, or a super block per the EA geometry.
10898    /// Shared by write_chunk and write_compressed_chunk.
10899    ///
10900    /// The index lookup happens *before* the bytes are placed, because the
10901    /// entry it finds is what tells [`place_chunk`](Self::place_chunk) whether
10902    /// this is a rewrite that can stay put.
10903    fn record_ea_chunk(
10904        &self,
10905        index: usize,
10906        chunk_idx: u64,
10907        final_bytes: &[u8],
10908        filter_mask: u32,
10909    ) -> IoResult<()> {
10910        let compressed_size = final_bytes.len() as u64;
10911        let ds = self.ds(index);
10912        // Hold one slot guard for the whole method: every dataset-state access
10913        // below goes through `m`, while `self.handle`/`self.allocator`/`self.ctx`
10914        // are disjoint fields safe to touch with the guard held.
10915        let mut m = ds.lock();
10916        let is_filtered = m.filter_pipeline.is_some();
10917        // For a filtered dataset the chunk's stored size is encoded in the
10918        // `chunk_size_len`-byte field of each filtered EA entry
10919        // (`FilteredChunkEntry::encode` writes `nbytes[..chunk_size_len]`,
10920        // which truncates silently). Reject a size that would not fit, the way
10921        // libhdf5's H5D_CHUNK_ENCODE_SIZE_CHECK does, instead of corrupting the
10922        // index. The compress path never exceeds this (chunk_size_len holds the
10923        // uncompressed chunk size); a direct/raw write with caller-supplied
10924        // bytes can.
10925        if is_filtered {
10926            let chunk_size_len = m.chunked.as_ref().unwrap().chunk_size_len as usize;
10927            if chunk_size_len < 8 && compressed_size >= (1u64 << (chunk_size_len * 8)) {
10928                return Err(crate::io::IoError::InvalidState(format!(
10929                    "filtered chunk size {compressed_size} does not fit in the \
10930                     {chunk_size_len}-byte extensible-array chunk-size field"
10931                )));
10932            }
10933        }
10934        let idx_blk_elmts = {
10935            let c = m.chunked.as_ref().unwrap();
10936            c.earray_params.idx_blk_elmts as u64
10937        };
10938
10939        if chunk_idx < idx_blk_elmts {
10940            let chunked = m.chunked.as_mut().unwrap();
10941            if is_filtered {
10942                if let Some(ref mut fiblk) = chunked.filt_iblk {
10943                    let old = fiblk.elements[chunk_idx as usize];
10944                    let chunk_addr =
10945                        self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
10946                    self.handle.write_at(chunk_addr, final_bytes)?;
10947                    fiblk.elements[chunk_idx as usize] = FilteredChunkEntry {
10948                        addr: chunk_addr,
10949                        nbytes: compressed_size,
10950                        filter_mask,
10951                    };
10952                }
10953            } else {
10954                // An unfiltered chunk's stored size is fixed by the chunk
10955                // shape, so a rewrite always fits where it already is.
10956                let old = chunked.ea_iblk.elements[chunk_idx as usize];
10957                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
10958                self.handle.write_at(chunk_addr, final_bytes)?;
10959                chunked.ea_iblk.elements[chunk_idx as usize] = chunk_addr;
10960            }
10961            chunked.chunks_written += 1;
10962            if chunk_idx + 1 > chunked.ea_header.max_idx_set {
10963                chunked.ea_header.max_idx_set = chunk_idx + 1;
10964            }
10965            if chunked.ea_header.num_elmts_realized < idx_blk_elmts {
10966                chunked.ea_header.num_elmts_realized = idx_blk_elmts;
10967            }
10968        } else {
10969            // chunk_idx >= idx_blk_elmts: place the chunk through the EA
10970            // data-block / super-block hierarchy (libhdf5-compatible geometry).
10971            let (geo, max_nelmts_bits, chunk_size_len, ea_header_addr) = {
10972                let c = m.chunked.as_ref().unwrap();
10973                let p = &c.earray_params;
10974                (
10975                    EaGeometry::new(
10976                        p.idx_blk_elmts,
10977                        p.data_blk_min_elmts,
10978                        p.sup_blk_min_data_ptrs,
10979                        p.max_nelmts_bits,
10980                        p.max_dblk_page_nelmts_bits,
10981                    )?,
10982                    p.max_nelmts_bits,
10983                    c.chunk_size_len,
10984                    c.ea_header_addr,
10985                )
10986            };
10987            let loc = match geo.locate(chunk_idx)? {
10988                EaLoc::Dblk(l) => l,
10989                EaLoc::Index { .. } => unreachable!("chunk_idx >= idx_blk_elmts"),
10990            };
10991            if loc.paged {
10992                return Err(crate::io::IoError::InvalidState(format!(
10993                    "chunk index {} needs a paged extensible-array data block, \
10994                     which is not yet supported",
10995                    chunk_idx
10996                )));
10997            }
10998            let class_id = if is_filtered {
10999                EA_CLS_FILT_CHUNK
11000            } else {
11001                EA_CLS_CHUNK
11002            };
11003            let dblk_nelmts = loc.dblk_nelmts as usize;
11004
11005            // Resolve the data block's current address and its parent slot,
11006            // creating the owning super block on demand.
11007            let parent: DblkParent;
11008            let mut dblk_addr: u64;
11009            match loc.path {
11010                EaDblkPath::Direct { idx: di } => {
11011                    let c = m.chunked.as_ref().unwrap();
11012                    dblk_addr = if is_filtered {
11013                        c.filt_iblk.as_ref().unwrap().dblk_addrs[di]
11014                    } else {
11015                        c.ea_iblk.dblk_addrs[di]
11016                    };
11017                    parent = DblkParent::IndexBlock(di);
11018                }
11019                EaDblkPath::ViaSblk {
11020                    sblk_off,
11021                    local_dblk,
11022                    ndblks_in_sblk,
11023                    sblk_block_offset,
11024                } => {
11025                    let mut sblk_addr = {
11026                        let c = m.chunked.as_ref().unwrap();
11027                        if is_filtered {
11028                            c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
11029                        } else {
11030                            c.ea_iblk.sblk_addrs[sblk_off]
11031                        }
11032                    };
11033                    if sblk_addr == UNDEF_ADDR {
11034                        let sb = ExtensibleArraySuperBlock::new(
11035                            class_id,
11036                            ea_header_addr,
11037                            sblk_block_offset,
11038                            ndblks_in_sblk,
11039                        );
11040                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11041                        sblk_addr = self
11042                            .allocator
11043                            .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11044                        self.handle.write_at(sblk_addr, &enc)?;
11045                        let c = m.chunked.as_mut().unwrap();
11046                        if is_filtered {
11047                            c.filt_iblk.as_mut().unwrap().sblk_addrs[sblk_off] = sblk_addr;
11048                        } else {
11049                            c.ea_iblk.sblk_addrs[sblk_off] = sblk_addr;
11050                        }
11051                        c.ea_header.num_sblks_created += 1;
11052                        c.ea_header.size_sblks_created += enc.len() as u64;
11053                    }
11054                    let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
11055                    // The writer never creates paged super blocks (it errors
11056                    // before the paging threshold), so page_init_total is 0.
11057                    let sb = ExtensibleArraySuperBlock::decode(
11058                        &sb_buf,
11059                        &self.ctx,
11060                        max_nelmts_bits,
11061                        ndblks_in_sblk,
11062                        0,
11063                    )?;
11064                    dblk_addr = sb.dblk_addrs[local_dblk];
11065                    parent = DblkParent::SuperBlock {
11066                        sblk_addr,
11067                        ndblks_in_sblk,
11068                        local_dblk,
11069                    };
11070                }
11071            }
11072
11073            // Create or update the data block holding this chunk's entry.
11074            let created = dblk_addr == UNDEF_ADDR;
11075            if is_filtered {
11076                let mut dblk = if created {
11077                    FilteredDataBlock::new(ea_header_addr, loc.dblk_block_offset, dblk_nelmts)
11078                } else {
11079                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11080                    FilteredDataBlock::decode(
11081                        &buf,
11082                        &self.ctx,
11083                        max_nelmts_bits,
11084                        dblk_nelmts,
11085                        chunk_size_len,
11086                    )?
11087                };
11088                // A freshly created data block holds only undefined addresses,
11089                // so this reads as "no previous chunk" without a special case.
11090                let old = dblk.elements[loc.offset_in_dblk as usize];
11091                let chunk_addr = self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11092                self.handle.write_at(chunk_addr, final_bytes)?;
11093                let entry = FilteredChunkEntry {
11094                    addr: chunk_addr,
11095                    nbytes: compressed_size,
11096                    filter_mask,
11097                };
11098                dblk.elements[loc.offset_in_dblk as usize] = entry;
11099                let enc = dblk.encode(&self.ctx, max_nelmts_bits, chunk_size_len);
11100                if created {
11101                    dblk_addr = self
11102                        .allocator
11103                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11104                }
11105                self.handle.write_at(dblk_addr, &enc)?;
11106                if created {
11107                    let c = m.chunked.as_mut().unwrap();
11108                    c.ea_header.num_dblks_created += 1;
11109                    c.ea_header.size_dblks_created += enc.len() as u64;
11110                }
11111            } else {
11112                let mut dblk = if created {
11113                    ExtensibleArrayDataBlock::new(
11114                        ea_header_addr,
11115                        loc.dblk_block_offset,
11116                        dblk_nelmts,
11117                    )
11118                } else {
11119                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11120                    ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, dblk_nelmts)?
11121                };
11122                // Unfiltered: the stored size is fixed by the chunk shape, so
11123                // a rewrite always fits its old block. A freshly created data
11124                // block holds undefined addresses and falls through to a new
11125                // allocation.
11126                let old = dblk.elements[loc.offset_in_dblk as usize];
11127                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11128                self.handle.write_at(chunk_addr, final_bytes)?;
11129                dblk.elements[loc.offset_in_dblk as usize] = chunk_addr;
11130                let enc = dblk.encode(&self.ctx, max_nelmts_bits);
11131                if created {
11132                    dblk_addr = self
11133                        .allocator
11134                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11135                }
11136                self.handle.write_at(dblk_addr, &enc)?;
11137                if created {
11138                    let c = m.chunked.as_mut().unwrap();
11139                    c.ea_header.num_dblks_created += 1;
11140                    c.ea_header.size_dblks_created += enc.len() as u64;
11141                }
11142            }
11143
11144            // Record a newly-created data block's address in its parent.
11145            if created {
11146                match parent {
11147                    DblkParent::IndexBlock(di) => {
11148                        let c = m.chunked.as_mut().unwrap();
11149                        if is_filtered {
11150                            c.filt_iblk.as_mut().unwrap().dblk_addrs[di] = dblk_addr;
11151                        } else {
11152                            c.ea_iblk.dblk_addrs[di] = dblk_addr;
11153                        }
11154                    }
11155                    DblkParent::SuperBlock {
11156                        sblk_addr,
11157                        ndblks_in_sblk,
11158                        local_dblk,
11159                    } => {
11160                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
11161                        let mut sb = ExtensibleArraySuperBlock::decode(
11162                            &buf,
11163                            &self.ctx,
11164                            max_nelmts_bits,
11165                            ndblks_in_sblk,
11166                            0,
11167                        )?;
11168                        sb.dblk_addrs[local_dblk] = dblk_addr;
11169                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11170                        self.handle.write_at(sblk_addr, &enc)?;
11171                    }
11172                }
11173            }
11174
11175            // Statistics.
11176            let c = m.chunked.as_mut().unwrap();
11177            c.chunks_written += 1;
11178            if chunk_idx + 1 > c.ea_header.max_idx_set {
11179                c.ea_header.max_idx_set = chunk_idx + 1;
11180            }
11181            if created {
11182                c.ea_header.num_elmts_realized += loc.dblk_nelmts;
11183            }
11184        }
11185        Ok(())
11186    }
11187
11188    /// Write a slice (hyperslab) of data to a dataset, contiguous or chunked.
11189    ///
11190    /// `starts` and `counts` define the N-dimensional selection.
11191    /// `data` must be exactly `product(counts) * element_size` bytes.
11192    ///
11193    /// The selection is validated once here and then handed to the layout's
11194    /// own writer, so a caller never has to know which storage the dataset
11195    /// uses.
11196    pub fn write_slice(
11197        &self,
11198        index: usize,
11199        starts: &[u64],
11200        counts: &[u64],
11201        data: &[u8],
11202    ) -> IoResult<()> {
11203        let ds = self.ds(index);
11204        let _op = ds.op.lock();
11205        self.write_slice_inner(index, starts, counts, data)
11206    }
11207
11208    /// [`Self::write_slice`] body; the caller holds the dataset's op lock or
11209    /// the writer exclusively.
11210    pub(crate) fn write_slice_inner(
11211        &self,
11212        index: usize,
11213        starts: &[u64],
11214        counts: &[u64],
11215        data: &[u8],
11216    ) -> IoResult<()> {
11217        let ds_ref = self.ds(index);
11218        let ds = ds_ref.lock();
11219        let is_chunked = ds.is_chunked();
11220
11221        let dims = &ds.dataspace.dims;
11222        let element_size = ds.datatype.element_size() as u64;
11223        let ndims = dims.len();
11224
11225        // Every hyperslab edge must stay inside the dataset; without this an
11226        // out-of-bounds selection writes raw bytes over neighbouring data.
11227        check_hyperslab(dims, starts, counts)?;
11228        if ndims == 0 {
11229            return Err(crate::io::IoError::InvalidState(
11230                "write_slice does not support scalar datasets; use write_dataset_raw".into(),
11231            ));
11232        }
11233
11234        let out_elems: u64 = counts.iter().product();
11235        if data.len() as u64 != out_elems * element_size {
11236            return Err(crate::io::IoError::InvalidState(format!(
11237                "data size mismatch: expected {} bytes, got {}",
11238                out_elems * element_size,
11239                data.len()
11240            )));
11241        }
11242
11243        // `dims` borrows the dataset slot; collect what the writers below need
11244        // so the guard can be dropped before they re-lock it.
11245        let dims = dims.clone();
11246        let target = ds.contiguous_target();
11247        drop(ds);
11248
11249        if is_chunked {
11250            // Rows the append buffer holds are not in the chunks yet; writing
11251            // them there anyway would be undone when the buffer flushes at
11252            // close. Hand them to the chunks first.
11253            self.flush_append_buffer_if_intersecting(index, starts[0], starts[0] + counts[0])?;
11254            return self.write_slice_chunked(index, starts, counts, data);
11255        }
11256        let Some(target) = target else {
11257            return Err(crate::io::IoError::InvalidState(
11258                "dataset has no data allocated".into(),
11259            ));
11260        };
11261
11262        // Write each maximal contiguous run in one write. Trailing
11263        // full-selected dimensions coalesce, mirroring the read path: a slice
11264        // with a full last axis becomes one write per outer index instead of
11265        // one write per last-axis row.
11266        for_each_contiguous_run(
11267            &dims,
11268            starts,
11269            counts,
11270            element_size,
11271            |dst_off, src_off, len| {
11272                self.write_contiguous_bytes(&target, dst_off, &data[src_off..src_off + len])
11273            },
11274        )?;
11275
11276        Ok(())
11277    }
11278
11279    /// Write a hyperslab into a chunked dataset, one chunk at a time.
11280    ///
11281    /// The selection is already validated by [`write_slice`](Self::write_slice).
11282    /// For each chunk the selection touches, the chunk's share of `data` is
11283    /// scattered into a whole-chunk buffer and the chunk is rewritten:
11284    ///
11285    /// - a chunk the selection covers completely is built from `data` alone —
11286    ///   nothing needs reading back (libhdf5 takes the same shortcut with the
11287    ///   `relax` flag of `H5D__chunk_lock`);
11288    /// - a chunk covered only in part starts from what is already stored, or
11289    ///   from a fill-value buffer when the chunk has never been written, so
11290    ///   neighbouring elements survive and untouched ones read as fill.
11291    ///
11292    /// An edge chunk that hangs past the dataset extent is always the partial
11293    /// case, so the region beyond the extent keeps its fill value.
11294    fn write_slice_chunked(
11295        &self,
11296        index: usize,
11297        starts: &[u64],
11298        counts: &[u64],
11299        data: &[u8],
11300    ) -> IoResult<()> {
11301        if counts.contains(&0) {
11302            return Ok(());
11303        }
11304        let geo = self.chunk_geometry(index)?;
11305        let ndims = geo.dims.len();
11306        if geo.chunk_dims.len() != ndims {
11307            return Err(crate::io::IoError::InvalidState(format!(
11308                "dataset chunk shape has {} dimensions but the dataspace has {}",
11309                geo.chunk_dims.len(),
11310                ndims
11311            )));
11312        }
11313        if geo.chunk_dims.contains(&0) {
11314            return Err(crate::io::IoError::InvalidState(
11315                "chunk shape has a zero-length dimension".into(),
11316            ));
11317        }
11318        let chunk_bytes = geo.chunk_bytes() as usize;
11319
11320        // Grid range the selection touches, inclusive on both ends.
11321        let first: Vec<u64> = (0..ndims).map(|d| starts[d] / geo.chunk_dims[d]).collect();
11322        let last: Vec<u64> = (0..ndims)
11323            .map(|d| (starts[d] + counts[d] - 1) / geo.chunk_dims[d])
11324            .collect();
11325
11326        let mut coords = first.clone();
11327        loop {
11328            // Intersect the selection with this chunk. `in_chunk` is the
11329            // region's origin inside the chunk, `in_data` its origin inside
11330            // the caller's counts-shaped buffer, `extent` its size.
11331            let mut in_chunk = vec![0u64; ndims];
11332            let mut in_data = vec![0u64; ndims];
11333            let mut extent = vec![0u64; ndims];
11334            let mut covers_whole_chunk = true;
11335            for d in 0..ndims {
11336                let chunk_origin = coords[d] * geo.chunk_dims[d];
11337                let lo = starts[d].max(chunk_origin);
11338                let hi = (starts[d] + counts[d]).min(chunk_origin + geo.chunk_dims[d]);
11339                in_chunk[d] = lo - chunk_origin;
11340                in_data[d] = lo - starts[d];
11341                extent[d] = hi - lo;
11342                if in_chunk[d] != 0 || extent[d] != geo.chunk_dims[d] {
11343                    covers_whole_chunk = false;
11344                }
11345            }
11346
11347            let mut buf = if covers_whole_chunk {
11348                // Every byte is overwritten below.
11349                vec![0u8; chunk_bytes]
11350            } else {
11351                match self.read_chunk_at_coords(index, &coords)? {
11352                    Some(existing) => {
11353                        if existing.len() != chunk_bytes {
11354                            return Err(crate::io::IoError::InvalidState(format!(
11355                                "stored chunk at {coords:?} is {} bytes but the chunk shape \
11356                                 needs {chunk_bytes}",
11357                                existing.len()
11358                            )));
11359                        }
11360                        existing
11361                    }
11362                    None => self.new_write_chunk_buffer(index, chunk_bytes),
11363                }
11364            };
11365
11366            for_each_dual_run(
11367                &geo.chunk_dims,
11368                &in_chunk,
11369                counts,
11370                &in_data,
11371                &extent,
11372                geo.element_size,
11373                |dst_off, src_off, len| {
11374                    let dst = dst_off as usize;
11375                    let src = src_off as usize;
11376                    buf[dst..dst + len].copy_from_slice(&data[src..src + len]);
11377                    Ok(())
11378                },
11379            )?;
11380            self.write_chunk_at_coords(index, &coords, &buf)?;
11381
11382            // Odometer over the touched grid range.
11383            let mut d = ndims;
11384            loop {
11385                if d == 0 {
11386                    return Ok(());
11387                }
11388                d -= 1;
11389                if coords[d] < last[d] {
11390                    coords[d] += 1;
11391                    break;
11392                }
11393                coords[d] = first[d];
11394            }
11395        }
11396    }
11397
11398    /// Add an attribute to the root group (file-level attribute), replacing
11399    /// a same-name attribute. See [`set_attribute`](Self::set_attribute).
11400    pub fn add_root_attribute(&self, attr: AttributeMessage) -> IoResult<()> {
11401        self.set_attribute(AttrTarget::Root, attr)
11402    }
11403
11404    /// Insert `attr` into the attribute list `target` names, replacing a
11405    /// same-name attribute.
11406    ///
11407    /// The single owner of attribute-list mutation: an `AttributeMessage`
11408    /// that leaves a list here has its vlen global-heap objects released, so
11409    /// no replacement — vlen over vlen, numeric over vlen — can strand heap
11410    /// space (the attribute counterpart of issue #10's dataset fix).
11411    ///
11412    /// Under SWMR every attribute mutation is refused, matching libhdf5's
11413    /// rule for SWMR writes. Object headers are frozen once streaming
11414    /// starts — a change was committed at close only when the header
11415    /// happened to be rebuilt (group attrs always, dataset attrs only if
11416    /// the dataset also got chunk writes) and silently dropped otherwise —
11417    /// and a replacement's superseded vlen value could never be reclaimed,
11418    /// since a streaming reader may hold its heap references.
11419    pub fn set_attribute(&self, target: AttrTarget<'_>, attr: AttributeMessage) -> IoResult<()> {
11420        self.insert_attribute(target, attr, Created)
11421    }
11422
11423    /// The body of [`set_attribute`](Self::set_attribute), told whether the
11424    /// attribute it is inserting is genuinely new — see [`AttrOrigin`].
11425    fn insert_attribute(
11426        &self,
11427        target: AttrTarget<'_>,
11428        attr: AttributeMessage,
11429        origin: AttrOrigin,
11430    ) -> IoResult<()> {
11431        if self.swmr_active {
11432            return Err(swmr_attr_error(&attr.name));
11433        }
11434        // Whatever this name meant before, it means the incoming message now.
11435        self.forget_attribute_reference(self.attr_scope(target)?, &attr.name);
11436        // No size gate: an attribute whose message is too large for the
11437        // 16-bit size field an object header message has spills the object's
11438        // whole attribute set to dense storage at finalize, exactly as
11439        // `H5O__attr_create` does. See `attributes_need_dense`.
11440        let mut entry = AttributeEntry::from(attr);
11441        let old = self.with_attr_list(target, |attrs| {
11442            if let Some(pos) = attrs.iter().position(|a| a.name() == entry.name()) {
11443                // `H5O__attr_write` replaces an existing attribute's value and
11444                // leaves its `crt_idx` alone: the attribute was not created
11445                // again, so its creation index does not move.
11446                entry.set_creation_index(attrs[pos].creation_index());
11447                Some(std::mem::replace(&mut attrs[pos], entry))
11448            } else {
11449                // `H5O__attr_create` stamps the set's running maximum onto the
11450                // new attribute and post-increments it — but only a create
11451                // reaches for it.
11452                entry.set_creation_index(match origin {
11453                    Created => Some(next_creation_index(attrs)),
11454                    Rewritten(kept) => kept,
11455                });
11456                attrs.push(entry);
11457                None
11458            }
11459        })?;
11460        match old {
11461            Some(old) => self.release_attr_vlen(&old),
11462            None => Ok(()),
11463        }
11464    }
11465
11466    /// Set a variable-length string attribute on `target`, replacing any
11467    /// same-name attribute.
11468    ///
11469    /// Owns the whole replacement sequence: the superseded attribute is
11470    /// removed and its heap objects released *before* the new value's
11471    /// collection is allocated — the free-before-alloc order (issue #10)
11472    /// that lets a reopen-replace loop land in the block it just freed
11473    /// instead of growing the file every session. The cost, as on the
11474    /// dataset path: a failure between the eviction and the insert below
11475    /// loses the attribute rather than leaking its heap space.
11476    pub fn set_vlen_string_attribute(
11477        &self,
11478        target: AttrTarget<'_>,
11479        name: &str,
11480        value: &str,
11481    ) -> IoResult<()> {
11482        let origin = self.evict_attr(target, name)?;
11483        let attr = self.vlen_string_attribute(name, value)?;
11484        self.insert_attribute(target, attr, origin)
11485    }
11486
11487    /// The array counterpart of
11488    /// [`set_vlen_string_attribute`](Self::set_vlen_string_attribute).
11489    pub fn set_vlen_string_array_attribute(
11490        &self,
11491        target: AttrTarget<'_>,
11492        name: &str,
11493        values: &[&str],
11494        dims: &[u64],
11495    ) -> IoResult<()> {
11496        let origin = self.evict_attr(target, name)?;
11497        let attr = self.vlen_string_array_attribute(name, values, dims)?;
11498        self.insert_attribute(target, attr, origin)
11499    }
11500
11501    /// Set an attribute on `target` whose value is the object references
11502    /// naming `paths` — h5py's `obj.attrs['ref'] = f['/target'].ref`.
11503    ///
11504    /// `dims` is the attribute's dataspace: empty for the scalar shape a
11505    /// single reference takes, `&[n]` for an array of them. Each path names a
11506    /// dataset or a group (`/` is the root group) and must already exist. What
11507    /// reaches the file is each target's object header address, which finalize
11508    /// assigns — so the paths are what is stored, and the attribute's message
11509    /// is built from them every time an object header is
11510    /// ([`object_attributes`](Self::object_attributes)). The message carries a
11511    /// zero image of the final width until then.
11512    pub fn set_object_reference_attribute(
11513        &self,
11514        target: AttrTarget<'_>,
11515        name: &str,
11516        paths: &[&str],
11517        dims: &[u64],
11518    ) -> IoResult<()> {
11519        let scope = self.attr_scope(target)?;
11520        // An empty `dims` is the scalar shape, whose one element the empty
11521        // product already reports.
11522        let elements: u64 = dims.iter().product();
11523        if elements != paths.len() as u64 {
11524            return Err(crate::io::IoError::InvalidState(format!(
11525                "attribute '{name}' shape {dims:?} needs {elements} references, got {}",
11526                paths.len()
11527            )));
11528        }
11529        // Resolve now as well as at finalize, so a path that names nothing is
11530        // reported at the call that got it wrong.
11531        for path in paths {
11532            self.object_reference_target(path)?;
11533        }
11534        let datatype = DatatypeMessage::object_reference(&self.ctx);
11535        let image = vec![0u8; paths.len() * datatype.element_size() as usize];
11536        let attr = if dims.is_empty() {
11537            AttributeMessage::scalar_numeric(name, datatype, image)
11538        } else {
11539            AttributeMessage::array_numeric(name, datatype, dims, image)
11540        };
11541        // Through the same owner as every other attribute, which is also what
11542        // drops any value this name carried before.
11543        self.set_attribute(target, attr)?;
11544        self.attribute_references
11545            .lock()
11546            .push(AttributeReferenceValue {
11547                scope,
11548                name: name.to_string(),
11549                targets: paths.iter().map(|p| (*p).to_string()).collect(),
11550            });
11551        Ok(())
11552    }
11553
11554    /// Take the attribute `name` off `target`'s list, releasing its heap
11555    /// objects. No-op when absent. Refused under SWMR — see
11556    /// [`set_attribute`](Self::set_attribute).
11557    ///
11558    /// What it answers is what the insert that follows it must be told: an
11559    /// attribute that was there is being rewritten and keeps its creation
11560    /// index, and one that was not is created.
11561    fn evict_attr(&self, target: AttrTarget<'_>, name: &str) -> IoResult<AttrOrigin> {
11562        if self.swmr_active {
11563            return Err(swmr_attr_error(name));
11564        }
11565        self.forget_attribute_reference(self.attr_scope(target)?, name);
11566        let old = self.with_attr_list(target, |attrs| {
11567            attrs
11568                .iter()
11569                .position(|a| a.name() == name)
11570                .map(|pos| attrs.remove(pos))
11571        })?;
11572        match old {
11573            Some(old) => {
11574                let origin = Rewritten(old.creation_index());
11575                self.release_attr_vlen(&old)?;
11576                Ok(origin)
11577            }
11578            None => Ok(Created),
11579        }
11580    }
11581
11582    /// Release the global-heap objects a superseded attribute owned.
11583    /// Recognizes top-level vlen datatypes only: a *compound* attribute
11584    /// with vlen members — which this crate cannot write, only a foreign
11585    /// file can carry — keeps its members' heap objects when replaced or
11586    /// deleted, the storage cost the foreign writer accepted. Every other
11587    /// class stores its value inline in the message. Per-object removal
11588    /// keeps collections shared with other refs (libhdf5-written files)
11589    /// intact.
11590    fn release_attr_vlen(&self, old: &AttributeEntry) -> IoResult<()> {
11591        use crate::format::messages::datatype::DatatypeMessage;
11592        // An attribute whose message this crate could not decode keeps
11593        // whatever heap space it references: releasing objects named by bytes
11594        // we cannot interpret would free storage that is still live.
11595        let Some(old) = old.readable() else {
11596            return Ok(());
11597        };
11598        if matches!(
11599            old.datatype,
11600            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
11601        ) {
11602            self.release_vlen_references(&old.data)?;
11603        }
11604        Ok(())
11605    }
11606
11607    /// Run `f` on the attribute list `target` names — the accessor every
11608    /// attribute mutation shares.
11609    fn with_attr_list<R>(
11610        &self,
11611        target: AttrTarget<'_>,
11612        f: impl FnOnce(&mut Vec<AttributeEntry>) -> R,
11613    ) -> IoResult<R> {
11614        match target {
11615            AttrTarget::Root => Ok(f(&mut self.root_attributes.lock())),
11616            AttrTarget::Group(path) => {
11617                let path = self.canonical_group_path(path);
11618                for grp in self.group_refs() {
11619                    let mut g = grp.lock();
11620                    if g.name == path && !g.deleted {
11621                        return Ok(f(&mut g.attributes));
11622                    }
11623                }
11624                Err(crate::io::IoError::NotFound(format!(
11625                    "group '{path}' not found"
11626                )))
11627            }
11628            AttrTarget::Dataset(index) => {
11629                let count = self.dataset_count();
11630                if index >= count {
11631                    return Err(crate::io::IoError::InvalidState(format!(
11632                        "dataset index {index} out of range (have {count})"
11633                    )));
11634                }
11635                let ds = self.ds(index);
11636                let mut m = ds.lock();
11637                // Every caller of this mutates the list, and a reopened
11638                // dataset's header is rewritten only when it is marked stale.
11639                m.header_dirty = true;
11640                Ok(f(&mut m.attributes))
11641            }
11642        }
11643    }
11644
11645    /// Store each of `items` as a global heap object and return its
11646    /// placement `(collection address, object index)`, in input order —
11647    /// the writer side of libhdf5's `H5HG_insert`.
11648    ///
11649    /// Placement follows libhdf5: a collection from the CWFS list takes an
11650    /// item when its free space holds the object *and* a residual
11651    /// free-space marker header (`encode_at_size` always emits the
11652    /// marker); what no listed collection can take goes into a fresh
11653    /// collection, spilling into another at the 65535-object index cap.
11654    /// One batch may therefore span several collections — invisible to
11655    /// readers, which resolve each reference's own collection address. An
11656    /// empty batch allocates nothing: an empty collection still encodes
11657    /// to the 4096-byte `H5HG_MINALLOC` minimum, a block nothing would
11658    /// reference. libhdf5 additionally tries to extend a nearly-full
11659    /// collection's block in place (`H5MF_try_extend`); this writer does
11660    /// not — an oversized item always starts a fresh collection.
11661    ///
11662    /// The `cwfs` lock is held across every read-modify-rewrite of a
11663    /// listed collection block: it serializes concurrent inserts (two
11664    /// datasets' writers can pack the same block) and inserts against
11665    /// [`release_vlen_references`](Self::release_vlen_references), which
11666    /// rewrites the same blocks when objects are freed.
11667    ///
11668    /// Under SWMR the CWFS list is neither consulted nor updated and every
11669    /// batch gets fresh collections: packing rewrites a block a streaming
11670    /// reader may be mid-walk on — the same reason `place_chunk` keeps a
11671    /// relocated chunk's old block.
11672    fn insert_vlen_objects(&self, items: &[&[u8]]) -> IoResult<Vec<(u64, u16)>> {
11673        use crate::format::global_heap::{GlobalHeapCollection, GlobalHeapObject};
11674
11675        if items.is_empty() {
11676            return Ok(Vec::new());
11677        }
11678        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
11679        let mut placements = Vec::with_capacity(items.len());
11680        let mut i = 0;
11681
11682        // Pack into listed collections while one can take the next item.
11683        if !self.swmr_active {
11684            let mut cwfs = self.cwfs.lock();
11685            while i < items.len() {
11686                let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
11687                let Some(pos) = cwfs.iter().position(|e| e.free >= need + objhdr) else {
11688                    // Second pass of libhdf5's H5F_cwfs_find_free_heap: no
11689                    // listed collection has room, so try to grow one in
11690                    // place before falling back to a fresh collection.
11691                    if self.extend_listed_collection(&mut cwfs, need + objhdr)? {
11692                        continue;
11693                    }
11694                    break;
11695                };
11696                let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
11697                let image = self.handle.read_at(addr, size)?;
11698                let (mut gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
11699                // The disk is the truth for free space; the entry is a hint.
11700                let Some(mut free) = gcol.free_space_at(&self.ctx, size) else {
11701                    cwfs.remove(pos);
11702                    continue;
11703                };
11704                let mut next_idx = gcol.max_index();
11705                let mut took = false;
11706                while i < items.len() && next_idx < u16::MAX {
11707                    let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
11708                    if free < need + objhdr {
11709                        break;
11710                    }
11711                    next_idx += 1;
11712                    gcol.objects.push(GlobalHeapObject {
11713                        index: next_idx,
11714                        ref_count: 0,
11715                        data: items[i].to_vec(),
11716                    });
11717                    placements.push((addr, next_idx));
11718                    free -= need;
11719                    took = true;
11720                    i += 1;
11721                }
11722                if took {
11723                    let rewritten = gcol.encode_at_size(&self.ctx, size)?;
11724                    self.handle.write_at(addr, &rewritten)?;
11725                    // Correct the entry to the measured free space and move
11726                    // it to the front — libhdf5 keeps `cwfs` in
11727                    // most-recently-used order.
11728                    let mut e = cwfs.remove(pos);
11729                    e.free = free;
11730                    cwfs.insert(0, e);
11731                } else if next_idx == u16::MAX {
11732                    // At the index cap nothing can be inserted no matter the
11733                    // free space; drop the entry or the scan re-picks it
11734                    // forever. (A removal can lower the top index again, and
11735                    // the release side re-lists the collection then.)
11736                    cwfs.remove(pos);
11737                } else {
11738                    // The hint overstated the block's free space — shrink it
11739                    // to the measured value so the scan moves on.
11740                    cwfs[pos].free = free;
11741                }
11742            }
11743        }
11744
11745        // What remains goes into fresh collections.
11746        while i < items.len() {
11747            let mut gcol = GlobalHeapCollection::new();
11748            // Objects are pushed with a running index: `add_object` rescans
11749            // for the max index per call, O(n²) across a spill-sized batch.
11750            let mut next_idx: u16 = 0;
11751            while i < items.len() && next_idx < u16::MAX {
11752                next_idx += 1;
11753                gcol.objects.push(GlobalHeapObject {
11754                    index: next_idx,
11755                    ref_count: 0,
11756                    data: items[i].to_vec(),
11757                });
11758                i += 1;
11759            }
11760            let encoded = gcol.encode(&self.ctx);
11761            let addr = self
11762                .allocator
11763                .allocate(encoded.len() as u64, FreeSpaceClass::RawData);
11764            self.handle.write_at(addr, &encoded)?;
11765            for idx in 1..=next_idx {
11766                placements.push((addr, idx));
11767            }
11768            // List the block's leftover free space for later inserts — the
11769            // minimum-size padding of a small batch is most of 4096 bytes.
11770            // Below two object headers not even an empty object fits.
11771            if !self.swmr_active {
11772                if let Some(free) = gcol.free_space_at(&self.ctx, encoded.len()) {
11773                    if free >= 2 * objhdr {
11774                        cwfs_note(&mut self.cwfs.lock(), addr, encoded.len(), free);
11775                    }
11776                }
11777            }
11778        }
11779        Ok(placements)
11780    }
11781
11782    /// Try to extend one listed collection in place so it can take an
11783    /// object needing `want` bytes of free space — the second pass of
11784    /// libhdf5's `H5F_cwfs_find_free_heap`: grow the file allocation
11785    /// ([`FileAllocator::try_extend`], mirroring `H5MF_try_extend`) and then
11786    /// the collection itself (`H5HG_extend`: a larger declared size and a
11787    /// free-space marker covering the new tail — here by re-encoding at the
11788    /// grown size, which writes exactly those two things).
11789    ///
11790    /// Extension size is `max(collection_size, shortfall)` — at least a
11791    /// doubling — capped so the result stays within [`GCOL_MAX_SIZE`], both
11792    /// as upstream computes them. On success the grown entry moves to the
11793    /// front of the list and the caller's scan re-picks it; the free-space
11794    /// measurement is taken from the block on disk, not the list's hint, so
11795    /// the rewrite and the entry agree.
11796    ///
11797    /// Caller holds the `cwfs` lock (it passes the guarded list), which is
11798    /// what serializes this read-modify-rewrite against concurrent inserts
11799    /// and releases.
11800    fn extend_listed_collection(&self, cwfs: &mut Vec<CwfsEntry>, want: usize) -> IoResult<bool> {
11801        use crate::format::global_heap::{GlobalHeapCollection, GCOL_MAX_SIZE};
11802
11803        let mut pos = 0;
11804        while pos < cwfs.len() {
11805            let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
11806            let image = self.handle.read_at(addr, size)?;
11807            let (gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
11808            // The disk is the truth for free space; the entry is a hint.
11809            let Some(free) = gcol.free_space_at(&self.ctx, size) else {
11810                cwfs.remove(pos);
11811                continue;
11812            };
11813            // A hint can understate the block (upstream's FREE_SIZE is its
11814            // in-memory truth and cannot): if the block already has room,
11815            // correct the hint instead of doubling the collection.
11816            if free >= want {
11817                cwfs[pos].free = free;
11818                return Ok(true);
11819            }
11820            let new_need = size.max(want.saturating_sub(free));
11821            if size + new_need > GCOL_MAX_SIZE
11822                || !self.allocator.try_extend(
11823                    addr,
11824                    size as u64,
11825                    new_need as u64,
11826                    FreeSpaceClass::RawData,
11827                )
11828            {
11829                pos += 1;
11830                continue;
11831            }
11832            let new_size = size + new_need;
11833            let rewritten = gcol.encode_at_size(&self.ctx, new_size)?;
11834            self.handle.write_at(addr, &rewritten)?;
11835            let mut e = cwfs.remove(pos);
11836            e.size = new_size;
11837            e.free = free + new_need;
11838            cwfs.insert(0, e);
11839            return Ok(true);
11840        }
11841        Ok(false)
11842    }
11843
11844    /// Create a variable-length string dataset and write string data.
11845    ///
11846    /// Stores strings in the global heap. The dataset raw data consists of
11847    /// vlen references (collection_addr + object_index pairs).
11848    ///
11849    /// `charset` is the datatype's declared character set (0 = ASCII,
11850    /// 1 = UTF-8); the strings are checked against it before anything is
11851    /// written, so the type never misdescribes the bytes under it.
11852    pub fn create_vlen_string_dataset(
11853        &self,
11854        name: &str,
11855        strings: &[&str],
11856        charset: u8,
11857    ) -> IoResult<usize> {
11858        use crate::format::global_heap::encode_vlen_reference;
11859        use crate::format::messages::datatype::DatatypeMessage;
11860
11861        ensure_vlen_charset(charset, strings)?;
11862
11863        let create = self.begin_create(name)?;
11864        let name = create.name.as_str();
11865        let num_strings = strings.len() as u64;
11866
11867        // Store the strings as heap objects; a batch that fits an earlier
11868        // collection's free space shares its block.
11869        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
11870        let placements = self.insert_vlen_objects(&items)?;
11871
11872        // Build raw data: vlen references
11873        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
11874        let data_size = (num_strings as usize) * ref_size;
11875        let mut raw_data = Vec::with_capacity(data_size);
11876        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
11877            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
11878            raw_data.extend_from_slice(&encode_vlen_reference(
11879                seq_len,
11880                gcol_addr,
11881                obj_idx as u32,
11882                &self.ctx,
11883            ));
11884        }
11885
11886        // Allocate and write raw data
11887        let data_addr = self
11888            .allocator
11889            .allocate(data_size as u64, FreeSpaceClass::RawData);
11890        self.handle.write_at(data_addr, &raw_data)?;
11891
11892        // Create the dataset with vlen string datatype
11893        let datatype = DatatypeMessage::VarLenString {
11894            padding: 0,
11895            charset,
11896        };
11897        let dataspace =
11898            crate::format::messages::dataspace::DataspaceMessage::simple(&[num_strings]);
11899
11900        let idx = self.push_dataset(
11901            &create,
11902            DatasetInfo {
11903                name: name.to_string(),
11904                datatype,
11905                committed_type: None,
11906                external: None,
11907                virtual_storage: None,
11908                dataspace,
11909                read_format: None,
11910                obj_header_addr: 0,
11911                data_addr,
11912                data_size: data_size as u64,
11913                compact: None,
11914                attributes: Vec::new(),
11915                obj_header_written_addr: None,
11916                obj_header_blocks: Vec::new(),
11917                filter_pipeline: None,
11918                deleted: false,
11919                extent_dirty: false,
11920                header_dirty: false,
11921                nlink_written: 1,
11922                creation_seq: self.take_creation_seq(),
11923                track_attr_order: self.track_order.attrs,
11924                fill_value: None,
11925                fill_time: FILL_TIME_IFSET,
11926                layout_version: 4,
11927                times: self.created_object_times(),
11928                chunked: None,
11929                fixed_array: None,
11930                implicit: None,
11931                single_chunk: None,
11932                btree_v1: None,
11933                btree_v2: None,
11934                append: None,
11935            },
11936        );
11937
11938        Ok(idx)
11939    }
11940
11941    /// Create a 1-D variable-length byte-array dataset.
11942    ///
11943    /// The `u8` case of [`create_vlen_sequence_dataset`], where an item's
11944    /// byte image and its element count are the same number.
11945    ///
11946    /// [`create_vlen_sequence_dataset`]: Self::create_vlen_sequence_dataset
11947    ///
11948    /// Superseded in production by [`write_vlen_numeric`](crate::H5File::write_vlen_numeric)
11949    /// (`H5Group::write_vlen_bytes` routes through it, not through here);
11950    /// kept as a direct entry point for this crate's own white-box tests.
11951    #[cfg(test)]
11952    pub fn create_vlen_bytes_dataset(&self, name: &str, items: &[&[u8]]) -> IoResult<usize> {
11953        use crate::format::messages::datatype::DatatypeMessage;
11954
11955        self.create_vlen_sequence_dataset(name, DatatypeMessage::u8_type(), items)
11956    }
11957
11958    /// Create a 1-D variable-length sequence dataset over `base`.
11959    ///
11960    /// Each item is the encoded image of one sequence — `n * base.element_size()`
11961    /// bytes in the base type's own byte order — and is stored as a global-heap
11962    /// object; the dataset holds one vlen reference per item, the same on-disk
11963    /// shape a vlen string dataset has. The `H5T_VLEN` length field counts base
11964    /// elements rather than bytes, so an image whose length is not a whole
11965    /// number of elements is refused here rather than stored under a length
11966    /// that misreads it.
11967    pub fn create_vlen_sequence_dataset(
11968        &self,
11969        name: &str,
11970        base: DatatypeMessage,
11971        items: &[&[u8]],
11972    ) -> IoResult<usize> {
11973        use crate::format::global_heap::encode_vlen_reference;
11974        use crate::format::messages::datatype::DatatypeMessage;
11975
11976        let elem_size = base.element_size() as usize;
11977        if elem_size == 0 {
11978            return Err(crate::io::IoError::InvalidState(format!(
11979                "vlen base datatype {base} has no element size"
11980            )));
11981        }
11982        for (i, item) in items.iter().enumerate() {
11983            if !item.len().is_multiple_of(elem_size) {
11984                return Err(crate::io::IoError::InvalidState(format!(
11985                    "sequence {i} is {} bytes, not a whole number of {elem_size}-byte elements",
11986                    item.len()
11987                )));
11988            }
11989        }
11990
11991        let create = self.begin_create(name)?;
11992        let name = create.name.as_str();
11993        let num_items = items.len() as u64;
11994
11995        // Store the sequence images as heap objects, sharing collection
11996        // blocks as `create_vlen_string_dataset` does.
11997        let placements = self.insert_vlen_objects(items)?;
11998
11999        // Build raw data: one vlen reference per item.
12000        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12001        let data_size = (num_items as usize) * ref_size;
12002        let mut raw_data = Vec::with_capacity(data_size);
12003        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12004            let seq_len = crate::format::global_heap::vlen_seq_len(items[i].len() / elem_size)?;
12005            raw_data.extend_from_slice(&encode_vlen_reference(
12006                seq_len,
12007                gcol_addr,
12008                obj_idx as u32,
12009                &self.ctx,
12010            ));
12011        }
12012
12013        // Allocate and write raw data.
12014        let data_addr = self
12015            .allocator
12016            .allocate(data_size as u64, FreeSpaceClass::RawData);
12017        self.handle.write_at(data_addr, &raw_data)?;
12018
12019        let datatype = DatatypeMessage::VarLenSequence {
12020            base: Box::new(base),
12021        };
12022        let dataspace = crate::format::messages::dataspace::DataspaceMessage::simple(&[num_items]);
12023
12024        let idx = self.push_dataset(
12025            &create,
12026            DatasetInfo {
12027                name: name.to_string(),
12028                datatype,
12029                committed_type: None,
12030                external: None,
12031                virtual_storage: None,
12032                dataspace,
12033                read_format: None,
12034                obj_header_addr: 0,
12035                data_addr,
12036                data_size: data_size as u64,
12037                compact: None,
12038                attributes: Vec::new(),
12039                obj_header_written_addr: None,
12040                obj_header_blocks: Vec::new(),
12041                filter_pipeline: None,
12042                deleted: false,
12043                extent_dirty: false,
12044                header_dirty: false,
12045                nlink_written: 1,
12046                creation_seq: self.take_creation_seq(),
12047                track_attr_order: self.track_order.attrs,
12048                fill_value: None,
12049                fill_time: FILL_TIME_IFSET,
12050                layout_version: 4,
12051                times: self.created_object_times(),
12052                chunked: None,
12053                fixed_array: None,
12054                implicit: None,
12055                single_chunk: None,
12056                btree_v1: None,
12057                btree_v2: None,
12058                append: None,
12059            },
12060        );
12061
12062        Ok(idx)
12063    }
12064
12065    /// Create a chunked, compressed variable-length string dataset.
12066    ///
12067    /// Strings are stored in the global heap (same as `create_vlen_string_dataset`),
12068    /// but the vlen references are stored in chunked layout with the given filter
12069    /// pipeline (e.g., deflate, zstd). `chunk_size` is the number of strings per chunk.
12070    pub fn create_vlen_string_dataset_compressed(
12071        &self,
12072        name: &str,
12073        strings: &[&str],
12074        chunk_size: usize,
12075        pipeline: FilterPipeline,
12076    ) -> IoResult<usize> {
12077        use crate::format::global_heap::encode_vlen_reference;
12078        use crate::format::messages::datatype::DatatypeMessage;
12079
12080        let create = self.begin_create(name)?;
12081        let name = create.name.as_str();
12082        let num_strings = strings.len() as u64;
12083        validate_chunk_geometry(&[num_strings], &[num_strings], &[chunk_size as u64])?;
12084
12085        // Store the strings as heap objects; the geometry validation above
12086        // must precede this so a refused call allocates nothing.
12087        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12088        let placements = self.insert_vlen_objects(&items)?;
12089
12090        // Build raw data: vlen references
12091        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12092        let data_size = (num_strings as usize) * ref_size;
12093        let mut raw_data = Vec::with_capacity(data_size);
12094        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12095            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12096            raw_data.extend_from_slice(&encode_vlen_reference(
12097                seq_len,
12098                gcol_addr,
12099                obj_idx as u32,
12100                &self.ctx,
12101            ));
12102        }
12103
12104        // Set up chunked compressed layout
12105        let datatype = DatatypeMessage::vlen_string_utf8();
12106        let element_size = datatype.element_size_ctx(&self.ctx) as u64;
12107        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12108        let dims: Vec<u64> = vec![num_strings];
12109        let max_dims: Vec<u64> = vec![num_strings];
12110        let chunk_bytes = chunk_size as u64 * element_size;
12111        let layout_version = self.chunk_layout_version(true, chunk_bytes);
12112        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
12113
12114        let earray_params = EarrayParams::default_params();
12115        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
12116        let nsblk_addrs = compute_nsblk_addrs(
12117            earray_params.idx_blk_elmts,
12118            earray_params.data_blk_min_elmts,
12119            earray_params.sup_blk_min_data_ptrs,
12120            earray_params.max_nelmts_bits,
12121        )?;
12122
12123        // Create filtered EA header
12124        let mut ea_header =
12125            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
12126        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
12127        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
12128        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
12129        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
12130        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
12131
12132        let hdr_encoded = ea_header.encode(&self.ctx);
12133        let ea_header_addr = self
12134            .allocator
12135            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
12136
12137        // Create filtered index block
12138        let filt_iblk = FilteredIndexBlock::new(
12139            ea_header_addr,
12140            earray_params.idx_blk_elmts,
12141            ndblk_addrs,
12142            nsblk_addrs,
12143        );
12144        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
12145        let ea_iblk_addr = self
12146            .allocator
12147            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
12148
12149        ea_header.idx_blk_addr = ea_iblk_addr;
12150
12151        let hdr_encoded = ea_header.encode(&self.ctx);
12152        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
12153        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
12154
12155        let dataspace = DataspaceMessage {
12156            // Chunked storage always requires at least one dimension, so
12157            // this is never Scalar or Null.
12158            class: DataspaceClass::Simple,
12159            dims: dims.to_vec(),
12160            max_dims: Some(max_dims.to_vec()),
12161        };
12162
12163        let ea_iblk = ExtensibleArrayIndexBlock::new(
12164            ea_header_addr,
12165            earray_params.idx_blk_elmts,
12166            ndblk_addrs,
12167            nsblk_addrs,
12168        );
12169
12170        let idx = self.push_dataset(
12171            &create,
12172            DatasetInfo {
12173                name: name.to_string(),
12174                datatype,
12175                committed_type: None,
12176                external: None,
12177                virtual_storage: None,
12178                dataspace,
12179                read_format: None,
12180                obj_header_addr: 0,
12181                data_addr: UNDEF_ADDR,
12182                data_size: 0,
12183                compact: None,
12184                attributes: Vec::new(),
12185                obj_header_written_addr: None,
12186                obj_header_blocks: Vec::new(),
12187                filter_pipeline: Some(pipeline),
12188                deleted: false,
12189                extent_dirty: false,
12190                header_dirty: false,
12191                nlink_written: 1,
12192                creation_seq: self.take_creation_seq(),
12193                track_attr_order: self.track_order.attrs,
12194                fill_value: None,
12195                fill_time: FILL_TIME_IFSET,
12196                layout_version,
12197                times: self.created_object_times(),
12198                fixed_array: None,
12199                implicit: None,
12200                single_chunk: None,
12201                btree_v1: None,
12202                btree_v2: None,
12203                chunked: Some(ChunkedDatasetInfo {
12204                    chunk_dims: chunk_dims.clone(),
12205                    earray_params,
12206                    ea_header_addr,
12207                    ea_iblk_addr,
12208                    ea_header,
12209                    ea_iblk,
12210                    chunks_written: 0,
12211                    filt_iblk: Some(filt_iblk),
12212                    chunk_size_len,
12213                }),
12214                append: None,
12215            },
12216        );
12217
12218        // Write chunks of vlen references with compression
12219        let chunk_byte_size = chunk_bytes as usize;
12220        let num_chunks = raw_data.len().div_ceil(chunk_byte_size);
12221        for chunk_i in 0..num_chunks {
12222            let start = chunk_i * chunk_byte_size;
12223            let end = (start + chunk_byte_size).min(raw_data.len());
12224            let chunk_data = if end - start < chunk_byte_size {
12225                // Pad last chunk to full size (vlen datasets carry no user
12226                // fill value, so this resolves to zero = null vlen reference).
12227                let mut padded = self.new_chunk_buffer(idx, chunk_byte_size);
12228                padded[..end - start].copy_from_slice(&raw_data[start..end]);
12229                padded
12230            } else {
12231                raw_data[start..end].to_vec()
12232            };
12233            self.write_chunk(idx, chunk_i as u64, &chunk_data)?;
12234        }
12235
12236        Ok(idx)
12237    }
12238
12239    /// Create an empty chunked vlen string dataset ready for incremental appends.
12240    ///
12241    /// The dataset starts with `dims = [0]` and `max_dims = [unlimited]`.
12242    /// Use `append_vlen_strings` to add data.
12243    pub fn create_appendable_vlen_string_dataset(
12244        &self,
12245        name: &str,
12246        chunk_size: usize,
12247        pipeline: Option<FilterPipeline>,
12248    ) -> IoResult<usize> {
12249        let datatype = DatatypeMessage::vlen_string_utf8();
12250        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12251        let dims: Vec<u64> = vec![0];
12252        let max_dims: Vec<u64> = vec![u64::MAX];
12253
12254        if let Some(ref pl) = pipeline {
12255            self.create_chunked_dataset_with_pipeline(
12256                name,
12257                datatype,
12258                &dims,
12259                &max_dims,
12260                &chunk_dims,
12261                pl.clone(),
12262            )
12263        } else {
12264            self.create_chunked_dataset(name, datatype, &dims, &max_dims, &chunk_dims)
12265        }
12266    }
12267
12268    /// Append variable-length strings to an existing chunked vlen string dataset.
12269    ///
12270    /// Creates a new global heap collection for the strings, builds vlen
12271    /// references, and appends them as new chunks to the dataset.
12272    pub fn append_vlen_strings(&self, ds_index: usize, strings: &[&str]) -> IoResult<()> {
12273        use crate::format::global_heap::encode_vlen_reference;
12274        use crate::format::messages::datatype::DatatypeMessage;
12275
12276        if strings.is_empty() {
12277            return Ok(());
12278        }
12279
12280        // Whole-operation guard: buffer take, frame writes, re-buffer and
12281        // extend below are separate slot acquisitions that a concurrent
12282        // same-dataset append must not interleave with.
12283        let cell = self.ds(ds_index);
12284        let _op = cell.op.lock();
12285
12286        // The elements about to be written are vlen references; any other
12287        // element type would be overwritten with them as raw bytes.
12288        let charset = {
12289            let ds = self.ds(ds_index);
12290            let m = ds.lock();
12291            match m.datatype {
12292                DatatypeMessage::VarLenString { charset, .. } => charset,
12293                _ => {
12294                    return Err(crate::io::IoError::InvalidState(
12295                        "append_vlen_strings is only for variable-length string datasets".into(),
12296                    ))
12297                }
12298            }
12299        };
12300        ensure_vlen_charset(charset, strings)?;
12301
12302        // Every deterministic rejection must precede the heap write below:
12303        // a collection written for a batch the append then refuses (a
12304        // contiguous dataset, or a reopened dataset whose chunk index was
12305        // not reconstructed) is a 4096-byte orphan nothing references.
12306        let chunk_dims = self
12307            .dataset_chunk_dims(ds_index)
12308            .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?
12309            .to_vec();
12310        let dims = self.dataset_dims(ds_index).to_vec();
12311
12312        // Store the batch's strings as heap objects; a batch that fits an
12313        // earlier collection's free space shares its block.
12314        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12315        let placements = self.insert_vlen_objects(&items)?;
12316
12317        // Build raw vlen reference bytes
12318        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12319        let mut raw = Vec::with_capacity(strings.len() * ref_size);
12320        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12321            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12322            raw.extend_from_slice(&encode_vlen_reference(
12323                seq_len,
12324                gcol_addr,
12325                obj_idx as u32,
12326                &self.ctx,
12327            ));
12328        }
12329
12330        let n_new_frames = strings.len();
12331        let current_dim0 = dims[0] as usize;
12332        let chunk_dim0 = chunk_dims[0] as usize;
12333        let frame_bytes = ref_size;
12334
12335        // Merge the buffer with the new frames when it is the dataset's tail;
12336        // a buffer left mid-extent (the extent moved past it) keeps its
12337        // recorded place — flush it and start fresh at the current end.
12338        let taken = { self.ds(ds_index).lock().append.take() };
12339        let (base_dim0, buffered_frames, mut combined) = match taken {
12340            Some(b) if b.base + b.frames == current_dim0 as u64 => {
12341                (b.base as usize, b.frames as usize, b.bytes)
12342            }
12343            Some(b) => {
12344                self.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
12345                (current_dim0, 0, Vec::new())
12346            }
12347            None => (current_dim0, 0, Vec::new()),
12348        };
12349        combined.extend_from_slice(&raw);
12350
12351        let total_frames = buffered_frames + n_new_frames;
12352
12353        // Rows up to the last chunk boundary are written now; the tail that
12354        // does not complete a chunk goes back in the buffer for the next
12355        // append (or the flush at close). The boundary can precede
12356        // `base_dim0` — a reopened file's flushed partial chunk leaves the
12357        // base mid-chunk — in which case everything is tail.
12358        let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
12359        let write_frames = last_boundary.saturating_sub(base_dim0);
12360        let tail_frames = total_frames - write_frames;
12361        if write_frames > 0 {
12362            self.write_append_frames(
12363                ds_index,
12364                base_dim0 as u64,
12365                write_frames as u64,
12366                &combined[..write_frames * frame_bytes],
12367            )?;
12368        }
12369        if tail_frames > 0 {
12370            let ds = self.ds(ds_index);
12371            let mut m = ds.lock();
12372            m.append = Some(AppendBuffer {
12373                base: (base_dim0 + write_frames) as u64,
12374                frames: tail_frames as u64,
12375                bytes: combined[write_frames * frame_bytes..].to_vec(),
12376            });
12377        }
12378
12379        // Extend dims
12380        let logical_dim0 = base_dim0 + total_frames;
12381        let mut new_dims = dims;
12382        new_dims[0] = logical_dim0 as u64;
12383        self.extend_dataset_inner(ds_index, &new_dims)?;
12384
12385        Ok(())
12386    }
12387
12388    /// Replace elements `start .. start + strings.len()` of a 1-D
12389    /// variable-length string dataset, leaving its extent and every other
12390    /// element alone.
12391    ///
12392    /// The replacements go into the global heap and only the vlen
12393    /// references of the named elements are rewritten, so the cost is the
12394    /// new strings plus the chunks those references live in — not the column.
12395    /// The objects the old references pointed at are freed *before* the
12396    /// replacement is allocated, so repeated updates reuse space instead of
12397    /// growing the file — including across close/reopen cycles, where the
12398    /// in-memory free list starts empty and only this free-first order lets
12399    /// the session reuse the block it just released. This is what libhdf5
12400    /// does: `H5T__vlen_disk_write` deletes the reference it read into the
12401    /// conversion background buffer before storing the new one.
12402    ///
12403    /// Elements the append buffer still holds are flushed to their chunks
12404    /// first, so the whole range is on disk and one write path covers it.
12405    pub fn write_vlen_strings_slice(
12406        &self,
12407        ds_index: usize,
12408        start: u64,
12409        strings: &[&str],
12410    ) -> IoResult<()> {
12411        use crate::format::global_heap::{encode_vlen_reference, vlen_reference_size};
12412        use crate::format::messages::datatype::DatatypeMessage;
12413
12414        // An empty batch is a no-op: nothing to replace, nothing to free.
12415        if strings.is_empty() {
12416            return Ok(());
12417        }
12418
12419        // Whole-operation guard: the flush, the old-reference reads and the
12420        // slice write below must not interleave with a concurrent
12421        // same-dataset operation.
12422        let cell = self.ds(ds_index);
12423        let _op = cell.op.lock();
12424
12425        // Snapshot what the write needs, then drop the guard: `write_slice`
12426        // below re-locks the same slot.
12427        let (charset, dims, writable) = {
12428            let ds = self.ds(ds_index);
12429            let m = ds.lock();
12430            let charset = match m.datatype {
12431                DatatypeMessage::VarLenString { charset, .. } => charset,
12432                _ => {
12433                    return Err(crate::io::IoError::InvalidState(
12434                        "write_vlen_strings_slice is only for variable-length string datasets"
12435                            .into(),
12436                    ))
12437                }
12438            };
12439            let writable = if m.is_chunked() {
12440                Ok(())
12441            } else {
12442                match m.contiguous_target() {
12443                    Some(ContiguousTarget::Virtual) => Err(virtual_write_refused()),
12444                    Some(_) => Ok(()),
12445                    None => Err(crate::io::IoError::InvalidState(
12446                        "dataset has no data allocated".into(),
12447                    )),
12448                }
12449            };
12450            (charset, m.dataspace.dims.clone(), writable)
12451        };
12452
12453        // `write_slice_inner` rejects a dataset with neither chunk machinery
12454        // nor allocated data (a reopened dataset whose index was not
12455        // reconstructed), and refuses a virtual one outright — those
12456        // rejections must come before the heap write below, or every failed
12457        // call orphans a 4096-byte collection.
12458        writable?;
12459
12460        if dims.len() != 1 {
12461            return Err(crate::io::IoError::InvalidState(format!(
12462                "write_vlen_strings_slice is only for 1-dimension datasets, this one has {}",
12463                dims.len()
12464            )));
12465        }
12466        let end = start + strings.len() as u64;
12467        if end > dims[0] {
12468            return Err(crate::io::IoError::InvalidState(format!(
12469                "elements {start}..{end} are outside the dataset's {} elements",
12470                dims[0]
12471            )));
12472        }
12473        ensure_vlen_charset(charset, strings)?;
12474
12475        let ref_size = vlen_reference_size(&self.ctx);
12476
12477        // Elements the append buffer holds are not in the chunks yet: hand
12478        // them to the chunks first so the whole range is on disk and the one
12479        // write path below covers it.
12480        self.flush_append_buffer_if_intersecting(ds_index, start, end)?;
12481
12482        // The on-disk references about to be overwritten, read before anything
12483        // moves. libhdf5 reads the same bytes into the conversion background
12484        // buffer (`H5D__scatgath_write` gathers the file's current elements
12485        // when `need_bkg` is set) and hands them to `H5T__vlen_disk_write`,
12486        // which deletes them before storing the new reference.
12487        let superseded = self.current_element_bytes(ds_index, start, end - start, ref_size)?;
12488
12489        // Free the superseded objects *before* allocating the replacement,
12490        // the order `H5T__vlen_disk_write` uses. The freed block satisfies
12491        // the allocation below within this same session, so a reopen-and-
12492        // replace loop keeps the file flat — no persisted free-space
12493        // information exists to carry it across sessions (issue #10). The
12494        // cost, shared with libhdf5: a failure between here and the ref
12495        // write below leaves the dataset's old references dangling.
12496        self.release_vlen_references(&superseded)?;
12497
12498        // The insert comes after the release above so the space the release
12499        // recovered — a freed block, or in-collection bytes the release just
12500        // listed in `cwfs` — can satisfy this batch.
12501        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12502        let placements = self.insert_vlen_objects(&items)?;
12503
12504        let mut refs = Vec::with_capacity(strings.len() * ref_size);
12505        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12506            refs.extend_from_slice(&encode_vlen_reference(
12507                crate::format::global_heap::vlen_seq_len(strings[i].len())?,
12508                gcol_addr,
12509                obj_idx as u32,
12510                &self.ctx,
12511            ));
12512        }
12513
12514        self.write_slice_inner(ds_index, &[start], &[strings.len() as u64], &refs)?;
12515
12516        Ok(())
12517    }
12518
12519    /// The bytes elements `start .. start + count` of a 1-D dataset currently
12520    /// hold, whichever layout stores them.
12521    ///
12522    /// Elements no write has reached yet read as zeros — for a vlen dataset
12523    /// that is the nil reference, which names no heap object.
12524    fn current_element_bytes(
12525        &self,
12526        ds_index: usize,
12527        start: u64,
12528        count: u64,
12529        element_size: usize,
12530    ) -> IoResult<Vec<u8>> {
12531        let mut out = vec![0u8; count as usize * element_size];
12532        if count == 0 {
12533            return Ok(out);
12534        }
12535
12536        let (is_chunked, data_addr) = {
12537            let ds = self.ds(ds_index);
12538            let m = ds.lock();
12539            (m.is_chunked(), m.data_addr)
12540        };
12541
12542        if !is_chunked {
12543            if data_addr != UNDEF_ADDR {
12544                // `read_at_most`, not `read_at`: a contiguous dataset's block is
12545                // reserved when it is created, so the file can still be shorter
12546                // than the block until something writes it. What is missing has
12547                // never been written, which is the zeros above.
12548                let at = data_addr + start * element_size as u64;
12549                let got = self.handle.read_at_most(at, out.len())?;
12550                out[..got.len()].copy_from_slice(&got);
12551            }
12552            return Ok(out);
12553        }
12554
12555        let geo = self.chunk_geometry(ds_index)?;
12556        let per_chunk = geo.chunk_dims[0];
12557        // Only a corrupt or crafted file declares a zero-length chunk
12558        // dimension; the divisions below must reject it the way
12559        // `write_slice` does, not panic.
12560        if per_chunk == 0 {
12561            return Err(crate::io::IoError::InvalidState(
12562                "chunk shape has a zero-length dimension".into(),
12563            ));
12564        }
12565        let end = start + count;
12566        for c in (start / per_chunk)..=((end - 1) / per_chunk) {
12567            let origin = c * per_chunk;
12568            let lo = start.max(origin);
12569            let hi = end.min(origin + per_chunk);
12570            // A chunk with no block yet leaves this span as the zeros above.
12571            let Some(chunk) = self.read_chunk_at_coords(ds_index, &[c])? else {
12572                continue;
12573            };
12574            let src = ((lo - origin) as usize) * element_size;
12575            let dst = ((lo - start) as usize) * element_size;
12576            let len = ((hi - lo) as usize) * element_size;
12577            if src + len > chunk.len() {
12578                return Err(crate::io::IoError::InvalidState(format!(
12579                    "chunk {c} is {} bytes, too short for elements {lo}..{hi}",
12580                    chunk.len()
12581                )));
12582            }
12583            out[dst..dst + len].copy_from_slice(&chunk[src..src + len]);
12584        }
12585        Ok(out)
12586    }
12587
12588    /// Free the global heap objects `refs` names, so replacing a vlen element
12589    /// does not strand what it used to point at.
12590    ///
12591    /// Callers pass refs only for *top-level* vlen datatypes (the
12592    /// `collect_refs` / `is_vlen` decisions at the prune, delete and
12593    /// attribute-release sites all match `VarLenString`/`VarLenSequence`).
12594    /// A compound datatype with vlen members — writable only by a foreign
12595    /// library, never by this crate — keeps its members' heap objects when
12596    /// its storage is pruned, deleted or replaced.
12597    ///
12598    /// This is libhdf5's `H5HG_remove` reached through `H5T__vlen_disk_delete`:
12599    /// the object leaves its collection, the collection is rewritten at its
12600    /// existing size with the recovered bytes given to the free-space marker,
12601    /// and a collection that ends up empty returns its block to the allocator.
12602    /// A rewritten collection's recovered space is listed in `cwfs` for
12603    /// [`insert_vlen_objects`](Self::insert_vlen_objects) to pack into; a
12604    /// freed block leaves the list.
12605    /// A nil reference (address 0 or `UNDEF_ADDR`) names no object. The
12606    /// address decides, not the sequence length: this crate's writers store
12607    /// even the empty string as a real heap object, so a zero-length reference
12608    /// with a defined address still holds one that must be released. libhdf5
12609    /// diverges here against itself — `H5T__vlen_disk_delete` returns before
12610    /// `H5HG_remove` when the sequence length is zero, yet its write path
12611    /// (`H5VL__native_blob_put`) inserts a heap object even for an empty
12612    /// sequence, stranding it forever. The address rule frees those objects.
12613    ///
12614    /// Heap objects carry no reference count on this path, matching libhdf5:
12615    /// its vlen code never calls `H5HG_link` (only the virtual-dataset layer
12616    /// does). Releasing the same reference twice is absorbed by the
12617    /// missing-index check below, but a crafted file in which two elements
12618    /// share one heap object would lose it for the survivor when either is
12619    /// replaced — the same exposure the file has under libhdf5. This crate's
12620    /// writers never share: each element write inserts its own object.
12621    ///
12622    /// Under SWMR nothing is freed and no collection is rewritten: a reader may
12623    /// be following those references, the same reason `place_chunk` keeps a
12624    /// relocated chunk's old block.
12625    fn release_vlen_references(&self, refs: &[u8]) -> IoResult<()> {
12626        use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
12627
12628        let ref_size = vlen_reference_size(&self.ctx);
12629        if ref_size == 0 || refs.len() < ref_size {
12630            return Ok(());
12631        }
12632
12633        // Group by collection so one holding several replaced objects is read,
12634        // rewritten and judged empty exactly once.
12635        let mut per_collection: std::collections::BTreeMap<u64, Vec<u16>> = Default::default();
12636        for r in refs.chunks_exact(ref_size) {
12637            let (_seq_len, addr, obj_idx) = decode_vlen_reference(r, &self.ctx)?;
12638            if addr == 0 || addr == UNDEF_ADDR {
12639                continue;
12640            }
12641            let Ok(idx) = u16::try_from(obj_idx) else {
12642                return Err(crate::io::IoError::InvalidState(format!(
12643                    "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
12644                )));
12645            };
12646            per_collection.entry(addr).or_default().push(idx);
12647        }
12648        self.remove_heap_objects(per_collection)
12649    }
12650
12651    /// Remove global heap objects — `H5HG_remove` — given the object indices
12652    /// grouped by the collection they live in.
12653    ///
12654    /// The single owner of heap-object removal: the vlen release path above
12655    /// reaches it with the objects a replaced element used to name, and
12656    /// [`release_dataset_storage`](Self::release_dataset_storage) with the
12657    /// one mapping-list object a deleted virtual dataset owned, which is what
12658    /// `H5D__virtual_delete` frees the same way.
12659    fn remove_heap_objects(
12660        &self,
12661        per_collection: std::collections::BTreeMap<u64, Vec<u16>>,
12662    ) -> IoResult<()> {
12663        use crate::format::global_heap::GlobalHeapCollection;
12664
12665        if self.swmr_active {
12666            return Ok(());
12667        }
12668
12669        // The `cwfs` lock is held across the sweep: it serializes these
12670        // collection-block rewrites (and frees) against
12671        // `insert_vlen_objects`, which may be packing new objects into the
12672        // same blocks.
12673        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
12674        let mut cwfs = self.cwfs.lock();
12675        for (addr, indices) in per_collection {
12676            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
12677            // exactly that, so one read usually covers the whole image; only
12678            // an oversized collection needs a second read at its declared size.
12679            let mut image = self.handle.read_at_most(addr, 4096)?;
12680            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
12681            if declared > image.len() {
12682                image = self.handle.read_at(addr, declared)?;
12683            }
12684            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
12685            let mut removed_any = false;
12686            for idx in indices {
12687                removed_any |= gcol.remove_object(idx);
12688            }
12689            // Every index already gone (a stale or duplicate reference):
12690            // leave the image alone. Rewriting is not just wasted I/O — a
12691            // 100%-full collection written by libhdf5 has no free-space
12692            // marker, so re-encoding it at its declared size cannot fit one
12693            // and the whole element update would fail.
12694            if !removed_any {
12695                continue;
12696            }
12697            if gcol.is_empty() {
12698                self.allocator
12699                    .free(addr, declared as u64, FreeSpaceClass::RawData);
12700                // The block is gone; a lingering entry would let an insert
12701                // pack into space the allocator can hand to anything.
12702                cwfs.retain(|e| e.addr != addr);
12703            } else {
12704                let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
12705                self.handle.write_at(addr, &rewritten)?;
12706                // The recovered bytes are packable now — list them, the way
12707                // libhdf5's `H5HG_remove` adds the heap to `cwfs`.
12708                if let Some(free) = gcol.free_space_at(&self.ctx, declared) {
12709                    if free >= 2 * objhdr {
12710                        cwfs_note(&mut cwfs, addr, declared, free);
12711                    }
12712                }
12713            }
12714        }
12715        Ok(())
12716    }
12717
12718    /// Add an attribute to a dataset.
12719    ///
12720    /// The attribute will be written as a message in the dataset's object
12721    /// header when the file is finalized.
12722    pub fn add_dataset_attribute(&self, ds_index: usize, attr: AttributeMessage) -> IoResult<()> {
12723        self.set_attribute(AttrTarget::Dataset(ds_index), attr)
12724    }
12725
12726    /// Build a variable-length UTF-8 string attribute message.
12727    ///
12728    /// The string is stored as one object in a global heap collection and the
12729    /// returned [`AttributeMessage`] carries the vlen reference as its data,
12730    /// with a vlen-string datatype and scalar dataspace. h5py reads the value
12731    /// back as a Python `str` (not `bytes`).
12732    ///
12733    /// This is the single owner of vlen-string-attribute construction: every
12734    /// public string-attribute setter (dataset, group, root, and the SWMR
12735    /// equivalents) routes through it, so a `VarLenUnicode` /
12736    /// `set_attr_string` value is always stored as a true variable-length
12737    /// string rather than the fixed-length string it used to be.
12738    ///
12739    /// The string's heap object is placed by
12740    /// [`insert_vlen_objects`](Self::insert_vlen_objects), so consecutive
12741    /// attributes pack into a shared collection instead of each paying the
12742    /// 4096-byte `H5HG_MINALLOC` minimum for a block that holds one string.
12743    fn vlen_string_attribute(&self, name: &str, value: &str) -> IoResult<AttributeMessage> {
12744        use crate::format::global_heap::encode_vlen_reference;
12745        use crate::format::messages::dataspace::DataspaceMessage;
12746        use crate::format::messages::datatype::DatatypeMessage;
12747
12748        let (gcol_addr, obj_idx) = self.insert_vlen_objects(&[value.as_bytes()])?[0];
12749        let seq_len = crate::format::global_heap::vlen_seq_len(value.len())?;
12750        let data = encode_vlen_reference(seq_len, gcol_addr, obj_idx as u32, &self.ctx);
12751        Ok(AttributeMessage {
12752            name: name.to_string(),
12753            datatype: DatatypeMessage::vlen_string_utf8(),
12754            dataspace: DataspaceMessage::scalar(),
12755            data,
12756        })
12757    }
12758
12759    /// Build a variable-length UTF-8 string **array** attribute message.
12760    ///
12761    /// The N-dimensional counterpart of
12762    /// [`vlen_string_attribute`](Self::vlen_string_attribute): every element
12763    /// string is stored as one object in a single global heap collection, and
12764    /// the attribute data is the row-major concatenation of one vlen reference
12765    /// per element. The datatype is the same vlen-string datatype; the dataspace
12766    /// is the simple dataspace described by `shape` (an empty `shape` is a
12767    /// scalar). h5py reads the value back as a numpy array of Python `str` with
12768    /// that shape.
12769    ///
12770    /// The caller owns the invariant that `values.len()` equals the product of
12771    /// `shape` (the public setters validate it before calling). The element
12772    /// objects are placed by
12773    /// [`insert_vlen_objects`](Self::insert_vlen_objects) — a zero-element
12774    /// array allocates nothing, and each reference carries its element's
12775    /// own collection address.
12776    fn vlen_string_array_attribute(
12777        &self,
12778        name: &str,
12779        values: &[&str],
12780        shape: &[u64],
12781    ) -> IoResult<AttributeMessage> {
12782        use crate::format::global_heap::encode_vlen_reference;
12783        use crate::format::messages::dataspace::DataspaceMessage;
12784        use crate::format::messages::datatype::DatatypeMessage;
12785
12786        debug_assert_eq!(
12787            values.len() as u64,
12788            shape.iter().product::<u64>(),
12789            "vlen_string_array_attribute values.len() must equal product(shape)"
12790        );
12791
12792        let items: Vec<&[u8]> = values.iter().map(|v| v.as_bytes()).collect();
12793        let placements = self.insert_vlen_objects(&items)?;
12794
12795        let mut data = Vec::with_capacity(values.len() * 16);
12796        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12797            data.extend_from_slice(&encode_vlen_reference(
12798                crate::format::global_heap::vlen_seq_len(values[i].len())?,
12799                gcol_addr,
12800                obj_idx as u32,
12801                &self.ctx,
12802            ));
12803        }
12804        Ok(AttributeMessage {
12805            name: name.to_string(),
12806            datatype: DatatypeMessage::vlen_string_utf8(),
12807            dataspace: DataspaceMessage::simple(shape),
12808            data,
12809        })
12810    }
12811
12812    /// Set a user-defined fill value for a dataset.
12813    ///
12814    /// `bytes` must be exactly one element wide (matching the dataset's
12815    /// datatype). The value is emitted as a `fill_defined = 2` fill-value
12816    /// message in the dataset object header when the file is finalized.
12817    ///
12818    /// IMPORTANT: for a *contiguous* dataset this also immediately writes
12819    /// the tiled fill value across the whole data block, so it must be
12820    /// called BEFORE any `write_dataset_raw` / `write_slice` — otherwise the
12821    /// fill write clobbers data already written. (The high-level builder
12822    /// always calls this right after creating the dataset.)
12823    pub fn set_dataset_fill_value(&self, ds_index: usize, bytes: Vec<u8>) -> IoResult<()> {
12824        let count = self.dataset_count();
12825        if ds_index >= count {
12826            return Err(crate::io::IoError::InvalidState(format!(
12827                "dataset index {} out of range",
12828                ds_index
12829            )));
12830        }
12831        let ds_ref = self.ds(ds_index);
12832        let mut ds = ds_ref.lock();
12833        let es = ds.datatype.element_size() as usize;
12834        if bytes.len() != es {
12835            return Err(crate::io::IoError::InvalidState(format!(
12836                "fill value is {} bytes but dataset element size is {}",
12837                bytes.len(),
12838                es
12839            )));
12840        }
12841        // For a dataset with no per-chunk fill path the fill-value message
12842        // only declares fill-on-allocation — tile the fill value across the
12843        // storage itself now, so unwritten elements read back as the fill
12844        // value. Which storage that is depends on the layout: a compact
12845        // dataset's is the image inside its layout message, a contiguous
12846        // one's is its data block. (The high-level builder calls this
12847        // immediately after create, before any data is written; a subsequent
12848        // write_raw/write_slice overwrites its region.)
12849        // An implicitly indexed dataset is filled here too, and for the same
12850        // reason: that index has no per-chunk fill path because it has no
12851        // per-chunk anything — its whole chunk grid is one run of space,
12852        // allocated and filled at create like a contiguous block. So the test
12853        // is not "is it chunked" but "does something else fill its chunks".
12854        let fills_per_chunk = ds
12855            .chunk_index_kind()
12856            .is_some_and(|k| k != ChunkIndexKind::Implicit);
12857        // `H5D_FILL_TIME_NEVER` means exactly this: the library never writes
12858        // the fill value into allocated storage. Call `set_dataset_fill_time`
12859        // before this method to have it observed here — the storage this
12860        // would otherwise tile keeps whatever zero bytes its allocation
12861        // already gave it.
12862        if !fills_per_chunk && ds.fill_time != FILL_TIME_NEVER {
12863            if let Some(len) = ds.compact.as_ref().map(Vec::len) {
12864                ds.compact = Some(crate::format::messages::fill_value::tiled_fill(
12865                    len,
12866                    Some(&bytes),
12867                ));
12868            } else {
12869                // An implicit index's chunk grid is filled as one run, the
12870                // same way a contiguous block is, and storage this file did
12871                // not allocate is not filled at all; `allocated_storage_run`
12872                // is where both of those are decided.
12873                let run = ds.allocated_storage_run();
12874                if let Some((target, data_size)) = run.filter(|&(_, size)| size > 0) {
12875                    let filled = crate::format::messages::fill_value::tiled_fill(
12876                        data_size as usize,
12877                        Some(&bytes),
12878                    );
12879                    self.write_contiguous_bytes(&target, 0, &filled)?;
12880                }
12881            }
12882        }
12883
12884        ds.fill_value = Some(bytes);
12885        ds.header_dirty = true;
12886        Ok(())
12887    }
12888
12889    /// Set when the fill value is written into allocated storage —
12890    /// `H5Pset_fill_time`. `time` is one of [`FILL_TIME_ALLOC`],
12891    /// [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]; anything else is rejected
12892    /// the way `H5Pset_fill_time` rejects an out-of-range `H5D_fill_time_t`.
12893    ///
12894    /// Call this before [`set_dataset_fill_value`](Self::set_dataset_fill_value)
12895    /// so that a `FILL_TIME_NEVER` policy is in place before that call
12896    /// decides whether to eager-tile the value into storage. (The
12897    /// high-level builder always calls it first.)
12898    pub fn set_dataset_fill_time(&self, ds_index: usize, time: u8) -> IoResult<()> {
12899        if !matches!(time, FILL_TIME_ALLOC | FILL_TIME_NEVER | FILL_TIME_IFSET) {
12900            return Err(crate::io::IoError::InvalidState(format!(
12901                "invalid fill time {time}; must be {FILL_TIME_ALLOC} (alloc), \
12902                 {FILL_TIME_NEVER} (never) or {FILL_TIME_IFSET} (if-set)"
12903            )));
12904        }
12905        let count = self.dataset_count();
12906        if ds_index >= count {
12907            return Err(crate::io::IoError::InvalidState(format!(
12908                "dataset index {} out of range",
12909                ds_index
12910            )));
12911        }
12912        let ds_ref = self.ds(ds_index);
12913        let mut ds = ds_ref.lock();
12914        ds.fill_time = time;
12915        ds.header_dirty = true;
12916        Ok(())
12917    }
12918
12919    /// Allocate a `chunk_bytes`-sized buffer pre-filled with dataset
12920    /// `ds_index`'s fill value (tiled one element wide), or zeros when no
12921    /// user-defined fill value exists.
12922    ///
12923    /// Every partial chunk the writer emits must be built on top of a
12924    /// buffer from this method, so that the unwritten element region of an
12925    /// allocated chunk reads back as the fill value rather than zero.
12926    ///
12927    /// Unconditional: a shrink's straddler refill
12928    /// (`refill_chunk_beyond_extent`) calls this to repair data about to
12929    /// become reachable again, which libhdf5's `H5D__chunk_prune_fill` does
12930    /// regardless of the fill-time policy. [`new_write_chunk_buffer`](Self::new_write_chunk_buffer)
12931    /// is the gated counterpart for a chunk touched for the first time
12932    /// during a write, where the policy does apply.
12933    pub(crate) fn new_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
12934        let ds = self.ds(ds_index);
12935        let m = ds.lock();
12936        let fv = m.fill_value.as_deref();
12937        crate::format::messages::fill_value::tiled_fill(chunk_bytes, fv)
12938    }
12939
12940    /// The buffer a chunk gets the first time a write touches it — this
12941    /// dataset's allocation-time fill gate. `H5D__chunk_lock`'s cache-miss
12942    /// path (H5Dchunk.c:4894) fills such a buffer only for `ALLOC`, or for
12943    /// `IFSET` with a fill value defined; `NEVER` leaves it as the zeros a
12944    /// fresh buffer already has. Everything else about the buffer is
12945    /// [`new_chunk_buffer`](Self::new_chunk_buffer)'s.
12946    fn new_write_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
12947        let never = {
12948            let ds = self.ds(ds_index);
12949            let m = ds.lock();
12950            m.fill_time == FILL_TIME_NEVER
12951        };
12952        if never {
12953            vec![0u8; chunk_bytes]
12954        } else {
12955            self.new_chunk_buffer(ds_index, chunk_bytes)
12956        }
12957    }
12958
12959    /// Write `n_frames` whole frames whose first row is `base_frame`, for
12960    /// whichever chunk index the dataset uses and whatever its chunk shape.
12961    ///
12962    /// The single owner of an append's chunk writes. The frames are one
12963    /// hyperslab — rows `base_frame .. base_frame + n_frames` over the full
12964    /// row shape — so the write goes through
12965    /// [`write_slice_chunked`](Self::write_slice_chunked), the same engine
12966    /// `write_slice` uses: a chunk the span covers completely is written
12967    /// straight through, a partial one is read-modify-write on top of what
12968    /// is stored (or the fill value), and a chunk row narrower or wider
12969    /// than the frame row is scattered at the chunk stride. The previous
12970    /// owner required the extensible-array index and packed rows at the
12971    /// frame stride, so appends to a fixed-array or v2 B-tree dataset
12972    /// failed at close and lost the buffered rows.
12973    ///
12974    /// The caller holds the dataset's op lock or the writer exclusively.
12975    pub(crate) fn write_append_frames(
12976        &self,
12977        ds_index: usize,
12978        base_frame: u64,
12979        n_frames: u64,
12980        frames: &[u8],
12981    ) -> IoResult<()> {
12982        if n_frames == 0 {
12983            return Ok(());
12984        }
12985        let geo = self.chunk_geometry(ds_index)?;
12986        let mut starts = vec![0u64; geo.dims.len()];
12987        starts[0] = base_frame;
12988        let mut counts = geo.dims.clone();
12989        counts[0] = n_frames;
12990        let expected = counts.iter().product::<u64>() * geo.element_size;
12991        if frames.len() as u64 != expected {
12992            return Err(crate::io::IoError::InvalidState(format!(
12993                "{n_frames} frames at rows {base_frame}.. need {expected} bytes, got {}",
12994                frames.len()
12995            )));
12996        }
12997        self.write_slice_chunked(ds_index, &starts, &counts, frames)
12998    }
12999
13000    /// Write the dataset's append buffer (if any) into its chunks and clear
13001    /// it. The single owner of the buffer-to-chunks transition: the flush at
13002    /// close, an append meeting a non-contiguous buffer, and any operation
13003    /// about to write rows the buffer holds all come through here.
13004    ///
13005    /// The caller holds the dataset's op lock or the writer exclusively —
13006    /// the take and the frame writes are separate acquisitions.
13007    pub(crate) fn flush_append_buffer(&self, ds_index: usize) -> IoResult<()> {
13008        let taken = { self.ds(ds_index).lock().append.take() };
13009        match taken {
13010            Some(b) => self.write_append_frames(ds_index, b.base, b.frames, &b.bytes),
13011            None => Ok(()),
13012        }
13013    }
13014
13015    /// Flush the append buffer when rows `start_row .. end_row` intersect
13016    /// the buffered range — those rows' current content is the buffer, and
13017    /// writing them on disk while the buffer still holds them would be
13018    /// undone by the flush at close.
13019    ///
13020    /// The caller holds the dataset's op lock or the writer exclusively.
13021    pub(crate) fn flush_append_buffer_if_intersecting(
13022        &self,
13023        ds_index: usize,
13024        start_row: u64,
13025        end_row: u64,
13026    ) -> IoResult<()> {
13027        let intersects = {
13028            let ds = self.ds(ds_index);
13029            let m = ds.lock();
13030            m.append
13031                .as_ref()
13032                .is_some_and(|b| start_row < b.base + b.frames && end_row > b.base)
13033        };
13034        if intersects {
13035            self.flush_append_buffer(ds_index)
13036        } else {
13037            Ok(())
13038        }
13039    }
13040
13041    /// Read an already-written chunk's *decompressed* bytes when the chunk
13042    /// is allocated and resolvable from the in-memory extensible-array
13043    /// index. Handles index-block and data-block chunks, filtered and
13044    /// unfiltered.
13045    ///
13046    /// Returns `Ok(None)` only when the chunk has never been written
13047    /// (address `UNDEF`) or the index genuinely does not reach it, which for
13048    /// a read-modify-write means the chunk's content is the fill value.
13049    pub(crate) fn read_chunk_if_present(
13050        &self,
13051        ds_index: usize,
13052        chunk_idx: u64,
13053    ) -> IoResult<Option<Vec<u8>>> {
13054        // Phase 1: resolve the chunk's location from the in-memory index.
13055        // Hold the slot guard through Phase 1: `chunked` borrows it, while the
13056        // `self.handle`/`self.ctx` reads below touch disjoint fields.
13057        let ds = self.ds(ds_index);
13058        let m = ds.lock();
13059        let element_size = m.datatype.element_size() as u64;
13060        let pipeline = m.filter_pipeline.clone();
13061        let Some(chunked) = m.chunked.as_ref() else {
13062            return Ok(None);
13063        };
13064        let chunk_bytes = chunked.chunk_dims.iter().product::<u64>() * element_size;
13065        let max_nelmts_bits = chunked.earray_params.max_nelmts_bits;
13066        let chunk_size_len = chunked.chunk_size_len;
13067        let is_filtered = chunked.filt_iblk.is_some();
13068
13069        // The chunk entry is either read straight from an index block, or
13070        // located via a data block that must itself be read from disk.
13071        enum Loc {
13072            Direct(u64, u64, u32),
13073            DataBlock {
13074                dblk_addr: u64,
13075                offset: usize,
13076                nelmts: usize,
13077            },
13078        }
13079
13080        // Resolve the chunk's location with the libhdf5-compatible EA
13081        // geometry (super-block-grouped data blocks), matching `record_ea_chunk`.
13082        let ea_loc = {
13083            let p = &chunked.earray_params;
13084            EaGeometry::new(
13085                p.idx_blk_elmts,
13086                p.data_blk_min_elmts,
13087                p.sup_blk_min_data_ptrs,
13088                p.max_nelmts_bits,
13089                p.max_dblk_page_nelmts_bits,
13090            )?
13091            .locate(chunk_idx)?
13092        };
13093        let loc = match ea_loc {
13094            EaLoc::Index { elem } => {
13095                if is_filtered {
13096                    let e = &chunked.filt_iblk.as_ref().unwrap().elements[elem];
13097                    Loc::Direct(e.addr, e.nbytes, e.filter_mask)
13098                } else {
13099                    Loc::Direct(chunked.ea_iblk.elements[elem], chunk_bytes, 0)
13100                }
13101            }
13102            EaLoc::Dblk(l) => {
13103                if l.paged {
13104                    return Err(crate::io::IoError::InvalidState(format!(
13105                        "chunk index {} lives in a paged extensible-array data \
13106                         block, which is not yet supported for read-modify-write",
13107                        chunk_idx
13108                    )));
13109                }
13110                let dblk_addr = match l.path {
13111                    EaDblkPath::Direct { idx } => {
13112                        if is_filtered {
13113                            chunked.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
13114                        } else {
13115                            chunked.ea_iblk.dblk_addrs[idx]
13116                        }
13117                    }
13118                    EaDblkPath::ViaSblk {
13119                        sblk_off,
13120                        local_dblk,
13121                        ndblks_in_sblk,
13122                        ..
13123                    } => {
13124                        let sblk_addr = if is_filtered {
13125                            chunked.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
13126                        } else {
13127                            chunked.ea_iblk.sblk_addrs[sblk_off]
13128                        };
13129                        if sblk_addr == UNDEF_ADDR {
13130                            return Ok(None);
13131                        }
13132                        let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
13133                        let sb = ExtensibleArraySuperBlock::decode(
13134                            &sb_buf,
13135                            &self.ctx,
13136                            max_nelmts_bits,
13137                            ndblks_in_sblk,
13138                            0,
13139                        )?;
13140                        sb.dblk_addrs[local_dblk]
13141                    }
13142                };
13143                if dblk_addr == UNDEF_ADDR {
13144                    return Ok(None);
13145                }
13146                Loc::DataBlock {
13147                    dblk_addr,
13148                    offset: l.offset_in_dblk as usize,
13149                    nelmts: l.dblk_nelmts as usize,
13150                }
13151            }
13152        };
13153
13154        // Phase 2: resolve through the data block (if needed) and read. The
13155        // mask is the chunk's filter mask (0 for unfiltered), so a chunk
13156        // written via a direct chunk write with a skipped filter is reversed
13157        // correctly during read-modify-write.
13158        let (addr, nbytes, mask) = match loc {
13159            Loc::Direct(a, n, m) => (a, n, m),
13160            Loc::DataBlock {
13161                dblk_addr,
13162                offset,
13163                nelmts,
13164            } => {
13165                let buf = self.handle.read_at_most(dblk_addr, 65536)?;
13166                if is_filtered {
13167                    let dblk = FilteredDataBlock::decode(
13168                        &buf,
13169                        &self.ctx,
13170                        max_nelmts_bits,
13171                        nelmts,
13172                        chunk_size_len,
13173                    )?;
13174                    let e = &dblk.elements[offset];
13175                    (e.addr, e.nbytes, e.filter_mask)
13176                } else {
13177                    let dblk =
13178                        ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, nelmts)?;
13179                    (dblk.elements[offset], chunk_bytes, 0)
13180                }
13181            }
13182        };
13183        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13184    }
13185
13186    /// Read one stored chunk block and undo its filters.
13187    ///
13188    /// `nbytes` is the *stored* length and `mask` the chunk's filter mask, so
13189    /// a chunk written by a direct chunk write with a skipped filter is
13190    /// reversed correctly. `Ok(None)` means the chunk has no block yet — the
13191    /// single place that judgement is made, shared by every chunk index.
13192    fn read_chunk_block(
13193        &self,
13194        pipeline: Option<&FilterPipeline>,
13195        addr: u64,
13196        nbytes: u64,
13197        mask: u32,
13198    ) -> IoResult<Option<Vec<u8>>> {
13199        if addr == UNDEF_ADDR || nbytes == 0 {
13200            return Ok(None);
13201        }
13202        let raw = self.handle.read_at(addr, nbytes as usize)?;
13203        match pipeline {
13204            Some(pl) => Ok(Some(filter::reverse_filters_masked(pl, &raw, mask)?)),
13205            None => Ok(Some(raw)),
13206        }
13207    }
13208
13209    /// Read the *decompressed* bytes of the chunk at `chunk_coords`, whichever
13210    /// chunk index the dataset uses, or `Ok(None)` when that chunk has never
13211    /// been written.
13212    ///
13213    /// This is the read half of a partial-chunk read-modify-write: a hyperslab
13214    /// write that covers only part of a chunk must start from what is already
13215    /// there. Keeping one entry point for all three index types is what lets
13216    /// [`write_slice`](Self::write_slice) stay index-agnostic.
13217    pub(crate) fn read_chunk_at_coords(
13218        &self,
13219        ds_index: usize,
13220        chunk_coords: &[u64],
13221    ) -> IoResult<Option<Vec<u8>>> {
13222        let geo = self.chunk_geometry(ds_index)?;
13223        // Only the linearly-addressed indexes compute a slot; a v2 B-tree is
13224        // keyed by the coordinates themselves (and may hold unlimited inner
13225        // dimensions, which have no linear slot).
13226        match geo.kind {
13227            ChunkIndexKind::ExtensibleArray => {
13228                let linear = geo.linear_index(chunk_coords)?;
13229                self.read_chunk_if_present(ds_index, linear)
13230            }
13231            ChunkIndexKind::FixedArray => {
13232                let linear = geo.linear_index(chunk_coords)?;
13233                let ds = self.ds(ds_index);
13234                let m = ds.lock();
13235                let pipeline = m.filter_pipeline.clone();
13236                let fa = m.fixed_array.as_ref().unwrap();
13237                let lidx = linear as usize;
13238                let (addr, nbytes, mask) = if pipeline.is_some() {
13239                    match fa.fa_dblk.filtered_elements.get(lidx) {
13240                        Some(e) => (e.address, e.chunk_size, e.filter_mask),
13241                        None => return Ok(None),
13242                    }
13243                } else {
13244                    match fa.fa_dblk.elements.get(lidx) {
13245                        Some(&a) => (a, geo.chunk_bytes(), 0),
13246                        None => return Ok(None),
13247                    }
13248                };
13249                drop(m);
13250                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13251            }
13252            ChunkIndexKind::BtreeV2 => {
13253                let ds = self.ds(ds_index);
13254                let m = ds.lock();
13255                let pipeline = m.filter_pipeline.clone();
13256                let bt2 = m.btree_v2.as_ref().unwrap();
13257                // A filtered index records the stored size and mask per chunk;
13258                // an unfiltered one stores whole chunks, so their size is the
13259                // chunk shape and no filter ran.
13260                let found = if bt2.index.filtered {
13261                    bt2.index
13262                        .lookup_filtered(chunk_coords)
13263                        .map(|r| (r.chunk_address, r.chunk_size, r.filter_mask))
13264                } else {
13265                    bt2.index
13266                        .lookup(chunk_coords)
13267                        .map(|r| (r.chunk_address, geo.chunk_bytes(), 0))
13268                };
13269                drop(m);
13270                match found {
13271                    Some((addr, nbytes, mask)) => {
13272                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13273                    }
13274                    None => Ok(None),
13275                }
13276            }
13277            // Every chunk of an implicitly indexed dataset exists from the
13278            // moment the dataset does, so there is no "never written" answer
13279            // to give: an untouched chunk reads back as the fill value the
13280            // create wrote there.
13281            ChunkIndexKind::Implicit => {
13282                let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
13283                self.read_chunk_block(None, grid + offset, geo.chunk_bytes(), 0)
13284            }
13285            // A single-chunk dataset's one chunk is never written until its
13286            // first write (unless the dataset was early-allocated and
13287            // unfiltered, in which case create already gave it an address) —
13288            // unlike Implicit, `UNDEF_ADDR` here is a real "never written".
13289            ChunkIndexKind::SingleChunk => {
13290                let ds = self.ds(ds_index);
13291                let m = ds.lock();
13292                let pipeline = m.filter_pipeline.clone();
13293                let sc = m.single_chunk.as_ref().unwrap();
13294                if sc.data_addr == UNDEF_ADDR {
13295                    return Ok(None);
13296                }
13297                let (addr, nbytes, mask) = if pipeline.is_some() {
13298                    (sc.data_addr, sc.nbytes, sc.filter_mask)
13299                } else {
13300                    (sc.data_addr, geo.chunk_bytes(), 0)
13301                };
13302                drop(m);
13303                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13304            }
13305            ChunkIndexKind::BtreeV1 => {
13306                let ds = self.ds(ds_index);
13307                let m = ds.lock();
13308                let pipeline = m.filter_pipeline.clone();
13309                let bt1 = m.btree_v1.as_ref().unwrap();
13310                let found = bt1
13311                    .position(chunk_coords)
13312                    .ok()
13313                    .map(|i| &bt1.records[i])
13314                    .map(|r| (r.address, r.nbytes as u64, r.filter_mask));
13315                drop(m);
13316                match found {
13317                    Some((addr, nbytes, mask)) => {
13318                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13319                    }
13320                    None => Ok(None),
13321                }
13322            }
13323        }
13324    }
13325
13326    /// The slot one chunk of an implicitly indexed dataset occupies: the
13327    /// address its whole chunk grid starts at, and the chunk's offset within
13328    /// that grid. `data_addr + linear_index * chunk_bytes` is the whole of
13329    /// that index (`H5D__none_idx_get_addr`, H5Dnone.c).
13330    ///
13331    /// The one place a chunk of such a dataset is placed — read and write both
13332    /// come through here, so the bounds check below covers both. The grid it
13333    /// names is [`DatasetInfo::implicit_grid`], which is why the write side
13334    /// can hand [`ContiguousTarget::Local`] to
13335    /// [`write_contiguous_bytes`](Self::write_contiguous_bytes) without asking
13336    /// anything: the external and virtual destinations that owner also knows
13337    /// about are unreachable from a chunked dataset.
13338    fn implicit_chunk_slot(
13339        &self,
13340        ds_index: usize,
13341        geo: &ChunkGeometry,
13342        chunk_coords: &[u64],
13343    ) -> IoResult<(u64, u64)> {
13344        let linear = geo.linear_index(chunk_coords)?;
13345        let ds = self.ds(ds_index);
13346        let m = ds.lock();
13347        let (grid, grid_size) = m.implicit_grid().ok_or_else(|| {
13348            crate::io::IoError::InvalidState("no implicitly indexed chunk grid".into())
13349        })?;
13350        let offset = linear.checked_mul(geo.chunk_bytes()).ok_or_else(|| {
13351            crate::io::IoError::InvalidState("implicit chunk offset overflows u64".into())
13352        })?;
13353        if offset + geo.chunk_bytes() > grid_size {
13354            return Err(crate::io::IoError::InvalidState(format!(
13355                "chunk {chunk_coords:?} lies outside the {grid_size} bytes of chunk space \
13356                 this implicitly indexed dataset was created with"
13357            )));
13358        }
13359        Ok((grid, offset))
13360    }
13361
13362    /// Write one whole chunk addressed by its grid coordinates, whichever
13363    /// chunk index the dataset uses. `data` is the chunk's unfiltered bytes;
13364    /// the dataset's filter pipeline (if any) runs here.
13365    ///
13366    /// The write half of the pair with
13367    /// [`read_chunk_at_coords`](Self::read_chunk_at_coords). Unlike the
13368    /// dataset-level `write_chunk_at`, this never grows the dataspace — a
13369    /// hyperslab write is bounded by the current extent by definition.
13370    ///
13371    /// The caller holds the dataset's op lock or the writer exclusively.
13372    pub(crate) fn write_chunk_at_coords(
13373        &self,
13374        ds_index: usize,
13375        chunk_coords: &[u64],
13376        data: &[u8],
13377    ) -> IoResult<()> {
13378        let geo = self.chunk_geometry(ds_index)?;
13379        match geo.kind {
13380            ChunkIndexKind::ExtensibleArray => {
13381                let linear = geo.linear_index(chunk_coords)?;
13382                self.write_chunk_inner(ds_index, linear, data)
13383            }
13384            ChunkIndexKind::FixedArray => {
13385                self.write_chunk_fixed_array_inner(ds_index, chunk_coords, data)
13386            }
13387            ChunkIndexKind::BtreeV2 => {
13388                self.write_chunk_btree_v2_inner(ds_index, chunk_coords, data)
13389            }
13390            ChunkIndexKind::Implicit => {
13391                self.write_chunk_implicit_inner(ds_index, chunk_coords, data)
13392            }
13393            ChunkIndexKind::SingleChunk => {
13394                self.write_chunk_single_chunk_inner(ds_index, chunk_coords, data)
13395            }
13396            ChunkIndexKind::BtreeV1 => {
13397                self.write_chunk_btree_v1_inner(ds_index, chunk_coords, data)
13398            }
13399        }
13400    }
13401
13402    /// Write one whole chunk to a dataset indexed by a version-1 B-tree.
13403    ///
13404    /// `chunk_coords` is the chunk's grid position. `data` is the chunk's
13405    /// unfiltered bytes; the dataset's filter pipeline runs here if it has
13406    /// one, and the key records the stored size and mask the way libhdf5's
13407    /// does (`H5D__btree_new_node`).
13408    ///
13409    /// The caller holds the dataset's op lock or the writer exclusively.
13410    pub(crate) fn write_chunk_btree_v1_inner(
13411        &self,
13412        ds_index: usize,
13413        chunk_coords: &[u64],
13414        data: &[u8],
13415    ) -> IoResult<()> {
13416        // Read what the write needs under a brief guard, then filter OUTSIDE
13417        // the lock, as every other index's write path does.
13418        let ds = self.ds(ds_index);
13419        let (chunk_bytes, pipeline) = {
13420            let m = ds.lock();
13421            let element_size = m.datatype.element_size() as u64;
13422            let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
13423                crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
13424            })?;
13425            (
13426                bt1.chunk_dims.iter().product::<u64>() * element_size,
13427                m.filter_pipeline.clone(),
13428            )
13429        };
13430        if data.len() as u64 != chunk_bytes {
13431            return Err(crate::io::IoError::InvalidState(format!(
13432                "chunk data size mismatch: expected {} bytes, got {}",
13433                chunk_bytes,
13434                data.len()
13435            )));
13436        }
13437
13438        let filtered;
13439        let stored = match pipeline {
13440            Some(ref pl) => {
13441                filtered = filter::apply_filters(pl, data)?;
13442                &filtered[..]
13443            }
13444            None => data,
13445        };
13446        self.record_btree_v1_chunk(ds_index, chunk_coords, stored, 0)
13447    }
13448
13449    /// Write a pre-filtered chunk verbatim to a version-1 B-tree dataset,
13450    /// recording the caller-supplied `filter_mask` — the classic-index half
13451    /// of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
13452    ///
13453    /// The caller holds the dataset's op lock or the writer exclusively.
13454    pub(crate) fn write_compressed_chunk_btree_v1_inner(
13455        &self,
13456        ds_index: usize,
13457        chunk_coords: &[u64],
13458        data: &[u8],
13459        filter_mask: u32,
13460    ) -> IoResult<()> {
13461        if self.ds(ds_index).lock().filter_pipeline.is_none() {
13462            return Err(crate::io::IoError::InvalidState(
13463                "write_chunk_raw requires a filtered dataset (an unfiltered chunk \
13464                 is stored at its full size, so there is nothing for a stored size \
13465                 or a filter mask to say)"
13466                    .into(),
13467            ));
13468        }
13469        self.record_btree_v1_chunk(ds_index, chunk_coords, data, filter_mask)
13470    }
13471
13472    /// Place a chunk's already-final bytes in the file and record them in the
13473    /// version-1 B-tree under the caller-supplied `filter_mask`.
13474    ///
13475    /// Shared by the two writes above, so both reach the index through one
13476    /// placement rule. The records are kept in key order here — the bulk load
13477    /// at flush walks them in that order and a lookup bisects them.
13478    fn record_btree_v1_chunk(
13479        &self,
13480        ds_index: usize,
13481        chunk_coords: &[u64],
13482        final_bytes: &[u8],
13483        filter_mask: u32,
13484    ) -> IoResult<()> {
13485        let stored_len = final_bytes.len() as u64;
13486        // The key's size field is 32 bits wide (`H5D_btree_key_t::nbytes`),
13487        // which is also libhdf5's limit on a chunk in this index.
13488        let Ok(nbytes) = u32::try_from(stored_len) else {
13489            return Err(crate::io::IoError::InvalidState(format!(
13490                "stored chunk size {stored_len} does not fit in the 32-bit size \
13491                 field of a version-1 B-tree chunk key"
13492            )));
13493        };
13494        let ds = self.ds(ds_index);
13495        let mut m = ds.lock();
13496        let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
13497            crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
13498        })?;
13499        if chunk_coords.len() != bt1.chunk_dims.len() {
13500            return Err(crate::io::IoError::InvalidState(format!(
13501                "chunk_coords has {} entries but the dataset has {} dimensions",
13502                chunk_coords.len(),
13503                bt1.chunk_dims.len()
13504            )));
13505        }
13506        // A coordinate past the maximum extent has no chunk to be: unlike the
13507        // array indexes there is no slot to run out of, so the bound is
13508        // checked here or not at all. An unlimited dimension has none.
13509        for (d, ((&c, &cd), &max)) in chunk_coords
13510            .iter()
13511            .zip(&bt1.chunk_dims)
13512            .zip(&bt1.max_dims)
13513            .enumerate()
13514        {
13515            if max != u64::MAX && c.saturating_mul(cd) >= max {
13516                return Err(crate::io::IoError::InvalidState(format!(
13517                    "chunk coordinate {c} in dimension {d} is outside the maximum \
13518                     extent {max}"
13519                )));
13520            }
13521        }
13522        let slot = bt1.position(chunk_coords);
13523        let old = slot.ok().map(|i| {
13524            let r = &bt1.records[i];
13525            (r.address, r.nbytes as u64)
13526        });
13527        // A rewrite whose stored size is unchanged stays where it is (always
13528        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
13529        let address = self.place_chunk(old, stored_len);
13530        self.handle.write_at(address, final_bytes)?;
13531
13532        let bt1 = m.btree_v1.as_mut().unwrap();
13533        let record = BtreeV1ChunkRecord {
13534            scaled: chunk_coords.to_vec(),
13535            address,
13536            nbytes,
13537            filter_mask,
13538        };
13539        match slot {
13540            Ok(i) => bt1.records[i] = record,
13541            Err(i) => bt1.records.insert(i, record),
13542        }
13543        bt1.chunks_written += 1;
13544        Ok(())
13545    }
13546
13547    /// Write one whole chunk of an implicitly indexed dataset into the slot
13548    /// its coordinates name. There is no index to record anything in — the
13549    /// slot is where it always was — so this is the write in full.
13550    ///
13551    /// The bytes go through [`write_contiguous_bytes`](Self::write_contiguous_bytes),
13552    /// the one owner of a raw-byte write, against the grid
13553    /// [`implicit_chunk_slot`](Self::implicit_chunk_slot) names.
13554    ///
13555    /// The caller holds the dataset's op lock or the writer exclusively.
13556    pub(crate) fn write_chunk_implicit_inner(
13557        &self,
13558        ds_index: usize,
13559        chunk_coords: &[u64],
13560        data: &[u8],
13561    ) -> IoResult<()> {
13562        let geo = self.chunk_geometry(ds_index)?;
13563        let chunk_bytes = geo.chunk_bytes();
13564        if data.len() as u64 != chunk_bytes {
13565            return Err(crate::io::IoError::InvalidState(format!(
13566                "chunk data size mismatch: expected {} bytes, got {}",
13567                chunk_bytes,
13568                data.len()
13569            )));
13570        }
13571        let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
13572        self.write_contiguous_bytes(&ContiguousTarget::Local(grid), offset, data)
13573    }
13574
13575    /// Snapshot the geometry needed to address a chunked dataset's grid.
13576    ///
13577    /// Taken under one brief slot guard so the callers below — which re-lock
13578    /// the slot through `write_chunk`/`read_chunk_*` — never hold it across
13579    /// compression or I/O.
13580    fn chunk_geometry(&self, ds_index: usize) -> IoResult<ChunkGeometry> {
13581        let ds = self.ds(ds_index);
13582        let m = ds.lock();
13583        let Some(kind) = m.chunk_index_kind() else {
13584            return Err(crate::io::IoError::InvalidState(
13585                "not a chunked dataset".into(),
13586            ));
13587        };
13588        let chunk_dims = match kind {
13589            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
13590            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
13591            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
13592            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
13593            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
13594            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
13595        };
13596        Ok(ChunkGeometry {
13597            kind,
13598            dims: m.dataspace.dims.clone(),
13599            max_dims: m.dataspace.max_dims.clone(),
13600            chunk_dims,
13601            element_size: m.datatype.element_size() as u64,
13602        })
13603    }
13604
13605    /// Index-grid slot of the chunk at grid `coords` (see
13606    /// [`crate::io::chunk_grid`]).
13607    pub(crate) fn chunk_slot(&self, ds_index: usize, coords: &[u64]) -> IoResult<u64> {
13608        self.chunk_geometry(ds_index)?.linear_index(coords)
13609    }
13610
13611    /// Grid coordinates of the chunk recorded under index-grid slot `linear`
13612    /// — the inverse of [`Self::chunk_slot`].
13613    pub(crate) fn chunk_coords_from_slot(
13614        &self,
13615        ds_index: usize,
13616        linear: u64,
13617    ) -> IoResult<Vec<u64>> {
13618        let geo = self.chunk_geometry(ds_index)?;
13619        crate::io::chunk_grid::coords_of(
13620            &geo.dims,
13621            geo.max_dims.as_deref(),
13622            &geo.chunk_dims,
13623            linear,
13624        )
13625    }
13626
13627    /// Define a chunked dataset indexed by a fixed array, fixed at its
13628    /// current shape (`max_dims == dims`). `chunk_dims` defines the chunk
13629    /// shape. Returns the dataset index.
13630    pub fn create_fixed_array_dataset(
13631        &self,
13632        name: &str,
13633        datatype: DatatypeMessage,
13634        dims: &[u64],
13635        chunk_dims: &[u64],
13636    ) -> IoResult<usize> {
13637        self.create_fixed_array_dataset_with_max(name, datatype, dims, dims, chunk_dims, None)
13638    }
13639
13640    /// Define a fixed-shape compressed chunked dataset indexed by a
13641    /// *filtered* Fixed Array (`max_dims == dims`).
13642    ///
13643    /// Like `create_fixed_array_dataset`, but the FA header carries the filtered
13644    /// client id and a `chunk_size_len`-wide compressed-size field per chunk
13645    /// (`FixedArrayFilteredChunkElement`), and the dataset gets a filter
13646    /// pipeline. Chunks written via `write_chunk_fixed_array` are compressed and
13647    /// their compressed size + filter mask are recorded in the data block.
13648    ///
13649    /// A convenience over [`create_fixed_array_dataset_with_max`]'s own
13650    /// pipeline argument; production dataset creation calls that directly,
13651    /// so this is kept as a direct entry point for this crate's own
13652    /// white-box tests.
13653    ///
13654    /// [`create_fixed_array_dataset_with_max`]: Self::create_fixed_array_dataset_with_max
13655    #[cfg(all(test, feature = "deflate"))]
13656    pub fn create_fixed_array_dataset_with_pipeline(
13657        &self,
13658        name: &str,
13659        datatype: DatatypeMessage,
13660        dims: &[u64],
13661        chunk_dims: &[u64],
13662        pipeline: FilterPipeline,
13663    ) -> IoResult<usize> {
13664        self.create_fixed_array_dataset_with_max(
13665            name,
13666            datatype,
13667            dims,
13668            dims,
13669            chunk_dims,
13670            Some(pipeline),
13671        )
13672    }
13673
13674    /// Define a chunked dataset indexed by a fixed array, growable up to
13675    /// `max_dims` (every maximum finite — libhdf5 picks this index exactly
13676    /// when no dimension is unlimited).
13677    ///
13678    /// The array is sized for the chunk grid of the *maximum* extent, the
13679    /// libhdf5 rule (`H5D__farray_idx_create` uses `max_nchunks`), so the
13680    /// dataset can be extended to `max_dims` without re-indexing chunks.
13681    pub fn create_fixed_array_dataset_with_max(
13682        &self,
13683        name: &str,
13684        datatype: DatatypeMessage,
13685        dims: &[u64],
13686        max_dims: &[u64],
13687        chunk_dims: &[u64],
13688        pipeline: Option<FilterPipeline>,
13689    ) -> IoResult<usize> {
13690        let create = self.begin_create(name)?;
13691        let name = create.name.as_str();
13692        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
13693        if max_dims.contains(&u64::MAX) {
13694            return Err(crate::io::IoError::InvalidState(
13695                "a fixed-array index requires a fixed maximum shape (no unlimited dimension)"
13696                    .into(),
13697            ));
13698        }
13699        let mut num_chunks: u64 = 1;
13700        for g in crate::io::chunk_grid::index_grid(dims, Some(max_dims), chunk_dims)? {
13701            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
13702                crate::io::IoError::InvalidState("chunk count overflows u64".into())
13703            })?;
13704        }
13705
13706        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13707        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
13708
13709        // Create the FA header. For a filtered FA, chunk_size_len is sized
13710        // the same way the filtered Extensible Array path computes it:
13711        // derived from the uncompressed chunk byte count under layout v4,
13712        // the fixed `sizeof_size` under layout v5.
13713        let mut fa_header = if pipeline.is_some() {
13714            let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
13715            FixedArrayHeader::new_for_filtered_chunks(&self.ctx, num_chunks, chunk_size_len)
13716        } else {
13717            FixedArrayHeader::new_for_chunks(&self.ctx, num_chunks)
13718        };
13719        let hdr_encoded = fa_header.encode(&self.ctx);
13720        let fa_header_addr = self
13721            .allocator
13722            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
13723
13724        // Create the FA data block. libhdf5 switches to a paged layout once
13725        // num_elmts exceeds dblk_page_nelmts; both layouts allocate space
13726        // for `num_chunks` entries up front, but the paged layout also
13727        // reserves the page-init bitmap and a per-page checksum.
13728        let fa_dblk = if pipeline.is_some() {
13729            FixedArrayDataBlock::new_filtered(fa_header_addr, num_chunks as usize)
13730        } else {
13731            FixedArrayDataBlock::new_unfiltered(fa_header_addr, num_chunks as usize)
13732        };
13733        let dblk_size = fixed_array_dblk_disk_size(&self.ctx, &fa_header);
13734        let fa_dblk_addr = self.allocator.allocate(dblk_size, FreeSpaceClass::Metadata);
13735
13736        // Update header with data block address
13737        fa_header.data_blk_addr = fa_dblk_addr;
13738
13739        // Write both. The data block content is finalized in `flush_dataset`
13740        // once all chunk addresses are known; here we just reserve space and
13741        // write the header so the file is structurally consistent.
13742        let hdr_encoded = fa_header.encode(&self.ctx);
13743        self.handle.write_at(fa_header_addr, &hdr_encoded)?;
13744        let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa_header, &fa_dblk);
13745        debug_assert_eq!(dblk_encoded.len() as u64, dblk_size);
13746        self.handle.write_at(fa_dblk_addr, &dblk_encoded)?;
13747
13748        // The maximum is stored even when it equals the dims: it is what
13749        // `extend_dataset` checks growth against, and the FA capacity above
13750        // is exactly its chunk grid.
13751        let dataspace = DataspaceMessage {
13752            // Chunked storage always requires at least one dimension, so
13753            // this is never Scalar or Null.
13754            class: DataspaceClass::Simple,
13755            dims: dims.to_vec(),
13756            max_dims: Some(max_dims.to_vec()),
13757        };
13758
13759        let idx = self.push_dataset(
13760            &create,
13761            DatasetInfo {
13762                name: name.to_string(),
13763                datatype,
13764                committed_type: None,
13765                external: None,
13766                virtual_storage: None,
13767                dataspace,
13768                read_format: None,
13769                obj_header_addr: 0,
13770                data_addr: UNDEF_ADDR,
13771                data_size: 0,
13772                compact: None,
13773                attributes: Vec::new(),
13774                obj_header_written_addr: None,
13775                obj_header_blocks: Vec::new(),
13776                filter_pipeline: pipeline,
13777                deleted: false,
13778                extent_dirty: false,
13779                header_dirty: false,
13780                nlink_written: 1,
13781                creation_seq: self.take_creation_seq(),
13782                track_attr_order: self.track_order.attrs,
13783                fill_value: None,
13784                fill_time: FILL_TIME_IFSET,
13785                layout_version,
13786                times: self.created_object_times(),
13787                chunked: None,
13788                btree_v2: None,
13789                implicit: None,
13790                single_chunk: None,
13791                btree_v1: None,
13792                fixed_array: Some(FixedArrayDatasetInfo {
13793                    chunk_dims: chunk_dims.to_vec(),
13794                    fa_header_addr,
13795                    fa_dblk_addr,
13796                    fa_header,
13797                    fa_dblk,
13798                    chunks_written: 0,
13799                }),
13800                append: None,
13801            },
13802        );
13803
13804        Ok(idx)
13805    }
13806
13807    /// Define a chunked dataset with the *implicit* index: no index structure
13808    /// at all, every chunk of the grid allocated at create in one contiguous
13809    /// run, addressed by arithmetic (`H5Dnone.c`).
13810    ///
13811    /// libhdf5 picks this index only where that arithmetic is total, and this
13812    /// enforces the same three conditions
13813    /// (`H5D__layout_set_latest_indexing`, H5Dlayout.c): no filter — a
13814    /// filtered chunk is not `chunk_bytes` long, so the run would not be a
13815    /// grid; no unlimited dimension — the run has to have a length; and early
13816    /// allocation, which is what this creator *does* rather than something it
13817    /// checks. The dataset's fill-value message says so
13818    /// (`build_dataset_header`), because a file claiming incremental
13819    /// allocation is one libhdf5 would never have chosen this index for.
13820    pub fn create_implicit_dataset(
13821        &self,
13822        name: &str,
13823        datatype: DatatypeMessage,
13824        dims: &[u64],
13825        chunk_dims: &[u64],
13826    ) -> IoResult<usize> {
13827        let create = self.begin_create(name)?;
13828        let name = create.name.as_str();
13829        validate_chunk_geometry(dims, dims, chunk_dims)?;
13830        let mut num_chunks: u64 = 1;
13831        for g in crate::io::chunk_grid::index_grid(dims, None, chunk_dims)? {
13832            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
13833                crate::io::IoError::InvalidState("chunk count overflows u64".into())
13834            })?;
13835        }
13836        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13837        let data_size = num_chunks.checked_mul(chunk_bytes).ok_or_else(|| {
13838            crate::io::IoError::InvalidState("implicit chunk storage overflows u64".into())
13839        })?;
13840        let layout_version = self.chunk_layout_version(false, chunk_bytes);
13841
13842        // Early allocation is the whole of this index: the run exists, and
13843        // holds the fill value, before any chunk is written. It is written
13844        // out rather than merely reserved because the file's end-of-file
13845        // address is what libhdf5 checks a file's completeness against — a
13846        // reserved-but-absent tail is a truncated file to it.
13847        let data_addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
13848        self.handle.write_at(
13849            data_addr,
13850            &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
13851        )?;
13852
13853        let dataspace = DataspaceMessage {
13854            // Chunked storage always requires at least one dimension, so
13855            // this is never Scalar or Null.
13856            class: DataspaceClass::Simple,
13857            dims: dims.to_vec(),
13858            max_dims: Some(dims.to_vec()),
13859        };
13860
13861        let idx = self.push_dataset(
13862            &create,
13863            DatasetInfo {
13864                name: name.to_string(),
13865                datatype,
13866                committed_type: None,
13867                external: None,
13868                virtual_storage: None,
13869                dataspace,
13870                read_format: None,
13871                obj_header_addr: 0,
13872                data_addr: UNDEF_ADDR,
13873                data_size: 0,
13874                compact: None,
13875                attributes: Vec::new(),
13876                obj_header_written_addr: None,
13877                obj_header_blocks: Vec::new(),
13878                filter_pipeline: None,
13879                deleted: false,
13880                extent_dirty: false,
13881                header_dirty: false,
13882                nlink_written: 1,
13883                creation_seq: self.take_creation_seq(),
13884                track_attr_order: self.track_order.attrs,
13885                fill_value: None,
13886                fill_time: FILL_TIME_IFSET,
13887                layout_version,
13888                times: self.created_object_times(),
13889                chunked: None,
13890                btree_v2: None,
13891                fixed_array: None,
13892                implicit: Some(ImplicitDatasetInfo {
13893                    chunk_dims: chunk_dims.to_vec(),
13894                    data_addr,
13895                    data_size,
13896                }),
13897                single_chunk: None,
13898                btree_v1: None,
13899                append: None,
13900            },
13901        );
13902
13903        Ok(idx)
13904    }
13905
13906    /// Define a chunked dataset indexed by the single-chunk index: a fixed
13907    /// shape covered by exactly one whole chunk (`chunk_dims == dims`), its
13908    /// address — and, once written, size and filter mask if filtered — held
13909    /// directly in the layout message instead of any index structure
13910    /// (`H5Dsingle.c`). libhdf5 selects this index ahead of both Implicit and
13911    /// Fixed Array whenever the shape qualifies, filtered or not, early
13912    /// allocation or not (`H5D__layout_set_latest_indexing`).
13913    ///
13914    /// `early_alloc` mirrors [`create_implicit_dataset`](Self::create_implicit_dataset):
13915    /// when true, the chunk's storage is allocated and filled with the fill
13916    /// value immediately, matching an early-allocated unfiltered dataset
13917    /// whose one chunk covers the whole shape. When false, the chunk has no
13918    /// address until its first write, the same as an unfiltered Fixed Array
13919    /// element.
13920    pub fn create_single_chunk_dataset(
13921        &self,
13922        name: &str,
13923        datatype: DatatypeMessage,
13924        dims: &[u64],
13925        chunk_dims: &[u64],
13926        early_alloc: bool,
13927    ) -> IoResult<usize> {
13928        let create = self.begin_create(name)?;
13929        let name = create.name.as_str();
13930        validate_chunk_geometry(dims, dims, chunk_dims)?;
13931        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13932        let layout_version = self.chunk_layout_version(false, data_size);
13933
13934        let data_addr = if early_alloc {
13935            // Same reasoning as `create_implicit_dataset`: the fill-value
13936            // bytes are written now, not merely reserved, because the
13937            // file's end-of-file address is what libhdf5 checks a file's
13938            // completeness against.
13939            let addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
13940            self.handle.write_at(
13941                addr,
13942                &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
13943            )?;
13944            addr
13945        } else {
13946            UNDEF_ADDR
13947        };
13948
13949        let dataspace = DataspaceMessage {
13950            // Chunked storage always requires at least one dimension, so
13951            // this is never Scalar or Null.
13952            class: DataspaceClass::Simple,
13953            dims: dims.to_vec(),
13954            max_dims: Some(dims.to_vec()),
13955        };
13956
13957        let idx = self.push_dataset(
13958            &create,
13959            DatasetInfo {
13960                name: name.to_string(),
13961                datatype,
13962                committed_type: None,
13963                external: None,
13964                virtual_storage: None,
13965                dataspace,
13966                read_format: None,
13967                obj_header_addr: 0,
13968                data_addr: UNDEF_ADDR,
13969                data_size: 0,
13970                compact: None,
13971                attributes: Vec::new(),
13972                obj_header_written_addr: None,
13973                obj_header_blocks: Vec::new(),
13974                filter_pipeline: None,
13975                deleted: false,
13976                extent_dirty: false,
13977                header_dirty: false,
13978                nlink_written: 1,
13979                creation_seq: self.take_creation_seq(),
13980                track_attr_order: self.track_order.attrs,
13981                fill_value: None,
13982                fill_time: FILL_TIME_IFSET,
13983                layout_version,
13984                times: self.created_object_times(),
13985                chunked: None,
13986                btree_v2: None,
13987                fixed_array: None,
13988                implicit: None,
13989                single_chunk: Some(SingleChunkDatasetInfo {
13990                    chunk_dims: chunk_dims.to_vec(),
13991                    data_addr,
13992                    data_size,
13993                    nbytes: if early_alloc { data_size } else { 0 },
13994                    filter_mask: 0,
13995                    chunks_written: 0,
13996                    early_alloc,
13997                }),
13998                btree_v1: None,
13999                append: None,
14000            },
14001        );
14002
14003        Ok(idx)
14004    }
14005
14006    /// Define a fixed-shape compressed chunked dataset — of exactly one
14007    /// whole chunk — indexed by a *filtered* single-chunk index
14008    /// (`H5O_LAYOUT_CHUNK_SINGLE_INDEX_WITH_FILTER`, H5Dsingle.c). The
14009    /// chunk's stored size and filter mask are recorded inline in the
14010    /// layout message once the chunk is written.
14011    ///
14012    /// Like [`create_fixed_array_dataset_with_pipeline`](Self::create_fixed_array_dataset_with_pipeline),
14013    /// there is nothing to allocate ahead of that first write — a filtered
14014    /// chunk's stored length isn't known until it is compressed — so this
14015    /// dataset is always incrementally allocated regardless of the caller's
14016    /// requested allocation time.
14017    pub fn create_single_chunk_dataset_with_pipeline(
14018        &self,
14019        name: &str,
14020        datatype: DatatypeMessage,
14021        dims: &[u64],
14022        chunk_dims: &[u64],
14023        pipeline: FilterPipeline,
14024    ) -> IoResult<usize> {
14025        let create = self.begin_create(name)?;
14026        let name = create.name.as_str();
14027        validate_chunk_geometry(dims, dims, chunk_dims)?;
14028        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14029        let layout_version = self.chunk_layout_version(true, data_size);
14030
14031        let dataspace = DataspaceMessage {
14032            // Chunked storage always requires at least one dimension, so
14033            // this is never Scalar or Null.
14034            class: DataspaceClass::Simple,
14035            dims: dims.to_vec(),
14036            max_dims: Some(dims.to_vec()),
14037        };
14038
14039        let idx = self.push_dataset(
14040            &create,
14041            DatasetInfo {
14042                name: name.to_string(),
14043                datatype,
14044                committed_type: None,
14045                external: None,
14046                virtual_storage: None,
14047                dataspace,
14048                read_format: None,
14049                obj_header_addr: 0,
14050                data_addr: UNDEF_ADDR,
14051                data_size: 0,
14052                compact: None,
14053                attributes: Vec::new(),
14054                obj_header_written_addr: None,
14055                obj_header_blocks: Vec::new(),
14056                filter_pipeline: Some(pipeline),
14057                deleted: false,
14058                extent_dirty: false,
14059                header_dirty: false,
14060                nlink_written: 1,
14061                creation_seq: self.take_creation_seq(),
14062                track_attr_order: self.track_order.attrs,
14063                fill_value: None,
14064                fill_time: FILL_TIME_IFSET,
14065                layout_version,
14066                times: self.created_object_times(),
14067                chunked: None,
14068                btree_v2: None,
14069                fixed_array: None,
14070                implicit: None,
14071                single_chunk: Some(SingleChunkDatasetInfo {
14072                    chunk_dims: chunk_dims.to_vec(),
14073                    data_addr: UNDEF_ADDR,
14074                    data_size,
14075                    nbytes: 0,
14076                    filter_mask: 0,
14077                    chunks_written: 0,
14078                    early_alloc: false,
14079                }),
14080                btree_v1: None,
14081                append: None,
14082            },
14083        );
14084
14085        Ok(idx)
14086    }
14087
14088    /// Define a chunked dataset indexed by a version-1 B-tree — the classic
14089    /// chunk index, and the only one a version-0/1 superblock file can carry.
14090    ///
14091    /// The tree itself is not created here: libhdf5 leaves the layout
14092    /// message's address undefined until the first chunk is inserted
14093    /// (`H5D__btree_idx_create` runs on that insert), and so does this — the
14094    /// flush that bulk-loads the records is what puts a node in the file.
14095    ///
14096    /// Unlike the array indexes this one has no grid to size, so it takes any
14097    /// number of unlimited dimensions: a key *is* the chunk's position, and
14098    /// the tree is ordered by it.
14099    pub fn create_btree_v1_dataset(
14100        &self,
14101        name: &str,
14102        datatype: DatatypeMessage,
14103        dims: &[u64],
14104        max_dims: &[u64],
14105        chunk_dims: &[u64],
14106        pipeline: Option<FilterPipeline>,
14107    ) -> IoResult<usize> {
14108        let create = self.begin_create(name)?;
14109        let name = create.name.as_str();
14110        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14111        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14112        if chunk_bytes > u32::MAX as u64 {
14113            return Err(crate::io::IoError::InvalidState(format!(
14114                "a {chunk_bytes}-byte chunk does not fit the 32-bit size field of a \
14115                 version-1 B-tree chunk key"
14116            )));
14117        }
14118
14119        let dataspace = DataspaceMessage {
14120            // Chunked storage always requires at least one dimension, so
14121            // this is never Scalar or Null.
14122            class: DataspaceClass::Simple,
14123            dims: dims.to_vec(),
14124            max_dims: Some(max_dims.to_vec()),
14125        };
14126
14127        let idx = self.push_dataset(
14128            &create,
14129            DatasetInfo {
14130                name: name.to_string(),
14131                datatype,
14132                committed_type: None,
14133                external: None,
14134                virtual_storage: None,
14135                dataspace,
14136                read_format: None,
14137                obj_header_addr: 0,
14138                data_addr: UNDEF_ADDR,
14139                data_size: 0,
14140                compact: None,
14141                attributes: Vec::new(),
14142                obj_header_written_addr: None,
14143                obj_header_blocks: Vec::new(),
14144                filter_pipeline: pipeline,
14145                deleted: false,
14146                extent_dirty: false,
14147                header_dirty: false,
14148                nlink_written: 1,
14149                creation_seq: self.take_creation_seq(),
14150                track_attr_order: self.track_order.attrs,
14151                fill_value: None,
14152                fill_time: FILL_TIME_IFSET,
14153                // The version-3 data layout message this index encodes as:
14154                // `H5O_LAYOUT_VERSION_DEFAULT`, which is the floor of
14155                // `H5D__chunk_set_info`'s final MAX and the whole of it below
14156                // the version-4 gate — a bound whose row is lower does not
14157                // push the message down, it only keeps the v1.10 indexes out.
14158                layout_version: LAYOUT_VERSION_DEFAULT,
14159                times: self.created_object_times(),
14160                chunked: None,
14161                fixed_array: None,
14162                btree_v2: None,
14163                implicit: None,
14164                single_chunk: None,
14165                btree_v1: Some(BtreeV1DatasetInfo {
14166                    chunk_dims: chunk_dims.to_vec(),
14167                    max_dims: max_dims.to_vec(),
14168                    config: self.btree_v1_config(),
14169                    records: Vec::new(),
14170                    node_addrs: Vec::new(),
14171                    root_addr: UNDEF_ADDR,
14172                    chunks_written: 0,
14173                }),
14174                append: None,
14175            },
14176        );
14177
14178        Ok(idx)
14179    }
14180
14181    /// Define a chunked dataset indexed by a B-tree v2 (multiple unlimited dimensions).
14182    ///
14183    /// Returns the dataset index.
14184    pub fn create_btree_v2_dataset(
14185        &self,
14186        name: &str,
14187        datatype: DatatypeMessage,
14188        dims: &[u64],
14189        max_dims: &[u64],
14190        chunk_dims: &[u64],
14191    ) -> IoResult<usize> {
14192        self.create_btree_v2_dataset_inner(name, datatype, dims, max_dims, chunk_dims, None)
14193    }
14194
14195    /// Define a *filtered* chunked dataset indexed by a B-tree v2.
14196    ///
14197    /// The v2 B-tree counterpart of
14198    /// [`create_chunked_dataset_with_pipeline`](Self::create_chunked_dataset_with_pipeline):
14199    /// chunks are compressed on write and the index records each chunk's
14200    /// stored size and filter mask (record type 11), the same shape libhdf5
14201    /// builds when a multi-unlimited-dimension dataset has a filter pipeline
14202    /// (`H5Dbtree2.c`, `H5D_BT2_FILT`).
14203    pub fn create_btree_v2_dataset_with_pipeline(
14204        &self,
14205        name: &str,
14206        datatype: DatatypeMessage,
14207        dims: &[u64],
14208        max_dims: &[u64],
14209        chunk_dims: &[u64],
14210        pipeline: FilterPipeline,
14211    ) -> IoResult<usize> {
14212        self.create_btree_v2_dataset_inner(
14213            name,
14214            datatype,
14215            dims,
14216            max_dims,
14217            chunk_dims,
14218            Some(pipeline),
14219        )
14220    }
14221
14222    fn create_btree_v2_dataset_inner(
14223        &self,
14224        name: &str,
14225        datatype: DatatypeMessage,
14226        dims: &[u64],
14227        max_dims: &[u64],
14228        chunk_dims: &[u64],
14229        pipeline: Option<FilterPipeline>,
14230    ) -> IoResult<usize> {
14231        use crate::format::chunk_index::btree_v2::Bt2Header;
14232
14233        let create = self.begin_create(name)?;
14234        let name = create.name.as_str();
14235        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14236        let ndims = dims.len();
14237        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14238        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
14239
14240        // The filtered record's size field is as wide as libhdf5 will
14241        // recompute it — from the uncompressed chunk size under layout v4,
14242        // the fixed `sizeof_size` under layout v5 — exactly as the
14243        // extensible- and fixed-array filtered paths size theirs.
14244        let bt2_index = match pipeline {
14245            Some(_) => {
14246                let len = self.chunk_size_len_for(layout_version, chunk_bytes);
14247                Bt2ChunkIndex::new_filtered(ndims, len)
14248            }
14249            None => Bt2ChunkIndex::new_unfiltered(ndims),
14250        };
14251
14252        // The bulk loader spreads a level's records evenly over its nodes, one
14253        // separator between adjacent siblings, which needs room for a few
14254        // records per node. HDF5's rank limit of 32 leaves room for seven; a
14255        // wider rank than that has no valid geometry, so reject it here rather
14256        // than emit a tree no reader can walk.
14257        let record_size = bt2_index.record_size(&self.ctx) as usize;
14258        let node_size = bt2_index.node_size as usize;
14259        if node_size < 10 + 3 * record_size {
14260            return Err(crate::io::IoError::InvalidState(format!(
14261                "a {ndims}-dimension v2 B-tree record is {record_size} bytes, too wide \
14262                 for a {node_size}-byte node"
14263            )));
14264        }
14265
14266        // Only the header gets a home now: it names an empty tree, whose root
14267        // is undefined until the first flush bulk-loads the index into nodes.
14268        let hdr = if bt2_index.filtered {
14269            Bt2Header::new_for_filtered_chunks(&self.ctx, ndims, bt2_index.chunk_size_len)
14270        } else {
14271            Bt2Header::new_for_chunks(&self.ctx, ndims)
14272        };
14273        let hdr_encoded = hdr.encode(&self.ctx);
14274        let bt2_header_addr = self
14275            .allocator
14276            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14277        self.handle.write_at(bt2_header_addr, &hdr_encoded)?;
14278
14279        let dataspace = DataspaceMessage {
14280            // Chunked storage always requires at least one dimension, so
14281            // this is never Scalar or Null.
14282            class: DataspaceClass::Simple,
14283            dims: dims.to_vec(),
14284            max_dims: Some(max_dims.to_vec()),
14285        };
14286
14287        let idx = self.push_dataset(
14288            &create,
14289            DatasetInfo {
14290                name: name.to_string(),
14291                datatype,
14292                committed_type: None,
14293                external: None,
14294                virtual_storage: None,
14295                dataspace,
14296                read_format: None,
14297                obj_header_addr: 0,
14298                data_addr: UNDEF_ADDR,
14299                data_size: 0,
14300                compact: None,
14301                attributes: Vec::new(),
14302                obj_header_written_addr: None,
14303                obj_header_blocks: Vec::new(),
14304                filter_pipeline: pipeline,
14305                deleted: false,
14306                extent_dirty: false,
14307                header_dirty: false,
14308                nlink_written: 1,
14309                creation_seq: self.take_creation_seq(),
14310                track_attr_order: self.track_order.attrs,
14311                fill_value: None,
14312                fill_time: FILL_TIME_IFSET,
14313                layout_version,
14314                times: self.created_object_times(),
14315                chunked: None,
14316                fixed_array: None,
14317                implicit: None,
14318                single_chunk: None,
14319                btree_v1: None,
14320                btree_v2: Some(Bt2DatasetInfo {
14321                    chunk_dims: chunk_dims.to_vec(),
14322                    bt2_header_addr,
14323                    node_addrs: Vec::new(),
14324                    index: bt2_index,
14325                    chunks_written: 0,
14326                }),
14327                append: None,
14328            },
14329        );
14330
14331        Ok(idx)
14332    }
14333
14334    /// Create a chunked dataset with a custom filter pipeline.
14335    pub fn create_chunked_dataset_with_pipeline(
14336        &self,
14337        name: &str,
14338        datatype: DatatypeMessage,
14339        dims: &[u64],
14340        max_dims: &[u64],
14341        chunk_dims: &[u64],
14342        pipeline: FilterPipeline,
14343    ) -> IoResult<usize> {
14344        let create = self.begin_create(name)?;
14345        let name = create.name.as_str();
14346        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14347        ensure_at_most_one_unlimited(max_dims)?;
14348        let element_size = datatype.element_size() as u64;
14349        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * element_size;
14350        let layout_version = self.chunk_layout_version(true, chunk_bytes);
14351        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
14352
14353        let earray_params = EarrayParams::default_params();
14354        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
14355        let nsblk_addrs = compute_nsblk_addrs(
14356            earray_params.idx_blk_elmts,
14357            earray_params.data_blk_min_elmts,
14358            earray_params.sup_blk_min_data_ptrs,
14359            earray_params.max_nelmts_bits,
14360        )?;
14361
14362        let mut ea_header =
14363            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
14364        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
14365        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
14366        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
14367        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
14368        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
14369
14370        let hdr_encoded = ea_header.encode(&self.ctx);
14371        let ea_header_addr = self
14372            .allocator
14373            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14374
14375        let filt_iblk = FilteredIndexBlock::new(
14376            ea_header_addr,
14377            earray_params.idx_blk_elmts,
14378            ndblk_addrs,
14379            nsblk_addrs,
14380        );
14381        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
14382        let ea_iblk_addr = self
14383            .allocator
14384            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
14385
14386        ea_header.idx_blk_addr = ea_iblk_addr;
14387        let hdr_encoded = ea_header.encode(&self.ctx);
14388        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
14389        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
14390
14391        let dataspace = DataspaceMessage {
14392            // Chunked storage always requires at least one dimension, so
14393            // this is never Scalar or Null.
14394            class: DataspaceClass::Simple,
14395            dims: dims.to_vec(),
14396            max_dims: Some(max_dims.to_vec()),
14397        };
14398        let ea_iblk = ExtensibleArrayIndexBlock::new(
14399            ea_header_addr,
14400            earray_params.idx_blk_elmts,
14401            ndblk_addrs,
14402            nsblk_addrs,
14403        );
14404
14405        let idx = self.push_dataset(
14406            &create,
14407            DatasetInfo {
14408                name: name.to_string(),
14409                datatype,
14410                committed_type: None,
14411                external: None,
14412                virtual_storage: None,
14413                dataspace,
14414                read_format: None,
14415                obj_header_addr: 0,
14416                data_addr: UNDEF_ADDR,
14417                data_size: 0,
14418                compact: None,
14419                attributes: Vec::new(),
14420                obj_header_written_addr: None,
14421                obj_header_blocks: Vec::new(),
14422                filter_pipeline: Some(pipeline),
14423                deleted: false,
14424                extent_dirty: false,
14425                header_dirty: false,
14426                nlink_written: 1,
14427                creation_seq: self.take_creation_seq(),
14428                track_attr_order: self.track_order.attrs,
14429                fill_value: None,
14430                fill_time: FILL_TIME_IFSET,
14431                layout_version,
14432                times: self.created_object_times(),
14433                fixed_array: None,
14434                implicit: None,
14435                single_chunk: None,
14436                btree_v1: None,
14437                btree_v2: None,
14438                chunked: Some(ChunkedDatasetInfo {
14439                    chunk_dims: chunk_dims.to_vec(),
14440                    earray_params,
14441                    ea_header_addr,
14442                    ea_iblk_addr,
14443                    ea_header,
14444                    ea_iblk,
14445                    chunks_written: 0,
14446                    filt_iblk: Some(filt_iblk),
14447                    chunk_size_len,
14448                }),
14449                append: None,
14450            },
14451        );
14452        Ok(idx)
14453    }
14454
14455    /// Write a chunk to a fixed-array-indexed dataset.
14456    ///
14457    /// `chunk_coords` is the multidimensional chunk index (e.g., [row_chunk, col_chunk]).
14458    /// The uncompressed `data` must be exactly one chunk wide; the filter
14459    /// pipeline (if any) runs here before the bytes reach the index.
14460    pub fn write_chunk_fixed_array(
14461        &self,
14462        index: usize,
14463        chunk_coords: &[u64],
14464        data: &[u8],
14465    ) -> IoResult<()> {
14466        let ds = self.ds(index);
14467        let _op = ds.op.lock();
14468        self.write_chunk_fixed_array_inner(index, chunk_coords, data)
14469    }
14470
14471    /// [`Self::write_chunk_fixed_array`] body; the caller holds the dataset's
14472    /// op lock or the writer exclusively.
14473    pub(crate) fn write_chunk_fixed_array_inner(
14474        &self,
14475        index: usize,
14476        chunk_coords: &[u64],
14477        data: &[u8],
14478    ) -> IoResult<()> {
14479        // Read what we need under one brief slot guard, then compress
14480        // OUTSIDE the lock: `record_fixed_array_chunk` re-locks the same slot,
14481        // so the guard must be dropped before it (and before apply_filters).
14482        let ds = self.ds(index);
14483        let (chunk_bytes, pipeline) = {
14484            let m = ds.lock();
14485            let element_size = m.datatype.element_size() as u64;
14486            let fa = m.fixed_array.as_ref().ok_or_else(|| {
14487                crate::io::IoError::InvalidState("not a fixed-array dataset".into())
14488            })?;
14489            (
14490                fa.chunk_dims.iter().product::<u64>() * element_size,
14491                m.filter_pipeline.clone(),
14492            )
14493        };
14494
14495        if data.len() as u64 != chunk_bytes {
14496            return Err(crate::io::IoError::InvalidState(format!(
14497                "chunk data size mismatch: expected {} bytes, got {}",
14498                chunk_bytes,
14499                data.len()
14500            )));
14501        }
14502        let write_data;
14503        let data_to_write = if let Some(ref pipeline) = pipeline {
14504            write_data = filter::apply_filters(pipeline, data)?;
14505            &write_data[..]
14506        } else {
14507            data
14508        };
14509        // filter_mask = 0: the whole pipeline ran (or the dataset is
14510        // unfiltered), so no filter is skipped for this chunk.
14511        self.record_fixed_array_chunk(index, chunk_coords, data_to_write, 0)
14512    }
14513
14514    /// Write a pre-filtered chunk verbatim to a fixed-array dataset, recording
14515    /// the caller-supplied `filter_mask`.
14516    ///
14517    /// The bytes are stored exactly as given (no filter pipeline is run); this
14518    /// is the fixed-array half of the HDF5 "direct chunk write"
14519    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
14520    /// means filter *i* of the pipeline was **not** applied to this chunk and
14521    /// must be skipped on read; pass 0 when the full pipeline was applied
14522    /// upstream.
14523    ///
14524    /// Requires a filtered dataset — only the filtered FA element carries the
14525    /// size+mask slot.
14526    ///
14527    /// The caller holds the dataset's op lock or the writer exclusively.
14528    pub(crate) fn write_compressed_chunk_fixed_array_inner(
14529        &self,
14530        index: usize,
14531        chunk_coords: &[u64],
14532        data: &[u8],
14533        filter_mask: u32,
14534    ) -> IoResult<()> {
14535        if self.ds(index).lock().filter_pipeline.is_none() {
14536            return Err(crate::io::IoError::InvalidState(
14537                "write_compressed_chunk_fixed_array requires a filtered dataset \
14538                 (no slot for a compressed size or filter mask on an unfiltered \
14539                 chunk index)"
14540                    .into(),
14541            ));
14542        }
14543        self.record_fixed_array_chunk(index, chunk_coords, data, filter_mask)
14544    }
14545
14546    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
14547    /// filtered if the dataset is filtered, raw otherwise) into a fixed-array
14548    /// dataset's data block, recording the caller-supplied `filter_mask`.
14549    /// Shared by [`write_chunk_fixed_array`](Self::write_chunk_fixed_array)
14550    /// and [`write_compressed_chunk_fixed_array`](Self::write_compressed_chunk_fixed_array).
14551    fn record_fixed_array_chunk(
14552        &self,
14553        index: usize,
14554        chunk_coords: &[u64],
14555        final_bytes: &[u8],
14556        filter_mask: u32,
14557    ) -> IoResult<()> {
14558        // Hold one slot guard for the whole method; `self.allocator`/`self.handle`/
14559        // `self.ctx` below touch disjoint fields safe to use with the guard held.
14560        let ds = self.ds(index);
14561        let mut m = ds.lock();
14562        let is_filtered = m.filter_pipeline.is_some();
14563        let fa = m
14564            .fixed_array
14565            .as_ref()
14566            .ok_or_else(|| crate::io::IoError::InvalidState("not a fixed-array dataset".into()))?;
14567
14568        // Linear chunk index in the maximum-extent grid — the slot the fixed
14569        // array (sized from that grid at create) records the chunk under.
14570        let linear_idx = crate::io::chunk_grid::linear_index(
14571            &m.dataspace.dims,
14572            m.dataspace.max_dims.as_deref(),
14573            &fa.chunk_dims,
14574            chunk_coords,
14575        )?;
14576
14577        // Update the fixed array data block. The slot is read before the bytes
14578        // are placed so a rewrite can stay where it is (see `place_chunk`).
14579        let fa = m.fixed_array.as_mut().unwrap();
14580        let lidx = linear_idx as usize;
14581        if is_filtered {
14582            // Filtered FA: store address + stored size + filter mask. A
14583            // non-zero mask bit means "filter i was skipped for this chunk".
14584            let stored_size = final_bytes.len();
14585            // The stored size is encoded in the FA header's `chunk_size_len`-byte
14586            // field; libhdf5 errors if it does not fit (H5D_CHUNK_ENCODE_SIZE_CHECK)
14587            // rather than truncating silently. element_size = sizeof_addr +
14588            // chunk_size_len + 4 by construction.
14589            let chunk_size_len = (fa.fa_header.element_size as usize)
14590                .checked_sub(self.ctx.sizeof_addr as usize + 4)
14591                .ok_or_else(|| {
14592                    crate::io::IoError::InvalidState(
14593                        "filtered fixed-array element size is too small".into(),
14594                    )
14595                })?;
14596            if chunk_size_len < 8 && stored_size >= (1usize << (chunk_size_len * 8)) {
14597                return Err(crate::io::IoError::InvalidState(format!(
14598                    "compressed chunk size {stored_size} does not fit in the \
14599                     {chunk_size_len}-byte fixed-array chunk-size field"
14600                )));
14601            }
14602            if lidx < fa.fa_dblk.filtered_elements.len() {
14603                let old = &fa.fa_dblk.filtered_elements[lidx];
14604                let chunk_addr =
14605                    self.place_chunk(Some((old.address, old.chunk_size)), stored_size as u64);
14606                self.handle.write_at(chunk_addr, final_bytes)?;
14607                fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
14608                    address: chunk_addr,
14609                    chunk_size: stored_size as u64,
14610                    filter_mask,
14611                };
14612                fa.chunks_written += 1;
14613            } else {
14614                return Err(crate::io::IoError::InvalidState(format!(
14615                    "chunk index {} out of range (max {})",
14616                    linear_idx,
14617                    fa.fa_dblk.filtered_elements.len()
14618                )));
14619            }
14620        } else {
14621            // An unfiltered fixed array stores only addresses — there is no
14622            // slot for a filter mask, so a non-zero mask cannot be honored.
14623            if filter_mask != 0 {
14624                return Err(crate::io::IoError::InvalidState(
14625                    "filter_mask is non-zero but the dataset is unfiltered".into(),
14626                ));
14627            }
14628            if lidx < fa.fa_dblk.elements.len() {
14629                // Unfiltered: the stored size is fixed by the chunk shape, so
14630                // a rewrite always fits its old block.
14631                let old = fa.fa_dblk.elements[lidx];
14632                let len = final_bytes.len() as u64;
14633                let chunk_addr = self.place_chunk(Some((old, len)), len);
14634                self.handle.write_at(chunk_addr, final_bytes)?;
14635                fa.fa_dblk.elements[lidx] = chunk_addr;
14636                fa.chunks_written += 1;
14637            } else {
14638                return Err(crate::io::IoError::InvalidState(format!(
14639                    "chunk index {} out of range (max {})",
14640                    linear_idx,
14641                    fa.fa_dblk.elements.len()
14642                )));
14643            }
14644        }
14645
14646        Ok(())
14647    }
14648
14649    /// Write the one chunk of a single-chunk indexed dataset.
14650    ///
14651    /// `chunk_coords` is validated against the grid the same way every other
14652    /// coordinate-addressed index does (`ChunkGeometry::linear_index`), even
14653    /// though the grid holds exactly one slot — this is what rejects an
14654    /// out-of-range coordinate instead of silently writing to that slot.
14655    /// `data` is the chunk's unfiltered bytes; the dataset's filter pipeline
14656    /// runs here if it has one.
14657    ///
14658    /// The caller holds the dataset's op lock or the writer exclusively.
14659    pub(crate) fn write_chunk_single_chunk_inner(
14660        &self,
14661        index: usize,
14662        chunk_coords: &[u64],
14663        data: &[u8],
14664    ) -> IoResult<()> {
14665        let geo = self.chunk_geometry(index)?;
14666        geo.linear_index(chunk_coords)?;
14667        let chunk_bytes = geo.chunk_bytes();
14668        if data.len() as u64 != chunk_bytes {
14669            return Err(crate::io::IoError::InvalidState(format!(
14670                "chunk data size mismatch: expected {} bytes, got {}",
14671                chunk_bytes,
14672                data.len()
14673            )));
14674        }
14675        let pipeline = self.ds(index).lock().filter_pipeline.clone();
14676        let write_data;
14677        let data_to_write = if let Some(ref pipeline) = pipeline {
14678            write_data = filter::apply_filters(pipeline, data)?;
14679            &write_data[..]
14680        } else {
14681            data
14682        };
14683        // filter_mask = 0: the whole pipeline ran (or the dataset is
14684        // unfiltered), so no filter is skipped for this chunk.
14685        self.record_single_chunk(index, data_to_write, 0)
14686    }
14687
14688    /// Write a pre-filtered chunk verbatim to a single-chunk dataset,
14689    /// recording the caller-supplied `filter_mask`.
14690    ///
14691    /// The bytes are stored exactly as given (no filter pipeline is run); this
14692    /// is the single-chunk half of the HDF5 "direct chunk write"
14693    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
14694    /// means filter *i* of the pipeline was **not** applied to this chunk and
14695    /// must be skipped on read; pass 0 when the full pipeline was applied
14696    /// upstream.
14697    ///
14698    /// Requires a filtered dataset — only the filtered single-chunk layout
14699    /// carries a size+mask slot.
14700    ///
14701    /// The caller holds the dataset's op lock or the writer exclusively.
14702    pub(crate) fn write_compressed_chunk_single_chunk_inner(
14703        &self,
14704        index: usize,
14705        chunk_coords: &[u64],
14706        data: &[u8],
14707        filter_mask: u32,
14708    ) -> IoResult<()> {
14709        if self.ds(index).lock().filter_pipeline.is_none() {
14710            return Err(crate::io::IoError::InvalidState(
14711                "write_compressed_chunk_single_chunk requires a filtered dataset \
14712                 (no slot for a compressed size or filter mask on an unfiltered \
14713                 chunk index)"
14714                    .into(),
14715            ));
14716        }
14717        let geo = self.chunk_geometry(index)?;
14718        geo.linear_index(chunk_coords)?;
14719        self.record_single_chunk(index, data, filter_mask)
14720    }
14721
14722    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
14723    /// filtered if the dataset is filtered, raw otherwise) into a single-chunk
14724    /// dataset's layout message fields, recording the caller-supplied
14725    /// `filter_mask`. Shared by
14726    /// [`write_chunk_single_chunk_inner`](Self::write_chunk_single_chunk_inner)
14727    /// and
14728    /// [`write_compressed_chunk_single_chunk_inner`](Self::write_compressed_chunk_single_chunk_inner).
14729    ///
14730    /// Unlike the array indexes there is no per-chunk slot to look up — the
14731    /// dataset has exactly one chunk, and its address/size/mask live directly
14732    /// in the layout message (`H5Dsingle.c`) — so this only ever rewrites the
14733    /// one chunk in place, via [`place_chunk`](Self::place_chunk) the same as
14734    /// every other index's rewrite path.
14735    fn record_single_chunk(
14736        &self,
14737        index: usize,
14738        final_bytes: &[u8],
14739        filter_mask: u32,
14740    ) -> IoResult<()> {
14741        let ds = self.ds(index);
14742        let mut m = ds.lock();
14743        let is_filtered = m.filter_pipeline.is_some();
14744        if !is_filtered && filter_mask != 0 {
14745            return Err(crate::io::IoError::InvalidState(
14746                "filter_mask is non-zero but the dataset is unfiltered".into(),
14747            ));
14748        }
14749        let sc = m
14750            .single_chunk
14751            .as_ref()
14752            .ok_or_else(|| crate::io::IoError::InvalidState("not a single-chunk dataset".into()))?;
14753
14754        // A rewrite whose stored size is unchanged stays where it is (always
14755        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
14756        let old = if sc.data_addr == UNDEF_ADDR {
14757            None
14758        } else {
14759            Some((
14760                sc.data_addr,
14761                if is_filtered { sc.nbytes } else { sc.data_size },
14762            ))
14763        };
14764        let stored_size = final_bytes.len() as u64;
14765        let addr = self.place_chunk(old, stored_size);
14766        self.handle.write_at(addr, final_bytes)?;
14767
14768        let sc = m.single_chunk.as_mut().unwrap();
14769        sc.data_addr = addr;
14770        sc.nbytes = stored_size;
14771        sc.filter_mask = filter_mask;
14772        sc.chunks_written = 1;
14773        Ok(())
14774    }
14775
14776    /// Write a chunk to a B-tree v2 indexed dataset.
14777    ///
14778    /// `chunk_coords` is the scaled chunk coordinates (one per dimension).
14779    /// `data` is the chunk's unfiltered bytes; if the dataset has a filter
14780    /// pipeline it runs here and the index records the stored size and mask.
14781    ///
14782    /// Production writes call [`write_chunk_btree_v2_inner`](Self::write_chunk_btree_v2_inner)
14783    /// directly (they already hold the dataset's op lock); this self-locking
14784    /// form is kept as a direct entry point for this crate's own white-box
14785    /// tests.
14786    #[cfg(test)]
14787    pub fn write_chunk_btree_v2(
14788        &self,
14789        index: usize,
14790        chunk_coords: &[u64],
14791        data: &[u8],
14792    ) -> IoResult<()> {
14793        let ds = self.ds(index);
14794        let _op = ds.op.lock();
14795        self.write_chunk_btree_v2_inner(index, chunk_coords, data)
14796    }
14797
14798    /// [`Self::write_chunk_btree_v2`] body; the caller holds the dataset's op
14799    /// lock or the writer exclusively.
14800    pub(crate) fn write_chunk_btree_v2_inner(
14801        &self,
14802        index: usize,
14803        chunk_coords: &[u64],
14804        data: &[u8],
14805    ) -> IoResult<()> {
14806        // Read what the write needs under a brief guard, then compress OUTSIDE
14807        // the lock — filtering a chunk must not hold the dataset slot.
14808        let ds = self.ds(index);
14809        let (chunk_bytes, pipeline) = {
14810            let m = ds.lock();
14811            let element_size = m.datatype.element_size() as u64;
14812            let bt2 = m.btree_v2.as_ref().ok_or_else(|| {
14813                crate::io::IoError::InvalidState("not a B-tree v2 dataset".into())
14814            })?;
14815            (
14816                bt2.chunk_dims.iter().product::<u64>() * element_size,
14817                m.filter_pipeline.clone(),
14818            )
14819        };
14820
14821        if data.len() as u64 != chunk_bytes {
14822            return Err(crate::io::IoError::InvalidState(format!(
14823                "chunk data size mismatch: expected {} bytes, got {}",
14824                chunk_bytes,
14825                data.len()
14826            )));
14827        }
14828
14829        let filtered;
14830        let stored = match pipeline {
14831            Some(ref pl) => {
14832                filtered = filter::apply_filters(pl, data)?;
14833                &filtered[..]
14834            }
14835            None => data,
14836        };
14837
14838        // filter_mask = 0: the whole pipeline ran (or the dataset is
14839        // unfiltered), so no filter is skipped.
14840        self.record_btree_v2_chunk(index, chunk_coords, stored, 0)
14841    }
14842
14843    /// Write a pre-filtered chunk verbatim to a BT2-indexed dataset, recording
14844    /// the caller-supplied `filter_mask`.
14845    ///
14846    /// The v2-B-tree half of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
14847    /// The bytes are stored exactly as given; `filter_mask` bit *i* set means
14848    /// filter *i* of the pipeline was **not** applied and must be skipped on
14849    /// read. Requires a filtered dataset — only a type-11 record has a slot for
14850    /// a stored size and mask.
14851    ///
14852    /// The caller holds the dataset's op lock or the writer exclusively.
14853    pub(crate) fn write_compressed_chunk_btree_v2_inner(
14854        &self,
14855        index: usize,
14856        chunk_coords: &[u64],
14857        data: &[u8],
14858        filter_mask: u32,
14859    ) -> IoResult<()> {
14860        if self.ds(index).lock().filter_pipeline.is_none() {
14861            return Err(crate::io::IoError::InvalidState(
14862                "write_compressed_chunk_btree_v2 requires a filtered dataset (no \
14863                 slot for a compressed size or filter mask on an unfiltered chunk \
14864                 index)"
14865                    .into(),
14866            ));
14867        }
14868        self.record_btree_v2_chunk(index, chunk_coords, data, filter_mask)
14869    }
14870
14871    /// Place a chunk's already-final bytes (filtered if the dataset is
14872    /// filtered, raw otherwise) in the file and record them in the v2 B-tree,
14873    /// under the caller-supplied `filter_mask`.
14874    ///
14875    /// Shared by [`write_chunk_btree_v2`](Self::write_chunk_btree_v2) and
14876    /// [`write_compressed_chunk_btree_v2`](Self::write_compressed_chunk_btree_v2),
14877    /// so both reach the index through one placement rule.
14878    fn record_btree_v2_chunk(
14879        &self,
14880        index: usize,
14881        chunk_coords: &[u64],
14882        final_bytes: &[u8],
14883        filter_mask: u32,
14884    ) -> IoResult<()> {
14885        let stored_len = final_bytes.len() as u64;
14886        let ds = self.ds(index);
14887        let mut m = ds.lock();
14888        let element_size = m.datatype.element_size() as u64;
14889        let bt2 = m
14890            .btree_v2
14891            .as_ref()
14892            .ok_or_else(|| crate::io::IoError::InvalidState("not a B-tree v2 dataset".into()))?;
14893        let chunk_bytes = bt2.chunk_dims.iter().product::<u64>() * element_size;
14894        // A filtered record encodes the stored size in a `chunk_size_len`-byte
14895        // field that truncates silently. Reject a size that would not fit, as
14896        // the extensible-array path does — the compress path never exceeds it,
14897        // but a direct write with caller-supplied bytes can.
14898        if bt2.index.filtered {
14899            let chunk_size_len = bt2.index.chunk_size_len as usize;
14900            if chunk_size_len < 8 && stored_len >= (1u64 << (chunk_size_len * 8)) {
14901                return Err(crate::io::IoError::InvalidState(format!(
14902                    "filtered chunk size {stored_len} does not fit in the \
14903                     {chunk_size_len}-byte v2 B-tree chunk-size field"
14904                )));
14905            }
14906        }
14907        // Place the bytes: a rewrite whose stored size is unchanged stays
14908        // where it is (always so when unfiltered — the size is fixed by the
14909        // chunk shape), and one that no longer fits moves, releasing its old
14910        // block. See `place_chunk`.
14911        let old = if bt2.index.filtered {
14912            bt2.index
14913                .lookup_filtered(chunk_coords)
14914                .map(|r| (r.chunk_address, r.chunk_size))
14915        } else {
14916            bt2.index
14917                .lookup(chunk_coords)
14918                .map(|r| (r.chunk_address, chunk_bytes))
14919        };
14920        let chunk_addr = self.place_chunk(old, stored_len);
14921        self.handle.write_at(chunk_addr, final_bytes)?;
14922
14923        let bt2 = m.btree_v2.as_mut().unwrap();
14924        if bt2.index.filtered {
14925            bt2.index
14926                .insert_filtered(chunk_coords.to_vec(), chunk_addr, stored_len, filter_mask);
14927        } else {
14928            bt2.index.insert(chunk_coords.to_vec(), chunk_addr);
14929        }
14930        bt2.chunks_written += 1;
14931
14932        Ok(())
14933    }
14934
14935    /// Write multiple chunks in a batch, optionally compressing in parallel.
14936    ///
14937    /// `chunks` is a list of (chunk_idx, data) pairs for an EA-indexed dataset.
14938    pub fn write_chunks_batch(&self, ds_index: usize, chunks: &[(u64, &[u8])]) -> IoResult<()> {
14939        let ds = self.ds(ds_index);
14940        let _op = ds.op.lock();
14941        self.write_chunks_batch_inner(ds_index, chunks)
14942    }
14943
14944    /// [`Self::write_chunks_batch`] body; the caller holds the dataset's op
14945    /// lock or the writer exclusively.
14946    pub(crate) fn write_chunks_batch_inner(
14947        &self,
14948        ds_index: usize,
14949        chunks: &[(u64, &[u8])],
14950    ) -> IoResult<()> {
14951        #[cfg(feature = "parallel")]
14952        {
14953            // If filter pipeline is set, compress all chunks in parallel.
14954            // Clone the pipeline out under a brief slot guard so the parallel
14955            // compression below runs off the lock.
14956            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
14957            if let Some(ref pipeline) = pipeline {
14958                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
14959                // Propagate a filter error rather than storing raw bytes under a
14960                // filter_mask that claims the pipeline ran (see
14961                // apply_filters_parallel). Ok reaching here means every chunk
14962                // compressed fully, so filter_mask = 0 is truthful.
14963                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
14964                for ((idx, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
14965                    self.write_compressed_chunk_inner(ds_index, *idx, compressed_data, 0)?;
14966                }
14967                return Ok(());
14968            }
14969        }
14970        // Fallback: sequential
14971        for (idx, data) in chunks {
14972            self.write_chunk_inner(ds_index, *idx, data)?;
14973        }
14974        Ok(())
14975    }
14976
14977    /// Write multiple fixed-array chunks in a batch, compressing them in
14978    /// parallel when a filter pipeline is set and the `parallel` feature is on.
14979    ///
14980    /// The fixed-array analogue of [`write_chunks_batch`](Self::write_chunks_batch):
14981    /// chunks are addressed by grid coordinates rather than a linear index.
14982    /// `record_fixed_array_chunk` writes already-compressed bytes verbatim, so
14983    /// the parallel compressor is the only place a filter runs. Falls back to
14984    /// per-chunk [`write_chunk_fixed_array`](Self::write_chunk_fixed_array) when
14985    /// unfiltered or when `parallel` is off.
14986    ///
14987    /// The caller holds the dataset's op lock or the writer exclusively.
14988    pub(crate) fn write_chunks_fixed_array_batch_inner(
14989        &self,
14990        ds_index: usize,
14991        chunks: &[(&[u64], &[u8])],
14992    ) -> IoResult<()> {
14993        #[cfg(feature = "parallel")]
14994        {
14995            // Clone the pipeline out under a brief slot guard so the parallel
14996            // compression below runs off the lock.
14997            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
14998            if let Some(ref pipeline) = pipeline {
14999                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
15000                // Same single owner as the EA batch: apply_filters_parallel
15001                // propagates a filter error instead of storing raw bytes under a
15002                // filter_mask that claims the pipeline ran. Ok here means every
15003                // chunk compressed fully, so filter_mask = 0 is truthful.
15004                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15005                for ((coords, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15006                    self.record_fixed_array_chunk(ds_index, coords, compressed_data, 0)?;
15007                }
15008                return Ok(());
15009            }
15010        }
15011        // Fallback: sequential (write_chunk_fixed_array_inner compresses per
15012        // chunk).
15013        for (coords, data) in chunks {
15014            self.write_chunk_fixed_array_inner(ds_index, coords, data)?;
15015        }
15016        Ok(())
15017    }
15018
15019    /// Write a pre-filtered chunk verbatim to an EA-indexed dataset, recording
15020    /// the caller-supplied `filter_mask`.
15021    ///
15022    /// The bytes are stored exactly as given (no filter pipeline is run); this
15023    /// is the extensible-array half of the HDF5 "direct chunk write"
15024    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15025    /// means filter *i* of the pipeline was **not** applied to this chunk and
15026    /// must be skipped on read; pass 0 when the full pipeline was applied
15027    /// upstream.
15028    ///
15029    /// Requires a filtered dataset — only the filtered EA entry carries the
15030    /// size+mask slot. An unfiltered dataset has nowhere to record either.
15031    ///
15032    /// The caller holds the dataset's op lock or the writer exclusively.
15033    pub(crate) fn write_compressed_chunk_inner(
15034        &self,
15035        index: usize,
15036        chunk_idx: u64,
15037        compressed_data: &[u8],
15038        filter_mask: u32,
15039    ) -> IoResult<()> {
15040        if self.ds(index).lock().filter_pipeline.is_none() {
15041            return Err(crate::io::IoError::InvalidState(
15042                "write_compressed_chunk requires a filtered dataset (no slot for \
15043                 a compressed size or filter mask on an unfiltered chunk index)"
15044                    .into(),
15045            ));
15046        }
15047        self.record_ea_chunk(index, chunk_idx, compressed_data, filter_mask)
15048    }
15049
15050    /// Extend the dimensions of a chunked dataset.
15051    pub fn extend_dataset(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15052        let ds = self.ds(index);
15053        let _op = ds.op.lock();
15054        self.extend_dataset_inner(index, new_dims)
15055    }
15056
15057    /// [`Self::extend_dataset`] body; the caller holds the dataset's op lock
15058    /// or the writer exclusively.
15059    pub(crate) fn extend_dataset_inner(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15060        let ds = self.ds(index);
15061        let mut m = ds.lock();
15062        if !m.is_chunked() {
15063            return Err(crate::io::IoError::InvalidState(
15064                "can only extend chunked datasets".into(),
15065            ));
15066        }
15067        if new_dims.len() != m.dataspace.dims.len() {
15068            return Err(crate::io::IoError::InvalidState(format!(
15069                "extend_dataset rank mismatch: dataset has {} dimensions, got {}",
15070                m.dataspace.dims.len(),
15071                new_dims.len()
15072            )));
15073        }
15074        // The chunk index and append buffers assume the logical size only
15075        // grows; shrinking below already-written data desynchronizes them.
15076        for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15077            if new < cur {
15078                return Err(crate::io::IoError::InvalidState(format!(
15079                    "extend_dataset cannot shrink dimension {d} from {cur} to {new}"
15080                )));
15081            }
15082            // An absent maximum shape means the shape is fixed (libhdf5
15083            // defaults maxdims to dims at creation), so any growth exceeds it.
15084            match m.dataspace.max_dims {
15085                Some(ref max) if new > max[d] => {
15086                    return Err(crate::io::IoError::InvalidState(format!(
15087                        "extend_dataset dimension {d} ({new}) exceeds the maximum {}",
15088                        max[d]
15089                    )));
15090                }
15091                None if new > cur => {
15092                    return Err(crate::io::IoError::InvalidState(format!(
15093                        "extend_dataset dimension {d} ({new}) exceeds the maximum {cur}: \
15094                         a dataset without a stored maximum shape is fixed at its extent"
15095                    )));
15096                }
15097                _ => {}
15098            }
15099        }
15100        if m.dataspace.dims != new_dims {
15101            m.dataspace.dims = new_dims.to_vec();
15102            m.extent_dirty = true;
15103        }
15104        Ok(())
15105    }
15106
15107    /// Set the logical extent of a chunked dataset, growing **or shrinking**
15108    /// any dimension (unlike [`extend_dataset`](Self::extend_dataset), which
15109    /// only grows).
15110    ///
15111    /// A shrink prunes the stored chunks the way libhdf5's
15112    /// `H5D__chunk_prune_by_extent` (H5Dchunk.c) does: a chunk entirely
15113    /// beyond the new extent leaves the chunk index and its block is freed
15114    /// for reuse (kept under SWMR, where a live reader may still hold its
15115    /// address — the rule `H5Dearray.c` applies in `idx_remove`), and a
15116    /// chunk the new extent cuts through has its out-of-extent region
15117    /// overwritten with the fill value, so growing the extent back exposes
15118    /// fill values rather than the stale data.
15119    pub fn set_dataset_extent(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15120        let ds = self.ds(index);
15121        let _op = ds.op.lock();
15122        let old_dims = {
15123            let m = ds.lock();
15124            if !m.is_chunked() {
15125                return Err(crate::io::IoError::InvalidState(
15126                    "can only set the extent of chunked datasets".into(),
15127                ));
15128            }
15129            if new_dims.len() != m.dataspace.dims.len() {
15130                return Err(crate::io::IoError::InvalidState(format!(
15131                    "set_extent rank mismatch: dataset has {} dimensions, got {}",
15132                    m.dataspace.dims.len(),
15133                    new_dims.len()
15134                )));
15135            }
15136            // A shrink can cut into buffered rows, whose recorded base would
15137            // then point past the extent; refuse rather than reconcile.
15138            if m.append.is_some() {
15139                return Err(crate::io::IoError::InvalidState(
15140                    "set_extent cannot run while the dataset has buffered appends; \
15141                     flush them first"
15142                        .into(),
15143                ));
15144            }
15145            // An absent maximum shape means the shape is fixed (libhdf5
15146            // defaults maxdims to dims at creation), so growth is bounded by
15147            // the extent.
15148            match m.dataspace.max_dims {
15149                Some(ref max) => {
15150                    for (d, (&new, &mx)) in new_dims.iter().zip(max).enumerate() {
15151                        if new > mx {
15152                            return Err(crate::io::IoError::InvalidState(format!(
15153                                "set_extent dimension {d} ({new}) exceeds the maximum {mx}"
15154                            )));
15155                        }
15156                    }
15157                }
15158                None => {
15159                    for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15160                        if new > cur {
15161                            return Err(crate::io::IoError::InvalidState(format!(
15162                                "set_extent dimension {d} ({new}) exceeds the maximum {cur}: \
15163                                 a dataset without a stored maximum shape is fixed at its extent"
15164                            )));
15165                        }
15166                    }
15167                }
15168            }
15169            m.dataspace.dims.clone()
15170        };
15171        // A shrink strands chunks; prune them (and refill the straddlers)
15172        // *before* the dims update — chunk addressing uses the
15173        // maximum-extent grid, which the update does not change, and the
15174        // helpers re-lock the slot themselves.
15175        if new_dims.iter().zip(&old_dims).any(|(&n, &o)| n < o) {
15176            self.prune_chunks_beyond(index, new_dims)?;
15177        }
15178        let mut m = ds.lock();
15179        if m.dataspace.dims != new_dims {
15180            m.dataspace.dims = new_dims.to_vec();
15181            m.extent_dirty = true;
15182        }
15183        Ok(())
15184    }
15185
15186    /// Remove and refill the chunks a shrink to `new_dims` strands — the
15187    /// libhdf5 `H5D__chunk_prune_by_extent` behavior. A chunk entirely
15188    /// beyond the new extent leaves the index and its block is freed (kept
15189    /// under SWMR, where a live reader may still hold its address); a chunk
15190    /// the extent cuts through gets its out-of-extent region refilled with
15191    /// the fill value, so a later regrow reads fill, not stale elements.
15192    ///
15193    /// Runs *before* the dims update: the index grid chunks are addressed in
15194    /// comes from the maximum extent, which a shrink never changes, so every
15195    /// stored entry still resolves. The caller holds the dataset's op lock.
15196    fn prune_chunks_beyond(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15197        let geo = self.chunk_geometry(index)?;
15198        // A vlen dataset's elements are global-heap IDs: the pruned chunks
15199        // still reference live heap objects, so the walkers read each dead
15200        // chunk's bytes before freeing its block and the heap objects are
15201        // released here — otherwise every shrink strands its strings in the
15202        // file. `release_vlen_references` is a SWMR no-op, so the reads are
15203        // skipped under SWMR too.
15204        let collect_refs = !self.swmr_active && {
15205            let ds = self.ds(index);
15206            let m = ds.lock();
15207            matches!(
15208                m.datatype,
15209                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
15210            )
15211        };
15212        let (straddlers, dead_refs) = match geo.kind {
15213            ChunkIndexKind::ExtensibleArray => {
15214                self.prune_ea_chunks(index, &geo, new_dims, collect_refs)?
15215            }
15216            ChunkIndexKind::FixedArray => {
15217                self.prune_fa_chunks(index, &geo, new_dims, collect_refs)?
15218            }
15219            ChunkIndexKind::BtreeV2 => {
15220                self.prune_bt2_chunks(index, &geo, new_dims, collect_refs)?
15221            }
15222            // Removing a chunk from the implicit index is
15223            // `H5D__none_idx_remove`: a no-op, because the chunk's space is
15224            // the dataset's space and stays allocated either way. Only the
15225            // straddlers matter, and they are refilled by the caller.
15226            ChunkIndexKind::Implicit => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15227            // A single-chunk index has no per-chunk remove either — its one
15228            // chunk's address lives in the layout message, not an index
15229            // structure, and stays exactly where it is; a shrink only ever
15230            // straddles that one chunk (`H5D__single_idx_remove` is likewise
15231            // a no-op).
15232            ChunkIndexKind::SingleChunk => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15233            ChunkIndexKind::BtreeV1 => {
15234                self.prune_btree_v1_chunks(index, &geo, new_dims, collect_refs)?
15235            }
15236        };
15237        if !dead_refs.is_empty() {
15238            self.release_vlen_references(&dead_refs)?;
15239        }
15240        // Whole-chunk read-modify-write per straddler: an unfiltered chunk
15241        // rewrites in place, a filtered one re-places through `place_chunk`.
15242        let chunk_bytes = geo.chunk_bytes() as usize;
15243        for coords in straddlers {
15244            let Some(mut data) = self.read_chunk_at_coords(index, &coords)? else {
15245                continue;
15246            };
15247            let fill = self.new_chunk_buffer(index, chunk_bytes);
15248            let replaced = refill_chunk_beyond_extent(
15249                &mut data,
15250                &fill,
15251                &coords,
15252                &geo.chunk_dims,
15253                new_dims,
15254                geo.element_size as usize,
15255            );
15256            // Release before the write-back: a filtered straddler re-places
15257            // its block, and freed heap space must be visible to that
15258            // allocation (free-before-alloc, as everywhere else).
15259            if collect_refs && !replaced.is_empty() {
15260                self.release_vlen_references(&replaced)?;
15261            }
15262            self.write_chunk_at_coords(index, &coords, &data)?;
15263        }
15264        Ok(())
15265    }
15266
15267    /// Extensible-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15268    /// walk every slot the array has ever set, free and clear the entries of
15269    /// chunks entirely beyond `new_dims`, and return the grid coordinates of
15270    /// the chunks that straddle it, plus — when `collect_refs` — the dead
15271    /// chunks' element bytes so the caller can release their heap objects.
15272    fn prune_ea_chunks(
15273        &self,
15274        index: usize,
15275        geo: &ChunkGeometry,
15276        new_dims: &[u64],
15277        collect_refs: bool,
15278    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15279        let ds = self.ds(index);
15280        // One slot guard for the whole walk, the `record_ea_chunk` pattern:
15281        // `self.handle`/`self.allocator`/`self.ctx` are disjoint fields.
15282        let mut m = ds.lock();
15283        let is_filtered = m.filter_pipeline.is_some();
15284        let pipeline = m.filter_pipeline.clone();
15285        let chunk_bytes = geo.chunk_bytes();
15286        let (ea_geo, max_nelmts_bits, chunk_size_len, max_idx) = {
15287            let c = m.chunked.as_ref().unwrap();
15288            let p = &c.earray_params;
15289            (
15290                EaGeometry::new(
15291                    p.idx_blk_elmts,
15292                    p.data_blk_min_elmts,
15293                    p.sup_blk_min_data_ptrs,
15294                    p.max_nelmts_bits,
15295                    p.max_dblk_page_nelmts_bits,
15296                )?,
15297                p.max_nelmts_bits,
15298                c.chunk_size_len,
15299                c.ea_header.max_idx_set,
15300            )
15301        };
15302
15303        let mut straddlers = Vec::new();
15304        let mut dead_refs = Vec::new();
15305
15306        // The decoded data block the walk is currently inside, written back
15307        // when the walk leaves it (or ends) having cleared an entry.
15308        enum Dblk {
15309            Unfiltered(ExtensibleArrayDataBlock),
15310            Filtered(FilteredDataBlock),
15311        }
15312        let mut cache: Option<(u64, Dblk, bool)> = None;
15313        let flush = |cache: &mut Option<(u64, Dblk, bool)>| -> IoResult<()> {
15314            if let Some((addr, blk, dirty)) = cache.take() {
15315                if dirty {
15316                    let enc = match &blk {
15317                        Dblk::Unfiltered(d) => d.encode(&self.ctx, max_nelmts_bits),
15318                        Dblk::Filtered(d) => d.encode(&self.ctx, max_nelmts_bits, chunk_size_len),
15319                    };
15320                    self.handle.write_at(addr, &enc)?;
15321                }
15322            }
15323            Ok(())
15324        };
15325        // Consecutive slots resolve through the same super block, so keep
15326        // the last decode. Super blocks are only read here — clearing a
15327        // data-block element never moves the block — so it never dirties.
15328        let mut sblk_cache: Option<(usize, ExtensibleArraySuperBlock)> = None;
15329
15330        let mut slot = 0u64;
15331        while slot < max_idx {
15332            let coords = crate::io::chunk_grid::coords_of(
15333                &geo.dims,
15334                geo.max_dims.as_deref(),
15335                &geo.chunk_dims,
15336                slot,
15337            )?;
15338            if !chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
15339                if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15340                    straddlers.push(coords);
15341                }
15342                slot += 1;
15343                continue;
15344            }
15345            match ea_geo.locate(slot)? {
15346                EaLoc::Index { elem } => {
15347                    let c = m.chunked.as_mut().unwrap();
15348                    if is_filtered {
15349                        let fiblk = c.filt_iblk.as_mut().unwrap();
15350                        let e = fiblk.elements[elem];
15351                        if e.addr != UNDEF_ADDR {
15352                            if collect_refs {
15353                                if let Some(bytes) = self.read_chunk_block(
15354                                    pipeline.as_ref(),
15355                                    e.addr,
15356                                    e.nbytes,
15357                                    e.filter_mask,
15358                                )? {
15359                                    dead_refs.extend_from_slice(&bytes);
15360                                }
15361                            }
15362                            if !self.swmr_active {
15363                                self.allocator
15364                                    .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
15365                            }
15366                            fiblk.elements[elem] = FilteredChunkEntry {
15367                                addr: UNDEF_ADDR,
15368                                nbytes: 0,
15369                                filter_mask: 0,
15370                            };
15371                        }
15372                    } else {
15373                        let a = c.ea_iblk.elements[elem];
15374                        if a != UNDEF_ADDR {
15375                            if collect_refs {
15376                                if let Some(bytes) =
15377                                    self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
15378                                {
15379                                    dead_refs.extend_from_slice(&bytes);
15380                                }
15381                            }
15382                            if !self.swmr_active {
15383                                self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
15384                            }
15385                            c.ea_iblk.elements[elem] = UNDEF_ADDR;
15386                        }
15387                    }
15388                    slot += 1;
15389                }
15390                EaLoc::Dblk(l) => {
15391                    if l.paged {
15392                        return Err(crate::io::IoError::InvalidState(format!(
15393                            "chunk index {slot} lives in a paged extensible-array \
15394                             data block, which is not yet supported"
15395                        )));
15396                    }
15397                    let dblk_start = slot - l.offset_in_dblk;
15398                    let dblk_end = dblk_start + l.dblk_nelmts;
15399                    // Resolve the data block's address; an undefined super or
15400                    // data block means nothing in its whole element range was
15401                    // ever written, so the walk skips the range.
15402                    let dblk_addr = {
15403                        let c = m.chunked.as_ref().unwrap();
15404                        match l.path {
15405                            EaDblkPath::Direct { idx } => {
15406                                if is_filtered {
15407                                    c.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
15408                                } else {
15409                                    c.ea_iblk.dblk_addrs[idx]
15410                                }
15411                            }
15412                            EaDblkPath::ViaSblk {
15413                                sblk_off,
15414                                local_dblk,
15415                                ndblks_in_sblk,
15416                                ..
15417                            } => {
15418                                let sblk_addr = if is_filtered {
15419                                    c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
15420                                } else {
15421                                    c.ea_iblk.sblk_addrs[sblk_off]
15422                                };
15423                                if sblk_addr == UNDEF_ADDR {
15424                                    UNDEF_ADDR
15425                                } else {
15426                                    if sblk_cache.as_ref().map(|&(o, _)| o) != Some(sblk_off) {
15427                                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
15428                                        let sb = ExtensibleArraySuperBlock::decode(
15429                                            &buf,
15430                                            &self.ctx,
15431                                            max_nelmts_bits,
15432                                            ndblks_in_sblk,
15433                                            0,
15434                                        )?;
15435                                        sblk_cache = Some((sblk_off, sb));
15436                                    }
15437                                    sblk_cache.as_ref().unwrap().1.dblk_addrs[local_dblk]
15438                                }
15439                            }
15440                        }
15441                    };
15442                    if dblk_addr == UNDEF_ADDR {
15443                        slot = dblk_end;
15444                        continue;
15445                    }
15446                    if cache.as_ref().map(|&(a, _, _)| a) != Some(dblk_addr) {
15447                        flush(&mut cache)?;
15448                        let buf = self.handle.read_at_most(dblk_addr, 65536)?;
15449                        let blk = if is_filtered {
15450                            Dblk::Filtered(FilteredDataBlock::decode(
15451                                &buf,
15452                                &self.ctx,
15453                                max_nelmts_bits,
15454                                l.dblk_nelmts as usize,
15455                                chunk_size_len,
15456                            )?)
15457                        } else {
15458                            Dblk::Unfiltered(ExtensibleArrayDataBlock::decode(
15459                                &buf,
15460                                &self.ctx,
15461                                max_nelmts_bits,
15462                                l.dblk_nelmts as usize,
15463                            )?)
15464                        };
15465                        cache = Some((dblk_addr, blk, false));
15466                    }
15467                    let (_, blk, dirty) = cache.as_mut().unwrap();
15468                    match blk {
15469                        Dblk::Filtered(d) => {
15470                            let e = d.elements[l.offset_in_dblk as usize];
15471                            if e.addr != UNDEF_ADDR {
15472                                if collect_refs {
15473                                    if let Some(bytes) = self.read_chunk_block(
15474                                        pipeline.as_ref(),
15475                                        e.addr,
15476                                        e.nbytes,
15477                                        e.filter_mask,
15478                                    )? {
15479                                        dead_refs.extend_from_slice(&bytes);
15480                                    }
15481                                }
15482                                if !self.swmr_active {
15483                                    self.allocator
15484                                        .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
15485                                }
15486                                d.elements[l.offset_in_dblk as usize] = FilteredChunkEntry {
15487                                    addr: UNDEF_ADDR,
15488                                    nbytes: 0,
15489                                    filter_mask: 0,
15490                                };
15491                                *dirty = true;
15492                            }
15493                        }
15494                        Dblk::Unfiltered(d) => {
15495                            let a = d.elements[l.offset_in_dblk as usize];
15496                            if a != UNDEF_ADDR {
15497                                if collect_refs {
15498                                    if let Some(bytes) =
15499                                        self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
15500                                    {
15501                                        dead_refs.extend_from_slice(&bytes);
15502                                    }
15503                                }
15504                                if !self.swmr_active {
15505                                    self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
15506                                }
15507                                d.elements[l.offset_in_dblk as usize] = UNDEF_ADDR;
15508                                *dirty = true;
15509                            }
15510                        }
15511                    }
15512                    slot += 1;
15513                }
15514            }
15515        }
15516        flush(&mut cache)?;
15517        Ok((straddlers, dead_refs))
15518    }
15519
15520    /// Fixed-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15521    /// the whole element array is in memory and flushed at close, so
15522    /// clearing an entry is pure bookkeeping.
15523    fn prune_fa_chunks(
15524        &self,
15525        index: usize,
15526        geo: &ChunkGeometry,
15527        new_dims: &[u64],
15528        collect_refs: bool,
15529    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15530        let ds = self.ds(index);
15531        let mut m = ds.lock();
15532        let is_filtered = m.filter_pipeline.is_some();
15533        let pipeline = m.filter_pipeline.clone();
15534        let chunk_bytes = geo.chunk_bytes();
15535        let mut straddlers = Vec::new();
15536        let mut dead_refs = Vec::new();
15537        let fa = m.fixed_array.as_mut().unwrap();
15538        let nslots = if is_filtered {
15539            fa.fa_dblk.filtered_elements.len()
15540        } else {
15541            fa.fa_dblk.elements.len()
15542        };
15543        for lidx in 0..nslots {
15544            let (addr, stored, mask) = if is_filtered {
15545                let e = &fa.fa_dblk.filtered_elements[lidx];
15546                (e.address, e.chunk_size, e.filter_mask)
15547            } else {
15548                (fa.fa_dblk.elements[lidx], chunk_bytes, 0)
15549            };
15550            if addr == UNDEF_ADDR {
15551                continue;
15552            }
15553            let coords = crate::io::chunk_grid::coords_of(
15554                &geo.dims,
15555                geo.max_dims.as_deref(),
15556                &geo.chunk_dims,
15557                lidx as u64,
15558            )?;
15559            if chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
15560                if collect_refs {
15561                    if let Some(bytes) =
15562                        self.read_chunk_block(pipeline.as_ref(), addr, stored, mask)?
15563                    {
15564                        dead_refs.extend_from_slice(&bytes);
15565                    }
15566                }
15567                if !self.swmr_active {
15568                    self.allocator.free(addr, stored, FreeSpaceClass::RawData);
15569                }
15570                if is_filtered {
15571                    fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
15572                        address: UNDEF_ADDR,
15573                        chunk_size: 0,
15574                        filter_mask: 0,
15575                    };
15576                } else {
15577                    fa.fa_dblk.elements[lidx] = UNDEF_ADDR;
15578                }
15579            } else if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15580                straddlers.push(coords);
15581            }
15582        }
15583        Ok((straddlers, dead_refs))
15584    }
15585
15586    /// Implicit half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15587    /// the grid coordinates of the chunks a shrink to `new_dims` cuts
15588    /// through. Nothing is freed or cleared — this index has no per-chunk
15589    /// state to clear and no per-chunk block to free — so the chunks wholly
15590    /// beyond the extent keep their bytes, exactly as `H5D__none_idx_remove`
15591    /// leaves them. That also means their elements stay reachable, so a
15592    /// variable-length dataset's heap objects must *not* be released here.
15593    fn implicit_straddlers(
15594        &self,
15595        geo: &ChunkGeometry,
15596        new_dims: &[u64],
15597    ) -> IoResult<Vec<Vec<u64>>> {
15598        let mut nchunks: u64 = 1;
15599        for g in
15600            crate::io::chunk_grid::index_grid(&geo.dims, geo.max_dims.as_deref(), &geo.chunk_dims)?
15601        {
15602            nchunks = nchunks.checked_mul(g).ok_or_else(|| {
15603                crate::io::IoError::InvalidState("chunk count overflows u64".into())
15604            })?;
15605        }
15606        let mut straddlers = Vec::new();
15607        for lidx in 0..nchunks {
15608            let coords = crate::io::chunk_grid::coords_of(
15609                &geo.dims,
15610                geo.max_dims.as_deref(),
15611                &geo.chunk_dims,
15612                lidx,
15613            )?;
15614            if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15615                straddlers.push(coords);
15616            }
15617        }
15618        Ok(straddlers)
15619    }
15620
15621    /// V2-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15622    /// drop the records of chunks beyond the extent — the next flush
15623    /// re-serializes the smaller tree over the node pool and releases the
15624    /// surplus node blocks.
15625    fn prune_bt2_chunks(
15626        &self,
15627        index: usize,
15628        geo: &ChunkGeometry,
15629        new_dims: &[u64],
15630        collect_refs: bool,
15631    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15632        let ds = self.ds(index);
15633        let mut m = ds.lock();
15634        let pipeline = m.filter_pipeline.clone();
15635        let chunk_bytes = geo.chunk_bytes();
15636        let swmr = self.swmr_active;
15637        let mut straddlers = Vec::new();
15638        let mut dead_refs = Vec::new();
15639        let bt2 = m.btree_v2.as_mut().unwrap();
15640        if bt2.index.filtered {
15641            let records = std::mem::take(&mut bt2.index.filtered_records);
15642            let mut kept = Vec::with_capacity(records.len());
15643            for r in records {
15644                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15645                    if collect_refs {
15646                        if let Some(bytes) = self.read_chunk_block(
15647                            pipeline.as_ref(),
15648                            r.chunk_address,
15649                            r.chunk_size,
15650                            r.filter_mask,
15651                        )? {
15652                            dead_refs.extend_from_slice(&bytes);
15653                        }
15654                    }
15655                    if !swmr {
15656                        self.allocator
15657                            .free(r.chunk_address, r.chunk_size, FreeSpaceClass::RawData);
15658                    }
15659                } else {
15660                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15661                        straddlers.push(r.scaled_offsets.clone());
15662                    }
15663                    kept.push(r);
15664                }
15665            }
15666            bt2.index.filtered_records = kept;
15667        } else {
15668            let records = std::mem::take(&mut bt2.index.records);
15669            let mut kept = Vec::with_capacity(records.len());
15670            for r in records {
15671                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15672                    if collect_refs {
15673                        if let Some(bytes) = self.read_chunk_block(
15674                            pipeline.as_ref(),
15675                            r.chunk_address,
15676                            chunk_bytes,
15677                            0,
15678                        )? {
15679                            dead_refs.extend_from_slice(&bytes);
15680                        }
15681                    }
15682                    if !swmr {
15683                        self.allocator
15684                            .free(r.chunk_address, chunk_bytes, FreeSpaceClass::RawData);
15685                    }
15686                } else {
15687                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15688                        straddlers.push(r.scaled_offsets.clone());
15689                    }
15690                    kept.push(r);
15691                }
15692            }
15693            bt2.index.records = kept;
15694        }
15695        Ok((straddlers, dead_refs))
15696    }
15697
15698    /// Version-1-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15699    /// drop the records of chunks beyond the extent — the next flush
15700    /// re-serializes the smaller tree over the node pool and releases the
15701    /// surplus node blocks.
15702    fn prune_btree_v1_chunks(
15703        &self,
15704        index: usize,
15705        geo: &ChunkGeometry,
15706        new_dims: &[u64],
15707        collect_refs: bool,
15708    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15709        let ds = self.ds(index);
15710        let mut m = ds.lock();
15711        let pipeline = m.filter_pipeline.clone();
15712        let swmr = self.swmr_active;
15713        let mut straddlers = Vec::new();
15714        let mut dead_refs = Vec::new();
15715        let bt1 = m.btree_v1.as_mut().unwrap();
15716        let records = std::mem::take(&mut bt1.records);
15717        let mut kept = Vec::with_capacity(records.len());
15718        for r in records {
15719            if chunk_outside_extent(&r.scaled, &geo.chunk_dims, new_dims) {
15720                if collect_refs {
15721                    if let Some(bytes) = self.read_chunk_block(
15722                        pipeline.as_ref(),
15723                        r.address,
15724                        r.nbytes as u64,
15725                        r.filter_mask,
15726                    )? {
15727                        dead_refs.extend_from_slice(&bytes);
15728                    }
15729                }
15730                if !swmr {
15731                    self.allocator
15732                        .free(r.address, r.nbytes as u64, FreeSpaceClass::RawData);
15733                }
15734            } else {
15735                if chunk_straddles_extent(&r.scaled, &geo.chunk_dims, new_dims) {
15736                    straddlers.push(r.scaled.clone());
15737                }
15738                kept.push(r);
15739            }
15740        }
15741        m.btree_v1.as_mut().unwrap().records = kept;
15742        Ok((straddlers, dead_refs))
15743    }
15744
15745    /// Flush a chunked dataset's index structures to disk (durable).
15746    ///
15747    /// Writes the index blocks and issues an `fdatasync` so the data is
15748    /// durable — the guarantee SWMR readers and standalone callers rely on.
15749    pub fn flush_dataset(&self, index: usize) -> IoResult<()> {
15750        let ds = self.ds(index);
15751        let _op = ds.op.lock();
15752        self.flush_dataset_synced(index, true)
15753    }
15754
15755    /// Flush a chunked dataset's index structures, syncing only if `sync`.
15756    ///
15757    /// `finalize` threads its own durability choice here so that a
15758    /// [`close_no_sync`](Self::close_no_sync) skips this per-dataset
15759    /// `sync_data` too — otherwise gating only the final `sync_all` would
15760    /// leave one `fdatasync` per indexed dataset and defeat the fast close.
15761    fn flush_dataset_synced(&self, index: usize, sync: bool) -> IoResult<()> {
15762        // Hold one slot guard for the whole method; `self.handle`/`self.ctx`/
15763        // `self.allocator` below touch disjoint fields.
15764        let ds = self.ds(index);
15765        let mut m = ds.lock();
15766
15767        // EA-indexed dataset
15768        if let Some(ref chunked) = m.chunked {
15769            if let Some(ref fiblk) = chunked.filt_iblk {
15770                // Filtered EA
15771                let iblk_encoded = fiblk.encode(&self.ctx, chunked.chunk_size_len);
15772                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
15773            } else {
15774                // Unfiltered EA
15775                let iblk_encoded = chunked.ea_iblk.encode(&self.ctx);
15776                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
15777            }
15778            let hdr_encoded = chunked.ea_header.encode(&self.ctx);
15779            self.handle.write_at(chunked.ea_header_addr, &hdr_encoded)?;
15780            if sync {
15781                self.handle.sync_data()?;
15782            }
15783            return Ok(());
15784        }
15785
15786        // Fixed-array-indexed dataset
15787        if let Some(ref fa) = m.fixed_array {
15788            let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa.fa_header, &fa.fa_dblk);
15789            self.handle.write_at(fa.fa_dblk_addr, &dblk_encoded)?;
15790            let hdr_encoded = fa.fa_header.encode(&self.ctx);
15791            self.handle.write_at(fa.fa_header_addr, &hdr_encoded)?;
15792            if sync {
15793                self.handle.sync_data()?;
15794            }
15795            return Ok(());
15796        }
15797
15798        // BT2-indexed dataset
15799        if let Some(ref bt2) = m.btree_v2 {
15800            // Bulk-load the index into fixed-size nodes and lay them over the
15801            // dataset's block pool. Because every node is the same size, the
15802            // blocks already on disk are reused in place and only the shortfall
15803            // is allocated — the pool is the single owner of these addresses,
15804            // so no flush leaves a block behind. The addresses a reader already
15805            // holds stay valid, which is also what SWMR needs.
15806            let tree = bt2.index.build_tree(&self.ctx);
15807            let mut node_addrs = bt2.node_addrs.clone();
15808            while node_addrs.len() < tree.nodes.len() {
15809                node_addrs.push(
15810                    self.allocator
15811                        .allocate(tree.node_size as u64, FreeSpaceClass::Metadata),
15812                );
15813            }
15814            // A tree with fewer nodes than last flush releases the surplus
15815            // rather than leaving it recorded and unreachable, so the pool is
15816            // exactly one block per node whichever way the count moved. Under
15817            // SWMR a reader may still hold a header naming those blocks, so
15818            // keep them out of the free list — the same rule `place_chunk`
15819            // applies to a relocated chunk.
15820            for addr in node_addrs.split_off(tree.nodes.len()) {
15821                if !self.swmr_active {
15822                    self.allocator
15823                        .free(addr, tree.node_size as u64, FreeSpaceClass::Metadata);
15824                }
15825            }
15826
15827            for (image, &addr) in tree.encode(&self.ctx, &node_addrs).iter().zip(&node_addrs) {
15828                self.handle.write_at(addr, image)?;
15829            }
15830
15831            // The root is the last node the bulk load emits.
15832            let root_addr = match tree.nodes.len() {
15833                0 => UNDEF_ADDR,
15834                n => node_addrs[n - 1],
15835            };
15836            let hdr_encoded = tree.header(root_addr).encode(&self.ctx);
15837            self.handle.write_at(bt2.bt2_header_addr, &hdr_encoded)?;
15838
15839            m.btree_v2.as_mut().unwrap().node_addrs = node_addrs;
15840
15841            if sync {
15842                self.handle.sync_data()?;
15843            }
15844            return Ok(());
15845        }
15846
15847        // Version-1-B-tree-indexed dataset
15848        if let Some(ref bt1) = m.btree_v1 {
15849            // Bulk-loaded over the same block pool the v2 B-tree above uses,
15850            // and for the same reason: every node of a v1 tree is the width
15851            // its "K" value gives, so a block stays usable however the tree
15852            // reshapes, and only the shortfall is ever allocated.
15853            let element_size = m.datatype.element_size() as u64;
15854            let tree = bt1.build_tree(element_size, self.ctx.sizeof_addr as usize);
15855            let node_size = tree.node_size() as u64;
15856            let mut node_addrs = bt1.node_addrs.clone();
15857            while node_addrs.len() < tree.node_count() {
15858                node_addrs.push(self.allocator.allocate(node_size, FreeSpaceClass::Metadata));
15859            }
15860            // A tree with fewer nodes than last flush releases the surplus
15861            // straight away, where the v2 B-tree has to keep it out of the
15862            // free list for a live SWMR reader: this index lives only in a
15863            // classic file, which `start_swmr` refuses outright (and upstream
15864            // says the same in `H5D_COPS_BTREE`).
15865            for addr in node_addrs.split_off(tree.node_count()) {
15866                self.allocator
15867                    .free(addr, node_size, FreeSpaceClass::Metadata);
15868            }
15869            for (image, &addr) in tree.encode(&node_addrs)?.iter().zip(&node_addrs) {
15870                self.handle.write_at(addr, image)?;
15871            }
15872            // The root is the last node the bulk load emits, and is undefined
15873            // while the dataset has no chunks — what the version-3 data
15874            // layout message then carries, exactly as libhdf5 leaves it.
15875            let root_addr = tree.root_address(&node_addrs);
15876            let bt1 = m.btree_v1.as_mut().unwrap();
15877            bt1.node_addrs = node_addrs;
15878            bt1.root_addr = root_addr;
15879
15880            if sync {
15881                self.handle.sync_data()?;
15882            }
15883            return Ok(());
15884        }
15885
15886        Ok(())
15887    }
15888
15889    /// Finalize and close the file.
15890    ///
15891    /// Writes the dataset object headers, root group object header, and
15892    /// superblock. After this call the file is a valid HDF5 file.
15893    pub fn close(mut self) -> IoResult<()> {
15894        self.close_in_place()
15895    }
15896
15897    /// [`close`](Self::close) for a holder that cannot give the writer up by
15898    /// value because it has a `Drop` of its own ([`SwmrWriter`]): the same
15899    /// one-shot commit, after which this writer's `Drop` is a no-op.
15900    ///
15901    /// [`SwmrWriter`]: crate::io::swmr::SwmrWriter
15902    pub(crate) fn close_in_place(&mut self) -> IoResult<()> {
15903        // Mark closed BEFORE finalizing: finalize writes external truth
15904        // (object headers + superblock) and must run exactly once. If we
15905        // finalized first and it failed, the `?` would return with `closed`
15906        // still false, and dropping `self` would re-run `finalize` a second
15907        // time over a half-written file (and print the "call close()" notice
15908        // the caller already heeded). Committing to the close path first makes
15909        // `Drop` (the only other finalize site) a no-op regardless of outcome,
15910        // so the error is reported exactly once via this `Result`.
15911        self.closed = true;
15912        self.finalize(true)
15913    }
15914
15915    /// Finalize and close the file without a final `fsync`.
15916    ///
15917    /// Identical to [`close`](Self::close) — the same object headers and
15918    /// superblock are written, so on return the file is a complete, valid HDF5
15919    /// file readable by any process — except that the trailing `sync_all`
15920    /// (fsync) is skipped. The bytes are handed to the OS but are not
15921    /// guaranteed durable against power loss or an OS crash until the OS
15922    /// flushes its page cache; a normal process exit or a same-machine reader
15923    /// sees the full file regardless.
15924    ///
15925    /// This trades durability for speed: `sync_all` typically dominates close
15926    /// latency, so bulk writers that do not need crash durability (the file can
15927    /// be regenerated) can use this to avoid that cost. Use [`close`](Self::close)
15928    /// when durability matters. `Drop` always finalizes durably, so a writer
15929    /// finalized this way must reach `close_no_sync` explicitly.
15930    pub fn close_no_sync(mut self) -> IoResult<()> {
15931        // Same close-once discipline as `close`: commit to the close path
15932        // before finalizing so `Drop` cannot re-run `finalize` on failure.
15933        self.closed = true;
15934        self.finalize(false)
15935    }
15936
15937    /// Provide mutable access to the underlying file handle.
15938    pub fn handle(&mut self) -> &mut FileHandle {
15939        &mut self.handle
15940    }
15941
15942    /// The superblock version this file will be written with.
15943    ///
15944    /// `H5F__super_init` takes the oldest version that can describe the file
15945    /// and raises it to the one the file's library-version low bound implies:
15946    /// `super_vers = MAX(super_vers, HDF5_superblock_ver_bounds[low_bound])`,
15947    /// with the bounds table reading 0, 2, 3, 3, 3, 3, 3 for EARLIEST, V18,
15948    /// V110, V112, V114, V200, LATEST (H5Fsuper.c:68, :1128-1154). A file
15949    /// created at `H5F_LIBVER_EARLIEST` takes that bound's entry directly
15950    /// ([`SuperblockVersion::Chosen`], and the classic branch below) — version
15951    /// 0, or version 2 when the file carries shared messages, whose master
15952    /// table needs the superblock extension only a version-2 superblock has
15953    /// (H5Fsuper.c:1135). For every other file the bound is read back from
15954    /// what this crate writes:
15955    ///
15956    /// * The floor is `H5F_LIBVER_V18`, hence version 2. Every group such a
15957    ///   file holds is a link-message group, which libhdf5 only writes at a
15958    ///   low bound of V18 or newer (`use_at_least_v18`, H5Gobj.c:179), and
15959    ///   every object header in it is version 2, which `H5O_obj_ver_bounds`
15960    ///   likewise puts at V18 (H5Oint.c:125). A version-0 superblock over
15961    ///   this content would claim a file libhdf5 1.6 can read, and no libhdf5
15962    ///   writes that combination.
15963    /// * A chunked dataset — extensible array, fixed array or version-2
15964    ///   B-tree, all reached through a version-4 or -5 data layout message —
15965    ///   reads back as V110 (`H5O_layout_ver_bounds`, H5Dlayout.c:44), hence
15966    ///   version 3.
15967    /// * SWMR writes version 3 outright (H5Fsuper.c:1129).
15968    ///
15969    /// A file whose caller *named* a bound skips the read-back and takes that
15970    /// bound's row directly, so `V18` stays at version 2 however its chunked
15971    /// datasets are indexed — which is what libhdf5 does, the layout version
15972    /// being no input to `H5F__super_init` at all.
15973    ///
15974    /// None of that applies to a reopened file. `H5F__super_read` validates
15975    /// the version it finds and never recomputes one, so the version written
15976    /// back is the version read — see [`SuperblockVersion`], which is also
15977    /// where the other half of that rule lives: the version floors the bound
15978    /// the appended structures are written at, which is why nothing this
15979    /// session adds can need a newer one.
15980    fn superblock_version_for(&self, flags: u8) -> u8 {
15981        let chosen = match self.superblock_version {
15982            SuperblockVersion::Existing(version) => return version,
15983            SuperblockVersion::Chosen(version) => version,
15984        };
15985        if self.is_legacy() {
15986            // A classic file keeps the version it was created at — 0, or 2
15987            // when its shared messages needed the extension. Nothing a session
15988            // can add reaches past that: its objects get symbol-table links,
15989            // its chunked datasets the version-1 B-tree behind a version-3
15990            // layout message, and the two features that would raise the bound
15991            // — SWMR and the 2.0 format — are refused where the caller asks
15992            // for them.
15993            return chosen;
15994        }
15995        let mut version = chosen
15996            .max(SUPERBLOCK_V2)
15997            .max(self.effective_libver().superblock_version());
15998        if self.swmr_active || flags & FLAG_SWMR_WRITE != 0 {
15999            version = version.max(SUPERBLOCK_V3);
16000        }
16001        version
16002    }
16003
16004    /// The low bound a modern file this writer *created* is effectively
16005    /// written at: the one the caller named, or — with none named — the one
16006    /// its content reads back as. A reopened file never reaches here; its
16007    /// superblock version is not derived from its content at all.
16008    ///
16009    /// The read-back is what `superblock_version_for` needs and the field
16010    /// alone cannot give: this crate's default file names no bound, and the
16011    /// generation it writes is not one bound but two rows (see the `libver`
16012    /// field). The floor is `V18`, the oldest bound under which libhdf5 writes
16013    /// link-message groups (`use_at_least_v18`, H5Gobj.c:179) and version-2
16014    /// object headers (`H5O_obj_ver_bounds`, H5Oint.c:125), which is all such
16015    /// a file holds; a v1.10 chunk index in it raises that to `V110`, the
16016    /// oldest bound whose `H5O_layout_ver_bounds` row reaches the version-4
16017    /// layout message that index is written behind.
16018    fn effective_libver(&self) -> LibverBound {
16019        self.libver.unwrap_or_else(|| {
16020            if self.has_v110_chunk_index() {
16021                LibverBound::V110
16022            } else {
16023                LibverBound::V18
16024            }
16025        })
16026    }
16027
16028    /// Whether any dataset still in the file is indexed by a v1.10 chunk
16029    /// index — the markers `build_dataset_header` turns into a version-4/5
16030    /// data layout message, and nothing else it can emit reaches that
16031    /// version.
16032    ///
16033    /// Not "is any dataset chunked": the version-1 B-tree is a chunk index
16034    /// that encodes as a *version-3* layout message, the version
16035    /// `H5O_layout_ver_bounds` gives the earliest bound, so a dataset using
16036    /// it asks nothing of the superblock.
16037    fn has_v110_chunk_index(&self) -> bool {
16038        self.dataset_refs().iter().any(|d| {
16039            let m = d.lock();
16040            !m.deleted
16041                && m.chunk_index_kind()
16042                    .is_some_and(|k| k != ChunkIndexKind::BtreeV1)
16043        })
16044    }
16045
16046    /// Write the superblock at offset 0 with the given flags.
16047    ///
16048    /// Requires that the root group has already been written (via `finalize`
16049    /// or `finalize_for_swmr`).
16050    pub fn write_superblock(&mut self, flags: u8) -> IoResult<()> {
16051        let root_addr = self
16052            .root_group_addr
16053            .ok_or_else(|| crate::io::IoError::InvalidState("root group not yet written".into()))?;
16054        // The userblock this file was opened with. `H5F__super_read` prefers
16055        // the located address over this field, but `H5Pget_userblock` reports
16056        // it, so a rewrite that zeroed it would hide the block from every
16057        // reader that asks for its size.
16058        let base = self.handle.base();
16059        // The end of file is the one address in the superblock measured from
16060        // the start of the *file* rather than from the base: `H5F__super_read`
16061        // sets the EOA to `stored_eof - base_addr` (H5Fsuper.c:635) and calls
16062        // the file truncated when `eof + base_addr < stored_eof` (:573). The
16063        // allocator counts in the based space, so the userblock is added back.
16064        let eof = self.allocator.eof() + base;
16065        let version = self.superblock_version_for(flags);
16066        // Which of the two images is written follows the version, not the
16067        // generation: a classic file carrying shared messages is a version-2
16068        // superblock over version-1 messages and symbol-table groups
16069        // (H5Fsuper.c:1135), and only the version-2/3 image has the extension
16070        // address that table is reached through. Below version 2 the file is
16071        // always a classic one — the other branch floors at 2.
16072        if let Some(legacy) = self.legacy.as_deref().filter(|_| version < SUPERBLOCK_V2) {
16073            // Re-emitted, not rebuilt: the "K" ranks, the userblock size and
16074            // the driver info address are recorded nowhere else in the file,
16075            // and every node width in it is derived from the ranks. Only the
16076            // three things this session can have changed are recomputed.
16077            let root_stab = self
16078                .symbol_tables
16079                .written
16080                .lock()
16081                .get(&LinkScope::Root)
16082                .copied();
16083            let mut sb = legacy.superblock.clone();
16084            sb.version = version;
16085            sb.file_consistency_flags = flags as u32;
16086            sb.end_of_file_address = eof;
16087            sb.root_symbol_table_entry.obj_header_addr = root_addr;
16088            // `H5G__stab_valid` (H5Groot.c) reads this pair back and compares
16089            // it against the root header's Symbol Table message, repairing the
16090            // superblock when they disagree. Writing the pair that message now
16091            // names is what keeps the file from needing that repair. A root
16092            // that keeps its links in messages has no such pair and no entry
16093            // in `written`, and gets `H5G_NOTHING_CACHED` — what libhdf5
16094            // writes for the same root.
16095            sb.root_symbol_table_entry.cache = match root_stab {
16096                Some(s) => SymbolTableCache::SymbolTable {
16097                    btree_addr: s.btree_addr,
16098                    heap_addr: s.heap_addr,
16099                },
16100                None => SymbolTableCache::Nothing,
16101            };
16102            self.handle.write_at(0, &sb.encode())?;
16103            return Ok(());
16104        }
16105        let sb = SuperblockV2V3 {
16106            version,
16107            sizeof_offsets: self.ctx.sizeof_addr,
16108            sizeof_lengths: self.ctx.sizeof_size,
16109            file_consistency_flags: flags,
16110            base_address: base,
16111            // Whatever `write_superblock_extension` put there, which is the
16112            // only place an extension is written.
16113            superblock_extension_address: self.extension.addr.lock().unwrap_or(UNDEF_ADDR),
16114            end_of_file_address: eof,
16115            root_group_object_header_address: root_addr,
16116        };
16117        self.handle.write_at(0, &sb.encode())?;
16118        Ok(())
16119    }
16120
16121    /// Re-write a dataset's object header in place (SWMR update).
16122    ///
16123    /// The header must have been previously written via `finalize_for_swmr`.
16124    /// Only the dataspace dimensions change; the encoded size must not exceed
16125    /// the originally allocated space.
16126    pub fn write_dataset_header_inplace(&mut self, index: usize) -> IoResult<()> {
16127        // Scope the slot guard: `build_dataset_header` re-locks the same slot.
16128        let (addr, original_size) = {
16129            let ds = self.ds(index);
16130            let m = ds.lock();
16131            // One block, because a finalize writes every header as one
16132            // chunk: an in-place rewrite has that block's room and no more.
16133            match m.obj_header_blocks.as_slice() {
16134                [(addr, size)] => (*addr, *size as usize),
16135                _ => {
16136                    return Err(crate::io::IoError::InvalidState(
16137                        "dataset header not yet written as a single chunk".into(),
16138                    ))
16139                }
16140            }
16141        };
16142
16143        let header = self.build_dataset_header(index)?;
16144        let nlink = self.object_link_count(HardLinkTarget::Dataset(index));
16145        let encoded =
16146            self.encode_header_at(&header, nlink, self.dataset_header_format(index), addr)?;
16147
16148        if encoded.len() > original_size {
16149            return Err(crate::io::IoError::InvalidState(format!(
16150                "dataset header grew from {} to {} bytes; cannot rewrite in place",
16151                original_size,
16152                encoded.len()
16153            )));
16154        }
16155
16156        // Pad to original size with zeros (the trailing zeros after the
16157        // checksum won't be parsed by readers since chunk0_data_size is fixed).
16158        let mut padded = encoded;
16159        padded.resize(original_size, 0);
16160
16161        self.handle.write_at(addr, &padded)?;
16162        // Only after the bytes are down: a failed write leaves the registry
16163        // describing the header the file still holds.
16164        self.ds(index).lock().header_written(nlink);
16165        Ok(())
16166    }
16167
16168    /// Perform a full finalize for SWMR mode.
16169    ///
16170    /// This writes all dataset object headers, the root group header, and the
16171    /// superblock with SWMR flags. After this call, the file is valid for
16172    /// SWMR readers. Subsequent writes use in-place updates.
16173    pub fn finalize_for_swmr(&mut self) -> IoResult<()> {
16174        self.reject_swmr()?;
16175        // 0. Flush all chunked dataset index structures.
16176        for i in 0..self.dataset_count() {
16177            let is_indexed = {
16178                let ds = self.ds(i);
16179                let m = ds.lock();
16180                !m.deleted && m.is_chunked()
16181            };
16182            if is_indexed {
16183                self.flush_dataset(i)?;
16184            }
16185        }
16186
16187        // 1. Allocate every object header (none for a dataset deleted before
16188        // start_swmr — its storage was freed at delete time). Same three
16189        // phases as the full finalize, and for the same reason: nothing a
16190        // header names can be laid out until every object has an address.
16191        let live: Vec<usize> = (0..self.dataset_count())
16192            .filter(|&i| !self.ds(i).lock().deleted)
16193            .collect();
16194        // Before any dataset header: a sharing dataset's header names the
16195        // committed type's address.
16196        self.write_committed_datatype_headers()?;
16197        let layout = self.allocate_object_headers(&live)?;
16198
16199        // 2. Build content against those addresses.
16200        self.prepare_dense_attributes(&live)?;
16201        self.prepare_link_storage()?;
16202        self.write_reference_values()?;
16203
16204        // 3. Write every object header.
16205        self.write_object_headers(&layout)?;
16206        // What SWMR alone needs to know afterwards: where each dataset's
16207        // header is published and how much room it has, which is what
16208        // `write_dataset_header_inplace` rewrites within.
16209        for &(i, addr, size) in &layout.datasets {
16210            let ds = self.ds(i);
16211            let mut m = ds.lock();
16212            m.obj_header_written_addr = Some(addr);
16213            m.obj_header_blocks = vec![(addr, size as u64)];
16214        }
16215        self.root_group_encoded_size = layout.root.1;
16216
16217        // 4. Write superblock with SWMR flags.
16218        self.write_superblock(FLAG_WRITE_ACCESS | FLAG_SWMR_WRITE)?;
16219        self.handle.set_eof(self.allocator.eof())?;
16220
16221        self.handle.sync_all()?;
16222        // Readers can now be following this file, so a chunk that moves must
16223        // leave its old block intact for whoever is still holding the previous
16224        // index (see `swmr_active`).
16225        self.swmr_active = true;
16226        Ok(())
16227    }
16228
16229    // ------------------------------------------------------------------
16230    // Internal helpers
16231    // ------------------------------------------------------------------
16232
16233    /// Flush every dataset's append buffer into the chunks it belongs to,
16234    /// through [`flush_append_buffer`](Self::flush_append_buffer): frames
16235    /// already in the chunk survive, and the rest of it reads back as the
16236    /// dataset's fill value (zeros when none is defined).
16237    fn flush_append_buffers(&mut self) -> IoResult<()> {
16238        for i in 0..self.dataset_count() {
16239            if self.ds(i).lock().deleted {
16240                continue;
16241            }
16242            self.flush_append_buffer(i)?;
16243        }
16244        Ok(())
16245    }
16246
16247    /// Write all object headers and the superblock, producing a complete,
16248    /// valid HDF5 file.
16249    ///
16250    /// `sync == true` issues a final `sync_all` (fsync) so the bytes are
16251    /// durable against power loss / OS crash before returning. `sync == false`
16252    /// skips that fsync: the file is still fully written to the OS and readable
16253    /// by any process, but durability is left to the OS page-cache flush. This
16254    /// is the only difference between [`close`](Self::close) (durable) and
16255    /// [`close_no_sync`](Self::close_no_sync) (fast).
16256    fn finalize(&mut self, sync: bool) -> IoResult<()> {
16257        // Flush any partial append buffers before finalizing
16258        self.flush_append_buffers()?;
16259
16260        // A SWMR session (`finalize_for_swmr` already ran, so
16261        // `root_group_addr` is `Some`) is closed by the same full finalize as
16262        // a fresh write: every object header is rebuilt at a fresh address and
16263        // the superblock is written with clean-close flags. A full rebuild —
16264        // rather than the in-place header rewrite used by the live
16265        // `SwmrWriter::flush` path — is required so any structural change made
16266        // after `start_swmr` is committed to the final file. A hard link, in
16267        // particular, both grows its target's header with an object
16268        // reference-count message and adds a `MSG_LINK` record to a group
16269        // header; an in-place rewrite cannot accommodate the grown header and
16270        // never re-emits group/root headers. The fall-through below already
16271        // handles datasets whose header was written by `finalize_for_swmr`
16272        // (`obj_header_written_addr.is_some()`).
16273
16274        // 0. Flush chunked dataset index structures (only modified datasets).
16275        for i in 0..self.dataset_count() {
16276            let ds = self.ds(i);
16277            {
16278                let m = ds.lock();
16279                if m.deleted {
16280                    continue;
16281                }
16282                if m.obj_header_written_addr.is_some() && !m.storage_dirty() {
16283                    continue;
16284                }
16285                let is_indexed = m.is_chunked();
16286                if !is_indexed {
16287                    continue;
16288                }
16289            }
16290            self.flush_dataset_synced(i, sync)?;
16291        }
16292
16293        // Every header block this finalize supersedes — a reopened root or
16294        // group header, a modified dataset's reopened header — is freed
16295        // before its replacement is allocated, so the rewrite reuses the
16296        // block instead of growing the file on every open/close cycle.
16297        // Never under SWMR: a live reader may be walking the old headers,
16298        // the same rule `release_vlen_references` and `place_chunk` follow.
16299        // Hard links can alias one header under several names; the set keeps
16300        // an aliased block from entering the free list twice.
16301        let mut freed_headers = std::collections::HashSet::new();
16302
16303        // 1. Plan. Which datasets get a header (deleted datasets get none —
16304        // their storage was already freed at delete time) is settled first,
16305        // because everything the next phases lay out is laid out only for the
16306        // headers this finalize actually rewrites; and every header block
16307        // those phases supersede is returned here, before the first
16308        // allocation, so a rewrite can land in it.
16309        let mut rewritten: Vec<usize> = Vec::new();
16310        // A finalize that lays the shared-message table out afresh reassigns
16311        // every heap ID in the file, so no existing header can keep its bytes:
16312        // the pointers in them name heap objects the new table does not have.
16313        let table_replaced = self.rebuilds_shared_messages();
16314        for i in 0..self.dataset_count() {
16315            // Before the slot guard: `object_link_count` re-locks every
16316            // dataset and group slot, this one included.
16317            let nlink = self.object_link_count(HardLinkTarget::Dataset(i));
16318            let ds = self.ds(i);
16319            let mut m = ds.lock();
16320            if m.deleted {
16321                continue;
16322            }
16323            if m.obj_header_written_addr.is_some() {
16324                // An existing dataset from append mode keeps its header — and
16325                // everything that header names — unless this session changed
16326                // what the header says.
16327                if !table_replaced && !m.header_stale_with(nlink) {
16328                    // Keep the original object header address for the root group link.
16329                    m.obj_header_addr = m.obj_header_written_addr.unwrap();
16330                    continue;
16331                }
16332                if !self.swmr_active && !m.obj_header_blocks.is_empty() {
16333                    let old = m.obj_header_written_addr.take().unwrap();
16334                    let blocks = std::mem::take(&mut m.obj_header_blocks);
16335                    if freed_headers.insert(old) {
16336                        for (addr, len) in blocks {
16337                            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16338                        }
16339                    }
16340                }
16341            }
16342            rewritten.push(i);
16343        }
16344        if !self.swmr_active {
16345            for gi in 0..self.group_count() {
16346                let grp = self.grp(gi);
16347                let mut g = grp.lock();
16348                if let Some(old) = g
16349                    .obj_header_written_addr
16350                    .take()
16351                    .filter(|_| !g.obj_header_blocks.is_empty())
16352                {
16353                    let blocks = std::mem::take(&mut g.obj_header_blocks);
16354                    if freed_headers.insert(old) {
16355                        for (addr, len) in blocks {
16356                            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16357                        }
16358                    }
16359                }
16360            }
16361            let root_blocks = std::mem::take(&mut self.superseded_root_header);
16362            if root_blocks
16363                .first()
16364                .is_some_and(|&(addr, _)| freed_headers.insert(addr))
16365            {
16366                for (addr, len) in root_blocks {
16367                    self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16368                }
16369            }
16370        }
16371
16372        // 2. Allocate. Committed datatype headers go down whole: a header of
16373        // theirs holds a datatype and a reference count, so it waits on
16374        // nothing, while a dataset sharing the type and the group naming it
16375        // both store its address. They are written before the shared-message
16376        // phase opens, so a committed type reaches the file as itself.
16377        self.write_committed_datatype_headers()?;
16378        self.begin_shared_message_layout();
16379        let layout = self.allocate_object_headers(&rewritten)?;
16380
16381        // 3. Build content, with every object header's address known. Dense
16382        // attribute storage holds the attribute messages themselves — an
16383        // object reference among them is a header address; dense links and
16384        // symbol tables name header addresses; a reference dataset's elements
16385        // are header addresses. Nothing here is a fixup: each is written once,
16386        // with the value the file keeps. The shared-message table comes last:
16387        // it counts the bodies the headers will hold, and the three above are
16388        // what settle them.
16389        self.prepare_dense_attributes(&rewritten)?;
16390        self.prepare_link_storage()?;
16391        self.write_reference_values()?;
16392        self.prepare_shared_messages(&rewritten)?;
16393        self.write_superblock_extension()?;
16394
16395        // 4. Write every object header over the block phase 2 reserved for it.
16396        self.write_object_headers(&layout)?;
16397
16398        // 5. Write superblock at offset 0.
16399        self.write_superblock(0)?;
16400
16401        // 6. End the file where its address space ends (`H5FD_truncate`, which
16402        // `H5F__dest` calls on every close). Allocated-but-unwritten space at
16403        // the end would otherwise leave the file shorter than the end-of-file
16404        // address the superblock just recorded, which libhdf5 reads as a
16405        // truncated file.
16406        self.handle.set_eof(self.allocator.eof())?;
16407
16408        // Durability is opt-in per call: `close` passes `true`, `close_no_sync`
16409        // passes `false`, and `Drop` passes `true` so an un-`close`d writer is
16410        // still finalized durably by default.
16411        if sync {
16412            self.handle.sync_all()?;
16413        }
16414        Ok(())
16415    }
16416
16417    /// Give every object header this finalize writes an address, before
16418    /// anything that names one is built.
16419    ///
16420    /// INVARIANT: from the moment this returns until the file is closed, every
16421    /// object in it has the object header address it will be found at. That is
16422    /// what lets the phase after this one say an address wherever the format
16423    /// wants one — in a link message, in a symbol table entry, in a reference
16424    /// dataset's elements, and in an attribute's value, which is the one of the
16425    /// four that cannot be revisited after its header is written.
16426    ///
16427    /// Measuring a header before its content is final is sound because no
16428    /// address changes its length: every address is a fixed-width field, and an
16429    /// object that has none yet reads as zero, which is the same width. The
16430    /// storage a header names is laid out between the two passes for the same
16431    /// reason and answers the same way — `emit_attributes` and `emit_links`
16432    /// each fall back to a size-equal placeholder message. It is
16433    /// [`write_object_headers`](Self::write_object_headers) that checks this
16434    /// held, rather than either pass assuming it.
16435    fn allocate_object_headers(&mut self, datasets: &[usize]) -> IoResult<HeaderLayout> {
16436        let mut layout = HeaderLayout {
16437            datasets: Vec::with_capacity(datasets.len()),
16438            groups: Vec::new(),
16439            root: (0, 0),
16440        };
16441        for &i in datasets {
16442            let rc = self.object_link_count(HardLinkTarget::Dataset(i));
16443            let header = self.build_dataset_header(i)?;
16444            let size = self.header_encoded_size(&header, rc, self.dataset_header_format(i))?;
16445            let addr = self
16446                .allocator
16447                .allocate(size as u64, FreeSpaceClass::Metadata);
16448            self.ds(i).lock().obj_header_addr = addr;
16449            layout.datasets.push((i, addr, size));
16450        }
16451        for gi in 0..self.group_count() {
16452            if self.grp(gi).lock().deleted {
16453                continue;
16454            }
16455            let rc = self.object_link_count(HardLinkTarget::Group(gi));
16456            let header = self.build_group_header(gi)?;
16457            let size = self.header_encoded_size(&header, rc, self.group_header_format(gi))?;
16458            let addr = self
16459                .allocator
16460                .allocate(size as u64, FreeSpaceClass::Metadata);
16461            self.grp(gi).lock().obj_header_addr = addr;
16462            layout.groups.push((gi, addr, size));
16463        }
16464        let header = self.build_root_group_header()?;
16465        let size =
16466            self.header_encoded_size(&header, 1, self.header_format(self.root_track_order))?;
16467        let addr = self
16468            .allocator
16469            .allocate(size as u64, FreeSpaceClass::Metadata);
16470        self.root_group_addr = Some(addr);
16471        layout.root = (addr, size);
16472        Ok(layout)
16473    }
16474
16475    /// Write every object header over the block
16476    /// [`allocate_object_headers`](Self::allocate_object_headers) reserved for
16477    /// it.
16478    ///
16479    /// The single owner of object header writing in both finalize paths, and
16480    /// the only place a header's body meets its block: a body that does not
16481    /// fill its measurement exactly fails the finalize here rather than
16482    /// overrunning the next object or leaving a tail of the previous one, which
16483    /// is how a message whose length turns out to depend on an address would
16484    /// show up.
16485    fn write_object_headers(&mut self, layout: &HeaderLayout) -> IoResult<()> {
16486        for &(i, addr, size) in &layout.datasets {
16487            let rc = self.object_link_count(HardLinkTarget::Dataset(i));
16488            let header = self.build_dataset_header(i)?;
16489            let encoded =
16490                self.encode_header_at(&header, rc, self.dataset_header_format(i), addr)?;
16491            check_header_size(&encoded, size, || {
16492                format!("dataset '{}'", self.ds(i).lock().name)
16493            })?;
16494            self.handle.write_at(addr, &encoded)?;
16495            // Only after the bytes are down: a failed write leaves the registry
16496            // describing the header the file still holds.
16497            self.ds(i).lock().header_written(rc);
16498        }
16499        for &(gi, addr, size) in &layout.groups {
16500            let rc = self.object_link_count(HardLinkTarget::Group(gi));
16501            let header = self.build_group_header(gi)?;
16502            let encoded = self.encode_header_at(&header, rc, self.group_header_format(gi), addr)?;
16503            check_header_size(&encoded, size, || {
16504                format!("group '{}'", self.grp(gi).lock().name)
16505            })?;
16506            self.handle.write_at(addr, &encoded)?;
16507        }
16508        let (addr, size) = layout.root;
16509        let header = self.build_root_group_header()?;
16510        let encoded =
16511            self.encode_header_at(&header, 1, self.header_format(self.root_track_order), addr)?;
16512        check_header_size(&encoded, size, || "the root group".to_string())?;
16513        self.handle.write_at(addr, &encoded)?;
16514        Ok(())
16515    }
16516
16517    fn build_dataset_header(&self, index: usize) -> IoResult<ObjectHeader> {
16518        // Compute the link count first: object_link_count re-locks dataset and
16519        // group slots (including this one), so it must run before we take this
16520        // dataset's slot guard — otherwise it would deadlock on the same slot.
16521        let rc = self.object_link_count(HardLinkTarget::Dataset(index));
16522        // Same reason: reading the committed type's address locks the
16523        // committed-datatype registry, which the slot guard below must not be
16524        // held across.
16525        let committed = self.ds(index).lock().committed_type;
16526        let committed_addr = committed.map(|r| match r {
16527            CommittedTypeRef::Session(ci) => self.committed_datatypes.lock()[ci].obj_header_addr,
16528            CommittedTypeRef::Preserved(addr) => addr,
16529        });
16530        // And again: an attribute holding an object reference is said in the
16531        // target's header address, which is read off that object's slot.
16532        let attributes = self.object_attributes(AttrScope::Dataset(index))?;
16533
16534        // Hold one slot guard for the whole header build.
16535        let ds = self.ds(index);
16536        let m = ds.lock();
16537        let mut header = ObjectHeader::new();
16538
16539        // Every message below is written in the format this dataset already
16540        // has, not the one this session would pick. libhdf5 grows a header in
16541        // place and never re-encodes a message it did not touch, so reopening
16542        // a superblock-v2 file — which raises the low bound to V18
16543        // (hdf5_1.14.6 H5Fsuper.c:460-462) — leaves the version-1 dataspaces
16544        // an EARLIEST-bound creating session wrote exactly as they are. This
16545        // writer has to lay the whole header out again whenever the
16546        // shared-message heap moves, so preserving the encoding is the only
16547        // way to land on the same bytes.
16548        let format = m.read_format.unwrap_or_else(|| self.message_format());
16549        let libver = match format {
16550            ObjectFormat::Legacy => LibverBound::Earliest,
16551            ObjectFormat::Modern => self.encoding_libver(),
16552        };
16553
16554        // Dataspace message (type 0x01)
16555        let ds_msg = m.dataspace.encode_for(&self.ctx, format);
16556        let owner = ShareOwner::Header(m.obj_header_addr);
16557        let (flags, ds_msg) = self.share_message(owner, MSG_DATASPACE, 0x00, ds_msg);
16558        header.add_message(MSG_DATASPACE, flags, ds_msg);
16559
16560        // Datatype message (type 0x03). A dataset built on a committed type
16561        // stores a pointer to that object header in place of the message, and
16562        // the shared flag is what says the body is a pointer — the two are one
16563        // statement, so they are written together.
16564        match committed_addr {
16565            Some(addr) => header.add_message(
16566                MSG_DATATYPE,
16567                MSG_FLAG_CONSTANT | MSG_FLAG_SHARED,
16568                SharedMessagePointer::encode_committed(addr, &self.ctx),
16569            ),
16570            None => {
16571                let body = m.datatype.encode_at(&self.ctx, libver);
16572                let (flags, body) = if self.dataset_datatype_shareable(&m.datatype, libver) {
16573                    self.share_message(owner, MSG_DATATYPE, MSG_FLAG_CONSTANT, body)
16574                } else {
16575                    (MSG_FLAG_CONSTANT, body)
16576                };
16577                header.add_message(MSG_DATATYPE, flags, body)
16578            }
16579        }
16580
16581        // Fill Value message (type 0x05)
16582        let is_chunked = m.is_chunked();
16583        // `H5P__init_def_layout` gives each storage class its own default
16584        // allocation time: incremental for chunked and for virtual (whose
16585        // source datasets are allocated as they are written), early for
16586        // compact (the space is the header, so it exists as soon as the
16587        // dataset does), late for contiguous. An implicitly indexed dataset is
16588        // the one chunked exception, and not by default but by definition:
16589        // early allocation is a *condition* of that index
16590        // (`H5D__layout_set_latest_indexing`), so a header claiming
16591        // incremental would describe a file libhdf5 would never have chosen
16592        // this index for. A single-chunk dataset can go either way — unlike
16593        // Implicit, early allocation is not one of its selection conditions
16594        // — so its `early_alloc` flag (set only for an unfiltered dataset
16595        // created that way) is what this checks instead.
16596        let alloc_time = if m.compact.is_some()
16597            || m.implicit.is_some()
16598            || m.single_chunk.as_ref().is_some_and(|s| s.early_alloc)
16599        {
16600            1 // early
16601        } else if is_chunked || m.virtual_storage.is_some() {
16602            3 // incremental
16603        } else {
16604            2 // late
16605        };
16606        // `H5D__update_oh_info` (H5Dint.c:927-943): a variable-length
16607        // datatype with no explicit fill value forces ALLOC regardless of
16608        // the declared policy — its heap-reference encoding has no safe
16609        // all-zero "no fill" representation, so libhdf5 always writes the
16610        // (empty) fill value at allocation for such a dataset. `IFSET` is
16611        // the only declared policy this touches: an explicit `ALLOC` is
16612        // already what it forces, and upstream rejects `NEVER` for a
16613        // VL-typed dataset at `H5Dcreate` outright — this crate's
16614        // VL-typed datasets have no builder path to declare `NEVER` in the
16615        // first place, so that branch cannot be reached here.
16616        let is_vlen = matches!(
16617            m.datatype,
16618            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
16619        );
16620        let fill_write_time = if is_vlen && m.fill_value.is_none() && m.fill_time == FILL_TIME_IFSET
16621        {
16622            FILL_TIME_ALLOC
16623        } else {
16624            m.fill_time
16625        };
16626        let fv = if let Some(ref bytes) = m.fill_value {
16627            // User-defined fill value (fill_defined = 2).
16628            FillValueMessage {
16629                alloc_time,
16630                fill_write_time,
16631                fill_defined: 2,
16632                fill_value: Some(bytes.clone()),
16633            }
16634        } else {
16635            // No fill value of the dataset's own (fill_defined = 1, the
16636            // implicit default zero fill) — `alloc_time` above already
16637            // carries the per-layout-class default (`H5P__set_layout`,
16638            // H5Pdcpl.c:1864-1877), so this branch must use it too instead
16639            // of `FillValueMessage::default()`'s hardcoded LATE: that was
16640            // wrong for a compact (EARLY) or virtual (INCR) dataset with no
16641            // fill value, only coincidentally right for contiguous.
16642            FillValueMessage {
16643                alloc_time,
16644                fill_write_time,
16645                fill_defined: 1, // default value (zeros)
16646                fill_value: None,
16647            }
16648        };
16649        // `H5O_MSG_FLAG_CONSTANT`, as `H5D__update_oh_info` appends it
16650        // (H5Dint.c:965) — the same flag the datatype message beside it
16651        // carries (H5Dint.c:961) and the old fill value below (H5Dint.c:981).
16652        // A dataset's fill value is fixed at creation: `H5Pset_fill_value` is
16653        // a creation property, so nothing can rewrite the message in place and
16654        // libhdf5 tells the header so.
16655        let fv_msg = fv.encode_for(format);
16656        let (flags, fv_msg) = self.share_message(owner, MSG_FILL_VALUE, MSG_FLAG_CONSTANT, fv_msg);
16657        header.add_message(MSG_FILL_VALUE, flags, fv_msg);
16658
16659        // The "fill value (old)" message (type 0x04) beside the new one, for a
16660        // user-defined fill value below the v1.8 bound. `H5D__update_oh_info`
16661        // (H5Dint.c:1024-1035) appends `H5O_FILL_ID` whenever `fill_prop->buf`
16662        // is set and `use_at_least_v18` — `H5F_LOW_BOUND(file) >= V18`, which
16663        // here is exactly a non-`Legacy` message format — is false, so that a
16664        // reader that predates the new message still finds the value. The body
16665        // is the size and the bytes and nothing else: no allocation time, no
16666        // write time, no defined flag (`H5O__fill_old_encode`, H5Ofill.c:512).
16667        if matches!(format, ObjectFormat::Legacy) {
16668            if let Some(ref bytes) = m.fill_value {
16669                let mut old = Vec::with_capacity(4 + bytes.len());
16670                old.extend_from_slice(&(bytes.len() as u32).to_le_bytes());
16671                old.extend_from_slice(bytes);
16672                let (flags, old) =
16673                    self.share_message(owner, MSG_FILL_VALUE_OLD, MSG_FLAG_CONSTANT, old);
16674                header.add_message(MSG_FILL_VALUE_OLD, flags, old);
16675            }
16676        }
16677
16678        // External Data Files message (type 0x07), before the layout message
16679        // and marked constant, exactly where `H5D__layout_oh_create` puts it.
16680        // It is what makes a reader route the dataset's I/O through the files
16681        // it names rather than through the undefined address the layout
16682        // message below still declares.
16683        if let Some(ref ext) = m.external {
16684            header.add_message(
16685                MSG_EXTERNAL_FILE_LIST,
16686                MSG_FLAG_CONSTANT,
16687                ext.message().encode(&self.ctx),
16688            );
16689        }
16690
16691        // Data Layout message (type 0x08)
16692        let layout = if let Some(ref chunked) = m.chunked {
16693            let mut layout_dims = chunked.chunk_dims.clone();
16694            layout_dims.push(m.datatype.element_size() as u64);
16695            DataLayoutMessage::chunked_v4_earray(
16696                m.layout_version,
16697                layout_dims,
16698                chunked.earray_params.clone(),
16699                chunked.ea_header_addr,
16700            )
16701        } else if let Some(ref fa) = m.fixed_array {
16702            let mut layout_dims = fa.chunk_dims.clone();
16703            layout_dims.push(m.datatype.element_size() as u64);
16704            DataLayoutMessage::chunked_v4_farray(
16705                m.layout_version,
16706                layout_dims,
16707                FixedArrayParams::default_params(),
16708                fa.fa_header_addr,
16709            )
16710        } else if let Some(ref bt2) = m.btree_v2 {
16711            let mut layout_dims = bt2.chunk_dims.clone();
16712            layout_dims.push(m.datatype.element_size() as u64);
16713            DataLayoutMessage::chunked_v4_btree_v2(
16714                m.layout_version,
16715                layout_dims,
16716                crate::format::messages::data_layout::Bt2Params {
16717                    node_size: bt2.index.node_size,
16718                    split_percent: bt2.index.split_percent,
16719                    merge_percent: bt2.index.merge_percent,
16720                },
16721                bt2.bt2_header_addr,
16722            )
16723        } else if let Some(ref imp) = m.implicit {
16724            let mut layout_dims = imp.chunk_dims.clone();
16725            layout_dims.push(m.datatype.element_size() as u64);
16726            DataLayoutMessage::chunked_v4_implicit(m.layout_version, layout_dims, imp.data_addr)
16727        } else if let Some(ref sc) = m.single_chunk {
16728            let mut layout_dims = sc.chunk_dims.clone();
16729            layout_dims.push(m.datatype.element_size() as u64);
16730            if m.filter_pipeline.is_some() {
16731                DataLayoutMessage::chunked_v4_single_filtered(
16732                    layout_dims,
16733                    sc.data_addr,
16734                    sc.nbytes,
16735                    sc.filter_mask,
16736                )
16737            } else {
16738                DataLayoutMessage::chunked_v4_single(layout_dims, sc.data_addr)
16739            }
16740        } else if let Some(ref bt1) = m.btree_v1 {
16741            // The classic index: a version-3 layout message carrying the
16742            // address of the tree's root node, which is undefined until a
16743            // chunk is written.
16744            let mut layout_dims = bt1.chunk_dims.clone();
16745            layout_dims.push(m.datatype.element_size() as u64);
16746            DataLayoutMessage::chunked_v3_btree_v1(layout_dims, bt1.root_addr)
16747        } else if let Some(ref image) = m.compact {
16748            DataLayoutMessage::compact(image.clone())
16749        } else if let Some(ref virt) = m.virtual_storage {
16750            // Version 4 always: the virtual layout class did not exist before
16751            // it, so the default virtual layout is created at version 4 and
16752            // `H5Pset_virtual` raises any lower one to it (H5Pdcpl.c),
16753            // whatever the file's library-version bounds say — which is why a
16754            // v0-superblock file can still hold one.
16755            DataLayoutMessage::virtual_layout(4, virt.heap_addr, virt.heap_index)
16756        } else {
16757            DataLayoutMessage::contiguous(m.data_addr, m.data_size)
16758        };
16759        // `H5D__layout_oh_create` (H5Dlayout.c:530-536) marks the layout
16760        // message constant only where the storage it names is certain to be
16761        // there already: allocation time is early, the class is not compact,
16762        // no filter can change a chunk's size, and the dataspace holds at
16763        // least one element. Anything else leaves the address undefined at
16764        // creation and rewrites the message when the space is allocated, so
16765        // the flag would be a lie. `H5S_GET_EXTENT_NPOINTS` is zero for a
16766        // NULL dataspace and for any extent with a zero-length dimension.
16767        let npoints: u64 = if m.dataspace.is_null() {
16768            0
16769        } else {
16770            m.dataspace.dims.iter().product()
16771        };
16772        let filtered = m
16773            .filter_pipeline
16774            .as_ref()
16775            .is_some_and(|p| !p.filters.is_empty());
16776        let layout_flags = if alloc_time == 1 && m.compact.is_none() && !filtered && npoints != 0 {
16777            MSG_FLAG_CONSTANT
16778        } else {
16779            0x00
16780        };
16781        let layout_msg = layout.encode(&self.ctx);
16782        header.add_message(MSG_DATA_LAYOUT, layout_flags, layout_msg);
16783
16784        // Filter Pipeline message (type 0x0B) -- only if filters are
16785        // configured. `H5D__layout_oh_create` appends it with
16786        // `H5O_MSG_FLAG_CONSTANT` (H5Dlayout.c:462), as does the group
16787        // pipeline for dense links (H5Gobj.c:264): the pipeline is a creation
16788        // property, and every chunk already written was filtered through it,
16789        // so it can never be rewritten in place.
16790        if let Some(ref pipeline) = m.filter_pipeline {
16791            if !pipeline.filters.is_empty() {
16792                let (flags, filter_msg) = self.share_message(
16793                    owner,
16794                    MSG_FILTER_PIPELINE,
16795                    MSG_FLAG_CONSTANT,
16796                    pipeline.encode_for(format),
16797                );
16798                header.add_message(MSG_FILTER_PIPELINE, flags, filter_msg);
16799            }
16800        }
16801
16802        // A dataset has no links, so only attribute creation order can raise
16803        // its header past version 1 (`H5O__set_version`).
16804        let format = self.header_format(TrackOrder {
16805            links: CreationOrder::default(),
16806            attrs: m.track_attr_order,
16807        });
16808
16809        // Modification time, here and not earlier: `H5D__update_oh_info` makes
16810        // this the last message it writes (H5Dint.c:1022-1026), and the
16811        // attributes below it are added by `H5A` calls that come after the
16812        // dataset exists.
16813        touch_oh(&mut header, format, m.times, true);
16814
16815        // Attribute Info (type 0x15) + attribute messages (type 0x0C).
16816        self.emit_attributes(
16817            &mut header,
16818            AttrScope::Dataset(index),
16819            &attributes,
16820            m.track_attr_order,
16821            format,
16822            owner,
16823        );
16824
16825        self.emit_refcount(&mut header, rc, format);
16826
16827        Ok(header)
16828    }
16829
16830    /// Write the object header of every committed datatype something still
16831    /// reaches, recording the address each one landed at.
16832    ///
16833    /// Runs before the dataset and group headers because both name these
16834    /// addresses — a sharing dataset in its datatype message, the parent
16835    /// group in the link. One pass is enough: the header holds a datatype
16836    /// message and at most a reference count, neither of which depends on an
16837    /// address.
16838    fn write_committed_datatype_headers(&mut self) -> IoResult<()> {
16839        // The count is bound first: a lock guard in the `for` iterator
16840        // expression would live for the whole loop body, which locks the same
16841        // registry again.
16842        let count = self.committed_datatypes.lock().len();
16843        for i in 0..count {
16844            let rc = self.committed_datatype_refcount(i);
16845            if rc == 0 {
16846                // Its name's group was deleted and no dataset shares it, so
16847                // nothing in the file could reach the header.
16848                continue;
16849            }
16850            let format = self.committed_datatype_header_format();
16851            let encoded = self
16852                .build_committed_datatype_header(i, rc, format)
16853                .encode_for(format, rc)?;
16854            let addr = self
16855                .allocator
16856                .allocate(encoded.len() as u64, FreeSpaceClass::Metadata);
16857            self.handle.write_at(addr, &encoded)?;
16858            self.committed_datatypes.lock()[i].obj_header_addr = addr;
16859        }
16860        Ok(())
16861    }
16862
16863    /// The header format a committed datatype gets.
16864    ///
16865    /// `H5T__commit` creates the header from the datatype creation property
16866    /// list (H5Tcommit.c:468), which carries no link order and, by default, no
16867    /// attribute order — so the version is the file's floor exactly as
16868    /// `H5O__set_version` computes it, and a committed datatype in a classic
16869    /// file is a version-1 header like every other object in it.
16870    fn committed_datatype_header_format(&self) -> ObjectFormat {
16871        self.header_format(TrackOrder::default())
16872    }
16873
16874    /// Build the object header for a committed datatype: the type, and the
16875    /// reference count when more than one name reaches it.
16876    fn build_committed_datatype_header(
16877        &self,
16878        index: usize,
16879        rc: u32,
16880        format: ObjectFormat,
16881    ) -> ObjectHeader {
16882        let (datatype, times) = {
16883            let reg = self.committed_datatypes.lock();
16884            (reg[index].datatype.clone(), reg[index].times)
16885        };
16886        let mut header = ObjectHeader::new();
16887        // No attributes to emit, so nothing else would apply the file-wide
16888        // floor to this header. `store_msg_crt_idx` is a property of the file,
16889        // not of the object: every header created under it records creation
16890        // indices, a committed datatype's included.
16891        header.set_attribute_creation_order(self.header_attr_order(CreationOrder::default()));
16892        // `H5T__commit` marks the message constant and unshareable: this
16893        // header is where shared datatype bodies are read *from*, so its own
16894        // message must never become a pointer into the shared-message heap.
16895        header.add_message(
16896            MSG_DATATYPE,
16897            MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
16898            datatype.encode_at(&self.ctx, self.encoding_libver()),
16899        );
16900        touch_oh(&mut header, format, times, false);
16901        // Through the same owner as every other object's count: a dataset
16902        // sharing this type raises it (`H5O__shared_link_adj`, H5Oshared.c:249)
16903        // just as a second name does, and where that count is recorded is the
16904        // header version's business, not the caller's.
16905        self.emit_refcount(&mut header, rc, format);
16906        header
16907    }
16908
16909    /// Build the object header for a subgroup.
16910    fn build_group_header(&self, group_idx: usize) -> IoResult<ObjectHeader> {
16911        let mut header = ObjectHeader::new();
16912
16913        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
16914        // themselves, compact or dense.
16915        // Snapshot what the header needs, then drop the slot guard: the calls
16916        // below re-lock group slots (including this one).
16917        let (track_order, times, owner) = {
16918            let grp = self.grp(group_idx);
16919            let g = grp.lock();
16920            (
16921                g.track_order,
16922                g.times,
16923                ShareOwner::Header(g.obj_header_addr),
16924            )
16925        };
16926        let attributes = self.object_attributes(AttrScope::Group(group_idx))?;
16927        touch_oh(&mut header, self.header_format(track_order), times, false);
16928
16929        let links = self.group_links(LinkScope::Group(group_idx), track_order.links);
16930        self.emit_links(
16931            &mut header,
16932            LinkScope::Group(group_idx),
16933            &links,
16934            track_order.links,
16935        );
16936
16937        // Attribute Info (type 0x15) + attributes (type 0x0C) -- e.g. NeXus
16938        // `NX_class`.
16939        let format = self.header_format(track_order);
16940        self.emit_attributes(
16941            &mut header,
16942            AttrScope::Group(group_idx),
16943            &attributes,
16944            track_order.attrs,
16945            format,
16946            owner,
16947        );
16948
16949        self.emit_refcount(
16950            &mut header,
16951            self.object_link_count(HardLinkTarget::Group(group_idx)),
16952            format,
16953        );
16954
16955        Ok(header)
16956    }
16957
16958    fn build_root_group_header(&self) -> IoResult<ObjectHeader> {
16959        let mut header = ObjectHeader::new();
16960        touch_oh(
16961            &mut header,
16962            self.header_format(self.root_track_order),
16963            self.root_times,
16964            false,
16965        );
16966
16967        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
16968        // themselves, compact or dense.
16969        let links = self.group_links(LinkScope::Root, self.root_track_order.links);
16970        self.emit_links(
16971            &mut header,
16972            LinkScope::Root,
16973            &links,
16974            self.root_track_order.links,
16975        );
16976
16977        // Root-level attributes
16978        let root_attributes = self.object_attributes(AttrScope::Root)?;
16979        self.emit_attributes(
16980            &mut header,
16981            AttrScope::Root,
16982            &root_attributes,
16983            self.root_track_order.attrs,
16984            self.header_format(self.root_track_order),
16985            ShareOwner::Header(self.root_group_addr.unwrap_or(0)),
16986        );
16987
16988        Ok(header)
16989    }
16990}
16991
16992impl Drop for Hdf5Writer {
16993    fn drop(&mut self) {
16994        if !self.closed {
16995            // Best-effort finalize on drop. Drop cannot return a Result, so a
16996            // failure here is otherwise invisible: it would leave a truncated
16997            // or unflushed file on disk while the caller believes the write
16998            // succeeded. Surface it on stderr instead of swallowing it.
16999            // Callers that need to handle the error must call
17000            // `H5File::close()` explicitly, which returns the Result.
17001            if let Err(e) = self.finalize(true) {
17002                eprintln!(
17003                    "rust-hdf5: failed to finalize HDF5 file on drop: {e}. \
17004                     The file may be incomplete or corrupt; call \
17005                     H5File::close() to handle this error explicitly."
17006                );
17007            }
17008        }
17009    }
17010}
17011
17012#[cfg(test)]
17013mod tests {
17014    use super::*;
17015    use crate::format::messages::datatype::DatatypeMessage;
17016    use crate::io::reader::Hdf5Reader;
17017
17018    fn fixture(name: &str) -> std::path::PathBuf {
17019        std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
17020            .join("tests/fixtures")
17021            .join(name)
17022    }
17023
17024    /// Copy a fixture so a test that appends does not edit the checked-in file.
17025    fn fixture_copy(name: &str, tag: &str) -> std::path::PathBuf {
17026        let path = temp_path(tag);
17027        std::fs::copy(fixture(name), &path).unwrap();
17028        path
17029    }
17030
17031    fn temp_path(tag: &str) -> std::path::PathBuf {
17032        use std::sync::atomic::{AtomicU64, Ordering};
17033        static COUNTER: AtomicU64 = AtomicU64::new(0);
17034        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
17035        std::env::temp_dir().join(format!(
17036            "rust_hdf5_w_{}_{}_{}.h5",
17037            std::process::id(),
17038            tag,
17039            n
17040        ))
17041    }
17042
17043    /// A group past the link phase change keeps its links in a fractal heap
17044    /// with a v2 B-tree name index. The reopen that rewrites that group's
17045    /// header lays a fresh pair out, so both blocks the old header named must
17046    /// come back to the allocator — every block of the heap, and the index
17047    /// header with its nodes.
17048    ///
17049    /// Asserted on the free list rather than on the file size: a reopen does
17050    /// not yet carry dense links forward, so the rewritten group's links (and
17051    /// the datasets they name) are dropped, and the file size that follows
17052    /// says more about that than about this.
17053    #[test]
17054    fn a_reopen_frees_the_dense_link_storage_its_rewrite_supersedes() {
17055        let path = temp_path("dense_link_reclaim");
17056
17057        let writer = Hdf5Writer::create(&path).unwrap();
17058        writer.create_group("/", "run").unwrap();
17059        for i in 0..12 {
17060            writer
17061                .create_dataset(&format!("run/d{i:02}"), DatatypeMessage::i32_type(), &[2])
17062                .unwrap();
17063        }
17064        writer.close().unwrap();
17065
17066        let writer = Hdf5Writer::open_append(&path).unwrap();
17067        let gidx = (0..writer.group_count())
17068            .find(|&g| writer.grp(g).lock().name == "/run")
17069            .expect("the reopen registered the group");
17070        let linfo = writer
17071            .superseded_dense
17072            .lock()
17073            .as_ref()
17074            .and_then(|s| s.links.get(&LinkScope::Group(gidx)).cloned())
17075            .expect("the reopen recorded the group's dense link storage");
17076        assert_ne!(linfo.fractal_heap_address, UNDEF_ADDR);
17077        assert_ne!(linfo.name_btree_address, UNDEF_ADDR);
17078
17079        writer
17080            .release_superseded_dense_links(LinkScope::Group(gidx))
17081            .unwrap();
17082        let freed = writer.allocator.free_blocks();
17083        let covers = |addr: u64| {
17084            freed
17085                .iter()
17086                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17087        };
17088        assert!(covers(linfo.fractal_heap_address), "heap header: {freed:?}");
17089        assert!(covers(linfo.name_btree_address), "name index: {freed:?}");
17090
17091        // And exactly once: the entry is gone, so the finalize that follows
17092        // cannot hand the same blocks back a second time.
17093        assert!(writer
17094            .superseded_dense
17095            .lock()
17096            .as_ref()
17097            .is_none_or(|s| s.links.is_empty()));
17098        writer
17099            .release_superseded_dense_links(LinkScope::Group(gidx))
17100            .unwrap();
17101        assert_eq!(writer.allocator.free_blocks(), freed);
17102
17103        writer.close().unwrap();
17104        std::fs::remove_file(&path).ok();
17105    }
17106
17107    /// The rewrite frees what it supersedes even when the replacement is not
17108    /// dense at all. An attribute set that drops back under `max_compact`
17109    /// goes into the object header, so nothing names the old heap any more —
17110    /// and a free driven by "the new set needs dense storage" would never
17111    /// reach this one.
17112    #[test]
17113    fn a_rewrite_that_drops_out_of_dense_storage_still_frees_it() {
17114        let path = temp_path("dense_attr_to_compact");
17115        let numeric = |name: &str| {
17116            AttributeMessage::scalar_numeric(
17117                name,
17118                DatatypeMessage::i32_type(),
17119                7i32.to_le_bytes().to_vec(),
17120            )
17121        };
17122
17123        let writer = Hdf5Writer::create(&path).unwrap();
17124        for i in 0..12 {
17125            writer
17126                .add_root_attribute(numeric(&format!("a{i:02}")))
17127                .unwrap();
17128        }
17129        writer.close().unwrap();
17130
17131        let writer = Hdf5Writer::open_append(&path).unwrap();
17132        let ainfo = writer
17133            .superseded_dense
17134            .lock()
17135            .as_ref()
17136            .and_then(|s| s.attrs.get(&AttrScope::Root).cloned())
17137            .expect("the reopen recorded the root's dense attribute storage");
17138        for i in 0..10 {
17139            writer
17140                .evict_attr(AttrTarget::Root, &format!("a{i:02}"))
17141                .unwrap();
17142        }
17143        assert!(!writer.attributes_need_dense(&writer.root_attributes.lock(), ObjectFormat::Modern));
17144
17145        writer.prepare_dense_attributes(&[]).unwrap();
17146        let freed = writer.allocator.free_blocks();
17147        let covers = |addr: u64| {
17148            freed
17149                .iter()
17150                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17151        };
17152        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17153        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17154        assert!(writer
17155            .superseded_dense
17156            .lock()
17157            .as_ref()
17158            .is_none_or(|s| s.attrs.is_empty()));
17159
17160        writer.close().unwrap();
17161        std::fs::remove_file(&path).ok();
17162    }
17163
17164    /// Deleting a reopened object supersedes its dense storage as surely as
17165    /// rewriting one does: nothing in the finalized file names the heap, so
17166    /// the delete owner frees it through the same entry.
17167    #[test]
17168    fn deleting_a_reopened_group_frees_its_dense_attribute_storage() {
17169        let path = temp_path("dense_attr_delete");
17170        let numeric = |name: &str| {
17171            AttributeMessage::scalar_numeric(
17172                name,
17173                DatatypeMessage::i32_type(),
17174                7i32.to_le_bytes().to_vec(),
17175            )
17176        };
17177
17178        let writer = Hdf5Writer::create(&path).unwrap();
17179        writer.create_group("/", "run").unwrap();
17180        for i in 0..12 {
17181            writer
17182                .set_attribute(AttrTarget::Group("/run"), numeric(&format!("a{i:02}")))
17183                .unwrap();
17184        }
17185        writer.close().unwrap();
17186
17187        let writer = Hdf5Writer::open_append(&path).unwrap();
17188        let gidx = (0..writer.group_count())
17189            .find(|&g| writer.grp(g).lock().name == "/run")
17190            .expect("the reopen registered the group");
17191        let ainfo = writer
17192            .superseded_dense
17193            .lock()
17194            .as_ref()
17195            .and_then(|s| s.attrs.get(&AttrScope::Group(gidx)).cloned())
17196            .expect("the reopen recorded the group's dense attribute storage");
17197
17198        writer.delete_group("/run").unwrap();
17199        let freed = writer.allocator.free_blocks();
17200        let covers = |addr: u64| {
17201            freed
17202                .iter()
17203                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17204        };
17205        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17206        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17207        assert!(writer
17208            .superseded_dense
17209            .lock()
17210            .as_ref()
17211            .is_none_or(|s| s.attrs.is_empty()));
17212
17213        writer.close().unwrap();
17214        std::fs::remove_file(&path).ok();
17215    }
17216
17217    /// The charset rule is one owner shared by every vlen string writer:
17218    /// appends into an ASCII-declared dataset reject non-ASCII strings the
17219    /// same way the slice writer does, and a dataset whose elements are not
17220    /// vlen references at all is refused instead of overwritten with them.
17221    #[test]
17222    fn append_vlen_strings_checks_the_datatype_and_charset() {
17223        let path = temp_path("append_vlen_charset");
17224
17225        let writer = Hdf5Writer::create(&path).unwrap();
17226        let idx = writer
17227            .create_appendable_vlen_string_dataset("d", 4, None)
17228            .unwrap();
17229        writer.ds(idx).lock().datatype = DatatypeMessage::vlen_string_ascii();
17230        let err = writer
17231            .append_vlen_strings(idx, &["ok", "안녕"])
17232            .unwrap_err();
17233        assert!(
17234            err.to_string().contains("is not ASCII"),
17235            "unexpected error: {err}"
17236        );
17237        writer.append_vlen_strings(idx, &["ok", "fine"]).unwrap();
17238
17239        let nums = writer
17240            .create_chunked_dataset("n", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
17241            .unwrap();
17242        let err = writer.append_vlen_strings(nums, &["x"]).unwrap_err();
17243        assert!(
17244            err.to_string()
17245                .contains("only for variable-length string datasets"),
17246            "unexpected error: {err}"
17247        );
17248
17249        writer.close().unwrap();
17250        std::fs::remove_file(&path).ok();
17251    }
17252
17253    /// `create_chunked_dataset` builds an extensible-array index unconditionally
17254    /// (the caller — the high-level dataset API — is the one that decides when
17255    /// two-or-more unlimited dimensions should go to a v2 B-tree instead), so
17256    /// its own guard is the last line of defense against a shape that index
17257    /// can't represent at all.
17258    #[test]
17259    fn create_chunked_dataset_rejects_two_unlimited_dimensions() {
17260        let path = temp_path("earray_two_unlimited");
17261        let writer = Hdf5Writer::create(&path).unwrap();
17262        let err = writer
17263            .create_chunked_dataset(
17264                "d",
17265                DatatypeMessage::i32_type(),
17266                &[4, 4],
17267                &[u64::MAX, u64::MAX],
17268                &[2, 2],
17269            )
17270            .unwrap_err();
17271        assert!(err.to_string().contains("at most one unlimited"), "{err}");
17272        writer.close().unwrap();
17273        std::fs::remove_file(&path).ok();
17274    }
17275
17276    /// Every creator must enter through `begin_create`; the four that used
17277    /// to bypass it could push a second dataset under an existing name and
17278    /// emit an invalid file with two same-named links.
17279    #[test]
17280    fn every_creator_rejects_an_existing_dataset_name() {
17281        let path = temp_path("create_gate");
17282
17283        let writer = Hdf5Writer::create(&path).unwrap();
17284        writer
17285            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
17286            .unwrap();
17287
17288        let attempts: [(&str, IoResult<usize>); 4] = [
17289            (
17290                "vlen_string",
17291                writer.create_vlen_string_dataset("d", &["x"], 1),
17292            ),
17293            ("vlen_bytes", writer.create_vlen_bytes_dataset("d", &[b"x"])),
17294            (
17295                "vlen_string_compressed",
17296                writer.create_vlen_string_dataset_compressed(
17297                    "d",
17298                    &["x"],
17299                    1,
17300                    FilterPipeline::deflate(6),
17301                ),
17302            ),
17303            (
17304                "chunked_with_pipeline",
17305                writer.create_chunked_dataset_with_pipeline(
17306                    "d",
17307                    DatatypeMessage::i32_type(),
17308                    &[0],
17309                    &[u64::MAX],
17310                    &[4],
17311                    FilterPipeline::deflate(6),
17312                ),
17313            ),
17314        ];
17315        for (which, res) in attempts {
17316            match res {
17317                Ok(_) => panic!("{which} accepted a duplicate name"),
17318                Err(e) => assert!(
17319                    e.to_string().contains("already exists"),
17320                    "{which}: unexpected error: {e}"
17321                ),
17322            }
17323        }
17324
17325        writer.close().unwrap();
17326        std::fs::remove_file(&path).ok();
17327    }
17328
17329    /// Every creator and every kind of name meet at `ensure_name_free`.
17330    ///
17331    /// The gate's whole value is that it is one list: a creator must be
17332    /// blind neither to a name kind it does not itself make nor to one added
17333    /// after it. This crosses the two — six names, one of each kind the
17334    /// writer can put in a group, against every creator — so a creator that
17335    /// grows its own check, or a name kind that stops being on the list,
17336    /// fails here rather than in a file holding two links of one name.
17337    #[test]
17338    fn every_creator_refuses_every_kind_of_taken_name() {
17339        let path = temp_path("create_gate_matrix");
17340        let writer = Hdf5Writer::create(&path).unwrap();
17341
17342        let i32t = || DatatypeMessage::i32_type();
17343        writer.create_dataset("d", i32t(), &[2]).unwrap();
17344        writer.create_compact_dataset("c", i32t(), &[2]).unwrap();
17345        writer.create_group("/", "g").unwrap();
17346        writer.commit_datatype("t", i32t()).unwrap();
17347        writer.create_hard_link("/", "h", "d").unwrap();
17348        writer
17349            .create_symbolic_link(
17350                "/",
17351                "s",
17352                LinkTarget::Soft {
17353                    target: "/d".into(),
17354                },
17355            )
17356            .unwrap();
17357        writer
17358            .create_symbolic_link(
17359                "/",
17360                "e",
17361                LinkTarget::External {
17362                    file: "other.h5".into(),
17363                    path: "/x".into(),
17364                },
17365            )
17366            .unwrap();
17367
17368        for taken in ["d", "c", "g", "t", "h", "s", "e"] {
17369            let attempts: [(&str, IoResult<()>); 8] = [
17370                (
17371                    "dataset",
17372                    writer.create_dataset(taken, i32t(), &[2]).map(|_| ()),
17373                ),
17374                (
17375                    "compact",
17376                    writer
17377                        .create_compact_dataset(taken, i32t(), &[2])
17378                        .map(|_| ()),
17379                ),
17380                (
17381                    "chunked",
17382                    writer
17383                        .create_chunked_dataset(taken, i32t(), &[0], &[u64::MAX], &[4])
17384                        .map(|_| ()),
17385                ),
17386                (
17387                    "vlen_string",
17388                    writer
17389                        .create_vlen_string_dataset(taken, &["x"], 1)
17390                        .map(|_| ()),
17391                ),
17392                (
17393                    "committed datatype",
17394                    writer.commit_datatype(taken, i32t()).map(|_| ()),
17395                ),
17396                ("group", writer.create_group("/", taken).map(|_| ())),
17397                ("hard link", writer.create_hard_link("/", taken, "d")),
17398                (
17399                    "soft link",
17400                    writer.create_symbolic_link(
17401                        "/",
17402                        taken,
17403                        LinkTarget::Soft {
17404                            target: "/d".into(),
17405                        },
17406                    ),
17407                ),
17408            ];
17409            for (which, res) in attempts {
17410                match res {
17411                    Ok(()) => panic!("{which} accepted the taken name '{taken}'"),
17412                    Err(e) => assert!(
17413                        e.to_string().contains("already exists"),
17414                        "{which} on '{taken}': unexpected error: {e}"
17415                    ),
17416                }
17417            }
17418        }
17419
17420        writer.close().unwrap();
17421        std::fs::remove_file(&path).ok();
17422    }
17423
17424    /// The `H5T_VLEN` length field counts base elements, so an image that is
17425    /// not a whole number of them has no length that reads back as what was
17426    /// handed over; it is refused at the call rather than stored truncated.
17427    #[test]
17428    fn vlen_sequence_refuses_a_partial_element() {
17429        let path = temp_path("vlen_partial_element");
17430
17431        let writer = Hdf5Writer::create(&path).unwrap();
17432        let err = writer
17433            .create_vlen_sequence_dataset("d", DatatypeMessage::i32_type(), &[&[1u8, 2, 3, 4, 5]])
17434            .unwrap_err()
17435            .to_string();
17436        assert!(err.contains("5 bytes"), "unexpected error: {err}");
17437        assert!(err.contains("4-byte elements"), "unexpected error: {err}");
17438
17439        // The refusal is the length rule alone: the same base takes a whole
17440        // number of elements, and an empty sequence is a legal one.
17441        writer
17442            .create_vlen_sequence_dataset(
17443                "d",
17444                DatatypeMessage::i32_type(),
17445                &[&[1u8, 2, 3, 4], &[][..]],
17446            )
17447            .unwrap();
17448
17449        writer.close().unwrap();
17450        std::fs::remove_file(&path).ok();
17451    }
17452
17453    /// A corrupt file can declare a zero-length chunk dimension; the
17454    /// superseded-reference read must reject it the way `write_slice` does,
17455    /// not divide by it.
17456    #[test]
17457    fn vlen_slice_rejects_a_zero_chunk_dimension() {
17458        let path = temp_path("vlen_slice_zero_chunk");
17459
17460        let writer = Hdf5Writer::create(&path).unwrap();
17461        let idx = writer
17462            .create_appendable_vlen_string_dataset("d", 2, None)
17463            .unwrap();
17464        writer.append_vlen_strings(idx, &["a", "b"]).unwrap();
17465        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 0;
17466        let err = writer.write_vlen_strings_slice(idx, 0, &["x"]).unwrap_err();
17467        assert!(
17468            err.to_string().contains("zero-length dimension"),
17469            "unexpected error: {err}"
17470        );
17471
17472        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 2;
17473        writer.close().unwrap();
17474        std::fs::remove_file(&path).ok();
17475    }
17476
17477    /// A libhdf5-written collection can be 100% full — no free-space marker,
17478    /// content exactly the declared size. When a stale reference names an
17479    /// index that is not there, nothing is removed, and the collection must
17480    /// be left alone: re-encoding it at its declared size cannot fit the
17481    /// free-space marker and would fail the whole update.
17482    #[test]
17483    fn release_leaves_a_full_collection_it_removed_nothing_from() {
17484        use crate::format::global_heap::encode_vlen_reference;
17485
17486        let path = temp_path("release_full_gcol");
17487        let writer = Hdf5Writer::create(&path).unwrap();
17488
17489        // Hand-built full collection: 16-byte header + one 16+8-byte object,
17490        // declared size exactly 40, no free-space marker.
17491        let mut img = Vec::new();
17492        img.extend_from_slice(b"GCOL");
17493        img.push(1);
17494        img.extend_from_slice(&[0u8; 3]);
17495        img.extend_from_slice(&40u64.to_le_bytes());
17496        img.extend_from_slice(&1u16.to_le_bytes()); // object index 1
17497        img.extend_from_slice(&1u16.to_le_bytes()); // ref_count
17498        img.extend_from_slice(&0u32.to_le_bytes()); // reserved
17499        img.extend_from_slice(&8u64.to_le_bytes()); // data size
17500        img.extend_from_slice(b"deadbeef");
17501        assert_eq!(img.len(), 40);
17502        let addr = writer
17503            .allocator
17504            .allocate(img.len() as u64, FreeSpaceClass::RawData);
17505        writer.handle.write_at(addr, &img).unwrap();
17506
17507        // The superseded reference names index 2, which the collection does
17508        // not hold — a no-op removal.
17509        let refs = encode_vlen_reference(3, addr, 2, &writer.ctx);
17510        writer.release_vlen_references(&refs).unwrap();
17511        assert_eq!(writer.handle.read_at(addr, 40).unwrap(), img);
17512
17513        writer.close().unwrap();
17514        std::fs::remove_file(&path).ok();
17515    }
17516
17517    /// The CWFS second pass (`H5F_cwfs_find_free_heap`): an object too big
17518    /// for the listed collection's remaining free space extends the
17519    /// collection in place — the file allocation grows off the end of the
17520    /// file (`H5MF_try_extend`) and the collection's declared size and
17521    /// free-space marker grow with it (`H5HG_extend`) — instead of opening
17522    /// a second collection.
17523    #[test]
17524    fn an_oversized_vlen_insert_extends_the_listed_collection() {
17525        use crate::format::global_heap::GlobalHeapCollection;
17526
17527        let path = temp_path("cwfs_extend_tail");
17528        let writer = Hdf5Writer::create(&path).unwrap();
17529        // A small object opens a minimum-size (4096) listed collection —
17530        // the file's last allocation, so the extension grows the file end.
17531        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
17532        let big = vec![0x41u8; 5000]; // more than the ~4 KiB remaining
17533        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
17534        assert_eq!(
17535            p2[0].0, p1[0].0,
17536            "the big object opened a second collection"
17537        );
17538
17539        // The block on disk is one grown collection holding both objects.
17540        let img = writer.handle.read_at_most(p1[0].0, 65536).unwrap();
17541        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
17542        assert!(csize > 4096, "declared size did not grow: {csize}");
17543        assert_eq!(gcol.objects.len(), 2);
17544        assert_eq!(gcol.objects[1].data, big);
17545
17546        writer.close().unwrap();
17547        let bytes = std::fs::read(&path).unwrap();
17548        assert_eq!(
17549            bytes.windows(4).filter(|w| *w == b"GCOL").count(),
17550            1,
17551            "a second collection signature is in the file"
17552        );
17553        std::fs::remove_file(&path).ok();
17554    }
17555
17556    /// The non-tail counterpart: the collection is pinned away from the end
17557    /// of the file, but a released block starts right after it, so the
17558    /// extension consumes the front of that block (`H5MF_try_extend`'s
17559    /// free-section path) and the remainder stays reusable.
17560    #[test]
17561    fn extension_consumes_a_freed_block_after_the_collection() {
17562        use crate::format::global_heap::GlobalHeapCollection;
17563
17564        let path = temp_path("cwfs_extend_freed");
17565        let writer = Hdf5Writer::create(&path).unwrap();
17566        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
17567        let addr = p1[0].0;
17568        // Land a block right after the collection, pin the file end past
17569        // it, then release it: extension must use the released space.
17570        let spacer = writer.allocator.allocate(8192, FreeSpaceClass::RawData);
17571        assert_eq!(spacer, addr + 4096, "spacer not adjacent; layout changed");
17572        writer.allocator.allocate(8, FreeSpaceClass::RawData);
17573        writer.allocator.free(spacer, 8192, FreeSpaceClass::RawData);
17574
17575        let big = vec![0x42u8; 5000];
17576        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
17577        assert_eq!(p2[0].0, addr, "the big object opened a second collection");
17578
17579        let img = writer.handle.read_at_most(addr, 65536).unwrap();
17580        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
17581        assert_eq!(csize, 8192, "grew by max(size, shortfall) = 4096");
17582        assert_eq!(gcol.objects.len(), 2);
17583
17584        // The remainder of the released block is still allocatable.
17585        assert_eq!(
17586            writer.allocator.allocate(4096, FreeSpaceClass::RawData),
17587            addr + 8192,
17588            "the freed block's tail was lost"
17589        );
17590        writer.close().unwrap();
17591        std::fs::remove_file(&path).ok();
17592    }
17593
17594    /// Issue #10: a reopen-and-replace loop on a vlen string must not grow
17595    /// the file. The superseded heap objects are freed *before* the
17596    /// replacement is allocated, so each session reuses the block it just
17597    /// released even though the free list starts empty on reopen. The old
17598    /// free-after-alloc order failed this by one collection per session.
17599    #[test]
17600    fn vlen_replace_across_reopen_keeps_the_file_flat() {
17601        let path = temp_path("vlen_reopen_flat");
17602        let payload_a = "a".repeat(64 * 1024);
17603        let payload_b = "b".repeat(64 * 1024);
17604
17605        let writer = Hdf5Writer::create(&path).unwrap();
17606        writer
17607            .create_vlen_string_dataset("notes", &["initial"], 1)
17608            .unwrap();
17609        writer.close().unwrap();
17610
17611        let mut sizes = Vec::new();
17612        for i in 0..8 {
17613            let writer = Hdf5Writer::open_append(&path).unwrap();
17614            let payload = if i % 2 == 0 { &payload_a } else { &payload_b };
17615            writer
17616                .write_vlen_strings_slice(0, 0, &[payload.as_str()])
17617                .unwrap();
17618            writer.close().unwrap();
17619            sizes.push(std::fs::metadata(&path).unwrap().len());
17620        }
17621        // The first replacement grows the file once (the initial collection
17622        // cannot hold 64 KiB); every later equal-size replacement must land
17623        // in the block its own session just freed.
17624        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
17625
17626        // The reused blocks still form a valid file holding the last value.
17627        let mut reader = Hdf5Reader::open(&path).unwrap();
17628        assert_eq!(
17629            reader.read_vlen_strings("notes").unwrap(),
17630            vec![payload_b.clone()]
17631        );
17632
17633        std::fs::remove_file(&path).ok();
17634    }
17635
17636    /// Replacing a vlen string attribute must release the superseded
17637    /// global-heap collection *before* the replacement's collection is
17638    /// allocated, so a reopen-replace loop lands each new value in the block
17639    /// it just freed instead of growing the file by one collection per
17640    /// session — the attribute counterpart of
17641    /// [`vlen_replace_across_reopen_keeps_the_file_flat`].
17642    #[test]
17643    fn vlen_attr_replace_across_reopen_keeps_the_file_flat() {
17644        let path = temp_path("vlen_attr_reopen_flat");
17645        let payload_a = "a".repeat(8 * 1024);
17646        let payload_b = "b".repeat(8 * 1024);
17647
17648        let writer = Hdf5Writer::create(&path).unwrap();
17649        writer
17650            .set_vlen_string_attribute(AttrTarget::Root, "note", &payload_a)
17651            .unwrap();
17652        writer.close().unwrap();
17653
17654        let mut sizes = Vec::new();
17655        for i in 0..8 {
17656            let writer = Hdf5Writer::open_append(&path).unwrap();
17657            let payload = if i % 2 == 0 { &payload_b } else { &payload_a };
17658            writer
17659                .set_vlen_string_attribute(AttrTarget::Root, "note", payload)
17660                .unwrap();
17661            writer.close().unwrap();
17662            sizes.push(std::fs::metadata(&path).unwrap().len());
17663        }
17664        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
17665
17666        // The reused blocks still hold the last value.
17667        let reader = Hdf5Reader::open(&path).unwrap();
17668        let attr = reader.root_attr("note").unwrap().clone();
17669        let mut reader = reader;
17670        assert_eq!(reader.attr_string_value(&attr).unwrap(), payload_a);
17671
17672        std::fs::remove_file(&path).ok();
17673    }
17674
17675    /// A numeric attribute replacing a vlen one goes through the same list
17676    /// owner, so the superseded collection is released even though the new
17677    /// value holds no heap reference: a later same-size vlen attribute must
17678    /// land in the freed block, making the file exactly as large as one that
17679    /// never stored the replaced value.
17680    #[test]
17681    fn numeric_replacing_a_vlen_attr_releases_its_collection() {
17682        let payload = "x".repeat(8 * 1024);
17683        let numeric = || {
17684            AttributeMessage::scalar_numeric(
17685                "x",
17686                DatatypeMessage::i32_type(),
17687                7i32.to_le_bytes().to_vec(),
17688            )
17689        };
17690
17691        let path_a = temp_path("vlen_attr_cross_a");
17692        let writer = Hdf5Writer::create(&path_a).unwrap();
17693        writer
17694            .set_vlen_string_attribute(AttrTarget::Root, "x", &payload)
17695            .unwrap();
17696        writer.add_root_attribute(numeric()).unwrap();
17697        writer
17698            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
17699            .unwrap();
17700        writer.close().unwrap();
17701
17702        // The same end state written without the replaced vlen value.
17703        let path_b = temp_path("vlen_attr_cross_b");
17704        let writer = Hdf5Writer::create(&path_b).unwrap();
17705        writer.add_root_attribute(numeric()).unwrap();
17706        writer
17707            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
17708            .unwrap();
17709        writer.close().unwrap();
17710
17711        assert_eq!(
17712            std::fs::metadata(&path_a).unwrap().len(),
17713            std::fs::metadata(&path_b).unwrap().len()
17714        );
17715
17716        let reader = Hdf5Reader::open(&path_a).unwrap();
17717        let y = reader.root_attr("y").unwrap().clone();
17718        let mut reader = reader;
17719        assert_eq!(reader.attr_string_value(&y).unwrap(), payload);
17720
17721        std::fs::remove_file(&path_a).ok();
17722        std::fs::remove_file(&path_b).ok();
17723    }
17724
17725    /// Reopen/write/close cycles must not leak the object-header blocks
17726    /// finalize rewrites: the reopened root header, the reopened group
17727    /// header, and the modified chunked dataset's header are each freed
17728    /// before their replacements are allocated. The chunk rewrite itself is
17729    /// in place (unfiltered chunks never move), so a leak of any header
17730    /// block shows up as monotonic growth here.
17731    #[test]
17732    fn reopen_cycles_reuse_superseded_header_blocks() {
17733        let path = temp_path("header_reuse");
17734        {
17735            let writer = Hdf5Writer::create(&path).unwrap();
17736            writer.create_group("/", "g").unwrap();
17737            let idx = writer
17738                .create_chunked_dataset(
17739                    "g/data",
17740                    DatatypeMessage::i32_type(),
17741                    &[4],
17742                    &[u64::MAX],
17743                    &[4],
17744                )
17745                .unwrap();
17746            let seed: Vec<u8> = [1i32, 2, 3, 4]
17747                .iter()
17748                .flat_map(|v| v.to_le_bytes())
17749                .collect();
17750            writer.write_chunk(idx, 0, &seed).unwrap();
17751            writer.close().unwrap();
17752        }
17753
17754        let mut sizes = Vec::new();
17755        for i in 0..6i32 {
17756            let writer = Hdf5Writer::open_append(&path).unwrap();
17757            let data: Vec<u8> = [i; 4].iter().flat_map(|v| v.to_le_bytes()).collect();
17758            writer.write_chunk(0, 0, &data).unwrap();
17759            writer.close().unwrap();
17760            sizes.push(std::fs::metadata(&path).unwrap().len());
17761        }
17762        assert_eq!(&sizes[1..], &vec![sizes[0]; 5][..], "sizes: {sizes:?}");
17763
17764        // The reused header blocks still form a valid file.
17765        let mut reader = Hdf5Reader::open(&path).unwrap();
17766        let raw = reader.read_dataset_raw("g/data").unwrap();
17767        let values: Vec<i32> = raw
17768            .chunks(4)
17769            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
17770            .collect();
17771        assert_eq!(values, vec![5, 5, 5, 5]);
17772
17773        std::fs::remove_file(&path).ok();
17774    }
17775
17776    #[test]
17777    fn create_empty_file() {
17778        let path = temp_path("empty");
17779
17780        let writer = Hdf5Writer::create(&path).unwrap();
17781        writer.close().unwrap();
17782
17783        // Verify we can read it back
17784        let reader = Hdf5Reader::open(&path).unwrap();
17785        assert!(reader.dataset_names().is_empty());
17786
17787        std::fs::remove_file(&path).ok();
17788    }
17789
17790    #[test]
17791    fn create_single_dataset() {
17792        let path = temp_path("single");
17793
17794        let writer = Hdf5Writer::create(&path).unwrap();
17795        let idx = writer
17796            .create_dataset("data", DatatypeMessage::f64_type(), &[4])
17797            .unwrap();
17798        let values: Vec<f64> = vec![1.0, 2.0, 3.0, 4.0];
17799        let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17800        writer.write_dataset_raw(idx, &raw).unwrap();
17801        writer.close().unwrap();
17802
17803        // Read back
17804        let mut reader = Hdf5Reader::open(&path).unwrap();
17805        assert_eq!(reader.dataset_names(), vec!["data"]);
17806        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4]);
17807        let readback = reader.read_dataset_raw("data").unwrap();
17808        assert_eq!(readback, raw);
17809
17810        std::fs::remove_file(&path).ok();
17811    }
17812
17813    #[test]
17814    fn create_multiple_datasets() {
17815        let path = temp_path("multi");
17816
17817        let writer = Hdf5Writer::create(&path).unwrap();
17818
17819        let idx0 = writer
17820            .create_dataset("ints", DatatypeMessage::i32_type(), &[3])
17821            .unwrap();
17822        let i_data: Vec<u8> = [10i32, 20, 30]
17823            .iter()
17824            .flat_map(|v| v.to_le_bytes())
17825            .collect();
17826        writer.write_dataset_raw(idx0, &i_data).unwrap();
17827
17828        let idx1 = writer
17829            .create_dataset("floats", DatatypeMessage::f32_type(), &[2, 2])
17830            .unwrap();
17831        let f_data: Vec<u8> = [1.0f32, 2.0, 3.0, 4.0]
17832            .iter()
17833            .flat_map(|v| v.to_le_bytes())
17834            .collect();
17835        writer.write_dataset_raw(idx1, &f_data).unwrap();
17836
17837        writer.close().unwrap();
17838
17839        let mut reader = Hdf5Reader::open(&path).unwrap();
17840        let names = reader.dataset_names();
17841        assert!(names.contains(&"ints"));
17842        assert!(names.contains(&"floats"));
17843        assert_eq!(reader.dataset_shape("ints").unwrap(), vec![3]);
17844        assert_eq!(reader.dataset_shape("floats").unwrap(), vec![2, 2]);
17845        assert_eq!(reader.read_dataset_raw("ints").unwrap(), i_data);
17846        assert_eq!(reader.read_dataset_raw("floats").unwrap(), f_data);
17847
17848        std::fs::remove_file(&path).ok();
17849    }
17850
17851    #[test]
17852    fn data_size_mismatch() {
17853        let path = temp_path("mismatch");
17854
17855        let writer = Hdf5Writer::create(&path).unwrap();
17856        let idx = writer
17857            .create_dataset("x", DatatypeMessage::u8_type(), &[4])
17858            .unwrap();
17859        let err = writer.write_dataset_raw(idx, &[1, 2, 3]); // 3 bytes instead of 4
17860        assert!(err.is_err());
17861
17862        std::fs::remove_file(&path).ok();
17863    }
17864
17865    #[test]
17866    fn create_chunked_dataset_simple() {
17867        let path = temp_path("chunked_simple");
17868
17869        let writer = Hdf5Writer::create(&path).unwrap();
17870        let idx = writer
17871            .create_chunked_dataset(
17872                "data",
17873                DatatypeMessage::f64_type(),
17874                &[0, 4],        // start empty
17875                &[u64::MAX, 4], // unlimited first dim
17876                &[1, 4],        // chunk = [1, 4]
17877            )
17878            .unwrap();
17879
17880        // Write 3 frames (chunks)
17881        for frame in 0..3u64 {
17882            let values: Vec<f64> = (0..4).map(|i| (frame * 4 + i) as f64).collect();
17883            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17884            writer.write_chunk(idx, frame, &raw).unwrap();
17885        }
17886
17887        // Extend dimensions
17888        writer.extend_dataset(idx, &[3, 4]).unwrap();
17889
17890        writer.close().unwrap();
17891
17892        // Read back
17893        let mut reader = Hdf5Reader::open(&path).unwrap();
17894        assert_eq!(reader.dataset_names(), vec!["data"]);
17895        assert_eq!(reader.dataset_shape("data").unwrap(), vec![3, 4]);
17896
17897        let raw = reader.read_dataset_raw("data").unwrap();
17898        let values: Vec<f64> = raw
17899            .chunks(8)
17900            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
17901            .collect();
17902        assert_eq!(values.len(), 12);
17903        for (i, val) in values.iter().enumerate() {
17904            assert_eq!(*val, i as f64);
17905        }
17906
17907        std::fs::remove_file(&path).ok();
17908    }
17909
17910    #[test]
17911    fn chunked_dataset_many_frames() {
17912        let path = temp_path("chunked_many");
17913
17914        let writer = Hdf5Writer::create(&path).unwrap();
17915        let idx = writer
17916            .create_chunked_dataset(
17917                "frames",
17918                DatatypeMessage::i32_type(),
17919                &[0, 2],
17920                &[u64::MAX, 2],
17921                &[1, 2],
17922            )
17923            .unwrap();
17924
17925        let n_frames = 10u64;
17926        for frame in 0..n_frames {
17927            let values = [(frame * 2) as i32, (frame * 2 + 1) as i32];
17928            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17929            writer.write_chunk(idx, frame, &raw).unwrap();
17930        }
17931
17932        writer.extend_dataset(idx, &[n_frames, 2]).unwrap();
17933        writer.close().unwrap();
17934
17935        // Read back
17936        let mut reader = Hdf5Reader::open(&path).unwrap();
17937        assert_eq!(reader.dataset_shape("frames").unwrap(), vec![10, 2]);
17938
17939        let raw = reader.read_dataset_raw("frames").unwrap();
17940        let values: Vec<i32> = raw
17941            .chunks(4)
17942            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
17943            .collect();
17944        assert_eq!(values.len(), 20);
17945        for (i, val) in values.iter().enumerate() {
17946            assert_eq!(*val, i as i32);
17947        }
17948
17949        std::fs::remove_file(&path).ok();
17950    }
17951
17952    #[test]
17953    fn create_fixed_array_dataset_roundtrip() {
17954        let path = temp_path("fixed_array");
17955
17956        let writer = Hdf5Writer::create(&path).unwrap();
17957        let idx = writer
17958            .create_fixed_array_dataset(
17959                "grid",
17960                DatatypeMessage::i32_type(),
17961                &[4, 6], // 4x6 grid
17962                &[2, 3], // chunk = 2x3
17963            )
17964            .unwrap();
17965
17966        // Write all chunks: 2x2 = 4 chunks
17967        // chunk (0,0): rows 0-1, cols 0-2
17968        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
17969            .iter()
17970            .flat_map(|v| v.to_le_bytes())
17971            .collect();
17972        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
17973
17974        // chunk (0,1): rows 0-1, cols 3-5
17975        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
17976            .iter()
17977            .flat_map(|v| v.to_le_bytes())
17978            .collect();
17979        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
17980
17981        // chunk (1,0): rows 2-3, cols 0-2
17982        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
17983            .iter()
17984            .flat_map(|v| v.to_le_bytes())
17985            .collect();
17986        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
17987
17988        // chunk (1,1): rows 2-3, cols 3-5
17989        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
17990            .iter()
17991            .flat_map(|v| v.to_le_bytes())
17992            .collect();
17993        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
17994
17995        writer.close().unwrap();
17996
17997        // Read back
17998        let mut reader = Hdf5Reader::open(&path).unwrap();
17999        assert_eq!(reader.dataset_names(), vec!["grid"]);
18000        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18001
18002        let raw = reader.read_dataset_raw("grid").unwrap();
18003        let values: Vec<i32> = raw
18004            .chunks(4)
18005            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18006            .collect();
18007        assert_eq!(values.len(), 24);
18008        for (i, val) in values.iter().enumerate() {
18009            assert_eq!(*val, i as i32);
18010        }
18011
18012        std::fs::remove_file(&path).ok();
18013    }
18014
18015    #[test]
18016    fn fixed_array_paged_dblk_disk_size() {
18017        let ctx = FormatContext {
18018            sizeof_addr: 8,
18019            sizeof_size: 8,
18020        };
18021        // 1024 elements per page (bits=10). 3000 chunks => 3 pages.
18022        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 3000);
18023        assert!(hdr.is_paged());
18024        assert_eq!(hdr.npages(), 3);
18025        // prefix: 4+1+1+8 + bitmap(1) + cksum(4) = 19
18026        // elements: 3000 * 8 = 24000 ; per-page cksum: 3 * 4 = 12
18027        assert_eq!(fixed_array_dblk_disk_size(&ctx, &hdr), 19 + 24000 + 12);
18028
18029        // Non-paged: 1000 elements. prefix(14) + 1000*8 + cksum(4).
18030        let small = FixedArrayHeader::new_for_chunks(&ctx, 1000);
18031        assert!(!small.is_paged());
18032        assert_eq!(fixed_array_dblk_disk_size(&ctx, &small), 14 + 8000 + 4);
18033    }
18034
18035    #[test]
18036    fn fixed_array_paged_encode_matches_reader_layout() {
18037        let ctx = FormatContext {
18038            sizeof_addr: 8,
18039            sizeof_size: 8,
18040        };
18041        let mut hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18042        hdr.data_blk_addr = 0x9000;
18043        let npages = hdr.npages() as usize; // ceil(2500/1024) = 3
18044
18045        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18046        for (i, e) in dblk.elements.iter_mut().enumerate() {
18047            *e = 0x10000 + (i as u64) * 0x100;
18048        }
18049
18050        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18051        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18052
18053        // Decode the prefix and pages exactly as the reader does.
18054        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18055        assert_eq!(prefix.header_addr, 0x1000);
18056        for p in 0..npages {
18057            assert!(prefix.page_initialized(p), "page {p} should be initialized");
18058        }
18059
18060        let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
18061        let page_stride = dblk_page_nelmts * 8 + 4;
18062        let mut recovered = Vec::new();
18063        for p in 0..npages {
18064            let page_nelmts = if p + 1 == npages {
18065                2500 - p * dblk_page_nelmts
18066            } else {
18067                dblk_page_nelmts
18068            };
18069            let off = prefix.prefix_size + p * page_stride;
18070            let page_buf = &encoded[off..];
18071            let addrs = crate::format::chunk_index::fixed_array::decode_unfiltered_page(
18072                page_buf,
18073                &ctx,
18074                page_nelmts,
18075            )
18076            .unwrap();
18077            recovered.extend(addrs);
18078        }
18079        assert_eq!(recovered, dblk.elements);
18080    }
18081
18082    #[test]
18083    fn fixed_array_paged_decode_roundtrip_with_uninitialized_page() {
18084        let ctx = FormatContext {
18085            sizeof_addr: 8,
18086            sizeof_size: 8,
18087        };
18088        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18089        let npages = hdr.npages() as usize; // 3
18090        let page = hdr.dblk_page_nelmts() as usize; // 1024
18091
18092        // Populate pages 0 and 2; leave page 1 entirely undefined so its
18093        // bitmap bit stays clear on encode.
18094        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18095        for i in (0..page).chain(2 * page..2500) {
18096            dblk.elements[i] = 0x10000 + (i as u64) * 0x100;
18097        }
18098
18099        let mut encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18100        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18101        assert!(prefix.page_initialized(0));
18102        assert!(!prefix.page_initialized(1));
18103        assert!(prefix.page_initialized(2));
18104
18105        // Corrupt the uninitialized page's bytes the way libhdf5 leaves
18106        // them: arbitrary, no valid checksum. Decode must not look at it.
18107        let page_stride = page * 8 + 4;
18108        let p1 = prefix.prefix_size + page_stride;
18109        for b in &mut encoded[p1..p1 + page_stride] {
18110            *b = 0x5A;
18111        }
18112
18113        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, 0).unwrap();
18114        assert_eq!(decoded.elements, dblk.elements);
18115        assert_eq!(decoded.header_addr, 0x1000);
18116    }
18117
18118    #[test]
18119    fn fixed_array_paged_decode_filtered_roundtrip() {
18120        let ctx = FormatContext {
18121            sizeof_addr: 8,
18122            sizeof_size: 8,
18123        };
18124        let chunk_size_len = 4usize;
18125        let hdr = FixedArrayHeader::new_for_filtered_chunks(&ctx, 1500, chunk_size_len as u8);
18126        assert!(hdr.is_paged());
18127
18128        let mut dblk = FixedArrayDataBlock::new_filtered(0x2000, 1500);
18129        for (i, e) in dblk.filtered_elements.iter_mut().enumerate() {
18130            e.address = 0x8000 + (i as u64) * 0x40;
18131            e.chunk_size = 100 + i as u64;
18132            e.filter_mask = (i % 3) as u32;
18133        }
18134
18135        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18136        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18137        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, chunk_size_len).unwrap();
18138        assert_eq!(decoded.filtered_elements, dblk.filtered_elements);
18139        assert_eq!(decoded.client_id, FA_CLIENT_FILT_CHUNK);
18140    }
18141
18142    #[test]
18143    fn create_fixed_array_paged_dataset_roundtrip() {
18144        let path = temp_path("fixed_array_paged");
18145
18146        // 1D dataset of 3000 elements, chunk size 1 => 3000 chunks.
18147        // 3000 > 1024 (one page) => the FA data block must be paged.
18148        let n: usize = 3000;
18149        let writer = Hdf5Writer::create(&path).unwrap();
18150        let idx = writer
18151            .create_fixed_array_dataset("paged", DatatypeMessage::i32_type(), &[n as u64], &[1])
18152            .unwrap();
18153
18154        for i in 0..n {
18155            let v = (i as i32).to_le_bytes();
18156            writer
18157                .write_chunk_fixed_array(idx, &[i as u64], &v)
18158                .unwrap();
18159        }
18160        writer.close().unwrap();
18161
18162        let mut reader = Hdf5Reader::open(&path).unwrap();
18163        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18164        let raw = reader.read_dataset_raw("paged").unwrap();
18165        let values: Vec<i32> = raw
18166            .chunks(4)
18167            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18168            .collect();
18169        assert_eq!(values.len(), n);
18170        for (i, v) in values.iter().enumerate() {
18171            assert_eq!(*v, i as i32, "element {i}");
18172        }
18173
18174        std::fs::remove_file(&path).ok();
18175    }
18176
18177    #[cfg(feature = "deflate")]
18178    #[test]
18179    fn create_filtered_fixed_array_dataset_roundtrip() {
18180        // Small compressed fixed-shape chunked dataset: flat filtered FA.
18181        let path = temp_path("fixed_array_filt");
18182
18183        let writer = Hdf5Writer::create(&path).unwrap();
18184        let idx = writer
18185            .create_fixed_array_dataset_with_pipeline(
18186                "grid",
18187                DatatypeMessage::i32_type(),
18188                &[4, 6], // 4x6 grid
18189                &[2, 3], // chunk = 2x3 => 2x2 = 4 chunks
18190                FilterPipeline::deflate(6),
18191            )
18192            .unwrap();
18193
18194        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18195            .iter()
18196            .flat_map(|v| v.to_le_bytes())
18197            .collect();
18198        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18199        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18200            .iter()
18201            .flat_map(|v| v.to_le_bytes())
18202            .collect();
18203        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
18204        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
18205            .iter()
18206            .flat_map(|v| v.to_le_bytes())
18207            .collect();
18208        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
18209        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
18210            .iter()
18211            .flat_map(|v| v.to_le_bytes())
18212            .collect();
18213        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
18214
18215        writer.close().unwrap();
18216
18217        let mut reader = Hdf5Reader::open(&path).unwrap();
18218        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18219        let raw = reader.read_dataset_raw("grid").unwrap();
18220        let values: Vec<i32> = raw
18221            .chunks(4)
18222            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18223            .collect();
18224        assert_eq!(values.len(), 24);
18225        for (i, v) in values.iter().enumerate() {
18226            assert_eq!(*v, i as i32, "element {i}");
18227        }
18228
18229        std::fs::remove_file(&path).ok();
18230    }
18231
18232    #[cfg(feature = "deflate")]
18233    #[test]
18234    fn create_filtered_fixed_array_paged_dataset_roundtrip() {
18235        // Large compressed fixed-shape chunked dataset (>1024 chunks): the
18236        // filtered FA data block must be paged.
18237        let path = temp_path("fixed_array_filt_paged");
18238
18239        let n: usize = 3000;
18240        let writer = Hdf5Writer::create(&path).unwrap();
18241        let idx = writer
18242            .create_fixed_array_dataset_with_pipeline(
18243                "paged",
18244                DatatypeMessage::i32_type(),
18245                &[n as u64],
18246                &[1],
18247                FilterPipeline::deflate(6),
18248            )
18249            .unwrap();
18250
18251        for i in 0..n {
18252            let v = (i as i32).to_le_bytes();
18253            writer
18254                .write_chunk_fixed_array(idx, &[i as u64], &v)
18255                .unwrap();
18256        }
18257        writer.close().unwrap();
18258
18259        let mut reader = Hdf5Reader::open(&path).unwrap();
18260        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18261        let raw = reader.read_dataset_raw("paged").unwrap();
18262        let values: Vec<i32> = raw
18263            .chunks(4)
18264            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18265            .collect();
18266        assert_eq!(values.len(), n);
18267        for (i, v) in values.iter().enumerate() {
18268            assert_eq!(*v, i as i32, "element {i}");
18269        }
18270
18271        std::fs::remove_file(&path).ok();
18272    }
18273
18274    #[test]
18275    fn filtered_fixed_array_dblk_disk_size_and_encode() {
18276        // Cross-check filtered FA data-block sizing against the encoded length,
18277        // for both flat and paged layouts.
18278        let ctx = FormatContext {
18279            sizeof_addr: 8,
18280            sizeof_size: 8,
18281        };
18282        let csl = 3u8; // chunk_size_len
18283        let elem_size = 8 + csl as usize + 4; // addr + size + filter_mask
18284
18285        // Flat: 100 chunks. prefix(14) + 100*elem_size + cksum(4).
18286        let mut flat = FixedArrayHeader::new_for_filtered_chunks(&ctx, 100, csl);
18287        flat.data_blk_addr = 0x4000;
18288        assert!(!flat.is_paged());
18289        assert_eq!(
18290            fixed_array_dblk_disk_size(&ctx, &flat),
18291            (14 + 100 * elem_size + 4) as u64
18292        );
18293        let flat_dblk = FixedArrayDataBlock::new_filtered(0x1000, 100);
18294        assert_eq!(
18295            encode_fixed_array_dblk(&ctx, &flat, &flat_dblk).len() as u64,
18296            fixed_array_dblk_disk_size(&ctx, &flat)
18297        );
18298
18299        // Paged: 2500 chunks => 3 pages. prefix(4+1+1+8+1+4=19)
18300        // + 2500*elem_size + 3*cksum(4).
18301        let mut paged = FixedArrayHeader::new_for_filtered_chunks(&ctx, 2500, csl);
18302        paged.data_blk_addr = 0x9000;
18303        assert!(paged.is_paged());
18304        assert_eq!(paged.npages(), 3);
18305        assert_eq!(
18306            fixed_array_dblk_disk_size(&ctx, &paged),
18307            (19 + 2500 * elem_size + 12) as u64
18308        );
18309        let mut paged_dblk = FixedArrayDataBlock::new_filtered(0x1000, 2500);
18310        for (i, e) in paged_dblk.filtered_elements.iter_mut().enumerate() {
18311            e.address = 0x10000 + (i as u64) * 0x100;
18312            e.chunk_size = (i % 200) as u64;
18313        }
18314        let encoded = encode_fixed_array_dblk(&ctx, &paged, &paged_dblk);
18315        assert_eq!(
18316            encoded.len() as u64,
18317            fixed_array_dblk_disk_size(&ctx, &paged)
18318        );
18319
18320        // Decode the paged prefix + pages as the reader does.
18321        let npages = paged.npages() as usize;
18322        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18323        for p in 0..npages {
18324            assert!(prefix.page_initialized(p), "page {p}");
18325        }
18326        let dblk_page_nelmts = paged.dblk_page_nelmts() as usize;
18327        let page_stride = dblk_page_nelmts * elem_size + 4;
18328        let mut recovered = Vec::new();
18329        for p in 0..npages {
18330            let page_nelmts = if p + 1 == npages {
18331                2500 - p * dblk_page_nelmts
18332            } else {
18333                dblk_page_nelmts
18334            };
18335            let off = prefix.prefix_size + p * page_stride;
18336            let elems = crate::format::chunk_index::fixed_array::decode_filtered_page(
18337                &encoded[off..],
18338                &ctx,
18339                page_nelmts,
18340                csl as usize,
18341            )
18342            .unwrap();
18343            recovered.extend(elems);
18344        }
18345        assert_eq!(recovered, paged_dblk.filtered_elements);
18346    }
18347
18348    #[test]
18349    fn create_btree_v2_dataset_roundtrip() {
18350        let path = temp_path("btree_v2");
18351
18352        let writer = Hdf5Writer::create(&path).unwrap();
18353        let idx = writer
18354            .create_btree_v2_dataset(
18355                "data",
18356                DatatypeMessage::f64_type(),
18357                &[0, 0],               // start empty
18358                &[u64::MAX, u64::MAX], // both dims unlimited
18359                &[2, 3],               // chunk = 2x3
18360            )
18361            .unwrap();
18362
18363        // Write chunks for a 4x6 dataset
18364        // chunk (0,0)
18365        let c00: Vec<u8> = [0.0f64, 1.0, 2.0, 6.0, 7.0, 8.0]
18366            .iter()
18367            .flat_map(|v| v.to_le_bytes())
18368            .collect();
18369        writer.write_chunk_btree_v2(idx, &[0, 0], &c00).unwrap();
18370
18371        // chunk (0,1)
18372        let c01: Vec<u8> = [3.0f64, 4.0, 5.0, 9.0, 10.0, 11.0]
18373            .iter()
18374            .flat_map(|v| v.to_le_bytes())
18375            .collect();
18376        writer.write_chunk_btree_v2(idx, &[0, 1], &c01).unwrap();
18377
18378        // chunk (1,0)
18379        let c10: Vec<u8> = [12.0f64, 13.0, 14.0, 18.0, 19.0, 20.0]
18380            .iter()
18381            .flat_map(|v| v.to_le_bytes())
18382            .collect();
18383        writer.write_chunk_btree_v2(idx, &[1, 0], &c10).unwrap();
18384
18385        // chunk (1,1)
18386        let c11: Vec<u8> = [15.0f64, 16.0, 17.0, 21.0, 22.0, 23.0]
18387            .iter()
18388            .flat_map(|v| v.to_le_bytes())
18389            .collect();
18390        writer.write_chunk_btree_v2(idx, &[1, 1], &c11).unwrap();
18391
18392        writer.extend_dataset(idx, &[4, 6]).unwrap();
18393        writer.close().unwrap();
18394
18395        // Read back
18396        let mut reader = Hdf5Reader::open(&path).unwrap();
18397        assert_eq!(reader.dataset_names(), vec!["data"]);
18398        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4, 6]);
18399
18400        let raw = reader.read_dataset_raw("data").unwrap();
18401        let values: Vec<f64> = raw
18402            .chunks(8)
18403            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
18404            .collect();
18405        assert_eq!(values.len(), 24);
18406        for (i, val) in values.iter().enumerate() {
18407            assert_eq!(*val, i as f64);
18408        }
18409
18410        std::fs::remove_file(&path).ok();
18411    }
18412
18413    /// Bytes one chunk of [`btree_v2_flush_probe`]'s dataset occupies — an
18414    /// f64 element, so the allocator's alignment neither pads nor merges it and
18415    /// the file's growth is exactly the bytes asked for.
18416    const BT2_PROBE_CHUNK: u64 = 8;
18417
18418    /// Write chunks of a 1x1-chunked 2-D BT2 dataset, flushing at each batch
18419    /// boundary, and report `(node addresses, file length)` after every flush.
18420    /// Chunks are addressed down column 0 so the record count — and hence the
18421    /// tree's shape — grows one record at a time.
18422    fn btree_v2_flush_probe(path: &std::path::Path, batches: &[u64]) -> Vec<(Vec<u64>, u64)> {
18423        let writer = Hdf5Writer::create(path).unwrap();
18424        let idx = writer
18425            .create_btree_v2_dataset(
18426                "data",
18427                DatatypeMessage::f64_type(),
18428                &[0, 0],
18429                &[u64::MAX, u64::MAX],
18430                &[1, 1],
18431            )
18432            .unwrap();
18433        let mut written = 0u64;
18434        let mut out = Vec::new();
18435        for &upto in batches {
18436            while written < upto {
18437                writer
18438                    .write_chunk_btree_v2(idx, &[written, 0], &(written as f64).to_le_bytes())
18439                    .unwrap();
18440                written += 1;
18441            }
18442            writer.flush_dataset(idx).unwrap();
18443            let addrs = writer
18444                .ds(idx)
18445                .lock()
18446                .btree_v2
18447                .as_ref()
18448                .unwrap()
18449                .node_addrs
18450                .clone();
18451            out.push((addrs, std::fs::metadata(path).unwrap().len()));
18452        }
18453        writer.extend_dataset(idx, &[written.max(1), 1]).unwrap();
18454        writer.close().unwrap();
18455        out
18456    }
18457
18458    /// The node pool tracks the tree in both directions. Dropping records is
18459    /// what a removal path would do — [`Bt2ChunkIndex`] has none today, so the
18460    /// test drops them itself — and the flush that follows must hand the blocks
18461    /// its smaller tree no longer needs back to the allocator instead of
18462    /// leaving them recorded and unreachable.
18463    #[test]
18464    fn a_btree_v2_flush_frees_the_node_blocks_its_tree_gave_up() {
18465        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
18466
18467        let path = temp_path("bt2_node_shrink");
18468        let writer = Hdf5Writer::create(&path).unwrap();
18469        let idx = writer
18470            .create_btree_v2_dataset(
18471                "data",
18472                DatatypeMessage::f64_type(),
18473                &[0, 0],
18474                &[u64::MAX, u64::MAX],
18475                &[1, 1],
18476            )
18477            .unwrap();
18478        // 85 records is one past a leaf, so the tree is two leaves and a root.
18479        for i in 0..85u64 {
18480            writer
18481                .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18482                .unwrap();
18483        }
18484        writer.flush_dataset(idx).unwrap();
18485        let grown = writer
18486            .ds(idx)
18487            .lock()
18488            .btree_v2
18489            .as_ref()
18490            .unwrap()
18491            .node_addrs
18492            .clone();
18493        assert_eq!(grown.len(), 3, "expected two leaves and a root");
18494
18495        // Back to 84 records: one leaf, so two of the three blocks are surplus.
18496        writer
18497            .ds(idx)
18498            .lock()
18499            .btree_v2
18500            .as_mut()
18501            .unwrap()
18502            .index
18503            .records
18504            .truncate(84);
18505        writer.flush_dataset(idx).unwrap();
18506        let shrunk = writer
18507            .ds(idx)
18508            .lock()
18509            .btree_v2
18510            .as_ref()
18511            .unwrap()
18512            .node_addrs
18513            .clone();
18514        assert_eq!(
18515            shrunk,
18516            grown[..1],
18517            "the pool still records the surplus blocks"
18518        );
18519
18520        // The surplus went back to the allocator, not on the floor: the next
18521        // node-sized allocation lands inside the region the two blocks covered.
18522        let reused = writer
18523            .allocator
18524            .allocate(BT2_NODE_SIZE as u64, FreeSpaceClass::Metadata);
18525        assert!(
18526            (grown[1]..grown[1] + 2 * BT2_NODE_SIZE as u64).contains(&reused),
18527            "a node block allocated at {reused:#x}, outside the freed \
18528             [{:#x}, {:#x}) the flush gave up",
18529            grown[1],
18530            grown[1] + 2 * BT2_NODE_SIZE as u64
18531        );
18532
18533        writer.extend_dataset(idx, &[85, 1]).unwrap();
18534        writer.close().unwrap();
18535        std::fs::remove_file(&path).ok();
18536    }
18537
18538    /// A v2 B-tree whose header declares a non-default node size — libhdf5
18539    /// built with a different `H5D_BT2_NODE_SIZE`, or any other writer —
18540    /// reopens for append: the reconstruction adopts the header's node_size,
18541    /// split and merge instead of refusing everything but 2048, and the next
18542    /// flush re-serializes at that size (upstream allocates every node at
18543    /// `hdr->node_size`, H5B2leaf.c / H5B2internal.c).
18544    #[test]
18545    fn a_btree_v2_with_a_foreign_node_size_reopens_and_grows() {
18546        let path = temp_path("bt2_foreign_node_size");
18547        {
18548            let writer = Hdf5Writer::create(&path).unwrap();
18549            let idx = writer
18550                .create_btree_v2_dataset(
18551                    "data",
18552                    DatatypeMessage::f64_type(),
18553                    &[0, 0],
18554                    &[u64::MAX, u64::MAX],
18555                    &[1, 1],
18556                )
18557                .unwrap();
18558            // Act as a foreign writer: 512-byte nodes, non-default tuning.
18559            // record_size 24 => a 512-byte leaf holds 20 records, so 85
18560            // records make a depth-1 tree of 512-byte blocks.
18561            {
18562                let ds = writer.ds(idx);
18563                let mut m = ds.lock();
18564                let index = &mut m.btree_v2.as_mut().unwrap().index;
18565                index.node_size = 512;
18566                index.split_percent = 90;
18567                index.merge_percent = 30;
18568            }
18569            for i in 0..85u64 {
18570                writer
18571                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18572                    .unwrap();
18573            }
18574            writer.extend_dataset(idx, &[85, 1]).unwrap();
18575            writer.close().unwrap();
18576        }
18577        {
18578            let writer = Hdf5Writer::open_append(&path).unwrap();
18579            let idx = writer.dataset_index("data").unwrap();
18580            {
18581                let ds = writer.ds(idx);
18582                let m = ds.lock();
18583                let index = &m.btree_v2.as_ref().unwrap().index;
18584                assert_eq!(index.node_size, 512, "header node_size not adopted");
18585                assert_eq!(index.split_percent, 90);
18586                assert_eq!(index.merge_percent, 30);
18587                assert_eq!(index.records.len(), 85, "records not walked back");
18588            }
18589            for i in 85..115u64 {
18590                writer
18591                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18592                    .unwrap();
18593            }
18594            writer.extend_dataset(idx, &[115, 1]).unwrap();
18595            writer.close().unwrap();
18596        }
18597
18598        let mut reader = Hdf5Reader::open(&path).unwrap();
18599        let raw = reader.read_dataset_raw("data").unwrap();
18600        let values: Vec<f64> = raw
18601            .chunks(8)
18602            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
18603            .collect();
18604        assert_eq!(values.len(), 115);
18605        for (i, v) in values.iter().enumerate() {
18606            assert_eq!(*v, i as f64, "element {i}");
18607        }
18608        std::fs::remove_file(&path).ok();
18609    }
18610
18611    /// A node's record count falls as well as rises: the tree's first leaf goes
18612    /// from a full 84 records to 42 when 85 records force it to split. The node
18613    /// image is padded to the whole block so re-serializing overwrites the
18614    /// block, not a prefix of it — otherwise that leaf keeps the tail of its
18615    /// 84-record self, stale records sitting in a live node block.
18616    #[test]
18617    fn a_shrinking_btree_v2_node_leaves_no_stale_records_behind() {
18618        use crate::format::chunk_index::btree_v2::{Bt2ChunkIndex, BT2_NODE_SIZE};
18619
18620        let path = temp_path("bt2_node_blocks");
18621        let probe = btree_v2_flush_probe(&path, &[84, 85]);
18622        let node0 = probe.last().unwrap().0[0];
18623
18624        // What the first leaf holds once the tree has split.
18625        let ctx = FormatContext {
18626            sizeof_addr: 8,
18627            sizeof_size: 8,
18628        };
18629        let mut index = Bt2ChunkIndex::new_unfiltered(2);
18630        for i in 0..85u64 {
18631            index.insert(vec![i, 0], 0);
18632        }
18633        let tree = index.build_tree(&ctx);
18634        assert!(
18635            tree.nodes[0].num_records < 84,
18636            "this test needs the first leaf to shrink, got {}",
18637            tree.nodes[0].num_records
18638        );
18639        // signature(4) + version(1) + type(1) + records + checksum(4)
18640        let used = 10 + tree.nodes[0].num_records as usize * tree.record_size as usize;
18641
18642        let bytes = std::fs::read(&path).unwrap();
18643        let block = &bytes[node0 as usize..node0 as usize + BT2_NODE_SIZE as usize];
18644        assert!(
18645            block[used..].iter().all(|&b| b == 0),
18646            "leaf block at {node0:#x} still holds {} bytes of its previous, larger image",
18647            block[used..].iter().rposition(|&b| b != 0).unwrap_or(0) + 1
18648        );
18649        std::fs::remove_file(&path).ok();
18650    }
18651
18652    /// The node pool is the single owner of the tree's block addresses: a flush
18653    /// reuses every block already in it and allocates only the shortfall. So
18654    /// re-flushing an unchanged index must cost nothing, and a flush that grows
18655    /// the tree must cost exactly the blocks it added — anything more means a
18656    /// block was stranded.
18657    #[test]
18658    fn a_btree_v2_flush_allocates_only_the_node_blocks_it_adds() {
18659        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
18660
18661        let path = temp_path("bt2_pool_growth");
18662        // Re-flush at 84 (still one leaf), then cross into a three-node depth-1
18663        // tree, then keep growing.
18664        let batches = [84u64, 84, 85, 200, 200];
18665        let probe = btree_v2_flush_probe(&path, &batches);
18666        for i in 1..probe.len() {
18667            let (prev_addrs, prev_len) = &probe[i - 1];
18668            let (addrs, len) = &probe[i];
18669            assert!(
18670                addrs.starts_with(prev_addrs),
18671                "flush {i} moved a node block instead of reusing it"
18672            );
18673            let new_blocks = (addrs.len() - prev_addrs.len()) as u64 * BT2_NODE_SIZE as u64;
18674            let new_chunks = (batches[i] - batches[i - 1]) * BT2_PROBE_CHUNK;
18675            assert_eq!(
18676                len - prev_len,
18677                new_blocks + new_chunks,
18678                "flush {i} grew the file by more than the blocks it added"
18679            );
18680        }
18681        // The unchanged re-flushes must be free.
18682        assert_eq!(probe[1].1, probe[0].1);
18683        assert_eq!(probe[4].1, probe[3].1);
18684        std::fs::remove_file(&path).ok();
18685    }
18686
18687    #[cfg(feature = "parallel")]
18688    #[test]
18689    fn parallel_batch_write_roundtrip() {
18690        let path = temp_path("parallel_batch");
18691
18692        let writer = Hdf5Writer::create(&path).unwrap();
18693        let idx = writer
18694            .create_chunked_dataset(
18695                "data",
18696                DatatypeMessage::i32_type(),
18697                &[0, 4],
18698                &[u64::MAX, 4],
18699                &[1, 4],
18700            )
18701            .unwrap();
18702
18703        // Prepare chunks
18704        let chunks_data: Vec<(u64, Vec<u8>)> = (0..8u64)
18705            .map(|frame| {
18706                let values: Vec<i32> = (0..4).map(|i| (frame * 4 + i) as i32).collect();
18707                let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18708                (frame, raw)
18709            })
18710            .collect();
18711
18712        let batch: Vec<(u64, &[u8])> = chunks_data
18713            .iter()
18714            .map(|(idx, data)| (*idx, data.as_slice()))
18715            .collect();
18716
18717        writer.write_chunks_batch(idx, &batch).unwrap();
18718        writer.extend_dataset(idx, &[8, 4]).unwrap();
18719        writer.close().unwrap();
18720
18721        // Read back
18722        let mut reader = Hdf5Reader::open(&path).unwrap();
18723        assert_eq!(reader.dataset_shape("data").unwrap(), vec![8, 4]);
18724        let raw = reader.read_dataset_raw("data").unwrap();
18725        let values: Vec<i32> = raw
18726            .chunks(4)
18727            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18728            .collect();
18729        assert_eq!(values.len(), 32);
18730        for (i, val) in values.iter().enumerate() {
18731            assert_eq!(*val, i as i32);
18732        }
18733
18734        std::fs::remove_file(&path).ok();
18735    }
18736
18737    #[test]
18738    fn swmr_writer_append_frames() {
18739        use crate::io::swmr::SwmrWriter;
18740
18741        // Per-call unique path so concurrent cargo invocations and
18742        // kernel-side flock release races cannot collide.
18743        use std::sync::atomic::{AtomicU64, Ordering};
18744        static COUNTER: AtomicU64 = AtomicU64::new(0);
18745        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18746        let path = std::env::temp_dir().join(format!(
18747            "rust_hdf5_swmr_append_{}_{}.h5",
18748            std::process::id(),
18749            n
18750        ));
18751
18752        let mut swmr = SwmrWriter::create(&path).unwrap();
18753        let idx = swmr
18754            .create_streaming_dataset("detector", DatatypeMessage::u16_type(), &[4, 4])
18755            .unwrap();
18756
18757        swmr.start_swmr().unwrap();
18758
18759        // Append 5 frames
18760        for frame in 0..5u16 {
18761            let data: Vec<u16> = (0..16).map(|i| frame * 16 + i).collect();
18762            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18763            swmr.append_frame(idx, &raw).unwrap();
18764        }
18765
18766        swmr.flush().unwrap();
18767        swmr.close().unwrap();
18768
18769        // Read back
18770        let mut reader = Hdf5Reader::open(&path).unwrap();
18771        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![5, 4, 4]);
18772
18773        let raw = reader.read_dataset_raw("detector").unwrap();
18774        let values: Vec<u16> = raw
18775            .chunks(2)
18776            .map(|chunk| u16::from_le_bytes(chunk.try_into().unwrap()))
18777            .collect();
18778        assert_eq!(values.len(), 80); // 5 * 4 * 4
18779                                      // Verify first frame
18780        for (i, val) in values.iter().enumerate().take(16) {
18781            assert_eq!(*val, i as u16);
18782        }
18783        // Verify last frame
18784        for (i, val) in values[64..80].iter().enumerate() {
18785            assert_eq!(*val, 4 * 16 + i as u16);
18786        }
18787
18788        std::fs::remove_file(&path).ok();
18789    }
18790
18791    #[test]
18792    fn swmr_writer_tiled_frames() {
18793        use crate::io::swmr::SwmrWriter;
18794        use std::sync::atomic::{AtomicU64, Ordering};
18795        static COUNTER: AtomicU64 = AtomicU64::new(0);
18796        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18797        let path = std::env::temp_dir().join(format!(
18798            "rust_hdf5_swmr_tiled_{}_{}.h5",
18799            std::process::id(),
18800            n
18801        ));
18802
18803        let mut swmr = SwmrWriter::create(&path).unwrap();
18804        // 4x4 frames, tiled into 2x2 chunks -> 4 chunks per frame.
18805        let idx = swmr
18806            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[4, 4], &[2, 2])
18807            .unwrap();
18808        swmr.start_swmr().unwrap();
18809
18810        for frame in 0..3u16 {
18811            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
18812            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18813            swmr.append_frame(idx, &raw).unwrap();
18814        }
18815        swmr.flush().unwrap();
18816        swmr.close().unwrap();
18817
18818        let mut reader = Hdf5Reader::open(&path).unwrap();
18819        assert_eq!(reader.dataset_shape("det").unwrap(), vec![3, 4, 4]);
18820        let raw = reader.read_dataset_raw("det").unwrap();
18821        let values: Vec<u16> = raw
18822            .chunks(2)
18823            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18824            .collect();
18825        assert_eq!(values.len(), 48);
18826        // Every element must survive the frame -> tile split and the
18827        // tile -> frame reassembly on read.
18828        for frame in 0..3u16 {
18829            for i in 0..16usize {
18830                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
18831            }
18832        }
18833        std::fs::remove_file(&path).ok();
18834    }
18835
18836    /// A chunk tile larger than the frame is geometry libhdf5 refuses to
18837    /// create (`H5D__chunk_construct`: chunk must not exceed a fixed maximum
18838    /// dimension), so no libhdf5-based writer — including the NDFileHDF5
18839    /// tiling controls this API mirrors — can produce such a file. Until
18840    /// 0.4.1 we accepted it and zero-padded the frame up to the tile; now
18841    /// the create is rejected like every other creator's.
18842    #[test]
18843    fn swmr_writer_tiled_chunk_larger_than_frame_is_rejected() {
18844        use crate::io::swmr::SwmrWriter;
18845        use std::sync::atomic::{AtomicU64, Ordering};
18846        static COUNTER: AtomicU64 = AtomicU64::new(0);
18847        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18848        let path = std::env::temp_dir().join(format!(
18849            "rust_hdf5_swmr_bigchunk_{}_{}.h5",
18850            std::process::id(),
18851            n
18852        ));
18853
18854        let mut swmr = SwmrWriter::create(&path).unwrap();
18855        let err = swmr
18856            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[3, 3], &[8, 8])
18857            .unwrap_err();
18858        assert!(
18859            err.to_string().contains("maximum dimension size"),
18860            "unexpected error: {err}"
18861        );
18862        swmr.close().unwrap();
18863        std::fs::remove_file(&path).ok();
18864    }
18865
18866    #[test]
18867    fn swmr_writer_multi_frame_chunks() {
18868        use crate::io::swmr::SwmrWriter;
18869        use std::sync::atomic::{AtomicU64, Ordering};
18870        static COUNTER: AtomicU64 = AtomicU64::new(0);
18871        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18872        let path = std::env::temp_dir().join(format!(
18873            "rust_hdf5_swmr_mfc_{}_{}.h5",
18874            std::process::id(),
18875            n
18876        ));
18877
18878        // 3x3 frames, chunk = 4 frames x full frame. 10 frames -> 3 bands
18879        // of 4, 4, 2 (the last band partial).
18880        let mut swmr = SwmrWriter::create(&path).unwrap();
18881        let idx = swmr
18882            .create_streaming_dataset_chunked(
18883                "det",
18884                DatatypeMessage::u16_type(),
18885                &[3, 3],
18886                &[4, 3, 3],
18887            )
18888            .unwrap();
18889        swmr.start_swmr().unwrap();
18890        for frame in 0..10u16 {
18891            let data: Vec<u16> = (0..9).map(|i| frame * 100 + i).collect();
18892            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18893            swmr.append_frame(idx, &raw).unwrap();
18894        }
18895        swmr.flush().unwrap();
18896        swmr.close().unwrap();
18897
18898        let mut reader = Hdf5Reader::open(&path).unwrap();
18899        // The partial last band must not over-extend the frame count.
18900        assert_eq!(reader.dataset_shape("det").unwrap(), vec![10, 3, 3]);
18901        let raw = reader.read_dataset_raw("det").unwrap();
18902        let values: Vec<u16> = raw
18903            .chunks(2)
18904            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18905            .collect();
18906        assert_eq!(values.len(), 90);
18907        for frame in 0..10u16 {
18908            for i in 0..9usize {
18909                assert_eq!(values[frame as usize * 9 + i], frame * 100 + i as u16);
18910            }
18911        }
18912        std::fs::remove_file(&path).ok();
18913    }
18914
18915    #[test]
18916    fn swmr_writer_multi_frame_tiled_chunks() {
18917        use crate::io::swmr::SwmrWriter;
18918        use std::sync::atomic::{AtomicU64, Ordering};
18919        static COUNTER: AtomicU64 = AtomicU64::new(0);
18920        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18921        let path = std::env::temp_dir().join(format!(
18922            "rust_hdf5_swmr_mftc_{}_{}.h5",
18923            std::process::id(),
18924            n
18925        ));
18926
18927        // 4x4 frames, chunk = 2 frames x 2x2 tiles. 5 frames -> bands of
18928        // 2, 2, 1; every frame is also split into a 2x2 tile grid.
18929        let mut swmr = SwmrWriter::create(&path).unwrap();
18930        let idx = swmr
18931            .create_streaming_dataset_chunked(
18932                "det",
18933                DatatypeMessage::u16_type(),
18934                &[4, 4],
18935                &[2, 2, 2],
18936            )
18937            .unwrap();
18938        swmr.start_swmr().unwrap();
18939        for frame in 0..5u16 {
18940            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
18941            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18942            swmr.append_frame(idx, &raw).unwrap();
18943        }
18944        swmr.flush().unwrap();
18945        swmr.close().unwrap();
18946
18947        let mut reader = Hdf5Reader::open(&path).unwrap();
18948        assert_eq!(reader.dataset_shape("det").unwrap(), vec![5, 4, 4]);
18949        let raw = reader.read_dataset_raw("det").unwrap();
18950        let values: Vec<u16> = raw
18951            .chunks(2)
18952            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18953            .collect();
18954        assert_eq!(values.len(), 80);
18955        for frame in 0..5u16 {
18956            for i in 0..16usize {
18957                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
18958            }
18959        }
18960        std::fs::remove_file(&path).ok();
18961    }
18962
18963    #[cfg(feature = "deflate")]
18964    #[test]
18965    fn swmr_writer_compressed_frames() {
18966        use crate::io::swmr::SwmrWriter;
18967        use std::sync::atomic::{AtomicU64, Ordering};
18968        static COUNTER: AtomicU64 = AtomicU64::new(0);
18969        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18970        let path = std::env::temp_dir().join(format!(
18971            "rust_hdf5_swmr_comp_{}_{}.h5",
18972            std::process::id(),
18973            n
18974        ));
18975
18976        let mut swmr = SwmrWriter::create(&path).unwrap();
18977        let pipeline = crate::format::messages::filter::FilterPipeline::deflate(4);
18978        let idx = swmr
18979            .create_streaming_dataset_compressed(
18980                "detector",
18981                DatatypeMessage::i32_type(),
18982                &[8],
18983                pipeline,
18984            )
18985            .unwrap();
18986        swmr.start_swmr().unwrap();
18987
18988        for frame in 0..40i32 {
18989            let raw: Vec<u8> = (0..8).flat_map(|i| (frame * 8 + i).to_le_bytes()).collect();
18990            swmr.append_frame(idx, &raw).unwrap();
18991            if frame % 7 == 0 {
18992                swmr.flush().unwrap();
18993            }
18994        }
18995        swmr.flush().unwrap();
18996        swmr.close().unwrap();
18997
18998        let mut reader = Hdf5Reader::open(&path).unwrap();
18999        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![40, 8]);
19000        let raw = reader.read_dataset_raw("detector").unwrap();
19001        let values: Vec<i32> = raw
19002            .chunks(4)
19003            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19004            .collect();
19005        assert_eq!(values, (0..320).collect::<Vec<i32>>());
19006
19007        std::fs::remove_file(&path).ok();
19008    }
19009
19010    #[test]
19011    fn group_hierarchy_writer_reader() {
19012        let path = temp_path("group_hierarchy");
19013
19014        let writer = Hdf5Writer::create(&path).unwrap();
19015
19016        // Create groups
19017        let g0 = writer.create_group("/", "group1").unwrap();
19018        let g1 = writer.create_group("/group1", "sub").unwrap();
19019        assert_eq!(g0, 0);
19020        assert_eq!(g1, 1);
19021
19022        // Create datasets
19023        let ds_root = writer
19024            .create_dataset("root_data", DatatypeMessage::f64_type(), &[2])
19025            .unwrap();
19026        let raw_root: Vec<u8> = [1.0f64, 2.0].iter().flat_map(|v| v.to_le_bytes()).collect();
19027        writer.write_dataset_raw(ds_root, &raw_root).unwrap();
19028
19029        let ds_g0 = writer
19030            .create_dataset("group1/data", DatatypeMessage::i32_type(), &[3])
19031            .unwrap();
19032        let raw_g0: Vec<u8> = [10i32, 20, 30]
19033            .iter()
19034            .flat_map(|v| v.to_le_bytes())
19035            .collect();
19036        writer.write_dataset_raw(ds_g0, &raw_g0).unwrap();
19037
19038        let ds_g1 = writer
19039            .create_dataset("group1/sub/values", DatatypeMessage::u8_type(), &[4])
19040            .unwrap();
19041        writer.write_dataset_raw(ds_g1, &[1u8, 2, 3, 4]).unwrap();
19042
19043        writer.close().unwrap();
19044
19045        // Read back
19046        let mut reader = Hdf5Reader::open(&path).unwrap();
19047        let names = reader.dataset_names();
19048        assert!(names.contains(&"root_data"), "names: {:?}", names);
19049        assert!(names.contains(&"group1/data"), "names: {:?}", names);
19050        assert!(names.contains(&"group1/sub/values"), "names: {:?}", names);
19051
19052        let raw = reader.read_dataset_raw("root_data").unwrap();
19053        let vals: Vec<f64> = raw
19054            .chunks(8)
19055            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19056            .collect();
19057        assert_eq!(vals, vec![1.0, 2.0]);
19058
19059        let raw = reader.read_dataset_raw("group1/data").unwrap();
19060        let vals: Vec<i32> = raw
19061            .chunks(4)
19062            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19063            .collect();
19064        assert_eq!(vals, vec![10, 20, 30]);
19065
19066        let raw = reader.read_dataset_raw("group1/sub/values").unwrap();
19067        assert_eq!(raw, vec![1, 2, 3, 4]);
19068
19069        std::fs::remove_file(&path).ok();
19070    }
19071
19072    /// libhdf5 (`H5D__chunk_construct`) rejects a chunk dimension that
19073    /// exceeds a fixed maximum dimension. Before this check, such a dataset
19074    /// was created and appends landed rows at the chunk stride instead of
19075    /// the row stride, reading back [1, 2, 0, 0] for [1, 2, 3, 4].
19076    #[test]
19077    fn create_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
19078        let path = temp_path("chunk_wider_than_max");
19079
19080        let writer = Hdf5Writer::create(&path).unwrap();
19081        let err = writer
19082            .create_chunked_dataset(
19083                "data",
19084                DatatypeMessage::f64_type(),
19085                &[0, 2],
19086                &[u64::MAX, 2],
19087                &[2, 4],
19088            )
19089            .unwrap_err();
19090        assert!(
19091            err.to_string().contains("maximum dimension size"),
19092            "unexpected error: {err}"
19093        );
19094
19095        // The fixed-array creators derive the maximum from the fixed dims.
19096        let err = writer
19097            .create_fixed_array_dataset("fa", DatatypeMessage::f64_type(), &[3], &[5])
19098            .unwrap_err();
19099        assert!(
19100            err.to_string().contains("maximum dimension size"),
19101            "unexpected error: {err}"
19102        );
19103
19104        writer.close().unwrap();
19105        std::fs::remove_file(&path).ok();
19106    }
19107
19108    /// libhdf5 exempts a dimension whose *current* size is zero from the
19109    /// chunk-vs-maximum check (`curr_dims[u] &&` in `H5D__chunk_construct`),
19110    /// and rejects a zero chunk dimension on every path.
19111    #[test]
19112    fn create_mirrors_the_libhdf5_chunk_geometry_exemptions() {
19113        let path = temp_path("chunk_geometry_exemptions");
19114
19115        let writer = Hdf5Writer::create(&path).unwrap();
19116        // dims[1] == 0: chunk 4 > max 2 is allowed, as libhdf5 allows it.
19117        writer
19118            .create_chunked_dataset(
19119                "exempt",
19120                DatatypeMessage::f64_type(),
19121                &[0, 0],
19122                &[u64::MAX, 2],
19123                &[2, 4],
19124            )
19125            .unwrap();
19126
19127        let err = writer
19128            .create_chunked_dataset("zero", DatatypeMessage::f64_type(), &[0], &[u64::MAX], &[0])
19129            .unwrap_err();
19130        assert!(
19131            err.to_string().contains("chunk dimension 0 is zero"),
19132            "unexpected error: {err}"
19133        );
19134
19135        writer.close().unwrap();
19136        std::fs::remove_file(&path).ok();
19137    }
19138
19139    /// A file written by 0.4.0 can carry a chunk row wider than the frame
19140    /// row — create now rejects that geometry, but reopened files keep it.
19141    /// Appends must scatter frames at the chunk stride, not pack them at
19142    /// the frame stride (which read back `[1, 2, 0, 0]` for `[1, 2, 3, 4]`).
19143    /// The wide shape is simulated by widening the registered chunk dims
19144    /// after create, which also lands in the layout message at close.
19145    #[test]
19146    fn append_scatters_into_a_legacy_wider_than_row_chunk() {
19147        let path = temp_path("legacy_wide_chunk_append");
19148
19149        let writer = Hdf5Writer::create(&path).unwrap();
19150        let idx = writer
19151            .create_chunked_dataset(
19152                "data",
19153                DatatypeMessage::i32_type(),
19154                &[0, 2],
19155                &[u64::MAX, 2],
19156                &[2, 2],
19157            )
19158            .unwrap();
19159        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims = vec![2, 4];
19160
19161        let frames: Vec<u8> = [1i32, 2, 3, 4]
19162            .iter()
19163            .flat_map(|v| v.to_le_bytes())
19164            .collect();
19165        writer.write_append_frames(idx, 0, 2, &frames).unwrap();
19166        writer.extend_dataset(idx, &[2, 2]).unwrap();
19167        writer.close().unwrap();
19168
19169        let mut reader = Hdf5Reader::open(&path).unwrap();
19170        assert_eq!(reader.dataset_shape("data").unwrap(), vec![2, 2]);
19171        let raw = reader.read_dataset_raw("data").unwrap();
19172        let values: Vec<i32> = raw
19173            .chunks(4)
19174            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19175            .collect();
19176        assert_eq!(values, vec![1, 2, 3, 4]);
19177        std::fs::remove_file(&path).ok();
19178    }
19179
19180    /// The compressed vlen creator sizes its chunked layout from a
19181    /// caller-supplied chunk size; it goes through the same geometry
19182    /// validation as every other creator (empty inputs are exempt because
19183    /// their current size is zero).
19184    #[test]
19185    #[cfg(feature = "deflate")]
19186    fn compressed_vlen_create_validates_its_chunk_size() {
19187        use crate::format::messages::filter::FilterPipeline;
19188        let path = temp_path("vlen_compressed_chunk");
19189
19190        let writer = Hdf5Writer::create(&path).unwrap();
19191        let err = writer
19192            .create_vlen_string_dataset_compressed(
19193                "texts",
19194                &["a", "b", "c"],
19195                100,
19196                FilterPipeline::deflate(6),
19197            )
19198            .unwrap_err();
19199        assert!(
19200            err.to_string().contains("maximum dimension size"),
19201            "unexpected error: {err}"
19202        );
19203
19204        writer
19205            .create_vlen_string_dataset_compressed("empty", &[], 16, FilterPipeline::deflate(6))
19206            .unwrap();
19207
19208        writer.close().unwrap();
19209        std::fs::remove_file(&path).ok();
19210    }
19211
19212    /// `set_libver_latest` moves *filtered* chunked datasets to layout v5 with
19213    /// fixed 8-byte chunk-size fields; unfiltered chunked and pre-opt-in
19214    /// datasets keep v4 with the derived width, matching libhdf5's
19215    /// `version_perf` rule (only the filtered index arms bump to 5).
19216    #[cfg(feature = "deflate")]
19217    #[test]
19218    fn libver_latest_selects_v5_for_filtered_chunks_only() {
19219        let path = temp_path("libver_v5_select");
19220
19221        let mut writer = Hdf5Writer::create(&path).unwrap();
19222        let before = writer
19223            .create_chunked_dataset_with_pipeline(
19224                "d4",
19225                DatatypeMessage::i32_type(),
19226                &[0],
19227                &[u64::MAX],
19228                &[16],
19229                FilterPipeline::deflate(4),
19230            )
19231            .unwrap();
19232        writer.set_libver_latest(true).unwrap();
19233        let ea5 = writer
19234            .create_chunked_dataset_with_pipeline(
19235                "ea5",
19236                DatatypeMessage::i32_type(),
19237                &[0],
19238                &[u64::MAX],
19239                &[16],
19240                FilterPipeline::deflate(4),
19241            )
19242            .unwrap();
19243        let plain = writer
19244            .create_chunked_dataset(
19245                "plain",
19246                DatatypeMessage::i32_type(),
19247                &[0],
19248                &[u64::MAX],
19249                &[16],
19250            )
19251            .unwrap();
19252        let fa5 = writer
19253            .create_fixed_array_dataset_with_pipeline(
19254                "fa5",
19255                DatatypeMessage::i32_type(),
19256                &[4, 6],
19257                &[2, 3],
19258                FilterPipeline::deflate(6),
19259            )
19260            .unwrap();
19261        let bt5 = writer
19262            .create_btree_v2_dataset_with_pipeline(
19263                "bt5",
19264                DatatypeMessage::i32_type(),
19265                &[0, 0],
19266                &[u64::MAX, u64::MAX],
19267                &[2, 3],
19268                FilterPipeline::deflate(6),
19269            )
19270            .unwrap();
19271
19272        {
19273            let d4 = writer.ds(before);
19274            let d4 = d4.lock();
19275            assert_eq!(d4.layout_version, 4);
19276            assert_eq!(
19277                d4.chunked.as_ref().unwrap().chunk_size_len,
19278                compute_chunk_size_len(16 * 4)
19279            );
19280            let e5 = writer.ds(ea5);
19281            let e5 = e5.lock();
19282            assert_eq!(e5.layout_version, 5);
19283            assert_eq!(e5.chunked.as_ref().unwrap().chunk_size_len, 8);
19284            assert_eq!(writer.ds(plain).lock().layout_version, 4);
19285            assert_eq!(writer.ds(fa5).lock().layout_version, 5);
19286            assert_eq!(writer.ds(bt5).lock().layout_version, 5);
19287        }
19288
19289        // Write through the FA and BT2 v5 indexes so their 8-byte chunk-size
19290        // fields are exercised end to end, not just selected.
19291        for (coords, vals) in [
19292            ([0u64, 0], [0i32, 1, 2, 6, 7, 8]),
19293            ([0, 1], [3, 4, 5, 9, 10, 11]),
19294            ([1, 0], [12, 13, 14, 18, 19, 20]),
19295            ([1, 1], [15, 16, 17, 21, 22, 23]),
19296        ] {
19297            let bytes: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
19298            writer
19299                .write_chunk_fixed_array(fa5, &coords, &bytes)
19300                .unwrap();
19301            writer.write_chunk_btree_v2(bt5, &coords, &bytes).unwrap();
19302        }
19303        writer.extend_dataset(bt5, &[4, 6]).unwrap();
19304        writer.close().unwrap();
19305
19306        let mut reader = Hdf5Reader::open(&path).unwrap();
19307        for name in ["fa5", "bt5"] {
19308            let raw = reader.read_dataset_raw(name).unwrap();
19309            let values: Vec<i32> = raw
19310                .chunks(4)
19311                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19312                .collect();
19313            assert_eq!(values, (0..24).collect::<Vec<i32>>(), "dataset {name}");
19314        }
19315
19316        std::fs::remove_file(&path).ok();
19317    }
19318
19319    /// A v5 file reopened for append must stay v5: the decode → `DatasetInfo`
19320    /// → finalize path carries the version through, so the re-encoded layout
19321    /// message matches the 8-byte size fields the filtered index was built
19322    /// with. A silent v4 downgrade here would make libhdf5 derive a narrower
19323    /// field width than the index uses.
19324    #[cfg(feature = "deflate")]
19325    #[test]
19326    fn v5_layout_survives_reopen_and_append() {
19327        let path = temp_path("libver_v5_reopen");
19328        let chunk: usize = 8;
19329
19330        let mut writer = Hdf5Writer::create(&path).unwrap();
19331        writer.set_libver_latest(true).unwrap();
19332        let idx = writer
19333            .create_chunked_dataset_with_pipeline(
19334                "d",
19335                DatatypeMessage::i32_type(),
19336                &[0],
19337                &[u64::MAX],
19338                &[chunk as u64],
19339                FilterPipeline::deflate(4),
19340            )
19341            .unwrap();
19342        for c in 0..2u64 {
19343            let data: Vec<u8> = (0..chunk as i32)
19344                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
19345                .collect();
19346            writer.write_chunk(idx, c, &data).unwrap();
19347        }
19348        writer.extend_dataset(idx, &[2 * chunk as u64]).unwrap();
19349        writer.close().unwrap();
19350
19351        // Reopen: the decoded layout version must be preserved, and appends
19352        // must keep working against the 8-byte-size-field index.
19353        let writer = Hdf5Writer::open_append(&path).unwrap();
19354        assert_eq!(writer.ds(0).lock().layout_version, 5);
19355        for c in 2..4u64 {
19356            let data: Vec<u8> = (0..chunk as i32)
19357                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
19358                .collect();
19359            writer.write_chunk(0, c, &data).unwrap();
19360        }
19361        writer.extend_dataset(0, &[4 * chunk as u64]).unwrap();
19362        writer.close().unwrap();
19363
19364        // Still v5 after the second finalize, and fully readable.
19365        let writer = Hdf5Writer::open_append(&path).unwrap();
19366        assert_eq!(writer.ds(0).lock().layout_version, 5);
19367        writer.close().unwrap();
19368
19369        let mut reader = Hdf5Reader::open(&path).unwrap();
19370        let raw = reader.read_dataset_raw("d").unwrap();
19371        let values: Vec<i32> = raw
19372            .chunks(4)
19373            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19374            .collect();
19375        assert_eq!(values, (0..4 * chunk as i32).collect::<Vec<i32>>());
19376
19377        std::fs::remove_file(&path).ok();
19378    }
19379
19380    /// A chunk strictly larger than `u32::MAX` bytes forces layout v5 with no
19381    /// opt-in — v4's size field cannot represent it — while a chunk of exactly
19382    /// `u32::MAX` bytes stays v4, matching libhdf5's `version_req` boundary
19383    /// (`> 0xffffffff`, filtered or not).
19384    #[test]
19385    fn oversized_chunk_forces_v5_without_opt_in() {
19386        let path = temp_path("libver_4gib_force");
19387
19388        let writer = Hdf5Writer::create(&path).unwrap();
19389        let at_limit = writer
19390            .create_chunked_dataset_with_pipeline(
19391                "at_limit",
19392                DatatypeMessage::u8_type(),
19393                &[0],
19394                &[u64::MAX],
19395                &[u32::MAX as u64],
19396                FilterPipeline::deflate(4),
19397            )
19398            .unwrap();
19399        let over = writer
19400            .create_chunked_dataset_with_pipeline(
19401                "over",
19402                DatatypeMessage::u8_type(),
19403                &[0],
19404                &[u64::MAX],
19405                &[u32::MAX as u64 + 1],
19406                FilterPipeline::deflate(4),
19407            )
19408            .unwrap();
19409        let over_unfiltered = writer
19410            .create_chunked_dataset(
19411                "over_plain",
19412                DatatypeMessage::u8_type(),
19413                &[0],
19414                &[u64::MAX],
19415                &[u32::MAX as u64 + 1],
19416            )
19417            .unwrap();
19418
19419        assert_eq!(writer.ds(at_limit).lock().layout_version, 4);
19420        {
19421            let ds = writer.ds(over);
19422            let ds = ds.lock();
19423            assert_eq!(ds.layout_version, 5);
19424            assert_eq!(ds.chunked.as_ref().unwrap().chunk_size_len, 8);
19425        }
19426        assert_eq!(writer.ds(over_unfiltered).lock().layout_version, 5);
19427        writer.close().unwrap();
19428        std::fs::remove_file(&path).ok();
19429    }
19430
19431    /// SWMR reaches version 3 on its own, without a chunked dataset to raise
19432    /// the bound — through the flags `finalize_for_swmr` passes, and then
19433    /// through `swmr_active` for every superblock written after it. Only a
19434    /// file with nothing else newer in it can tell the two arms apart, and
19435    /// the public SWMR API always creates a chunked streaming dataset.
19436    #[test]
19437    fn swmr_reaches_version_3_with_no_chunked_dataset_in_the_file() {
19438        let path = temp_path("swmr_superblock");
19439
19440        let mut writer = Hdf5Writer::create(&path).unwrap();
19441        writer
19442            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
19443            .unwrap();
19444        assert_eq!(writer.superblock_version_for(0), SUPERBLOCK_V2);
19445
19446        writer.finalize_for_swmr().unwrap();
19447        // What `start_swmr` does after finalizing, and what lets a second
19448        // handle read the file while this writer lives — the writer's
19449        // exclusive lock is mandatory on Windows.
19450        writer.handle().release_lock().unwrap();
19451        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
19452
19453        // The close-time finalize carries no SWMR flag; the file is still an
19454        // SWMR file and must not be handed back a version older than the one
19455        // its readers attached to.
19456        writer.close().unwrap();
19457        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
19458        std::fs::remove_file(&path).ok();
19459    }
19460
19461    /// A named bound below `H5F_LIBVER_V110` refuses the session instead —
19462    /// the two checks `H5F__start_swmr_write` opens with, a version-3
19463    /// superblock (H5Fint.c:3814) and a low bound of at least V110
19464    /// (H5Fint.c:3818). Naming no bound at all is what the test above does,
19465    /// and that file is free to become version 3.
19466    #[test]
19467    fn a_named_bound_below_v110_refuses_an_swmr_session() {
19468        for bound in [LibverBound::Earliest, LibverBound::V18] {
19469            let path = temp_path(&format!("swmr_refused_{bound:?}"));
19470            let mut writer = Hdf5Writer::create_with_options(
19471                &path,
19472                FileCreateOptions {
19473                    libver: Some(bound),
19474                    ..Default::default()
19475                },
19476            )
19477            .unwrap();
19478            writer
19479                .create_dataset("d", DatatypeMessage::i32_type(), &[2])
19480                .unwrap();
19481
19482            let err = writer.finalize_for_swmr().unwrap_err().to_string();
19483            assert!(err.contains("SWMR"), "{bound:?}: {err}");
19484            assert!(err.contains("H5F_LIBVER_V110"), "{bound:?}: {err}");
19485
19486            // Refused, not half-done: nothing was published, and the close
19487            // writes the file the bound asked for.
19488            writer.close().unwrap();
19489            let version = std::fs::read(&path).unwrap()[8];
19490            assert_eq!(version, bound.superblock_version(), "{bound:?}");
19491            std::fs::remove_file(&path).ok();
19492        }
19493    }
19494
19495    /// After every writer of a dataset object header, `nlink_written` is the
19496    /// count that writer encoded.
19497    ///
19498    /// `header_stale_with` is the one authority for "does the on-disk header
19499    /// still describe this dataset?", and it reads `nlink_written`; the three
19500    /// writers — `finalize`, `finalize_for_swmr` and
19501    /// `write_dataset_header_inplace` — therefore all record through
19502    /// `DatasetInfo::header_written`. This walks the SWMR sequence, where the
19503    /// in-place writer is the one that could drift, and pins why it does not:
19504    /// a name added after the publish grows the header past the block it was
19505    /// published into, so the rewrite is refused rather than half-applied and
19506    /// the count on disk stays the one the registry names.
19507    #[test]
19508    fn every_dataset_header_write_records_its_link_count() {
19509        let path = temp_path("header_write_records_nlink");
19510        let writer = Hdf5Writer::create(&path).unwrap();
19511        let idx = writer
19512            .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
19513            .unwrap();
19514        let mut writer = writer;
19515        writer.finalize_for_swmr().unwrap();
19516        assert_eq!(
19517            writer.ds(idx).lock().nlink_written,
19518            1,
19519            "the SWMR publish put one name in the header"
19520        );
19521        writer.write_dataset_header_inplace(idx).unwrap();
19522        assert_eq!(writer.ds(idx).lock().nlink_written, 1);
19523
19524        // A second name after the publish: the reference-count message it
19525        // adds does not fit the published block.
19526        writer.create_hard_link("/", "alias", "d").unwrap();
19527        assert_eq!(writer.object_link_count(HardLinkTarget::Dataset(idx)), 2);
19528        let grew = writer
19529            .write_dataset_header_inplace(idx)
19530            .unwrap_err()
19531            .to_string();
19532        assert!(
19533            grew.contains("cannot rewrite in place"),
19534            "a header that outgrew its block must be refused: {grew}"
19535        );
19536        assert_eq!(
19537            writer.ds(idx).lock().nlink_written,
19538            1,
19539            "a refused rewrite leaves the registry describing the header the file holds"
19540        );
19541
19542        // The close-time finalize is the writer that commits the second name,
19543        // and a reopen reads the same count back off the link graph.
19544        writer.close().unwrap();
19545        let writer = Hdf5Writer::open_append(&path).unwrap();
19546        assert_eq!(
19547            writer.ds(0).lock().nlink_written,
19548            2,
19549            "finalize wrote two names and the reopen reads two"
19550        );
19551        writer.close().unwrap();
19552        std::fs::remove_file(&path).ok();
19553    }
19554
19555    /// `H5D__chunk_set_info`'s `version_req` (H5Dchunk.c:909, :936): version 5
19556    /// is required for a chunk over 4 GiB — the version-4 layout message's
19557    /// stored-size field is 32 bits and cannot record one — and
19558    /// `LAYOUT_VERSION_DEFAULT` (3, `H5O_LAYOUT_VERSION_DEFAULT`) is the floor
19559    /// for everything at or under that limit. Pure arithmetic on the byte
19560    /// count: no chunk is ever allocated.
19561    #[test]
19562    fn required_chunk_layout_version_pins_5_past_4_gib() {
19563        assert_eq!(
19564            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64),
19565            LAYOUT_VERSION_DEFAULT
19566        );
19567        assert_eq!(
19568            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64 + 1),
19569            5
19570        );
19571    }
19572
19573    /// `H5D__chunk_set_info`'s index-selection gate (H5Dchunk.c:936): a chunk
19574    /// over 4 GiB reaches the v1.10 chunk indexes even under a bound whose
19575    /// `H5O_layout_ver_bounds` row (`LibverBound::layout_version`) is below
19576    /// 4 — `V18` (row 3) and `Earliest` (row 1) both normally keep an
19577    /// ordinary chunk on the version-1 B-tree, but
19578    /// `required_chunk_layout_version`'s own escape to 5 overrides that row
19579    /// for this one chunk. The default bound (`V110`, row 4) already crosses
19580    /// the threshold on its own, so it is asserted only as the baseline, not
19581    /// as a distinguishing case for the escape.
19582    #[test]
19583    fn uses_v110_chunk_indexing_escapes_past_4_gib_at_every_bound() {
19584        let over_4gib = u32::MAX as u64 + 1;
19585        let small = 1024u64;
19586
19587        let path = temp_path("uses_v110_default");
19588        let writer = Hdf5Writer::create(&path).unwrap();
19589        assert!(writer.uses_v110_chunk_indexing(small));
19590        assert!(writer.uses_v110_chunk_indexing(over_4gib));
19591        writer.close().unwrap();
19592        std::fs::remove_file(&path).ok();
19593
19594        let path = temp_path("uses_v110_v18");
19595        let mut writer = Hdf5Writer::create(&path).unwrap();
19596        writer.set_libver_bound(LibverBound::V18).unwrap();
19597        assert!(
19598            !writer.uses_v110_chunk_indexing(small),
19599            "V18's layout row (3) stays below the v1.10 gate for an ordinary chunk"
19600        );
19601        assert!(
19602            writer.uses_v110_chunk_indexing(over_4gib),
19603            "the >4 GiB escape reaches v1.10 indexing despite V18's row"
19604        );
19605        writer.close().unwrap();
19606        std::fs::remove_file(&path).ok();
19607
19608        let path = temp_path("uses_v110_earliest");
19609        let mut writer = Hdf5Writer::create(&path).unwrap();
19610        writer.set_libver_bound(LibverBound::Earliest).unwrap();
19611        assert!(
19612            !writer.uses_v110_chunk_indexing(small),
19613            "Earliest's layout row (1) stays below the v1.10 gate for an ordinary chunk"
19614        );
19615        assert!(
19616            writer.uses_v110_chunk_indexing(over_4gib),
19617            "the >4 GiB escape reaches v1.10 indexing despite Earliest's row"
19618        );
19619        writer.close().unwrap();
19620        std::fs::remove_file(&path).ok();
19621    }
19622
19623    /// `H5D__chunk_set_info`'s closing `MAX3` (H5Dchunk.c:1046): the same
19624    /// escape pins the layout message itself at version 5 for a chunk over
19625    /// 4 GiB regardless of bound — `required_chunk_layout_version` dominates
19626    /// the max chain ahead of both the bound-derived preference and
19627    /// `LAYOUT_VERSION_DEFAULT`.
19628    #[test]
19629    fn chunk_layout_version_pins_5_past_4_gib_at_every_bound() {
19630        let over_4gib = u32::MAX as u64 + 1;
19631        let small = 1024u64;
19632
19633        let path = temp_path("chunk_ver_default");
19634        let writer = Hdf5Writer::create(&path).unwrap();
19635        assert_eq!(writer.chunk_layout_version(false, small), 4);
19636        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19637        writer.close().unwrap();
19638        std::fs::remove_file(&path).ok();
19639
19640        let path = temp_path("chunk_ver_v18");
19641        let mut writer = Hdf5Writer::create(&path).unwrap();
19642        writer.set_libver_bound(LibverBound::V18).unwrap();
19643        assert_eq!(writer.chunk_layout_version(false, small), 3);
19644        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19645        writer.close().unwrap();
19646        std::fs::remove_file(&path).ok();
19647
19648        let path = temp_path("chunk_ver_earliest");
19649        let mut writer = Hdf5Writer::create(&path).unwrap();
19650        writer.set_libver_bound(LibverBound::Earliest).unwrap();
19651        assert_eq!(
19652            writer.chunk_layout_version(false, small),
19653            LAYOUT_VERSION_DEFAULT
19654        );
19655        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19656        writer.close().unwrap();
19657        std::fs::remove_file(&path).ok();
19658    }
19659    /// `fsm_persist.h5` persists two managers — metadata and raw data. The
19660    /// reopen reads both, hands their merged sections to the allocator, and
19661    /// claims the four blocks the managers themselves occupy.
19662    #[test]
19663    fn a_persisting_file_reopens_with_its_free_sections() {
19664        let path = fixture_copy("fsm_persist.h5", "fsm_read");
19665        let writer = Hdf5Writer::open_append(&path).unwrap();
19666        let fs = writer.free_space.as_deref().expect("managers were read");
19667
19668        assert!(fs.info.persist);
19669        assert_eq!(fs.info.strategy, FileSpaceStrategy::FsmAggr);
19670        assert_eq!(fs.info.threshold, 1);
19671
19672        let sections = writer.allocator.free_blocks();
19673        // h5stat -S reports 1910 bytes of tracked free space for this file.
19674        assert_eq!(sections.iter().map(|s| s.1).sum::<u64>(), 1910);
19675        // Address-ordered, and no two sections touch: what the two managers
19676        // held separately came out coalesced.
19677        for w in sections.windows(2) {
19678            assert!(w[0].0 + w[0].1 < w[1].0, "{sections:?}");
19679        }
19680        // Two headers plus the two sections blocks they name.
19681        assert_eq!(fs.superseded.len(), 4);
19682        for &(addr, len) in &fs.superseded {
19683            assert!(len > 0);
19684            assert!(
19685                !sections
19686                    .iter()
19687                    .any(|&(a, l)| addr < a + l && a < addr + len),
19688                "manager block {addr:#x}+{len} sits in a free section"
19689            );
19690        }
19691        drop(writer);
19692        let _ = std::fs::remove_file(&path);
19693    }
19694
19695    /// A file created with non-default file-space properties carries the
19696    /// message that declares them, and one created to persist gets real
19697    /// managers as soon as anything is freed.
19698    #[test]
19699    fn a_created_file_declares_the_strategy_it_was_made_with() {
19700        let path = temp_path("fsm_create");
19701        {
19702            let w = Hdf5Writer::create_with_options(
19703                &path,
19704                FileCreateOptions {
19705                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
19706                    ..Default::default()
19707                },
19708            )
19709            .unwrap();
19710            let i = w
19711                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19712                .unwrap();
19713            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19714            w.close().unwrap();
19715        }
19716
19717        let info = read_only_append(&path)
19718            .free_space
19719            .as_deref()
19720            .expect("the created file declares a strategy")
19721            .info
19722            .clone();
19723        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
19724        assert!(info.persist);
19725        assert_eq!(info.threshold, 1);
19726        assert_eq!(info.page_size, 4096);
19727        // The alignment fragments the creation left behind are the file's
19728        // first free space, so the metadata manager already has an address
19729        // and the raw-data one, which nothing freed into, does not.
19730        assert_ne!(info.fs_addr[0], UNDEF_ADDR);
19731        assert!(info.fs_addr.iter().skip(1).all(|&a| a == UNDEF_ADDR));
19732
19733        // An append supersedes the root header and the extension, and that
19734        // freed space is what the managers now record.
19735        append_one(&path, "added", false);
19736        assert!(
19737            tracked_free_space(&path) > 0,
19738            "the append recorded no free space"
19739        );
19740        let _ = std::fs::remove_file(&path);
19741    }
19742
19743    /// The two strategies without managers, and the default. All three are
19744    /// `H5Pset_file_space_strategy` settings; only the default leaves the file
19745    /// without the message.
19746    #[test]
19747    fn a_strategy_without_managers_still_declares_itself() {
19748        for (strategy, persist) in [
19749            (FileSpaceStrategy::Aggr, true),
19750            (FileSpaceStrategy::None, false),
19751        ] {
19752            let path = temp_path("fsm_nomgr");
19753            {
19754                let w = Hdf5Writer::create_with_options(
19755                    &path,
19756                    FileCreateOptions {
19757                        file_space: FileSpaceConfig::new(strategy, persist, 7),
19758                        ..Default::default()
19759                    },
19760                )
19761                .unwrap();
19762                w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19763                    .unwrap();
19764                w.close().unwrap();
19765            }
19766            // Read through the reader, not the writer: a reopen only builds
19767            // free-space state for a file it will rewrite managers for, and
19768            // these two have none.
19769            let info = declared_file_space(&path).expect("the strategy is declared");
19770            assert_eq!(info.strategy, strategy);
19771            // `H5P__set_file_space_strategy` stores neither for a strategy
19772            // that has no managers, so both keep the library defaults.
19773            assert!(!info.persist);
19774            assert_eq!(info.threshold, 1);
19775            let _ = std::fs::remove_file(&path);
19776        }
19777    }
19778
19779    /// The library defaults are what a file says by saying nothing.
19780    #[test]
19781    fn the_default_strategy_writes_no_message() {
19782        let path = temp_path("fsm_default");
19783        {
19784            let w = Hdf5Writer::create_with_options(
19785                &path,
19786                FileCreateOptions {
19787                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, false, 1),
19788                    ..Default::default()
19789                },
19790            )
19791            .unwrap();
19792            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19793                .unwrap();
19794            w.close().unwrap();
19795        }
19796        assert!(declared_file_space(&path).is_none());
19797        let _ = std::fs::remove_file(&path);
19798    }
19799
19800    /// The file-space info message a file carries, read back the way any
19801    /// reader sees it.
19802    fn declared_file_space(path: &std::path::Path) -> Option<FileSpaceInfoMessage> {
19803        crate::io::reader::Hdf5Reader::open(path)
19804            .unwrap()
19805            .superblock_extension()
19806            .file_space_info
19807            .clone()
19808    }
19809
19810    /// A created paged file is laid out on its page grid: the superblock takes
19811    /// the whole of page zero and the rest of that page is the metadata
19812    /// manager's first section, which is what `H5MF__alloc_pagefs` gives
19813    /// `H5F__super_init`'s `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`.
19814    #[test]
19815    fn a_created_paged_file_lays_its_pages_out() {
19816        let path = temp_path("fsm_paged_created");
19817        {
19818            let w = Hdf5Writer::create_with_options(
19819                &path,
19820                FileCreateOptions {
19821                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
19822                    ..Default::default()
19823                },
19824            )
19825            .unwrap();
19826            let i = w
19827                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19828                .unwrap();
19829            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19830            w.close().unwrap();
19831        }
19832        let info = read_only_append(&path)
19833            .free_space
19834            .as_deref()
19835            .expect("the created file declares a strategy")
19836            .info
19837            .clone();
19838        assert_eq!(info.strategy, FileSpaceStrategy::Page);
19839        assert!(info.persist);
19840        assert_eq!(info.page_size, 4096);
19841        assert_eq!(
19842            std::fs::metadata(&path).unwrap().len() % info.page_size,
19843            0,
19844            "a paged file ends on a page boundary"
19845        );
19846        let _ = std::fs::remove_file(&path);
19847    }
19848
19849    /// A userblock has to be a whole number of pages, or every page boundary
19850    /// after it is off the file's own grid — `H5F__super_init` refuses one
19851    /// that is not (H5Fsuper.c:1182-1192).
19852    #[test]
19853    fn a_paged_file_refuses_a_userblock_smaller_than_its_page() {
19854        let path = temp_path("fsm_paged_userblock");
19855        let Err(err) = Hdf5Writer::create_with_options(
19856            &path,
19857            FileCreateOptions {
19858                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
19859                userblock: 512,
19860                ..Default::default()
19861            },
19862        ) else {
19863            panic!("a 512-byte userblock was accepted on a 4096-byte page");
19864        };
19865        assert!(
19866            format!("{err}").contains("multiple of its 4096-byte"),
19867            "{err}"
19868        );
19869        let _ = std::fs::remove_file(&path);
19870    }
19871
19872    /// A page size the builder names is the page the file is actually laid
19873    /// out in, not just a number the message repeats: every allocation is
19874    /// shaped by it and the file ends on one of its boundaries.
19875    #[test]
19876    fn a_file_created_at_a_non_default_page_size_allocates_by_it() {
19877        let path = temp_path("fsm_page_size_8k");
19878        {
19879            let w = Hdf5Writer::create_with_options(
19880                &path,
19881                FileCreateOptions {
19882                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19883                        .with_page_size(8192),
19884                    ..Default::default()
19885                },
19886            )
19887            .unwrap();
19888            let i = w
19889                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19890                .unwrap();
19891            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19892            w.close().unwrap();
19893        }
19894        let info = read_only_append(&path)
19895            .free_space
19896            .as_deref()
19897            .expect("the created file declares a strategy")
19898            .info
19899            .clone();
19900        assert_eq!(info.page_size, 8192);
19901        assert_eq!(
19902            std::fs::metadata(&path).unwrap().len() % 8192,
19903            0,
19904            "the file ends on one of the pages it was created with"
19905        );
19906        let _ = std::fs::remove_file(&path);
19907    }
19908
19909    /// The page size is the fourth of the four properties `H5F__super_init`
19910    /// compares against the library defaults (H5Fsuper.c:1092-1097), so
19911    /// naming it is on its own enough to give a file the message — under the
19912    /// default strategy, which allocates without it.
19913    #[test]
19914    fn a_non_default_page_size_alone_gives_the_file_a_message() {
19915        let path = temp_path("fsm_page_size_only");
19916        {
19917            let w = Hdf5Writer::create_with_options(
19918                &path,
19919                FileCreateOptions {
19920                    file_space: FileSpaceConfig::default().with_page_size(1024),
19921                    ..Default::default()
19922                },
19923            )
19924            .unwrap();
19925            w.close().unwrap();
19926        }
19927        let info = declared_file_space(&path)
19928            .expect("a file naming only a page size still carries the message");
19929        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
19930        assert!(!info.persist);
19931        assert_eq!(info.page_size, 1024);
19932        let _ = std::fs::remove_file(&path);
19933    }
19934
19935    /// `H5Pset_file_space_page_size` refuses anything below 512 or above
19936    /// 1 GiB (H5Pfcpl.c:1389-1393), and nothing between: no power of two is
19937    /// required, so a size the bounds admit is one the file may carry.
19938    #[test]
19939    fn a_page_size_outside_the_library_bounds_is_refused() {
19940        for size in [0, 1, 511, PAGE_SIZE_MAX + 1] {
19941            let path = temp_path(&format!("fsm_page_size_bad_{size}"));
19942            let Err(err) = Hdf5Writer::create_with_options(
19943                &path,
19944                FileCreateOptions {
19945                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19946                        .with_page_size(size),
19947                    ..Default::default()
19948                },
19949            ) else {
19950                panic!("a {size}-byte file-space page was accepted");
19951            };
19952            assert!(
19953                format!("{err}").contains("between 512 bytes and 1073741824"),
19954                "{err}"
19955            );
19956            let _ = std::fs::remove_file(&path);
19957        }
19958        let path = temp_path("fsm_page_size_odd");
19959        let w = Hdf5Writer::create_with_options(
19960            &path,
19961            FileCreateOptions {
19962                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19963                    .with_page_size(513),
19964                ..Default::default()
19965            },
19966        )
19967        .expect("513 is inside the bounds, and no power of two is required");
19968        w.close().unwrap();
19969        let _ = std::fs::remove_file(&path);
19970    }
19971
19972    /// A paged file's managers are read on reopen, the same as any other
19973    /// file's: paged aggregation changes which manager a request maps to, not
19974    /// whether the file has managers to rewrite.
19975    #[test]
19976    fn a_paged_file_reports_the_managers_it_persists() {
19977        let path = fixture_copy("fsm_persist_page.h5", "fsm_read_paged");
19978        let writer = Hdf5Writer::open_append(&path).unwrap();
19979        let fs = writer.free_space.as_deref().expect("no managers read");
19980        assert_eq!(fs.info.strategy, FileSpaceStrategy::Page);
19981        assert!(
19982            !writer.allocator.free_extents().is_empty(),
19983            "the sections the file records were not put back in circulation"
19984        );
19985        drop(writer);
19986        let _ = std::fs::remove_file(&path);
19987    }
19988
19989    /// A file with no file-space info message at all — every file this crate
19990    /// creates — has nothing to read and nothing to write back.
19991    #[test]
19992    fn a_file_without_a_strategy_has_no_managers() {
19993        let path = temp_path("fsm_none");
19994        {
19995            let w = Hdf5Writer::create(&path).unwrap();
19996            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19997                .unwrap();
19998            w.close().unwrap();
19999        }
20000        let writer = Hdf5Writer::open_append(&path).unwrap();
20001        assert!(writer.free_space.is_none());
20002        drop(writer);
20003        let _ = std::fs::remove_file(&path);
20004    }
20005    /// Sum of the sections the managers a file names actually hold — what
20006    /// `h5stat -S` prints as "Amount of tracked free space", read back through
20007    /// this crate's own decoder so a test can assert on it. A reopen seeds the
20008    /// allocator with exactly those sections, so its free list is the number.
20009    fn tracked_free_space(path: &std::path::Path) -> u64 {
20010        read_only_append(path)
20011            .allocator
20012            .free_blocks()
20013            .iter()
20014            .map(|b| b.1)
20015            .sum()
20016    }
20017
20018    /// Open for append and mark the writer closed, so dropping it releases the
20019    /// file lock instead of finalizing and rewriting what is being inspected.
20020    fn read_only_append(path: &std::path::Path) -> Hdf5Writer {
20021        let mut w = Hdf5Writer::open_append(path).unwrap();
20022        w.closed = true;
20023        w
20024    }
20025
20026    /// Add one small dataset, the smallest append that still rewrites the root
20027    /// header, the superblock extension and — on a persisting file — the
20028    /// free-space manager.
20029    fn append_one(path: &std::path::Path, name: &str, disable_managers: bool) {
20030        let mut w = Hdf5Writer::open_append(path).unwrap();
20031        if disable_managers {
20032            // Both halves of the change, so the control is the file as this
20033            // crate wrote it before: the session neither allocates from the
20034            // recorded sections nor writes any back.
20035            w.free_space = None;
20036            w.allocator.reset_free_list(&[]);
20037        }
20038        let i = w
20039            .create_dataset(name, DatatypeMessage::i32_type(), &[8])
20040            .unwrap();
20041        w.write_dataset_raw(
20042            i,
20043            &(0..8i32).flat_map(|v| v.to_le_bytes()).collect::<Vec<u8>>(),
20044        )
20045        .unwrap();
20046        w.close().unwrap();
20047    }
20048
20049    /// The block list a reopen carries for the superblock extension covers
20050    /// every chunk of the header, not just the first. The fixture's extension
20051    /// is a two-chunk header — libhdf5 put the file-space info message in a
20052    /// continuation — and freeing chunk zero alone left the continuation
20053    /// allocated with nothing naming it.
20054    #[test]
20055    fn a_reopen_carries_every_chunk_of_the_superblock_extension() {
20056        let path = fixture_copy("fsm_persist.h5", "fsm_ext_chunks");
20057        let blocks = read_only_append(&path).extension.superseded.clone();
20058        assert!(
20059            blocks.len() > 1,
20060            "the fixture's extension is one chunk, so this proves nothing: {blocks:?}"
20061        );
20062        let _ = std::fs::remove_file(&path);
20063    }
20064
20065    /// An append on a persisting file both spends and records the space its
20066    /// managers track: the new dataset comes out of the sections the file
20067    /// already had, and what the rewrite frees goes back into them.
20068    #[test]
20069    fn an_append_reuses_and_records_the_space_the_managers_track() {
20070        let path = fixture_copy("fsm_persist.h5", "fsm_write");
20071        let original = std::fs::metadata(&path).unwrap().len();
20072        let before = tracked_free_space(&path);
20073        assert_eq!(before, 1910, "the fixture's own managers");
20074
20075        append_one(&path, "added", false);
20076        let size = std::fs::metadata(&path).unwrap().len();
20077        let tracked = tracked_free_space(&path);
20078
20079        // Negative control: the same append with both halves of this off — no
20080        // allocating out of the recorded sections and no writing any back —
20081        // which is what this crate did before it read free space at all.
20082        let control = fixture_copy("fsm_persist.h5", "fsm_write_control");
20083        append_one(&control, "added", true);
20084        let control_size = std::fs::metadata(&control).unwrap().len();
20085        assert_eq!(
20086            tracked_free_space(&control),
20087            before,
20088            "with the manager rewrite disabled the number must not move"
20089        );
20090
20091        // The new dataset's raw data comes out of the raw-data sections the
20092        // file already recorded, so the append grows the file by less than the
20093        // same append with the reuse off. It does not stop the growth:
20094        // `H5MF_alloc` asks one manager and no other, and of this fixture's
20095        // 1910 free bytes 1848 are raw-data ones, so the metadata the append
20096        // writes still comes from the end of the file.
20097        assert!(
20098            size < control_size,
20099            "the append took nothing from the {before} bytes free: \
20100             {original} grew to {size}, the control to {control_size}"
20101        );
20102        assert!(
20103            control_size > original,
20104            "the control has to grow or it proves nothing"
20105        );
20106        // Space no manager and no object claims — `h5stat -S`'s "unaccounted
20107        // space" — is what the leak was, and it is smaller now.
20108        assert!(
20109            size - tracked < control_size - before,
20110            "unaccounted space went from {} to {}",
20111            control_size - before,
20112            size - tracked
20113        );
20114
20115        for p in [&path, &control] {
20116            let _ = std::fs::remove_file(p);
20117        }
20118    }
20119
20120    /// The set the writer holds free when it finishes is exactly the set the
20121    /// manager it just wrote records — the invariant that makes the on-disk
20122    /// managers a faithful account of the file's free space.
20123    #[test]
20124    fn the_manager_records_the_free_list_the_close_ends_with() {
20125        let path = fixture_copy("fsm_persist.h5", "fsm_roundtrip");
20126        let internal = {
20127            let mut w = Hdf5Writer::open_append(&path).unwrap();
20128            let i = w
20129                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20130                .unwrap();
20131            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20132            w.finalize(true).unwrap();
20133            let blocks = w.allocator.free_extents();
20134            w.closed = true;
20135            blocks
20136        };
20137        assert!(!internal.is_empty(), "the append freed nothing");
20138
20139        // Classes included: a section read back out of the wrong manager is a
20140        // section libhdf5 would offer to the wrong kind of allocation.
20141        let reread = {
20142            let w = read_only_append(&path);
20143            assert!(w.free_space.is_some(), "managers were written");
20144            w.allocator.free_extents()
20145        };
20146        assert_eq!(internal, reread);
20147        let _ = std::fs::remove_file(&path);
20148    }
20149
20150    /// The paged half of
20151    /// [`the_manager_records_the_free_list_the_close_ends_with`]: a paged
20152    /// file's sections carry a page and a class as well as an address, and a
20153    /// section written into the wrong manager or split across a page boundary
20154    /// would come back different.
20155    #[test]
20156    fn the_manager_records_the_free_list_a_paged_close_ends_with() {
20157        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_roundtrip");
20158        let internal = {
20159            let mut w = Hdf5Writer::open_append(&path).unwrap();
20160            let i = w
20161                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20162                .unwrap();
20163            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20164            w.finalize(true).unwrap();
20165            let blocks = w.allocator.free_extents();
20166            w.closed = true;
20167            blocks
20168        };
20169        assert!(!internal.is_empty(), "the append freed nothing");
20170
20171        let reread = {
20172            let w = read_only_append(&path);
20173            assert!(w.free_space.is_some(), "managers were written");
20174            w.allocator.free_extents()
20175        };
20176        assert_eq!(internal, reread);
20177        let _ = std::fs::remove_file(&path);
20178    }
20179
20180    /// Negative control for the paged managers: with the read and the rewrite
20181    /// both off — the file as this crate handled a paged file before — the
20182    /// space the append frees is recorded nowhere, and the number this crate
20183    /// reads back is the fixture's own.
20184    #[test]
20185    fn a_paged_append_records_nothing_without_the_manager_rewrite() {
20186        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_measured");
20187        let control = fixture_copy("fsm_persist_page.h5", "fsm_paged_control");
20188        let before = tracked_free_space(&path);
20189        let original = std::fs::metadata(&path).unwrap().len();
20190
20191        append_one(&path, "added", false);
20192        append_one(&control, "added", true);
20193
20194        assert_eq!(
20195            tracked_free_space(&control),
20196            before,
20197            "the control moved the number it is there to hold still"
20198        );
20199        assert_eq!(
20200            std::fs::metadata(&path).unwrap().len(),
20201            original,
20202            "the append grew a paged file with {before} bytes recorded free"
20203        );
20204        assert!(
20205            std::fs::metadata(&control).unwrap().len() > original,
20206            "the control has to grow or it proves nothing"
20207        );
20208        assert_ne!(
20209            tracked_free_space(&path),
20210            before,
20211            "the managers came back holding what the fixture wrote"
20212        );
20213        for p in [&path, &control] {
20214            let _ = std::fs::remove_file(p);
20215        }
20216    }
20217
20218    /// A block released from a dataset's raw data is recorded by the manager
20219    /// `H5MF_ALLOC_TO_FS_AGGR_TYPE` maps `H5FD_MEM_DRAW` to, and nothing else
20220    /// is: the dichotomy the sec2 driver installs is what decides, and the two
20221    /// managers it collapses to are the file-space info message's slots 0 and
20222    /// 2.
20223    #[test]
20224    fn a_released_raw_block_lands_in_the_raw_data_manager() {
20225        let path = temp_path("fsm_dichotomy");
20226        {
20227            let w = Hdf5Writer::create_with_options(
20228                &path,
20229                FileCreateOptions {
20230                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
20231                    ..Default::default()
20232                },
20233            )
20234            .unwrap();
20235            let i = w
20236                .create_dataset("bulk", DatatypeMessage::i32_type(), &[256])
20237                .unwrap();
20238            w.write_dataset_raw(i, &vec![0u8; 1024]).unwrap();
20239            w.create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20240                .unwrap();
20241            w.close().unwrap();
20242        }
20243        let (raw_addr, raw_len) = {
20244            let w = read_only_append(&path);
20245            let i = w.dataset_index("bulk").unwrap();
20246            let ds = w.ds(i);
20247            let m = ds.lock();
20248            (m.data_addr, m.data_size)
20249        };
20250        assert!(raw_len >= 1024, "the raw block is {raw_len} bytes");
20251        {
20252            let w = Hdf5Writer::open_append(&path).unwrap();
20253            w.delete_dataset("bulk").unwrap();
20254            w.close().unwrap();
20255        }
20256
20257        let mut w = read_only_append(&path);
20258        let info = w
20259            .free_space
20260            .as_deref()
20261            .expect("the file persists managers")
20262            .info
20263            .clone();
20264        assert_ne!(info.fs_addr[0], UNDEF_ADDR, "no metadata manager");
20265        assert_ne!(info.fs_addr[2], UNDEF_ADDR, "no raw-data manager");
20266        for (slot, &addr) in info.fs_addr.iter().enumerate() {
20267            if slot != 0 && slot != 2 {
20268                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
20269            }
20270        }
20271
20272        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20273        let inside = |b: &FreeBlock| b.addr >= raw_addr && b.addr + b.len <= raw_addr + raw_len;
20274        let raw: Vec<&FreeBlock> = found
20275            .sections
20276            .iter()
20277            .filter(|b| b.manager == FreeSpaceManager::RawData)
20278            .collect();
20279        assert!(
20280            !raw.is_empty(),
20281            "the deleted dataset's bytes were not recorded"
20282        );
20283        assert!(
20284            raw.iter().all(|b| inside(b)),
20285            "a raw-data section is outside the deleted dataset's block: {raw:?}"
20286        );
20287        assert!(
20288            found
20289                .sections
20290                .iter()
20291                .filter(|b| b.manager == FreeSpaceManager::Metadata)
20292                .all(|b| !inside(b)),
20293            "raw-data bytes were recorded by the metadata manager"
20294        );
20295        drop(w);
20296        let _ = std::fs::remove_file(&path);
20297    }
20298
20299    /// A reopened paged file's managers are this writer's to rewrite, and the
20300    /// three the sec2 driver can reach are the only ones it names.
20301    ///
20302    /// `H5MF__alloc_to_fs_type` (H5MF.c:265) sends a request of at least one
20303    /// page to `H5F_MEM_PAGE_GENERIC` unless the driver declares
20304    /// `H5FD_FEAT_PAGED_AGGR`, which only the multi and split drivers do, so a
20305    /// sec2 file has the dichotomy's two small managers and that one large
20306    /// one: message slots 0, 2 and 6.
20307    #[test]
20308    fn a_paged_file_names_only_the_managers_sec2_can_reach() {
20309        let path = fixture_copy("fsm_persist_page.h5", "fsm_write_paged");
20310        assert!(
20311            read_only_append(&path).free_space.is_some(),
20312            "the paged fixture's managers were not read"
20313        );
20314        append_one(&path, "added", false);
20315
20316        let mut w = read_only_append(&path);
20317        let info = w
20318            .free_space
20319            .as_deref()
20320            .expect("the file persists managers")
20321            .info
20322            .clone();
20323        assert_eq!(info.strategy, FileSpaceStrategy::Page);
20324        for (slot, &addr) in info.fs_addr.iter().enumerate() {
20325            if !matches!(slot, 0 | 2 | 6) {
20326                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
20327            }
20328        }
20329        assert!(
20330            info.fs_addr.iter().any(|&a| a != UNDEF_ADDR),
20331            "the rewritten file records nothing free"
20332        );
20333        crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20334        drop(w);
20335        let _ = std::fs::remove_file(&path);
20336    }
20337
20338    /// Every section a paged file records sits inside one page, and the pages
20339    /// its small managers use are pages of their own kind — the invariant
20340    /// `H5MF__alloc_pagefs` maintains by giving each small request a whole
20341    /// page of its class and recording the rest of it in that class's manager.
20342    #[test]
20343    fn a_paged_files_small_sections_stay_inside_one_page_of_one_kind() {
20344        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_pages");
20345        append_one(&path, "added", false);
20346
20347        let mut w = read_only_append(&path);
20348        let info = w
20349            .free_space
20350            .as_deref()
20351            .expect("the file persists managers")
20352            .info
20353            .clone();
20354        let page = info.page_size;
20355        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20356        let mut kind_of_page: std::collections::HashMap<u64, FreeSpaceManager> =
20357            std::collections::HashMap::new();
20358        for section in &found.sections {
20359            if section.manager == FreeSpaceManager::Large {
20360                continue;
20361            }
20362            assert_eq!(
20363                section.addr / page,
20364                (section.addr + section.len - 1) / page,
20365                "the section at {:#x} crosses a page boundary",
20366                section.addr
20367            );
20368            let owner = kind_of_page
20369                .entry(section.addr / page)
20370                .or_insert(section.manager);
20371            assert_eq!(
20372                *owner,
20373                section.manager,
20374                "page {} holds sections of two kinds",
20375                section.addr / page
20376            );
20377        }
20378        drop(w);
20379        let _ = std::fs::remove_file(&path);
20380    }
20381}