Skip to main content

rust_hdf5/io/
writer.rs

1//! HDF5 file writer.
2//!
3//! Produces a valid HDF5 file with superblock v3, a root group object header,
4//! and datasets with contiguous or chunked storage. The output is readable by `h5dump`.
5
6use std::collections::{HashMap, HashSet};
7use std::path::{Path, PathBuf};
8
9use crate::dataset::DatasetAccess;
10use crate::format::btree_v1::{BTreeV1Config, ChunkBTreeV1Node, ChunkBTreeV1Tree, ChunkKey};
11use crate::format::chunk_index::btree_v2::Bt2ChunkIndex;
12use crate::format::chunk_index::extensible_array::{
13    compute_chunk_size_len, compute_ndblk_addrs, compute_nsblk_addrs, EaDblkPath, EaGeometry,
14    EaLoc, ExtensibleArrayDataBlock, ExtensibleArrayHeader, ExtensibleArrayIndexBlock,
15    ExtensibleArraySuperBlock, FilteredChunkEntry, FilteredDataBlock, FilteredIndexBlock,
16    EA_CLS_CHUNK, EA_CLS_FILT_CHUNK,
17};
18use crate::format::chunk_index::fixed_array::{
19    decode_filtered_page, decode_unfiltered_page, encode_filtered_page, encode_unfiltered_page,
20    FixedArrayDataBlock, FixedArrayFilteredChunkElement, FixedArrayHeader, FixedArrayPagedPrefix,
21    FA_CLIENT_FILT_CHUNK,
22};
23use crate::format::creation_order::CreationOrder;
24use crate::format::dense_attr::build_dense_attributes;
25use crate::format::dense_link::build_dense_links;
26use crate::format::free_space::{
27    self, FreeSection, FreeSpaceClass, FreeSpaceHeader, FreeSpaceManager,
28};
29use crate::format::local_heap::{
30    local_heap_header_size, LocalHeapHeader, LocalHeapImage, LOCAL_HEAP_FREE_NULL,
31};
32use crate::format::messages::attr_info::{next_creation_index, AttributeInfoMessage};
33use crate::format::messages::attribute::{
34    AttributeEntry, AttributeMessage, ATTR_FLAG_SPACE_SHARED, ATTR_FLAG_TYPE_SHARED,
35};
36use crate::format::messages::data_layout::{
37    DataLayoutMessage, EarrayParams, FixedArrayParams, LAYOUT_VERSION_DEFAULT,
38};
39use crate::format::messages::dataspace::{DataspaceClass, DataspaceMessage};
40use crate::format::messages::datatype::{DatatypeMessage, ReferenceKind};
41use crate::format::messages::external_file_list::{ExternalFileListMessage, UNLIMITED};
42use crate::format::messages::fill_value::{
43    FillValueMessage, FILL_TIME_ALLOC, FILL_TIME_IFSET, FILL_TIME_NEVER,
44};
45use crate::format::messages::filter::{self, FilterPipeline};
46use crate::format::messages::group_info::GroupInfoMessage;
47use crate::format::messages::link::{CharacterSet, LinkMessage, LinkTarget};
48use crate::format::messages::link_info::LinkInfoMessage;
49use crate::format::messages::mod_time::ModificationTime;
50use crate::format::messages::superblock_ext::{
51    FileSpaceInfoMessage, FileSpaceStrategy, SharedMessageTableMessage,
52    DEFAULT_FILE_SPACE_PAGE_SIZE, FS_ADDR_COUNT_V1, PAGE_SIZE_MAX, PAGE_SIZE_MIN,
53};
54use crate::format::messages::virtual_mapping::{
55    parse_source_name, VirtualMapping, VirtualMappingList,
56};
57use crate::format::messages::*;
58use crate::format::object_header::{ObjectHeader, ObjectTimes, MAX_MESSAGE_SIZE};
59use crate::format::reference::{
60    encode_reference_element, encode_revised_blob, ReferenceElementImage, ReferenceTarget,
61    REVISED_BLOB_TOKEN_OFFSET,
62};
63use crate::format::selection::Selection;
64use crate::format::sohm::{
65    type_flag, SharedMessagePointer, MAX_SOHM_INDEXES, SOHM_HEAP_ID_LEN, SOHM_POINTER_HEAP_ID_AT,
66};
67use crate::format::sohm_write::{
68    build_shared_messages, NestedShare, SharedMessage, SohmIndexContent, SohmIndexSpec,
69};
70use crate::format::superblock::*;
71use crate::format::{FormatContext, LibverBound, ObjectFormat, UNDEF_ADDR};
72
73use crate::io::allocator::{FileAllocator, FreeBlock};
74use crate::io::file_handle::FileHandle;
75use crate::io::hyperslab::{for_each_contiguous_run, for_each_dual_run};
76use crate::io::symbol_table_io::{free_stab, write_stab, Stab, StabExtents, StabLink, StabTarget};
77use crate::io::{FileMeta, IoResult};
78
79/// On-disk size in bytes of a fixed-array data block, for the layout (paged or
80/// flat) implied by `hdr`.
81///
82/// Mirrors `H5FA_DBLOCK_SIZE` (`H5FApkg.h`):
83///   - non-paged: `prefix + nelmts * raw_elmt_size + checksum`
84///   - paged: `prefix + page_init_bitmap + nelmts * raw_elmt_size
85///     + npages * checksum`, where the prefix checksum covers the bitmap.
86///
87/// `raw_elmt_size` is `sizeof_addr` for an unfiltered array, and
88/// `sizeof_addr + chunk_size_len + 4` (the filtered element: address +
89/// compressed size + filter mask) for a filtered array. libhdf5 carries this
90/// value as `hdr->cparam.raw_elmt_size`, i.e. exactly `hdr.element_size`.
91fn fixed_array_dblk_disk_size(ctx: &FormatContext, hdr: &FixedArrayHeader) -> u64 {
92    let elem_size = hdr.element_size as u64;
93    let sa = ctx.sizeof_addr as u64;
94    let nelmts = hdr.num_elmts;
95    // Common metadata prefix: signature(4) + version(1) + client_id(1) + header_addr(sa).
96    let meta_prefix = 4 + 1 + 1 + sa;
97    if hdr.is_paged() {
98        let npages = hdr.npages();
99        let bitmap_size = npages.div_ceil(8);
100        // prefix (incl. its own 4-byte checksum) + elements + per-page checksums.
101        (meta_prefix + bitmap_size + 4) + nelmts * elem_size + npages * 4
102    } else {
103        // prefix + elements + single 4-byte checksum.
104        meta_prefix + nelmts * elem_size + 4
105    }
106}
107
108/// A walk of a v2 B-tree: the file and node geometry the descent reads
109/// through, and the two collections it fills — every node's raw record
110/// bytes and every node block's address, the latter because `open_append`
111/// needs it so the reconstructed [`Bt2DatasetInfo::node_addrs`] pool owns
112/// the on-disk nodes (the next flush re-serializes the tree over them, and
113/// a delete frees them).
114///
115/// `record_size`, `node_size` and `geo` are constant for the whole walk, so
116/// [`descend`](Self::descend) takes only what changes per level: the node's
117/// address, its depth, and how many records it holds.
118struct Bt2Walk<'a> {
119    handle: &'a FileHandle,
120    ctx: &'a FormatContext,
121    record_size: u16,
122    node_size: u32,
123    geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
124    records: Vec<u8>,
125    node_addrs: Vec<u64>,
126}
127
128impl<'a> Bt2Walk<'a> {
129    fn new(
130        handle: &'a FileHandle,
131        ctx: &'a FormatContext,
132        record_size: u16,
133        node_size: u32,
134        geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
135    ) -> Self {
136        Self {
137            handle,
138            ctx,
139            record_size,
140            node_size,
141            geo,
142            records: Vec::new(),
143            node_addrs: Vec::new(),
144        }
145    }
146
147    /// Walk the subtree rooted at `addr`, at depth `depth` with `nrec`
148    /// records, collecting every node's raw record bytes and every node
149    /// block's address.
150    fn descend(&mut self, addr: u64, depth: u16, nrec: u16) -> IoResult<()> {
151        use crate::format::chunk_index::btree_v2::{Bt2InternalNode, Bt2LeafNode};
152
153        self.node_addrs.push(addr);
154        let buf = self.handle.read_at_most(addr, self.node_size as usize)?;
155        if depth == 0 {
156            let leaf = Bt2LeafNode::decode(&buf, nrec, self.record_size)?;
157            self.records.extend_from_slice(&leaf.record_data);
158        } else {
159            let node = Bt2InternalNode::decode(
160                &buf,
161                self.ctx,
162                depth,
163                nrec,
164                self.record_size,
165                self.geo.max_nrec_size,
166                self.geo.child_total_size(depth),
167            )?;
168            // In-order: an internal node's records separate its children, so each
169            // one belongs between the subtrees on either side of it.
170            let children: Vec<(u64, u16)> = node
171                .child_addrs
172                .iter()
173                .zip(node.child_nrecords.iter())
174                .map(|(&a, &n)| (a, n))
175                .collect();
176            let rec = self.record_size as usize;
177            for (i, (child_addr, child_nrec)) in children.into_iter().enumerate() {
178                self.descend(child_addr, depth - 1, child_nrec)?;
179                if let Some(record) = node.record_data.get(i * rec..(i + 1) * rec) {
180                    self.records.extend_from_slice(record);
181                }
182            }
183        }
184        Ok(())
185    }
186}
187
188/// A walk of a version-1 raw-data-chunk B-tree: the file and geometry the
189/// descent reads through, and the two collections it fills.
190///
191/// The v1 counterpart of [`Bt2Walk`], and for the same reason: the
192/// records are what [`BtreeV1DatasetInfo::build_tree`] bulk-loads on the next
193/// flush, and the addresses are the block pool that flush re-serializes over,
194/// so a reopened tree owns the nodes it found instead of leaking them and
195/// allocating a second set beside them.
196///
197/// One value rather than nine parameters threaded through the recursion: only
198/// `addr` and `depth` change between one level and the next, so they are what
199/// [`descend`](Self::descend) takes and everything else lives here.
200struct BtreeV1Walk<'a> {
201    handle: &'a FileHandle,
202    ctx: &'a FormatContext,
203    config: &'a BTreeV1Config,
204    /// The chunk edge lengths, *without* the trailing element-size dimension,
205    /// so `chunk_dims.len()` is the rank the node keys are decoded at.
206    chunk_dims: &'a [u64],
207    file_size: u64,
208    records: Vec<BtreeV1ChunkRecord>,
209    node_addrs: Vec<u64>,
210}
211
212impl<'a> BtreeV1Walk<'a> {
213    fn new(
214        handle: &'a FileHandle,
215        ctx: &'a FormatContext,
216        config: &'a BTreeV1Config,
217        chunk_dims: &'a [u64],
218        file_size: u64,
219    ) -> Self {
220        Self {
221            handle,
222            ctx,
223            config,
224            chunk_dims,
225            file_size,
226            records: Vec::new(),
227            node_addrs: Vec::new(),
228        }
229    }
230
231    /// Walk the subtree rooted at `addr`, collecting every leaf entry as a
232    /// [`BtreeV1ChunkRecord`] and every node block's address.
233    ///
234    /// Records come out in key order because a v1 B-tree's leaves are in key
235    /// order and this descends left to right, which is what
236    /// [`BtreeV1DatasetInfo::position`]'s binary search needs. The keys store
237    /// element offsets (`scaled * chunk_dim`, `H5D__btree_encode_key`), so the
238    /// grid position this records is the quotient.
239    fn descend(&mut self, addr: u64, depth: u32) -> IoResult<()> {
240        // The same bound the reader's walk uses: a node's level is one byte, so
241        // no honest tree is deeper than that, and a cyclic index stops here.
242        if depth > 256 {
243            return Err(crate::io::IoError::InvalidState(
244                "chunk B-tree v1 exceeds maximum depth".into(),
245            ));
246        }
247        if addr == UNDEF_ADDR || addr >= self.file_size {
248            return Ok(());
249        }
250        let rank = self.chunk_dims.len();
251        let sa = self.ctx.sizeof_addr as usize;
252        let node_size = self.config.chunk_btree_node_size(sa, rank);
253        let buf = self.handle.read_at_most(addr, node_size)?;
254        let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.config.chunk_max_entries())?;
255        self.node_addrs.push(addr);
256
257        if node.level == 0 {
258            for (i, &child_addr) in node.children.iter().enumerate() {
259                let key = &node.keys[i];
260                let scaled: Vec<u64> = key.offsets[..rank]
261                    .iter()
262                    .zip(self.chunk_dims)
263                    .map(|(&offset, &dim)| offset.checked_div(dim).unwrap_or(0))
264                    .collect();
265                self.records.push(BtreeV1ChunkRecord {
266                    scaled,
267                    address: child_addr,
268                    nbytes: key.chunk_size,
269                    filter_mask: key.filter_mask,
270                });
271            }
272        } else {
273            for &child_addr in &node.children {
274                self.descend(child_addr, depth + 1)?;
275            }
276        }
277        Ok(())
278    }
279}
280
281/// Encode a fixed-array data block for the layout implied by `hdr`, using the
282/// chunk addresses held in `dblk.elements` (unfiltered) or the filtered chunk
283/// entries in `dblk.filtered_elements` (filtered, `client_id == 1`).
284///
285/// For the paged layout (`hdr.is_paged()`), emits the `FADB` prefix with a
286/// page-init bitmap followed by `npages` checksummed element pages. A page is
287/// marked initialized iff at least one of its chunk addresses is defined,
288/// mirroring libhdf5's lazy `H5FA__dblk_page_create`. Uninitialized pages are
289/// still written (all `UNDEF_ADDR`, valid checksum) so the file contains no
290/// uninitialized bytes; the reader skips them via the bitmap.
291fn encode_fixed_array_dblk(
292    ctx: &FormatContext,
293    hdr: &FixedArrayHeader,
294    dblk: &FixedArrayDataBlock,
295) -> Vec<u8> {
296    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
297    let sa = ctx.sizeof_addr as usize;
298    // chunk_size_len for filtered entries = element_size - sizeof_addr - 4.
299    // libhdf5 carries element_size = sizeof_addr + chunk_size_len + 4.
300    let chunk_size_len = (hdr.element_size as usize).saturating_sub(sa + 4);
301
302    if !hdr.is_paged() {
303        return if is_filtered {
304            dblk.encode_filtered(ctx, chunk_size_len)
305        } else {
306            dblk.encode_unfiltered(ctx)
307        };
308    }
309
310    let npages = hdr.npages() as usize;
311    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
312
313    // Build the page-init bitmap (MSB-first): a page is initialized iff any of
314    // its elements points at a defined address.
315    let mut bitmap = vec![0u8; npages.div_ceil(8)];
316    let nelmts = if is_filtered {
317        dblk.filtered_elements.len()
318    } else {
319        dblk.elements.len()
320    };
321    for p in 0..npages {
322        let start = p * dblk_page_nelmts;
323        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
324        let initialized = if is_filtered {
325            dblk.filtered_elements[start..end]
326                .iter()
327                .any(|e| e.address != UNDEF_ADDR)
328        } else {
329            dblk.elements[start..end].iter().any(|&a| a != UNDEF_ADDR)
330        };
331        if initialized {
332            bitmap[p / 8] |= 0x80u8 >> (p % 8);
333        }
334    }
335
336    let prefix = FixedArrayPagedPrefix {
337        client_id: hdr.client_id,
338        header_addr: dblk.header_addr,
339        page_init_bitmap: bitmap,
340        prefix_size: 4 + 1 + 1 + sa + npages.div_ceil(8) + 4,
341    };
342
343    let mut buf = prefix.encode(ctx);
344    debug_assert_eq!(buf.len(), prefix.prefix_size);
345
346    // Append each page: all pages use the full `dblk_page_nelmts` stride;
347    // only the last page holds fewer elements (libhdf5 H5FA.c).
348    for p in 0..npages {
349        let start = p * dblk_page_nelmts;
350        let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
351        if is_filtered {
352            buf.extend_from_slice(&encode_filtered_page(
353                &dblk.filtered_elements[start..end],
354                ctx,
355                chunk_size_len,
356            ));
357        } else {
358            buf.extend_from_slice(&encode_unfiltered_page(&dblk.elements[start..end], ctx));
359        }
360    }
361    buf
362}
363
364/// Decode a fixed-array data block for the layout implied by `hdr` — the
365/// inverse of [`encode_fixed_array_dblk`], and the single decode dispatch
366/// over non-paged/paged × unfiltered/filtered.
367///
368/// For the paged layout, pages whose bitmap bit is clear are skipped, not
369/// decoded: libhdf5 never writes an uninitialized page, so its bytes are
370/// arbitrary and carry no valid checksum. Their elements stay at the
371/// undefined-address defaults, which is exactly what the bitmap means.
372fn decode_fixed_array_dblk(
373    ctx: &FormatContext,
374    hdr: &FixedArrayHeader,
375    buf: &[u8],
376    chunk_size_len: usize,
377) -> crate::format::FormatResult<FixedArrayDataBlock> {
378    let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
379    let num_elmts = hdr.num_elmts as usize;
380
381    if !hdr.is_paged() {
382        return if is_filtered {
383            FixedArrayDataBlock::decode_filtered(buf, ctx, num_elmts, chunk_size_len)
384        } else {
385            FixedArrayDataBlock::decode_unfiltered(buf, ctx, num_elmts)
386        };
387    }
388
389    let npages = hdr.npages() as usize;
390    let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
391    let prefix = FixedArrayPagedPrefix::decode(buf, ctx, npages as u64)?;
392
393    let mut dblk = if is_filtered {
394        FixedArrayDataBlock::new_filtered(prefix.header_addr, num_elmts)
395    } else {
396        FixedArrayDataBlock::new_unfiltered(prefix.header_addr, num_elmts)
397    };
398    dblk.client_id = hdr.client_id;
399
400    // Pages follow the prefix back to back; every page spans the full
401    // `dblk_page_nelmts` stride except the last, which holds the remainder.
402    let mut pos = prefix.prefix_size;
403    for p in 0..npages {
404        let start = p * dblk_page_nelmts;
405        let end = ((p + 1) * dblk_page_nelmts).min(num_elmts);
406        let nelmts = end - start;
407        if prefix.page_initialized(p) {
408            let page_buf = buf.get(pos..).unwrap_or(&[]);
409            if is_filtered {
410                let elems = decode_filtered_page(page_buf, ctx, nelmts, chunk_size_len)?;
411                dblk.filtered_elements[start..end].clone_from_slice(&elems);
412            } else {
413                let addrs = decode_unfiltered_page(page_buf, ctx, nelmts)?;
414                dblk.elements[start..end].copy_from_slice(&addrs);
415            }
416        }
417        pos += nelmts * hdr.element_size as usize + 4;
418    }
419    Ok(dblk)
420}
421
422/// Interior-mutability cell for per-dataset write state, selected by feature.
423///
424/// This is the §5-B "cfg-selected interior types" from
425/// `docs/threadsafe-fine-grained-locking.md`: the single-threaded build uses a
426/// `RefCell` (zero overhead, no atomics), while the `threadsafe` build uses a
427/// `Mutex` so two threads can write *different* datasets concurrently while the
428/// same dataset's writes serialize. Call sites are identical across both via
429/// [`Slot::lock`].
430#[cfg(not(feature = "threadsafe"))]
431pub(crate) struct Slot<T>(std::cell::RefCell<T>);
432
433#[cfg(not(feature = "threadsafe"))]
434impl<T> Slot<T> {
435    pub(crate) fn new(value: T) -> Self {
436        Slot(std::cell::RefCell::new(value))
437    }
438    /// Borrow the contents mutably (an uncontended `RefCell` borrow).
439    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, T> {
440        self.0.borrow_mut()
441    }
442}
443
444#[cfg(feature = "threadsafe")]
445pub(crate) struct Slot<T>(std::sync::Mutex<T>);
446
447#[cfg(feature = "threadsafe")]
448impl<T> Slot<T> {
449    pub(crate) fn new(value: T) -> Self {
450        Slot(std::sync::Mutex::new(value))
451    }
452    /// Lock the contents. Different datasets hold different slots, so this
453    /// only contends when two threads write the *same* dataset.
454    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, T> {
455        self.0.lock().unwrap()
456    }
457}
458
459/// Proof that the create gate (`create_lock`) is held and the new dataset's
460/// name passed the uniqueness check. Only [`Hdf5Writer::begin_create`]
461/// constructs one and [`Hdf5Writer::push_dataset`] demands one, so a creator
462/// cannot reach the dataset registry while skipping either step. Carries
463/// the canonical (link-resolved) name the creator must store, so the
464/// registry only ever holds tree paths.
465pub(crate) struct CreateGuard<'a> {
466    #[cfg(not(feature = "threadsafe"))]
467    _gate: std::cell::RefMut<'a, ()>,
468    #[cfg(feature = "threadsafe")]
469    _gate: std::sync::MutexGuard<'a, ()>,
470    /// The dataset name with every group hard link in it resolved.
471    pub(crate) name: String,
472    /// The group that will hold the new dataset's link, resolved from the
473    /// path components of `name`; `None` is the root group. Carried here so
474    /// [`Hdf5Writer::push_dataset`] registers the child itself and no creator
475    /// can leave a dataset whose name says one thing and whose parent group
476    /// says another.
477    pub(crate) parent: Option<usize>,
478}
479
480/// Reference-counted shared pointer, feature-selected. The single-thread
481/// build uses `Rc` (no atomics); the `threadsafe` build uses `Arc` so a
482/// dataset/group slot can be cloned out of the registry and locked on its
483/// own — letting writes to *different* datasets proceed concurrently without
484/// holding the registry lock. See `docs/threadsafe-fine-grained-locking.md`
485/// (Stage 3).
486#[cfg(not(feature = "threadsafe"))]
487pub(crate) type Shared<T> = std::rc::Rc<T>;
488#[cfg(feature = "threadsafe")]
489pub(crate) type Shared<T> = std::sync::Arc<T>;
490
491/// One dataset's cell in the registry: its metadata slot plus the operation
492/// lock that serializes whole logical operations on it. Both live in one
493/// allocation so they cannot fall out of step — every dataset has its op
494/// lock by construction.
495pub(crate) struct DatasetCell {
496    /// Serializes one *whole* logical operation on this dataset.
497    ///
498    /// The metadata slot below serializes each individual acquisition, but a
499    /// multi-acquisition operation — take the append buffer → write chunks →
500    /// re-buffer the tail → extend, or flush-then-overwrite in a slice write
501    /// — would interleave with a concurrent same-dataset operation *between*
502    /// its acquisitions under `threadsafe`. Public write entries take this
503    /// lock and delegate to `_inner` variants; `_inner` variants and the
504    /// `pub(crate)` write helpers require the caller to hold it (or to hold
505    /// the writer exclusively via `&mut`, as close and the SWMR wrapper do).
506    ///
507    /// Not reentrant: the single-thread build's `RefCell` panics instantly
508    /// on a nested acquisition, so a missed entry/inner split fails loudly
509    /// in every test run rather than deadlocking only under `threadsafe`.
510    ///
511    /// Lock order: `create_lock → op → registry spine → metadata slot`. An
512    /// op lock is never held across another dataset's op lock, and no
513    /// op-lock holder takes `create_lock`, so the order is acyclic.
514    pub(crate) op: Slot<()>,
515    info: Slot<DatasetInfo>,
516}
517
518impl DatasetCell {
519    pub(crate) fn new(info: DatasetInfo) -> Self {
520        DatasetCell {
521            op: Slot::new(()),
522            info: Slot::new(info),
523        }
524    }
525
526    /// Borrow the metadata slot (a single acquisition; see [`Self::op`] for
527    /// whole-operation serialization).
528    #[cfg(not(feature = "threadsafe"))]
529    pub(crate) fn lock(&self) -> std::cell::RefMut<'_, DatasetInfo> {
530        self.info.lock()
531    }
532
533    /// Lock the metadata slot (a single acquisition; see [`Self::op`] for
534    /// whole-operation serialization).
535    #[cfg(feature = "threadsafe")]
536    pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, DatasetInfo> {
537        self.info.lock()
538    }
539}
540
541/// A single dataset's [`DatasetCell`], reference-counted so a writer can
542/// clone it out of the registry (releasing the registry lock) and then lock
543/// just this one dataset. Two threads writing different datasets take
544/// different `DatasetRef` locks and never contend; the same dataset's writes
545/// serialize, which is required because one chunk index is not concurrently
546/// mutable.
547pub(crate) type DatasetRef = Shared<DatasetCell>;
548
549/// A single group's metadata behind its own [`Slot`], reference-counted like
550/// [`DatasetRef`].
551pub(crate) type GroupRef = Shared<Slot<GroupInfo>>;
552
553/// Appended frames held back until they complete a chunk.
554///
555/// The buffer is the sole authority for rows `base .. base + frames`: the
556/// file's chunks do not hold them yet, and any operation that writes those
557/// rows must go through [`Hdf5Writer::flush_append_buffer`] first. `base` is
558/// recorded when the frames are buffered — never derived from the current
559/// extent, which an `extend_dataset` can move independently.
560pub struct AppendBuffer {
561    /// Absolute row of the first buffered frame.
562    pub base: u64,
563    /// Number of buffered frames.
564    pub frames: u64,
565    /// The frames' bytes, `frames` whole rows, row-major.
566    pub bytes: Vec<u8>,
567}
568
569/// One file a dataset's raw data lives in, as the writer holds it: the name
570/// the I/O path opens, together with the local-heap offset the External File
571/// List message stores that name as.
572///
573/// The two halves are one entry rather than two parallel lists because they
574/// describe one slot — the message encodes `name_offset`, and every read or
575/// write of the slot's bytes opens `name`; splitting them is what lets a
576/// rewrite pair a name with another slot's offset.
577#[derive(Debug, Clone, PartialEq, Eq)]
578pub struct ExternalFile {
579    /// The file name exactly as the heap stores it. Resolved against
580    /// `HDF5_EXTFILE_PREFIX` at I/O time, never here — the same rule the read
581    /// side follows.
582    pub name: String,
583    /// Where `name` sits in the local heap at [`ExternalStorage::heap_addr`].
584    pub name_offset: u64,
585    /// Byte offset within `name` where this slot's region begins.
586    pub offset: u64,
587    /// Bytes of the dataset's raw data this slot holds.
588    pub size: u64,
589}
590
591/// A dataset whose contiguous raw data lives outside this file — the External
592/// File List message (`H5O_EFL_ID`) and the local heap its names are in.
593///
594/// The data layout message of such a dataset still says `Contiguous`, with
595/// its address left undefined: it is this message's presence that makes
596/// libhdf5 route the dataset's I/O through `H5D_LOPS_EFL` (H5Dlayout.c).
597#[derive(Debug, Clone)]
598pub struct ExternalStorage {
599    /// Address of the local heap header holding every slot's name.
600    pub heap_addr: u64,
601    /// The files, in the order their regions concatenate into the dataset's
602    /// logical byte range.
603    pub files: Vec<ExternalFile>,
604    /// The prefix every one of those names is joined against, and the open
605    /// that settled it. Lives here rather than on [`DatasetInfo`] so a
606    /// dataset with no external storage cannot carry a prefix and a dataset
607    /// with external storage cannot lack one.
608    prefix: EfilePrefix,
609}
610
611/// The expanded external file prefix in force for one dataset, and the open
612/// that decided it — libhdf5's `dset->shared->extfile_prefix`.
613///
614/// `H5D__build_file_prefix` runs it once per open of the shared info, from
615/// the dapl of `H5D__create` (H5Dint.c:1318) or of the `H5D__open` that
616/// found no shared info yet (:1537), and both `H5D__efl_read` and
617/// `H5D__efl_write` then join against that one answer (H5Defl.c:315-317,
618/// :429-431). Measured under libhdf5 1.14.6 and 2.0.0: `H5Dcreate2` with a
619/// dapl naming a directory creates the raw data file there at `H5Dwrite`,
620/// and `HDF5_EXTFILE_PREFIX` shadows that property on the write path exactly
621/// as it does on the read path.
622#[derive(Debug, Clone, Default)]
623struct EfilePrefix {
624    /// The expansion itself; `None` is "no prefix", which leaves a stored
625    /// name to resolve against the process's current directory.
626    expanded: Option<PathBuf>,
627    /// The open that decided [`expanded`](Self::expanded). An expired handle
628    /// means no open is holding the answer any more, so the next one settles
629    /// it afresh — which is the state a dataset this session reopened starts
630    /// in, `H5Fopen` opening no dataset of its own.
631    open: std::sync::Weak<()>,
632}
633
634impl ExternalStorage {
635    /// The message this storage encodes to (`H5O_efl_t`).
636    fn message(&self) -> ExternalFileListMessage {
637        ExternalFileListMessage {
638            heap_addr: self.heap_addr,
639            slots: self
640                .files
641                .iter()
642                .map(
643                    |f| crate::format::messages::external_file_list::ExternalFileSlot {
644                        name_offset: f.name_offset,
645                        offset: f.offset,
646                        size: f.size,
647                    },
648                )
649                .collect(),
650        }
651    }
652
653    /// Bytes the slots reserve in total (`H5O_efl_total_size`), saturating
654    /// rather than wrapping so an overflowing list reads as "as large as it
655    /// gets" and passes any size check instead of failing one.
656    fn total_size(&self) -> u64 {
657        self.files
658            .iter()
659            .fold(0u64, |acc, f| acc.saturating_add(f.size))
660    }
661}
662
663/// A dataset whose elements are read out of other datasets — the virtual
664/// layout message (`H5D_VIRTUAL`) and the mapping list it points at.
665///
666/// The mappings live in one global heap object rather than in the header
667/// (`H5D__virtual_store_layout`), so the layout message carries only its
668/// address and index; the list itself is kept here so a rewrite of the header
669/// can re-emit the message pointing at the same object.
670#[derive(Debug, Clone, PartialEq, Eq)]
671pub struct VirtualStorage {
672    /// Address of the global heap collection holding the mapping list.
673    pub heap_addr: u64,
674    /// Index of the mapping-list object within that collection.
675    pub heap_index: u32,
676    /// The mappings themselves, in the order they were declared — which is
677    /// the order libhdf5 resolves overlapping ones in.
678    pub mappings: Vec<VirtualMapping>,
679}
680
681/// Where a contiguous dataset's raw bytes live, read off its registry entry
682/// so the write itself can run with the slot unlocked.
683///
684/// The one place the local-versus-external-versus-nowhere choice is made; see
685/// [`DatasetInfo::contiguous_target`].
686enum ContiguousTarget {
687    /// A block in this file, starting at this address.
688    Local(u64),
689    /// The files an External File List names, in dataset order, and the
690    /// prefix in force for the open doing the writing — carried together
691    /// because a slot name means nothing without it.
692    External {
693        files: Vec<ExternalFile>,
694        prefix: Option<PathBuf>,
695    },
696    /// Nowhere: the dataset is virtual, and every element of it is stored in
697    /// whichever source dataset its mappings send that element to.
698    Virtual,
699}
700
701/// What a writer-mode `H5Dataset` handle is built from — the shape and
702/// element width it answers questions with, the chunk index it writes
703/// through, and the open it holds.
704pub(crate) struct DatasetHandleParts {
705    pub(crate) shape: Vec<usize>,
706    pub(crate) element_size: usize,
707    /// `None` for storage that is not chunked.
708    pub(crate) chunk_index: Option<ChunkIndexKind>,
709    /// Keeps this open alive; see [`Hdf5Writer::bind_efile_prefix`].
710    pub(crate) open: Option<crate::io::reader::DatasetOpenToken>,
711}
712
713impl ContiguousTarget {
714    /// Whether this target is storage bytes can be written into at all —
715    /// false only for [`ContiguousTarget::Virtual`], which names sources
716    /// rather than storage.
717    fn is_storage(&self) -> bool {
718        !matches!(self, Self::Virtual)
719    }
720}
721
722/// The one refusal of a write into a virtual dataset, so the two paths that
723/// can reach one — [`Hdf5Writer::write_contiguous_bytes`] and the pre-insert
724/// gate of [`Hdf5Writer::write_vlen_strings_slice`] — say the same thing.
725///
726/// libhdf5 does take this write, pushing each element through the mapping
727/// that covers it into the source dataset holding it (`H5D__virtual_write`);
728/// this writer never opens a source file, so it refuses rather than dropping
729/// the bytes somewhere they cannot be read back from.
730/// The legality checks `H5Pset_virtual` runs over one mapping —
731/// `H5D_virtual_check_mapping_pre` and `H5D_virtual_check_mapping_post`
732/// (H5Dvirtual.c).
733///
734/// The two upstream checks that need the *source dataset's* own extent (the
735/// limited/limited element-count match, and a printf mapping's single-block
736/// match) are not run here for the same reason upstream skips them when the
737/// source space status is `H5O_VIRTUAL_STATUS_INVALID`: a mapping may name a
738/// source that does not exist yet, and nothing here opens one.
739fn check_virtual_mapping(dataset: &str, m: &VirtualMapping) -> IoResult<()> {
740    for (which, sel) in [
741        ("virtual", &m.virtual_selection),
742        ("source", &m.source_selection),
743    ] {
744        if matches!(sel, Selection::Points(_)) {
745            return Err(crate::io::IoError::Unsupported(format!(
746                "virtual dataset '{dataset}' has a point {which} selection, which \
747                 H5D_virtual_check_mapping_pre refuses for every virtual dataset mapping \
748                 (\"point selections not currently supported with virtual datasets\")"
749            )));
750        }
751    }
752
753    let unlim_virtual = m.virtual_selection.unlim_dim().is_some();
754    let unlim_source = m.source_selection.unlim_dim().is_some();
755
756    // Both sides unbounded: the mapping grows with its source, so the slices
757    // they exchange must be the same shape whatever either extent becomes.
758    if unlim_virtual && unlim_source {
759        if let (Some(v), Some(sr)) = (
760            regular_hyperslab(&m.virtual_selection),
761            regular_hyperslab(&m.source_selection),
762        ) {
763            let (nv, ns) = (v.num_elem_non_unlim(), sr.num_elem_non_unlim());
764            if nv != ns {
765                return Err(crate::io::IoError::InvalidState(format!(
766                    "virtual dataset '{dataset}' maps an unlimited source selection onto an \
767                     unlimited virtual selection, but a slice of the non-unlimited \
768                     dimensions holds {ns:?} source elements and {nv:?} virtual ones"
769                )));
770            }
771        }
772    }
773
774    // `H5D_virtual_check_mapping_post`: an unlimited virtual selection over a
775    // limited source selection is the printf shape, where each block of the
776    // virtual selection is filled by a *different* source dataset named by
777    // substituting that block's index. It needs a `%b` to name them, and a
778    // hyperslab virtual selection to have blocks at all; every other shape
779    // needs the opposite, since a substitution with only one block to fill
780    // has nothing to vary over.
781    let nsubs = parse_source_name(&m.source_file_name)
782        .and_then(|f| Ok(f.nsubs() + parse_source_name(&m.source_dset_name)?.nsubs()))
783        .map_err(|e| {
784            crate::io::IoError::InvalidState(format!(
785                "virtual dataset '{dataset}' source name: {e}"
786            ))
787        })?;
788    if unlim_virtual && !unlim_source {
789        if nsubs == 0 {
790            return Err(crate::io::IoError::InvalidState(format!(
791                "virtual dataset '{dataset}' has an unlimited virtual selection, a limited \
792                 source selection, and no printf specifiers in source names"
793            )));
794        }
795        if !matches!(m.virtual_selection, Selection::Hyperslab { .. }) {
796            return Err(crate::io::IoError::InvalidState(format!(
797                "virtual dataset '{dataset}' has a printf mapping whose virtual selection is \
798                 not a hyperslab; the substitution runs over the blocks of that hyperslab"
799            )));
800        }
801    } else if nsubs > 0 {
802        return Err(crate::io::IoError::InvalidState(format!(
803            "virtual dataset '{dataset}' has printf specifier(s) in source name(s) without \
804             an unlimited virtual selection and limited source selection"
805        )));
806    }
807    Ok(())
808}
809
810/// The regular (start, stride, count, block) form behind a selection, or
811/// `None` — the only form that can carry `H5S_UNLIMITED`, so every unlimited
812/// check goes through it.
813fn regular_hyperslab(sel: &Selection) -> Option<&crate::format::selection::RegularHyperslab> {
814    match sel {
815        Selection::Hyperslab {
816            form: crate::format::selection::Hyperslab::Regular(r),
817            ..
818        } => Some(r),
819        _ => None,
820    }
821}
822
823fn virtual_write_refused() -> crate::io::IoError {
824    crate::io::IoError::Unsupported(
825        "cannot write into a virtual dataset: its elements live in the source datasets \
826         its mappings name, and this writer does not write through to them — write the \
827         source datasets themselves"
828            .into(),
829    )
830}
831
832/// Metadata for a dataset being written.
833///
834/// The whole struct lives behind a per-dataset [`Slot`] (via [`DatasetRef`]).
835/// The streaming write path locks it only briefly — compression runs *outside*
836/// the lock — so writes to different datasets do not contend, and a structural
837/// op (create/delete) that scans names only momentarily touches a sibling
838/// slot.
839pub struct DatasetInfo {
840    /// Link name within the root group.
841    pub name: String,
842    /// Element datatype.
843    pub datatype: DatatypeMessage,
844    /// The committed datatype this dataset shares, when it was created from
845    /// one. The type itself stays in [`datatype`](Self::datatype) — the
846    /// dataspace, the element width and every payload check need it — and
847    /// this says the header must store a pointer to that object instead of a
848    /// datatype message of its own.
849    pub committed_type: Option<CommittedTypeRef>,
850    /// Dataspace (dimensionality).
851    pub dataspace: DataspaceMessage,
852    /// The object format the reopen found this dataset's messages written in,
853    /// `None` for a dataset this session created.
854    ///
855    /// A rewrite re-encodes the whole header — the shared-message table is
856    /// laid out whole, so every heap ID moves and every header naming one has
857    /// to be written again. Re-deriving the message format from the reopened
858    /// session's bounds would upgrade messages the file already has, which
859    /// libhdf5 never does: it grows a header in place and leaves every
860    /// message it did not touch alone. The same rule the reopen already
861    /// applies to a group it found in a symbol table
862    /// ([`uses_symbol_table`](Hdf5Writer::uses_symbol_table)) — what the file
863    /// says governs, not what this session's bound would have chosen.
864    pub read_format: Option<ObjectFormat>,
865    /// File offset of the dataset's object header (set during finalize).
866    pub obj_header_addr: u64,
867    /// File offset of the raw data block (contiguous only).
868    pub data_addr: u64,
869    /// Size of the raw data in bytes (contiguous only).
870    pub data_size: u64,
871    /// The raw data itself, for a compact dataset — the whole image, which
872    /// [`build_dataset_header`](Hdf5Writer::build_dataset_header) puts inside
873    /// the data layout message rather than in a block of its own. `Some` is
874    /// what makes a dataset compact, and the buffer is created at its final
875    /// length (filled, as `H5D__compact_fill` does, before any write), so it
876    /// is also the dataset's byte count; `data_addr`/`data_size` stay at the
877    /// "no block in the file" values a compact dataset shares with a NULL one.
878    pub compact: Option<Vec<u8>>,
879    /// The files this dataset's contiguous raw data lives in, when it lives
880    /// outside this HDF5 file. `Some` is what makes a contiguous dataset
881    /// externally stored: its `data_addr` stays [`UNDEF_ADDR`] and every byte
882    /// goes to the files named here instead of to a block of this file's own.
883    pub external: Option<ExternalStorage>,
884    /// The source datasets this dataset's elements are read from, when it is
885    /// virtual. `Some` is what makes it virtual, and it stores nothing of its
886    /// own: `data_addr`/`data_size` keep the "no block in this file" values a
887    /// compact dataset also has.
888    pub virtual_storage: Option<VirtualStorage>,
889    /// Chunked storage info (None for contiguous).
890    pub chunked: Option<ChunkedDatasetInfo>,
891    /// Fixed array chunked storage info.
892    pub fixed_array: Option<FixedArrayDatasetInfo>,
893    /// B-tree v2 chunked storage info.
894    pub btree_v2: Option<Bt2DatasetInfo>,
895    /// Implicit (no structure) chunked storage info.
896    pub implicit: Option<ImplicitDatasetInfo>,
897    /// Single-chunk chunked storage info: the whole (fixed) dataspace is
898    /// exactly one chunk.
899    pub single_chunk: Option<SingleChunkDatasetInfo>,
900    /// Version-1 B-tree chunked storage info — the classic-format index.
901    pub btree_v1: Option<BtreeV1DatasetInfo>,
902    /// Appended frames not yet written to chunks, `None` when empty.
903    pub append: Option<AppendBuffer>,
904    /// Attributes attached to this dataset.
905    pub attributes: Vec<AttributeEntry>,
906    /// File offset where the dataset object header was written (for SWMR in-place rewrites).
907    pub obj_header_written_addr: Option<u64>,
908    /// Encoded size of the dataset object header (for verifying in-place rewrites fit).
909    /// Every block the object's on-disk header occupies, chunk 0 first, or
910    /// empty when it has none yet. All of them are freed together: a rewrite
911    /// re-encodes the whole chain into one fresh chunk, so a continuation
912    /// block left behind is space no free-space manager records.
913    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
914    /// Filter pipeline for compressed chunks.
915    pub filter_pipeline: Option<FilterPipeline>,
916    /// Soft-deleted: excluded from finalize output.
917    pub deleted: bool,
918    /// The dataspace extent changed this session (`extend_dataset` /
919    /// `set_dataset_extent`). On a reopened dataset the finalize gate
920    /// otherwise infers "modified" from `chunks_written` alone, and a
921    /// session that only changed the extent would keep the old on-disk
922    /// header — silently dropping the new shape.
923    pub extent_dirty: bool,
924    /// Something the object header encodes changed this session without
925    /// touching the dataset's storage — an attribute set or removed, a fill
926    /// value defined. See [`header_stale`](DatasetInfo::header_stale).
927    pub header_dirty: bool,
928    /// The hard link count the on-disk header was written with, so finalize
929    /// can tell that this session changed it.
930    ///
931    /// A count, not a flag, because the count is what the header records and
932    /// the ways to change it are many: creating a link, unlinking one,
933    /// deleting a link's parent group, promoting a link to a primary name.
934    /// Comparing the value closes all of them at once, where a dirty flag
935    /// would have to be set at each and would be forgotten at the next one
936    /// added.
937    pub nlink_written: u32,
938    /// When the link naming this dataset was created; see
939    /// [`GroupInfo::creation_seq`].
940    pub creation_seq: u64,
941    /// How this dataset records creation order for its attributes — the
942    /// file's creation-order policy captured when the dataset was created,
943    /// the way libhdf5 captures the DCPL. A dataset holds no links, so only
944    /// the attribute half of [`TrackOrder`] applies to it.
945    pub track_attr_order: CreationOrder,
946    /// User-defined fill value bytes (exactly one element wide). `None`
947    /// means default zero-fill; `Some` is emitted as a `fill_defined = 2`
948    /// fill-value message in the dataset object header.
949    pub fill_value: Option<Vec<u8>>,
950    /// Fill value write time (`H5Pset_fill_time`'s `H5D_fill_time_t`, one of
951    /// [`FILL_TIME_ALLOC`], [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]),
952    /// emitted verbatim into the fill-value message's write-time field.
953    /// Defaults to `FILL_TIME_IFSET`, `H5D_CRT_FILL_TIME_DEF` — what a fresh
954    /// dataset creation property list carries until `set_dataset_fill_time`
955    /// says otherwise.
956    pub fill_time: u8,
957    /// Layout message version for chunked storage: 4, or 5 when the chunk
958    /// index encodes stored chunk sizes in a fixed `sizeof_size` field
959    /// (libhdf5 2.0). Chosen at create by `Hdf5Writer::chunk_layout_version`,
960    /// preserved from the file on reopen, and emitted verbatim at finalize.
961    /// Contiguous datasets ignore it.
962    pub layout_version: u8,
963    /// The times this object tracks: `Some` exactly when it was created with
964    /// `H5Pset_obj_track_times(true)`, `None` when it was not.
965    ///
966    /// One meaning on both header versions, which store them differently and
967    /// store different amounts of them: a version-2 header keeps all four in
968    /// its prefix, and a version-1 dataset keeps one, in an `H5O_MTIME_NEW`
969    /// message. [`touch_oh`] is the single place that turns this into either
970    /// of those, so the four fields are here whichever version the object
971    /// has, exactly as `H5O_t` carries `atime`/`mtime`/`ctime`/`btime` for a
972    /// version-1 header it never serialises them from.
973    pub times: Option<ObjectTimes>,
974}
975
976impl DatasetInfo {
977    /// Which chunk index this dataset uses, `None` for storage that is not
978    /// chunked — the one place the index-carrying fields are turned into an
979    /// answer.
980    ///
981    /// INVARIANT: a chunk index added to this struct is added here. A site
982    /// that spells the disjunction out itself is what classifies a new index
983    /// as contiguous storage, and contiguous storage is read and written at
984    /// [`data_addr`](Self::data_addr) — which a chunked dataset leaves
985    /// undefined, so the misclassification is a read or a write at
986    /// `UNDEF_ADDR` rather than an error.
987    pub(crate) fn chunk_index_kind(&self) -> Option<ChunkIndexKind> {
988        if self.chunked.is_some() {
989            Some(ChunkIndexKind::ExtensibleArray)
990        } else if self.fixed_array.is_some() {
991            Some(ChunkIndexKind::FixedArray)
992        } else if self.btree_v2.is_some() {
993            Some(ChunkIndexKind::BtreeV2)
994        } else if self.implicit.is_some() {
995            Some(ChunkIndexKind::Implicit)
996        } else if self.single_chunk.is_some() {
997            Some(ChunkIndexKind::SingleChunk)
998        } else if self.btree_v1.is_some() {
999            Some(ChunkIndexKind::BtreeV1)
1000        } else {
1001            None
1002        }
1003    }
1004
1005    /// Whether this dataset's raw data is stored in chunks — the question
1006    /// every storage-form test asks, asked in one place.
1007    pub(crate) fn is_chunked(&self) -> bool {
1008        self.chunk_index_kind().is_some()
1009    }
1010
1011    /// Where this dataset's contiguous raw bytes live, or `None` when it has
1012    /// no contiguous storage to write into at all — a chunked dataset, a
1013    /// compact one (whose bytes *are* the layout message), or one whose block
1014    /// was never allocated.
1015    ///
1016    /// INVARIANT: every write of a contiguous dataset's raw bytes picks its
1017    /// destination here and reaches it through
1018    /// [`Hdf5Writer::write_contiguous_bytes`]. A site that read `data_addr`
1019    /// itself would write an externally-stored dataset's data into this file
1020    /// — at [`UNDEF_ADDR`], the far end of the address space — instead of into
1021    /// the files its header names, and would do the same to a virtual one,
1022    /// whose bytes are not this file's to write at all.
1023    ///
1024    /// Chunked storage is excluded through
1025    /// [`chunk_index_kind`](Self::chunk_index_kind) rather than by naming the
1026    /// index-carrying fields, so an index added to this struct cannot arrive
1027    /// here as contiguous storage: an implicit-indexed dataset reads
1028    /// `data_addr` as the base of its chunk grid, which as a contiguous
1029    /// destination would take a raw write meant for one chunk and lay it over
1030    /// the whole grid.
1031    fn contiguous_target(&self) -> Option<ContiguousTarget> {
1032        if self.is_chunked() || self.compact.is_some() {
1033            return None;
1034        }
1035        if self.virtual_storage.is_some() {
1036            return Some(ContiguousTarget::Virtual);
1037        }
1038        match &self.external {
1039            Some(ext) => Some(ContiguousTarget::External {
1040                files: ext.files.clone(),
1041                prefix: ext.prefix.expanded.clone(),
1042            }),
1043            None => {
1044                (self.data_addr != UNDEF_ADDR).then_some(ContiguousTarget::Local(self.data_addr))
1045            }
1046        }
1047    }
1048
1049    /// The one run of file bytes an implicitly indexed dataset's chunk grid
1050    /// is — its start and its length — or `None` when the dataset is indexed
1051    /// some other way or its space is not allocated yet.
1052    ///
1053    /// That index has no per-chunk structure to hold an address in: every
1054    /// chunk sits at `data_addr + linear_index * chunk_bytes` and the grid is
1055    /// allocated whole at create (`H5D__none_idx_get_addr`, H5Dnone.c). So the
1056    /// run is file space this writer allocated, and it is the *only* storage a
1057    /// chunk of such a dataset can occupy — the builder refuses external and
1058    /// virtual storage together with chunked storage, which is why
1059    /// [`allocated_storage_run`](Self::allocated_storage_run) can name it
1060    /// [`ContiguousTarget::Local`] and no chunk write can reach the other two.
1061    fn implicit_grid(&self) -> Option<(u64, u64)> {
1062        let imp = self.implicit.as_ref()?;
1063        (imp.data_addr != UNDEF_ADDR).then_some((imp.data_addr, imp.data_size))
1064    }
1065
1066    /// The run of raw storage this writer *allocated* for the dataset — the
1067    /// target to initialise it through and its size — or `None` when it
1068    /// allocated none.
1069    ///
1070    /// The two storage forms that are one run of bytes: a contiguous
1071    /// dataset's data block, and an implicitly indexed dataset's chunk grid.
1072    /// A compact dataset is excluded (its bytes are its layout message) and so
1073    /// is every other chunk index, whose chunks are placed one at a time.
1074    ///
1075    /// External storage is excluded because this writer does not allocate it:
1076    /// `H5D__alloc_storage` skips its whole body — the space reservation and
1077    /// the `H5D__init_storage` that would tile the fill value into it — for a
1078    /// dataset with an external file list or an empty extent, "we assume that
1079    /// external storage is already allocated by the caller, or at least will
1080    /// be before I/O is performed" (H5Dint.c:2270-2274). Measured under
1081    /// libhdf5 1.14.6 and 2.0.0: a user fill value, `H5D_FILL_TIME_ALLOC` and
1082    /// `H5D_ALLOC_TIME_EARLY` together leave the raw data file uncreated at
1083    /// `H5Dcreate2`, and a read before any write fails with "unable to open
1084    /// external raw data file" rather than reporting the fill.
1085    ///
1086    /// INVARIANT: only storage whose bytes this file owns is initialised as
1087    /// one run, so the allocate-time fill cannot reach the files an external
1088    /// file list names or the sources a virtual dataset maps.
1089    fn allocated_storage_run(&self) -> Option<(ContiguousTarget, u64)> {
1090        match self.implicit_grid() {
1091            Some((addr, size)) => Some((ContiguousTarget::Local(addr), size)),
1092            // Not a fallthrough for an unallocated implicit grid:
1093            // `contiguous_target` answers `None` for every chunked dataset.
1094            None => match self.contiguous_target() {
1095                Some(t @ ContiguousTarget::Local(_)) => Some((t, self.data_size)),
1096                _ => None,
1097            },
1098        }
1099    }
1100
1101    /// Whether this session wrote chunk data or changed the extent, so the
1102    /// dataset's index structures have to be re-flushed.
1103    fn storage_dirty(&self) -> bool {
1104        self.chunked.as_ref().is_some_and(|c| c.chunks_written > 0)
1105            || self
1106                .fixed_array
1107                .as_ref()
1108                .is_some_and(|f| f.chunks_written > 0)
1109            || self.btree_v2.as_ref().is_some_and(|b| b.chunks_written > 0)
1110            || self.btree_v1.as_ref().is_some_and(|b| b.chunks_written > 0)
1111            || self
1112                .single_chunk
1113                .as_ref()
1114                .is_some_and(|s| s.chunks_written > 0)
1115            || self.extent_dirty
1116    }
1117
1118    /// Whether a reopened dataset's on-disk object header no longer describes
1119    /// it.
1120    ///
1121    /// INVARIANT: every mutation of something `build_dataset_header` encodes
1122    /// must show up here. Finalize keeps the original header when this is
1123    /// false, so a change this misses is not deferred — it is discarded, with
1124    /// no error to say so. Attributes were the case that proved it: they are
1125    /// invisible to the chunk-write counters, so an attribute set on a
1126    /// reopened dataset vanished at close.
1127    fn header_stale(&self) -> bool {
1128        self.storage_dirty() || self.header_dirty
1129    }
1130
1131    /// The same question for the one thing the dataset itself cannot see: how
1132    /// many hard links resolve to it. That count lives in the header — an
1133    /// Object Reference Count message in a version-2 header, the `nlink`
1134    /// prefix field of a version-1 one — but it is a property of the file's
1135    /// link graph, so the caller supplies today's value.
1136    fn header_stale_with(&self, nlink: u32) -> bool {
1137        self.header_stale() || nlink != self.nlink_written
1138    }
1139
1140    /// Record that this dataset's on-disk object header was just written with
1141    /// `nlink` in it.
1142    ///
1143    /// INVARIANT: every write of a dataset object header passes through here.
1144    /// [`header_stale_with`](Self::header_stale_with) is the one authority for
1145    /// "does what is on disk still describe this dataset?", and it answers by
1146    /// comparing against [`nlink_written`](Self::nlink_written) — so a site
1147    /// that writes a header without saying so leaves that answer describing an
1148    /// older write. There are three writers: `finalize`, `finalize_for_swmr`
1149    /// and `write_dataset_header_inplace`. The last recorded nothing; it could
1150    /// not drift today only because a count it could write is a count that
1151    /// makes the header outgrow its block, which it refuses. That is a
1152    /// property of the reference-count message's size, not a rule anything
1153    /// states, and it is not what the field's definition rests on.
1154    fn header_written(&mut self, nlink: u32) {
1155        self.nlink_written = nlink;
1156    }
1157}
1158
1159/// Runtime metadata for a chunked dataset.
1160pub struct ChunkedDatasetInfo {
1161    /// Chunk dimension sizes.
1162    pub chunk_dims: Vec<u64>,
1163    /// Extensible array parameters.
1164    pub earray_params: EarrayParams,
1165    /// File offset of the EA header.
1166    pub ea_header_addr: u64,
1167    /// File offset of the EA index block.
1168    pub ea_iblk_addr: u64,
1169    /// In-memory copy of the EA header (for updating statistics).
1170    pub ea_header: ExtensibleArrayHeader,
1171    /// In-memory copy of the EA index block (for unfiltered datasets).
1172    pub ea_iblk: ExtensibleArrayIndexBlock,
1173    /// Number of chunks written so far.
1174    pub chunks_written: u64,
1175    /// Filtered index block (for compressed datasets).
1176    pub filt_iblk: Option<FilteredIndexBlock>,
1177    /// chunk_size_len for filtered entries.
1178    pub chunk_size_len: u8,
1179}
1180
1181/// Where a newly-created EA data block's address must be recorded.
1182enum DblkParent {
1183    /// Slot `index_block.dblk_addrs[idx]`.
1184    IndexBlock(usize),
1185    /// Slot `super_block.dblk_addrs[local_dblk]` of the super block at `sblk_addr`.
1186    SuperBlock {
1187        sblk_addr: u64,
1188        ndblks_in_sblk: usize,
1189        local_dblk: usize,
1190    },
1191}
1192
1193/// Which attribute list an attribute operation targets: the root group's,
1194/// a group's (by full path), or a dataset's (by writer index).
1195#[derive(Clone, Copy)]
1196pub enum AttrTarget<'a> {
1197    /// The root group's (file-level) attributes.
1198    Root,
1199    /// A group's attributes, by full path.
1200    Group(&'a str),
1201    /// A dataset's attributes, by writer index.
1202    Dataset(usize),
1203}
1204
1205/// Which chunk index a dataset uses.
1206///
1207/// The five above the line are what `H5D__layout_set_latest_indexing`
1208/// (H5Dlayout.c) picks between once the file format allows a version-4 data
1209/// layout message, in this precedence: a v2 B-tree for two or more unlimited
1210/// dimensions, an extensible array for exactly one, and — for a fixed shape —
1211/// the single-chunk index whenever exactly one chunk covers the whole
1212/// dataspace (`dims == max_dims == chunk_dims`, checked before either
1213/// alternative below and taken regardless of filter or allocation-time), else
1214/// the implicit index when nothing has to be recorded per chunk (no filter,
1215/// early allocation), else a fixed array. [`BtreeV1`](Self::BtreeV1) is not
1216/// one of them: it belongs to the version-3 layout message, and a file whose
1217/// superblock is older than version 2 can carry no other.
1218#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1219pub(crate) enum ChunkIndexKind {
1220    ExtensibleArray,
1221    FixedArray,
1222    BtreeV2,
1223    Implicit,
1224    SingleChunk,
1225    BtreeV1,
1226}
1227
1228/// A chunked dataset's grid geometry, snapshotted out of its slot.
1229///
1230/// The single owner of chunk-grid arithmetic: how many chunks span each
1231/// dimension, where a coordinate sits in the row-major order the array
1232/// indices record, and how many bytes one chunk holds.
1233struct ChunkGeometry {
1234    kind: ChunkIndexKind,
1235    dims: Vec<u64>,
1236    max_dims: Option<Vec<u64>>,
1237    chunk_dims: Vec<u64>,
1238    element_size: u64,
1239}
1240
1241impl ChunkGeometry {
1242    /// Unfiltered byte size of one whole chunk.
1243    fn chunk_bytes(&self) -> u64 {
1244        self.chunk_dims.iter().product::<u64>() * self.element_size
1245    }
1246
1247    /// Row-major position of `coords` in the chunk grid — the linear index an
1248    /// extensible or fixed array records the chunk under, computed against
1249    /// the maximum-extent grid by [`crate::io::chunk_grid::linear_index`].
1250    fn linear_index(&self, coords: &[u64]) -> IoResult<u64> {
1251        crate::io::chunk_grid::linear_index(
1252            &self.dims,
1253            self.max_dims.as_deref(),
1254            &self.chunk_dims,
1255            coords,
1256        )
1257    }
1258}
1259
1260/// The refusal every attribute mutation gets while SWMR streaming is
1261/// active, from the two owners of attribute-list change
1262/// ([`Hdf5Writer::set_attribute`] and `evict_attr`).
1263fn swmr_attr_error(name: &str) -> crate::io::IoError {
1264    crate::io::IoError::InvalidState(format!(
1265        "cannot add or modify attribute '{name}' during SWMR streaming: object \
1266         headers are frozen while readers stream, and a superseded variable-length \
1267         value's heap storage could never be reclaimed; set attributes before \
1268         start_swmr (libhdf5 forbids attribute changes during SWMR writes too)"
1269    ))
1270}
1271
1272/// Where an attribute arriving at [`Hdf5Writer::insert_attribute`] came from.
1273///
1274/// The variable-length setters have to evict before they allocate — the
1275/// free-before-alloc order — so by the time the replacement is inserted the
1276/// list no longer holds the entry it replaces, and the ordinary "already
1277/// present, so keep its index" test cannot see it. `H5A__attr_write` does not
1278/// create the attribute again, so the index travels with the eviction rather
1279/// than being stamped afresh; without it a rewritten attribute takes the set's
1280/// running maximum and moves to the end of the creation order.
1281#[derive(Debug, Clone, Copy)]
1282enum AttrOrigin {
1283    /// A new attribute, which takes the set's next creation index.
1284    Created,
1285    /// A value written over an attribute this writer has just evicted, which
1286    /// keeps that attribute's creation index — `None` when the object tracks
1287    /// no order, and so records none. An eviction that found nothing to remove
1288    /// answers `Created`: what follows it is a create like any other.
1289    Rewritten(Option<u16>),
1290}
1291use AttrOrigin::{Created, Rewritten};
1292
1293/// Take an object's attributes into the append session, or refuse the reopen.
1294///
1295/// Append mode rebuilds every object header it touches out of the attributes
1296/// read from it, so what this returns is what the object will still have when
1297/// the session finalizes. An attribute set that could not be read whole —
1298/// `ObjectAttributes::into_complete` refuses it — would come back as the part
1299/// that did read, silently deleting the rest.
1300///
1301/// Left to surface at `finalize`, that failure would land after this session's
1302/// chunk data and indices had already been written past the allocation point
1303/// the superblock still records, leaving a file libhdf5 reads as truncated.
1304/// Refusing the open leaves it untouched.
1305///
1306/// Size is no longer a reason to refuse: an attribute too large for a header
1307/// message goes back out through dense storage, the form libhdf5 read it from.
1308///
1309/// The set comes back in creation-index order, which is the order the registry
1310/// holds attributes in for an object made in this session too. A dense set is
1311/// read through the name index, so the order it arrives in is the order a hash
1312/// walk took; sorting here is what makes "the list is in creation order" true
1313/// of a reopened object as well, without any later stage having to know which
1314/// storage form the attributes came out of. Attributes of an untracked object
1315/// carry no index and keep the order they were read in.
1316fn take_reopened_attributes(
1317    attrs: crate::io::reader::ObjectAttributes,
1318    owner: &str,
1319) -> IoResult<Vec<AttributeEntry>> {
1320    let mut attrs = attrs.into_complete(owner)?;
1321    attrs.sort_by_key(|a| a.creation_index());
1322    Ok(attrs)
1323}
1324
1325/// The creation-order policy an on-disk object header declares — the single
1326/// owner of the recovery rule, used for the root group, every reopened group
1327/// and (through its attribute half) every reopened dataset.
1328///
1329/// The two halves come from two different places, and reading one for both is
1330/// how a file that sets only one of them came back with both or neither:
1331///
1332///   * links — the `Link Info` message's flag bits, which is what
1333///     `H5Pget_link_creation_order` reads (`H5G__get_create_plist`). A group
1334///     with no such message (or one this crate cannot decode) tracks nothing;
1335///     so does every dataset, which has no links to order.
1336///   * attributes — the object header's own flag bits, which is what
1337///     `H5Pget_attr_creation_order` reads (`H5Pocpl.c`). The `Attribute Info`
1338///     message carries the same two bits, but the header is the authority
1339///     libhdf5 consults, and it is present even when the object has no
1340///     attributes yet.
1341fn recover_track_order(
1342    header: &crate::format::object_header::ObjectHeader,
1343    ctx: &FormatContext,
1344) -> TrackOrder {
1345    let links = header
1346        .messages
1347        .iter()
1348        .find(|m| m.msg_type == crate::format::messages::MSG_LINK_INFO)
1349        .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
1350        .map(|(info, _)| info.creation_order())
1351        .unwrap_or_default();
1352    TrackOrder {
1353        links,
1354        attrs: header.attribute_creation_order(),
1355    }
1356}
1357
1358/// `H5O_touch_oh` (H5Oint.c:1273): put an object's tracked times where its
1359/// header version keeps them.
1360///
1361/// INVARIANT: every object header this writer builds passes its times through
1362/// here. The version decides the storage and nothing else does — a caller that
1363/// set `ObjectHeader::times` itself would hand a version-1 encode a prefix
1364/// field that version has no room for, and one that added the message itself
1365/// would put a second copy in a version-2 header.
1366///
1367/// `force` is upstream's own parameter, and it is what splits datasets from
1368/// everything else: it creates the version-1 `H5O_MTIME_NEW` message when the
1369/// header has none, and only `H5D__update_oh_info` passes it true
1370/// (H5Dint.c:1022-1026). Every other caller passes false and so creates no
1371/// message at all, which is why a version-1 group or committed datatype
1372/// records no time even when it is tracking them. A version-2 header keeps all
1373/// four times in its prefix whatever `force` says.
1374fn touch_oh(
1375    header: &mut ObjectHeader,
1376    format: ObjectFormat,
1377    times: Option<ObjectTimes>,
1378    force: bool,
1379) {
1380    let Some(times) = touched_times(times) else {
1381        return;
1382    };
1383    match format {
1384        ObjectFormat::Modern => header.times = Some(times),
1385        ObjectFormat::Legacy if force => header.add_message(
1386            crate::format::messages::MSG_MOD_TIME,
1387            0x00,
1388            ModificationTime(times.change).encode(),
1389        ),
1390        ObjectFormat::Legacy => {}
1391    }
1392}
1393
1394/// The times a header being (re)written carries, given what the object had.
1395///
1396/// Every object header this writer emits is one it is writing *now*, which is
1397/// what `H5O_touch_oh` is called for: an object that stores times gets its
1398/// access and change time moved to now, and one that does not store them stays
1399/// that way — the flag belongs to the object's creation property list, and a
1400/// rewrite is not a creation.
1401fn touched_times(times: Option<ObjectTimes>) -> Option<ObjectTimes> {
1402    times.map(|t| t.touched(now_seconds()))
1403}
1404
1405/// Seconds since the epoch, as an object header stores them (`H5_now`).
1406///
1407/// Saturates rather than wrapping: the field is a 32-bit count, and a clock
1408/// past 2106 is better reported as the largest time the format can express
1409/// than as a time in 1970. A clock before the epoch yields 0, which is what
1410/// libhdf5 writes for "no time recorded".
1411fn now_seconds() -> u32 {
1412    std::time::SystemTime::now()
1413        .duration_since(std::time::UNIX_EPOCH)
1414        .map_or(0, |d| u32::try_from(d.as_secs()).unwrap_or(u32::MAX))
1415}
1416
1417/// The dense storage an on-disk object header names: the fractal heap and the
1418/// indices its `Attribute Info` and `Link Info` messages point at.
1419///
1420/// A rewrite of that header lays fresh storage out and stops naming this, so
1421/// what this returns is exactly what the rewrite supersedes and must free.
1422/// Compact storage names no heap and yields `None` — there is nothing to free
1423/// and nothing that could be freed twice.
1424fn superseded_dense(
1425    header: &crate::format::object_header::ObjectHeader,
1426    ctx: &FormatContext,
1427) -> (Option<AttributeInfoMessage>, Option<LinkInfoMessage>) {
1428    let decode = |msg_type: u8| {
1429        header
1430            .messages
1431            .iter()
1432            .find(|m| m.msg_type == msg_type)
1433            .map(|m| m.data.as_slice())
1434    };
1435    let attrs = decode(crate::format::messages::MSG_ATTR_INFO)
1436        .and_then(|d| AttributeInfoMessage::decode(d, ctx).ok())
1437        .map(|(info, _)| info)
1438        .filter(|info| info.is_dense());
1439    let links = decode(crate::format::messages::MSG_LINK_INFO)
1440        .and_then(|d| LinkInfoMessage::decode(d, ctx).ok())
1441        .map(|(info, _)| info)
1442        .filter(|info| info.is_dense());
1443    (attrs, links)
1444}
1445
1446/// One collection block with free space that a later vlen insert may
1447/// fill — an entry in the writer's CWFS list (libhdf5 `f->shared->cwfs`).
1448struct CwfsEntry {
1449    /// Block address of the collection.
1450    addr: u64,
1451    /// Declared block size; never changes after allocation.
1452    size: usize,
1453    /// Bytes its free-space marker owns, per
1454    /// [`GlobalHeapCollection::free_space_at`](crate::format::global_heap::GlobalHeapCollection::free_space_at).
1455    free: usize,
1456}
1457
1458/// Maximum CWFS entries tracked — libhdf5's `H5HG_NCWFS` (H5HGpkg.h).
1459const H5HG_NCWFS: usize = 16;
1460
1461/// Record a collection with `free` bytes in the CWFS list: update its
1462/// entry if present, append while the list is short, and otherwise
1463/// replace the entry with the least free space when this one has more —
1464/// the retention rule of libhdf5's `H5HG_insert`.
1465fn cwfs_note(cwfs: &mut Vec<CwfsEntry>, addr: u64, size: usize, free: usize) {
1466    if let Some(p) = cwfs.iter().position(|e| e.addr == addr) {
1467        cwfs[p].free = free;
1468        return;
1469    }
1470    if cwfs.len() < H5HG_NCWFS {
1471        cwfs.insert(0, CwfsEntry { addr, size, free });
1472        return;
1473    }
1474    if let Some(p) = (0..cwfs.len()).min_by_key(|&p| cwfs[p].free) {
1475        if free > cwfs[p].free {
1476            cwfs[p] = CwfsEntry { addr, size, free };
1477        }
1478    }
1479}
1480
1481/// The uniform rejection for `delete_dataset` / `delete_group` while SWMR
1482/// streaming is active: deleting frees the object's blocks, and a live
1483/// reader may hold any of their addresses.
1484fn swmr_delete_error(name: &str) -> crate::io::IoError {
1485    crate::io::IoError::InvalidState(format!(
1486        "cannot delete '{name}' during SWMR streaming: a reader may hold the \
1487         object's header and storage addresses (libhdf5 forbids link deletion \
1488         during SWMR writes too)"
1489    ))
1490}
1491
1492/// Whether the chunk at grid `coords` lies entirely at or beyond `extent` in
1493/// some dimension — no element of it would survive a shrink to that extent.
1494fn chunk_outside_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1495    coords
1496        .iter()
1497        .zip(chunk_dims)
1498        .zip(extent)
1499        .any(|((&c, &cd), &e)| c.saturating_mul(cd) >= e)
1500}
1501
1502/// Whether the chunk at grid `coords` keeps elements under `extent` but
1503/// extends past it in some dimension — a shrink must refill its
1504/// out-of-extent region with the fill value.
1505fn chunk_straddles_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1506    !chunk_outside_extent(coords, chunk_dims, extent)
1507        && coords
1508            .iter()
1509            .zip(chunk_dims)
1510            .zip(extent)
1511            .any(|((&c, &cd), &e)| (c + 1).saturating_mul(cd) > e)
1512}
1513
1514/// Overwrite, in `data` (one whole chunk, unfiltered, row-major), every
1515/// element at or beyond `extent` with the matching bytes of `fill` — a
1516/// same-sized buffer tiled with the fill value. The caller guarantees the
1517/// chunk at `coords` straddles `extent`, so every dimension keeps at least
1518/// one element. Returns the replaced bytes, so a vlen dataset's dead
1519/// heap references can be released rather than stranded.
1520fn refill_chunk_beyond_extent(
1521    data: &mut [u8],
1522    fill: &[u8],
1523    coords: &[u64],
1524    chunk_dims: &[u64],
1525    extent: &[u64],
1526    element_size: usize,
1527) -> Vec<u8> {
1528    let ndims = chunk_dims.len();
1529    let keep: Vec<usize> = (0..ndims)
1530        .map(|d| {
1531            let origin = coords[d] * chunk_dims[d];
1532            chunk_dims[d].min(extent[d].saturating_sub(origin)) as usize
1533        })
1534        .collect();
1535    // Row-major walk: for every row (all dimensions but the last),
1536    // overwrite the whole row when its prefix is outside the keep box,
1537    // else only the row's out-of-extent tail.
1538    let row_elems = chunk_dims[ndims - 1] as usize;
1539    let keep_last = keep[ndims - 1];
1540    let nrows: u64 = chunk_dims[..ndims - 1].iter().product();
1541    let mut replaced = Vec::new();
1542    for r in 0..nrows {
1543        let mut rem = r;
1544        let mut in_keep = true;
1545        for d in (0..ndims - 1).rev() {
1546            let c = rem % chunk_dims[d];
1547            rem /= chunk_dims[d];
1548            if c as usize >= keep[d] {
1549                in_keep = false;
1550            }
1551        }
1552        let start = if in_keep { keep_last } else { 0 };
1553        if start == row_elems {
1554            continue;
1555        }
1556        let a = (r as usize * row_elems + start) * element_size;
1557        let b = (r as usize + 1) * row_elems * element_size;
1558        replaced.extend_from_slice(&data[a..b]);
1559        data[a..b].copy_from_slice(&fill[a..b]);
1560    }
1561    replaced
1562}
1563
1564/// Validate caller-supplied chunk geometry at dataset definition, the rule
1565/// libhdf5 applies in `H5D__chunk_construct` (H5Dchunk.c): the chunk rank
1566/// must match the dataspace rank, no chunk dimension may be zero, and a
1567/// chunk dimension may not exceed a fixed maximum dimension — except in a
1568/// dimension whose current size is zero, which libhdf5 exempts.
1569fn validate_chunk_geometry(dims: &[u64], max_dims: &[u64], chunk_dims: &[u64]) -> IoResult<()> {
1570    let ndims = dims.len();
1571    if chunk_dims.len() != ndims {
1572        return Err(crate::io::IoError::InvalidState(format!(
1573            "chunk shape has {} dimensions but the dataspace has {}",
1574            chunk_dims.len(),
1575            ndims
1576        )));
1577    }
1578    if max_dims.len() != ndims {
1579        return Err(crate::io::IoError::InvalidState(format!(
1580            "maximum shape has {} dimensions but the dataspace has {}",
1581            max_dims.len(),
1582            ndims
1583        )));
1584    }
1585    for d in 0..ndims {
1586        if chunk_dims[d] == 0 {
1587            return Err(crate::io::IoError::InvalidState(format!(
1588                "chunk dimension {d} is zero"
1589            )));
1590        }
1591        if dims[d] != 0 && max_dims[d] != u64::MAX && max_dims[d] < chunk_dims[d] {
1592            return Err(crate::io::IoError::InvalidState(format!(
1593                "chunk dimension {} is {} but the maximum dimension size is {}",
1594                d, chunk_dims[d], max_dims[d]
1595            )));
1596        }
1597    }
1598    Ok(())
1599}
1600
1601/// An extensible-array index requires at most one unlimited dimension —
1602/// `H5D__chunk_construct` (H5Dchunk.c) only selects this index for exactly
1603/// one — at any position: `chunk_grid::linear_index` seeds the unlimited
1604/// dimension into the slot no down-chunks multiplier touches, the same
1605/// address libhdf5 reaches by swizzling it to the slowest position
1606/// (`H5VM_swizzle_coords`, H5Dearray.c). Two or more unlimited dimensions
1607/// have no finite grid at all; that shape needs a v2 B-tree index instead.
1608fn ensure_at_most_one_unlimited(max_dims: &[u64]) -> IoResult<()> {
1609    let unlimited: Vec<usize> = max_dims
1610        .iter()
1611        .enumerate()
1612        .filter(|&(_, &m)| m == u64::MAX)
1613        .map(|(d, _)| d)
1614        .collect();
1615    if unlimited.len() > 1 {
1616        return Err(crate::io::IoError::InvalidState(format!(
1617            "an extensible-array index supports at most one unlimited dimension, \
1618             but dimensions {unlimited:?} are all unlimited; a v2 B-tree index \
1619             handles two or more"
1620        )));
1621    }
1622    Ok(())
1623}
1624
1625/// Reject strings the dataset's declared character set cannot label.
1626///
1627/// A Rust `&str` is always UTF-8, so only an ASCII declaration (charset 0)
1628/// can be violated. libhdf5 stores the bytes unvalidated — its vlen write
1629/// path has no cset check anywhere — which mislabels them for every reader
1630/// that trusts the declaration (h5py raises on the same mismatch).
1631fn ensure_vlen_charset(charset: u8, strings: &[&str]) -> IoResult<()> {
1632    if charset == 0 {
1633        if let Some((i, s)) = strings.iter().enumerate().find(|(_, s)| !s.is_ascii()) {
1634            return Err(crate::io::IoError::InvalidState(format!(
1635                "string {i} ({s:?}) is not ASCII, but the dataset's character set is"
1636            )));
1637        }
1638    }
1639    Ok(())
1640}
1641
1642/// Runtime metadata for a fixed-array-indexed chunked dataset.
1643pub struct FixedArrayDatasetInfo {
1644    /// Chunk dimension sizes.
1645    pub chunk_dims: Vec<u64>,
1646    /// File offset of the FA header.
1647    pub fa_header_addr: u64,
1648    /// File offset of the FA data block.
1649    pub fa_dblk_addr: u64,
1650    /// In-memory copy of the FA header.
1651    pub fa_header: FixedArrayHeader,
1652    /// In-memory copy of the FA data block.
1653    pub fa_dblk: FixedArrayDataBlock,
1654    /// Number of chunks written so far.
1655    pub chunks_written: u64,
1656}
1657
1658/// Runtime metadata for an implicitly indexed chunked dataset — the index
1659/// that is no structure at all (`H5Dnone.c`).
1660///
1661/// Every chunk of the maximum-extent grid is allocated at create in one
1662/// contiguous run, in the row-major order [`crate::io::chunk_grid`] defines,
1663/// so a chunk's address is `data_addr + linear_index * chunk_bytes` and
1664/// nothing has to be recorded when one is written. libhdf5 picks this index
1665/// only when that arithmetic is total: no filter (every chunk is exactly
1666/// `chunk_bytes` long), no unlimited dimension (the run has a finite length),
1667/// and early allocation (the run exists before any write).
1668pub struct ImplicitDatasetInfo {
1669    /// Chunk dimension sizes.
1670    pub chunk_dims: Vec<u64>,
1671    /// File offset of the first chunk — the layout message's index address.
1672    pub data_addr: u64,
1673    /// Byte length of the whole chunk run: `nchunks * chunk_bytes`.
1674    pub data_size: u64,
1675}
1676
1677/// Runtime metadata for a single-chunk indexed dataset (`H5Dsingle.c`): a
1678/// fixed dataspace exactly one chunk wide in every dimension
1679/// (`dims == max_dims == chunk_dims`), so there is exactly one chunk and its
1680/// address — and, when filtered, its stored size and filter mask — are held
1681/// directly in the layout message rather than in any index structure.
1682///
1683/// libhdf5 selects this index ahead of the implicit and fixed-array indexes
1684/// whenever the shape qualifies, whether or not the dataset is filtered or
1685/// early-allocated (`H5D__layout_set_latest_indexing`, H5Dlayout.c).
1686pub struct SingleChunkDatasetInfo {
1687    /// Chunk dimension sizes (equal to the dataspace's `dims`).
1688    pub chunk_dims: Vec<u64>,
1689    /// File offset of the chunk, [`UNDEF_ADDR`] until the chunk is written
1690    /// (or immediately, for an unfiltered dataset created with early
1691    /// allocation).
1692    pub data_addr: u64,
1693    /// The chunk's full unfiltered byte length — `chunk_dims.product() *
1694    /// element_size`, fixed for the dataset's lifetime.
1695    pub data_size: u64,
1696    /// Stored (on-disk) byte length: equal to `data_size` when the dataset
1697    /// carries no filter pipeline; the filtered length once the chunk has
1698    /// been written, 0 before then.
1699    pub nbytes: u64,
1700    /// Filter mask recorded for the stored chunk (bit *i* set means filter
1701    /// *i* was skipped); meaningful only when the dataset is filtered.
1702    pub filter_mask: u32,
1703    /// Chunks written this session (0 or 1) — `storage_dirty`'s signal that
1704    /// the layout message's address/size/mask fields must be re-flushed.
1705    pub chunks_written: u64,
1706    /// Whether this dataset was created with early allocation
1707    /// (`H5D_ALLOC_TIME_EARLY`) — distinct from `data_addr` being defined,
1708    /// which also becomes true the moment an incrementally allocated
1709    /// dataset's one chunk is written; `build_dataset_header` needs this to
1710    /// tell the two apart when it reports the fill-value message's
1711    /// allocation time. Only ever set for an unfiltered dataset: a filtered
1712    /// chunk's stored length is not known until it is compressed, so there
1713    /// is nothing to allocate ahead of that write regardless of alloc time
1714    /// (the same gap `create_fixed_array_dataset_with_pipeline` has).
1715    pub early_alloc: bool,
1716}
1717
1718/// One chunk as the version-1 B-tree records it — the key libhdf5 stores
1719/// (`H5D_btree_key_t`) plus the address it keys.
1720pub struct BtreeV1ChunkRecord {
1721    /// Grid position of the chunk. The key's element offsets are derived from
1722    /// it at encode time (`scaled * chunk_dim`), so this is the one place the
1723    /// position is stored and the sort order is over these coordinates.
1724    pub scaled: Vec<u64>,
1725    /// File offset of the chunk's bytes.
1726    pub address: u64,
1727    /// Stored byte length — the filtered length when the dataset is filtered,
1728    /// the full chunk otherwise. `u32` because the key's field is.
1729    pub nbytes: u32,
1730    /// Filter mask: bit `i` set means filter `i` was skipped for this chunk.
1731    pub filter_mask: u32,
1732}
1733
1734/// Runtime metadata for a chunked dataset indexed by a version-1 B-tree —
1735/// the classic-format chunk index (`H5Dbtree.c`), and the only one a
1736/// version-0/1 superblock file can carry.
1737pub struct BtreeV1DatasetInfo {
1738    /// Chunk dimension sizes.
1739    pub chunk_dims: Vec<u64>,
1740    /// Maximum dimensions (u64::MAX = unlimited).
1741    pub max_dims: Vec<u64>,
1742    /// The file's v1-B-tree "K" ranks. Every node's width is derived from
1743    /// them, and they are recorded only in the superblock this file was
1744    /// opened with — so they are carried rather than re-derived.
1745    pub config: BTreeV1Config,
1746    /// The chunks, in key order (`scaled` ascending, lexicographically).
1747    pub records: Vec<BtreeV1ChunkRecord>,
1748    /// Pool of node-size blocks holding the tree's nodes, on the same terms
1749    /// as [`Bt2DatasetInfo::node_addrs`]: a flush re-serializes the whole
1750    /// bulk-loaded tree over them and allocates only the shortfall, so no
1751    /// flush can orphan a block it replaced.
1752    pub node_addrs: Vec<u64>,
1753    /// Address of the tree's root node — what the version-3 data layout
1754    /// message carries. `UNDEF_ADDR` until a flush puts a node in the file,
1755    /// which is the state libhdf5 leaves a chunked dataset in until its first
1756    /// chunk is written.
1757    pub root_addr: u64,
1758    /// Number of chunks written so far.
1759    pub chunks_written: u64,
1760}
1761
1762impl BtreeV1DatasetInfo {
1763    /// The chunk shape a key's offsets are scaled by: the chunk dimensions
1764    /// with the element size appended, which is also what the layout message
1765    /// stores.
1766    fn key_dims(&self, element_size: u64) -> Vec<u64> {
1767        let mut dims = self.chunk_dims.clone();
1768        dims.push(element_size);
1769        dims
1770    }
1771
1772    /// Bulk-load the tree this index's records describe.
1773    fn build_tree(&self, element_size: u64, sizeof_addr: usize) -> ChunkBTreeV1Tree {
1774        let dims = self.key_dims(element_size);
1775        let entries: Vec<(ChunkKey, u64)> = self
1776            .records
1777            .iter()
1778            .map(|r| {
1779                (
1780                    ChunkKey::for_chunk(&r.scaled, &dims, r.nbytes, r.filter_mask),
1781                    r.address,
1782                )
1783            })
1784            .collect();
1785        // The right boundary closes the tree past its greatest key, which is
1786        // the last record's — the records are kept in key order.
1787        let last = self
1788            .records
1789            .last()
1790            .map_or_else(|| vec![0; self.chunk_dims.len()], |r| r.scaled.clone());
1791        ChunkBTreeV1Tree::build(
1792            &entries,
1793            ChunkKey::right_bound(&last, &dims),
1794            &self.config,
1795            sizeof_addr,
1796        )
1797    }
1798
1799    /// Where `scaled` sits in [`records`](Self::records): `Ok` at its record,
1800    /// `Err` at the position one would be inserted at.
1801    fn position(&self, scaled: &[u64]) -> Result<usize, usize> {
1802        self.records
1803            .binary_search_by(|r| r.scaled.as_slice().cmp(scaled))
1804    }
1805}
1806
1807/// Runtime metadata for a B-tree v2 indexed chunked dataset.
1808pub struct Bt2DatasetInfo {
1809    /// Chunk dimension sizes.
1810    pub chunk_dims: Vec<u64>,
1811    /// File offset of the BT2 header.
1812    pub bt2_header_addr: u64,
1813    /// Pool of node-size blocks (the index's
1814    /// [`node_size`](Bt2ChunkIndex::node_size) bytes each) holding the tree's
1815    /// nodes, in the order [`Bt2Tree::encode`] emits them.
1816    ///
1817    /// The single owner of the tree's node addresses: a flush re-serializes the
1818    /// whole tree over these blocks and allocates only the shortfall, so no
1819    /// flush can orphan a block it replaced. Every node is the same size, so a
1820    /// block stays usable however the tree reshapes.
1821    ///
1822    /// The pool holds exactly one block per node after every flush, in both
1823    /// directions: a taller tree allocates the shortfall, a smaller one frees
1824    /// the surplus. Nothing here depends on the record count only ever rising,
1825    /// so a record-removal path can be added to [`Bt2ChunkIndex`] without the
1826    /// blocks it drops going unreachable.
1827    pub node_addrs: Vec<u64>,
1828    /// In-memory chunk index.
1829    pub index: Bt2ChunkIndex,
1830    /// Number of chunks written so far.
1831    pub chunks_written: u64,
1832}
1833
1834/// Metadata for a group being written.
1835pub struct GroupInfo {
1836    /// Full path of this group (e.g. "/detector" or "/detector/raw").
1837    pub name: String,
1838    /// Index of the parent group in the groups vec, or None for root-level groups.
1839    pub parent: Option<usize>,
1840    /// Indices of child datasets (into `datasets` vec).
1841    pub child_datasets: Vec<usize>,
1842    /// Indices of child groups (into `groups` vec).
1843    pub child_groups: Vec<usize>,
1844    /// File offset of this group's object header (set during finalize).
1845    pub obj_header_addr: u64,
1846    /// File offset of the on-disk header a reopen found for this group, so
1847    /// finalize can free the block it supersedes.
1848    pub obj_header_written_addr: Option<u64>,
1849    /// Encoded size of that on-disk header (first block).
1850    /// Every block the object's on-disk header occupies, chunk 0 first, or
1851    /// empty when it has none yet. All of them are freed together: a rewrite
1852    /// re-encodes the whole chain into one fresh chunk, so a continuation
1853    /// block left behind is space no free-space manager records.
1854    pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
1855    /// Soft-deleted: excluded from finalize output.
1856    pub deleted: bool,
1857    /// Attributes attached to this group (e.g. NeXus `NX_class`).
1858    pub attributes: Vec<AttributeEntry>,
1859    /// When the link naming this group was created, on the writer's single
1860    /// monotonic sequence. Groups, datasets and hard links share it, so a
1861    /// parent can order its links the way they were actually made.
1862    pub creation_seq: u64,
1863    /// How this group records creation order for its links and, separately,
1864    /// for its attributes. Creation-order tracking is a property of the
1865    /// object's creation property list in libhdf5, so it is captured here
1866    /// when the group is created rather than read from the writer at
1867    /// finalize: a later change of policy must not rewrite an object already
1868    /// made.
1869    pub track_order: TrackOrder,
1870    /// The times this group tracks, on the same terms as
1871    /// [`DatasetInfo::times`]. A version-1 group header records none of them:
1872    /// nothing calls `H5O_touch_oh` with `force` for a group, so the message a
1873    /// version-1 dataset gets is never created for one.
1874    pub times: Option<ObjectTimes>,
1875}
1876
1877/// One object's creation-order policy, with the two subsystems libhdf5 keeps
1878/// apart kept apart here too.
1879///
1880/// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` are separate
1881/// calls reading back out of separate places on disk — the Link Info message
1882/// and the object header's own flag bits — and a file may set either alone.
1883/// Carrying them as one flag made a reopen give a one-of-two file both or
1884/// neither.
1885#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
1886pub struct TrackOrder {
1887    /// Creation order of the links this group holds. Meaningless for a
1888    /// dataset, which is why `DatasetInfo` keeps only the attribute half.
1889    pub links: CreationOrder,
1890    /// Creation order of the attributes attached to this object.
1891    pub attrs: CreationOrder,
1892}
1893
1894impl TrackOrder {
1895    /// The policy the crate's single `track_order` knob selects: both
1896    /// subsystems tracked *and* indexed, or neither — the pair h5py's
1897    /// `File(track_order=True)` writes.
1898    pub fn uniform(track: bool) -> Self {
1899        let order = if track {
1900            CreationOrder::Indexed
1901        } else {
1902            CreationOrder::Untracked
1903        };
1904        Self {
1905            links: order,
1906            attrs: order,
1907        }
1908    }
1909}
1910
1911/// The object a [`HardLink`] resolves to.
1912#[derive(Clone, Copy)]
1913pub enum HardLinkTarget {
1914    /// Index into the writer's `datasets` vec.
1915    Dataset(usize),
1916    /// Index into the writer's `groups` vec.
1917    Group(usize),
1918}
1919
1920/// A user-created hard link: an additional name, in some group, for an
1921/// object that already exists under its own name.
1922///
1923/// The HDF5 file format makes every group entry a `name -> object header
1924/// address` mapping, so a hard link is just a second such entry pointing at
1925/// an already-written object. No data is copied.
1926#[derive(Clone)]
1927pub struct HardLink {
1928    /// Parent group index (`None` = the root group).
1929    pub parent: Option<usize>,
1930    /// Leaf name of the link within the parent group.
1931    pub name: String,
1932    /// Object this link resolves to.
1933    pub target: HardLinkTarget,
1934    /// When this link was created; see [`GroupInfo::creation_seq`].
1935    pub creation_seq: u64,
1936}
1937
1938/// A user-created symbolic link: a name in a group whose value is a path
1939/// rather than an object header address.
1940///
1941/// A soft link holds a path within this file; an external link holds a file
1942/// name and a path within that file. Neither names an object this writer
1943/// owns, so — unlike [`HardLink`] — nothing about it is resolved: the link is
1944/// stored as written and answered at traversal time, exactly as `H5Lcreate_soft`
1945/// and `H5Lcreate_external` store theirs.
1946#[derive(Clone)]
1947pub struct SymbolicLink {
1948    /// Parent group index (`None` = the root group).
1949    pub parent: Option<usize>,
1950    /// Leaf name of the link within the parent group.
1951    pub name: String,
1952    /// The path (and, for an external link, the file) this link names.
1953    pub target: LinkTarget,
1954    /// When this link was created; see [`GroupInfo::creation_seq`].
1955    pub creation_seq: u64,
1956}
1957
1958/// A committed (named) datatype: an object header holding one datatype
1959/// message and nothing else, reached by a link like any other object.
1960///
1961/// `H5Tcommit2` makes the type an object in its own right so several datasets
1962/// can declare they share it; each of those datasets then stores a pointer to
1963/// this object header in place of its own datatype message. The object's
1964/// reference count is therefore the links naming it *plus* the datasets
1965/// sharing it — `H5O__shared_link_adj` counts a share as a link — and an
1966/// object no link and no dataset reaches is not written at all.
1967#[derive(Clone)]
1968pub struct CommittedDatatype {
1969    /// Full path with no leading `/`, the form dataset names take.
1970    pub name: String,
1971    /// Parent group index (`None` = the root group).
1972    pub parent: Option<usize>,
1973    /// The committed type.
1974    pub datatype: DatatypeMessage,
1975    /// When the link naming it was created; see [`GroupInfo::creation_seq`].
1976    pub creation_seq: u64,
1977    /// The times it tracks, on the same terms as [`DatasetInfo::times`]. A
1978    /// version-1 committed datatype header records none of them, for the same
1979    /// reason a version-1 group's does not.
1980    pub times: Option<ObjectTimes>,
1981    /// File offset of its object header (set during finalize).
1982    pub obj_header_addr: u64,
1983}
1984
1985/// Where the object header a dataset's shared datatype pointer must name
1986/// comes from.
1987///
1988/// A dataset built on a committed type stores no datatype message: it stores
1989/// the address of the type's object header. Only the address matters at
1990/// encode time, but it is knowable at two different moments — a type this
1991/// session commits has no address until finalize lays the file out, while one
1992/// a reopen found is already at an address this session will not move. Naming
1993/// both here keeps [`build_dataset_header`](Hdf5Writer::build_dataset_header)
1994/// the one place that turns a share into a pointer, whichever way the share
1995/// arrived.
1996#[derive(Clone, Copy, Debug, PartialEq, Eq)]
1997pub enum CommittedTypeRef {
1998    /// A type committed in this session, by its index in
1999    /// [`committed_datatypes`](Hdf5Writer::committed_datatypes); its address
2000    /// is read from that registry once finalize has stamped one.
2001    Session(usize),
2002    /// A committed datatype a reopen kept by its bytes, at the object header
2003    /// address it already occupies.
2004    Preserved(u64),
2005}
2006
2007/// A link a reopened file already held that this writer cannot express.
2008///
2009/// Soft, external and user-defined links have no creation, retarget or delete
2010/// operation here — only hard links do — so a header rewrite that emits what
2011/// the registry models would erase them. Their encoded `Link` message rides
2012/// along instead and is written back byte for byte, which preserves every
2013/// field (name character set, creation order, the link value) without this
2014/// writer having to model any of them.
2015///
2016/// A *hard* link is preserved the same way when the object it names is one
2017/// the reopen could not model: writing the link back unchanged leaves that
2018/// object's header exactly where it is, which is the only way the rewrite can
2019/// keep what it cannot rebuild.
2020#[derive(Clone)]
2021pub struct PreservedLink {
2022    /// Parent group index (`None` = the root group).
2023    pub parent: Option<usize>,
2024    /// Leaf name of the link within the parent group.
2025    pub name: String,
2026    /// The link's class, decoded once at collection so listings can report
2027    /// it. Never the source of what gets written — `encoded` is.
2028    pub class: crate::io::reader::LinkClass,
2029    /// The encoded `Link` message body, exactly as read from the file.
2030    pub encoded: Vec<u8>,
2031    /// Why the object this link names could not be modelled, for the callers
2032    /// that ask for it by name. `None` when the link's own class — not its
2033    /// target — is what this writer cannot express.
2034    pub reason: Option<String>,
2035    /// What the object this link names is, when the walk could tell. A
2036    /// listing asks this; `reason` is prose for the caller that asks why.
2037    pub kind: PreservedKind,
2038}
2039
2040/// Every link a reopen walk met, split by what the writer can do with it.
2041/// A header rewrite emits both halves, so a link in neither half is a link
2042/// the close would destroy.
2043#[derive(Default)]
2044struct CollectedLinks {
2045    /// Hard links whose target the reopen modelled, with the plan that says
2046    /// how to rebuild it.
2047    hard: Vec<(HardEntry, CollectedObject)>,
2048    /// Links written back unchanged: the class this writer cannot express,
2049    /// and the hard links whose object it cannot model.
2050    preserved: Vec<PreservedEntry>,
2051}
2052
2053/// One hard link the reopen walk met: what it names, and the exact message
2054/// that names it.
2055#[derive(Clone)]
2056struct HardEntry {
2057    /// Full link path, in the no-leading-`/` form the registry uses.
2058    path: String,
2059    /// Object header address the link names.
2060    address: u64,
2061    /// The encoded `Link` message body, exactly as read from the file.
2062    encoded: Vec<u8>,
2063}
2064
2065/// A link the rewrite writes back exactly as it read it.
2066struct PreservedEntry {
2067    path: String,
2068    class: crate::io::reader::LinkClass,
2069    encoded: Vec<u8>,
2070    /// Why the object it names could not be modelled; `None` when the link's
2071    /// own class is what this writer cannot express.
2072    reason: Option<String>,
2073    /// What the object is, when the walk could tell.
2074    kind: PreservedKind,
2075}
2076
2077/// What a reopen can do with one object it reached.
2078///
2079/// A header rewrite emits a modelled object out of the registry, so the
2080/// registry may hold an object only when *every* message the model consumes
2081/// decoded. A partial read is not a smaller object, it is a different one:
2082/// before this rule a dataset whose datatype message did not decode was
2083/// registered as a group, and the close rewrote its header as one.
2084enum ObjectPlan {
2085    /// A dataset the rewrite can rebuild.
2086    Dataset(Box<DatasetParts>),
2087    /// A group the rewrite can rebuild, and the links it holds.
2088    Group(GroupParts),
2089    /// An object this writer cannot model, and why. Its header is never
2090    /// rewritten and never freed; the link naming it is written back byte for
2091    /// byte, so the object stays exactly as the file already had it — what
2092    /// libhdf5 does with the parts of a file it does not understand.
2093    ///
2094    /// `kind` is what the walk could still tell about the object it is
2095    /// keeping. Not modelling an object is not the same as not knowing what
2096    /// it is, and answering the second question with the first is what made
2097    /// `named_datatype_names` deny, in write mode, a datatype the same file
2098    /// reports in read mode.
2099    Preserve { why: String, kind: PreservedKind },
2100}
2101
2102/// What a preserved object is, as far as the reopen walk could tell.
2103///
2104/// Deliberately not a copy of the reader's `ObjectKind`: that one carries the
2105/// decoded object, and a preserved object is precisely the one whose contents
2106/// the writer does not decode. This says only what a listing needs.
2107#[derive(Clone, Copy, PartialEq, Eq, Debug)]
2108pub enum PreservedKind {
2109    /// The walk did not classify it — or the link's own class, not its
2110    /// target, is what could not be expressed.
2111    Unclassified,
2112    /// A committed (named) datatype, by
2113    /// [`header_is_committed_datatype`](crate::io::reader::header_is_committed_datatype).
2114    NamedDatatype,
2115}
2116
2117impl ObjectPlan {
2118    /// An object kept by its bytes, of a kind the walk did not classify.
2119    ///
2120    /// Every reason that is a *failure* to read reaches this: a message that
2121    /// did not decode says nothing about what the object was.
2122    fn preserve(why: impl Into<String>) -> Self {
2123        ObjectPlan::Preserve {
2124            why: why.into(),
2125            kind: PreservedKind::Unclassified,
2126        }
2127    }
2128}
2129
2130/// The messages a dataset's rewrite is built from, all decoded.
2131struct DatasetParts {
2132    /// Every block the header chain occupies, chunk 0 first. All of them are
2133    /// superseded: the rewrite re-encodes the whole chain into one fresh
2134    /// chunk, so a continuation left unfreed is space nothing claims.
2135    header_blocks: crate::io::object_header_io::HeaderBlocks,
2136    datatype: DatatypeMessage,
2137    /// The committed datatype object header `datatype` was read *through*,
2138    /// when the header stores a pointer instead of a message of its own.
2139    ///
2140    /// The literal type is in `datatype` either way, because the read resolves
2141    /// the pointer before anything decodes it; this is what a rewrite needs to
2142    /// put the pointer back rather than inline a copy of the named type and
2143    /// leave `H5Tcommitted` false.
2144    committed_type: Option<u64>,
2145    dataspace: crate::format::messages::dataspace::DataspaceMessage,
2146    /// The object format the reopen found this dataset's messages written in,
2147    /// read from the dataspace message's own version byte.
2148    ///
2149    /// A version-2 superblock does not settle it: `H5F__super_init` raises the
2150    /// superblock for a shared-message table or non-default file-space
2151    /// properties without touching `H5F_LOW_BOUND` (H5Fsuper.c:1135, :1144), so
2152    /// a file created at the earliest bound with either can hold version-1
2153    /// messages under a version-2 superblock — which is what
2154    /// `tests/fixtures/sohm_*.h5` are.
2155    read_format: ObjectFormat,
2156    layout: crate::format::messages::data_layout::DataLayoutMessage,
2157    filter_pipeline: Option<FilterPipeline>,
2158    fill_value: Option<Vec<u8>>,
2159    /// The fill-value message's write-time byte, preserved across a
2160    /// rewrite the same way `fill_value` is — an appended-to dataset must
2161    /// keep the policy libhdf5 (or this writer) declared for it, not fall
2162    /// back to the `H5D_CRT_FILL_TIME_DEF` a fresh dataset gets.
2163    fill_write_time: u8,
2164    attributes: Vec<AttributeEntry>,
2165    /// The creation-order policy the on-disk header declares; a rewrite that
2166    /// read it from the writer instead would stamp this session's policy onto
2167    /// an object libhdf5 created under another.
2168    track_order: TrackOrder,
2169    /// The times the on-disk header records, for the same reason: whether an
2170    /// object tracks them is settled when it is created, not when it is
2171    /// rewritten. Recovered by [`ObjectHeader::recorded_times`].
2172    times: Option<ObjectTimes>,
2173    /// The dense storage the rewrite supersedes and must free.
2174    dense: DenseCarry,
2175    /// The External File List the header carries, with each slot's name
2176    /// already read back out of the local heap the message points at. `None`
2177    /// for a dataset whose raw data is in this file.
2178    ///
2179    /// Carried rather than re-derived because the rewrite has to re-emit the
2180    /// message: a contiguous layout with an undefined address and no EFL
2181    /// beside it is a dataset with no data at all, so dropping this on a
2182    /// header rewrite would silently unlink every external byte.
2183    external: Option<ExternalStorage>,
2184}
2185
2186/// The same for a group, plus the links it holds — decoded once, with the
2187/// bytes they came from, so the walk and the rewrite agree on its contents.
2188struct GroupParts {
2189    header_blocks: crate::io::object_header_io::HeaderBlocks,
2190    attributes: Vec<AttributeEntry>,
2191    links: Vec<(crate::format::messages::link::LinkMessage, Vec<u8>)>,
2192    track_order: TrackOrder,
2193    times: Option<ObjectTimes>,
2194    dense: DenseCarry,
2195    /// The symbol-table storage a classic group's header names — the blocks
2196    /// the rewrite supersedes. `None` for a link-message group, which has
2197    /// none. Its links are already in `links`: the walk turns each symbol
2198    /// table entry into the link message it stands for, so nothing downstream
2199    /// has to know which of the two forms the group was in.
2200    stab: Option<StabExtents>,
2201}
2202
2203/// The dense storage one reopened object's header names, which the rewrite of
2204/// that header stops naming and therefore has to free. Both halves are read
2205/// back before this is built — a heap that could not be read makes the object
2206/// [`ObjectPlan::Preserve`], so nothing here describes storage whose contents
2207/// were lost.
2208#[derive(Default)]
2209struct DenseCarry {
2210    attrs: Option<AttributeInfoMessage>,
2211    links: Option<LinkInfoMessage>,
2212}
2213
2214/// A modelled object, as the walk hands it to the registry rebuild. A group's
2215/// links are not here: the walk followed them, and each child is an entry of
2216/// its own.
2217enum CollectedObject {
2218    Dataset(Box<DatasetParts>),
2219    Group {
2220        header_blocks: crate::io::object_header_io::HeaderBlocks,
2221        attributes: Vec<AttributeEntry>,
2222        track_order: TrackOrder,
2223        times: Option<ObjectTimes>,
2224        dense: DenseCarry,
2225        stab: Option<StabExtents>,
2226    },
2227}
2228
2229/// The reopen's discovery pass: one walk that classifies every object it
2230/// reaches and descends into the groups among them.
2231///
2232/// Every object the close will touch is decided here and nowhere else, so
2233/// "modelled or preserved" is a property of the walk rather than of whatever
2234/// each later stage happened to be able to decode.
2235struct ReopenWalk<'a> {
2236    handle: &'a mut FileHandle,
2237    meta: &'a crate::io::FileMeta,
2238    out: CollectedLinks,
2239    /// Object headers already descended into, so hard-link cycles end.
2240    visited: std::collections::HashSet<u64>,
2241}
2242
2243impl<'a> ReopenWalk<'a> {
2244    fn new(handle: &'a mut FileHandle, meta: &'a crate::io::FileMeta) -> Self {
2245        Self {
2246            handle,
2247            meta,
2248            out: CollectedLinks::default(),
2249            visited: std::collections::HashSet::new(),
2250        }
2251    }
2252
2253    /// Everything the walk found.
2254    fn finish(self) -> CollectedLinks {
2255        self.out
2256    }
2257
2258    /// Decide what the reopen can do with the object at `addr`.
2259    ///
2260    /// The single gate: every object the rewrite touches is classified here,
2261    /// and an object is modelled only when each message the model consumes
2262    /// decoded. See [`ObjectPlan`] for why anything else must keep its bytes.
2263    fn plan(&mut self, addr: u64) -> IoResult<ObjectPlan> {
2264        let (handle, meta) = (&mut *self.handle, self.meta);
2265        let ctx = &meta.ctx;
2266        use crate::format::messages::data_layout::DataLayoutMessage;
2267        use crate::format::messages::dataspace::DataspaceMessage;
2268        use crate::format::messages::link::{CharacterSet, LinkMessage};
2269        use crate::format::messages::link_info::LinkInfoMessage;
2270        use crate::format::messages::shared::MSG_FLAG_SHARED;
2271        use crate::format::messages::{
2272            MSG_ATTRIBUTE, MSG_DATASPACE, MSG_DATATYPE, MSG_DATA_LAYOUT, MSG_EXTERNAL_FILE_LIST,
2273            MSG_FILL_VALUE, MSG_FILTER_PIPELINE, MSG_LINK, MSG_LINK_INFO, MSG_SYMBOL_TABLE,
2274        };
2275
2276        // The whole chain, messages and blocks alike: a filter pipeline or an
2277        // attribute that spilled into a continuation is one the rewrite would
2278        // otherwise drop, and a continuation block it does not know about is
2279        // one the rewrite would orphan.
2280        let (header, header_blocks) =
2281            match crate::io::object_header_io::read_object_header_with_blocks(handle, meta, addr) {
2282                Ok(h) => h,
2283                Err(e) => {
2284                    return Ok(ObjectPlan::preserve(format!(
2285                        "its object header chain does not read: {e}"
2286                    )))
2287                }
2288            };
2289
2290        // The policy, the times and the storage the header declares, read once
2291        // from the whole chain: all three are properties of the object, not of
2292        // any one message the loop below happens to reach.
2293        let track_order = recover_track_order(&header, ctx);
2294        let times = header.recorded_times();
2295        let (dense_attrs, dense_links) = superseded_dense(&header, ctx);
2296
2297        // Attributes come from the reader's collector rather than from the
2298        // loop below, so compact, dense and shared attributes all reach the
2299        // rewrite by the one path that knows how to read each of them. An
2300        // object whose set did not read whole is preserved: a short set here
2301        // would be a rewrite deleting the attributes it could not read.
2302        let attributes = match take_reopened_attributes(
2303            crate::io::reader::collect_object_attributes(handle, ctx, &header),
2304            &format!("the object at {addr:#x}"),
2305        ) {
2306            Ok(a) => a,
2307            Err(e) => {
2308                return Ok(ObjectPlan::preserve(format!(
2309                    "its attributes do not read back whole: {e}"
2310                )))
2311            }
2312        };
2313
2314        let mut datatype = None;
2315        let mut dataspace = None;
2316        let mut layout = None;
2317        let mut filter_pipeline = None;
2318        let mut fill_value = None;
2319        // No fill-value message at all is the library default, the same
2320        // convention the reader-side decode (`Hdf5Reader::dataset_info`)
2321        // uses for `fill_defined`.
2322        let mut fill_write_time: u8 = FILL_TIME_IFSET;
2323        let mut external = None;
2324        let mut links = Vec::new();
2325        let mut stab = None;
2326        // A datatype, dataspace or layout message says the object is not a
2327        // group, whether or not the three a dataset needs are all there.
2328        let mut dataset_shaped = false;
2329
2330        for msg in &header.messages {
2331            let consumed = matches!(
2332                msg.msg_type,
2333                MSG_DATATYPE
2334                    | MSG_DATASPACE
2335                    | MSG_DATA_LAYOUT
2336                    | MSG_FILTER_PIPELINE
2337                    | MSG_FILL_VALUE
2338                    | MSG_EXTERNAL_FILE_LIST
2339                    | MSG_ATTRIBUTE
2340                    | MSG_LINK
2341                    | MSG_LINK_INFO
2342                    | MSG_SYMBOL_TABLE
2343            );
2344            // A shared message holds a reference to where its body lives, not
2345            // the body. Decoding those bytes as one does not fail loudly — the
2346            // reference's version byte reads as a version and a class of its
2347            // own — so the guard is the only thing between a shared datatype
2348            // and a rewrite that invents a type for it.
2349            if consumed && msg.flags & MSG_FLAG_SHARED != 0 {
2350                return Ok(ObjectPlan::preserve(format!(
2351                    "its message of type {:#04x} is a shared-message reference, which this \
2352                     writer does not resolve",
2353                    msg.msg_type
2354                )));
2355            }
2356            macro_rules! consume {
2357                ($decode:expr, $what:literal) => {
2358                    match $decode {
2359                        Ok(v) => v,
2360                        Err(e) => {
2361                            return Ok(ObjectPlan::preserve(format!(
2362                                "its {} message does not decode: {e}",
2363                                $what
2364                            )))
2365                        }
2366                    }
2367                };
2368            }
2369            match msg.msg_type {
2370                // The pre-1.6 modification time, a formatted date string
2371                // (`H5O_MTIME`, type 0x0E). `recorded_times` reads only the
2372                // modern form, and a rewrite emits only that, so an object
2373                // carrying this one would come back out with the time it
2374                // recorded gone. Keeping its bytes is the same answer an
2375                // undecodable message already gets.
2376                crate::format::messages::MSG_MOD_TIME_OLD => {
2377                    return Ok(ObjectPlan::preserve(
2378                        "it carries a pre-1.6 modification time message, which this writer \
2379                         reads but does not write",
2380                    ))
2381                }
2382                MSG_DATATYPE => {
2383                    dataset_shaped = true;
2384                    let (dt, _) = consume!(DatatypeMessage::decode(&msg.data, ctx), "datatype");
2385                    datatype = Some(dt);
2386                }
2387                MSG_DATASPACE => {
2388                    dataset_shaped = true;
2389                    let version = msg.data.first().copied().unwrap_or(1);
2390                    let (ds, _) = consume!(DataspaceMessage::decode(&msg.data, ctx), "dataspace");
2391                    dataspace = Some((ds, version));
2392                }
2393                MSG_DATA_LAYOUT => {
2394                    dataset_shaped = true;
2395                    let (dl, _) =
2396                        consume!(DataLayoutMessage::decode(&msg.data, ctx), "data layout");
2397                    layout = Some(dl);
2398                }
2399                MSG_FILTER_PIPELINE => {
2400                    let (p, _) = consume!(FilterPipeline::decode(&msg.data), "filter pipeline");
2401                    if !p.filters.is_empty() {
2402                        filter_pipeline = Some(p);
2403                    }
2404                }
2405                MSG_FILL_VALUE => {
2406                    let (fv, _) = consume!(FillValueMessage::decode(&msg.data), "fill value");
2407                    if fv.fill_defined == 2 {
2408                        fill_value = fv.fill_value;
2409                    }
2410                    fill_write_time = fv.fill_write_time;
2411                }
2412                MSG_EXTERNAL_FILE_LIST => {
2413                    dataset_shaped = true;
2414                    let (efl, _) = consume!(
2415                        ExternalFileListMessage::decode(&msg.data, ctx),
2416                        "external file list"
2417                    );
2418                    // The names live in a local heap of their own, so the
2419                    // rewrite cannot re-emit the message from its bytes alone
2420                    // — it has to be able to point at the same strings. A heap
2421                    // that does not read back leaves the object preserved,
2422                    // which is what keeps its data reachable.
2423                    let resolved = match crate::io::reader::Hdf5Reader::resolve_external_file_slots(
2424                        handle, ctx, &efl,
2425                    ) {
2426                        Ok(r) => r,
2427                        Err(e) => {
2428                            return Ok(ObjectPlan::preserve(format!(
2429                                "its external file list names do not read back: {e}"
2430                            )))
2431                        }
2432                    };
2433                    external = Some(ExternalStorage {
2434                        heap_addr: efl.heap_addr,
2435                        // `H5Fopen` opens no dataset, so nothing has read a
2436                        // dapl for this one yet; the first handle it hands
2437                        // out settles the prefix.
2438                        prefix: EfilePrefix::default(),
2439                        files: efl
2440                            .slots
2441                            .iter()
2442                            .zip(resolved)
2443                            .map(|(slot, seg)| ExternalFile {
2444                                name: seg.name,
2445                                name_offset: slot.name_offset,
2446                                offset: slot.offset,
2447                                size: slot.size,
2448                            })
2449                            .collect(),
2450                    });
2451                }
2452                MSG_LINK => {
2453                    let (l, _) = consume!(LinkMessage::decode(&msg.data, ctx), "link");
2454                    links.push((l, msg.data.clone()));
2455                }
2456                MSG_LINK_INFO => {
2457                    let (li, _) = consume!(LinkInfoMessage::decode(&msg.data, ctx), "link info");
2458                    // Once a group holds enough links libhdf5 moves them into
2459                    // the fractal heap this message names and writes no `Link`
2460                    // messages at all. Reading them back is what makes the
2461                    // rewrite emit the group with its children; a rewrite from
2462                    // the header messages alone emitted it empty, orphaning
2463                    // every object below it.
2464                    if li.fractal_heap_address != UNDEF_ADDR {
2465                        let dense = match crate::io::reader::Hdf5Reader::read_dense_links(
2466                            handle,
2467                            ctx,
2468                            li.fractal_heap_address,
2469                        ) {
2470                            Ok(l) => l,
2471                            Err(e) => {
2472                                return Ok(ObjectPlan::preserve(format!(
2473                                    "its dense link storage does not read: {e}"
2474                                )))
2475                            }
2476                        };
2477                        // Re-encoded rather than carried as bytes: a heap
2478                        // object is not a header message, so there are no
2479                        // message bytes to carry. The encoding round-trips
2480                        // through the same decoder that just read it.
2481                        links.extend(dense.into_iter().map(|l| {
2482                            let bytes = l.encode(ctx);
2483                            (l, bytes)
2484                        }));
2485                    }
2486                }
2487                MSG_SYMBOL_TABLE => {
2488                    // A classic group keeps no link message at all: its links
2489                    // are symbol table entries in the B-tree this message
2490                    // names. Turning each into the link message it stands for
2491                    // is what lets the rest of the reopen — the walk, the
2492                    // registry, the preserve path — work on one link model
2493                    // whichever form the group is in.
2494                    let Some(s) = Stab::decode(&msg.data, ctx) else {
2495                        return Ok(ObjectPlan::preserve(
2496                            "its symbol table message is shorter than the two addresses it \
2497                             must carry",
2498                        ));
2499                    };
2500                    let contents = match crate::io::symbol_table_io::read_stab(handle, meta, s) {
2501                        Ok(c) => c,
2502                        Err(e) => {
2503                            return Ok(ObjectPlan::preserve(format!(
2504                                "its symbol table does not read: {e}"
2505                            )))
2506                        }
2507                    };
2508                    stab = Some(contents.extents);
2509                    links.extend(contents.links.into_iter().map(|l| {
2510                        let msg = match l.target {
2511                            StabTarget::Hard { addr, .. } => LinkMessage::hard(&l.name, addr),
2512                            StabTarget::Soft { value } => LinkMessage::soft(&l.name, &value),
2513                        };
2514                        // An entry carries no character set field, so the link
2515                        // it stands for has the file default whatever its name
2516                        // looks like (`H5G__ent_to_link`, H5Gent.c:372).
2517                        // Deriving one from the name would take a group
2518                        // libhdf5 wrote with a high-byte ASCII name out of its
2519                        // symbol table on the rewrite.
2520                        let msg = msg.with_cset(CharacterSet::Ascii);
2521                        let bytes = msg.encode(ctx);
2522                        (msg, bytes)
2523                    }));
2524                }
2525                _ => {}
2526            }
2527        }
2528
2529        match (datatype, dataspace, layout) {
2530            // A layout `rebuild_dataset` has no arm for leaves the registry
2531            // entry with an undefined data address, and the close then rewrites
2532            // the header as a contiguous, unallocated dataset — every element
2533            // gone, silently. Only the layouts that rebuild are modelled; the
2534            // rest keep their bytes, as an undecodable message already does.
2535            // The virtual layout is this.
2536            (Some(_), Some(_), Some(layout)) if !layout_rebuilds(&layout) => {
2537                Ok(ObjectPlan::preserve(format!(
2538                    "its data layout is {}, which this writer reads but does not build",
2539                    layout.describe()
2540                )))
2541            }
2542            (Some(datatype), Some((dataspace, dataspace_version)), Some(layout)) => {
2543                // Asked of the raw chain, not of `header`: the read above has
2544                // already put the named type's message in place of the pointer.
2545                let committed_type = match crate::io::object_header_io::committed_datatype_address(
2546                    handle, meta, addr,
2547                ) {
2548                    Ok(c) => c,
2549                    Err(e) => {
2550                        return Ok(ObjectPlan::preserve(format!(
2551                            "its shared datatype pointer does not decode: {e}"
2552                        )))
2553                    }
2554                };
2555                Ok(ObjectPlan::Dataset(Box::new(DatasetParts {
2556                    header_blocks,
2557                    datatype,
2558                    committed_type,
2559                    dataspace,
2560                    read_format: if dataspace_version <= 1 {
2561                        ObjectFormat::Legacy
2562                    } else {
2563                        ObjectFormat::Modern
2564                    },
2565                    layout,
2566                    filter_pipeline,
2567                    fill_value,
2568                    fill_write_time,
2569                    attributes,
2570                    track_order,
2571                    times,
2572                    dense: DenseCarry {
2573                        attrs: dense_attrs,
2574                        links: dense_links,
2575                    },
2576                    external,
2577                })))
2578            }
2579            // A committed (named) datatype has a datatype message and neither
2580            // of the other two; so does a dataset whose header this crate only
2581            // half understands. Neither is a group, and modelling either as
2582            // one is what rewrote them into empty groups. They part company
2583            // here and nowhere else: the datatype is kept by its bytes like
2584            // the other, but a listing can still name it.
2585            _ if crate::io::reader::header_is_committed_datatype(&header) => {
2586                Ok(ObjectPlan::Preserve {
2587                    why: "it is a committed (named) datatype, which this writer carries by \
2588                          its bytes rather than re-encoding"
2589                        .into(),
2590                    kind: PreservedKind::NamedDatatype,
2591                })
2592            }
2593            _ if dataset_shaped => Ok(ObjectPlan::preserve(
2594                "it carries a datatype, dataspace or layout message but not the three a \
2595                 dataset is built from; this writer models only groups and datasets",
2596            )),
2597            _ => Ok(ObjectPlan::Group(GroupParts {
2598                header_blocks,
2599                attributes,
2600                links,
2601                track_order,
2602                times,
2603                dense: DenseCarry {
2604                    attrs: dense_attrs,
2605                    links: dense_links,
2606                },
2607                stab,
2608            })),
2609        }
2610    }
2611
2612    /// Walk `links` (one group's, already decoded), classifying every object
2613    /// they name and descending into the groups among them.
2614    fn group(
2615        &mut self,
2616        links: &[(crate::format::messages::link::LinkMessage, Vec<u8>)],
2617        prefix: &str,
2618        depth: usize,
2619    ) -> IoResult<()> {
2620        // Bound nesting depth so a pathologically deep group chain cannot
2621        // overflow the stack (the `visited` set bounds total work but not
2622        // recursion depth).
2623        if depth > 256 {
2624            return Ok(());
2625        }
2626        use crate::format::messages::link::LinkTarget;
2627        for (link, encoded) in links {
2628            let full_name = if prefix.is_empty() {
2629                link.name.clone()
2630            } else {
2631                format!("{}/{}", prefix, link.name)
2632            };
2633
2634            // Only a hard link names an object this writer can rebuild. Every
2635            // other class is kept by its bytes, because a close that emitted
2636            // only what the registry models would drop it from the file.
2637            let LinkTarget::Hard { address } = &link.target else {
2638                self.out.preserved.push(PreservedEntry {
2639                    path: full_name,
2640                    class: crate::io::reader::LinkClass::from_target(&link.target),
2641                    encoded: encoded.clone(),
2642                    reason: None,
2643                    kind: PreservedKind::Unclassified,
2644                });
2645                continue;
2646            };
2647            let entry = HardEntry {
2648                path: full_name.clone(),
2649                address: *address,
2650                encoded: encoded.clone(),
2651            };
2652
2653            match self.plan(*address)? {
2654                // Kept by its bytes, exactly as a link class this writer
2655                // cannot express is: writing the link back unchanged is what
2656                // leaves the object's header where the file already has it.
2657                ObjectPlan::Preserve { why, kind } => self.out.preserved.push(PreservedEntry {
2658                    path: full_name,
2659                    class: crate::io::reader::LinkClass::Hard,
2660                    encoded: entry.encoded,
2661                    reason: Some(why),
2662                    kind,
2663                }),
2664                ObjectPlan::Dataset(parts) => {
2665                    self.out.hard.push((entry, CollectedObject::Dataset(parts)));
2666                }
2667                ObjectPlan::Group(parts) => {
2668                    self.out.hard.push((
2669                        entry,
2670                        CollectedObject::Group {
2671                            header_blocks: parts.header_blocks,
2672                            attributes: parts.attributes,
2673                            track_order: parts.track_order,
2674                            times: parts.times,
2675                            dense: parts.dense,
2676                            stab: parts.stab,
2677                        },
2678                    ));
2679                    // Recurse only into a group's header we have not entered
2680                    // before — breaks hard-link cycles.
2681                    if self.visited.insert(*address) {
2682                        self.group(&parts.links, &full_name, depth + 1)?;
2683                    }
2684                }
2685            }
2686        }
2687        Ok(())
2688    }
2689}
2690
2691/// Rebuild one reopened dataset's in-memory registry entry, storage and
2692/// all, from the header messages the walk decoded.
2693///
2694/// Fails when the chunk index the file names does not read back. The
2695/// caller answers that by preserving the object rather than registering
2696/// a dataset whose index has forgotten where its chunks are: the close
2697/// rewrites what the registry holds, so an index rebuilt from the part of
2698/// it that decoded would strand every chunk it could not read.
2699fn rebuild_dataset(
2700    handle: &mut FileHandle,
2701    meta: &FileMeta,
2702    file_size: u64,
2703    name: String,
2704    obj_addr: u64,
2705    parts: DatasetParts,
2706) -> IoResult<DatasetInfo> {
2707    let ctx = &meta.ctx;
2708    let DatasetParts {
2709        header_blocks,
2710        datatype: dt,
2711        committed_type,
2712        dataspace: ds,
2713        read_format,
2714        layout: dl,
2715        filter_pipeline: fp,
2716        fill_value,
2717        fill_write_time,
2718        attributes: attrs,
2719        track_order,
2720        times,
2721        dense: _,
2722        external,
2723    } = parts;
2724
2725    let mut info = DatasetInfo {
2726        name,
2727        datatype: dt,
2728        // The named type's own object is preserved by its bytes, so the
2729        // address the walk read the pointer from is the address it will still
2730        // be at when this header is written back.
2731        committed_type: committed_type.map(CommittedTypeRef::Preserved),
2732        read_format: Some(read_format),
2733        external,
2734        virtual_storage: None,
2735        dataspace: ds,
2736        obj_header_addr: obj_addr,
2737        data_addr: UNDEF_ADDR,
2738        data_size: 0,
2739        compact: None,
2740        chunked: None,
2741        fixed_array: None,
2742        implicit: None,
2743        single_chunk: None,
2744        btree_v1: None,
2745        btree_v2: None,
2746        append: None,
2747        attributes: attrs,
2748        obj_header_written_addr: Some(obj_addr),
2749        obj_header_blocks: header_blocks,
2750        filter_pipeline: fp,
2751        deleted: false,
2752        extent_dirty: false,
2753        header_dirty: false,
2754        // Stamped by the caller once the whole link graph is registered: it
2755        // is the count of links reaching this object, which one dataset's
2756        // parts cannot see.
2757        nlink_written: 1,
2758        // Stamped by the caller, which knows the order the walk met each
2759        // object; the rebuild sees one dataset at a time.
2760        creation_seq: 0,
2761        track_attr_order: track_order.attrs,
2762        fill_value,
2763        fill_time: fill_write_time,
2764        // Preserve the on-disk layout version so finalize re-encodes
2765        // what it read: a v5 file reopened and appended to must not be
2766        // silently downgraded to v4 (the filtered indexes keep their
2767        // 8-byte size fields, which v4 readers would mis-derive).
2768        layout_version: match &dl {
2769            DataLayoutMessage::ChunkedV4 { version, .. } => *version,
2770            // The classic index has no version above its own: a version-3
2771            // message is the whole of `H5D__chunk_set_info`'s MAX below the
2772            // version-4 gate, and re-encoding it any higher would name an
2773            // index the message cannot carry.
2774            DataLayoutMessage::ChunkedV3 { .. } => LAYOUT_VERSION_DEFAULT,
2775            _ => 4,
2776        },
2777        times,
2778    };
2779
2780    // Reconstruct storage-specific metadata
2781    debug_assert!(
2782        layout_rebuilds(&dl),
2783        "ReopenWalk::plan must preserve a layout this has no arm for"
2784    );
2785    match &dl {
2786        DataLayoutMessage::Contiguous { address, size } => {
2787            info.data_addr = *address;
2788            info.data_size = *size;
2789        }
2790        // The image is the layout message, so the rebuild carries it out of
2791        // the header it came from: anything that makes this dataset's header
2792        // stale rewrites the layout message from `compact`, and a rebuild
2793        // that left it empty would rewrite the dataset as an unallocated
2794        // contiguous one — dropping every byte.
2795        DataLayoutMessage::Compact { data } => {
2796            info.compact = Some(data.clone());
2797        }
2798        // The classic chunk index, reconstructed into the same
2799        // `BtreeV1DatasetInfo` a chunked dataset *created* in this format
2800        // gets, so the one set of machinery — `build_tree`, the flush's block
2801        // pool, `write_chunk`, `extend_dataset`, the prune a delete runs —
2802        // drives a reopened dataset and a fresh one alike. `root_addr` is what
2803        // the layout message carries and stays undefined for a dataset whose
2804        // chunks were never written, exactly as libhdf5 leaves it.
2805        DataLayoutMessage::ChunkedV3 {
2806            chunk_dims,
2807            b_tree_address,
2808        } => {
2809            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2810            let mut walk = BtreeV1Walk::new(handle, ctx, &meta.btree, &real_chunk_dims, file_size);
2811            walk.descend(*b_tree_address, 0)?;
2812            let BtreeV1Walk {
2813                records,
2814                node_addrs,
2815                ..
2816            } = walk;
2817            let max_dims = info
2818                .dataspace
2819                .max_dims
2820                .clone()
2821                .unwrap_or_else(|| info.dataspace.dims.clone());
2822            info.btree_v1 = Some(BtreeV1DatasetInfo {
2823                chunk_dims: real_chunk_dims,
2824                max_dims,
2825                // The file's own "K" ranks, not this session's defaults: they
2826                // set every node's width, so a tree bulk-loaded under the
2827                // wrong ones would re-serialize over blocks of the wrong size.
2828                config: meta.btree,
2829                records,
2830                node_addrs,
2831                root_addr: *b_tree_address,
2832                chunks_written: 0,
2833            });
2834        }
2835        DataLayoutMessage::ChunkedV4 {
2836            chunk_dims,
2837            index_address,
2838            index_type,
2839            earray_params,
2840            single_chunk_filter,
2841            ..
2842        } => {
2843            let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2844
2845            if *index_type == crate::format::messages::data_layout::ChunkIndexType::ExtensibleArray
2846            {
2847                if let Some(params) = earray_params {
2848                    let ep = EarrayParams {
2849                        max_nelmts_bits: params.max_nelmts_bits,
2850                        idx_blk_elmts: params.idx_blk_elmts,
2851                        sup_blk_min_data_ptrs: params.sup_blk_min_data_ptrs,
2852                        data_blk_min_elmts: params.data_blk_min_elmts,
2853                        max_dblk_page_nelmts_bits: params.max_dblk_page_nelmts_bits,
2854                    };
2855                    let ndblk_addrs = compute_ndblk_addrs(ep.sup_blk_min_data_ptrs)?;
2856                    let nsblk_addrs = compute_nsblk_addrs(
2857                        ep.idx_blk_elmts,
2858                        ep.data_blk_min_elmts,
2859                        ep.sup_blk_min_data_ptrs,
2860                        ep.max_nelmts_bits,
2861                    )?;
2862
2863                    // Read EA header
2864                    let hdr_buf = handle.read_at_most(*index_address, 256)?;
2865                    let ea_header = ExtensibleArrayHeader::decode(&hdr_buf, ctx)?;
2866
2867                    let is_filtered = ea_header.class_id
2868                        == crate::format::chunk_index::extensible_array::EA_CLS_FILT_CHUNK;
2869                    let chunk_size_len = if is_filtered {
2870                        ea_header.raw_elmt_size - ctx.sizeof_addr - 4
2871                    } else {
2872                        0
2873                    };
2874
2875                    // Read the EA index block. Filtered datasets
2876                    // store a `FilteredIndexBlock`; unfiltered ones a
2877                    // plain `ExtensibleArrayIndexBlock`. Both must be
2878                    // reconstructed so a reopened dataset can append
2879                    // (write_chunk consults whichever applies).
2880                    let ea_iblk_addr = ea_header.idx_blk_addr;
2881                    let (ea_iblk, filt_iblk) = if is_filtered {
2882                        let placeholder = ExtensibleArrayIndexBlock::new(
2883                            *index_address,
2884                            ep.idx_blk_elmts,
2885                            ndblk_addrs,
2886                            nsblk_addrs,
2887                        );
2888                        let fib = if ea_iblk_addr != UNDEF_ADDR {
2889                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2890                            FilteredIndexBlock::decode(
2891                                &iblk_buf,
2892                                ctx,
2893                                ep.idx_blk_elmts as usize,
2894                                ndblk_addrs,
2895                                nsblk_addrs,
2896                                chunk_size_len,
2897                            )?
2898                        } else {
2899                            FilteredIndexBlock::new(
2900                                *index_address,
2901                                ep.idx_blk_elmts,
2902                                ndblk_addrs,
2903                                nsblk_addrs,
2904                            )
2905                        };
2906                        (placeholder, Some(fib))
2907                    } else {
2908                        let eib = if ea_iblk_addr != UNDEF_ADDR {
2909                            let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2910                            ExtensibleArrayIndexBlock::decode(
2911                                &iblk_buf,
2912                                ctx,
2913                                ep.idx_blk_elmts as usize,
2914                                ndblk_addrs,
2915                                nsblk_addrs,
2916                            )?
2917                        } else {
2918                            ExtensibleArrayIndexBlock::new(
2919                                *index_address,
2920                                ep.idx_blk_elmts,
2921                                ndblk_addrs,
2922                                nsblk_addrs,
2923                            )
2924                        };
2925                        (eib, None)
2926                    };
2927
2928                    info.chunked = Some(ChunkedDatasetInfo {
2929                        chunk_dims: real_chunk_dims,
2930                        earray_params: ep,
2931                        ea_header_addr: *index_address,
2932                        ea_iblk_addr,
2933                        ea_header,
2934                        ea_iblk,
2935                        chunks_written: 0,
2936                        filt_iblk,
2937                        chunk_size_len,
2938                    });
2939                }
2940            } else if *index_type
2941                == crate::format::messages::data_layout::ChunkIndexType::FixedArray
2942            {
2943                // Read the FA header and data block back so a
2944                // reopened dataset is writable and deletable, not
2945                // re-link only — a placeholder made a delete free
2946                // just the header and leak every chunk plus the
2947                // index. Paged data blocks (any FA with more than
2948                // dblk_page_nelmts chunks, libhdf5 default 1024)
2949                // reconstruct through the same decode owner; only
2950                // pages the bitmap marks initialized are decoded.
2951                let hdr_buf = handle.read_at_most(*index_address, 256)?;
2952                let fa_header = FixedArrayHeader::decode(&hdr_buf, ctx)?;
2953                let is_filtered = fa_header.client_id == FA_CLIENT_FILT_CHUNK;
2954                let chunk_size_len = if is_filtered {
2955                    (fa_header.element_size as usize)
2956                        .checked_sub(ctx.sizeof_addr as usize + 4)
2957                        .ok_or_else(|| {
2958                            crate::io::IoError::InvalidState(
2959                                "fixed array filtered element_size too small".into(),
2960                            )
2961                        })?
2962                } else {
2963                    0
2964                };
2965                if fa_header.data_blk_addr != UNDEF_ADDR && chunk_size_len <= 8 {
2966                    let dblk_size = fixed_array_dblk_disk_size(ctx, &fa_header) as usize;
2967                    let dblk_buf = handle.read_at_most(fa_header.data_blk_addr, dblk_size)?;
2968                    let fa_dblk =
2969                        decode_fixed_array_dblk(ctx, &fa_header, &dblk_buf, chunk_size_len)?;
2970                    info.fixed_array = Some(FixedArrayDatasetInfo {
2971                        chunk_dims: real_chunk_dims,
2972                        fa_header_addr: *index_address,
2973                        fa_dblk_addr: fa_header.data_blk_addr,
2974                        fa_header,
2975                        fa_dblk,
2976                        // Chunks written this session, matching the
2977                        // EA reconstruction above.
2978                        chunks_written: 0,
2979                    });
2980                }
2981            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::BTreeV2 {
2982                use crate::format::chunk_index::btree_v2::{
2983                    Bt2Geometry, Bt2Header, BT2_TYPE_CHUNK_FILT, BT2_TYPE_CHUNK_UNFILT,
2984                };
2985
2986                // Walk the tree back into the in-memory index and
2987                // adopt its node blocks as the flush pool. The pool
2988                // re-serializes at the header's node_size, whatever
2989                // it is — libhdf5 sizes every node from
2990                // hdr->node_size (H5B2leaf.c, H5B2internal.c) — so
2991                // a foreign size reopens too. Only a record type
2992                // that is not a chunk record, or a node size below
2993                // the bulk loader's few-records-per-node floor
2994                // (the same bound creation enforces), stays
2995                // re-link only.
2996                let hdr_buf = handle.read_at_most(*index_address, 256)?;
2997                let bt2_hdr = Bt2Header::decode(&hdr_buf, ctx)?;
2998                let ndims = real_chunk_dims.len();
2999                let is_filt = match bt2_hdr.record_type {
3000                    BT2_TYPE_CHUNK_UNFILT => Some(false),
3001                    BT2_TYPE_CHUNK_FILT => Some(true),
3002                    _ => None,
3003                };
3004                if let (Some(is_filt), true) = (
3005                    is_filt,
3006                    bt2_hdr.node_size as usize >= 10 + 3 * bt2_hdr.record_size as usize,
3007                ) {
3008                    let mut index = if is_filt {
3009                        let csl = (bt2_hdr.record_size as usize)
3010                            .checked_sub(ctx.sizeof_addr as usize + 4 + ndims * 8)
3011                            .filter(|&c| c <= 8)
3012                            .ok_or_else(|| {
3013                                crate::io::IoError::InvalidState(
3014                                    "v2 B-tree filtered record size does not fit \
3015                                     its rank and address width"
3016                                        .into(),
3017                                )
3018                            })?;
3019                        Bt2ChunkIndex::new_filtered(ndims, csl as u8)
3020                    } else {
3021                        Bt2ChunkIndex::new_unfiltered(ndims)
3022                    };
3023                    // Re-serialize with the creator's parameters:
3024                    // node blocks keep their size and the rewritten
3025                    // header keeps its declared split/merge.
3026                    index.node_size = bt2_hdr.node_size;
3027                    index.split_percent = bt2_hdr.split_percent;
3028                    index.merge_percent = bt2_hdr.merge_percent;
3029                    let mut node_addrs = Vec::new();
3030                    if bt2_hdr.root_node_addr != UNDEF_ADDR && bt2_hdr.total_num_records > 0 {
3031                        let geo = Bt2Geometry::new(
3032                            bt2_hdr.node_size,
3033                            bt2_hdr.record_size,
3034                            bt2_hdr.depth,
3035                            ctx.sizeof_addr,
3036                        );
3037                        let mut walk =
3038                            Bt2Walk::new(handle, ctx, bt2_hdr.record_size, bt2_hdr.node_size, &geo);
3039                        walk.descend(
3040                            bt2_hdr.root_node_addr,
3041                            bt2_hdr.depth,
3042                            bt2_hdr.num_records_in_root,
3043                        )?;
3044                        node_addrs = walk.node_addrs;
3045                        let record_bytes = walk.records;
3046                        let total = if bt2_hdr.record_size > 0 {
3047                            record_bytes.len() / bt2_hdr.record_size as usize
3048                        } else {
3049                            0
3050                        };
3051                        if is_filt {
3052                            for r in Bt2ChunkIndex::decode_filtered_records(
3053                                &record_bytes,
3054                                total,
3055                                ndims,
3056                                bt2_hdr.record_size,
3057                                ctx,
3058                            )? {
3059                                index.insert_filtered(
3060                                    r.scaled_offsets,
3061                                    r.chunk_address,
3062                                    r.chunk_size,
3063                                    r.filter_mask,
3064                                );
3065                            }
3066                        } else {
3067                            for r in Bt2ChunkIndex::decode_unfiltered_records(
3068                                &record_bytes,
3069                                total,
3070                                ndims,
3071                                ctx,
3072                            )? {
3073                                index.insert(r.scaled_offsets, r.chunk_address);
3074                            }
3075                        }
3076                    }
3077                    info.btree_v2 = Some(Bt2DatasetInfo {
3078                        chunk_dims: real_chunk_dims,
3079                        bt2_header_addr: *index_address,
3080                        node_addrs,
3081                        index,
3082                        chunks_written: 0,
3083                    });
3084                }
3085            } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::Implicit
3086            {
3087                // Nothing to read back: the index *is* the run of chunk space
3088                // at `index_address`, and its length is the chunk grid times
3089                // the chunk size. Reconstructing that length is what lets a
3090                // delete free the storage and a write address it — a rebuild
3091                // that left this empty would rewrite the dataset as an
3092                // unallocated contiguous one, dropping every byte.
3093                let mut nchunks: u64 = 1;
3094                for g in crate::io::chunk_grid::index_grid(
3095                    &info.dataspace.dims,
3096                    info.dataspace.max_dims.as_deref(),
3097                    &real_chunk_dims,
3098                )? {
3099                    nchunks = nchunks.checked_mul(g).ok_or_else(|| {
3100                        crate::io::IoError::InvalidState("chunk count overflows u64".into())
3101                    })?;
3102                }
3103                let data_size = nchunks
3104                    .checked_mul(chunk_dims.iter().product::<u64>())
3105                    .ok_or_else(|| {
3106                        crate::io::IoError::InvalidState(
3107                            "implicit chunk storage overflows u64".into(),
3108                        )
3109                    })?;
3110                info.implicit = Some(ImplicitDatasetInfo {
3111                    chunk_dims: real_chunk_dims,
3112                    data_addr: *index_address,
3113                    data_size,
3114                });
3115            } else if *index_type
3116                == crate::format::messages::data_layout::ChunkIndexType::SingleChunk
3117            {
3118                // No index structure to read back either: the one chunk's
3119                // address, and its stored size and filter mask if the
3120                // layout's filtered flag is set, are the whole of the
3121                // layout message. `chunk_dims` already includes the
3122                // trailing element-size dimension, so its product is the
3123                // chunk's unfiltered byte length directly (see `data_size`
3124                // in the Implicit arm above).
3125                let data_size = chunk_dims.iter().product::<u64>();
3126                let (nbytes, filter_mask) = match single_chunk_filter {
3127                    Some(scf) => (scf.nbytes, scf.filter_mask),
3128                    None => (data_size, 0),
3129                };
3130                info.single_chunk = Some(SingleChunkDatasetInfo {
3131                    chunk_dims: real_chunk_dims,
3132                    data_addr: *index_address,
3133                    data_size,
3134                    nbytes,
3135                    filter_mask,
3136                    chunks_written: 0,
3137                    // Whether this was created with early allocation isn't
3138                    // recoverable here: `fill_value` above is only the
3139                    // decoded fill bytes, not the fill-value message's
3140                    // `alloc_time` byte the layout was chosen under. A
3141                    // reopened dataset that later gets a header rewrite
3142                    // therefore reports incremental allocation regardless
3143                    // of how it was actually created — the same
3144                    // imprecision a reopened `fixed_array`/`btree_v2`
3145                    // dataset already has, for the same reason.
3146                    early_alloc: false,
3147                });
3148            }
3149        }
3150        // Unreachable by `layout_rebuilds`, which is the gate
3151        // `ReopenWalk::plan` consults before it ever calls this.
3152        _ => {}
3153    }
3154
3155    Ok(info)
3156}
3157
3158/// Write `data` at *dataset-relative* byte offset `skip` into an external file
3159/// list, walking slots by cumulative declared size exactly like
3160/// `H5D__efl_write` (H5Defl.c).
3161///
3162/// Each slot's file is opened create-if-missing and never truncated, so a
3163/// write touches only the byte range that slot owns. A write past the *total*
3164/// declared size of the list is an error, matching upstream's "write past
3165/// logical end of file" check.
3166fn write_external_file_bytes(
3167    files: &[ExternalFile],
3168    extfile_prefix: Option<&Path>,
3169    mut skip: u64,
3170    data: &[u8],
3171) -> IoResult<()> {
3172    // `H5D__efl_write`'s slot walk: an `H5O_EFL_UNLIMITED` slot matches every
3173    // remaining offset (`skip >= u64::MAX` is never true), so the search stops
3174    // there and the write below takes the whole rest of the data.
3175    let mut slot_idx = 0usize;
3176    while slot_idx < files.len() && skip >= files[slot_idx].size {
3177        skip -= files[slot_idx].size;
3178        slot_idx += 1;
3179    }
3180
3181    let mut written = 0usize;
3182    while written < data.len() {
3183        let Some(slot) = files.get(slot_idx) else {
3184            return Err(crate::io::IoError::InvalidState(
3185                "write past the logical end of the external file list".into(),
3186            ));
3187        };
3188        let full_path = crate::io::reader::combine_prefixed_path(extfile_prefix, &slot.name);
3189        let ext_handle = FileHandle::open_or_create_readwrite_with_locking(
3190            &full_path,
3191            crate::io::locking::FileLocking::Disabled,
3192        )
3193        .map_err(|e| {
3194            crate::io::IoError::InvalidState(format!(
3195                "unable to open external raw data file {} for writing: {e}",
3196                full_path.display()
3197            ))
3198        })?;
3199        let this_write = (slot.size - skip).min((data.len() - written) as u64) as usize;
3200        ext_handle.write_at(slot.offset + skip, &data[written..written + this_write])?;
3201        // This handle is dropped at the end of the iteration, and `Drop` can
3202        // only print a flush failure. Empty the accumulator here instead, so a
3203        // full disk on an external raw-data file reaches the caller.
3204        ext_handle.flush()?;
3205
3206        written += this_write;
3207        skip = 0;
3208        slot_idx += 1;
3209    }
3210    Ok(())
3211}
3212
3213/// The directory the HDF5 file at `path` sits in — libhdf5's `H5F_t::extpath`,
3214/// which `H5D__build_file_prefix` expands `${ORIGIN}` to.
3215///
3216/// Canonicalized, so the value survives the process changing directory and so
3217/// a writer and a reader of the same file agree on it. Called once per open,
3218/// never per I/O, for exactly that reason.
3219fn source_dir_of(path: &Path) -> IoResult<PathBuf> {
3220    let canonical = std::fs::canonicalize(path)?;
3221    Ok(canonical
3222        .parent()
3223        .map(Path::to_path_buf)
3224        .unwrap_or_default())
3225}
3226
3227/// Whether [`rebuild_dataset`] has an arm that reconstructs this layout.
3228///
3229/// The single list: `ReopenWalk::plan` preserves an object whose layout this
3230/// says no to, so a layout added to one side and not the other cannot happen.
3231/// Keeping two lists is what would rewrite a modelled dataset as unallocated
3232/// contiguous storage, or preserve one the writer can now build.
3233fn layout_rebuilds(layout: &DataLayoutMessage) -> bool {
3234    matches!(
3235        layout,
3236        DataLayoutMessage::Contiguous { .. }
3237            | DataLayoutMessage::Compact { .. }
3238            | DataLayoutMessage::ChunkedV3 { .. }
3239            | DataLayoutMessage::ChunkedV4 { .. }
3240    )
3241}
3242
3243/// Encode an Object Reference Count message (type 0x16) body: a version
3244/// byte (`H5O_REFCOUNT_VERSION` = 0) followed by the little-endian u32
3245/// count. Emitted on objects reached by more than one hard link.
3246fn encode_refcount(refcount: u32) -> Vec<u8> {
3247    let mut v = Vec::with_capacity(5);
3248    v.push(0u8);
3249    v.extend_from_slice(&refcount.to_le_bytes());
3250    v
3251}
3252
3253/// The symbol-table storage of every group that has one, and the single owner
3254/// of which groups those are.
3255///
3256/// A group stores its links in a symbol table because the file was *made* that
3257/// way — `H5F_LIBVER_EARLIEST` is the one bound `H5G__obj_create_real`
3258/// (H5Gobj.c:179) writes them at — or because it already had one when the file
3259/// was reopened. The second is not the first: `H5G_obj_insert` inserts into
3260/// whatever storage the group is in and converts only when a link will not fit
3261/// an entry (H5Gobj.c:512), so a symbol table survives a reopen at any bound.
3262/// A file with shared messages is where the two come apart, because its
3263/// superblock extension forces a version-2 superblock over symbol-table groups
3264/// (H5Fsuper.c:1135) and `H5F__super_read` then raises the low bound to
3265/// `H5F_LIBVER_V18` on reopen — new objects are the modern generation while the
3266/// groups already there stay symbol tables.
3267struct SymbolTables {
3268    /// The scopes the reopen found a Symbol Table message on. Fixed for the
3269    /// session: a group already in that storage stays in it, whatever bound
3270    /// the objects added beside it are written at.
3271    found: HashSet<LinkScope>,
3272    /// The symbol-table storage each group's header already names, by the
3273    /// scope whose rewrite supersedes it.
3274    ///
3275    /// INVARIANT: every entry is freed exactly once, by
3276    /// [`Hdf5Writer::prepare_symbol_tables`], which removes it as it frees.
3277    superseded: Slot<HashMap<LinkScope, StabExtents>>,
3278    /// The storage that same pass laid out, read by the header builders.
3279    ///
3280    /// INVARIANT: an entry exists here only after every block of that group's
3281    /// heap and B-tree is on disk. `build_group_header` reads it and never
3282    /// builds — a header is sized and then written by two separate calls, so a
3283    /// build that allocated would allocate twice.
3284    written: Slot<HashMap<LinkScope, Stab>>,
3285}
3286
3287impl SymbolTables {
3288    /// What a file being created starts from: no group found in a symbol table
3289    /// because none was read, and nothing on disk to free.
3290    fn none_found() -> Self {
3291        Self {
3292            found: HashSet::new(),
3293            superseded: Slot::new(HashMap::new()),
3294            written: Slot::new(HashMap::new()),
3295        }
3296    }
3297}
3298
3299/// Everything a version-0/1 (symbol-table) file carries that a version-2/3 one
3300/// does not.
3301///
3302/// Its presence *is* the generation switch — [`Hdf5Writer::message_format`]
3303/// reads nothing else: libhdf5 at `H5F_LIBVER_EARLIEST` writes a version-0/1
3304/// superblock over version-1 object headers over symbol-table groups. Which
3305/// groups are symbol tables is the separate question [`SymbolTables`] answers,
3306/// because a reopen at a newer bound keeps the ones it finds.
3307///
3308/// Two things put one here, and only two: reopening a file that already is in
3309/// that format, and creating one at that bound
3310/// ([`LegacyFile::created`]). Neither is distinguished afterwards — a file is
3311/// classic or it is not, and every encoder asks only that.
3312struct LegacyFile {
3313    /// The superblock as it was read, or as [`LegacyFile::created`] built it.
3314    /// The close re-emits it with only the end of file and the root symbol
3315    /// table entry recomputed: the "K" ranks in particular are recorded
3316    /// nowhere else, and every node width in the file is derived from them.
3317    superblock: SuperblockV0V1,
3318}
3319
3320impl LegacyFile {
3321    /// The classic-format state a file created at `H5F_LIBVER_EARLIEST`
3322    /// starts from.
3323    ///
3324    /// A new file has no symbol table on disk to free and none laid out, so
3325    /// its [`SymbolTables`] starts empty and every group it makes takes that
3326    /// storage from the bound rather than from what was found.
3327    ///
3328    /// The superblock is the one `H5F__super_init` writes at that bound: the
3329    /// library-default "K" ranks (`H5F_CRT_SYM_LEAF_DEF`,
3330    /// `HDF5_BTREE_SNODE_IK_DEF`), no free-space info and no driver info. The
3331    /// root entry's object header address and cached symbol table are stamped
3332    /// in by [`Hdf5Writer::write_superblock`] once the root group has one;
3333    /// its name offset is the empty string at the front of every local heap.
3334    ///
3335    /// Version 0, not 1: a version-1 superblock exists only to carry a
3336    /// non-default chunked-storage "K" value (H5Fsuper.c:1150), and this
3337    /// writer has no property to set one.
3338    fn created(ctx: FormatContext, base_address: u64) -> Self {
3339        let btree = BTreeV1Config::default();
3340        Self {
3341            superblock: SuperblockV0V1 {
3342                version: SUPERBLOCK_V0,
3343                sizeof_offsets: ctx.sizeof_addr,
3344                sizeof_lengths: ctx.sizeof_size,
3345                file_consistency_flags: 0,
3346                sym_leaf_k: btree.sym_leaf_k,
3347                btree_internal_k: btree.snode_internal_k,
3348                indexed_storage_k: None,
3349                base_address,
3350                superblock_extension_address: UNDEF_ADDR,
3351                end_of_file_address: 0,
3352                driver_info_address: UNDEF_ADDR,
3353                root_symbol_table_entry: SymbolTableEntry {
3354                    name_offset: 0,
3355                    obj_header_addr: UNDEF_ADDR,
3356                    cache: SymbolTableCache::Nothing,
3357                },
3358            },
3359        }
3360    }
3361}
3362
3363/// The superblock extension a reopen found, and the single owner of the one
3364/// this file's close writes back.
3365///
3366/// The extension is external truth: it is where a file records the things its
3367/// superblock has no field for — non-default v1 B-tree "K" ranks, a driver's
3368/// settings, the file space strategy and its persisted free-space managers,
3369/// and the shared object header message table. `H5F__super_ext_write_msg`
3370/// modifies one message of it and leaves the rest alone, so a close that lays
3371/// a fresh extension out from what *this writer* models drops everything it
3372/// does not — and the K ranks are not decoration: a chunked dataset's version-1
3373/// B-tree nodes are sized from `chunk_internal_k`, so a reader that has lost
3374/// the message reads the tree at the default rank and fails outright.
3375///
3376/// INVARIANT: every message of the extension read is re-emitted by
3377/// [`Hdf5Writer::write_superblock_extension`], byte for byte, except the
3378/// shared-message table — the one message naming storage this session lays out
3379/// afresh, which [`SohmState`] recomputes. Nothing else here is interpreted,
3380/// so a message this crate does not model survives exactly as a modelled one
3381/// does.
3382struct CarriedExtension {
3383    /// Every block the extension header occupied — chunk 0 and each
3384    /// continuation it named — freed once the replacement is laid out. Empty
3385    /// for a file with no extension, and for one whose extension this session
3386    /// is the first to write. A rewrite re-encodes the whole chain into one
3387    /// chunk, so freeing only the first would leave the rest as space no
3388    /// free-space manager records and no object claims.
3389    superseded: crate::io::object_header_io::HeaderBlocks,
3390    /// Every message that header held — the shared-message table,
3391    /// continuations and null padding excepted. The first two are structure
3392    /// rather than content; the third is free space.
3393    carried: Vec<crate::io::object_header_io::ExtensionMessage>,
3394    /// Where [`Hdf5Writer::write_superblock_extension`] put the replacement,
3395    /// and the only value the superblock's extension address is read from.
3396    /// `None` until that pass runs, and for a file that needs no extension.
3397    addr: Slot<Option<u64>>,
3398}
3399
3400/// What a reopen learns from a file's free-space managers, split by who owns
3401/// it: the sections go to the allocator and the rest stays with the writer.
3402struct ReopenedFreeSpace {
3403    /// `None` for a file this writer records no free space for.
3404    state: Option<Box<FileSpaceState>>,
3405    /// Every section the managers held, each tagged with the manager it came
3406    /// out of and merged only within it, address-ordered. Empty whenever
3407    /// `state` is `None`.
3408    sections: Vec<FreeBlock>,
3409}
3410
3411/// The file-space info message this session is responsible for, and the
3412/// manager blocks it supersedes.
3413///
3414/// A file whose message says `persist` records the space its own edits
3415/// released in one free-space manager per allocation type: a header block
3416/// (`FSHD`) naming a sections block (`FSSE`) that lists every free region.
3417/// Nothing else in the file says those regions are free, so a session that
3418/// rewrites the file without reading them either leaks the space it frees or
3419/// hands out space a manager still claims.
3420///
3421/// Present for a file this writer *created* with non-default file-space
3422/// properties as well, where there is nothing to read and the message is this
3423/// session's to write. `None` — the field, not this struct — is the third
3424/// case: a reopened file whose message this session must not touch, which the
3425/// carried extension re-emits byte for byte.
3426///
3427/// INVARIANT: the sections read are handed to [`FileAllocator`] and tracked
3428/// there alone, so there is one account of the file's free space and not two.
3429/// What stays here is only what the allocator has no place for: the message to
3430/// write, and the managers' own blocks, which are not free space until the
3431/// close that replaces them frees them.
3432struct FileSpaceState {
3433    /// The message, as read or as the creation options declared it. It is the
3434    /// only place the manager addresses are recorded, so the close that moves
3435    /// them rewrites this message.
3436    info: FileSpaceInfoMessage,
3437    /// The manager blocks themselves — one header, and one sections block per
3438    /// manager that had any sections. Freed by the close that lays their
3439    /// replacements out, the rule every other superseded structure follows.
3440    /// Empty for a created file, which supersedes nothing.
3441    superseded: Vec<(u64, u64)>,
3442}
3443
3444impl FileSpaceState {
3445    /// Whether this file keeps free-space managers on disk. Both strategies
3446    /// that have managers do — paged aggregation has the same managers plus a
3447    /// large one — while the two aggregator-only strategies and
3448    /// `persist: false` still carry the message with nothing to write into it.
3449    fn records_free_space(&self) -> bool {
3450        self.info.persist
3451            && matches!(
3452                self.info.strategy,
3453                FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
3454            )
3455    }
3456}
3457
3458/// One free-space manager that has been given its own two blocks, and the
3459/// sections it will write into them.
3460///
3461/// Produced by
3462/// [`settle_free_space_managers`](Hdf5Writer::settle_free_space_managers).
3463/// Both blocks are ordinary allocations out of the same [`FileAllocator`] the
3464/// rest of the file uses, because upstream's are too:
3465/// `H5FS_vfd_alloc_hdr_and_section_info_if_needed` calls `H5MF_alloc`
3466/// (H5FSsection.c:2352, 2406).
3467struct PlacedManager {
3468    /// Which of the file's managers this is; its message slot names it in the
3469    /// file-space info message.
3470    manager: FreeSpaceManager,
3471    /// Header block address.
3472    hdr_addr: u64,
3473    /// Sections block address.
3474    sect_addr: u64,
3475    /// Bytes the sections block occupies. What the header records as both
3476    /// `sect_size` and `alloc_sect_size`, so an image shorter than the block
3477    /// is padded rather than reported short.
3478    sect_size: u64,
3479    /// The sections this manager records, in serialization order. Filled on
3480    /// the settling round, once no allocation can change them.
3481    sections: Vec<FreeSection>,
3482}
3483
3484/// The manager header for `sections`, before its own blocks have addresses.
3485///
3486/// Every width the section encoding uses comes from here, and the only one
3487/// that varies with the content is `serial_sections` — it decides how many
3488/// bytes a per-size run count takes — so sizing a layout and encoding it must
3489/// go through this one function or the two disagree.
3490fn manager_header(sections: &[FreeSection]) -> FreeSpaceHeader {
3491    FreeSpaceHeader {
3492        client: free_space::CLIENT_FILE,
3493        total_space: sections.iter().map(|s| s.len).sum(),
3494        total_sections: sections.len() as u64,
3495        // Every class the file client registers is serializable; only a
3496        // fractal heap's manager has ghost sections.
3497        serial_sections: sections.len() as u64,
3498        ghost_sections: 0,
3499        nclasses: free_space::FILE_SECT_CLASSES,
3500        shrink_percent: free_space::SHRINK_PERCENT,
3501        expand_percent: free_space::EXPAND_PERCENT,
3502        max_sect_addr: free_space::SEC2_MAX_SECT_ADDR,
3503        max_sect_size: free_space::SEC2_MAXADDR,
3504        sect_addr: UNDEF_ADDR,
3505        sect_size: 0,
3506        alloc_sect_size: 0,
3507    }
3508}
3509
3510impl Default for CarriedExtension {
3511    /// What a file with no extension carries: nothing to free, nothing to
3512    /// re-emit, and no address until a shared-message table gives it one.
3513    fn default() -> Self {
3514        Self {
3515            superseded: Vec::new(),
3516            carried: Vec::new(),
3517            addr: Slot::new(None),
3518        }
3519    }
3520}
3521
3522/// Where a file's superblock version comes from — the two cases libhdf5 keeps
3523/// strictly apart, and this writer's single source for both the version it
3524/// writes back and the generation it writes new structures in.
3525///
3526/// INVARIANT: reopening a file never changes its superblock version, and every
3527/// structure appended to it is written at a library-version bound of at least
3528/// the row that version belongs to.
3529///
3530/// libhdf5 splits the same way. `H5F__super_init` is the only place a version
3531/// is *decided* — content first, then `MAX(super_vers,
3532/// HDF5_superblock_ver_bounds[low_bound])` (H5Fsuper.c:1128-1154).
3533/// `H5F__super_read` never recomputes one; it validates what it read and
3534/// raises the file's low bound to match, version 2 to at least
3535/// `H5F_LIBVER_V18` and version 3 to at least `H5F_LIBVER_V110`
3536/// (hdf5_1.14.6 H5Fsuper.c:460-466). One direction only: the version bounds
3537/// the structures, the structures never bound the version back.
3538///
3539/// Two variants rather than one number with a rule attached, because the
3540/// number means different things on the two paths — a floor to raise on the
3541/// create path, a fixed value on the reopen path — and a single field would
3542/// have every reader re-derive which.
3543#[derive(Debug, Clone, Copy)]
3544enum SuperblockVersion {
3545    /// A file this writer created. The version its creation options start
3546    /// from, which [`superblock_version_for`](Hdf5Writer::superblock_version_for)
3547    /// raises to what the content and the named bound need. Nothing is on
3548    /// disk yet, so nothing floors the bound.
3549    Chosen(u8),
3550    /// A file this writer reopened: the version already in the file. Written
3551    /// back unchanged, and the floor under every bound this session writes at.
3552    Existing(u8),
3553}
3554
3555impl SuperblockVersion {
3556    /// The oldest library-version bound this file may be written at.
3557    ///
3558    /// `H5F__super_read`'s upgrade, as a table rather than two `if`s: the
3559    /// oldest row of `HDF5_superblock_ver_bounds` (H5Fsuper.c:68) whose entry
3560    /// is the version on disk. A created file has no superblock on disk, so
3561    /// its floor is the oldest bound there is.
3562    ///
3563    /// `Existing(0..=1)` and `Hdf5Writer::legacy` say the same thing from two
3564    /// directions and cannot disagree: `open_append_with_locking` builds the
3565    /// `LegacyFile` from exactly the versions this arm covers.
3566    fn libver_floor(self) -> LibverBound {
3567        match self {
3568            Self::Chosen(_) => LibverBound::Earliest,
3569            Self::Existing(0..=1) => LibverBound::Earliest,
3570            Self::Existing(2) => LibverBound::V18,
3571            Self::Existing(_) => LibverBound::V110,
3572        }
3573    }
3574}
3575
3576/// A registry entry that has held some name.
3577///
3578/// Datasets, groups and committed datatypes keep stable indices — their
3579/// registries only grow, deletion being a flag — so the index can name the
3580/// exact entry. The link registries shrink as links are unlinked, and a
3581/// link's path is derived from its parent group's current name, so for those
3582/// the index records only that the kind once claimed the name and the (short)
3583/// list itself answers.
3584#[derive(Clone, Copy, PartialEq, Eq)]
3585enum NameHit {
3586    Dataset(usize),
3587    Group(usize),
3588    Datatype(usize),
3589    HardLink,
3590    SymbolicLink,
3591    PreservedLink,
3592}
3593
3594/// Which names the file model already holds, so creating an object does not
3595/// have to walk every registry to find out.
3596///
3597/// INVARIANT: while `map` is `Some`, every name a registry entry currently
3598/// holds has an entry in `map` covering that entry. The converse is not
3599/// required: a hit whose object was since deleted, or whose name has since
3600/// changed, stays in the map and is filtered out by
3601/// [`Hdf5Writer::name_holder`], which re-runs the very predicates the linear
3602/// scan used. The index may therefore answer "maybe", never "free" for a name
3603/// that is taken.
3604///
3605/// MUST NOT: no code may give a registry entry a name, or move the path a
3606/// link is emitted under, without either registering the new name through
3607/// [`Hdf5Writer::register_name`] or dropping the index through
3608/// [`Hdf5Writer::forget_name_index`]. State a constructor puts straight into
3609/// the registries needs neither — `map` starts `None`, and the first query
3610/// builds it from the registries as they then stand.
3611struct NameIndex {
3612    map: Option<HashMap<String, Vec<NameHit>>>,
3613    /// Bumped whenever the registries move under a build in flight, so that
3614    /// build's result is discarded instead of being installed stale.
3615    epoch: u64,
3616}
3617
3618impl NameIndex {
3619    fn new() -> Self {
3620        NameIndex {
3621            map: None,
3622            epoch: 0,
3623        }
3624    }
3625
3626    /// Record that `hit` holds `name`. With no map built there is nothing to
3627    /// record, but the registries have moved, so any build in flight is
3628    /// invalidated rather than trusted.
3629    fn insert(&mut self, name: &str, hit: NameHit) {
3630        match self.map.as_mut() {
3631            None => self.epoch += 1,
3632            Some(map) => {
3633                let hits = map.entry(name.to_string()).or_default();
3634                if !hits.contains(&hit) {
3635                    hits.push(hit);
3636                }
3637            }
3638        }
3639    }
3640
3641    /// Throw the index away: the next query rebuilds it from the registries.
3642    fn forget(&mut self) {
3643        self.map = None;
3644        self.epoch += 1;
3645    }
3646}
3647
3648/// HDF5 file writer.
3649///
3650/// Usage:
3651/// 1. `Hdf5Writer::create(path)` to create a new file.
3652/// 2. `create_dataset(name, datatype, dims)` to define datasets.
3653/// 3. `write_dataset_raw(index, data)` to write raw data.
3654/// 4. `close()` to finalize the file (writes superblock, headers, etc.).
3655pub struct Hdf5Writer {
3656    handle: FileHandle,
3657    allocator: FileAllocator,
3658    ctx: FormatContext,
3659    /// Dataset registry. The outer [`Slot`] guards the spine (push on create,
3660    /// index/clone on access) and is held only briefly; each [`DatasetRef`]
3661    /// carries one dataset's metadata behind its own lock. A writer clones
3662    /// the `DatasetRef` out (releasing this lock) before doing the long
3663    /// per-dataset work, so a create never blocks an in-flight write.
3664    pub(crate) datasets: Slot<Vec<DatasetRef>>,
3665    /// Group registry, same shape as [`Self::datasets`].
3666    pub(crate) groups: Slot<Vec<GroupRef>>,
3667    /// User-created hard links (additional names for existing objects),
3668    /// resolved and emitted during finalize.
3669    pub(crate) hard_links: Slot<Vec<HardLink>>,
3670    /// User-created soft and external links. Held apart from
3671    /// [`Self::hard_links`] because they name a path rather than an object:
3672    /// nothing resolves them, and no object's reference count counts them.
3673    pub(crate) symbolic_links: Slot<Vec<SymbolicLink>>,
3674    /// Datatypes committed this session, each an object of its own; see
3675    /// [`CommittedDatatype`].
3676    pub(crate) committed_datatypes: Slot<Vec<CommittedDatatype>>,
3677    /// Links a reopened file held that this writer cannot express, carried
3678    /// through every header rewrite by their encoded bytes. Always empty for
3679    /// a freshly created file; see [`PreservedLink`].
3680    pub(crate) preserved_links: Slot<Vec<PreservedLink>>,
3681    /// Which names the registries above already hold; see [`NameIndex`].
3682    /// Boxed so this side table costs the writer one pointer: inline, its
3683    /// map shifted every field after it and cost the attribute path ~5%.
3684    name_index: Slot<Box<NameIndex>>,
3685    /// Attributes attached to the root group (file-level attributes).
3686    pub(crate) root_attributes: Slot<Vec<crate::format::messages::attribute::AttributeEntry>>,
3687    /// Serializes object creation so name-uniqueness check and registry insert
3688    /// happen atomically.
3689    ///
3690    /// INVARIANT: no two emitted links share a full-path name. Under
3691    /// `threadsafe`, create methods run on the shared read guard, so without
3692    /// this gate two threads could both pass the duplicate-name check (which
3693    /// snapshots a registry and drops its lock) and both push, writing an
3694    /// invalid HDF5 file with two same-named links. A create holds this lock
3695    /// across its check *and* its push; the streaming write path never takes
3696    /// it, so writes to existing datasets stay fully concurrent. It is the
3697    /// outermost lock a create acquires (create_lock → spine → slot), and no
3698    /// write path takes it, so it cannot deadlock with the registry locks.
3699    pub(crate) create_lock: Slot<()>,
3700    /// The low `H5Pset_libver_bounds` bound the *caller named*, or `None`
3701    /// when none was: the oldest libhdf5 a file this writer creates must stay
3702    /// readable by. It is the one switch the version-bearing messages read —
3703    /// the datatype message version (`H5O_dtype_ver_bounds`), the data layout
3704    /// message version (`H5O_layout_ver_bounds`) and with it the chunk index,
3705    /// and the superblock floor (`HDF5_superblock_ver_bounds`).
3706    ///
3707    /// `None` is not `Some(Earliest)`. No single libhdf5 bound describes this
3708    /// crate's default file: it takes the earliest row of the datatype and
3709    /// superblock tables (version-1 datatypes, a version-2 superblock raised
3710    /// to 3 only by what the content needs) over the v1.10 chunk indexes,
3711    /// which is the `H5F_LIBVER_V110` row of the layout table. Naming a bound
3712    /// asks for one whole libhdf5 generation instead, so the two cannot share
3713    /// a field.
3714    ///
3715    /// Nothing reads this directly:
3716    /// [`session_libver`](Hdf5Writer::session_libver) is the only reader, and
3717    /// it is where the default meets the floor the file's own superblock puts
3718    /// under it (see [`SuperblockVersion`]). A default is a bound the *writer*
3719    /// picks, and on a reopened file the writer has no say — which is exactly
3720    /// the difference this field cannot express on its own.
3721    libver: Option<LibverBound>,
3722    closed: bool,
3723    /// Set once `finalize_for_swmr` has published a readable file.
3724    ///
3725    /// A SWMR reader may hold a chunk index that still points at a block this
3726    /// writer has since replaced, so from that point on a relocated chunk's
3727    /// old block is kept rather than released for reuse — the same rule as
3728    /// libhdf5's `H5D__chunk_file_alloc`, which skips `H5MF_xfree` under
3729    /// `H5F_ACC_SWMR_WRITE`.
3730    swmr_active: bool,
3731    /// Collections with free space — libhdf5's `f->shared->cwfs` list. A
3732    /// vlen insert fills these partially-filled collection blocks before
3733    /// creating a new one, so many small writes share 4096-byte blocks
3734    /// instead of each taking their own. Entries hold `(addr, block size,
3735    /// free bytes)` hints; the block on disk stays the single truth for
3736    /// contents, and only the two functions that rewrite collection blocks
3737    /// ([`insert_vlen_objects`](Self::insert_vlen_objects) and
3738    /// [`release_vlen_references`](Self::release_vlen_references)) may
3739    /// update this list. In-memory only, like the allocator's free list:
3740    /// a reopened file's free space is rediscovered as releases touch its
3741    /// collections. Capped at [`H5HG_NCWFS`] entries.
3742    cwfs: Slot<Vec<CwfsEntry>>,
3743    /// Address of the root group object header (set after first finalize).
3744    root_group_addr: Option<u64>,
3745    /// Size of the encoded root group object header (for in-place rewrites).
3746    root_group_encoded_size: usize,
3747    /// The on-disk root header block a reopen found, `(addr, len)`, so
3748    /// finalize can free the block its rewrite supersedes.
3749    superseded_root_header: crate::io::object_header_io::HeaderBlocks,
3750    /// Where this file's superblock version comes from. The single owner of
3751    /// both halves of the reopen invariant — see [`SuperblockVersion`],
3752    /// [`superblock_version_for`](Self::superblock_version_for) and
3753    /// [`libver_floor`](Self::libver_floor).
3754    superblock_version: SuperblockVersion,
3755    /// Objects whose attributes this finalize spilled to dense storage, and
3756    /// the `Attribute Info` message naming what was written for each.
3757    ///
3758    /// INVARIANT: an entry exists here only after every block of that
3759    /// object's heap and name index is on disk, and only
3760    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes) may add
3761    /// one. `emit_attributes` reads it and never builds — a header is sized
3762    /// and then written by two separate `build_*_header` calls, so a build
3763    /// that allocated would allocate twice and leave the sized-for blocks
3764    /// stranded.
3765    dense_attributes: Slot<HashMap<AttrScope, AttributeInfoMessage>>,
3766    /// Groups whose links this finalize spilled to dense storage, and the
3767    /// `Link Info` message naming what was written for each.
3768    ///
3769    /// INVARIANT: an entry exists here only after every block of that group's
3770    /// heap and name index is on disk, and only
3771    /// [`prepare_dense_links`](Self::prepare_dense_links) may add one.
3772    dense_links: Slot<HashMap<LinkScope, LinkInfoMessage>>,
3773    /// The dense storage the reopened object headers already name — the heaps
3774    /// and indices this session's rewrites and deletes supersede.
3775    ///
3776    /// `None` for a file this session created: every block such a file will
3777    /// hold was allocated here, so there is nothing on disk to supersede and
3778    /// nothing to allocate for the bookkeeping either.
3779    ///
3780    /// INVARIANT: every entry is freed exactly once, by
3781    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs)
3782    /// or [`release_superseded_dense_links`](Self::release_superseded_dense_links),
3783    /// which remove it as they free. Nothing else may remove one: an entry
3784    /// that leaves without reaching the allocator is a leaked heap, and one
3785    /// that reaches it twice hands the same blocks to two objects.
3786    superseded_dense: Slot<Option<Box<SupersededDense>>>,
3787    /// The creation-order policy in force: whether an object created from
3788    /// now on records creation order for its links and its attributes. The
3789    /// h5py `track_order` analogue; see
3790    /// [`set_track_order`](Self::set_track_order). Each object captures this
3791    /// at creation, so changing it never rewrites an object already made.
3792    track_order: TrackOrder,
3793    /// Whether an object created from now on records the times its header can
3794    /// hold — `H5Pset_obj_track_times`, whose default is on
3795    /// (`H5O_CRT_OHDR_FLAGS_DEF` is `H5O_HDR_STORE_TIMES`, H5Opkg.h:74).
3796    /// Captured by each object at creation for the same reason
3797    /// [`track_order`](Self::track_order) is: it belongs to the creation
3798    /// property list, so a later change must not rewrite an object already
3799    /// made.
3800    track_times: bool,
3801    /// The root group's own captured policy. The root is created with the
3802    /// file, so its value comes from
3803    /// [`create_with_options`](Self::create_with_options) — or, on reopen,
3804    /// from the header already on disk.
3805    root_track_order: TrackOrder,
3806    /// The root group's stored times, on the same terms as
3807    /// [`GroupInfo::times`]: whatever a reopened file's root header had, and
3808    /// `None` for a file this writer created.
3809    root_times: Option<ObjectTimes>,
3810    /// Hands out the creation sequence numbers that order a group's links.
3811    next_creation_seq: Slot<u64>,
3812    /// Object-reference elements waiting for their target's object header
3813    /// address, which only exists once finalize has placed every header.
3814    pending_object_references: Slot<Vec<PendingObjectReference>>,
3815    /// Heap-backed reference objects waiting for the same address — the
3816    /// pre-1.12 region form and every 1.12 form whose element is a blob id.
3817    pending_heap_references: Slot<Vec<PendingHeapReference>>,
3818    /// What each object-reference attribute's value *means*, so
3819    /// [`object_attributes`](Hdf5Writer::object_attributes) can say it in
3820    /// addresses every time an object header is built.
3821    attribute_references: Slot<Vec<AttributeReferenceValue>>,
3822    /// Set when this file is in the classic (version-0/1 superblock) format,
3823    /// whether it was reopened in it or created at `H5F_LIBVER_EARLIEST`.
3824    /// See [`LegacyFile`]; [`is_legacy`](Self::is_legacy) is the only reader
3825    /// of whether it is there.
3826    legacy: Option<Box<LegacyFile>>,
3827    /// Which groups keep their links in a symbol table, and the storage each
3828    /// of them has. Empty for a file whose groups all store links in messages;
3829    /// see [`SymbolTables`], which owns the question.
3830    symbol_tables: SymbolTables,
3831    /// The v1 B-tree "K" ranks every node width in this file is derived from,
3832    /// after the superblock extension has had its say. A property of the file
3833    /// rather than of its generation: a version-2 superblock records no ranks
3834    /// of its own but its extension may, and a rewrite that used the library
3835    /// defaults there would write nodes of the wrong width.
3836    /// [`btree_v1_config`](Hdf5Writer::btree_v1_config) is the only reader.
3837    btree: BTreeV1Config,
3838    /// The superblock extension this file carries, and where the replacement
3839    /// went; see [`CarriedExtension`].
3840    extension: Box<CarriedExtension>,
3841    /// The free-space managers a reopened `persist: true` file carries; see
3842    /// [`FileSpaceState`]. `None` for every other file — one with no
3843    /// file-space info message, one that does not persist, one under paged
3844    /// aggregation, and every file this session created — and those files get
3845    /// no free-space manager written either.
3846    free_space: Option<Box<FileSpaceState>>,
3847    /// The file's shared-message indexes, when it was created with any.
3848    /// `None` — the default — is a file with no shared-message table, where
3849    /// [`share_message`](Self::share_message) is the identity.
3850    sohm: Option<Box<SohmState>>,
3851    /// The directory holding this HDF5 file, resolved once when it was opened
3852    /// — libhdf5's `H5F_t::extpath`, and the same value the read side keeps.
3853    /// External raw-data file names are joined against it when
3854    /// `HDF5_EXTFILE_PREFIX` names `${ORIGIN}`, so a write and a later read of
3855    /// the same dataset must resolve a relative name identically; capturing it
3856    /// at open time rather than reading the process's current directory per
3857    /// write is what makes that hold.
3858    source_dir: PathBuf,
3859}
3860
3861/// A file's shared object header messages, from creation to the table on disk.
3862///
3863/// INVARIANT: a message body reaches the file either literally or as a pointer
3864/// to exactly one heap object, never both, and the reference count of that
3865/// object is the number of headers that hold the pointer.
3866/// [`share_message`](Hdf5Writer::share_message) is the only place a body is
3867/// offered to an index, and
3868/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) is the
3869/// only place the phase changes — so counting and substituting are two passes
3870/// over the same call site rather than two pieces of logic that must agree.
3871struct SohmState {
3872    /// The indexes the file was created with, in table order.
3873    indexes: Vec<SohmIndexSpec>,
3874    /// What `share_message` does to an eligible message right now.
3875    phase: Slot<SohmPhase>,
3876    /// Address of the master table this session laid out, once it has one.
3877    /// Also the once-only latch on the layout: a second finalize keeps the
3878    /// table the first one published, and
3879    /// [`Hdf5Writer::write_superblock_extension`] reads it to name that table
3880    /// in the extension.
3881    table_addr: Slot<Option<u64>>,
3882    /// The blocks the table a reopen found occupies — the master table and
3883    /// each index's heap and index structure — taken by the finalize that
3884    /// replaces them. Empty for a file this session created.
3885    ///
3886    /// The table is laid out whole from the whole message set, so a reopen
3887    /// replaces it rather than inserting into it, and every header holding a
3888    /// pointer into the old one is rewritten in the same finalize
3889    /// ([`Hdf5Writer::rebuilds_shared_messages`]).
3890    superseded: Slot<Vec<(u64, u64)>>,
3891}
3892
3893/// The passes `share_message` runs in, and the state between them.
3894enum SohmPhase {
3895    /// Outside a finalize: every message stays literal.
3896    Idle,
3897    /// Measuring headers, before the bodies they will hold are final. A
3898    /// shareable message answers at the width of a heap pointer over a heap
3899    /// object that does not exist yet, which is the width the one it ends up
3900    /// pointing at has: a `H5O_shared_t` in heap form is the same size
3901    /// whatever it names. Nothing this pass produces is written — it exists so
3902    /// [`allocate_object_headers`](Hdf5Writer::allocate_object_headers) can
3903    /// reserve a block for a header whose messages are shared before the
3904    /// content phase has decided which heap object each one shares.
3905    ///
3906    /// The set is [`FirstCopies`], and it is why this pass has state at all:
3907    /// a message left literal is *wider* than a pointer, so a header can only
3908    /// be measured by making the same first-copy decision the substituting
3909    /// pass will make.
3910    Predict(FirstCopies),
3911    /// Counting the bodies the file will share. Messages still go in
3912    /// literally, so nothing this pass builds is written.
3913    Collect(SohmCollector),
3914    /// Substituting. A body the collect pass never saw stays literal, which
3915    /// is a valid file: the record it would have shared simply keeps a
3916    /// reference count one higher than the pointers that reach it.
3917    Resolve {
3918        /// Heap ID per body, from the table this finalize laid out.
3919        ids: HashMap<(u8, Vec<u8>), [u8; SOHM_HEAP_ID_LEN]>,
3920        /// The first copies this pass has already handed out; see
3921        /// [`FirstCopies`].
3922        first: FirstCopies,
3923    },
3924}
3925
3926/// The bodies a pass has already left literal in the header that offered them
3927/// first (`H5SM_IN_OH`, H5SM.c:1400-1417).
3928///
3929/// INVARIANT: the three passes walk the same object headers in the same order
3930/// — [`allocate_object_headers`](Hdf5Writer::allocate_object_headers),
3931/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) and
3932/// [`write_object_headers`](Hdf5Writer::write_object_headers) each build every
3933/// dataset in `datasets` order, then every group, then the root — so "the
3934/// header that offered this body first" is the same header in all three. Each
3935/// pass keeps its own set rather than sharing one, so a pass that does not run
3936/// cannot leave a stale decision behind for the next one. A divergence would
3937/// make a header wider than the block reserved for it, which
3938/// [`check_header_size`] refuses rather than writing.
3939type FirstCopies = std::collections::HashSet<(u8, Vec<u8>)>;
3940
3941/// The object header a message is being written into — `H5SM_try_share`'s
3942/// `open_oh` argument, which is what decides whether a first copy has a header
3943/// to stay literal in at all.
3944#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3945enum ShareOwner {
3946    /// `H5SM_try_share(f, NULL, ...)`: the message belongs to no object header
3947    /// of its own. An attribute's datatype and dataspace are offered this way
3948    /// (H5Aint.c:375-377) — they live inside the attribute's body, so there is
3949    /// no header message for a record to name and the body goes to the heap on
3950    /// first use however shareable its class is.
3951    Detached,
3952    /// `H5SM_try_share(f, oh, ...)`: the message is a message of the object
3953    /// header at this address (`H5O__msg_alloc`, H5Omessage.c:1735).
3954    Header(u64),
3955}
3956
3957impl SohmState {
3958    /// A file's indexes, plus the blocks of the table they were read out of
3959    /// when the file was reopened (empty when it was created this session).
3960    fn new(indexes: Vec<SohmIndexSpec>, superseded: Vec<(u64, u64)>) -> Self {
3961        Self {
3962            indexes,
3963            phase: Slot::new(SohmPhase::Idle),
3964            table_addr: Slot::new(None),
3965            superseded: Slot::new(superseded),
3966        }
3967    }
3968
3969    /// The index that would take a `msg_type` message of `body_len` bytes,
3970    /// as `H5SM_try_share` resolves one: the first index whose type mask
3971    /// covers the class, and then only if the message reaches that index's
3972    /// minimum. A message too small for its index is not offered to another —
3973    /// `H5SM__get_index` picks by type alone and the size check comes after.
3974    fn index_for(&self, msg_type: u8, body_len: usize) -> Option<usize> {
3975        let flag = type_flag(msg_type)?;
3976        let (at, spec) = self
3977            .indexes
3978            .iter()
3979            .enumerate()
3980            .find(|(_, spec)| spec.mesg_types & flag != 0)?;
3981        (body_len as u64 >= u64::from(spec.min_mesg_size)).then_some(at)
3982    }
3983
3984    /// Whether any index takes attribute messages, which is what makes the
3985    /// file record message creation indices — `H5SM_init` sets
3986    /// `store_msg_crt_idx` on exactly this condition (H5SM.c:220).
3987    fn shares_attributes(&self) -> bool {
3988        let Some(flag) = type_flag(MSG_ATTRIBUTE) else {
3989            return false;
3990        };
3991        self.indexes.iter().any(|spec| spec.mesg_types & flag != 0)
3992    }
3993}
3994
3995/// What decides whether two offers are the same shared message: the class,
3996/// the bytes, and the messages the bytes will end up pointing at.
3997type CollectedKey = (u8, Vec<u8>, Vec<NestedShare>);
3998
3999/// The shareable message bodies of one collect pass, in first-seen order.
4000struct SohmCollector {
4001    /// Per index, its bodies with the number of headers holding each.
4002    messages: Vec<Vec<SharedMessage>>,
4003    /// Where a body sits: `(index, position in that index's messages)`, keyed
4004    /// by everything that decides what will be stored — the class, the bytes,
4005    /// and the messages the bytes will end up pointing at.
4006    seen: HashMap<CollectedKey, (usize, usize)>,
4007}
4008
4009impl SohmCollector {
4010    fn new(nindexes: usize) -> Self {
4011        Self {
4012            messages: vec![Vec::new(); nindexes],
4013            seen: HashMap::new(),
4014        }
4015    }
4016
4017    /// Count one message against `index`, adding the body the first time it
4018    /// is seen, and say whether that body is new.
4019    ///
4020    /// `ohdr` is the header this offer would leave the body literal in when it
4021    /// is the first — `None` when the class cannot be shared in an object
4022    /// header or the offer names none. It is recorded only for a first copy:
4023    /// once a body is in the heap, later offers of it are pointers whatever
4024    /// header they come from.
4025    ///
4026    /// Two bodies are the same message only if their nesting agrees as well:
4027    /// the heap IDs a nesting body will hold are still zero here, so two
4028    /// attributes that differ only in their datatype are the same bytes at
4029    /// this point and different bytes on disk.
4030    fn record(
4031        &mut self,
4032        index: usize,
4033        msg_type: u8,
4034        body: &[u8],
4035        nested: &[NestedShare],
4036        ohdr: Option<u64>,
4037    ) -> bool {
4038        let key = (msg_type, body.to_vec(), nested.to_vec());
4039        match self.seen.get(&key) {
4040            Some(&(at, pos)) => {
4041                self.messages[at][pos].ref_count += 1;
4042                false
4043            }
4044            None => {
4045                let pos = self.messages[index].len();
4046                self.messages[index].push(SharedMessage {
4047                    msg_type,
4048                    body: body.to_vec(),
4049                    nested: nested.to_vec(),
4050                    ref_count: 1,
4051                    ohdr_addr: ohdr,
4052                });
4053                self.seen.insert(key, (index, pos));
4054                true
4055            }
4056        }
4057    }
4058
4059    /// Give back the reference [`record`](Self::record) took for a body whose
4060    /// container turned out to be a copy of one already here.
4061    ///
4062    /// A body reached only through a shared container is referenced once per
4063    /// container *record*, not once per object that has one: the pointer to
4064    /// it lives in the container's heap object, which exists once however
4065    /// many headers name it. `H5O__attr_create` reaches the same count from
4066    /// the other side, by building each attribute's components shared and
4067    /// then calling `H5O__attr_delete` — which decrements exactly the
4068    /// datatype and dataspace (H5Oattr.c:568-585) — whenever the attribute it
4069    /// built was not the first copy (H5Oattribute.c:331-366).
4070    fn release(&mut self, msg_type: u8, body: &[u8]) {
4071        if let Some(&(at, pos)) = self.seen.get(&(msg_type, body.to_vec(), Vec::new())) {
4072            let count = &mut self.messages[at][pos].ref_count;
4073            *count = count.saturating_sub(1);
4074        }
4075    }
4076}
4077
4078/// The file-creation properties a brand-new file is made with.
4079///
4080/// libhdf5 splits these across the file creation and file access property
4081/// lists (`H5Pset_userblock`, `H5Pset_link_creation_order`,
4082/// `H5Pset_libver_bounds`, the locking property); what they have in common is
4083/// that they are read once, when the file is created, and cannot be changed
4084/// afterwards without rewriting it. Options that *can* change mid-session —
4085/// the bound for objects created later, the creation-order policy for later
4086/// objects — have their own setters.
4087#[derive(Debug, Clone, Copy, Default)]
4088pub struct FileCreateOptions {
4089    /// OS-level locking policy for the new file.
4090    pub locking: crate::io::locking::FileLocking,
4091    /// Creation-order policy for the root group, and the default for every
4092    /// object created afterwards; see [`Hdf5Writer::set_track_order`].
4093    pub track_order: bool,
4094    /// Time-tracking policy for the root group, and the default for every
4095    /// object created afterwards; see [`Hdf5Writer::set_track_times`].
4096    pub track_times: bool,
4097    /// The file's low library-version bound (`H5Pset_libver_bounds`'s `low`),
4098    /// or `None` when the caller named none.
4099    ///
4100    /// The distinction is not decoration. `Some(LibverBound::Earliest)` is a
4101    /// request for the format libhdf5 writes at `H5F_LIBVER_EARLIEST` — a
4102    /// version-0 superblock over symbol-table groups and version-1 object
4103    /// headers, which is what [`ObjectFormat::Legacy`] encodes. `None` keeps
4104    /// what this crate has always written for a file whose creator said
4105    /// nothing: the version-2 superblock and link-message groups of the v1.8
4106    /// format, with the earliest bound's message versions where they can
4107    /// express the content. That combination is this crate's own, not one
4108    /// libhdf5 writes, so it cannot be spelled as a bound.
4109    pub libver: Option<LibverBound>,
4110    /// Bytes reserved in front of the superblock for the application's own
4111    /// use (`H5Pset_userblock`). Zero, the default, places the superblock at
4112    /// offset 0; otherwise a power of two of at least
4113    /// [`MIN_USERBLOCK`] bytes, since a reader finds the
4114    /// superblock by doubling its search offset from there.
4115    pub userblock: u64,
4116    /// Shared object header message indexes; see [`SharedMessageConfig`].
4117    pub shared_messages: SharedMessageConfig,
4118    /// How the file manages its own space; see [`FileSpaceConfig`].
4119    pub file_space: FileSpaceConfig,
4120}
4121
4122/// The file-space handling properties a new file is created with — the three
4123/// arguments of `H5Pset_file_space_strategy` and the one of
4124/// `H5Pset_file_space_page_size`.
4125///
4126/// The four together are what `H5F__super_init` compares against the library
4127/// defaults to decide whether the file needs a file-space info message at all
4128/// (H5Fsuper.c:1092-1097), which is why the page size belongs here even though
4129/// only paged aggregation allocates by it: a file that names a page size and
4130/// nothing else still carries the message.
4131#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4132pub struct FileSpaceConfig {
4133    /// `H5F_fspace_strategy_t`.
4134    pub strategy: FileSpaceStrategy,
4135    /// Whether the free-space managers are written to the file on close.
4136    pub persist: bool,
4137    /// The smallest section a manager records; a block freed below it is
4138    /// space the file leaks rather than tracks.
4139    pub threshold: u64,
4140    /// `H5Pset_file_space_page_size`: the file-space page every allocation of
4141    /// a paged file is shaped by, and the value the message carries whatever
4142    /// the strategy.
4143    pub page_size: u64,
4144}
4145
4146impl Default for FileSpaceConfig {
4147    /// `H5F_FILE_SPACE_STRATEGY_DEF`, `H5F_FREE_SPACE_PERSIST_DEF`,
4148    /// `H5F_FREE_SPACE_THRESHOLD_DEF` and `H5F_FILE_SPACE_PAGE_SIZE_DEF`
4149    /// (H5Fprivate.h:326-336).
4150    fn default() -> Self {
4151        Self {
4152            strategy: FileSpaceStrategy::FsmAggr,
4153            persist: false,
4154            threshold: 1,
4155            page_size: DEFAULT_FILE_SPACE_PAGE_SIZE,
4156        }
4157    }
4158}
4159
4160impl FileSpaceConfig {
4161    /// The properties as `H5P__set_file_space_strategy` (H5Pfcpl.c:1176)
4162    /// stores them: `persist` and `threshold` are set only for the two
4163    /// strategies that have free-space managers to persist, and keep their
4164    /// defaults for the two that do not.
4165    pub fn new(strategy: FileSpaceStrategy, persist: bool, threshold: u64) -> Self {
4166        let uses_managers = matches!(
4167            strategy,
4168            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
4169        );
4170        Self {
4171            strategy,
4172            persist: uses_managers && persist,
4173            threshold: if uses_managers {
4174                threshold
4175            } else {
4176                Self::default().threshold
4177            },
4178            ..Self::default()
4179        }
4180    }
4181
4182    /// `H5Pset_file_space_page_size`, the fourth file-space property and the
4183    /// one libhdf5 sets on its own call.
4184    ///
4185    /// Independent of the strategy, as upstream is: the value reaches the
4186    /// file-space info message whatever the strategy is, and only paged
4187    /// aggregation allocates by it. Out-of-range sizes are refused where the
4188    /// file is created ([`validate`](Self::validate)) rather than here, so a
4189    /// builder chain stays a builder chain.
4190    pub fn with_page_size(mut self, page_size: u64) -> Self {
4191        self.page_size = page_size;
4192        self
4193    }
4194
4195    /// Whether the file has to say any of this on disk. `H5F__super_init`
4196    /// writes the file-space info message only for a file that differs from
4197    /// the library defaults in one of the four properties (H5Fsuper.c:1092),
4198    /// and raises such a file's superblock to version 2 so it has an
4199    /// extension to write it into (H5Fsuper.c:1144).
4200    pub fn is_default(&self) -> bool {
4201        *self == Self::default()
4202    }
4203
4204    /// Refuse what this writer cannot make. `H5Pset_file_space_strategy`
4205    /// itself only refuses a strategy outside the enum (H5Pfcpl.c:1223), and
4206    /// `H5Pset_file_space_page_size` a page size outside `[512, 1 GiB]`
4207    /// (H5Pfcpl.c:1389-1393) — no power of two required, only the bounds.
4208    fn validate(&self) -> IoResult<()> {
4209        if !(PAGE_SIZE_MIN..=PAGE_SIZE_MAX).contains(&self.page_size) {
4210            return Err(crate::io::IoError::InvalidState(format!(
4211                "a file-space page size is between {PAGE_SIZE_MIN} bytes and \
4212                 {PAGE_SIZE_MAX}, not {}",
4213                self.page_size
4214            )));
4215        }
4216        match self.strategy {
4217            FileSpaceStrategy::FsmAggr
4218            | FileSpaceStrategy::Aggr
4219            | FileSpaceStrategy::None
4220            | FileSpaceStrategy::Page => Ok(()),
4221            FileSpaceStrategy::Unknown(b) => Err(crate::io::IoError::InvalidState(format!(
4222                "invalid file-space strategy {b}"
4223            ))),
4224        }
4225    }
4226
4227    /// The message a created file carries, before anything is allocated:
4228    /// every manager address undefined and no end-of-allocation recorded,
4229    /// which is what `H5F__super_init` writes (H5Fsuper.c:1369-1382).
4230    fn message(&self) -> FileSpaceInfoMessage {
4231        FileSpaceInfoMessage {
4232            // `H5O_fsinfo_set_version` starts at version 1 and only ever
4233            // raises it, so a created file never carries the version-0 form
4234            // however low its version bounds are.
4235            version: 1,
4236            strategy: self.strategy,
4237            persist: self.persist,
4238            threshold: self.threshold,
4239            page_size: self.page_size,
4240            pgend_meta_thres: 0,
4241            eoa_pre_fsm_fsalloc: UNDEF_ADDR,
4242            fs_addr: vec![UNDEF_ADDR; FS_ADDR_COUNT_V1],
4243        }
4244    }
4245}
4246
4247/// The shared object header message indexes a new file is created with.
4248///
4249/// libhdf5 sets these with three calls on the file creation property list:
4250/// `H5Pset_shared_mesg_nindexes` fixes how many indexes there are,
4251/// `H5Pset_shared_mesg_index` gives each one the message types it covers and
4252/// the smallest message it will take, and `H5Pset_shared_mesg_phase_change`
4253/// sets the list/B-tree thresholds for all of them at once. The default —
4254/// no indexes — is a file with no shared-message table, which is what every
4255/// file this crate wrote before the option existed.
4256#[derive(Debug, Clone, Copy, PartialEq)]
4257pub struct SharedMessageConfig {
4258    /// Indexes in table order; only the first `count` are in use.
4259    indexes: [SohmIndexSpec; MAX_SOHM_INDEXES],
4260    /// How many indexes the caller asked for. Kept even when it is more than
4261    /// the array holds, so file creation can refuse the count the way
4262    /// `H5Pset_shared_mesg_nindexes` does rather than silently drop indexes.
4263    count: usize,
4264}
4265
4266impl Default for SharedMessageConfig {
4267    fn default() -> Self {
4268        Self {
4269            indexes: [SohmIndexSpec {
4270                mesg_types: 0,
4271                min_mesg_size: 0,
4272                list_max: DEFAULT_SOHM_LIST_MAX,
4273                btree_min: DEFAULT_SOHM_BTREE_MIN,
4274            }; MAX_SOHM_INDEXES],
4275            count: 0,
4276        }
4277    }
4278}
4279
4280impl SharedMessageConfig {
4281    /// One index per `(mesg_types, min_mesg_size)` pair — the arguments
4282    /// `H5Pset_shared_mesg_index` takes, where `mesg_types` is the bit mask
4283    /// [`type_flag`](crate::format::sohm::type_flag) builds — with the
4284    /// file-wide phase change `H5Pset_shared_mesg_phase_change` sets: above
4285    /// `list_max` an index is a v2 B-tree, below `btree_min` it is a list
4286    /// again, and `list_max == 0` makes it a B-tree from its first message.
4287    ///
4288    /// Nothing is validated here; [`Hdf5Writer::create_with_options`] refuses
4289    /// a configuration libhdf5 would refuse, so an invalid one is reported
4290    /// where the file is made rather than where the value is typed.
4291    pub fn new(indexes: &[(u16, u32)], list_max: u16, btree_min: u16) -> Self {
4292        let mut config = Self {
4293            count: indexes.len(),
4294            ..Self::default()
4295        };
4296        for (slot, &(mesg_types, min_mesg_size)) in config.indexes.iter_mut().zip(indexes) {
4297            *slot = SohmIndexSpec {
4298                mesg_types,
4299                min_mesg_size,
4300                list_max,
4301                btree_min,
4302            };
4303        }
4304        config
4305    }
4306
4307    /// The indexes in use, in table order.
4308    pub(crate) fn specs(&self) -> &[SohmIndexSpec] {
4309        &self.indexes[..self.count.min(MAX_SOHM_INDEXES)]
4310    }
4311
4312    /// Refuse a configuration `H5Pset_shared_mesg_nindexes` or
4313    /// `H5Pset_shared_mesg_phase_change` would refuse.
4314    fn validate(&self) -> IoResult<()> {
4315        if self.count > MAX_SOHM_INDEXES {
4316            return Err(crate::io::IoError::InvalidState(format!(
4317                "a file may declare at most {MAX_SOHM_INDEXES} shared-message \
4318                 indexes, not {}",
4319                self.count
4320            )));
4321        }
4322        for spec in self.specs() {
4323            // The two thresholds must not overlap, or an index would convert
4324            // back and forth on every insert.
4325            if u32::from(spec.btree_min) > u32::from(spec.list_max) + 1 {
4326                return Err(crate::io::IoError::InvalidState(format!(
4327                    "shared-message phase change needs btree_min ({}) at most one \
4328                     past list_max ({}), or an index converts on every insert",
4329                    spec.btree_min, spec.list_max
4330                )));
4331            }
4332            if spec.mesg_types == 0 {
4333                return Err(crate::io::IoError::InvalidState(
4334                    "a shared-message index covering no message type would never \
4335                     be used; give it a type mask or drop it"
4336                        .into(),
4337                ));
4338            }
4339        }
4340        Ok(())
4341    }
4342}
4343
4344/// One object-reference element written before its value could be known.
4345///
4346/// An `H5R_OBJECT1` element is the target's object header address, and
4347/// addresses are assigned during finalize, so a write records the target by
4348/// path here and [`Hdf5Writer::write_object_reference_values`] puts the address
4349/// down once every header has one.
4350pub(crate) struct PendingObjectReference {
4351    /// Dataset holding the element.
4352    dataset: usize,
4353    /// Element index within that dataset.
4354    element: u64,
4355    /// Path of the object the element names; `/` is the root group.
4356    target: String,
4357}
4358
4359/// One heap-backed reference object written before its target's address could
4360/// be known.
4361///
4362/// The *element* of a `H5R_DATASET_REGION1`, and of every 1.12 reference whose
4363/// encoding does not fit inline, is final at write time — it is the global-heap
4364/// id of the object the write inserted. What waits is the `sizeof_addr` bytes
4365/// of that heap object holding the target's object header address, which
4366/// [`Hdf5Writer::write_heap_reference_values`] stamps in.
4367pub(crate) struct PendingHeapReference {
4368    /// Address of the global-heap collection holding the object.
4369    collection: u64,
4370    /// The object's index within that collection.
4371    index: u16,
4372    /// Where the target's token sits inside that object. The pre-1.12 region
4373    /// form leads with it (`H5R__encode_token_region_compat`); every 1.12 form
4374    /// puts the token's length byte first (`H5R__encode_obj_token`).
4375    token_offset: usize,
4376    /// What the reference names, and how strictly its path must resolve.
4377    target: PendingHeapTarget,
4378}
4379
4380/// What the path of a heap-backed reference must resolve to.
4381///
4382/// The two rules `H5R` applies: a region reference names a *dataset*, since
4383/// `H5Rcreate_region` takes one dataset's dataspace and every reader
4384/// dereferences it as one, while an attribute reference names the attribute's
4385/// owner, which `H5Rcreate_attr` lets be any object.
4386#[derive(Debug, Clone)]
4387pub(crate) enum PendingHeapTarget {
4388    Dataset(String),
4389    Object(String),
4390}
4391
4392/// The value of an attribute whose elements are object references, kept as
4393/// what it means rather than as what it encodes to.
4394///
4395/// An attribute's value is part of its object header message, so it cannot be
4396/// stamped after the fact the way a dataset element can — the header is one
4397/// block, written once. What is stored instead is the paths, and
4398/// [`Hdf5Writer::object_attributes`] turns them into addresses every time the
4399/// attribute set is built: the measuring pass reads the zeros of objects that
4400/// have no address yet, the content pass reads the addresses the file will
4401/// have, and the two agree in length because an address is a fixed-width
4402/// field. The entry in the object's attribute list carries a zero image of
4403/// exactly that length and is never itself written.
4404pub(crate) struct AttributeReferenceValue {
4405    /// The object the attribute hangs on.
4406    scope: AttrScope,
4407    /// The attribute's name within that object.
4408    name: String,
4409    /// Paths of the objects the elements name, in element order; `/` is the
4410    /// root group.
4411    targets: Vec<String>,
4412}
4413
4414/// Refuse an object header body that is not the length its block was reserved
4415/// at.
4416///
4417/// The one check standing behind
4418/// [`HeaderLayout`]'s premise that measuring a header before its content is
4419/// final gives the same length as encoding it after. `what` names the object
4420/// only when the check fails, so the caller pays for the lookup only then.
4421fn check_header_size(
4422    encoded: &[u8],
4423    reserved: usize,
4424    what: impl FnOnce() -> String,
4425) -> IoResult<()> {
4426    if encoded.len() == reserved {
4427        return Ok(());
4428    }
4429    Err(crate::io::IoError::InvalidState(format!(
4430        "the object header of {} encodes to {} bytes but was measured at {}; \
4431         a message in it changed length once the addresses it names were known",
4432        what(),
4433        encoded.len(),
4434        reserved
4435    )))
4436}
4437
4438/// Where every object header a finalize writes will sit, and how long the pass
4439/// that measured it said it is.
4440///
4441/// Produced by [`Hdf5Writer::allocate_object_headers`] and consumed by
4442/// [`Hdf5Writer::write_object_headers`]; between the two, everything a header
4443/// names is built against the addresses it records. The size travels with the
4444/// address because it is what the block was reserved at: the writing pass
4445/// checks its body against it rather than trusting that the two passes agreed.
4446struct HeaderLayout {
4447    /// `(dataset index, address, measured size)`, in write order.
4448    datasets: Vec<(usize, u64, usize)>,
4449    /// `(group index, address, measured size)`, in write order.
4450    groups: Vec<(usize, u64, usize)>,
4451    /// The root group's `(address, measured size)`.
4452    root: (u64, usize),
4453}
4454
4455/// Refuse a region-reference selection the target dataset's extent does not
4456/// admit — libhdf5's `H5S_select_valid`, which `H5Rcreate` applies before it
4457/// serializes anything.
4458///
4459/// The rank check comes from [`Selection::to_boxes`], which also refuses a
4460/// regular hyperslab with an unlimited count or block; a region reference has
4461/// no growable extent to resolve one against.
4462fn validate_region_selection(selection: &Selection, dims: &[u64], path: &str) -> IoResult<()> {
4463    let boxes = selection.to_boxes(dims).map_err(|e| {
4464        crate::io::IoError::InvalidState(format!("region reference over '{path}': {e}"))
4465    })?;
4466    for (start, count) in boxes {
4467        for (d, (&s, &c)) in start.iter().zip(&count).enumerate() {
4468            if s.checked_add(c).is_none_or(|end| end > dims[d]) {
4469                return Err(crate::io::IoError::InvalidState(format!(
4470                    "region reference over '{path}' selects {s}..{} in dimension {d}, \
4471                     outside the dataset's extent of {}",
4472                    s.saturating_add(c),
4473                    dims[d]
4474                )));
4475            }
4476        }
4477    }
4478    Ok(())
4479}
4480
4481/// What a reopen found already on disk in dense form, by the scope whose
4482/// header names it.
4483///
4484/// Both halves together because they are found together — one walk of the
4485/// reopened headers fills both — and released together only in the delete
4486/// path; a finalize supersedes attribute storage before it lays object
4487/// headers out and link storage after, so each half has its own owner.
4488#[derive(Debug, Default)]
4489struct SupersededDense {
4490    attrs: HashMap<AttrScope, AttributeInfoMessage>,
4491    links: HashMap<LinkScope, LinkInfoMessage>,
4492}
4493
4494/// Which object's attribute list a prepared dense layout belongs to.
4495#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4496pub(crate) enum AttrScope {
4497    Root,
4498    Group(usize),
4499    Dataset(usize),
4500}
4501
4502/// Which group's link list a prepared dense layout belongs to.
4503#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4504pub(crate) enum LinkScope {
4505    Root,
4506    Group(usize),
4507}
4508
4509/// Attributes an object header keeps before libhdf5 spills the whole set to
4510/// dense storage (`H5O_CRT_ATTR_MAX_COMPACT_DEF`).
4511const MAX_COMPACT_ATTRS: usize = 8;
4512
4513/// Links `H5G__obj_create_real` sizes a new group's object header for
4514/// (`H5G_CRT_GINFO_EST_NUM_ENTRIES`), and the name length it assumes for each
4515/// (`H5G_CRT_GINFO_EST_NAME_LEN`). Together with the link info and group info
4516/// messages they are the whole of chunk 0 — see
4517/// [`chunk0_capacity`](Hdf5Writer::chunk0_capacity).
4518const EST_LINK_COUNT: usize = 4;
4519/// See [`EST_LINK_COUNT`].
4520const EST_LINK_NAME_LEN: usize = 8;
4521
4522/// Messages a shared-message index keeps in list form before it becomes a v2
4523/// B-tree (`H5F_CRT_SHMSG_LIST_MAX_DEF`).
4524const DEFAULT_SOHM_LIST_MAX: u16 = 50;
4525
4526/// Messages a shared-message B-tree index drops to before it reverts to a
4527/// list (`H5F_CRT_SHMSG_BTREE_MIN_DEF`).
4528const DEFAULT_SOHM_BTREE_MIN: u16 = 40;
4529
4530/// Links a group header keeps before libhdf5 spills the whole set to dense
4531/// storage (`H5G_CRT_GINFO_MAX_COMPACT`). This writer emits no phase-change
4532/// values in the Group Info message, so the default is what applies.
4533const MAX_COMPACT_LINKS: usize = 8;
4534
4535/// Bytes a compact dataset's raw image may occupy.
4536///
4537/// `H5D__compact_construct` bounds it by `H5O_MESG_MAX_SIZE` less the layout
4538/// message's own four bytes (version, class, and the 2-byte data length).
4539/// The constant it subtracts from is 65536, one past what the object header's
4540/// 2-byte message size field can express, so the ceiling here is taken from
4541/// [`MAX_MESSAGE_SIZE`] — the largest message that actually encodes — and is
4542/// one byte below libhdf5's.
4543pub const MAX_COMPACT_DATA: usize = MAX_MESSAGE_SIZE - 4;
4544
4545/// Smallest userblock a file can be created with, and the granularity of
4546/// every larger one: `H5Pset_userblock` takes 0 or a power of two from here
4547/// up, because `H5FD_locate_signature` looks for the superblock at 0 and then
4548/// at this offset doubled repeatedly.
4549pub const MIN_USERBLOCK: u64 = 512;
4550
4551impl Hdf5Writer {
4552    /// Create a new HDF5 file at `path` using the env-var-derived locking
4553    /// policy (controlled by `HDF5_USE_FILE_LOCKING`).
4554    ///
4555    /// The superblock (48 bytes for v3 with 8-byte offsets) is reserved at
4556    /// offset 0 and written during `close()`.
4557    pub fn create(path: &Path) -> IoResult<Self> {
4558        Self::create_with_locking(
4559            path,
4560            crate::io::locking::FileLocking::from_env_or(Default::default()),
4561        )
4562    }
4563
4564    /// Create a new HDF5 file at `path` with an explicit locking policy.
4565    pub fn create_with_locking(
4566        path: &Path,
4567        locking: crate::io::locking::FileLocking,
4568    ) -> IoResult<Self> {
4569        Self::create_with_options(
4570            path,
4571            FileCreateOptions {
4572                locking,
4573                ..Default::default()
4574            },
4575        )
4576    }
4577
4578    /// Create a new HDF5 file at `path` with explicit file-creation options.
4579    pub fn create_with_options(path: &Path, options: FileCreateOptions) -> IoResult<Self> {
4580        let FileCreateOptions {
4581            locking,
4582            track_order,
4583            track_times,
4584            libver,
4585            userblock,
4586            shared_messages,
4587            file_space,
4588        } = options;
4589        shared_messages.validate()?;
4590        file_space.validate()?;
4591        if userblock != 0 && (userblock < MIN_USERBLOCK || !userblock.is_power_of_two()) {
4592            return Err(crate::io::IoError::InvalidState(format!(
4593                "a userblock is {MIN_USERBLOCK} bytes or a power of two above it, \
4594                 not {userblock}: a reader locates the superblock by doubling its \
4595                 search offset from {MIN_USERBLOCK}, so no other size can hold one"
4596            )));
4597        }
4598        let policy = free_space::SpacePolicy::for_message(&file_space.message());
4599        // `H5F__super_init` (H5Fsuper.c:1182-1192) refuses a userblock that is
4600        // not a whole number of allocation units, which for a paged file is
4601        // the file-space page: everything after the userblock is addressed
4602        // from its end, so a userblock that is not a page multiple would put
4603        // every page boundary off the file's own grid.
4604        if let Some(page) = policy.page() {
4605            if userblock != 0 && userblock % page != 0 {
4606                return Err(crate::io::IoError::InvalidState(format!(
4607                    "a paged file's userblock is a multiple of its {page}-byte \
4608                     file-space page, not {userblock}"
4609                )));
4610            }
4611        }
4612        let mut handle = FileHandle::create_with_locking(path, locking)?;
4613        if userblock != 0 {
4614            // Written while the handle is still unbased, so offset 0 is the
4615            // start of the file: the block belongs to the application, not to
4616            // the HDF5 address space that begins where it ends. libhdf5 zeroes
4617            // it the same way (`H5F__super_init`), leaving a file whose first
4618            // `userblock` bytes are the application's to overwrite.
4619            handle.write_at(0, &vec![0u8; userblock as usize])?;
4620            handle.set_base(userblock);
4621        }
4622        let ctx = FormatContext::default_v3();
4623
4624        // `H5F_LIBVER_EARLIEST` is the one bound under which libhdf5 writes
4625        // the classic generation — the version-1 rows of every
4626        // message-version table, the symbol-table group form
4627        // (`H5G__obj_create_real`, H5Gobj.c:179) and the version-0 superblock
4628        // row of `HDF5_superblock_ver_bounds`.
4629        //
4630        // Shared object header messages move the last of those three and
4631        // nothing else. Their master table lives in a superblock extension,
4632        // which only a version-2 superblock has, so `H5F__super_init` raises
4633        // the superblock to version 2 whatever the low bound says
4634        // (H5Fsuper.c:1135) — but it does not touch `H5F_LOW_BOUND`, which is
4635        // what every other rule reads. So such a file is a version-2
4636        // superblock over symbol-table groups and version-1 messages, which
4637        // is what the `tests/fixtures/sohm_*.h5` files libhdf5 itself wrote
4638        // are.
4639        let classic = libver == Some(LibverBound::Earliest);
4640        let legacy = classic.then(|| Box::new(LegacyFile::created(ctx, userblock)));
4641        // Non-default file-space properties raise the superblock the same way
4642        // a shared-message table does, and for the same reason: the message
4643        // that declares them lives in an extension, and only a version-2
4644        // superblock has one (H5Fsuper.c:1144).
4645        let superblock_version = SuperblockVersion::Chosen(
4646            if classic && shared_messages.specs().is_empty() && file_space.is_default() {
4647                SUPERBLOCK_V0
4648            } else {
4649                SUPERBLOCK_V2
4650            },
4651        );
4652
4653        // Reserve the superblock at offset 0. Which version it gets is only
4654        // known once the file's content is (see `superblock_version_for`),
4655        // but the two a version-2 file can reach — 2 and 3 — encode to the
4656        // same size, so the reservation follows the base version alone.
4657        let superblock_size = match legacy.as_deref() {
4658            Some(l) if matches!(superblock_version, SuperblockVersion::Chosen(v) if v < SUPERBLOCK_V2) => {
4659                l.superblock.encoded_size()
4660            }
4661            _ => SuperblockV2V3::size_for(ctx.sizeof_addr),
4662        };
4663        // The superblock is an ordinary allocation, not a reservation: under
4664        // paged aggregation it takes the whole of page zero and leaves the
4665        // rest of that page as a section of the metadata manager, which is
4666        // what `H5F__super_init` gets from `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`
4667        // going through `H5MF__alloc_pagefs`. Unpaged it returns offset zero
4668        // and moves the end of the file to `superblock_size`, which is what
4669        // reserving it did.
4670        let allocator = FileAllocator::with_policy(0, policy);
4671        allocator.allocate(superblock_size as u64, FreeSpaceClass::Metadata);
4672
4673        Ok(Self {
4674            handle,
4675            allocator,
4676            ctx,
4677            datasets: Slot::new(Vec::new()),
4678            groups: Slot::new(Vec::new()),
4679            hard_links: Slot::new(Vec::new()),
4680            symbolic_links: Slot::new(Vec::new()),
4681            committed_datatypes: Slot::new(Vec::new()),
4682            preserved_links: Slot::new(Vec::new()),
4683            name_index: Slot::new(Box::new(NameIndex::new())),
4684            root_attributes: Slot::new(Vec::new()),
4685            create_lock: Slot::new(()),
4686            libver,
4687            closed: false,
4688            swmr_active: false,
4689            cwfs: Slot::new(Vec::new()),
4690            root_group_addr: None,
4691            root_group_encoded_size: 0,
4692            superseded_root_header: Vec::new(),
4693            // A new file starts at the oldest superblock the generation it was
4694            // created in allows, and finalize raises it if the content needs a
4695            // newer one.
4696            superblock_version,
4697            dense_attributes: Slot::new(HashMap::new()),
4698            dense_links: Slot::new(HashMap::new()),
4699            superseded_dense: Slot::new(None),
4700            track_order: TrackOrder::uniform(track_order),
4701            track_times,
4702            root_track_order: TrackOrder::uniform(track_order),
4703            // The root group is created with the file, so it captures the
4704            // policy the same instant every other field of it is settled.
4705            root_times: track_times.then(|| ObjectTimes::created_at(now_seconds())),
4706            next_creation_seq: Slot::new(0),
4707            pending_object_references: Slot::new(Vec::new()),
4708            pending_heap_references: Slot::new(Vec::new()),
4709            attribute_references: Slot::new(Vec::new()),
4710            legacy,
4711            symbol_tables: SymbolTables::none_found(),
4712            // A created file has no extension to carry and no ranks but the
4713            // library defaults: `H5Pset_sym_k`/`H5Pset_istore_k` have no
4714            // equivalent on this writer's creation path.
4715            btree: BTreeV1Config::default(),
4716            extension: Box::default(),
4717            // A file created at the library defaults declares no file-space
4718            // strategy, so it has no message to write and no manager to keep;
4719            // one created with any other properties owns both.
4720            free_space: (!file_space.is_default()).then(|| {
4721                Box::new(FileSpaceState {
4722                    info: file_space.message(),
4723                    superseded: Vec::new(),
4724                })
4725            }),
4726            sohm: (!shared_messages.specs().is_empty())
4727                .then(|| Box::new(SohmState::new(shared_messages.specs().to_vec(), Vec::new()))),
4728            source_dir: source_dir_of(path)?,
4729        })
4730    }
4731
4732    /// Target the libhdf5 2.0 file format for datasets created after this
4733    /// call: filtered chunked datasets get layout message version 5, whose
4734    /// chunk indexes store chunk sizes in a fixed `sizeof_size`-byte field
4735    /// with no overflow limit (see [`Self::chunk_layout_version`]). Off by
4736    /// default, because readers older than libhdf5 2.0 — including the
4737    /// 1.14-based h5py wheels — reject version 5.
4738    ///
4739    /// `false` names `H5F_LIBVER_EARLIEST`, the far end of the same table,
4740    /// rather than un-naming the bound: it is `set_libver_bound`'s contract
4741    /// that applies, chunk index included.
4742    pub fn set_libver_latest(&mut self, latest: bool) -> IoResult<()> {
4743        self.set_libver_bound(if latest {
4744            LibverBound::V200
4745        } else {
4746            LibverBound::Earliest
4747        })
4748    }
4749
4750    /// Bytes this file reserves in front of its superblock
4751    /// (`H5Pget_userblock`).
4752    ///
4753    /// The same value for a file created with one and for a file reopened
4754    /// through [`open_append_with_locking`](Self::open_append_with_locking),
4755    /// which takes it from where the signature turned up: it is the base of
4756    /// the handle's address space either way.
4757    pub fn userblock_size(&self) -> u64 {
4758        self.handle.base()
4759    }
4760
4761    /// Set the file's low libver bound, the equivalent of
4762    /// `H5Pset_libver_bounds`'s `low` argument. Objects created after this
4763    /// call encode their messages at the versions that bound calls for.
4764    ///
4765    /// On a reopened file the bound is raised to the row the file's superblock
4766    /// version belongs to if it names an older one, exactly as
4767    /// `H5F__super_read` raises the fapl's value — see
4768    /// [`libver_floor`](Self::libver_floor). Only a bound the file's format
4769    /// cannot express at all is refused.
4770    pub fn set_libver_bound(&mut self, libver: LibverBound) -> IoResult<()> {
4771        // A classic file cannot honour a newer bound: every encoder in it
4772        // reads `H5F_LOW_BOUND`, and raising that is what makes libhdf5 write
4773        // the version-2/3 superblock this file does not have. Refused rather
4774        // than pinned silently, so the caller learns the bound did not take.
4775        if libver != LibverBound::Earliest && self.is_legacy() {
4776            return Err(crate::io::IoError::Unsupported(format!(
4777                "cannot set the library-version bound to {libver:?} on this file: it is                  in the classic (version-0/1 superblock) format, which libhdf5 writes                  only at H5F_LIBVER_EARLIEST"
4778            )));
4779        }
4780        self.libver = Some(libver);
4781        Ok(())
4782    }
4783
4784    /// The generation the *message* encoders follow — dataspace, datatype,
4785    /// fill value, attribute.
4786    ///
4787    /// A property of the file, not of the object: `H5S__set_version`,
4788    /// `H5O__fill_set_version`, `H5A__set_version` and `H5T_set_version` all
4789    /// read `H5F_LOW_BOUND(f)` and nothing about the object they are encoding
4790    /// for. So a creation-order-tracking group in a classic file still gets
4791    /// version-1 dataspaces and version-1 attribute messages, even though its
4792    /// own header is version 2.
4793    fn message_format(&self) -> ObjectFormat {
4794        match self.legacy {
4795            Some(_) => ObjectFormat::Legacy,
4796            None => ObjectFormat::Modern,
4797        }
4798    }
4799
4800    /// The object header version an object with this creation-order policy
4801    /// gets — `H5O__set_version` (H5Oint.c:251).
4802    ///
4803    /// Version 1 is the floor a classic file's low bound sets, but tracking
4804    /// creation order of *either* kind raises the object past it: the link
4805    /// creation index lives in the message envelope and the attribute tracking
4806    /// flags live in the header prefix, and version 1 has neither. This is a
4807    /// per-object question in a classic file, which is why the format is not
4808    /// one switch for the whole file — libhdf5 writes version-2 headers inside
4809    /// a version-0 superblock whenever the creation property list asks for
4810    /// creation order.
4811    fn header_format(&self, track: TrackOrder) -> ObjectFormat {
4812        let attrs = self.header_attr_order(track.attrs);
4813        if self.legacy.is_some() && !track.links.is_tracked() && !attrs.is_tracked() {
4814            ObjectFormat::Legacy
4815        } else {
4816            ObjectFormat::Modern
4817        }
4818    }
4819
4820    /// The attribute creation-order policy an object header records, given
4821    /// what the object's creation property list asked for.
4822    ///
4823    /// A file whose shared-message configuration covers attributes records a
4824    /// creation index on every object header message: a shared attribute is
4825    /// found again through it, so `H5SM_init` sets `store_msg_crt_idx`
4826    /// (H5SM.c:220) and `H5O__create_ohdr` then raises every header it creates
4827    /// to version 2 and ORs `H5O_HDR_ATTR_CRT_ORDER_TRACKED` into its flags
4828    /// (H5Oint.c:364, H5Oint.c:442) whatever the property list says. So on
4829    /// such a file the floor is `Tracked` — this is the only place that floor
4830    /// is applied, and both the header version and the header flags come
4831    /// through here.
4832    fn header_attr_order(&self, requested: CreationOrder) -> CreationOrder {
4833        if requested.is_tracked() || !self.tracks_message_creation_index() {
4834            return requested;
4835        }
4836        CreationOrder::Tracked
4837    }
4838
4839    /// Whether this finalize replaces the file's shared-message table.
4840    ///
4841    /// It does whenever the file has indexes and no table has been published
4842    /// this session — every finalize of a file created with them, and the
4843    /// first finalize after a reopen. `build_shared_messages` lays a table out
4844    /// whole from the whole message set rather than inserting into an existing
4845    /// one, so a reopen's table is a *replacement*: every heap ID in the file
4846    /// is reassigned, which makes every object header that holds one stale
4847    /// however little else about it changed. A second finalize (a SWMR close)
4848    /// keeps the table the first published and answers `false`.
4849    fn rebuilds_shared_messages(&self) -> bool {
4850        self.sohm
4851            .as_deref()
4852            .is_some_and(|s| s.table_addr.lock().is_none())
4853    }
4854
4855    /// Whether every object header this writer emits records message creation
4856    /// indices.
4857    fn tracks_message_creation_index(&self) -> bool {
4858        self.sohm
4859            .as_deref()
4860            .is_some_and(SohmState::shares_attributes)
4861    }
4862
4863    /// Whether the group at `scope` stores its links in a symbol table —
4864    /// `H5G__obj_create_real` (H5Gobj.c:129) and the conversion
4865    /// `H5G_obj_insert` performs (H5Gobj.c:512).
4866    ///
4867    /// The new group format is used unconditionally from `H5F_LIBVER_V18` up
4868    /// *for a group being created*, and below it only when the group tracks
4869    /// link creation order: a symbol table entry has no room for a creation
4870    /// index. The two axes are independent — a group that tracks only
4871    /// *attribute* creation order gets a version-2 header over a symbol table,
4872    /// which is what libhdf5 writes for it.
4873    ///
4874    /// A group the reopen found in a symbol table is not being created, and
4875    /// `H5G_obj_insert` never moves an existing group to the new format for
4876    /// the bound's sake. So [`SymbolTables::found`] answers for it whatever
4877    /// generation the rest of this session writes at.
4878    ///
4879    /// The content of the group is the third axis. A symbol table entry has
4880    /// three cache types and no room for a fourth, so an external or
4881    /// user-defined link cannot go in one; libhdf5 answers by converting that
4882    /// one group to link messages the moment such a link is inserted, leaving
4883    /// the superblock version, the object header version and every other group
4884    /// in the file alone. This writer builds each group's storage once at
4885    /// finalize rather than link by link, so the same rule reads as a question
4886    /// about the finished set.
4887    fn uses_symbol_table(&self, scope: LinkScope, links: CreationOrder) -> bool {
4888        (self.legacy.is_some() || self.symbol_tables.found.contains(&scope))
4889            && !links.is_tracked()
4890            && self.links_fit_symbol_table(scope, links)
4891    }
4892
4893    /// Whether every link `scope` holds is one a symbol table entry can
4894    /// express — `H5G_obj_insert`'s `obj_lnk->cset != H5T_CSET_ASCII ||
4895    /// obj_lnk->type > H5L_TYPE_BUILTIN_MAX` test (H5Gobj.c:514), asked of the
4896    /// whole set.
4897    ///
4898    /// A link a reopen carried through verbatim counts too, and one this
4899    /// writer cannot even decode counts as not fitting: the entry would have
4900    /// to be built from the decoded form, while a link message is re-emitted
4901    /// byte for byte.
4902    fn links_fit_symbol_table(&self, scope: LinkScope, order: CreationOrder) -> bool {
4903        self.group_links(scope, order)
4904            .iter()
4905            .all(LinkMessage::fits_symbol_table)
4906            && self.preserved_links_for(scope).iter().all(|encoded| {
4907                LinkMessage::decode(encoded, &self.ctx)
4908                    .is_ok_and(|(link, _)| link.fits_symbol_table())
4909            })
4910    }
4911
4912    /// The header format of the registered dataset at `index`.
4913    ///
4914    /// A dataset has no links, so only the attribute half of the policy can
4915    /// raise it past version 1.
4916    fn dataset_header_format(&self, index: usize) -> ObjectFormat {
4917        let ds = self.ds(index);
4918        let attrs = ds.lock().track_attr_order;
4919        self.header_format(TrackOrder {
4920            links: CreationOrder::default(),
4921            attrs,
4922        })
4923    }
4924
4925    /// The header format of the registered group at `index`.
4926    fn group_header_format(&self, index: usize) -> ObjectFormat {
4927        let grp = self.grp(index);
4928        let track = grp.lock().track_order;
4929        self.header_format(track)
4930    }
4931
4932    /// The oldest bound this file may be written at, and the single owner of
4933    /// the reopen half of the [`SuperblockVersion`] invariant.
4934    ///
4935    /// A reopened file's superblock version is the only thing on disk that
4936    /// says which generation the file is, and `H5F__super_read` reads it as
4937    /// exactly that: it raises `H5F_LOW_BOUND` to the row that version belongs
4938    /// to (hdf5_1.14.6 H5Fsuper.c:460-466). Every version-selecting site below
4939    /// goes through [`session_libver`](Self::session_libver) rather than
4940    /// reading the `libver` field, so none of them can hand a reopened file a
4941    /// structure older than the file already claims to hold.
4942    ///
4943    /// A floor, not a ceiling. `H5Fopen` takes a fapl like `H5Fcreate` does,
4944    /// and a bound named above this one applies: libhdf5 1.14.6 writes a
4945    /// version-4 layout message into a version-2 superblock when asked at
4946    /// `H5F_LIBVER_V110`, leaving the superblock version alone. The ceiling is
4947    /// the separate question [`set_libver_bound`](Self::set_libver_bound)
4948    /// answers — no bound but `Earliest` may be named on a classic file.
4949    fn libver_floor(&self) -> LibverBound {
4950        self.superblock_version.libver_floor()
4951    }
4952
4953    /// The bound one family of encoders is written at, and the single reader
4954    /// of the `libver` field.
4955    ///
4956    /// Three inputs, in the order libhdf5 applies them. A bound the caller
4957    /// named is the fapl's `low`, raised to the floor exactly as
4958    /// `H5F__super_read` raises it. With no bound named the answer depends on
4959    /// which superblock this file has:
4960    ///
4961    /// * A file this writer created has none yet, so the writer picks —
4962    ///   `create_default`, which differs per family because this crate's
4963    ///   default file is two rows rather than one bound (see the `libver`
4964    ///   field, [`encoding_libver`] and [`layout_version_bound`]). The
4965    ///   superblock is then written to match what was picked.
4966    /// * A reopened file has already said which generation it is, and its
4967    ///   superblock cannot be rewritten to match a newer pick. So the floor is
4968    ///   the whole answer — the same value `H5F_LOW_BOUND` has after
4969    ///   `H5F__super_read` under a default fapl.
4970    ///
4971    /// [`encoding_libver`]: Self::encoding_libver
4972    /// [`layout_version_bound`]: Self::layout_version_bound
4973    fn session_libver(&self, create_default: LibverBound) -> LibverBound {
4974        let floor = self.libver_floor();
4975        let bound = match (self.libver, self.superblock_version) {
4976            (Some(named), _) => named.max(floor),
4977            (None, SuperblockVersion::Existing(_)) => floor,
4978            (None, SuperblockVersion::Chosen(_)) => create_default,
4979        };
4980        match self.message_format() {
4981            // `H5F_LIBVER_EARLIEST` is the only low bound under which libhdf5
4982            // writes a version-0/1 superblock at all, so a newer structure
4983            // inside one is a combination no libhdf5 produces. Refused where
4984            // the caller asks for it (`set_libver_bound`) rather than silently
4985            // dropped; capping here is what keeps the encoders honest if a
4986            // path ever misses that gate.
4987            ObjectFormat::Legacy => bound.min(LibverBound::Earliest),
4988            ObjectFormat::Modern => bound,
4989        }
4990    }
4991
4992    /// The bound the message encoders see — dataspace, datatype, fill value,
4993    /// attribute.
4994    fn encoding_libver(&self) -> LibverBound {
4995        self.session_libver(LibverBound::Earliest)
4996    }
4997
4998    /// The data layout message version this file's bound calls for —
4999    /// `H5O_layout_ver_bounds[H5F_LOW_BOUND(f)]` (H5Dlayout.c:44), the term
5000    /// `H5D__chunk_set_info` weighs against the version a chunk *requires*
5001    /// (H5Dchunk.c:936, :1046).
5002    ///
5003    /// With no bound named the row is `H5F_LIBVER_V110`'s: this crate's
5004    /// default file uses the v1.10 chunk indexes, which is exactly what that
5005    /// row says and what no other row does (see the `libver` field for why the
5006    /// default is not `Earliest` here even though the datatype and superblock
5007    /// tables read it that way). A file whose superblock already places it on
5008    /// an older row takes that row instead — a reopened version-2 superblock
5009    /// is the `V18` row, whose layout version of 3 has no index-type field at
5010    /// all, so its appended chunked datasets go on the version-1 B-tree.
5011    fn layout_version_bound(&self) -> u8 {
5012        self.session_libver(LibverBound::V110).layout_version()
5013    }
5014
5015    /// The data layout version a chunk of `chunk_bytes` *requires* whatever
5016    /// the bound says — `version_req` in `H5D__chunk_set_info` (H5Dchunk.c:909).
5017    ///
5018    /// Only one thing raises it: a chunk over 4 GiB does not fit the version-4
5019    /// message's 32-bit stored-size field. The floor is the default the
5020    /// creation property list carries, `H5O_LAYOUT_VERSION_DEFAULT`
5021    /// (H5Oprivate.h:451), which is why a classic file's chunked dataset is a
5022    /// version-3 message rather than the version-1 its bound's row names.
5023    fn required_chunk_layout_version(chunk_bytes: u64) -> u8 {
5024        if chunk_bytes > u32::MAX as u64 {
5025            5
5026        } else {
5027            LAYOUT_VERSION_DEFAULT
5028        }
5029    }
5030
5031    /// Whether a new chunked dataset of this chunk size is indexed by one of
5032    /// the v1.10 indexes — extensible array, fixed array, v2 B-tree, single
5033    /// chunk or implicit — rather than by the version-1 B-tree.
5034    ///
5035    /// The gate `H5D__chunk_set_info` puts in front of the whole
5036    /// index-selection block (H5Dchunk.c:936): the bound's layout version
5037    /// reaches 4, or the chunk requires a version that does. Only inside it
5038    /// does the dataspace get to pick between the five; below it the layout
5039    /// message has no index-type field and the chunks go on the version-1
5040    /// B-tree. So the format decides before the shape does — a fixed shape
5041    /// covered by exactly one chunk takes the single-chunk index only on the
5042    /// near side of this gate.
5043    pub(crate) fn uses_v110_chunk_indexing(&self, chunk_bytes: u64) -> bool {
5044        self.layout_version_bound() >= 4 || Self::required_chunk_layout_version(chunk_bytes) >= 4
5045    }
5046
5047    /// Refuse an SWMR session this file's format cannot record.
5048    ///
5049    /// The two checks `H5F__start_swmr_write` opens with: the superblock must
5050    /// be at least version 3 (H5Fint.c:3814, hdf5_1.14.6 H5Fint.c:3751) — the
5051    /// only version with the status-flags field that says a writer is attached
5052    /// — and the low bound must be at least `H5F_LIBVER_V110` (H5Fint.c:3818),
5053    /// the oldest bound whose `HDF5_superblock_ver_bounds` row reaches version
5054    /// 3.
5055    ///
5056    /// Which of the two applies is the [`SuperblockVersion`] question. A
5057    /// reopened file already has its version and reopening never rewrites one,
5058    /// so the first check decides and the second cannot fail after it: the
5059    /// version-3 floor is `V110`. A file this writer created has no version on
5060    /// disk yet, so only the second is askable — and a caller who named no
5061    /// bound at all passes it, because nothing in such a file says the
5062    /// superblock may not be version 3 and SWMR is what makes it one.
5063    ///
5064    /// Named, not silently upgraded. libhdf5 upgrades in the one case where
5065    /// SWMR is asked for at *create* time (`H5F_ACC_SWMR_WRITE` raises the
5066    /// bound to V110 in `H5F__super_init`, H5Fsuper.c:1131); on the reopen
5067    /// path it refuses instead, and so does this.
5068    fn reject_swmr(&self) -> IoResult<()> {
5069        let why = match self.superblock_version {
5070            SuperblockVersion::Existing(version) if version >= SUPERBLOCK_V3 => return Ok(()),
5071            SuperblockVersion::Existing(version) => format!(
5072                "its superblock is version {version}, and reopening a file never \
5073                 rewrites that"
5074            ),
5075            SuperblockVersion::Chosen(_) if self.is_legacy() => {
5076                "it is in the classic (version-0/1 superblock) format that \
5077                 H5F_LIBVER_EARLIEST selects"
5078                    .to_string()
5079            }
5080            SuperblockVersion::Chosen(_) if self.libver.is_some_and(|b| b < LibverBound::V110) => {
5081                "it was asked for at a library-version bound below H5F_LIBVER_V110, \
5082                 whose superblock row is version 2"
5083                    .to_string()
5084            }
5085            SuperblockVersion::Chosen(_) => return Ok(()),
5086        };
5087        Err(crate::io::IoError::Unsupported(format!(
5088            "cannot start an SWMR session on this file: {why}, and SWMR needs a \
5089             version-3 superblock to record that a writer is attached; create the \
5090             file at H5F_LIBVER_V110 or newer"
5091        )))
5092    }
5093
5094    /// Whether this file is in the classic (version-0/1 superblock) format,
5095    /// whose groups store their links in symbol tables — either because it
5096    /// was reopened in it or because it was created at
5097    /// `H5F_LIBVER_EARLIEST`.
5098    pub(crate) fn is_legacy(&self) -> bool {
5099        self.legacy.is_some()
5100    }
5101
5102    /// The v1-B-tree "K" ranks in force for this file, from which every v1
5103    /// node's width is derived.
5104    ///
5105    /// A version-0/1 superblock records them in a field of its own and a
5106    /// version-2/3 one in a B-tree-K message in its superblock extension, so
5107    /// the file's generation says nothing about whether they are the defaults
5108    /// — `H5F__super_read` reads both into the same `H5F_shared_t`, and so
5109    /// does the reopen, into `btree`.
5110    fn btree_v1_config(&self) -> BTreeV1Config {
5111        self.btree
5112    }
5113
5114    /// Track and index creation order for the links and the attributes of
5115    /// every object created after this call — the equivalent of setting
5116    /// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` to
5117    /// `H5P_CRT_ORDER_TRACKED | H5P_CRT_ORDER_INDEXED` on the creation
5118    /// property lists those objects are made with.
5119    ///
5120    /// Objects already created keep the policy they were made under, exactly
5121    /// as libhdf5 keeps what their creation property list said. The root
5122    /// group is created with the file, so its policy comes from
5123    /// [`create_with_options`](Self::create_with_options) instead.
5124    pub fn set_track_order(&mut self, track: bool) {
5125        self.track_order = TrackOrder::uniform(track);
5126    }
5127
5128    /// Record the times of every object created after this call —
5129    /// `H5Pset_obj_track_times` on the creation property lists those objects
5130    /// are made with.
5131    ///
5132    /// Off by default, which is h5py's default and not libhdf5's: h5py's
5133    /// high-level API sets `track_times=False` on every object it makes
5134    /// (`_hl/files.py:189`, `_hl/dataset.py:39`, `_hl/group.py:42`), while a
5135    /// bare creation property list leaves it on (`H5O_CRT_OHDR_FLAGS_DEF` is
5136    /// `H5O_HDR_STORE_TIMES`, H5Opkg.h:74). A caller after libhdf5's own
5137    /// bytes turns it on here.
5138    ///
5139    /// Objects already created keep the policy they were made under, and the
5140    /// root group takes its own from
5141    /// [`create_with_options`](Self::create_with_options) — the same split
5142    /// [`set_track_order`](Self::set_track_order) has, and for the same
5143    /// reason: this is a creation property, not a file-wide setting.
5144    pub fn set_track_times(&mut self, track: bool) {
5145        self.track_times = track;
5146    }
5147
5148    /// The times an object created right now records — all four set to the
5149    /// current time, as `H5O_apply_ohdr` initialises them (H5Oint.c:411-414),
5150    /// or `None` when this session is not tracking times.
5151    ///
5152    /// INVARIANT: every object this writer registers takes its `times` from
5153    /// here. The policy belongs to the creation property list, so reading
5154    /// [`track_times`](Self::track_times) at any later moment — a finalize, a
5155    /// header rewrite — would stamp a policy the object was not made under.
5156    fn created_object_times(&self) -> Option<ObjectTimes> {
5157        self.track_times
5158            .then(|| ObjectTimes::created_at(now_seconds()))
5159    }
5160
5161    /// Layout message version for a new chunked dataset on one of the v1.10
5162    /// indexes — `H5D__chunk_set_info`'s closing
5163    /// `MAX3(layout->version, version_req, MIN(bound, version_perf))`
5164    /// (H5Dchunk.c:1046).
5165    ///
5166    /// Version 5 is *required* for a chunk over 4 GiB (pre-2.0 readers cannot
5167    /// handle one even though the v4 wire format could express it) and
5168    /// *preferred* for filtered chunks, which is why it takes the file's
5169    /// bound to get there: the preference is capped by the bound's own row,
5170    /// so only the 2.0 format lets it through. Everything else stays at
5171    /// version 4, which every 1.10+ reader accepts.
5172    fn chunk_layout_version(&self, filtered: bool, chunk_bytes: u64) -> u8 {
5173        // `version_perf`: 4 for the v1.10 indexes as such, 5 when a filter
5174        // can make a chunk expand past what version 4 can record.
5175        let preferred = if filtered { 5 } else { 4 };
5176        Self::required_chunk_layout_version(chunk_bytes)
5177            .max(self.layout_version_bound().min(preferred))
5178            .max(LAYOUT_VERSION_DEFAULT)
5179    }
5180
5181    /// Width of the stored-chunk-size field in a filtered chunk index:
5182    /// version 5 uses the fixed `sizeof_size`; version 4 derives it from the
5183    /// uncompressed chunk byte count (one spare byte included), the
5184    /// `H5D_*_COMPUTE_CHUNK_SIZE_LEN` rule shared by the extensible-array,
5185    /// fixed-array and v2-B-tree indexes.
5186    fn chunk_size_len_for(&self, layout_version: u8, chunk_bytes: u64) -> u8 {
5187        if layout_version >= 5 {
5188            self.ctx.sizeof_size
5189        } else {
5190            compute_chunk_size_len(chunk_bytes)
5191        }
5192    }
5193
5194    /// Provide public access to the format context.
5195    pub fn ctx(&self) -> &FormatContext {
5196        &self.ctx
5197    }
5198
5199    /// Number of dataset slots in the registry (including soft-deleted ones).
5200    pub(crate) fn dataset_count(&self) -> usize {
5201        self.datasets.lock().len()
5202    }
5203
5204    /// Clone out the [`DatasetRef`] for `index`, releasing the registry lock
5205    /// immediately. Lock the returned ref to read or mutate that one dataset.
5206    ///
5207    /// Panics on an out-of-range index, exactly like the `Vec` indexing it
5208    /// replaces; bounds-checking callers consult [`Self::dataset_count`] first.
5209    ///
5210    /// MUST NOT be called while the registry [`Slot`] is already locked (it
5211    /// would deadlock the `threadsafe` mutex / panic the single-thread
5212    /// `RefCell`): collect the refs you need, drop the registry guard, then work.
5213    pub(crate) fn ds(&self, index: usize) -> DatasetRef {
5214        Shared::clone(&self.datasets.lock()[index])
5215    }
5216
5217    /// Number of group slots in the registry (including soft-deleted ones).
5218    pub(crate) fn group_count(&self) -> usize {
5219        self.groups.lock().len()
5220    }
5221
5222    /// Clone out the [`GroupRef`] for `index`. Same contract as [`Self::ds`].
5223    pub(crate) fn grp(&self, index: usize) -> GroupRef {
5224        Shared::clone(&self.groups.lock()[index])
5225    }
5226
5227    /// Enter the create gate: take `create_lock` and check that `name` is not
5228    /// already taken. The returned witness is what [`Self::push_dataset`]
5229    /// requires, so the uniqueness check and the registry push are atomic
5230    /// (see `create_lock`) at every creator by construction.
5231    pub(crate) fn begin_create(&self, name: &str) -> IoResult<CreateGuard<'_>> {
5232        let gate = self.create_lock.lock();
5233        // A creation path through hard links lands in the link's target
5234        // group, as HDF5 traversal does. Canonicalizing here — the one
5235        // entry every creator passes — keeps alias forms out of the
5236        // registry.
5237        let name = self.canonical_dataset_path(name);
5238        // A path that leaves this file, or that runs into an object the
5239        // reopen kept verbatim, is refused here rather than at each creator:
5240        // this is the one gate every creation passes, so a creator added
5241        // later cannot forget the check. Both run before the parent lookup,
5242        // which would otherwise report the group such a path names as absent
5243        // instead of naming what stops the path. Uniqueness comes first among
5244        // them: a name already in the file is taken whatever holds it.
5245        self.reject_external_traversal(&name)?;
5246        self.ensure_name_free(&name)?;
5247        self.reject_preserved_object(&name)?;
5248        let (parent, _leaf) = self.split_parent(&name)?;
5249        Ok(CreateGuard {
5250            _gate: gate,
5251            name,
5252            parent,
5253        })
5254    }
5255
5256    /// Split an object path into the group that will hold its link and the
5257    /// leaf link name, resolving every component through the group registry.
5258    ///
5259    /// `path` is the registry form — no leading `/`, e.g. `"grp/sub/late"`.
5260    /// This is what keeps a `/` out of a link name: HDF5 link names are
5261    /// single path components (`H5G_traverse` splits on `/` before it ever
5262    /// reaches `H5L_link`), so a name that carries a path must name a group
5263    /// that exists, or be refused.
5264    ///
5265    /// A missing component is an error rather than an implicit group: the
5266    /// default link creation property list has `H5Pset_create_intermediate_group`
5267    /// off, and this writer exposes no property list to turn it on with.
5268    fn split_parent(&self, path: &str) -> IoResult<(Option<usize>, String)> {
5269        let (parent_path, leaf) = path.rsplit_once('/').unwrap_or(("", path));
5270        if leaf.is_empty() {
5271            return Err(crate::io::IoError::InvalidState(format!(
5272                "'{path}' does not end in a link name"
5273            )));
5274        }
5275        if parent_path.is_empty() {
5276            return Ok((None, leaf.to_string()));
5277        }
5278        let abs = format!("/{parent_path}");
5279        let groups = self.group_refs();
5280        let idx = groups
5281            .iter()
5282            .position(|g| {
5283                let gg = g.lock();
5284                gg.name == abs && !gg.deleted
5285            })
5286            .ok_or_else(|| {
5287                crate::io::IoError::NotFound(format!(
5288                    "cannot create '{path}': group '{abs}' does not exist"
5289                ))
5290            })?;
5291        Ok((Some(idx), leaf.to_string()))
5292    }
5293
5294    /// Push a freshly-built dataset into the registry and return its index.
5295    /// Takes the registry lock only for the push, so it does not block an
5296    /// in-flight write that already cloned its own [`DatasetRef`] out.
5297    /// The [`CreateGuard`] proves the caller entered through
5298    /// [`Self::begin_create`] and still holds the gate.
5299    pub(crate) fn push_dataset(&self, create: &CreateGuard<'_>, info: DatasetInfo) -> usize {
5300        let name = info.name.clone();
5301        let idx = {
5302            let mut reg = self.datasets.lock();
5303            let idx = reg.len();
5304            reg.push(Shared::new(DatasetCell::new(info)));
5305            idx
5306        };
5307        self.register_name(&name, NameHit::Dataset(idx));
5308        // The spine guard is dropped before the group slot is taken: the lock
5309        // order is spine -> slot and never the reverse.
5310        if let Some(pidx) = create.parent {
5311            self.grp(pidx).lock().child_datasets.push(idx);
5312        }
5313        idx
5314    }
5315
5316    /// Push a freshly-built group into the registry and return its index.
5317    pub(crate) fn push_group(&self, info: GroupInfo) -> usize {
5318        let name = info.name.trim_start_matches('/').to_string();
5319        let idx = {
5320            let mut reg = self.groups.lock();
5321            let idx = reg.len();
5322            reg.push(Shared::new(Slot::new(info)));
5323            idx
5324        };
5325        self.register_name(&name, NameHit::Group(idx));
5326        idx
5327    }
5328
5329    /// Snapshot every [`DatasetRef`] (spine lock held only for the clone).
5330    /// Iterate the snapshot to lock each dataset one at a time — this keeps
5331    /// the lock order *spine → slot* and never reacquires the spine while a
5332    /// slot is held, which is what makes the registry deadlock-free.
5333    pub(crate) fn dataset_refs(&self) -> Vec<DatasetRef> {
5334        self.datasets.lock().iter().map(Shared::clone).collect()
5335    }
5336
5337    /// Snapshot every [`GroupRef`]; see [`Self::dataset_refs`].
5338    pub(crate) fn group_refs(&self) -> Vec<GroupRef> {
5339        self.groups.lock().iter().map(Shared::clone).collect()
5340    }
5341
5342    /// Snapshot the hard-link list (the lock is held only for the clone), so
5343    /// callers can resolve each link's target/parent — which locks dataset and
5344    /// group slots — without holding the hard-link lock.
5345    /// The next creation sequence number.
5346    ///
5347    /// One monotonic counter for datasets, groups and hard links alike: a
5348    /// group orders its links by it, so an interleaved run of `create_group`
5349    /// and `create_dataset` comes back out in the order it was made rather
5350    /// than grouped by kind.
5351    fn take_creation_seq(&self) -> u64 {
5352        let mut next = self.next_creation_seq.lock();
5353        let seq = *next;
5354        *next += 1;
5355        seq
5356    }
5357
5358    pub(crate) fn hard_links_vec(&self) -> Vec<HardLink> {
5359        self.hard_links.lock().clone()
5360    }
5361
5362    /// Snapshot the symbolic-link list; see [`Self::hard_links_vec`].
5363    pub(crate) fn symbolic_links_vec(&self) -> Vec<SymbolicLink> {
5364        self.symbolic_links.lock().clone()
5365    }
5366
5367    /// Open an existing HDF5 file for appending new datasets, using the
5368    /// env-var-derived locking policy.
5369    ///
5370    /// Reads existing dataset object headers fully, reconstructing metadata
5371    /// for chunked datasets so that `write_chunk` and `extend_dataset` work
5372    /// on reopened datasets.
5373    pub fn open_append(path: &Path) -> IoResult<Self> {
5374        Self::open_append_with_locking(
5375            path,
5376            crate::io::locking::FileLocking::from_env_or(Default::default()),
5377        )
5378    }
5379
5380    /// Carry a reopened file's shared-message table into the writer's model:
5381    /// the index specifications the file was created with, and every block the
5382    /// table occupies so the finalize that replaces it can give them back.
5383    ///
5384    /// `H5SM_init` fixes the index count, each index's type mask, its minimum
5385    /// message size and the file-wide phase-change pair when the file is
5386    /// created, and nothing afterwards changes any of them — they are file
5387    /// creation properties. So the master table on disk *is* the
5388    /// [`SharedMessageConfig`] the file was made with, read back.
5389    ///
5390    /// Returns `None` for a file with no shared-message table, which is every
5391    /// file libhdf5 writes without `H5Pset_shared_mesg_nindexes`.
5392    /// Read the free-space managers a reopened file persists, if it does.
5393    ///
5394    /// `H5F__super_read` copies the file-space info message's addresses into
5395    /// `f->shared->fs_addr[]` and the library opens each manager lazily; this
5396    /// reads them all at once, because the writer needs the whole section set
5397    /// before it allocates anything.
5398    ///
5399    /// Returns `None` — nothing read, nothing to write back — for a file with
5400    /// no file-space info message, one that does not persist, and one whose
5401    /// strategy keeps no managers at all.
5402    fn reopen_free_space(
5403        handle: &mut FileHandle,
5404        meta: &crate::io::FileMeta,
5405        ext: &crate::io::reader::SuperblockExtension,
5406    ) -> IoResult<ReopenedFreeSpace> {
5407        let none = || ReopenedFreeSpace {
5408            state: None,
5409            sections: Vec::new(),
5410        };
5411        let Some(info) = ext.file_space_info.as_ref().filter(|i| i.persist) else {
5412            return Ok(none());
5413        };
5414        if !matches!(
5415            info.strategy,
5416            FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
5417        ) {
5418            return Ok(none());
5419        }
5420        let found = crate::io::free_space_io::read_managers(handle, &meta.ctx, info)?;
5421        Ok(ReopenedFreeSpace {
5422            state: Some(Box::new(FileSpaceState {
5423                info: info.clone(),
5424                superseded: found.blocks,
5425            })),
5426            sections: found.sections,
5427        })
5428    }
5429
5430    fn reopen_shared_messages(
5431        handle: &mut FileHandle,
5432        meta: &crate::io::FileMeta,
5433        ext: &crate::io::reader::SuperblockExtension,
5434    ) -> IoResult<Option<Box<SohmState>>> {
5435        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
5436        use crate::format::fractal_heap::collect_heap_extents;
5437        use crate::format::sohm::{list_size, SohmMasterTable, SOHM_INDEX_LIST};
5438
5439        let (Some(table), Some(smt)) = (
5440            meta.sohm.as_ref().filter(|t| !t.indexes.is_empty()),
5441            ext.shared_message_table.as_ref(),
5442        ) else {
5443            return Ok(None);
5444        };
5445        let ctx = &meta.ctx;
5446
5447        // The extension header itself is superseded by `CarriedExtension`,
5448        // which owns it whether or not the file has shared messages; what is
5449        // superseded here is only the storage the table message names.
5450        let mut superseded = Vec::new();
5451        superseded.push((
5452            smt.table_address,
5453            SohmMasterTable::encoded_size(ctx, smt.nindexes) as u64,
5454        ));
5455
5456        let mut specs = Vec::with_capacity(table.indexes.len());
5457        for index in &table.indexes {
5458            specs.push(SohmIndexSpec {
5459                mesg_types: index.mesg_types,
5460                min_mesg_size: index.min_mesg_size,
5461                list_max: index.list_max,
5462                btree_min: index.btree_min,
5463            });
5464            let mut reader = crate::io::reader::HandleBlockReader { handle };
5465            if index.heap_addr != UNDEF_ADDR {
5466                superseded.extend(collect_heap_extents(index.heap_addr, ctx, &mut reader)?);
5467            }
5468            if index.index_addr != UNDEF_ADDR {
5469                if index.index_type == SOHM_INDEX_LIST {
5470                    // `H5SM_LIST_SIZE`: the block is sized for `list_max`
5471                    // records however few are in it.
5472                    superseded.push((index.index_addr, list_size(ctx, index.list_max) as u64));
5473                } else {
5474                    superseded.extend(collect_btree_v2_extents(
5475                        index.index_addr,
5476                        ctx,
5477                        &mut reader,
5478                    )?);
5479                }
5480            }
5481        }
5482        Ok(Some(Box::new(SohmState::new(specs, superseded))))
5483    }
5484
5485    /// Open an existing HDF5 file for appending with an explicit locking
5486    /// policy.
5487    pub fn open_append_with_locking(
5488        path: &Path,
5489        locking: crate::io::locking::FileLocking,
5490    ) -> IoResult<Self> {
5491        let mut handle = FileHandle::open_readwrite_with_locking(path, locking)?;
5492        // The same `H5FD_locate_signature` search the read path makes, through
5493        // the same handle mechanism: the offset it finds is the file's base
5494        // address, so the allocator's end-of-file, every write and the
5495        // superblock rewrite all work in the HDF5 address space, and the
5496        // userblock in `[0, base)` is not addressable from this writer at all.
5497        let super_addr = handle
5498            .locate_signature()?
5499            .ok_or(crate::format::FormatError::InvalidSignature)?;
5500        handle.set_base(super_addr);
5501        let file_size = handle.file_size()?;
5502
5503        let sb_buf = handle.read_at_most(0, 256)?;
5504        // Which generation the file is decides everything the close then
5505        // writes back: version-1 object headers and symbol-table groups over a
5506        // version-0/1 superblock, or version-2 headers and link-message groups
5507        // over a version-2/3 one. libhdf5 writes those two combinations and no
5508        // mixture of them, so the branch is taken once, here, and carried as
5509        // `legacy`.
5510        let version = crate::format::superblock::detect_superblock_version(&sb_buf)?;
5511        let (ctx, sb_btree, root_addr, ext_addr, legacy) = if version <= 1 {
5512            let sb = SuperblockV0V1::decode(&sb_buf)?;
5513            let ctx = FormatContext {
5514                sizeof_addr: sb.sizeof_offsets,
5515                sizeof_size: sb.sizeof_lengths,
5516            };
5517            // Unlike a v2/v3 superblock, a classic one carries the "K" ranks
5518            // itself; every v1-B-tree and symbol-table node width in the file
5519            // comes from them.
5520            let btree = crate::format::btree_v1::BTreeV1Config {
5521                sym_leaf_k: sb.sym_leaf_k,
5522                snode_internal_k: sb.btree_internal_k,
5523                chunk_internal_k: sb.indexed_storage_k.unwrap_or(32),
5524            };
5525            let root = sb.root_symbol_table_entry.obj_header_addr;
5526            let ext = sb.superblock_extension_address;
5527            (ctx, btree, root, ext, Some(sb))
5528        } else {
5529            let sb = SuperblockV2V3::decode(&sb_buf)?;
5530            let ctx = FormatContext {
5531                sizeof_addr: sb.sizeof_offsets,
5532                sizeof_size: sb.sizeof_lengths,
5533            };
5534            (
5535                ctx,
5536                crate::format::btree_v1::BTreeV1Config::default(),
5537                sb.root_group_object_header_address,
5538                sb.superblock_extension_address,
5539                None,
5540            )
5541        };
5542
5543        // The reopen reads object headers exactly as the reader does, so it
5544        // needs the same file-level parameters: a v2/v3 superblock carries no
5545        // B-tree K values, and only the extension can override the defaults.
5546        let (meta, ext) = crate::io::reader::Hdf5Reader::read_extension_and_meta(
5547            &mut handle,
5548            ctx,
5549            sb_btree,
5550            ext_addr,
5551        )?;
5552
5553        // A file with shared object header messages keeps datatypes,
5554        // dataspaces and attributes in a fractal heap per index, and each
5555        // object header holds a heap ID pointing at one. The table is laid out
5556        // whole from the whole message set (`build_shared_messages`), never
5557        // grown insert by insert, so a reopen carries the indexes and the
5558        // bodies forward and the next finalize lays a new table out over the
5559        // old one's blocks — which is sound exactly while no header keeping
5560        // its bytes still points into the old heap. The walk below is what
5561        // settles that.
5562        let sohm = Self::reopen_shared_messages(&mut handle, &meta, &ext)?;
5563
5564        // The extension is external truth this close rewrites, so what it held
5565        // is captured whole here — before anything else reads the file — and
5566        // re-emitted by `write_superblock_extension`. Read from the raw chain
5567        // rather than from `ext`, which keeps only the messages this crate
5568        // models.
5569        let extension = if ext_addr == UNDEF_ADDR || ext_addr == 0 {
5570            Box::<CarriedExtension>::default()
5571        } else {
5572            let (carried, blocks) = crate::io::object_header_io::superblock_extension_messages(
5573                &mut handle,
5574                &meta,
5575                ext_addr,
5576            )?;
5577            Box::new(CarriedExtension {
5578                superseded: blocks,
5579                carried,
5580                addr: Slot::new(None),
5581            })
5582        };
5583
5584        // The managers that extension's file-space info message names, read
5585        // before anything allocates: the sections they hold are file space
5586        // this session may hand out, and the close rewrites them.
5587        let reopened_free_space = Self::reopen_free_space(&mut handle, &meta, &ext)?;
5588
5589        // Discover links from root group (and subgroups recursively). Every
5590        // object is classified before it is registered, and the root is the
5591        // one object with no alternative: its header must be rewritten to
5592        // hold anything new, so an unmodellable root is refused here rather
5593        // than rewritten into whatever this writer could read of it.
5594        let mut walk = ReopenWalk::new(&mut handle, &meta);
5595        let root = match walk.plan(root_addr)? {
5596            ObjectPlan::Group(parts) => parts,
5597            ObjectPlan::Dataset(_) => {
5598                return Err(crate::io::IoError::InvalidState(
5599                    "cannot open this file for appending: its root object is a dataset, \
5600                     not a group"
5601                        .into(),
5602                ))
5603            }
5604            ObjectPlan::Preserve { why, .. } => {
5605                return Err(crate::io::IoError::Unsupported(format!(
5606                    "cannot open this file for appending: {why}. Every append rewrites the \
5607                     root group's header, and this writer will not rewrite it from the part \
5608                     of it that it can read"
5609                )));
5610            }
5611        };
5612        let root_header_blocks = root.header_blocks;
5613        let root_attributes = root.attributes;
5614        let root_track_order = root.track_order;
5615        let root_times = root.times;
5616        let root_dense = root.dense;
5617        let root_stab = root.stab;
5618
5619        walk.group(&root.links, "", 0)?;
5620        let collected = walk.finish();
5621        let mut link_entries = collected.hard;
5622        let mut preserved = collected.preserved;
5623        // Objects the loop below could not rebuild, by header address, so the
5624        // other links to one are preserved with it rather than left pointing
5625        // at a registry entry that is no longer there.
5626        let mut unrebuilt: std::collections::HashMap<u64, String> = Default::default();
5627
5628        // Two link entries can share one object header — hard links. Only
5629        // the first-walked path becomes the object; the rest are rebuilt
5630        // as hard-link registry entries further down. Without this split
5631        // every alias came back as its own DatasetInfo carrying the same
5632        // storage addresses, so deleting (or finalizing) one freed blocks
5633        // the others still referenced.
5634        let mut seen_header_addrs = std::collections::HashSet::new();
5635        let mut alias_entries: Vec<HardEntry> = Vec::new();
5636        link_entries.retain(|(entry, _)| {
5637            if seen_header_addrs.insert(entry.address) {
5638                true
5639            } else {
5640                alias_entries.push(entry.clone());
5641                false
5642            }
5643        });
5644
5645        // The order the walk met each object, kept before the loop below
5646        // consumes the entries: `ensure_groups_for` needs parents to precede
5647        // children.
5648        let walk_order: Vec<String> = link_entries.iter().map(|(e, _)| e.path.clone()).collect();
5649
5650        let mut existing_datasets = Vec::new();
5651        // Non-dataset link targets (groups): the header's chunk-0 address and
5652        // every block its chain occupies, by link path — so finalize can free
5653        // the blocks its rewrite supersedes — plus the attributes the header
5654        // carries, which the group registry below must keep or finalize
5655        // rewrites the group without them.
5656        type GroupHeaderInfo = (
5657            u64,
5658            crate::io::object_header_io::HeaderBlocks,
5659            Vec<AttributeEntry>,
5660            TrackOrder,
5661            Option<ObjectTimes>,
5662        );
5663        let mut group_headers: std::collections::HashMap<String, GroupHeaderInfo> =
5664            Default::default();
5665        // The dense storage each rebuilt dataset's header named, by registry
5666        // index, so finalize frees exactly what its rewrite supersedes. Keyed
5667        // after the rebuild succeeded: a preserved dataset keeps its header,
5668        // and freeing the heap that header still names would strand it.
5669        let mut dataset_dense: Vec<(usize, AttributeInfoMessage)> = Vec::new();
5670        let mut group_dense: Vec<(String, DenseCarry)> = Vec::new();
5671        // The same, for the symbol-table storage a classic group's header
5672        // names: keyed by path here, by registry index once every group has
5673        // one.
5674        let mut group_stabs: Vec<(String, StabExtents)> = Vec::new();
5675        for (entry, object) in link_entries {
5676            let HardEntry {
5677                path: name,
5678                address: obj_addr,
5679                encoded,
5680            } = entry;
5681            let parts = match object {
5682                CollectedObject::Group {
5683                    header_blocks,
5684                    attributes,
5685                    track_order,
5686                    times,
5687                    dense,
5688                    stab,
5689                } => {
5690                    group_dense.push((name.clone(), dense));
5691                    if let Some(stab) = stab {
5692                        group_stabs.push((name.clone(), stab));
5693                    }
5694                    group_headers.insert(
5695                        name,
5696                        (obj_addr, header_blocks, attributes, track_order, times),
5697                    );
5698                    continue;
5699                }
5700                CollectedObject::Dataset(parts) => *parts,
5701            };
5702            let dense_attrs = parts.dense.attrs.clone();
5703            match rebuild_dataset(&mut handle, &meta, file_size, name.clone(), obj_addr, parts) {
5704                Ok(info) => {
5705                    if let Some(ainfo) = dense_attrs {
5706                        dataset_dense.push((existing_datasets.len(), ainfo));
5707                    }
5708                    existing_datasets.push(info);
5709                }
5710                // Kept by its bytes for the same reason a header this walk
5711                // could not decode is: the rewrite would otherwise emit an
5712                // object whose chunk index no longer names its chunks.
5713                Err(e) => {
5714                    let why = format!("this writer could not rebuild its chunk index: {e}");
5715                    unrebuilt.insert(obj_addr, why.clone());
5716                    preserved.push(PreservedEntry {
5717                        path: name,
5718                        class: crate::io::reader::LinkClass::Hard,
5719                        encoded,
5720                        reason: Some(why),
5721                        // A dataset whose chunk index would not rebuild: the
5722                        // walk classified it, and it is not a datatype.
5723                        kind: PreservedKind::Unclassified,
5724                    });
5725                }
5726            }
5727        }
5728
5729        // Reconstruct the group registry. Every group is a link entry of its
5730        // own, whether or not a dataset lives under it, so the registry is
5731        // built from the discovered links — rebuilding it from dataset paths
5732        // alone made attribute-only and empty groups vanish at close, and
5733        // dropped the attributes of the groups that survived.
5734        let mut groups: Vec<GroupInfo> = Vec::new();
5735        let mut group_index_map: std::collections::HashMap<String, usize> =
5736            std::collections::HashMap::new();
5737
5738        // Register the chain of groups "/a", "/a/b", … for the link-style
5739        // path `link_path` ("a/b"), taking each one's on-disk header block
5740        // and attributes out of `group_headers` when the link walk saw it.
5741        fn ensure_groups_for(
5742            link_path: &str,
5743            groups: &mut Vec<GroupInfo>,
5744            group_index_map: &mut std::collections::HashMap<String, usize>,
5745            group_headers: &mut std::collections::HashMap<String, GroupHeaderInfo>,
5746        ) {
5747            let mut path = String::new();
5748            for part in link_path.split('/') {
5749                let parent_path = if path.is_empty() {
5750                    "/".to_string()
5751                } else {
5752                    path.clone()
5753                };
5754                if path.is_empty() {
5755                    path = format!("/{}", part);
5756                } else {
5757                    path = format!("{}/{}", path, part);
5758                }
5759                if group_index_map.contains_key(&path) {
5760                    continue;
5761                }
5762                let parent = if parent_path == "/" {
5763                    None
5764                } else {
5765                    group_index_map.get(&parent_path).copied()
5766                };
5767                let gidx = groups.len();
5768                let (obj_header_written_addr, obj_header_blocks, attributes, track_order, times) =
5769                    group_headers.remove(path.trim_start_matches('/')).map_or(
5770                        (None, Vec::new(), Vec::new(), TrackOrder::default(), None),
5771                        |(addr, blocks, attrs, track, times)| {
5772                            (Some(addr), blocks, attrs, track, times)
5773                        },
5774                    );
5775                groups.push(GroupInfo {
5776                    name: path.clone(),
5777                    parent,
5778                    creation_seq: 0,
5779                    track_order,
5780                    times,
5781                    child_datasets: Vec::new(),
5782                    child_groups: Vec::new(),
5783                    obj_header_addr: 0,
5784                    obj_header_written_addr,
5785                    obj_header_blocks,
5786                    deleted: false,
5787                    attributes,
5788                });
5789                if let Some(pidx) = parent {
5790                    groups[pidx].child_groups.push(gidx);
5791                }
5792                group_index_map.insert(path.clone(), gidx);
5793            }
5794        }
5795
5796        // Every linked group, in link-walk order (parents precede children).
5797        for name in &walk_order {
5798            if group_headers.contains_key(name.as_str()) {
5799                ensure_groups_for(name, &mut groups, &mut group_index_map, &mut group_headers);
5800            }
5801        }
5802
5803        // Assign each dataset to its immediate parent group, creating any
5804        // group the link walk could not decode (its chain stays placeholder).
5805        for (di, ds) in existing_datasets.iter().enumerate() {
5806            let parts: Vec<&str> = ds.name.split('/').collect();
5807            if parts.len() <= 1 {
5808                continue; // root-level dataset, no group
5809            }
5810            let parent_link_path = parts[..parts.len() - 1].join("/");
5811            ensure_groups_for(
5812                &parent_link_path,
5813                &mut groups,
5814                &mut group_index_map,
5815                &mut group_headers,
5816            );
5817            let gidx = group_index_map[&format!("/{}", parent_link_path)];
5818            groups[gidx].child_datasets.push(di);
5819        }
5820
5821        // An object the rebuild above gave up on is preserved by its bytes,
5822        // so the other links to it are preserved too: there is no registry
5823        // entry for them to name.
5824        alias_entries.retain(|entry| match unrebuilt.get(&entry.address) {
5825            None => true,
5826            Some(why) => {
5827                preserved.push(PreservedEntry {
5828                    path: entry.path.clone(),
5829                    class: crate::io::reader::LinkClass::Hard,
5830                    encoded: entry.encoded.clone(),
5831                    reason: Some(why.clone()),
5832                    kind: PreservedKind::Unclassified,
5833                });
5834                false
5835            }
5836        });
5837
5838        // The one thing a rebuilt shared-message table can break: an object
5839        // kept by its bytes keeps the heap IDs its header holds, and the
5840        // finalize gives the heap those IDs name back to the allocator. Every
5841        // object the registry holds is rewritten instead
5842        // ([`rebuilds_shared_messages`](Self::rebuilds_shared_messages)), so
5843        // this asks only the preserved ones, and names the object rather than
5844        // the feature — the file is appendable the moment nothing preserved
5845        // holds a heap ID or hides a subtree that might.
5846        if sohm.is_some() {
5847            for entry in &preserved {
5848                if !matches!(entry.class, crate::io::reader::LinkClass::Hard) {
5849                    continue;
5850                }
5851                let Ok((link, _)) = LinkMessage::decode(&entry.encoded, &meta.ctx) else {
5852                    continue;
5853                };
5854                let LinkTarget::Hard { address } = link.target else {
5855                    continue;
5856                };
5857                if let Some(blocks) = crate::io::object_header_io::blocks_shared_message_rebuild(
5858                    &mut handle,
5859                    &meta,
5860                    address,
5861                )? {
5862                    let why = entry
5863                        .reason
5864                        .as_deref()
5865                        .unwrap_or("this writer cannot model it");
5866                    return Err(crate::io::IoError::Unsupported(format!(
5867                        "cannot open this file for appending: '{}' {blocks}, but {why}, so \
5868                         its header keeps the bytes it has while the append lays the \
5869                         shared-message table out afresh",
5870                        entry.path
5871                    )));
5872                }
5873            }
5874        }
5875
5876        // Rebuild the hard-link registry from the alias entries set aside
5877        // above, so the H5Ldelete semantics survive a reopen. An alias whose
5878        // target the walk could not model is not here at all: it was
5879        // preserved by its own bytes, exactly as the first link to that
5880        // object was.
5881        let mut hard_links: Vec<HardLink> = Vec::new();
5882        for HardEntry {
5883            path,
5884            address: addr,
5885            ..
5886        } in alias_entries
5887        {
5888            let target = if let Some(di) = existing_datasets
5889                .iter()
5890                .position(|d| d.obj_header_addr == addr)
5891            {
5892                HardLinkTarget::Dataset(di)
5893            } else if let Some(gi) = groups
5894                .iter()
5895                .position(|g| g.obj_header_written_addr == Some(addr))
5896            {
5897                HardLinkTarget::Group(gi)
5898            } else {
5899                continue;
5900            };
5901            let (parent, link_name) = match path.rsplit_once('/') {
5902                None => (None, path),
5903                Some((dir, leaf)) => {
5904                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
5905                    (
5906                        group_index_map.get(&format!("/{dir}")).copied(),
5907                        leaf.to_string(),
5908                    )
5909                }
5910            };
5911            hard_links.push(HardLink {
5912                parent,
5913                name: link_name,
5914                target,
5915                creation_seq: 0,
5916            });
5917        }
5918
5919        // Attach every link the writer cannot express to the group that
5920        // holds it, so the rewrite of that group's header emits it again.
5921        // `ensure_groups_for` registers the parent chain, which matters for
5922        // a group whose only content is such a link: nothing else would put
5923        // it in the registry, and the close would drop group and link alike.
5924        let mut preserved_links: Vec<PreservedLink> = Vec::new();
5925        for PreservedEntry {
5926            path,
5927            class,
5928            encoded,
5929            reason,
5930            kind,
5931        } in preserved
5932        {
5933            let (parent, link_name) = match path.rsplit_once('/') {
5934                None => (None, path),
5935                Some((dir, leaf)) => {
5936                    ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
5937                    (
5938                        group_index_map.get(&format!("/{dir}")).copied(),
5939                        leaf.to_string(),
5940                    )
5941                }
5942            };
5943            preserved_links.push(PreservedLink {
5944                parent,
5945                name: link_name,
5946                class,
5947                encoded,
5948                reason,
5949                kind,
5950            });
5951        }
5952
5953        // Stamp the creation sequence a reopened file cannot supply. Nothing
5954        // on disk says which link was made first unless the group tracked
5955        // creation order, and this reader does not carry that back out, so
5956        // discovery order is what there is: datasets, then groups, then the
5957        // hard links found beside them — the order the writer emitted links
5958        // in before it ordered them at all.
5959        let mut creation_seq = 0u64;
5960        for d in &mut existing_datasets {
5961            d.creation_seq = creation_seq;
5962            creation_seq += 1;
5963        }
5964        for g in &mut groups {
5965            g.creation_seq = creation_seq;
5966            creation_seq += 1;
5967        }
5968        for l in &mut hard_links {
5969            l.creation_seq = creation_seq;
5970            creation_seq += 1;
5971        }
5972
5973        // The strategy is the file's, not this session's: a paged file
5974        // allocates on its own page grid however it was opened, `persist`
5975        // deciding only whether the managers survive the close.
5976        let allocator = FileAllocator::with_policy(
5977            file_size,
5978            ext.file_space_info
5979                .as_ref()
5980                .map_or(free_space::SpacePolicy::Aggr, |info| {
5981                    free_space::SpacePolicy::for_message(info)
5982                }),
5983        );
5984        // The sections the file's own managers recorded are free space, so
5985        // they are what this session allocates from first — `H5MF_alloc` asks
5986        // the free-space manager before it bumps the end of the file, and a
5987        // reopen that skipped this would grow a file that had room.
5988        allocator.reset_free_list(&reopened_free_space.sections);
5989
5990        // Now that every object has its registry index, key the dense storage
5991        // found on disk by the scope that will supersede it. A group the link
5992        // walk saw but never registered is not rewritten either, so leaving it
5993        // out is what keeps its storage referenced.
5994        let mut superseded = SupersededDense {
5995            attrs: dataset_dense
5996                .into_iter()
5997                .map(|(di, ainfo)| (AttrScope::Dataset(di), ainfo))
5998                .collect(),
5999            links: HashMap::new(),
6000        };
6001        superseded
6002            .attrs
6003            .extend(root_dense.attrs.map(|a| (AttrScope::Root, a)));
6004        superseded
6005            .links
6006            .extend(root_dense.links.map(|l| (LinkScope::Root, l)));
6007        for (name, dense) in group_dense {
6008            let Some(&gidx) = group_index_map.get(&format!("/{name}")) else {
6009                continue;
6010            };
6011            superseded
6012                .attrs
6013                .extend(dense.attrs.map(|a| (AttrScope::Group(gidx), a)));
6014            superseded
6015                .links
6016                .extend(dense.links.map(|l| (LinkScope::Group(gidx), l)));
6017        }
6018        let superseded = (!superseded.attrs.is_empty() || !superseded.links.is_empty())
6019            .then(|| Box::new(superseded));
6020
6021        // The same keying for the symbol-table storage. Built from the headers
6022        // alone, not from the superblock version: a group whose header carried
6023        // no Symbol Table message contributes nothing — what happens to a group
6024        // libhdf5 wrote at a newer bound inside an otherwise classic file — and
6025        // one that carried it keeps its storage even where the superblock is
6026        // version 2, which is what a file with shared messages is.
6027        let mut stabs: HashMap<LinkScope, StabExtents> = HashMap::new();
6028        stabs.extend(root_stab.map(|s| (LinkScope::Root, s)));
6029        for (name, extents) in group_stabs {
6030            if let Some(&gidx) = group_index_map.get(&format!("/{name}")) {
6031                stabs.insert(LinkScope::Group(gidx), extents);
6032            }
6033        }
6034        let symbol_tables = SymbolTables {
6035            found: stabs.keys().copied().collect(),
6036            superseded: Slot::new(stabs),
6037            written: Slot::new(HashMap::new()),
6038        };
6039
6040        // The superblock the close re-emits, and the generation every message
6041        // this session encodes belongs to.
6042        let legacy = legacy.map(|superblock| Box::new(LegacyFile { superblock }));
6043
6044        // Wrap the reconstructed plain vecs into the per-slot registry. The
6045        // reconstruction logic above runs single-threaded on local `Vec`s;
6046        // only the final hand-off needs the `Shared<Slot<_>>` shape.
6047        let datasets = existing_datasets
6048            .into_iter()
6049            .map(|i| Shared::new(DatasetCell::new(i)))
6050            .collect();
6051        let groups = groups
6052            .into_iter()
6053            .map(|g| Shared::new(Slot::new(g)))
6054            .collect();
6055
6056        let writer = Self {
6057            handle,
6058            allocator,
6059            ctx,
6060            datasets: Slot::new(datasets),
6061            groups: Slot::new(groups),
6062            hard_links: Slot::new(hard_links),
6063            // A reopen carries the soft and external links it found as
6064            // `preserved_links`, byte for byte; this list holds only the ones
6065            // created in this session.
6066            symbolic_links: Slot::new(Vec::new()),
6067            committed_datatypes: Slot::new(Vec::new()),
6068            preserved_links: Slot::new(preserved_links),
6069            name_index: Slot::new(Box::new(NameIndex::new())),
6070            root_attributes: Slot::new(root_attributes),
6071            create_lock: Slot::new(()),
6072            // A reopen names no bound: the file already is whichever
6073            // generation it is, and the version in its superblock is what
6074            // says so — see `libver_floor`. `set_libver_bound` is where a
6075            // caller asks for a newer one, exactly as `H5Fopen` takes a fapl.
6076            libver: None,
6077            closed: false,
6078            swmr_active: false,
6079            cwfs: Slot::new(Vec::new()),
6080            root_group_addr: None,
6081            root_group_encoded_size: 0,
6082            superseded_root_header: root_header_blocks,
6083            // The version the file already has. It is written back unchanged
6084            // and it floors every bound this session writes at, so the append
6085            // hands the file back in the generation it found it in.
6086            superblock_version: SuperblockVersion::Existing(version),
6087            // The reopened file's own policy, so objects added in this
6088            // session are made the way the file already declares.
6089            root_track_order,
6090            root_times,
6091            dense_attributes: Slot::new(HashMap::new()),
6092            dense_links: Slot::new(HashMap::new()),
6093            superseded_dense: Slot::new(superseded),
6094            track_order: root_track_order,
6095            // Not recovered from the file the way the creation-order policy
6096            // is: a version-1 header leaves no trace of whether the object was
6097            // tracking times, so there is nothing on disk to read the policy
6098            // back from. An object added to a reopened file gets this writer's
6099            // own default, the same one a created file starts at.
6100            track_times: false,
6101            next_creation_seq: Slot::new(creation_seq),
6102            pending_object_references: Slot::new(Vec::new()),
6103            pending_heap_references: Slot::new(Vec::new()),
6104            attribute_references: Slot::new(Vec::new()),
6105            legacy,
6106            symbol_tables,
6107            // The ranks the superblock or its extension declared, which every
6108            // v1-B-tree and symbol-table node this session writes is sized by.
6109            btree: meta.btree,
6110            extension,
6111            free_space: reopened_free_space.state,
6112            // The indexes the file was created with, and the blocks its
6113            // current table occupies; the next finalize lays a new table out
6114            // over them from the whole message set.
6115            sohm,
6116            source_dir: source_dir_of(path)?,
6117        };
6118        // The link graph is complete only now, so this is the first point the
6119        // count each on-disk header was written with can be read off it: in a
6120        // well-formed file the links the walk found reaching an object *are*
6121        // that count, so nothing has to be decoded out of the headers.
6122        for i in 0..writer.dataset_count() {
6123            let nlink = writer.object_link_count(HardLinkTarget::Dataset(i));
6124            writer.ds(i).lock().nlink_written = nlink;
6125        }
6126        Ok(writer)
6127    }
6128
6129    /// Return the names of all datasets created so far.
6130    pub fn dataset_names(&self) -> Vec<String> {
6131        self.dataset_refs()
6132            .iter()
6133            .filter_map(|d| {
6134                let g = d.lock();
6135                (!g.deleted).then(|| g.name.clone())
6136            })
6137            .collect()
6138    }
6139
6140    /// Find a dataset index by name. Like `H5Dopen`, the name may be any
6141    /// link path to the dataset: a user hard link's path — or a path
6142    /// whose group components pass through such links — resolves to its
6143    /// target.
6144    pub fn dataset_index(&self, name: &str) -> Option<usize> {
6145        let name = self.canonical_dataset_path(name);
6146        self.dataset_refs()
6147            .iter()
6148            .position(|d| {
6149                let g = d.lock();
6150                g.name == name && !g.deleted
6151            })
6152            .or_else(|| {
6153                self.hard_links_vec().iter().find_map(|l| match l.target {
6154                    HardLinkTarget::Dataset(i)
6155                        if self.hard_link_emitted(l) && self.hard_link_full_path(l) == name =>
6156                    {
6157                        Some(i)
6158                    }
6159                    _ => None,
6160                })
6161            })
6162    }
6163
6164    /// Reconstruct the fields a writer-mode `H5Dataset` handle needs for the
6165    /// dataset at `index`, and open it under `access`. Single owner of this
6166    /// mapping so `H5File::dataset_writer`, `H5Group::dataset_writer`, and
6167    /// the vlen-string helpers all agree — including on
6168    /// [`bind_efile_prefix`](Self::bind_efile_prefix), which no handle site
6169    /// can then forget to run.
6170    pub(crate) fn dataset_handle_parts(
6171        &self,
6172        index: usize,
6173        access: &DatasetAccess,
6174    ) -> IoResult<DatasetHandleParts> {
6175        let open = self.bind_efile_prefix(index, access)?;
6176        let ds = self.ds(index);
6177        let g = ds.lock();
6178        Ok(DatasetHandleParts {
6179            shape: g.dataspace.dims.iter().map(|&d| d as usize).collect(),
6180            element_size: g.datatype.element_size() as usize,
6181            chunk_index: g.chunk_index_kind(),
6182            open,
6183        })
6184    }
6185
6186    /// Put `access`'s external file prefix in force for the dataset at
6187    /// `index`, or join the open that already settled one.
6188    ///
6189    /// INVARIANT: every write of an externally stored dataset's raw bytes
6190    /// joins its slot names against the prefix an *open* settled, and this is
6191    /// the only place that settles one. `write_contiguous_bytes` reads it and
6192    /// nothing else writes it, so a write cannot resolve a prefix of its own
6193    /// and land bytes where a read under the same properties would not look
6194    /// for them.
6195    ///
6196    /// First open wins, and a joining open may not disagree: `H5D__open_name`
6197    /// compares its own expanded prefix against the open dataset's and fails
6198    /// when they differ (H5Dint.c:1533-1545). Measured under libhdf5 1.14.6
6199    /// and 2.0.0, with a dataset created through a dapl naming a directory
6200    /// and its handle still alive: a second open naming another directory is
6201    /// refused, one naming the same directory joins, one naming none is
6202    /// refused too, and with `HDF5_EXTFILE_PREFIX` set — which shadows every
6203    /// property, so all three expand alike — none of them is. Dropping every
6204    /// handle releases the answer and the next open settles it afresh, which
6205    /// the same measurement confirms.
6206    ///
6207    /// Returns the token that keeps the open alive, `None` for a dataset
6208    /// whose raw data is in this file and which therefore has no prefix to
6209    /// agree about.
6210    pub(crate) fn bind_efile_prefix(
6211        &self,
6212        index: usize,
6213        access: &DatasetAccess,
6214    ) -> IoResult<Option<crate::io::reader::DatasetOpenToken>> {
6215        let ds = self.ds(index);
6216        let mut g = ds.lock();
6217        let source_dir = &self.source_dir;
6218        let Some(ext) = g.external.as_mut() else {
6219            return Ok(None);
6220        };
6221        let want =
6222            crate::io::reader::resolve_extfile_prefix(access.efile_prefix_value(), source_dir);
6223        if let Some(open) = ext.prefix.open.upgrade() {
6224            if ext.prefix.expanded != want {
6225                let name = g.name.clone();
6226                return Err(crate::io::IoError::InvalidState(format!(
6227                    "dataset {name:?} is already open under a different external file                      prefix, and libhdf5 refuses to join an open that disagrees about one"
6228                )));
6229            }
6230            return Ok(Some(open));
6231        }
6232        let token: crate::io::reader::DatasetOpenToken = std::sync::Arc::new(());
6233        ext.prefix = EfilePrefix {
6234            expanded: want,
6235            open: std::sync::Arc::downgrade(&token),
6236        };
6237        Ok(Some(token))
6238    }
6239
6240    /// Reject a name some other link in the file already occupies.
6241    ///
6242    /// `name` is the registry's full-path form, with no leading `/`. HDF5
6243    /// requires link names to be unique within their group, and every kind of
6244    /// link this writer can emit competes for the same name: a dataset's own
6245    /// link, a group's, a user hard link, a soft or external link, and a link
6246    /// a reopen is carrying through verbatim. This is the one place that list
6247    /// is written down, so a creator cannot be blind to a kind it does not
6248    /// itself make — nor a kind added after it.
6249    fn ensure_name_free(&self, name: &str) -> IoResult<()> {
6250        let holder = self.name_holder(name);
6251        // The index is a filter over the registries, not a second copy of
6252        // them, so a debug build re-derives the answer on every create: a
6253        // name it failed to record surfaces as a failing assertion in the
6254        // suite rather than as two links of one name in somebody's file.
6255        #[cfg(debug_assertions)]
6256        assert_eq!(
6257            holder,
6258            self.scan_name_holder(name),
6259            "the name index disagrees with the registries for '{name}'"
6260        );
6261        match holder {
6262            None => Ok(()),
6263            Some(kind) => Err(crate::io::IoError::InvalidState(format!(
6264                "a {kind} named '{name}' already exists"
6265            ))),
6266        }
6267    }
6268
6269    /// What already holds `name`, or `None` if it is free.
6270    ///
6271    /// The kinds answer in a fixed order — dataset, group, committed
6272    /// datatype, hard link, symbolic link, preserved link — because the
6273    /// refusal names the first one that holds it. [`NameIndex`] narrows each
6274    /// kind to the entries that ever took this name; every candidate is then
6275    /// put through the same predicate the full scan used, so a hit left
6276    /// behind by a delete or a rename answers exactly as an absent one does.
6277    fn name_holder(&self, name: &str) -> Option<&'static str> {
6278        self.build_name_index();
6279        let hits: Vec<NameHit> = {
6280            let index = self.name_index.lock();
6281            index.map.as_ref().and_then(|m| m.get(name))?.clone()
6282        };
6283        for hit in &hits {
6284            if let NameHit::Dataset(i) = *hit {
6285                let ds = self.ds(i);
6286                let d = ds.lock();
6287                if !d.deleted && d.name == name {
6288                    return Some("dataset");
6289                }
6290            }
6291        }
6292        for hit in &hits {
6293            if let NameHit::Group(i) = *hit {
6294                let grp = self.grp(i);
6295                let g = grp.lock();
6296                if !g.deleted && g.name.trim_start_matches('/') == name {
6297                    return Some("group");
6298                }
6299            }
6300        }
6301        for hit in &hits {
6302            if let NameHit::Datatype(i) = *hit {
6303                // The registry lock goes before `parent_alive` takes a group
6304                // slot, never across it.
6305                let (parent, held) = {
6306                    let reg = self.committed_datatypes.lock();
6307                    (reg[i].parent, reg[i].name == name)
6308                };
6309                if held && self.parent_alive(parent) {
6310                    return Some("committed datatype");
6311                }
6312            }
6313        }
6314        if hits.contains(&NameHit::HardLink)
6315            && self
6316                .hard_links_vec()
6317                .iter()
6318                .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6319        {
6320            return Some("hard link");
6321        }
6322        if hits.contains(&NameHit::SymbolicLink)
6323            && self
6324                .symbolic_links_vec()
6325                .iter()
6326                .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6327        {
6328            return Some("link");
6329        }
6330        // A preserved link occupies its name in the group just as a modelled
6331        // one does; both are emitted, and two link messages of one name in a
6332        // group is an invalid file.
6333        if hits.contains(&NameHit::PreservedLink)
6334            && self.preserved_link_paths().iter().any(|(p, _)| *p == name)
6335        {
6336            return Some("link");
6337        }
6338        None
6339    }
6340
6341    /// The same answer read straight off the registries, which is what the
6342    /// index is checked against in a debug build.
6343    #[cfg(debug_assertions)]
6344    fn scan_name_holder(&self, name: &str) -> Option<&'static str> {
6345        if self.dataset_refs().iter().any(|d| {
6346            let g = d.lock();
6347            !g.deleted && g.name == name
6348        }) {
6349            return Some("dataset");
6350        }
6351        if self.group_refs().iter().any(|g| {
6352            let gg = g.lock();
6353            !gg.deleted && gg.name.trim_start_matches('/') == name
6354        }) {
6355            return Some("group");
6356        }
6357        if self
6358            .committed_datatypes_vec()
6359            .iter()
6360            .any(|c| self.parent_alive(c.parent) && c.name == name)
6361        {
6362            return Some("committed datatype");
6363        }
6364        if self
6365            .hard_links_vec()
6366            .iter()
6367            .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6368        {
6369            return Some("hard link");
6370        }
6371        if self
6372            .symbolic_links_vec()
6373            .iter()
6374            .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6375        {
6376            return Some("link");
6377        }
6378        if self.preserved_link_paths().iter().any(|(p, _)| *p == name) {
6379            return Some("link");
6380        }
6381        None
6382    }
6383
6384    /// Build the name index unless it is already built.
6385    ///
6386    /// The walk takes the registry spines and their slots, so it runs with no
6387    /// index lock held — the writer never holds one lock across another — and
6388    /// the result is kept only if nothing renamed, created or unlinked
6389    /// anything while it ran.
6390    fn build_name_index(&self) {
6391        let epoch = {
6392            let index = self.name_index.lock();
6393            if index.map.is_some() {
6394                return;
6395            }
6396            index.epoch
6397        };
6398        let mut map: HashMap<String, Vec<NameHit>> = HashMap::new();
6399        for (i, ds) in self.dataset_refs().iter().enumerate() {
6400            let d = ds.lock();
6401            if !d.deleted {
6402                map.entry(d.name.clone())
6403                    .or_default()
6404                    .push(NameHit::Dataset(i));
6405            }
6406        }
6407        for (i, grp) in self.group_refs().iter().enumerate() {
6408            let g = grp.lock();
6409            if !g.deleted {
6410                map.entry(g.name.trim_start_matches('/').to_string())
6411                    .or_default()
6412                    .push(NameHit::Group(i));
6413            }
6414        }
6415        for (i, c) in self.committed_datatypes_vec().iter().enumerate() {
6416            map.entry(c.name.clone())
6417                .or_default()
6418                .push(NameHit::Datatype(i));
6419        }
6420        for l in self.hard_links_vec().iter() {
6421            map.entry(self.hard_link_full_path(l))
6422                .or_default()
6423                .push(NameHit::HardLink);
6424        }
6425        for l in self.symbolic_links_vec().iter() {
6426            map.entry(self.symbolic_link_full_path(l))
6427                .or_default()
6428                .push(NameHit::SymbolicLink);
6429        }
6430        for (path, _) in self.preserved_link_paths() {
6431            map.entry(path).or_default().push(NameHit::PreservedLink);
6432        }
6433        let mut index = self.name_index.lock();
6434        if index.map.is_none() && index.epoch == epoch {
6435            index.map = Some(map);
6436        }
6437    }
6438
6439    /// Record that `hit` now holds `name` — the one way a new name enters the
6440    /// index, called from every push that gives a registry entry a name.
6441    fn register_name(&self, name: &str, hit: NameHit) {
6442        self.name_index.lock().insert(name, hit);
6443    }
6444
6445    /// Drop the index because something moved names wholesale (a group
6446    /// rename carries its subtree and every link path under it).
6447    fn forget_name_index(&self) {
6448        self.name_index.lock().forget();
6449    }
6450
6451    /// Delete a dataset name, with libhdf5's `H5Ldelete` semantics: a name
6452    /// is only a link. If `name` is a user hard link's path, just that
6453    /// link is removed and the object is untouched. If it is the tree name
6454    /// and a user hard link still names the object, the object survives
6455    /// under it — the link becomes the primary name and nothing is freed.
6456    /// Only deleting the *last* name soft-deletes the object and frees the
6457    /// file space it owned: its chunk blocks and chunk-index structures
6458    /// (or contiguous data block), the global-heap objects of its
6459    /// variable-length data and attributes, and — on a reopened file — the
6460    /// on-disk object header block. The freed space is reused by later
6461    /// allocations in this session; the file does not shrink.
6462    ///
6463    /// Refused while SWMR streaming is active: a live reader may hold any
6464    /// of those addresses (libhdf5 forbids link deletion during SWMR
6465    /// writes too).
6466    pub fn delete_dataset(&self, name: &str) -> IoResult<()> {
6467        if self.swmr_active {
6468            return Err(swmr_delete_error(name));
6469        }
6470        self.reject_external_traversal(name)?;
6471        // The gate keeps the link list and child lists still while this
6472        // delete reads and rewrites them (create_lock → op → slot order,
6473        // the same as every creator).
6474        let _create = self.create_lock.lock();
6475        // `H5Ldelete` resolves the path through links only *up to* the
6476        // leaf — the leaf is what gets deleted, so a leaf naming a user
6477        // link must stay literal and be unlinked, not its target.
6478        let name = match name.rsplit_once('/') {
6479            None => name.to_string(),
6480            Some((dir, leaf)) => format!(
6481                "{}/{leaf}",
6482                self.canonical_group_path(&format!("/{dir}"))
6483                    .trim_start_matches('/')
6484            ),
6485        };
6486        let refs = self.dataset_refs();
6487        let idx = match refs.iter().position(|d| {
6488            let g = d.lock();
6489            g.name == name && !g.deleted
6490        }) {
6491            Some(i) => i,
6492            None => {
6493                // Not a tree name — the path may name a user hard link,
6494                // and deleting a link path unlinks just that link (the
6495                // creation collision checks keep the two namespaces
6496                // disjoint, so the order of the lookups cannot matter).
6497                let link = self.hard_links_vec().iter().position(|l| {
6498                    self.hard_link_emitted(l)
6499                        && matches!(l.target, HardLinkTarget::Dataset(_))
6500                        && self.hard_link_full_path(l) == name
6501                });
6502                let Some(pos) = link else {
6503                    return Err(crate::io::IoError::NotFound(name));
6504                };
6505                self.hard_links.lock().remove(pos);
6506                return Ok(());
6507            }
6508        };
6509        // A surviving hard link keeps the object: promote the first one to
6510        // the primary name and delete nothing.
6511        let promote = self.hard_links_vec().iter().position(|l| {
6512            self.hard_link_emitted(l) && matches!(l.target, HardLinkTarget::Dataset(i) if i == idx)
6513        });
6514        if let Some(pos) = promote {
6515            self.promote_dataset_to_link(idx, pos);
6516            return Ok(());
6517        }
6518        refs[idx].lock().deleted = true;
6519        // Remove from parent group's child_datasets
6520        for grp in self.group_refs() {
6521            grp.lock().child_datasets.retain(|&di| di != idx);
6522        }
6523        self.purge_dead_links();
6524        let ds = self.ds(idx);
6525        let _op = ds.op.lock();
6526        self.release_dataset_storage(idx)
6527    }
6528
6529    /// Soft-delete a group and all its child datasets and sub-groups,
6530    /// freeing every deleted object's file space the way
6531    /// [`delete_dataset`](Self::delete_dataset) does — with the same
6532    /// `H5Ldelete` semantics: a `name` that is a user hard link's path
6533    /// unlinks just that link, and hard links from *outside* the subtree
6534    /// keep their targets. A dataset or group such a link names survives,
6535    /// re-homed under the link (a group brings its whole subtree with
6536    /// it); a link naming the deleted group itself turns the call into a
6537    /// pure rename and nothing is freed. Refused while SWMR streaming is
6538    /// active, same rule as `delete_dataset`.
6539    pub fn delete_group(&self, name: &str) -> IoResult<()> {
6540        if self.swmr_active {
6541            return Err(swmr_delete_error(name));
6542        }
6543        self.reject_external_traversal(name)?;
6544        // Same gate as `delete_dataset`: the pre-scan below and the
6545        // promotions must see a still link list and child lists.
6546        let _create = self.create_lock.lock();
6547        let name = if name.starts_with('/') {
6548            name.to_string()
6549        } else {
6550            format!("/{}", name)
6551        };
6552        // Leaf stays literal, directory resolves through links — the
6553        // same `H5Ldelete` rule as `delete_dataset`.
6554        let name = match name.rsplit_once('/') {
6555            Some((dir, leaf)) if !dir.is_empty() => {
6556                format!("{}/{leaf}", self.canonical_group_path(dir))
6557            }
6558            _ => name,
6559        };
6560        let groups = self.group_refs();
6561        let gidx = match groups.iter().position(|g| {
6562            let gg = g.lock();
6563            gg.name == name && !gg.deleted
6564        }) {
6565            Some(i) => i,
6566            None => {
6567                // Same `H5Ldelete` rule as `delete_dataset`: a path naming
6568                // a user hard link to a group unlinks just that link.
6569                let trimmed = name.trim_start_matches('/');
6570                let link = self.hard_links_vec().iter().position(|l| {
6571                    self.hard_link_emitted(l)
6572                        && matches!(l.target, HardLinkTarget::Group(_))
6573                        && self.hard_link_full_path(l) == trimmed
6574                });
6575                let Some(pos) = link else {
6576                    return Err(crate::io::IoError::NotFound(name.clone()));
6577                };
6578                self.hard_links.lock().remove(pos);
6579                return Ok(());
6580            }
6581        };
6582
6583        // A link is "outside" when its parent group does not die with the
6584        // subtree; only outside links can keep their targets alive.
6585        fn outside(parent: Option<usize>, doomed_gs: &[usize]) -> bool {
6586            match parent {
6587                None => true,
6588                Some(pi) => !doomed_gs.contains(&pi),
6589            }
6590        }
6591        // A group an outside link names survives, re-homed with its whole
6592        // subtree under the link. Each promotion moves that subtree out of
6593        // the doomed set — and can turn a link inside it into an outside
6594        // one — so rescan from scratch until no promotable group is left.
6595        // Promoting `gidx` itself makes the delete a pure rename: return.
6596        let mut doomed_ds = Vec::new();
6597        let mut doomed_gs = Vec::new();
6598        loop {
6599            doomed_ds.clear();
6600            doomed_gs.clear();
6601            self.collect_live_subtree(gidx, &mut doomed_ds, &mut doomed_gs);
6602            let promote = self
6603                .hard_links_vec()
6604                .iter()
6605                .enumerate()
6606                .find_map(|(pos, l)| match l.target {
6607                    HardLinkTarget::Group(gi)
6608                        if self.hard_link_emitted(l)
6609                            && outside(l.parent, &doomed_gs)
6610                            && doomed_gs.contains(&gi) =>
6611                    {
6612                        Some((pos, gi))
6613                    }
6614                    _ => None,
6615                });
6616            let Some((pos, gi)) = promote else { break };
6617            self.promote_group_to_link(gi, pos);
6618            if gi == gidx {
6619                return Ok(());
6620            }
6621        }
6622        // A dataset an outside link names survives its container: re-home
6623        // it under the link now, so the marking pass below never sees it.
6624        for di in doomed_ds {
6625            let promote = self.hard_links_vec().iter().position(|l| {
6626                self.hard_link_emitted(l)
6627                    && outside(l.parent, &doomed_gs)
6628                    && matches!(l.target, HardLinkTarget::Dataset(i) if i == di)
6629            });
6630            if let Some(pos) = promote {
6631                self.promote_dataset_to_link(di, pos);
6632            }
6633        }
6634
6635        let mut ds_deleted = Vec::new();
6636        let mut gs_deleted = Vec::new();
6637        self.delete_group_recursive(gidx, &mut ds_deleted, &mut gs_deleted);
6638        // Remove from parent's child_groups
6639        let parent = groups[gidx].lock().parent;
6640        if let Some(pidx) = parent {
6641            groups[pidx].lock().child_groups.retain(|&gi| gi != gidx);
6642        }
6643        self.purge_dead_links();
6644        // Free storage only after the whole subtree is marked: the lists
6645        // hold each object exactly once (the marking pass skips anything
6646        // already deleted), so nothing is freed twice.
6647        for di in ds_deleted {
6648            let ds = self.ds(di);
6649            let _op = ds.op.lock();
6650            self.release_dataset_storage(di)?;
6651        }
6652        for gi in gs_deleted {
6653            self.release_group_storage(gi)?;
6654        }
6655        Ok(())
6656    }
6657
6658    /// Collect the live (not soft-deleted) members of `gidx`'s subtree,
6659    /// each exactly once, without changing anything — the read-only twin
6660    /// of [`delete_group_recursive`](Self::delete_group_recursive), for
6661    /// the pre-scan that must run before any marking.
6662    fn collect_live_subtree(&self, gidx: usize, ds_out: &mut Vec<usize>, gs_out: &mut Vec<usize>) {
6663        if gs_out.contains(&gidx) {
6664            return;
6665        }
6666        let (child_ds, child_gs) = {
6667            let grp = self.grp(gidx);
6668            let g = grp.lock();
6669            if g.deleted {
6670                return;
6671            }
6672            (g.child_datasets.clone(), g.child_groups.clone())
6673        };
6674        gs_out.push(gidx);
6675        for di in child_ds {
6676            if !self.ds(di).lock().deleted && !ds_out.contains(&di) {
6677                ds_out.push(di);
6678            }
6679        }
6680        for gi in child_gs {
6681            self.collect_live_subtree(gi, ds_out, gs_out);
6682        }
6683    }
6684
6685    /// Re-home dataset `idx` under the hard link at `pos` in the link
6686    /// list — the surviving half of `H5Ldelete`: the link leaves the user
6687    /// list and becomes the dataset's primary (tree) name, in the link's
6688    /// parent group. Storage is untouched; any further links to the
6689    /// dataset stay in the list and keep resolving.
6690    fn promote_dataset_to_link(&self, idx: usize, pos: usize) {
6691        let link = self.hard_links.lock().remove(pos);
6692        let new_name = self.hard_link_full_path(&link);
6693        for grp in self.group_refs() {
6694            grp.lock().child_datasets.retain(|&di| di != idx);
6695        }
6696        if let Some(pi) = link.parent {
6697            self.grp(pi).lock().child_datasets.push(idx);
6698        }
6699        self.ds(idx).lock().name = new_name.clone();
6700        self.register_name(&new_name, NameHit::Dataset(idx));
6701    }
6702
6703    /// The group counterpart of
6704    /// [`promote_dataset_to_link`](Self::promote_dataset_to_link): re-home
6705    /// group `gidx` under the hard link at `pos`, bringing its whole
6706    /// subtree with it. Names are stored as full paths, so every live
6707    /// descendant is renamed by prefix.
6708    fn promote_group_to_link(&self, gidx: usize, pos: usize) {
6709        let link = self.hard_links.lock().remove(pos);
6710        let new_name = format!("/{}", self.hard_link_full_path(&link));
6711        let old_name = self.grp(gidx).lock().name.clone();
6712        for grp in self.group_refs() {
6713            grp.lock().child_groups.retain(|&g| g != gidx);
6714        }
6715        {
6716            let grp = self.grp(gidx);
6717            let mut g = grp.lock();
6718            g.parent = link.parent;
6719            g.name = new_name.clone();
6720        }
6721        if let Some(pi) = link.parent {
6722            self.grp(pi).lock().child_groups.push(gidx);
6723        }
6724
6725        let mut ds_in = Vec::new();
6726        let mut gs_in = Vec::new();
6727        self.collect_live_subtree(gidx, &mut ds_in, &mut gs_in);
6728        // Group names carry a leading '/' ("/a/b"), dataset names none
6729        // ("a/b/ds") — two prefix forms of the same rename.
6730        let old_grp_prefix = format!("{old_name}/");
6731        let new_grp_prefix = format!("{new_name}/");
6732        let old_ds_prefix = old_grp_prefix.trim_start_matches('/').to_string();
6733        let new_ds_prefix = new_grp_prefix.trim_start_matches('/').to_string();
6734        for gi in gs_in {
6735            if gi == gidx {
6736                continue;
6737            }
6738            let grp = self.grp(gi);
6739            let mut g = grp.lock();
6740            let renamed = g
6741                .name
6742                .strip_prefix(&old_grp_prefix)
6743                .map(|rest| format!("{new_grp_prefix}{rest}"));
6744            if let Some(n) = renamed {
6745                g.name = n;
6746            }
6747        }
6748        for di in ds_in {
6749            let ds = self.ds(di);
6750            let mut d = ds.lock();
6751            let renamed = d
6752                .name
6753                .strip_prefix(&old_ds_prefix)
6754                .map(|rest| format!("{new_ds_prefix}{rest}"));
6755            if let Some(n) = renamed {
6756                d.name = n;
6757            }
6758        }
6759        // A group carries its subtree and every link path under it, so far
6760        // more names moved than this function can enumerate: start over.
6761        self.forget_name_index();
6762    }
6763
6764    /// Drop link entries that can no longer be emitted — their parent group
6765    /// or, for a hard link, their target object was just deleted — so the
6766    /// lists mirror what the file will hold instead of carrying suppressed
6767    /// zombies. Both kinds are purged here so a delete cannot clear one list
6768    /// and leave the other holding a name in a group that is gone.
6769    fn purge_dead_links(&self) {
6770        let dead: Vec<usize> = self
6771            .hard_links_vec()
6772            .iter()
6773            .enumerate()
6774            .filter(|(_, l)| !self.hard_link_emitted(l))
6775            .map(|(p, _)| p)
6776            .collect();
6777        let mut links = self.hard_links.lock();
6778        for p in dead.into_iter().rev() {
6779            links.remove(p);
6780        }
6781        drop(links);
6782
6783        let dead: Vec<usize> = self
6784            .symbolic_links_vec()
6785            .iter()
6786            .enumerate()
6787            .filter(|(_, l)| !self.symbolic_link_emitted(l))
6788            .map(|(p, _)| p)
6789            .collect();
6790        let mut links = self.symbolic_links.lock();
6791        for p in dead.into_iter().rev() {
6792            links.remove(p);
6793        }
6794    }
6795
6796    /// Mark `gidx` and its subtree deleted, appending each newly-deleted
6797    /// object's index to `ds_out` / `gs_out` exactly once — the caller
6798    /// frees their storage, and an object reachable twice (or a subtree
6799    /// already deleted) must not be freed twice.
6800    fn delete_group_recursive(
6801        &self,
6802        gidx: usize,
6803        ds_out: &mut Vec<usize>,
6804        gs_out: &mut Vec<usize>,
6805    ) {
6806        // Mark deleted and snapshot the child lists, releasing the group lock
6807        // before locking any dataset/child-group slot (spine → slot order).
6808        let (child_ds, child_gs) = {
6809            let grp = self.grp(gidx);
6810            let mut g = grp.lock();
6811            if g.deleted {
6812                return;
6813            }
6814            g.deleted = true;
6815            (g.child_datasets.clone(), g.child_groups.clone())
6816        };
6817        gs_out.push(gidx);
6818        for di in child_ds {
6819            let ds = self.ds(di);
6820            let mut d = ds.lock();
6821            if !d.deleted {
6822                d.deleted = true;
6823                ds_out.push(di);
6824            }
6825        }
6826        for gi in child_gs {
6827            self.delete_group_recursive(gi, ds_out, gs_out);
6828        }
6829    }
6830
6831    /// Free everything a soft-deleted dataset owned. The single owner of
6832    /// delete-time reclamation, called only from the two delete paths with
6833    /// the dataset already marked deleted and its op lock held.
6834    ///
6835    /// A deleted dataset contributes nothing to finalize (the header,
6836    /// index-flush and append-flush loops all skip it), so nothing in the
6837    /// finalized file can reference the blocks freed here. Never runs under
6838    /// SWMR — the delete entry points refuse first.
6839    fn release_dataset_storage(&self, index: usize) -> IoResult<()> {
6840        use crate::format::messages::datatype::DatatypeMessage;
6841        let (indexed, ndims, contiguous, is_vlen, attrs, header_blocks, mapping_list) = {
6842            let ds = self.ds(index);
6843            let mut m = ds.lock();
6844            // Buffered rows were never written to a chunk; they die with
6845            // the dataset instead of being flushed at close.
6846            m.append = None;
6847            let indexed = m.is_chunked();
6848            let contiguous = (!indexed && m.data_addr != UNDEF_ADDR && m.data_size > 0)
6849                .then_some((m.data_addr, m.data_size));
6850            m.data_addr = UNDEF_ADDR;
6851            m.data_size = 0;
6852            // The external files themselves are the application's, not this
6853            // file's, and neither is the name heap freed: `H5O_MSG_EFL`
6854            // installs no file-delete method, so libhdf5 leaves the heap block
6855            // behind too. Dropping the list is what stops a deleted dataset
6856            // still claiming storage.
6857            m.external = None;
6858            // The mapping list is this file's own metadata, so unlike the
6859            // external files above it *is* freed — `H5D__virtual_delete`
6860            // removes the heap object. The source datasets it named are
6861            // another file's and are left alone.
6862            let mapping_list = m
6863                .virtual_storage
6864                .take()
6865                .and_then(|v| u16::try_from(v.heap_index).ok().map(|i| (v.heap_addr, i)));
6866            let is_vlen = matches!(
6867                m.datatype,
6868                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
6869            );
6870            let attrs = std::mem::take(&mut m.attributes);
6871            m.obj_header_written_addr = None;
6872            let header_blocks = std::mem::take(&mut m.obj_header_blocks);
6873            (
6874                indexed,
6875                m.dataspace.dims.len(),
6876                contiguous,
6877                is_vlen,
6878                attrs,
6879                header_blocks,
6880                mapping_list,
6881            )
6882        };
6883        if let Some((addr, idx)) = mapping_list {
6884            self.remove_heap_objects([(addr, vec![idx])].into_iter().collect())?;
6885        }
6886        if indexed {
6887            // Prune to a zero extent: every stored chunk is entirely beyond
6888            // it, so the walk frees each chunk block and collects the vlen
6889            // references its bytes held (released inside).
6890            self.prune_chunks_beyond(index, &vec![0; ndims])?;
6891            self.free_chunk_index(index)?;
6892        } else if let Some((addr, size)) = contiguous {
6893            if is_vlen {
6894                let data = self.handle.read_at(addr, size as usize)?;
6895                self.release_vlen_references(&data)?;
6896            }
6897            self.allocator.free(addr, size, FreeSpaceClass::RawData);
6898        }
6899        for attr in &attrs {
6900            self.release_attr_vlen(attr)?;
6901        }
6902        self.release_superseded_dense_attrs(AttrScope::Dataset(index))?;
6903        for (addr, size) in header_blocks {
6904            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
6905        }
6906        Ok(())
6907    }
6908
6909    /// Free a deleted group's file space: its attributes' global-heap
6910    /// objects and, on a reopened file, the on-disk header block. The
6911    /// group counterpart of
6912    /// [`release_dataset_storage`](Self::release_dataset_storage).
6913    fn release_group_storage(&self, gidx: usize) -> IoResult<()> {
6914        let (attrs, header_blocks) = {
6915            let grp = self.grp(gidx);
6916            let mut g = grp.lock();
6917            let attrs = std::mem::take(&mut g.attributes);
6918            g.obj_header_written_addr = None;
6919            (attrs, std::mem::take(&mut g.obj_header_blocks))
6920        };
6921        for attr in &attrs {
6922            self.release_attr_vlen(attr)?;
6923        }
6924        self.release_superseded_dense_attrs(AttrScope::Group(gidx))?;
6925        self.release_superseded_dense_links(LinkScope::Group(gidx))?;
6926        for (addr, size) in header_blocks {
6927            self.allocator.free(addr, size, FreeSpaceClass::Metadata);
6928        }
6929        Ok(())
6930    }
6931
6932    /// Free the dense attribute storage a reopened header names, once, when
6933    /// this session stops naming it — because the header is being rewritten
6934    /// around fresh storage, or because the object was deleted.
6935    ///
6936    /// The single owner of that transition: nothing else removes an attribute
6937    /// entry from [`superseded_dense`](Self::superseded_dense), and this
6938    /// removes it as it frees, so no heap is freed twice or left half freed.
6939    /// An object whose storage was compact, or whose header this session
6940    /// keeps, has no entry and nothing happens.
6941    ///
6942    /// Never under SWMR: a live reader may still be walking the storage the
6943    /// published headers name, the same rule the superseded-header and
6944    /// relocated-chunk paths follow. The entry stays in place, unfreed.
6945    fn release_superseded_dense_attrs(&self, scope: AttrScope) -> IoResult<()> {
6946        if self.swmr_active {
6947            return Ok(());
6948        }
6949        let taken = self
6950            .superseded_dense
6951            .lock()
6952            .as_mut()
6953            .and_then(|s| s.attrs.remove(&scope));
6954        let Some(ainfo) = taken else {
6955            return Ok(());
6956        };
6957        self.release_dense_storage(
6958            ainfo.fractal_heap_address,
6959            ainfo.name_btree_address,
6960            ainfo.creation_order_btree_address,
6961        )
6962    }
6963
6964    /// The link counterpart of
6965    /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs),
6966    /// under the same invariant and the same SWMR rule. Split from it because
6967    /// the two are superseded at different points of a finalize: attribute
6968    /// storage before the object headers are laid out, link storage after
6969    /// every one of them has an address.
6970    fn release_superseded_dense_links(&self, scope: LinkScope) -> IoResult<()> {
6971        if self.swmr_active {
6972            return Ok(());
6973        }
6974        let taken = self
6975            .superseded_dense
6976            .lock()
6977            .as_mut()
6978            .and_then(|s| s.links.remove(&scope));
6979        let Some(linfo) = taken else {
6980            return Ok(());
6981        };
6982        self.release_dense_storage(
6983            linfo.fractal_heap_address,
6984            linfo.name_btree_address,
6985            linfo.creation_order_btree_address,
6986        )
6987    }
6988
6989    /// Return one dense storage's file space to the allocator: the fractal
6990    /// heap in full, its name index, and the creation-order index when the
6991    /// object had one.
6992    ///
6993    /// The extents come from walking the structures themselves rather than
6994    /// from re-deriving what a writer would have allocated, so storage
6995    /// libhdf5 laid out is freed as accurately as storage this crate wrote.
6996    /// Every walk here already ran once this session — the reopen read every
6997    /// attribute out of this heap through the same index — so a failure means
6998    /// the file changed underneath us, and surfacing it beats freeing a
6999    /// partial extent list.
7000    fn release_dense_storage(
7001        &self,
7002        heap_addr: u64,
7003        name_bt2_addr: u64,
7004        corder_bt2_addr: Option<u64>,
7005    ) -> IoResult<()> {
7006        use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
7007        use crate::format::fractal_heap::collect_heap_extents;
7008
7009        let mut reader = crate::io::reader::HandleBlockReader {
7010            handle: &self.handle,
7011        };
7012        let mut extents = Vec::new();
7013        if heap_addr != UNDEF_ADDR {
7014            extents.extend(collect_heap_extents(heap_addr, &self.ctx, &mut reader)?);
7015        }
7016        for addr in [Some(name_bt2_addr), corder_bt2_addr]
7017            .into_iter()
7018            .flatten()
7019            .filter(|&a| a != UNDEF_ADDR)
7020        {
7021            extents.extend(collect_btree_v2_extents(addr, &self.ctx, &mut reader)?);
7022        }
7023        for (addr, len) in extents {
7024            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
7025        }
7026        Ok(())
7027    }
7028
7029    /// Free a deleted dataset's chunk-index structures, after the chunks
7030    /// themselves were freed by a zero-extent prune. Takes the index info
7031    /// out of the slot, so the dataset no longer claims chunked storage.
7032    ///
7033    /// Every block's size is recovered the way its allocation computed it:
7034    /// re-encoding the in-memory copy (EA header and index block, FA
7035    /// header and data block, BT2 header) or sizing a same-shape dummy
7036    /// from the array geometry (EA data blocks, whose element counts come
7037    /// from [`EaGeometry`]; BT2 nodes are all `node_size`).
7038    fn free_chunk_index(&self, index: usize) -> IoResult<()> {
7039        let ds = self.ds(index);
7040        let mut m = ds.lock();
7041        let is_filtered = m.filter_pipeline.is_some();
7042        if let Some(c) = m.chunked.take() {
7043            let p = &c.earray_params;
7044            let bits = p.max_nelmts_bits;
7045            let csl = c.chunk_size_len;
7046            let geo = EaGeometry::new(
7047                p.idx_blk_elmts,
7048                p.data_blk_min_elmts,
7049                p.sup_blk_min_data_ptrs,
7050                bits,
7051                p.max_dblk_page_nelmts_bits,
7052            )?;
7053            let dblk_size = |nelmts: u64| -> u64 {
7054                if is_filtered {
7055                    FilteredDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7056                        .encode(&self.ctx, bits, csl)
7057                        .len() as u64
7058                } else {
7059                    ExtensibleArrayDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7060                        .encoded_size(&self.ctx, bits) as u64
7061                }
7062            };
7063            let (dblk_addrs, sblk_addrs, iblk_size) = if is_filtered {
7064                let f = c.filt_iblk.as_ref().unwrap();
7065                (
7066                    f.dblk_addrs.clone(),
7067                    f.sblk_addrs.clone(),
7068                    f.encode(&self.ctx, csl).len() as u64,
7069                )
7070            } else {
7071                (
7072                    c.ea_iblk.dblk_addrs.clone(),
7073                    c.ea_iblk.sblk_addrs.clone(),
7074                    c.ea_iblk.encoded_size(&self.ctx) as u64,
7075                )
7076            };
7077            // Data blocks addressed from the index block belong to the
7078            // first `iblock_nsblks` super blocks; each of those defines the
7079            // element count (and so the disk size) of its data blocks.
7080            let mut g = 0usize;
7081            'direct: for s in geo.sblk.iter().take(geo.iblock_nsblks) {
7082                for _ in 0..s.ndblks {
7083                    let Some(&a) = dblk_addrs.get(g) else {
7084                        break 'direct;
7085                    };
7086                    g += 1;
7087                    if a == UNDEF_ADDR {
7088                        continue;
7089                    }
7090                    if s.dblk_nelmts > geo.dblk_page_nelmts {
7091                        return Err(crate::io::IoError::InvalidState(
7092                            "cannot free a paged extensible-array data block, \
7093                             which is not yet supported"
7094                                .into(),
7095                        ));
7096                    }
7097                    self.allocator
7098                        .free(a, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7099                }
7100            }
7101            for (off, &sa) in sblk_addrs.iter().enumerate() {
7102                if sa == UNDEF_ADDR {
7103                    continue;
7104                }
7105                let s = geo.sblk[geo.iblock_nsblks + off];
7106                if s.dblk_nelmts > geo.dblk_page_nelmts {
7107                    return Err(crate::io::IoError::InvalidState(
7108                        "cannot free a paged extensible-array data block, \
7109                         which is not yet supported"
7110                            .into(),
7111                    ));
7112                }
7113                let buf = self.handle.read_at_most(sa, 65536)?;
7114                let sb =
7115                    ExtensibleArraySuperBlock::decode(&buf, &self.ctx, bits, s.ndblks as usize, 0)?;
7116                for &da in &sb.dblk_addrs {
7117                    if da != UNDEF_ADDR {
7118                        self.allocator
7119                            .free(da, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7120                    }
7121                }
7122                self.allocator.free(
7123                    sa,
7124                    sb.encode(&self.ctx, bits).len() as u64,
7125                    FreeSpaceClass::Metadata,
7126                );
7127            }
7128            self.allocator
7129                .free(c.ea_iblk_addr, iblk_size, FreeSpaceClass::Metadata);
7130            self.allocator.free(
7131                c.ea_header_addr,
7132                c.ea_header.encoded_size(&self.ctx) as u64,
7133                FreeSpaceClass::Metadata,
7134            );
7135            return Ok(());
7136        }
7137        if let Some(fa) = m.fixed_array.take() {
7138            self.allocator.free(
7139                fa.fa_dblk_addr,
7140                fixed_array_dblk_disk_size(&self.ctx, &fa.fa_header),
7141                FreeSpaceClass::Metadata,
7142            );
7143            self.allocator.free(
7144                fa.fa_header_addr,
7145                fa.fa_header.encode(&self.ctx).len() as u64,
7146                FreeSpaceClass::Metadata,
7147            );
7148            return Ok(());
7149        }
7150        // The implicit index has no structure to free, only the one run of
7151        // chunk space it was given at create — which is the whole of its
7152        // storage, so nothing else can be leaked or double-freed here.
7153        if let Some(imp) = m.implicit.take() {
7154            self.allocator
7155                .free(imp.data_addr, imp.data_size, FreeSpaceClass::RawData);
7156            return Ok(());
7157        }
7158        // The single-chunk index has no structure of its own either: its one
7159        // chunk is the whole of its storage, addressed directly from the
7160        // layout message rather than any index this function's doc comment's
7161        // "chunks already freed by a zero-extent prune" applies to — so
7162        // freeing it here, if it was ever allocated, is the only place it
7163        // happens.
7164        if let Some(sc) = m.single_chunk.take() {
7165            if sc.data_addr != UNDEF_ADDR {
7166                let len = if is_filtered { sc.nbytes } else { sc.data_size };
7167                self.allocator
7168                    .free(sc.data_addr, len, FreeSpaceClass::RawData);
7169            }
7170            return Ok(());
7171        }
7172        // The version-1 B-tree owns nothing but its node blocks: the header
7173        // every other index has is, here, the root pointer inside the layout
7174        // message.
7175        if let Some(bt1) = m.btree_v1.take() {
7176            let element_size = m.datatype.element_size() as u64;
7177            let node_size = bt1
7178                .build_tree(element_size, self.ctx.sizeof_addr as usize)
7179                .node_size();
7180            for &a in &bt1.node_addrs {
7181                self.allocator
7182                    .free(a, node_size as u64, FreeSpaceClass::Metadata);
7183            }
7184            return Ok(());
7185        }
7186        if let Some(bt2) = m.btree_v2.take() {
7187            let tree = bt2.index.build_tree(&self.ctx);
7188            for &a in &bt2.node_addrs {
7189                self.allocator
7190                    .free(a, tree.node_size as u64, FreeSpaceClass::Metadata);
7191            }
7192            self.allocator.free(
7193                bt2.bt2_header_addr,
7194                tree.header(UNDEF_ADDR).encode(&self.ctx).len() as u64,
7195                FreeSpaceClass::Metadata,
7196            );
7197        }
7198        Ok(())
7199    }
7200
7201    /// Return the chunk dimensions for a dataset, if chunked.
7202    ///
7203    /// Returns an owned `Vec` because the chunk geometry now lives behind the
7204    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7205    pub fn dataset_chunk_dims(&self, index: usize) -> Option<Vec<u64>> {
7206        let ds = self.ds(index);
7207        let m = ds.lock();
7208        m.chunk_index_kind().map(|kind| match kind {
7209            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
7210            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
7211            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
7212            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
7213            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
7214            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
7215        })
7216    }
7217
7218    /// Return the current dimensions of a dataset.
7219    ///
7220    /// Returns an owned `Vec` because the dataspace now lives behind the
7221    /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7222    pub fn dataset_dims(&self, index: usize) -> Vec<u64> {
7223        self.ds(index).lock().dataspace.dims.clone()
7224    }
7225
7226    /// Return the maximum extent a dataset declares, per dimension.
7227    ///
7228    /// An absent maximum shape means the shape is fixed at its current extent
7229    /// (libhdf5 defaults maxdims to dims at creation), so the current
7230    /// dimensions are returned; `H5S_UNLIMITED` is `u64::MAX`.
7231    pub fn dataset_max_dims(&self, index: usize) -> Vec<u64> {
7232        let ds = self.ds(index);
7233        let m = ds.lock();
7234        m.dataspace
7235            .max_dims
7236            .clone()
7237            .unwrap_or_else(|| m.dataspace.dims.clone())
7238    }
7239
7240    /// Whether a dataset stores its raw data through a filter pipeline.
7241    ///
7242    /// The write paths ask before choosing how to hand a chunk over: an
7243    /// unfiltered chunk's bytes go to the file exactly as the caller holds
7244    /// them, while a filtered one has to be compressed first.
7245    pub(crate) fn dataset_is_filtered(&self, index: usize) -> bool {
7246        self.ds(index).lock().filter_pipeline.is_some()
7247    }
7248
7249    /// Return the datatype a dataset declares on disk.
7250    ///
7251    /// The typed write paths need it to store bytes in the declared byte
7252    /// order; a reopened dataset handle has no copy of its own, and a cached
7253    /// one could disagree with what the header will say.
7254    pub fn dataset_datatype(&self, index: usize) -> DatatypeMessage {
7255        self.ds(index).lock().datatype.clone()
7256    }
7257
7258    /// Create a group in the file hierarchy.
7259    ///
7260    /// `parent_path` is the full path of the parent group (e.g., "/" for root).
7261    /// `name` is the name of the new group (e.g., "detector").
7262    ///
7263    /// Returns the group index in the writer's group list.
7264    pub fn create_group(&self, parent_path: &str, name: &str) -> IoResult<usize> {
7265        // Hold the create gate across the uniqueness check and the registry
7266        // push so the two are atomic (see `create_lock`).
7267        let _create = self.create_lock.lock();
7268        // A parent path through hard links creates in the link's target,
7269        // as HDF5 traversal does.
7270        let parent_path = self.canonical_group_path(parent_path);
7271        let parent_path = parent_path.as_str();
7272        let full_name = if parent_path == "/" {
7273            format!("/{}", name)
7274        } else {
7275            format!("{}/{}", parent_path, name)
7276        };
7277        // Same rule as dataset creation: a path through a carried external
7278        // link names a group in the other file, which this writer cannot make.
7279        self.reject_external_traversal(&full_name)?;
7280        // `name` may itself carry path components; resolving the whole thing
7281        // is what keeps a '/' out of the link this group will be reached by.
7282        let (parent_idx, _leaf) = self.split_parent(full_name.trim_start_matches('/'))?;
7283
7284        self.ensure_name_free(full_name.trim_start_matches('/'))?;
7285
7286        let group_idx = self.push_group(GroupInfo {
7287            name: full_name,
7288            parent: parent_idx,
7289            creation_seq: self.take_creation_seq(),
7290            track_order: self.track_order,
7291            times: self.created_object_times(),
7292            child_datasets: Vec::new(),
7293            child_groups: Vec::new(),
7294            obj_header_addr: 0,
7295            obj_header_written_addr: None,
7296            obj_header_blocks: Vec::new(),
7297            deleted: false,
7298            attributes: Vec::new(),
7299        });
7300
7301        // Register this group as a child of its parent
7302        if let Some(pidx) = parent_idx {
7303            self.grp(pidx).lock().child_groups.push(group_idx);
7304        }
7305
7306        Ok(group_idx)
7307    }
7308
7309    /// Register a dataset as belonging to a group.
7310    ///
7311    /// `group_path` is the full path of the group (e.g., "/detector").
7312    /// `ds_index` is the dataset index returned by `create_dataset`.
7313    pub fn assign_dataset_to_group(&self, group_path: &str, ds_index: usize) -> IoResult<()> {
7314        let group_path = self.canonical_group_path(group_path);
7315        let group_path = group_path.as_str();
7316        let groups = self.group_refs();
7317        let group_idx = groups
7318            .iter()
7319            .position(|g| {
7320                let gg = g.lock();
7321                gg.name == group_path && !gg.deleted
7322            })
7323            .ok_or_else(|| {
7324                crate::io::IoError::NotFound(format!("group '{}' not found", group_path))
7325            })?;
7326        // A move, not an addition: the create gate has already placed every
7327        // dataset from the path components of its name, so appending here
7328        // would leave one dataset linked from two groups at once.
7329        for g in &groups {
7330            g.lock().child_datasets.retain(|&d| d != ds_index);
7331        }
7332        groups[group_idx].lock().child_datasets.push(ds_index);
7333        Ok(())
7334    }
7335
7336    /// Create a hard link: an additional name for an object that already
7337    /// exists in the file.
7338    ///
7339    /// No data is copied — the link and its target share one object header,
7340    /// exactly as `h5py` / libhdf5 hard links do.
7341    ///
7342    /// * `parent_group_path` — full path of the group that will hold the
7343    ///   link (`"/"` for the root group).
7344    /// * `link_name` — leaf name of the new link within that group.
7345    /// * `target_path` — full path of an existing dataset or group, with or
7346    ///   without a leading `/`.
7347    pub fn create_hard_link(
7348        &self,
7349        parent_group_path: &str,
7350        link_name: &str,
7351        target_path: &str,
7352    ) -> IoResult<()> {
7353        if link_name.is_empty() || link_name.contains('/') {
7354            return Err(crate::io::IoError::InvalidState(format!(
7355                "hard link name '{link_name}' must be a non-empty leaf name"
7356            )));
7357        }
7358
7359        // Neither end may sit across a carried external link: the target
7360        // would be an object in the other file, and the link itself would be
7361        // a name in a group this writer does not own.
7362        self.reject_external_traversal(target_path)?;
7363        self.reject_external_traversal(&format!(
7364            "{}/{link_name}",
7365            parent_group_path.trim_end_matches('/')
7366        ))?;
7367
7368        // Hold the create gate across the collision check and the hard-link
7369        // push so the two are atomic (see `create_lock`).
7370        let _create = self.create_lock.lock();
7371        // Both paths resolve through hard links, as HDF5 traversal does.
7372        let parent_group_path = self.canonical_group_path(parent_group_path);
7373        let parent_group_path = parent_group_path.as_str();
7374
7375        // Resolve the parent group (None == root).
7376        let parent = if parent_group_path == "/" {
7377            None
7378        } else {
7379            Some(
7380                self.group_refs()
7381                    .iter()
7382                    .position(|g| {
7383                        let gg = g.lock();
7384                        gg.name == parent_group_path && !gg.deleted
7385                    })
7386                    .ok_or_else(|| {
7387                        crate::io::IoError::NotFound(format!(
7388                            "parent group '{parent_group_path}' not found"
7389                        ))
7390                    })?,
7391            )
7392        };
7393
7394        // Resolve the target. Dataset names are stored without a leading
7395        // '/', group names with one — compare on the trimmed form. A
7396        // trailing '/' is tolerated too.
7397        let target_rel = self.canonical_dataset_path(target_path.trim_matches('/'));
7398        let target_rel = target_rel.as_str();
7399        if target_rel.is_empty() {
7400            return Err(crate::io::IoError::InvalidState(
7401                "cannot hard-link the root group".into(),
7402            ));
7403        }
7404        let target = self.resolve_object(target_rel).ok_or_else(|| {
7405            crate::io::IoError::NotFound(format!("hard link target '{target_path}' not found"))
7406        })?;
7407
7408        // Reject a name already taken in the parent group.
7409        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7410
7411        self.hard_links.lock().push(HardLink {
7412            parent,
7413            name: link_name.to_string(),
7414            target,
7415            creation_seq: self.take_creation_seq(),
7416        });
7417        self.register_name(&self.link_full_path(parent, link_name), NameHit::HardLink);
7418        Ok(())
7419    }
7420
7421    /// Whether a hard link will actually be emitted: both its parent group
7422    /// and its target object must still be present (not soft-deleted).
7423    fn hard_link_emitted(&self, link: &HardLink) -> bool {
7424        let parent_ok = self.parent_alive(link.parent);
7425        let target_ok = match link.target {
7426            HardLinkTarget::Dataset(i) => !self.ds(i).lock().deleted,
7427            HardLinkTarget::Group(i) => !self.grp(i).lock().deleted,
7428        };
7429        parent_ok && target_ok
7430    }
7431
7432    /// The full path a link occupies, with no leading `/` — the same form
7433    /// dataset names are stored in. The one place a parent index and a leaf
7434    /// name become a path, so every link kind answers the collision check in
7435    /// the same spelling.
7436    fn link_full_path(&self, parent: Option<usize>, name: &str) -> String {
7437        match parent {
7438            None => name.to_string(),
7439            Some(pi) => format!(
7440                "{}/{name}",
7441                self.grp(pi).lock().name.trim_start_matches('/')
7442            ),
7443        }
7444    }
7445
7446    /// The full path a hard link occupies; see [`Self::link_full_path`].
7447    fn hard_link_full_path(&self, link: &HardLink) -> String {
7448        self.link_full_path(link.parent, &link.name)
7449    }
7450
7451    /// Whether a symbolic link will actually be emitted: its parent group
7452    /// must still be present. There is no target to check — a soft or
7453    /// external link is allowed to dangle, and `H5Lcreate_soft` does not look
7454    /// at the path it stores.
7455    fn symbolic_link_emitted(&self, link: &SymbolicLink) -> bool {
7456        self.parent_alive(link.parent)
7457    }
7458
7459    /// Whether the group that would hold a link still exists; `None` is the
7460    /// root group, which cannot be deleted.
7461    ///
7462    /// A deleted group's header is never written, so nothing it would have
7463    /// held is in the file — and the name is free again. Every registry
7464    /// decides that the same way, through here.
7465    fn parent_alive(&self, parent: Option<usize>) -> bool {
7466        match parent {
7467            None => true,
7468            Some(pi) => !self.grp(pi).lock().deleted,
7469        }
7470    }
7471
7472    /// The full path a symbolic link occupies; see [`Self::link_full_path`].
7473    fn symbolic_link_full_path(&self, link: &SymbolicLink) -> String {
7474        self.link_full_path(link.parent, &link.name)
7475    }
7476
7477    /// Create a soft or external link: a name in a group whose value is a
7478    /// path rather than an object.
7479    ///
7480    /// The single owner of symbolic-link creation — `H5Lcreate_soft` and
7481    /// `H5Lcreate_external` differ only in the value they store, and the
7482    /// name, parent and collision rules they share are all here.
7483    ///
7484    /// * `parent_group_path` — full path of the group that will hold the
7485    ///   link (`"/"` for the root group).
7486    /// * `link_name` — leaf name of the new link within that group.
7487    /// * `target` — the path this link names, and for an external link the
7488    ///   file holding it. Neither is resolved or required to exist: HDF5
7489    ///   answers a symbolic link at traversal time, so a dangling one is a
7490    ///   legal file.
7491    pub fn create_symbolic_link(
7492        &self,
7493        parent_group_path: &str,
7494        link_name: &str,
7495        target: LinkTarget,
7496    ) -> IoResult<()> {
7497        if link_name.is_empty() || link_name.contains('/') {
7498            return Err(crate::io::IoError::InvalidState(format!(
7499                "link name '{link_name}' must be a non-empty leaf name"
7500            )));
7501        }
7502        // `H5Lcreate_external` refuses an empty file or object name, and
7503        // stores the object path normalized; a link written here and one
7504        // libhdf5 writes from the same arguments then hold the same bytes.
7505        let target = match target {
7506            LinkTarget::External { file, path } => {
7507                if file.is_empty() || path.is_empty() {
7508                    return Err(crate::io::IoError::InvalidState(
7509                        "an external link needs both a file name and an object path".into(),
7510                    ));
7511                }
7512                LinkTarget::External {
7513                    file,
7514                    path: crate::format::messages::link::normalize_object_path(&path),
7515                }
7516            }
7517            other => other,
7518        };
7519        // The link itself would be a name in a group that lives in another
7520        // file; its *value* may name anything, including a path this writer
7521        // cannot follow, because nothing follows it here.
7522        self.reject_external_traversal(&format!(
7523            "{}/{link_name}",
7524            parent_group_path.trim_end_matches('/')
7525        ))?;
7526
7527        let _create = self.create_lock.lock();
7528        let parent_group_path = self.canonical_group_path(parent_group_path);
7529        let parent_group_path = parent_group_path.as_str();
7530        let parent = if parent_group_path == "/" {
7531            None
7532        } else {
7533            Some(
7534                self.group_refs()
7535                    .iter()
7536                    .position(|g| {
7537                        let gg = g.lock();
7538                        gg.name == parent_group_path && !gg.deleted
7539                    })
7540                    .ok_or_else(|| {
7541                        crate::io::IoError::NotFound(format!(
7542                            "parent group '{parent_group_path}' not found"
7543                        ))
7544                    })?,
7545            )
7546        };
7547
7548        self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7549        self.symbolic_links.lock().push(SymbolicLink {
7550            parent,
7551            name: link_name.to_string(),
7552            target,
7553            creation_seq: self.take_creation_seq(),
7554        });
7555        self.register_name(
7556            &self.link_full_path(parent, link_name),
7557            NameHit::SymbolicLink,
7558        );
7559        Ok(())
7560    }
7561
7562    // ---------------------------------------------------------- committed types
7563
7564    /// Snapshot the committed-datatype list; see [`Self::hard_links_vec`].
7565    pub(crate) fn committed_datatypes_vec(&self) -> Vec<CommittedDatatype> {
7566        self.committed_datatypes.lock().clone()
7567    }
7568
7569    /// The paths of every committed datatype a name still reaches, in
7570    /// creation order. One inside a deleted group is not among them: no link
7571    /// to it is emitted, so the file will not hold that name.
7572    ///
7573    /// Both halves of the file answer. A datatype an earlier session
7574    /// committed is carried by its bytes, not re-encoded, so it lives in the
7575    /// preserved-link list rather than the registry — and listing only the
7576    /// registry is what made this answer `[]` for a file whose every named
7577    /// type was committed before it was opened, while a reader of the same
7578    /// file named them all.
7579    pub(crate) fn committed_datatype_names(&self) -> Vec<String> {
7580        let mut out: Vec<String> = self
7581            .committed_datatypes_vec()
7582            .iter()
7583            .filter(|c| self.parent_alive(c.parent))
7584            .map(|c| c.name.clone())
7585            .collect();
7586        out.extend(
7587            self.preserved_links
7588                .lock()
7589                .iter()
7590                .filter(|l| l.kind == PreservedKind::NamedDatatype)
7591                .map(|l| self.preserved_link_full_path(l)),
7592        );
7593        out
7594    }
7595
7596    /// Commit `datatype` as an object of its own under `name` —
7597    /// `H5Tcommit2`. Returns its index in the committed-datatype registry.
7598    ///
7599    /// The object holds one datatype message and nothing else. It goes
7600    /// through [`begin_create`](Self::begin_create) like a dataset, so its
7601    /// name is resolved to a real parent group, refused if taken, and refused
7602    /// if it would cross a carried external link.
7603    pub fn commit_datatype(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
7604        let create = self.begin_create(name.trim_start_matches('/'))?;
7605        let entry = CommittedDatatype {
7606            name: create.name.clone(),
7607            parent: create.parent,
7608            datatype,
7609            creation_seq: self.take_creation_seq(),
7610            times: self.created_object_times(),
7611            obj_header_addr: 0,
7612        };
7613        let name = entry.name.clone();
7614        let idx = {
7615            let mut reg = self.committed_datatypes.lock();
7616            let idx = reg.len();
7617            reg.push(entry);
7618            idx
7619        };
7620        self.register_name(&name, NameHit::Datatype(idx));
7621        Ok(idx)
7622    }
7623
7624    /// Resolve a committed datatype's path to its registry index and the type
7625    /// it holds — the pair a dataset needs to be built on it.
7626    ///
7627    /// Returned together so the caller cannot pair one committed type's index
7628    /// with another's datatype: the dataset's element width, dataspace and
7629    /// payload checks all come from the type, and its header names the index.
7630    pub(crate) fn committed_datatype_for_share(
7631        &self,
7632        name: &str,
7633    ) -> IoResult<(usize, DatatypeMessage)> {
7634        let name = self.canonical_dataset_path(name.trim_start_matches('/'));
7635        let all = self.committed_datatypes_vec();
7636        all.iter()
7637            .position(|c| self.parent_alive(c.parent) && c.name == name)
7638            .map(|i| (i, all[i].datatype.clone()))
7639            .ok_or_else(|| {
7640                crate::io::IoError::NotFound(format!("no committed datatype named '{name}'"))
7641            })
7642    }
7643
7644    /// Record that dataset `dataset` stores its datatype as a pointer to the
7645    /// committed datatype `committed`.
7646    ///
7647    /// Takes an index [`committed_datatype_for_share`](Self::committed_datatype_for_share)
7648    /// produced, alongside the datatype from the same call, so the two cannot
7649    /// disagree and there is nothing here that can fail after the dataset
7650    /// exists.
7651    pub(crate) fn share_committed_type(&self, dataset: usize, committed: usize) {
7652        debug_assert!(committed < self.committed_datatypes.lock().len());
7653        self.ds(dataset).lock().committed_type = Some(CommittedTypeRef::Session(committed));
7654    }
7655
7656    /// How many names reach the committed datatype `index`: the link that
7657    /// gave it its name, plus every live dataset that shares it.
7658    ///
7659    /// `H5O__shared_link_adj` counts a share as a link, which is why a type
7660    /// h5py commits and then builds one dataset on reports `rc == 2`. Zero
7661    /// means nothing reaches it at all — the group holding its name was
7662    /// deleted and no dataset shares it — and then it is not written.
7663    fn committed_datatype_refcount(&self, index: usize) -> u32 {
7664        let linked = {
7665            let parent = self.committed_datatypes.lock()[index].parent;
7666            u32::from(self.parent_alive(parent))
7667        };
7668        let shares = self
7669            .dataset_refs()
7670            .iter()
7671            .filter(|d| {
7672                let m = d.lock();
7673                !m.deleted && m.committed_type == Some(CommittedTypeRef::Session(index))
7674            })
7675            .count() as u32;
7676        linked + shares
7677    }
7678
7679    /// Append the link naming each committed datatype whose parent group is
7680    /// `parent`. A committed datatype is reached by an ordinary hard link —
7681    /// what makes it a datatype rather than a group or a dataset is the one
7682    /// message in the header it points at.
7683    ///
7684    /// Only a live group's links are collected, and a live parent is itself a
7685    /// reference, so every address named here belongs to a header
7686    /// `write_committed_datatype_headers` wrote.
7687    fn push_committed_datatypes(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
7688        for cd in self.committed_datatypes_vec() {
7689            if cd.parent != parent {
7690                continue;
7691            }
7692            let leaf = cd.name.rsplit('/').next().unwrap_or(&cd.name);
7693            links.push((cd.creation_seq, LinkMessage::hard(leaf, cd.obj_header_addr)));
7694        }
7695    }
7696
7697    /// Rewrite a group path that passes through hard links into the tree
7698    /// path of the group it reaches — HDF5 traversal, where any link in a
7699    /// path component resolves to its target. Group-name form (leading
7700    /// `/`). Repeats because a substituted target's subtree can hold
7701    /// further links; bounded like libhdf5's link-traversal limit, so a
7702    /// link cycle cannot loop forever. A path with no link components
7703    /// (including one naming nothing at all) comes back unchanged.
7704    pub(crate) fn canonical_group_path(&self, path: &str) -> String {
7705        let mut path = path.to_string();
7706        for _ in 0..64 {
7707            // The longest emitted group-link path that is the whole of
7708            // `path` or a '/'-boundary prefix of it.
7709            let mut best: Option<(usize, usize)> = None; // (prefix len, target)
7710            for l in self.hard_links_vec() {
7711                let HardLinkTarget::Group(gi) = l.target else {
7712                    continue;
7713                };
7714                if !self.hard_link_emitted(&l) {
7715                    continue;
7716                }
7717                let lp = format!("/{}", self.hard_link_full_path(&l));
7718                let covers = path == lp || path.starts_with(&format!("{lp}/"));
7719                if covers && best.is_none_or(|(len, _)| lp.len() > len) {
7720                    best = Some((lp.len(), gi));
7721                }
7722            }
7723            let Some((len, gi)) = best else { break };
7724            let target_name = self.grp(gi).lock().name.clone();
7725            path = format!("{}{}", target_name, &path[len..]);
7726        }
7727        path
7728    }
7729
7730    /// [`canonical_group_path`](Self::canonical_group_path) in the
7731    /// dataset-name form (no leading `/`): the leaf is a dataset, so only
7732    /// group links can appear as components and the whole path can go
7733    /// through the group rewrite unchanged.
7734    fn canonical_dataset_path(&self, name: &str) -> String {
7735        self.canonical_group_path(&format!("/{name}"))
7736            .trim_start_matches('/')
7737            .to_string()
7738    }
7739
7740    /// Total number of hard links resolving to an object: its own tree link
7741    /// plus every emitted user-created hard link pointing at it.
7742    fn object_link_count(&self, target: HardLinkTarget) -> u32 {
7743        let same = |a: HardLinkTarget, b: HardLinkTarget| -> bool {
7744            matches!(
7745                (a, b),
7746                (HardLinkTarget::Dataset(x), HardLinkTarget::Dataset(y))
7747                    | (HardLinkTarget::Group(x), HardLinkTarget::Group(y))
7748                if x == y
7749            )
7750        };
7751        1 + self
7752            .hard_links_vec()
7753            .iter()
7754            .filter(|l| self.hard_link_emitted(l) && same(l.target, target))
7755            .count() as u32
7756    }
7757
7758    /// The object a path names, or `None` when nothing in the file does.
7759    ///
7760    /// `path` is the trimmed, hard-link-canonical form (no leading or
7761    /// trailing `/`) that dataset and group names compare against. The single
7762    /// owner of path→object resolution on the write side: hard links and
7763    /// object references must agree on what a path means, including that a
7764    /// path may itself be a user hard link — links have no chain (each points
7765    /// straight at the object header, as in libhdf5), so the existing link's
7766    /// target is the answer.
7767    pub(crate) fn resolve_object(&self, path: &str) -> Option<HardLinkTarget> {
7768        if let Some(idx) = self.dataset_refs().iter().position(|d| {
7769            let g = d.lock();
7770            !g.deleted && g.name.trim_start_matches('/') == path
7771        }) {
7772            return Some(HardLinkTarget::Dataset(idx));
7773        }
7774        if let Some(idx) = self.group_refs().iter().position(|g| {
7775            let gg = g.lock();
7776            !gg.deleted && gg.name.trim_start_matches('/') == path
7777        }) {
7778            return Some(HardLinkTarget::Group(idx));
7779        }
7780        self.hard_links_vec().iter().find_map(|l| {
7781            (self.hard_link_emitted(l) && self.hard_link_full_path(l) == path).then_some(l.target)
7782        })
7783    }
7784
7785    /// The address of dataset `index`'s own contiguous block, for the two
7786    /// writers that stamp single elements into it by file offset — object and
7787    /// region references, whose values are only known once finalize has placed
7788    /// every object header.
7789    ///
7790    /// Refuses, rather than handing back an address that is not one, every
7791    /// dataset that has no such block: chunked, compact, unallocated, or with
7792    /// its raw data in files outside this one.
7793    fn local_element_block(&self, index: usize, what: &str) -> IoResult<u64> {
7794        let ds = self.ds(index);
7795        let m = ds.lock();
7796        match m.contiguous_target() {
7797            Some(ContiguousTarget::Local(addr)) => Ok(addr),
7798            Some(ContiguousTarget::External { .. }) => {
7799                Err(crate::io::IoError::InvalidState(format!(
7800                    "{what} are stamped into the dataset's own contiguous block, and \
7801                 dataset '{}' has none: its raw data lives in external files",
7802                    m.name
7803                )))
7804            }
7805            Some(ContiguousTarget::Virtual) => Err(crate::io::IoError::InvalidState(format!(
7806                "{what} are stamped into the dataset's own contiguous block, and \
7807                 dataset '{}' has none: it is virtual, and its elements come from \
7808                 the source datasets its mappings name",
7809                m.name
7810            ))),
7811            None => Err(crate::io::IoError::InvalidState(format!(
7812                "{what} are stamped into contiguous storage; create the dataset \
7813                 without chunking"
7814            ))),
7815        }
7816    }
7817
7818    /// Store object references naming `paths` into the elements of dataset
7819    /// `index` starting at `start`.
7820    ///
7821    /// The value of an `H5R_OBJECT1` element is its target's object header
7822    /// address, which finalize assigns, so what lands here is the target path;
7823    /// [`Self::write_object_reference_values`] writes the addresses. Elements
7824    /// never written keep the zero image libhdf5 reads back as a null
7825    /// reference.
7826    pub fn write_object_references(
7827        &self,
7828        index: usize,
7829        start: u64,
7830        paths: &[&str],
7831    ) -> IoResult<()> {
7832        let elements = {
7833            let ds = self.ds(index);
7834            let m = ds.lock();
7835            match &m.datatype {
7836                // Both generations of object reference: `H5T_STD_REF_OBJ` and
7837                // the 1.12 `H5T_STD_REF`. They differ only in the element
7838                // image, which `encode_reference_element` owns.
7839                DatatypeMessage::Reference {
7840                    kind: ReferenceKind::Object1 | ReferenceKind::Object2,
7841                    ..
7842                } => {}
7843                other => {
7844                    return Err(crate::io::IoError::InvalidState(format!(
7845                        "dataset '{}' has datatype {other}, not an object reference",
7846                        m.name
7847                    )))
7848                }
7849            }
7850            m.dataspace
7851                .dims
7852                .iter()
7853                .fold(1u64, |a, &d| a.saturating_mul(d))
7854        };
7855        // Refused here as well as at fixup time, so a dataset whose storage
7856        // cannot hold stamped elements is reported at the call that chose it.
7857        self.local_element_block(index, "object references")?;
7858        let end = start.saturating_add(paths.len() as u64);
7859        if end > elements {
7860            return Err(crate::io::IoError::InvalidState(format!(
7861                "elements {start}..{end} are outside the dataset's {elements}"
7862            )));
7863        }
7864        // Resolve now as well as at fixup time, so a path that names nothing
7865        // is reported at the call that got it wrong.
7866        for path in paths {
7867            self.object_reference_target(path)?;
7868        }
7869        let mut pending = self.pending_object_references.lock();
7870        for (i, path) in paths.iter().enumerate() {
7871            pending.push(PendingObjectReference {
7872                dataset: index,
7873                element: start + i as u64,
7874                target: (*path).to_string(),
7875            });
7876        }
7877        Ok(())
7878    }
7879
7880    /// Record a hard link count of `rc` in `header`, if this file's format
7881    /// needs a message to carry it.
7882    ///
7883    /// A version-2 header carries the count in an Object Reference Count
7884    /// message, and only when more than one link reaches the object. A
7885    /// version-1 header carries it in its prefix and gets no message at all —
7886    /// `H5O_link_oh` gates every refcount-message operation on
7887    /// `oh->version > H5O_VERSION_1` (H5Oint.c:851), so a version-1 header
7888    /// holding one is a shape libhdf5 never writes.
7889    ///
7890    /// The message carries `H5O_MSG_FLAG_DONTSHARE`, which both refcount
7891    /// operations pass (H5Oint.c:874 append, H5Oint.c:864 write): the count is
7892    /// a property of this one object header, so a shared-message index that
7893    /// pointed several headers at one copy would make every object with the
7894    /// same link count share a single number.
7895    fn emit_refcount(&self, header: &mut ObjectHeader, rc: u32, format: ObjectFormat) {
7896        if rc > 1 && format == ObjectFormat::Modern {
7897            header.add_message(MSG_OBJ_REF_COUNT, MSG_FLAG_DONTSHARE, encode_refcount(rc));
7898        }
7899    }
7900
7901    /// Encode an object header for the block at `addr`, at the version this
7902    /// file's format calls for and with `rc` as the object's hard link count.
7903    ///
7904    /// The count is passed rather than read off the header because the two
7905    /// versions carry it in different places — the version-1 prefix's `nlink`
7906    /// field, the version-2 Reference Count message
7907    /// [`emit_refcount`](Self::emit_refcount) already added — and only the
7908    /// caller knows it.
7909    ///
7910    /// INVARIANT: every chunk of an object header lives in the one block its
7911    /// address and encoded size describe. A header whose messages overflow
7912    /// chunk 0 gets a continuation chunk immediately behind it in that same
7913    /// block, so the address is enough to free, relocate or supersede the
7914    /// whole header — which is what every caller already assumes. libhdf5
7915    /// would have grown chunk 0 into space that free rather than chaining
7916    /// onto it, but it reads a continuation chunk by the address and length
7917    /// its message states and cares nothing for where that lands.
7918    fn encode_header_at(
7919        &self,
7920        header: &ObjectHeader,
7921        rc: u32,
7922        format: ObjectFormat,
7923        addr: u64,
7924    ) -> IoResult<Vec<u8>> {
7925        if format == ObjectFormat::Legacy {
7926            return Ok(header.encode_for(format, rc)?);
7927        }
7928        let plan = header.plan_chunks(self.chunk0_capacity(header), &self.ctx)?;
7929        let (mut image, continuation) =
7930            header.encode_chunked(&plan, &self.ctx, addr + plan.chunk0_size as u64)?;
7931        if let Some(chunk) = continuation {
7932            image.extend_from_slice(&chunk);
7933        }
7934        Ok(image)
7935    }
7936
7937    /// The bytes [`encode_header_at`](Self::encode_header_at) will produce for
7938    /// `header`, without an address and without producing them.
7939    ///
7940    /// A header's encoded size does not depend on the addresses it carries,
7941    /// which is what lets the group pass hand every group header an address
7942    /// before it writes any of their content.
7943    fn header_encoded_size(
7944        &self,
7945        header: &ObjectHeader,
7946        rc: u32,
7947        format: ObjectFormat,
7948    ) -> IoResult<usize> {
7949        if format == ObjectFormat::Legacy {
7950            return Ok(header.encode_for(format, rc)?.len());
7951        }
7952        let plan = header.plan_chunks(self.chunk0_capacity(header), &self.ctx)?;
7953        Ok(plan.chunk0_size + plan.continuation_size)
7954    }
7955
7956    /// How many bytes of messages `header`'s chunk 0 holds before the rest
7957    /// spill into a continuation chunk.
7958    ///
7959    /// libhdf5 sizes chunk 0 once, when the object header is created, and can
7960    /// only grow it while the space behind it is still free — so an object
7961    /// whose creation-time estimate covered every message it would ever hold
7962    /// keeps one chunk, and one whose estimate was a guess does not. A dataset
7963    /// or a committed datatype is created from messages already in hand
7964    /// (`H5D__update_oh_info`, `H5T__commit`), so its estimate is exact and
7965    /// this writer's exact fit is the same answer.
7966    ///
7967    /// A group is the exception: `H5G__obj_create_real` (H5Gobj.c:219) sizes
7968    /// its header for the link info and group info messages plus
7969    /// `H5G_CRT_GINFO_EST_NUM_ENTRIES` links of `H5G_CRT_GINFO_EST_NAME_LEN`
7970    /// characters, and nothing else — attributes above all — is in that
7971    /// estimate. The Link Info message is what identifies one: it is the
7972    /// message that makes an object a new-format group, and
7973    /// `H5G__obj_get_linfo` uses it for exactly this question.
7974    fn chunk0_capacity(&self, header: &ObjectHeader) -> usize {
7975        let envelope = header.message_envelope_size();
7976        let sized = |msg_type: u8| {
7977            header
7978                .messages
7979                .iter()
7980                .find(|m| m.msg_type == msg_type)
7981                .map(|m| envelope + m.data.len())
7982        };
7983        let Some(link_info) = sized(MSG_LINK_INFO) else {
7984            return usize::MAX;
7985        };
7986        // One estimated hard link: version, flags, a one-byte name length for
7987        // a name this short, the name, and the object header address.
7988        let link = envelope + 1 + 1 + 1 + EST_LINK_NAME_LEN + self.ctx.sizeof_addr as usize;
7989        link_info + sized(MSG_GROUP_INFO).unwrap_or(0) + EST_LINK_COUNT * link
7990    }
7991
7992    /// The object an object reference's path names, as a hard-link target;
7993    /// `None` for the root group, which has no registry slot.
7994    fn object_reference_target(&self, path: &str) -> IoResult<Option<HardLinkTarget>> {
7995        let rel = self.canonical_dataset_path(path.trim_matches('/'));
7996        if rel.is_empty() {
7997            return Ok(None);
7998        }
7999        self.resolve_object(&rel)
8000            .map(Some)
8001            .ok_or_else(|| crate::io::IoError::NotFound(format!("reference target '{path}'")))
8002    }
8003
8004    /// The object header address an object reference's `path` names, or zero
8005    /// when that object has not been given one yet.
8006    ///
8007    /// Zero is where the superblock sits, so it is never an object header's
8008    /// address. It is what every object reads as before
8009    /// [`allocate_object_headers`](Self::allocate_object_headers) runs, which
8010    /// is what lets the pass that measures a header stand in for the pass that
8011    /// writes it: an address is a fixed-width field, so the placeholder is the
8012    /// same size as the answer.
8013    fn object_reference_address(&self, path: &str) -> IoResult<u64> {
8014        Ok(match self.object_reference_target(path)? {
8015            Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8016            Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8017            None => self.root_group_addr.unwrap_or(0),
8018        })
8019    }
8020
8021    /// `scope`'s attributes as this finalize will write them: the stored set,
8022    /// with every object-reference attribute's value said in the object header
8023    /// addresses assigned so far.
8024    ///
8025    /// The single owner of a reference attribute's value, and the only source
8026    /// an object header build may take an attribute set from. Nothing stored
8027    /// is mutated, so the pass that measures a header and the pass that writes
8028    /// it cannot disagree about anything but the addresses — which they cannot
8029    /// disagree about in length.
8030    ///
8031    /// INVARIANT: the stored attribute list is what says which attributes
8032    /// exist; a recorded reference value can only give a value to one already
8033    /// in it. So a value left behind by an object whose list was emptied — a
8034    /// deleted group or dataset — cannot put the attribute back, and a value
8035    /// whose attribute was replaced by one of another type is dropped at the
8036    /// replacement instead of reaching it (see
8037    /// [`forget_attribute_reference`](Self::forget_attribute_reference)).
8038    fn object_attributes(&self, scope: AttrScope) -> IoResult<Vec<AttributeEntry>> {
8039        let mut attrs = match scope {
8040            AttrScope::Root => self.root_attributes.lock().clone(),
8041            AttrScope::Group(gi) => self.grp(gi).lock().attributes.clone(),
8042            AttrScope::Dataset(i) => self.ds(i).lock().attributes.clone(),
8043        };
8044        // Snapshot first: resolving a path locks group and dataset slots.
8045        let values: Vec<(String, Vec<String>)> = self
8046            .attribute_references
8047            .lock()
8048            .iter()
8049            .filter(|r| r.scope == scope)
8050            .map(|r| (r.name.clone(), r.targets.clone()))
8051            .collect();
8052        let width = self.ctx.sizeof_addr as usize;
8053        for (name, targets) in values {
8054            let Some(pos) = attrs.iter().position(|a| a.name() == name) else {
8055                continue;
8056            };
8057            let Some(msg) = attrs[pos].readable() else {
8058                continue;
8059            };
8060            let mut msg = msg.clone();
8061            let mut data = Vec::with_capacity(targets.len() * width);
8062            for target in &targets {
8063                data.extend_from_slice(
8064                    &self.object_reference_address(target)?.to_le_bytes()[..width],
8065                );
8066            }
8067            msg.data = data;
8068            attrs[pos] = AttributeEntry::from(msg).with_creation_index(attrs[pos].creation_index());
8069        }
8070        Ok(attrs)
8071    }
8072
8073    /// The registry scope `target` names — the same object
8074    /// [`with_attr_list`](Self::with_attr_list) reaches, as the key the
8075    /// reference-value registry is indexed by. Refuses what that accessor
8076    /// refuses, and for the same reasons.
8077    fn attr_scope(&self, target: AttrTarget<'_>) -> IoResult<AttrScope> {
8078        match target {
8079            AttrTarget::Root => Ok(AttrScope::Root),
8080            AttrTarget::Group(path) => {
8081                let path = self.canonical_group_path(path);
8082                self.group_refs()
8083                    .iter()
8084                    .position(|g| {
8085                        let gg = g.lock();
8086                        gg.name == path && !gg.deleted
8087                    })
8088                    .map(AttrScope::Group)
8089                    .ok_or_else(|| {
8090                        crate::io::IoError::NotFound(format!("group '{path}' not found"))
8091                    })
8092            }
8093            AttrTarget::Dataset(index) => {
8094                let count = self.dataset_count();
8095                if index >= count {
8096                    return Err(crate::io::IoError::InvalidState(format!(
8097                        "dataset index {index} out of range (have {count})"
8098                    )));
8099                }
8100                Ok(AttrScope::Dataset(index))
8101            }
8102        }
8103    }
8104
8105    /// Drop the reference value recorded for `scope`'s attribute `name`.
8106    ///
8107    /// Called by both owners of attribute-list mutation —
8108    /// [`insert_attribute`](Self::insert_attribute) and
8109    /// [`evict_attr`](Self::evict_attr) — so an attribute that is replaced or
8110    /// removed cannot leave its value behind for whatever takes its name next.
8111    /// A string attribute written over a reference attribute is the case that
8112    /// needs it: without this the string's bytes would be overwritten with
8113    /// addresses at finalize.
8114    fn forget_attribute_reference(&self, scope: AttrScope, name: &str) {
8115        self.attribute_references
8116            .lock()
8117            .retain(|r| !(r.scope == scope && r.name == name));
8118    }
8119
8120    /// Write every pending object reference element as its target's object
8121    /// header address.
8122    ///
8123    /// INVARIANT: a reference element on disk holds its target's header
8124    /// address. Reached through [`write_reference_values`](Self::write_reference_values),
8125    /// which places it after every header has an address; a target that no
8126    /// longer resolves fails the finalize rather than leaving a placeholder
8127    /// behind.
8128    fn write_object_reference_values(&mut self) -> IoResult<()> {
8129        // Snapshot rather than drain: a SWMR session finalizes twice, and the
8130        // close-time finalize rebuilds every header at a fresh address, so the
8131        // elements must be stamped again with the addresses that survive.
8132        let pending: Vec<(usize, u64, String)> = self
8133            .pending_object_references
8134            .lock()
8135            .iter()
8136            .map(|p| (p.dataset, p.element, p.target.clone()))
8137            .collect();
8138        for (dataset, element, target) in &pending {
8139            let addr = match self.object_reference_target(target)? {
8140                Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8141                Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8142                None => self.root_group_addr.ok_or_else(|| {
8143                    crate::io::IoError::InvalidState(
8144                        "root group header address is not assigned yet".into(),
8145                    )
8146                })?,
8147            };
8148            // The element image is the dataset's own datatype's business: the
8149            // pre-1.12 and 1.12 forms differ in width and in layout, and the
8150            // dataset says which it holds.
8151            let (kind, width) = {
8152                let ds = self.ds(*dataset);
8153                let m = ds.lock();
8154                let DatatypeMessage::Reference { kind, size } = &m.datatype else {
8155                    return Err(crate::io::IoError::InvalidState(format!(
8156                        "dataset '{}' is no longer a reference dataset",
8157                        m.name
8158                    )));
8159                };
8160                (*kind, *size as usize)
8161            };
8162            let image = match kind {
8163                ReferenceKind::Object1 => ReferenceElementImage::Legacy(addr),
8164                ReferenceKind::Object2 => ReferenceElementImage::Inline(addr),
8165                other => {
8166                    return Err(crate::io::IoError::InvalidState(format!(
8167                        "dataset {dataset} now holds {other:?} elements, not object references"
8168                    )))
8169                }
8170            };
8171            let image = encode_reference_element(&image, width, &self.ctx)?;
8172            let data_addr = self.local_element_block(*dataset, "object references")?;
8173            let at = data_addr + element * width as u64;
8174            self.handle.write_at(at, &image)?;
8175        }
8176        Ok(())
8177    }
8178
8179    /// Store region references over `targets` into the elements of dataset
8180    /// `index` starting at `start`.
8181    ///
8182    /// Each target is the path of a dataset and a selection over it. What the
8183    /// element holds is a global-heap id — collection address then object index
8184    /// (`H5R__encode_heap`) — and the heap object it names is the target's
8185    /// object header address followed by the serialized selection
8186    /// (`H5R__encode_token_region_compat`). Both the object and the element are
8187    /// written here; only the address inside the object waits for
8188    /// [`Self::write_heap_reference_values`]. Elements never written keep the
8189    /// zero image libhdf5 reads back as a null reference.
8190    pub fn write_region_references(
8191        &self,
8192        index: usize,
8193        start: u64,
8194        targets: &[(&str, Selection)],
8195    ) -> IoResult<()> {
8196        let elements = {
8197            let ds = self.ds(index);
8198            let m = ds.lock();
8199            match &m.datatype {
8200                DatatypeMessage::Reference {
8201                    kind: ReferenceKind::DatasetRegion1,
8202                    ..
8203                } => {}
8204                other => {
8205                    return Err(crate::io::IoError::InvalidState(format!(
8206                        "dataset '{}' has datatype {other}, not a region reference",
8207                        m.name
8208                    )))
8209                }
8210            }
8211            m.dataspace
8212                .dims
8213                .iter()
8214                .fold(1u64, |a, &d| a.saturating_mul(d))
8215        };
8216        let data_addr = self.local_element_block(index, "region references")?;
8217        let end = start.saturating_add(targets.len() as u64);
8218        if end > elements {
8219            return Err(crate::io::IoError::InvalidState(format!(
8220                "elements {start}..{end} are outside the dataset's {elements}"
8221            )));
8222        }
8223
8224        // Build every heap object before inserting any: a path that names no
8225        // dataset, or a selection its extent does not admit, is reported at the
8226        // call that got it wrong rather than after half the batch is on disk.
8227        let sa = self.ctx.sizeof_addr as usize;
8228        let mut blobs = Vec::with_capacity(targets.len());
8229        for (path, selection) in targets {
8230            let target = self.region_reference_target(path)?;
8231            let dims = self.ds(target).lock().dataspace.dims.clone();
8232            validate_region_selection(selection, &dims, path)?;
8233            let mut blob = vec![0u8; sa];
8234            blob.extend_from_slice(&selection.encode()?);
8235            blobs.push(blob);
8236        }
8237        let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
8238        let placements = self.insert_vlen_objects(&items)?;
8239
8240        let width = (sa + 4) as u64;
8241        let mut pending = self.pending_heap_references.lock();
8242        for (i, &(collection, obj_index)) in placements.iter().enumerate() {
8243            let mut elem = Vec::with_capacity(width as usize);
8244            elem.extend_from_slice(&collection.to_le_bytes()[..sa]);
8245            elem.extend_from_slice(&u32::from(obj_index).to_le_bytes());
8246            self.handle
8247                .write_at(data_addr + (start + i as u64) * width, &elem)?;
8248            pending.push(PendingHeapReference {
8249                collection,
8250                index: obj_index,
8251                token_offset: 0,
8252                target: PendingHeapTarget::Dataset(targets[i].0.to_string()),
8253            });
8254        }
8255        Ok(())
8256    }
8257
8258    /// Store 1.12 references over `targets` into the elements of dataset
8259    /// `index` starting at `start` — the `H5T_STD_REF` trio.
8260    ///
8261    /// One datatype holds all three kinds, because a 1.12 element leads with
8262    /// the kind it holds; which is why this takes a [`ReferenceTarget`] per
8263    /// element rather than a fixed kind. `H5R_OBJECT2` needs nothing but the
8264    /// target's address, so its element is written inline by the same finalize
8265    /// pass every object reference goes through. The other two encode a
8266    /// selection or an attribute name alongside the token, which does not fit
8267    /// an element, so what is stored is a global-heap blob and the element is
8268    /// its id (`H5T__ref_disk_write`). Elements never written keep the zero
8269    /// image `H5T__ref_disk_isnull` reads back as a null reference.
8270    pub fn write_revised_references(
8271        &self,
8272        index: usize,
8273        start: u64,
8274        targets: &[(&str, ReferenceTarget)],
8275    ) -> IoResult<()> {
8276        let (width, elements) = {
8277            let ds = self.ds(index);
8278            let m = ds.lock();
8279            match &m.datatype {
8280                DatatypeMessage::Reference {
8281                    kind: ReferenceKind::Object2,
8282                    size,
8283                } => (
8284                    *size as u64,
8285                    m.dataspace
8286                        .dims
8287                        .iter()
8288                        .fold(1u64, |a, &d| a.saturating_mul(d)),
8289                ),
8290                other => {
8291                    return Err(crate::io::IoError::InvalidState(format!(
8292                        "dataset '{}' has datatype {other}, not the 1.12 H5T_STD_REF",
8293                        m.name
8294                    )))
8295                }
8296            }
8297        };
8298        let data_addr = self.local_element_block(index, "references")?;
8299        let end = start.saturating_add(targets.len() as u64);
8300        if end > elements {
8301            return Err(crate::io::IoError::InvalidState(format!(
8302                "elements {start}..{end} are outside the dataset's {elements}"
8303            )));
8304        }
8305
8306        // Build every blob before inserting any, so a path that names nothing,
8307        // a selection an extent does not admit or an attribute that does not
8308        // exist is reported at the call that got it wrong rather than after
8309        // half the batch is on disk.
8310        let mut blobs: Vec<(u64, ReferenceKind, PendingHeapTarget, Vec<u8>)> = Vec::new();
8311        let mut inline: Vec<(u64, String)> = Vec::new();
8312        for (i, (path, target)) in targets.iter().enumerate() {
8313            let element = start + i as u64;
8314            // The rank of the extent the selection is over, which only a region
8315            // reference encodes and takes from the target's dataspace.
8316            let mut extent_rank = 0;
8317            let (kind, pending) = match target {
8318                ReferenceTarget::Object => {
8319                    self.object_reference_target(path)?;
8320                    inline.push((element, (*path).to_string()));
8321                    continue;
8322                }
8323                ReferenceTarget::Region(selection) => {
8324                    let ds = self.region_reference_target(path)?;
8325                    let dims = self.ds(ds).lock().dataspace.dims.clone();
8326                    validate_region_selection(selection, &dims, path)?;
8327                    extent_rank = dims.len();
8328                    (
8329                        ReferenceKind::DatasetRegion2,
8330                        PendingHeapTarget::Dataset((*path).to_string()),
8331                    )
8332                }
8333                ReferenceTarget::Attribute(name) => {
8334                    let scope = match self.object_reference_target(path)? {
8335                        Some(HardLinkTarget::Dataset(i)) => AttrScope::Dataset(i),
8336                        Some(HardLinkTarget::Group(i)) => AttrScope::Group(i),
8337                        None => AttrScope::Root,
8338                    };
8339                    if !self
8340                        .object_attributes(scope)?
8341                        .iter()
8342                        .any(|a| a.name() == name)
8343                    {
8344                        return Err(crate::io::IoError::NotFound(format!(
8345                            "attribute '{name}' of reference target '{path}'"
8346                        )));
8347                    }
8348                    (
8349                        ReferenceKind::Attr,
8350                        PendingHeapTarget::Object((*path).to_string()),
8351                    )
8352                }
8353            };
8354            blobs.push((
8355                element,
8356                kind,
8357                pending,
8358                encode_revised_blob(0, target, extent_rank, &self.ctx)?,
8359            ));
8360        }
8361
8362        let items: Vec<&[u8]> = blobs.iter().map(|(_, _, _, b)| b.as_slice()).collect();
8363        let placements = self.insert_vlen_objects(&items)?;
8364
8365        let mut pending = self.pending_heap_references.lock();
8366        for ((element, kind, target, blob), &(collection, obj_index)) in
8367            blobs.iter().zip(&placements)
8368        {
8369            // The size the element declares is the heap object's own byte
8370            // count: `H5VL__native_blob_get` refuses to read one whose size
8371            // does not match what the element says.
8372            let image = encode_reference_element(
8373                &ReferenceElementImage::Blob {
8374                    kind: *kind,
8375                    size: blob.len() as u32,
8376                    collection,
8377                    index: u32::from(obj_index),
8378                },
8379                width as usize,
8380                &self.ctx,
8381            )?;
8382            self.handle.write_at(data_addr + element * width, &image)?;
8383            pending.push(PendingHeapReference {
8384                collection,
8385                index: obj_index,
8386                token_offset: REVISED_BLOB_TOKEN_OFFSET,
8387                target: target.clone(),
8388            });
8389        }
8390        drop(pending);
8391
8392        let mut pending = self.pending_object_references.lock();
8393        for (element, path) in inline {
8394            pending.push(PendingObjectReference {
8395                dataset: index,
8396                element,
8397                target: path,
8398            });
8399        }
8400        Ok(())
8401    }
8402
8403    /// The dataset a region reference's path names.
8404    ///
8405    /// A region reference names a *dataset*: `H5Rcreate` with
8406    /// `H5R_DATASET_REGION` takes the dataspace of one, and every reader
8407    /// dereferences it as one. A path that resolves to a group — or to the root
8408    /// group, which has no registry slot — is refused here rather than stored
8409    /// as a reference nothing can dereference.
8410    fn region_reference_target(&self, path: &str) -> IoResult<usize> {
8411        match self.object_reference_target(path)? {
8412            Some(HardLinkTarget::Dataset(i)) => Ok(i),
8413            _ => Err(crate::io::IoError::InvalidState(format!(
8414                "region reference target '{path}' is not a dataset"
8415            ))),
8416        }
8417    }
8418
8419    /// Stamp every pending heap-backed reference's object with its target's
8420    /// object header address.
8421    ///
8422    /// The references that are still stamped rather than written once: the
8423    /// *element* is a global-heap id, so the heap object has to exist at the
8424    /// call that stores the reference, long before any address does. The object
8425    /// was inserted with its token zeroed, so its size does not change here:
8426    /// each collection is read once, patched, and rewritten at its own declared
8427    /// size, which leaves every element's heap id valid — and leaves the
8428    /// object's byte count equal to the size the 1.12 element declares, which
8429    /// `H5VL__native_blob_get` refuses to read past.
8430    fn write_heap_reference_values(&mut self) -> IoResult<()> {
8431        use crate::format::global_heap::GlobalHeapCollection;
8432
8433        // Snapshot rather than drain, for the same reason the object-reference
8434        // pass does: a SWMR session finalizes twice and the close-time finalize
8435        // rebuilds every header at a fresh address.
8436        let pending: Vec<(u64, u16, usize, PendingHeapTarget)> = self
8437            .pending_heap_references
8438            .lock()
8439            .iter()
8440            .map(|p| (p.collection, p.index, p.token_offset, p.target.clone()))
8441            .collect();
8442        if pending.is_empty() {
8443            return Ok(());
8444        }
8445        let sa = self.ctx.sizeof_addr as usize;
8446        // Group by collection so one holding several references is read and
8447        // rewritten once.
8448        let mut per_collection: std::collections::BTreeMap<u64, Vec<(u16, usize, u64)>> =
8449            Default::default();
8450        for (collection, index, token_offset, target) in &pending {
8451            let addr = match target {
8452                PendingHeapTarget::Dataset(path) => {
8453                    let ds = self.region_reference_target(path)?;
8454                    self.ds(ds).lock().obj_header_addr
8455                }
8456                PendingHeapTarget::Object(path) => self.object_reference_address(path)?,
8457            };
8458            per_collection
8459                .entry(*collection)
8460                .or_default()
8461                .push((*index, *token_offset, addr));
8462        }
8463        for (collection, patches) in per_collection {
8464            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
8465            // exactly that, so one read usually covers the whole image.
8466            let mut image = self.handle.read_at_most(collection, 4096)?;
8467            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
8468            if declared > image.len() {
8469                image = self.handle.read_at(collection, declared)?;
8470            }
8471            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
8472            for (index, token_offset, addr) in patches {
8473                let token = gcol
8474                    .objects
8475                    .iter_mut()
8476                    .find(|o| o.index == index)
8477                    .and_then(|o| o.data.get_mut(token_offset..token_offset + sa))
8478                    .ok_or_else(|| {
8479                        crate::io::IoError::InvalidState(format!(
8480                            "object {index} of global heap collection {collection:#x} is no \
8481                             longer the reference written into it"
8482                        ))
8483                    })?;
8484                token.copy_from_slice(&addr.to_le_bytes()[..sa]);
8485            }
8486            let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
8487            self.handle.write_at(collection, &rewritten)?;
8488        }
8489        Ok(())
8490    }
8491
8492    /// Give every reference written this session its target's object header
8493    /// address.
8494    ///
8495    /// INVARIANT: no file is closed holding a reference whose target address is
8496    /// still the placeholder its write left. Both finalize paths call this in
8497    /// the content phase — after
8498    /// [`allocate_object_headers`](Self::allocate_object_headers), so every
8499    /// address exists, and before any object header is written — and this is
8500    /// the only caller of the per-kind passes, so a reference kind added later
8501    /// is written at both finalize sites or at neither. A target that no longer
8502    /// resolves fails the finalize rather than leaving a placeholder behind.
8503    ///
8504    /// This covers the two reference kinds whose value lives outside an object
8505    /// header. An attribute's value lives *inside* one, so it has no pass here:
8506    /// [`object_attributes`](Self::object_attributes) says it in addresses as
8507    /// the header is built.
8508    fn write_reference_values(&mut self) -> IoResult<()> {
8509        self.write_object_reference_values()?;
8510        self.write_heap_reference_values()
8511    }
8512
8513    /// Append every user-created hard link whose parent group is `parent`
8514    /// (`None` == the root group). Called while collecting a group's links,
8515    /// once every object's header address has been assigned.
8516    fn push_hard_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8517        for link in self.hard_links_vec() {
8518            if link.parent != parent || !self.hard_link_emitted(&link) {
8519                continue;
8520            }
8521            let addr = match link.target {
8522                HardLinkTarget::Dataset(i) => self.ds(i).lock().obj_header_addr,
8523                HardLinkTarget::Group(i) => self.grp(i).lock().obj_header_addr,
8524            };
8525            links.push((link.creation_seq, LinkMessage::hard(&link.name, addr)));
8526        }
8527    }
8528
8529    /// Append every user-created symbolic link whose parent group is `parent`
8530    /// (`None` == the root group).
8531    ///
8532    /// Nothing here waits on the layout pass — the link's value is a path, not
8533    /// an address — but it is collected with the rest so it takes its place in
8534    /// creation order and counts toward the phase change.
8535    fn push_symbolic_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8536        for link in self.symbolic_links_vec() {
8537            if link.parent != parent || !self.symbolic_link_emitted(&link) {
8538                continue;
8539            }
8540            links.push((
8541                link.creation_seq,
8542                LinkMessage {
8543                    name: link.name.clone(),
8544                    target: link.target.clone(),
8545                    creation_order: None,
8546                    cset: CharacterSet::for_name(&link.name),
8547                },
8548            ));
8549        }
8550    }
8551
8552    /// Refuse a caller path that would have to leave this file through one of
8553    /// the external links a reopened file brought in.
8554    ///
8555    /// The reader follows such a path into the file the link names; the writer
8556    /// cannot, because it models one file and would have to write into
8557    /// another. Saying which link stops the path — rather than reporting the
8558    /// name as absent, or worse, creating a second link of that name beside
8559    /// it — is the whole of what write mode does here.
8560    pub(crate) fn reject_external_traversal(&self, path: &str) -> IoResult<()> {
8561        let path = path.trim_start_matches('/');
8562        let crossing = self.preserved_link_paths().into_iter().find(|(p, class)| {
8563            matches!(class, crate::io::reader::LinkClass::External { .. })
8564                && (path == p || path.starts_with(&format!("{p}/")))
8565        });
8566        match crossing {
8567            None => Ok(()),
8568            Some((link, crate::io::reader::LinkClass::External { file, path: target })) => {
8569                Err(crate::io::IoError::Unsupported(format!(
8570                    "'{path}' resolves through the external link '{link}' to '{target}' in \
8571                     '{file}'; this writer carries external links through a rewrite but does \
8572                     not open the file they name"
8573                )))
8574            }
8575            // `find` matched on the External arm, so no other class reaches here.
8576            Some(_) => Ok(()),
8577        }
8578    }
8579
8580    /// Resolve `name` to a live dataset index, reporting *why* it does not
8581    /// resolve rather than collapsing every cause into absence.
8582    ///
8583    /// The write-mode counterpart of [`Hdf5Reader::open_dataset`]: the single
8584    /// gate every by-name dataset lookup in write mode goes through.
8585    ///
8586    /// [`Hdf5Reader::open_dataset`]: crate::io::reader::Hdf5Reader::open_dataset
8587    pub(crate) fn open_dataset_index(&self, name: &str) -> IoResult<usize> {
8588        self.reject_external_traversal(name)?;
8589        self.reject_preserved_object(name)?;
8590        self.dataset_index(name)
8591            .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
8592    }
8593
8594    /// Refuse a caller path that names an object the reopen kept by its bytes
8595    /// rather than modelling.
8596    ///
8597    /// Such an object is in the file and stays in it, but this writer holds
8598    /// none of what it would need to read or rewrite it. Saying so — with the
8599    /// reason the classification recorded — is the difference between an
8600    /// object the writer will not touch and a name the file does not have.
8601    pub(crate) fn reject_preserved_object(&self, path: &str) -> IoResult<()> {
8602        let path = path.trim_start_matches('/');
8603        let objects: Vec<(String, String)> = {
8604            let preserved = self.preserved_links.lock();
8605            preserved
8606                .iter()
8607                .filter_map(|l| {
8608                    l.reason
8609                        .as_ref()
8610                        .map(|why| (self.preserved_link_full_path(l), why.clone()))
8611                })
8612                .collect()
8613        };
8614        match objects
8615            .into_iter()
8616            .find(|(full, _)| path == full || path.starts_with(&format!("{full}/")))
8617        {
8618            None => Ok(()),
8619            Some((link, why)) => Err(crate::io::IoError::Unsupported(format!(
8620                "'{path}' is, or is inside, the object '{link}', which this file's reopen \
8621                 kept exactly as it found it because {why}"
8622            ))),
8623        }
8624    }
8625
8626    /// Every link this writer will emit that names a *path* rather than an
8627    /// object, with the class a listing reports for it: the soft and external
8628    /// links created this session, and the ones a reopen is carrying through.
8629    ///
8630    /// The object listings answer for hard links, so a write-mode link
8631    /// listing is this plus those; keeping both sources in one place is what
8632    /// stops a listing from seeing a kind the class lookup does not, or the
8633    /// reverse.
8634    pub(crate) fn path_link_classes(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8635        let mut out: Vec<(String, crate::io::reader::LinkClass)> = self
8636            .symbolic_links_vec()
8637            .iter()
8638            .filter(|l| self.symbolic_link_emitted(l))
8639            .map(|l| {
8640                (
8641                    self.symbolic_link_full_path(l),
8642                    crate::io::reader::LinkClass::from_target(&l.target),
8643                )
8644            })
8645            .collect();
8646        out.extend(self.preserved_link_paths());
8647        out
8648    }
8649
8650    /// Every link this writer is carrying but cannot express, by full path.
8651    pub(crate) fn preserved_link_paths(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8652        self.preserved_links
8653            .lock()
8654            .iter()
8655            .map(|l| (self.preserved_link_full_path(l), l.class.clone()))
8656            .collect()
8657    }
8658
8659    /// The full path of a preserved link: its parent group's path plus its
8660    /// leaf name, in the no-leading-`/` form the registry uses.
8661    fn preserved_link_full_path(&self, link: &PreservedLink) -> String {
8662        match link.parent {
8663            None => link.name.clone(),
8664            Some(gi) => {
8665                let group = self.grp(gi).lock().name.clone();
8666                format!("{}/{}", group.trim_start_matches('/'), link.name)
8667            }
8668        }
8669    }
8670
8671    /// The single owner of "which links does this group hold", in the order
8672    /// they were created and, when the file tracks creation order, stamped
8673    /// with it.
8674    ///
8675    /// Both the compact form (one `MSG_LINK` per link) and the dense form (the
8676    /// same messages inside a fractal heap) are built from this one list, so
8677    /// the phase-change decision, the storage it selects and the creation
8678    /// order recorded in either can never disagree about what the group
8679    /// contains.
8680    fn group_links(&self, scope: LinkScope, order: CreationOrder) -> Vec<LinkMessage> {
8681        let mut links: Vec<(u64, LinkMessage)> = Vec::new();
8682        match scope {
8683            LinkScope::Root => {
8684                // Datasets that belong to a subgroup are that group's links,
8685                // not the root's. Each group slot is locked one at a time.
8686                let mut datasets_in_subgroups: std::collections::HashSet<usize> =
8687                    std::collections::HashSet::new();
8688                for grp in self.group_refs() {
8689                    let g = grp.lock();
8690                    if g.deleted {
8691                        continue;
8692                    }
8693                    datasets_in_subgroups.extend(g.child_datasets.iter().copied());
8694                }
8695                // `dataset_refs` preserves registry order, so `enumerate`
8696                // yields each dataset's true index.
8697                for (i, ds) in self.dataset_refs().into_iter().enumerate() {
8698                    let m = ds.lock();
8699                    if m.deleted || datasets_in_subgroups.contains(&i) {
8700                        continue;
8701                    }
8702                    // The leaf, never the registry path: a link name is one
8703                    // path component, and `H5G_traverse` would split a '/'
8704                    // in it before `H5L_link` ever saw the name.
8705                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8706                    links.push((
8707                        m.creation_seq,
8708                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8709                    ));
8710                }
8711                for grp in self.group_refs() {
8712                    let g = grp.lock();
8713                    if g.deleted || g.parent.is_some() {
8714                        continue;
8715                    }
8716                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8717                    links.push((
8718                        g.creation_seq,
8719                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8720                    ));
8721                }
8722                self.push_hard_links(&mut links, None);
8723                self.push_symbolic_links(&mut links, None);
8724                self.push_committed_datatypes(&mut links, None);
8725            }
8726            LinkScope::Group(group_idx) => {
8727                // Snapshot the child lists, then drop the slot guard: the
8728                // per-child reads below re-lock dataset and group slots
8729                // (including this one).
8730                let (child_datasets, child_groups) = {
8731                    let grp = self.grp(group_idx);
8732                    let g = grp.lock();
8733                    (g.child_datasets.clone(), g.child_groups.clone())
8734                };
8735                for ds_idx in child_datasets {
8736                    let ds = self.ds(ds_idx);
8737                    let m = ds.lock();
8738                    if m.deleted {
8739                        continue;
8740                    }
8741                    let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8742                    links.push((
8743                        m.creation_seq,
8744                        LinkMessage::hard(leaf_name, m.obj_header_addr),
8745                    ));
8746                }
8747                for child_idx in child_groups {
8748                    let child_grp = self.grp(child_idx);
8749                    let g = child_grp.lock();
8750                    if g.deleted {
8751                        continue;
8752                    }
8753                    let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8754                    links.push((
8755                        g.creation_seq,
8756                        LinkMessage::hard(leaf_name, g.obj_header_addr),
8757                    ));
8758                }
8759                self.push_hard_links(&mut links, Some(group_idx));
8760                self.push_symbolic_links(&mut links, Some(group_idx));
8761                self.push_committed_datatypes(&mut links, Some(group_idx));
8762            }
8763        }
8764        // Creation order, not order by kind: a run of create_group and
8765        // create_dataset draws from one counter, so this is the order the
8766        // caller made them in. `H5G_obj_insert` numbers from zero within the
8767        // group, so the rank here is the link's creation order.
8768        links.sort_by_key(|(seq, _)| *seq);
8769        links
8770            .into_iter()
8771            .enumerate()
8772            .map(|(rank, (_, link))| {
8773                if order.is_tracked() {
8774                    link.with_creation_order(rank as i64)
8775                } else {
8776                    link
8777                }
8778            })
8779            .collect()
8780    }
8781
8782    /// Whether `links` must live in dense storage rather than in the group's
8783    /// object header — the `H5G_obj_insert` phase-change rule, applied to the
8784    /// whole set at once because this writer builds each header from scratch
8785    /// rather than inserting one link at a time.
8786    ///
8787    /// libhdf5 converts when the count *reaches* `max_compact` and another
8788    /// link arrives, so a set of exactly `max_compact` is still compact; and
8789    /// separately when one message would not fit the 16-bit size field an
8790    /// object header message has.
8791    ///
8792    /// The answer depends only on the link names and kinds, never on the
8793    /// addresses they point at, which is what lets a group header be sized
8794    /// before [`prepare_dense_links`](Self::prepare_dense_links) has run.
8795    fn links_need_dense(&self, links: &[LinkMessage]) -> bool {
8796        links.len() > MAX_COMPACT_LINKS
8797            || links
8798                .iter()
8799                .any(|l| l.encode(&self.ctx).len() > MAX_MESSAGE_SIZE)
8800    }
8801
8802    /// The single owner of link emission into a group object header: the Link
8803    /// Info and Group Info messages, and then either one `MSG_LINK` per link
8804    /// or nothing at all when the set has spilled to dense storage.
8805    ///
8806    /// The two storage forms are exclusive (`H5G_obj_insert` moves the whole
8807    /// set at once), and a header carrying both would report every link twice.
8808    ///
8809    /// A group whose links are dense but not yet laid out gets a compact Link
8810    /// Info message here. That is deliberate: the message encodes to the same
8811    /// length either way — two addresses, defined or not — so the sizing pass
8812    /// that runs before `prepare_dense_links` still reserves the right number
8813    /// of bytes, and the write pass that runs after it emits the real heap and
8814    /// index addresses. It is the same two-pass rule the child link addresses
8815    /// already follow.
8816    fn emit_links(
8817        &self,
8818        header: &mut ObjectHeader,
8819        scope: LinkScope,
8820        links: &[LinkMessage],
8821        order: CreationOrder,
8822    ) {
8823        // A symbol-table group holds no link messages at all: its links are the
8824        // entries of the symbol table `prepare_symbol_tables` laid out, and
8825        // the header carries only the two addresses naming it. Link Info and
8826        // Group Info are version-1.8 messages and have no business in a
8827        // version-1 header — `H5G__stab_valid` reads the Symbol Table message
8828        // and nothing else.
8829        if self.uses_symbol_table(scope, order) {
8830            // Sizing runs before the tables are laid out; the message is the
8831            // same two addresses wide either way, so the placeholder reserves
8832            // exactly what the real one needs. Same two-pass rule the child
8833            // link addresses already follow.
8834            let stab = self
8835                .symbol_tables
8836                .written
8837                .lock()
8838                .get(&scope)
8839                .copied()
8840                .unwrap_or(Stab {
8841                    btree_addr: UNDEF_ADDR,
8842                    heap_addr: UNDEF_ADDR,
8843                });
8844            header.add_message(MSG_SYMBOL_TABLE, 0x00, stab.encode(&self.ctx));
8845            return;
8846        }
8847        // Links a reopen carried through verbatim because this writer cannot
8848        // express them. They are emitted here rather than by a second caller
8849        // so that no header-rewrite path can drop them, and their presence
8850        // pins the group to compact storage: dense storage would have to
8851        // re-encode each link into the heap, which is exactly the byte
8852        // fidelity preserving them is for.
8853        let preserved = self.preserved_links_for(scope);
8854        let dense = preserved.is_empty() && self.links_need_dense(links);
8855        let link_info = self.dense_links.lock().get(&scope).cloned();
8856        let link_info = link_info.unwrap_or_else(|| {
8857            let mut info = LinkInfoMessage::compact();
8858            if order.is_tracked() {
8859                // `H5G__obj_insert` post-increments `max_corder`, so a group
8860                // holding n links reports n.
8861                info.max_creation_order = Some(links.len() as u64);
8862            }
8863            if order.is_indexed() {
8864                // The index address stays undefined while the links live in
8865                // the header, but the message must still carry the field:
8866                // `H5Pget_link_creation_order` reads INDEXED off this flag,
8867                // not off the address.
8868                info.creation_order_btree_address = Some(UNDEF_ADDR);
8869            }
8870            info
8871        });
8872        header.add_message(MSG_LINK_INFO, 0x00, link_info.encode(&self.ctx));
8873        // The link info message takes no flags and the group info message
8874        // takes `H5O_MSG_FLAG_CONSTANT`, exactly as `H5G__obj_create_real`
8875        // creates the pair (H5Gobj.c:255, :259) and as
8876        // `H5G__obj_insert`'s phase change re-creates it (H5Gobj.c:526). The
8877        // asymmetry is real: the link info message records the group's
8878        // storage and its creation-order counter, both of which change as
8879        // links come and go, while the group info message holds the phase
8880        // change and estimated-name-length constants of the creation property
8881        // list, which nothing after creation rewrites.
8882        header.add_message(
8883            MSG_GROUP_INFO,
8884            MSG_FLAG_CONSTANT,
8885            GroupInfoMessage::default().encode(),
8886        );
8887        if dense {
8888            return;
8889        }
8890        for link in links {
8891            header.add_message(MSG_LINK, 0x00, link.encode(&self.ctx));
8892        }
8893        for encoded in preserved {
8894            header.add_message(MSG_LINK, 0x00, encoded);
8895        }
8896    }
8897
8898    /// The verbatim link bodies a reopen carried into `scope`.
8899    fn preserved_links_for(&self, scope: LinkScope) -> Vec<Vec<u8>> {
8900        let parent = match scope {
8901            LinkScope::Root => None,
8902            LinkScope::Group(i) => Some(i),
8903        };
8904        self.preserved_links
8905            .lock()
8906            .iter()
8907            .filter(|l| l.parent == parent)
8908            .map(|l| l.encoded.clone())
8909            .collect()
8910    }
8911
8912    /// Lay out and write dense link storage for every group that needs it,
8913    /// recording the resulting `Link Info` message per group.
8914    ///
8915    /// The sole owner of that transition. It must run after every object
8916    /// header address is assigned — the heap holds encoded link messages, and
8917    /// those name their targets — and before any group header is written.
8918    ///
8919    /// Every group whose header this finalize rewrites passes through here,
8920    /// dense or not: the storage a reopened header named is superseded by the
8921    /// rewrite whichever form the new link set takes, and freeing it first is
8922    /// what lets the replacement reuse those blocks.
8923    fn prepare_dense_links(&self) -> IoResult<()> {
8924        let mut scopes: Vec<(LinkScope, Vec<LinkMessage>, CreationOrder)> = Vec::new();
8925        for gi in 0..self.group_count() {
8926            let (deleted, order) = {
8927                let grp = self.grp(gi);
8928                let g = grp.lock();
8929                (g.deleted, g.track_order.links)
8930            };
8931            // A symbol-table group is `prepare_symbol_tables`' business; it
8932            // has no Link Info message to hold a fractal heap address, and it
8933            // never had dense storage to release.
8934            if deleted || self.uses_symbol_table(LinkScope::Group(gi), order) {
8935                continue;
8936            }
8937            self.release_superseded_dense_links(LinkScope::Group(gi))?;
8938            let links = self.group_links(LinkScope::Group(gi), order);
8939            if self.links_need_dense(&links) {
8940                scopes.push((LinkScope::Group(gi), links, order));
8941            }
8942        }
8943        let root_order = self.root_track_order.links;
8944        if !self.uses_symbol_table(LinkScope::Root, root_order) {
8945            self.release_superseded_dense_links(LinkScope::Root)?;
8946            let root_links = self.group_links(LinkScope::Root, root_order);
8947            if self.links_need_dense(&root_links) {
8948                scopes.push((LinkScope::Root, root_links, root_order));
8949            }
8950        }
8951
8952        for (scope, links, order) in scopes {
8953            // `close` after `start_swmr` finalizes a second time over the same
8954            // groups, so rebuilding here would allocate a whole second heap
8955            // and strand the one the published headers already name.
8956            if self.dense_links.lock().contains_key(&scope) {
8957                continue;
8958            }
8959            let dense = build_dense_links(&links, &self.ctx, order, &mut |len| {
8960                self.allocator.allocate(len, FreeSpaceClass::Metadata)
8961            })?;
8962            for block in &dense.blocks {
8963                self.handle.write_at(block.addr, &block.image)?;
8964            }
8965            self.dense_links.lock().insert(scope, dense.linfo);
8966        }
8967        Ok(())
8968    }
8969
8970    /// Lay out whichever of the two forms of link storage this file uses,
8971    /// before any group header is written.
8972    ///
8973    /// The two are exclusive because the formats are: a classic group has no
8974    /// Link Info message to put a fractal heap address in, and a link-message
8975    /// group has no symbol table.
8976    fn prepare_link_storage(&self) -> IoResult<()> {
8977        self.prepare_dense_links()?;
8978        self.prepare_symbol_tables()
8979    }
8980
8981    /// Lay out and write the symbol table of every classic group, and free the
8982    /// storage each rewrite supersedes. A no-op on a link-message file.
8983    ///
8984    /// The classic counterpart of [`prepare_dense_links`](Self::prepare_dense_links),
8985    /// and the sole owner of that transition. The same two placement rules
8986    /// apply for the same two reasons: it runs after every object header has
8987    /// an address, because a symbol table entry names its target's header, and
8988    /// before any group header is written, because the header carries the
8989    /// Symbol Table message naming what this laid out.
8990    ///
8991    /// Deepest group first, root last. A hard link to a group caches that
8992    /// group's own B-tree and heap in the entry's scratch pad
8993    /// (`H5G__link_to_ent`), so the child's table must exist before the
8994    /// parent's is built; `H5G__stab_valid` checks the root entry's cache
8995    /// against the root header's Symbol Table message, so a stale pair there
8996    /// is not a slow lookup but a file `H5Fopen` rejects.
8997    ///
8998    /// Every classic group is rebuilt on every pass — there is no "already
8999    /// done" short-circuit like the dense one, because the only way this runs
9000    /// twice is a `Drop` retry after a failed `close`, and the entries of the
9001    /// first pass name header addresses the second pass has moved. (A SWMR
9002    /// session, the other double-finalize, cannot reach here: SWMR needs a
9003    /// version-3 superblock, so `start_swmr` refuses a classic file.)
9004    fn prepare_symbol_tables(&self) -> IoResult<()> {
9005        // Depth by parent chain, not by counting separators in the registry
9006        // path: the chain is what actually says which table has to exist first.
9007        let mut scopes: Vec<(usize, LinkScope, CreationOrder)> = Vec::new();
9008        for gi in 0..self.group_count() {
9009            let (deleted, order, mut parent) = {
9010                let grp = self.grp(gi);
9011                let g = grp.lock();
9012                (g.deleted, g.track_order.links, g.parent)
9013            };
9014            if deleted || !self.uses_symbol_table(LinkScope::Group(gi), order) {
9015                continue;
9016            }
9017            let mut depth = 1usize;
9018            while let Some(p) = parent {
9019                depth += 1;
9020                parent = self.grp(p).lock().parent;
9021            }
9022            scopes.push((depth, LinkScope::Group(gi), order));
9023        }
9024        scopes.sort_by_key(|&(depth, ..)| std::cmp::Reverse(depth));
9025        let root_order = self.root_track_order.links;
9026        if self.uses_symbol_table(LinkScope::Root, root_order) {
9027            scopes.push((0, LinkScope::Root, root_order));
9028        }
9029
9030        let meta = self.stab_meta();
9031        for (_, scope, order) in scopes {
9032            // Freed before the replacement is laid out, so a rewrite reuses
9033            // the same blocks instead of growing the file on every open/close
9034            // cycle — the rule `prepare_dense_links` and the header rewrite
9035            // already follow. Removed as it is freed, so no second pass can
9036            // free it twice.
9037            let superseded = self.symbol_tables.superseded.lock().remove(&scope);
9038            if let Some(extents) = superseded {
9039                free_stab(&self.allocator, &extents);
9040            }
9041            let links = self.stab_links_for(scope, order)?;
9042            let stab = write_stab(&self.handle, &self.allocator, &meta, &links)?;
9043            self.symbol_tables.written.lock().insert(scope, stab);
9044        }
9045        Ok(())
9046    }
9047
9048    /// The file-level parameters every symbol-table node width is derived from
9049    /// — the address/length widths and the B-tree "K" ranks. Only a version-0/1
9050    /// superblock records ranks of its own; [`btree_v1_config`] is the one
9051    /// place that decides whether this file has any.
9052    ///
9053    /// [`btree_v1_config`]: Self::btree_v1_config
9054    fn stab_meta(&self) -> FileMeta {
9055        FileMeta {
9056            ctx: self.ctx,
9057            btree: self.btree_v1_config(),
9058            sohm: None,
9059        }
9060    }
9061
9062    /// `scope`'s links as symbol table entries.
9063    ///
9064    /// A link a reopen carried through verbatim is decoded back out of its
9065    /// encoded Link message here, because a classic group has no link message
9066    /// to preserve it into. Nothing is lost in the round trip: the walk built
9067    /// that message from a symbol table entry in the first place, and the two
9068    /// forms carry the same three facts.
9069    fn stab_links_for(&self, scope: LinkScope, order: CreationOrder) -> IoResult<Vec<StabLink>> {
9070        let groups = self.group_header_scopes();
9071        let mut out = Vec::new();
9072        for link in self.group_links(scope, order) {
9073            out.push(self.stab_link(&link, &groups)?);
9074        }
9075        for encoded in self.preserved_links_for(scope) {
9076            let (link, _) = LinkMessage::decode(&encoded, &self.ctx)?;
9077            out.push(self.stab_link(&link, &groups)?);
9078        }
9079        Ok(out)
9080    }
9081
9082    /// Where each group's object header now sits, so a hard link that lands on
9083    /// one can cache that group's symbol table in its scratch pad.
9084    fn group_header_scopes(&self) -> HashMap<u64, LinkScope> {
9085        let mut map = HashMap::new();
9086        for gi in 0..self.group_count() {
9087            let grp = self.grp(gi);
9088            let g = grp.lock();
9089            if !g.deleted {
9090                map.insert(g.obj_header_addr, LinkScope::Group(gi));
9091            }
9092        }
9093        map
9094    }
9095
9096    /// One link as a symbol table entry.
9097    ///
9098    /// The scratch pad caches the target group's B-tree and heap when the
9099    /// target is a group this pass has already laid out — what
9100    /// `H5G__link_to_ent` does, and what lets `H5G__stab_lookup` walk a path
9101    /// without opening each header on the way. For anything else the pad stays
9102    /// `H5G_NOTHING_CACHED`, the value libhdf5 itself writes whenever the
9103    /// target has no Symbol Table message to read.
9104    fn stab_link(
9105        &self,
9106        link: &LinkMessage,
9107        groups: &HashMap<u64, LinkScope>,
9108    ) -> IoResult<StabLink> {
9109        let target = match &link.target {
9110            LinkTarget::Hard { address } => {
9111                let cached = groups
9112                    .get(address)
9113                    .and_then(|scope| self.symbol_tables.written.lock().get(scope).copied());
9114                StabTarget::Hard {
9115                    addr: *address,
9116                    cached,
9117                }
9118            }
9119            LinkTarget::Soft { target } => StabTarget::Soft {
9120                value: target.clone(),
9121            },
9122            // Unreachable by construction: a group holding one of these is
9123            // not a symbol-table group at all
9124            // ([`LinkMessage::fits_symbol_table`] is what
9125            // [`Hdf5Writer::uses_symbol_table`] asks), so this pass never
9126            // visits it. Reported rather than panicked so a future caller
9127            // that skips that gate learns which link it lost.
9128            LinkTarget::External { .. } | LinkTarget::UserDefined { .. } => {
9129                return Err(crate::io::IoError::InvalidState(format!(
9130                    "cannot store the link {:?} in a symbol table: it holds only \
9131                     hard and soft links, and this group was not converted to link \
9132                     messages the way `H5G_obj_insert` converts it",
9133                    link.name
9134                )))
9135            }
9136        };
9137        Ok(StabLink {
9138            name: link.name.clone(),
9139            target,
9140        })
9141    }
9142
9143    /// The single owner of attribute emission into an object header: appends
9144    /// the Attribute Info message and then one `MSG_ATTRIBUTE` per attribute.
9145    ///
9146    /// On a version-2 object header the two are inseparable.
9147    /// `H5O__attr_count_real` derives `H5Oget_info().num_attrs` from the
9148    /// Attribute Info message alone — with no such message the count reads as
9149    /// zero however many attribute messages follow, which is what made every
9150    /// rust-written file report `num_attrs == 0` to libhdf5 while
9151    /// `H5Aiterate2` still yielded the attributes. The message carries no
9152    /// count of its own: `H5A__get_ainfo` fills `nattrs` from the attribute
9153    /// messages the header loader actually saw, so compact storage needs
9154    /// nothing but the message's presence.
9155    ///
9156    /// When [`prepare_dense_attributes`](Self::prepare_dense_attributes) has
9157    /// spilled `scope`'s attributes to a fractal heap, the same message names
9158    /// that heap instead and *no* attribute message follows: the two storage
9159    /// forms are exclusive (`H5O__attr_create` moves the whole set at once),
9160    /// and a header carrying both would report every attribute twice.
9161    fn emit_attributes(
9162        &self,
9163        header: &mut ObjectHeader,
9164        scope: AttrScope,
9165        attributes: &[AttributeEntry],
9166        order: CreationOrder,
9167        format: ObjectFormat,
9168        owner: ShareOwner,
9169    ) {
9170        // `H5Pget_attr_creation_order` reads the object header's own flags,
9171        // not the Attribute Info message, so this is what makes the object
9172        // report creation-ordered attributes — and tracking widens every
9173        // message envelope by the creation index below.
9174        let order = self.header_attr_order(order);
9175        header.set_attribute_creation_order(order);
9176        if attributes.is_empty() {
9177            return;
9178        }
9179        // A version-1 object header gets the attribute messages alone.
9180        // `H5O__attr_create` gates every mention of the Attribute Info message
9181        // on `oh->version > H5O_VERSION_1` (H5Oattribute.c:218), and so does
9182        // `H5O__attr_count_real`, which is why the count still reads correctly
9183        // without it: on a version-1 header libhdf5 counts the messages.
9184        if format == ObjectFormat::Legacy {
9185            for attr in attributes {
9186                header.add_message(MSG_ATTRIBUTE, 0x00, self.encode_attribute(attr));
9187            }
9188            return;
9189        }
9190        // Whether the set spills is a property of the set alone, so it is the
9191        // same answer in the pass that measures this header and in the pass
9192        // that writes it — even though the storage itself is laid out between
9193        // the two, because it can only be laid out once every object header
9194        // has an address. Sizing therefore falls back to a placeholder message
9195        // of the same width: only the creation-order flags change the
9196        // Attribute Info message's length, so the header measured here holds
9197        // the header written against the storage that replaces it. Same
9198        // two-pass rule `emit_links` follows for dense links and symbol
9199        // tables.
9200        let dense = self.attributes_need_dense(attributes, format);
9201        let stored = self.dense_attributes.lock().get(&scope).cloned();
9202        let ainfo = stored.unwrap_or_else(|| {
9203            let mut ainfo = AttributeInfoMessage::compact();
9204            if order.is_tracked() {
9205                ainfo.max_creation_index = Some(next_creation_index(attributes));
9206            }
9207            if order.is_indexed() {
9208                // Compact storage has no index B-tree, but the message still
9209                // announces one so that its flags match the header's
9210                // (`H5O__attr_create` asserts they agree).
9211                ainfo.creation_order_btree_address = Some(UNDEF_ADDR);
9212            }
9213            ainfo
9214        });
9215        header.add_message(MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, ainfo.encode(&self.ctx));
9216        if dense {
9217            return;
9218        }
9219        // Each attribute states its own creation index — the one it was
9220        // created with here, or the one the file it was read from records. An
9221        // attribute with none belongs to an object that tracks no order, where
9222        // the field is not encoded at all.
9223        for attr in attributes {
9224            let (flags, body) = self.share_attribute(attr, format, owner);
9225            header.add_message_indexed(
9226                MSG_ATTRIBUTE,
9227                flags,
9228                body,
9229                attr.creation_index().unwrap_or(0),
9230            );
9231        }
9232    }
9233
9234    /// One attribute message body, at the version this file's low library
9235    /// bound calls for (`H5A__set_version`, which reads the bound and nothing
9236    /// about the object the attribute hangs on).
9237    fn encode_attribute(&self, attr: &AttributeEntry) -> Vec<u8> {
9238        attr.encode_for(&self.ctx, self.encoding_libver(), self.message_format())
9239    }
9240
9241    /// What a header stores for one attribute: the message flags and the body,
9242    /// with the attribute's own datatype and dataspace shared wherever an
9243    /// index covers them.
9244    ///
9245    /// `H5A__create` offers both to `H5SM_try_share` (H5Aint.c:375-377) before
9246    /// `H5O__attr_create` offers the attribute itself (H5Oattribute.c:726), so
9247    /// the attribute body that reaches the heap already holds their pointers
9248    /// and says which fields they are in its own flags byte
9249    /// (`H5O_ATTR_FLAG_TYPE_SHARED` / `H5O_ATTR_FLAG_SPACE_SHARED`,
9250    /// H5Oattr.c:358-359). Both offers go through
9251    /// [`share_message`](Self::share_message) like any other, so the pass that
9252    /// counts references and the pass that substitutes see the same three
9253    /// messages.
9254    fn share_attribute(
9255        &self,
9256        attr: &AttributeEntry,
9257        format: ObjectFormat,
9258        owner: ShareOwner,
9259    ) -> (u8, Vec<u8>) {
9260        let libver = self.encoding_libver();
9261        // Only a readable attribute has pieces to offer: an unreadable one is
9262        // the bytes it was read from, put back as they were. Version 1 has no
9263        // flags byte to record a shared field in — `H5O__attr_encode` writes a
9264        // reserved zero there — so a classic file shares the attribute whole
9265        // or not at all.
9266        let Some(message) = attr.readable().filter(|_| format.attribute_version() >= 2) else {
9267            return self.share_message(
9268                owner,
9269                MSG_ATTRIBUTE,
9270                0x00,
9271                attr.encode_for(&self.ctx, libver, format),
9272            );
9273        };
9274
9275        let datatype = message.datatype.encode_at(&self.ctx, libver);
9276        let dataspace = message.dataspace.encode_for(&self.ctx, format);
9277        // `H5A__create` passes no open header for either (H5Aint.c:375-377):
9278        // both live inside the attribute's body, so neither has a header
9279        // message a `H5SM_IN_OH` record could name and both reach the heap on
9280        // first use.
9281        let (dt_flags, dt_field) =
9282            self.share_message(ShareOwner::Detached, MSG_DATATYPE, 0x00, datatype.clone());
9283        let (ds_flags, ds_field) =
9284            self.share_message(ShareOwner::Detached, MSG_DATASPACE, 0x00, dataspace.clone());
9285
9286        let mut attr_flags = 0u8;
9287        if dt_flags & MSG_FLAG_SHARED != 0 {
9288            attr_flags |= ATTR_FLAG_TYPE_SHARED;
9289        }
9290        if ds_flags & MSG_FLAG_SHARED != 0 {
9291            attr_flags |= ATTR_FLAG_SPACE_SHARED;
9292        }
9293        let encoded = message.encode_with_fields(attr_flags, &dt_field, &ds_field);
9294
9295        // Each shared field's heap ID sits two bytes into the pointer that
9296        // replaced it; the body offered below carries whatever
9297        // `share_message` just produced, which is a zeroed ID in the pass that
9298        // counts and the real one in the pass that substitutes.
9299        let mut nested = Vec::new();
9300        if attr_flags & ATTR_FLAG_TYPE_SHARED != 0 {
9301            nested.push(NestedShare {
9302                heap_id_at: encoded.datatype_at + SOHM_POINTER_HEAP_ID_AT,
9303                target: (MSG_DATATYPE, datatype),
9304            });
9305        }
9306        if attr_flags & ATTR_FLAG_SPACE_SHARED != 0 {
9307            nested.push(NestedShare {
9308                heap_id_at: encoded.dataspace_at + SOHM_POINTER_HEAP_ID_AT,
9309                target: (MSG_DATASPACE, dataspace),
9310            });
9311        }
9312        self.share_nesting_message(owner, MSG_ATTRIBUTE, 0x00, encoded.body, nested)
9313    }
9314
9315    /// Whether `attributes` must live in dense storage rather than in the
9316    /// object header — the `H5O__attr_create` phase-change rule, applied to
9317    /// the whole set at once because this writer builds each header from
9318    /// scratch rather than inserting one attribute at a time.
9319    ///
9320    /// libhdf5 converts when the count *reaches* `max_compact` and another
9321    /// attribute arrives, so a set of exactly `max_compact` is still compact;
9322    /// and separately when one message would not fit the 16-bit size field an
9323    /// object header message has.
9324    ///
9325    /// Never in a classic file. Dense attribute storage is a fractal heap
9326    /// reached through an Attribute Info message, both introduced in the 1.8
9327    /// format; at `H5F_LIBVER_EARLIEST` libhdf5 keeps every attribute in the
9328    /// header however many there are (`H5O__attr_create` reaches the phase
9329    /// change only when the object header version allows it). An attribute
9330    /// too large for the 16-bit size field is then an error, which
9331    /// `ObjectHeader::encode_v1` raises, rather than a reason to spill.
9332    fn attributes_need_dense(&self, attributes: &[AttributeEntry], format: ObjectFormat) -> bool {
9333        if format == ObjectFormat::Legacy {
9334            return false;
9335        }
9336        attributes.len() > MAX_COMPACT_ATTRS
9337            || attributes
9338                .iter()
9339                .any(|a| self.encode_attribute(a).len() > MAX_MESSAGE_SIZE)
9340    }
9341
9342    /// Every object whose attributes this finalize re-lays-out, with the
9343    /// creation-order policy each one's storage must follow.
9344    ///
9345    /// `datasets` lists the datasets whose headers this finalize will
9346    /// actually write. A reopened dataset that took no writes keeps its
9347    /// original header — and with it whatever storage that header already
9348    /// names — so touching its attribute storage would strand every block of
9349    /// it.
9350    ///
9351    /// The policy is the one the *header* records, not the one the object's
9352    /// creation property list asked for: those differ on a file whose
9353    /// shared-message configuration covers attributes, where
9354    /// [`header_attr_order`](Self::header_attr_order) raises every object to
9355    /// tracked. Storage laid out against the property list would then omit the
9356    /// creation indices the header says are there — and, since the Attribute
9357    /// Info message carries a maximum creation index only when tracked, would
9358    /// be two bytes shorter than the message the sizing pass measured.
9359    fn attribute_scopes(&self, datasets: &[usize]) -> Vec<(AttrScope, CreationOrder)> {
9360        let order_of = |requested| self.header_attr_order(requested);
9361        let mut scopes = vec![(AttrScope::Root, order_of(self.root_track_order.attrs))];
9362        for gi in 0..self.group_count() {
9363            if self.grp(gi).lock().deleted {
9364                continue;
9365            }
9366            let order = self.grp(gi).lock().track_order.attrs;
9367            scopes.push((AttrScope::Group(gi), order_of(order)));
9368        }
9369        for &i in datasets {
9370            let order = self.ds(i).lock().track_attr_order;
9371            scopes.push((AttrScope::Dataset(i), order_of(order)));
9372        }
9373        scopes
9374    }
9375
9376    /// Lay out and write dense attribute storage for every object that needs
9377    /// it, recording the resulting `Attribute Info` message per object.
9378    ///
9379    /// The sole owner of that transition. It runs after every object header
9380    /// has an address — an attribute may hold an object reference, and the
9381    /// heap holds the encoded attribute messages — and before any object
9382    /// header is written, because the header carries the Attribute Info
9383    /// message naming what this laid out. Every block is on disk before the
9384    /// map naming it is populated, so a header written from that map can only
9385    /// point at bytes that exist. The same placement rule, for the same two
9386    /// reasons, as [`prepare_dense_links`](Self::prepare_dense_links).
9387    ///
9388    /// Which objects spill is not decided here: `emit_attributes` asks
9389    /// [`attributes_need_dense`](Self::attributes_need_dense) itself, so the
9390    /// header measured before this ran and the header written after it agree
9391    /// without either consulting the other.
9392    fn prepare_dense_attributes(&self, datasets: &[usize]) -> IoResult<()> {
9393        for (scope, order) in self.attribute_scopes(datasets) {
9394            // Every scope here has its header rewritten, so the storage a
9395            // reopen found on it is superseded whether or not the new set is
9396            // dense again — a free driven by "the new set needs a heap" would
9397            // never reach an object that dropped back to compact. Freed
9398            // immediately before its replacement is laid out, so the rewrite
9399            // lands in the blocks it just gave back instead of growing the
9400            // file on every open/close cycle.
9401            self.release_superseded_dense_attrs(scope)?;
9402            // `close` after `start_swmr` finalizes a second time over the same
9403            // attribute sets — SWMR refuses every attribute mutation — so
9404            // rebuilding here would allocate a whole second heap and strand
9405            // the one the published headers already name.
9406            if self.dense_attributes.lock().contains_key(&scope) {
9407                continue;
9408            }
9409            let attributes = self.object_attributes(scope)?;
9410            if !self.attributes_need_dense(&attributes, self.attr_scope_format(scope)) {
9411                continue;
9412            }
9413            let dense = build_dense_attributes(&attributes, &self.ctx, order, &mut |len| {
9414                self.allocator.allocate(len, FreeSpaceClass::Metadata)
9415            })?;
9416            for block in &dense.blocks {
9417                self.handle.write_at(block.addr, &block.image)?;
9418            }
9419            self.dense_attributes.lock().insert(scope, dense.ainfo);
9420        }
9421        Ok(())
9422    }
9423
9424    /// The object header format `scope`'s owner is written at, which is what
9425    /// decides whether its attributes may spill at all.
9426    fn attr_scope_format(&self, scope: AttrScope) -> ObjectFormat {
9427        match scope {
9428            AttrScope::Root => self.header_format(self.root_track_order),
9429            AttrScope::Group(gi) => self.group_header_format(gi),
9430            AttrScope::Dataset(i) => self.dataset_header_format(i),
9431        }
9432    }
9433
9434    /// Whether a dataset's datatype message may be offered to a
9435    /// shared-message index at all.
9436    ///
9437    /// The datatype is the one message class carrying a `can_share` callback
9438    /// (`H5O__dtype_can_share`, H5Odtype.c:99), and `H5SM__can_share_common`
9439    /// asks it before any index is consulted (H5SM.c:895-899). It refuses an
9440    /// immutable type and a committed one (H5Odtype.c:1893-1901); the
9441    /// committed half is already answered by address at the call site.
9442    ///
9443    /// A dataset's type reaches that predicate still immutable only when
9444    /// `H5D__init_type` kept the caller's own `H5T_t` rather than copying it,
9445    /// which it does exactly when the type is immutable, is not relocatable,
9446    /// and the low bound this dataset's messages are written at is below
9447    /// `H5F_LIBVER_V18` (H5Dint.c:569-572) — the bound the dataset was
9448    /// *created* under, which for a dataset a reopen found is not this
9449    /// session's.
9450    /// Any of the three failing produces an `H5T_COPY_ALL` copy, which is
9451    /// `H5T_STATE_RDONLY` rather than immutable (H5T.c:4461-4462) and so is
9452    /// shareable — which is why `H5Tcopy(H5T_STD_I32LE)` shares where
9453    /// `H5T_STD_I32LE` itself does not (tests/fixtures/gen_sohm.c).
9454    ///
9455    /// An attribute has no such branch: `H5A__create` copies unconditionally
9456    /// (H5Aint.c:341), so its datatype is always eligible and
9457    /// [`share_attribute`](Self::share_attribute) offers it without asking.
9458    fn dataset_datatype_shareable(&self, datatype: &DatatypeMessage, libver: LibverBound) -> bool {
9459        !datatype.is_predefined() || datatype.is_relocatable() || libver >= LibverBound::V18
9460    }
9461
9462    /// Whether the first copy of a `msg_type` message may stay literal in the
9463    /// object header that writes it.
9464    ///
9465    /// `H5O_msg_can_share_in_ohdr` reads the class's `H5O_SHARE_IN_OHDR` flag
9466    /// (H5Omessage.c:1426); the five classes that carry it are datatype
9467    /// (H5Odtype.c:89), dataspace (H5Osdspace.c:61), both fill value messages
9468    /// (H5Ofill.c:106 and :130) and the filter pipeline (H5Opline.c:65). The
9469    /// attribute class does not, which is why an attribute reaches the heap on
9470    /// its first use.
9471    const fn shares_in_ohdr(msg_type: u8) -> bool {
9472        matches!(
9473            msg_type,
9474            MSG_DATASPACE
9475                | MSG_DATATYPE
9476                | MSG_FILL_VALUE
9477                | MSG_FILL_VALUE_OLD
9478                | MSG_FILTER_PIPELINE
9479        )
9480    }
9481
9482    /// What a header stores for a message a shared-message index may cover:
9483    /// the body itself, or a pointer into the shared-message heap.
9484    ///
9485    /// The single point at which a message is offered to an index. Every
9486    /// header builder routes its shareable messages through here, so the pass
9487    /// that counts references and the pass that substitutes pointers walk
9488    /// exactly the same set — the counting and the substituting cannot drift
9489    /// apart, because they are one call site in two phases.
9490    ///
9491    /// `owner` is `H5SM_try_share`'s `open_oh`: the header this message
9492    /// belongs to, or [`ShareOwner::Detached`] for a body that is part of
9493    /// another message rather than a message of a header.
9494    ///
9495    /// Outside a finalize, and in any file created without indexes, this is
9496    /// the identity.
9497    fn share_message(
9498        &self,
9499        owner: ShareOwner,
9500        msg_type: u8,
9501        flags: u8,
9502        body: Vec<u8>,
9503    ) -> (u8, Vec<u8>) {
9504        self.share_nesting_message(owner, msg_type, flags, body, Vec::new())
9505    }
9506
9507    /// [`share_message`](Self::share_message) for a body that itself holds
9508    /// shared-message pointers.
9509    ///
9510    /// `nested` names each heap ID inside `body`, which is zero until the
9511    /// table is laid out. Two bodies that differ only in what they point at
9512    /// are the same bytes here and different bytes on disk, so the count and
9513    /// the substitute are keyed on the pair.
9514    fn share_nesting_message(
9515        &self,
9516        owner: ShareOwner,
9517        msg_type: u8,
9518        flags: u8,
9519        body: Vec<u8>,
9520        nested: Vec<NestedShare>,
9521    ) -> (u8, Vec<u8>) {
9522        let Some(sohm) = self.sohm.as_deref() else {
9523            return (flags, body);
9524        };
9525        // A message already carrying a pointer — a committed datatype — is
9526        // shared by address and must not be shared again, and the message
9527        // classes libhdf5 marks `H5O_MSG_FLAG_DONTSHARE` never reach an index.
9528        if flags & (MSG_FLAG_SHARED | MSG_FLAG_DONTSHARE) != 0 {
9529            return (flags, body);
9530        }
9531        let Some(index) = sohm.index_for(msg_type, body.len()) else {
9532            return (flags, body);
9533        };
9534        // `share_in_ohdr && open_oh` (H5SM.c:1400): the first copy of one of
9535        // these classes stays where it was written, marked shareable, and only
9536        // a second use moves the body to the heap.
9537        let ohdr = match owner {
9538            ShareOwner::Header(addr) if Self::shares_in_ohdr(msg_type) => Some(addr),
9539            _ => None,
9540        };
9541        // What a pointer to this body looks like: a zeroed heap ID until the
9542        // table exists, which is the width the real one has.
9543        let pointer = |id| {
9544            (
9545                flags | MSG_FLAG_SHARED,
9546                SharedMessagePointer::encode_sohm(id),
9547            )
9548        };
9549        match &mut *sohm.phase.lock() {
9550            SohmPhase::Idle => (flags, body),
9551            SohmPhase::Predict(first) => {
9552                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9553                    return (flags | MSG_FLAG_SHAREABLE, body);
9554                }
9555                pointer([0u8; SOHM_HEAP_ID_LEN])
9556            }
9557            // The same substitution `Predict` makes, so that what the collect
9558            // pass builds around a shared message is the width the resolve
9559            // pass will build — which is what lets an attribute body assembled
9560            // in this pass be the body assembled in that one, bar the heap IDs
9561            // it is here recording a need for.
9562            SohmPhase::Collect(collector) => {
9563                let first = collector.record(index, msg_type, &body, &nested, ohdr);
9564                if !first && !nested.is_empty() {
9565                    // This body is already here, so the pointers it holds
9566                    // already exist in the heap and the offers that built
9567                    // this copy of it must not count a second time.
9568                    for share in &nested {
9569                        collector.release(share.target.0, &share.target.1);
9570                    }
9571                }
9572                if ohdr.is_some() && first {
9573                    return (flags | MSG_FLAG_SHAREABLE, body);
9574                }
9575                pointer([0u8; SOHM_HEAP_ID_LEN])
9576            }
9577            SohmPhase::Resolve { ids, first } => {
9578                if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9579                    return (flags | MSG_FLAG_SHAREABLE, body);
9580                }
9581                let key = (msg_type, body);
9582                match ids.get(&key) {
9583                    Some(&id) => pointer(id),
9584                    // The collect pass never saw this body — a dataspace a
9585                    // SWMR extend changed after the table was laid out, say.
9586                    // Left literal, which leaves a heap object counted for one
9587                    // reference more than reaches it and nothing else.
9588                    None => (flags, key.1),
9589                }
9590            }
9591        }
9592    }
9593
9594    /// Answer every shareable message at a heap pointer's width for the rest
9595    /// of this finalize's allocation phase.
9596    ///
9597    /// Half of the bracket [`prepare_shared_messages`](Self::prepare_shared_messages)
9598    /// closes, and the reason the two can sit on opposite sides of the
9599    /// allocation: a header cannot be measured until it is known which of its
9600    /// messages are pointers, and a body cannot be counted until every address
9601    /// it names exists. Only the width is knowable in the first phase, and the
9602    /// width is all the measurement needs.
9603    ///
9604    /// A finalize that will not lay a table out — a `finalize_for_swmr`, a
9605    /// second finalize over a table already published — leaves the phase where
9606    /// it found it, so what that pass measures is what it writes.
9607    fn begin_shared_message_layout(&self) {
9608        let Some(sohm) = self.sohm.as_deref() else {
9609            return;
9610        };
9611        let mut phase = sohm.phase.lock();
9612        if matches!(*phase, SohmPhase::Idle) && sohm.table_addr.lock().is_none() {
9613            *phase = SohmPhase::Predict(FirstCopies::default());
9614        }
9615    }
9616
9617    /// Lay out the file's shared-message table: count the bodies every header
9618    /// this finalize writes would share, put them in their index's heap, and
9619    /// arm the substitution the header builders then apply.
9620    ///
9621    /// The sole owner of the transition to `Resolve`. It runs last in the
9622    /// content phase, after
9623    /// [`prepare_dense_attributes`](Self::prepare_dense_attributes),
9624    /// [`prepare_link_storage`](Self::prepare_link_storage) and
9625    /// [`write_reference_values`](Self::write_reference_values), because a
9626    /// body is only counted once it is the body the file will hold: an
9627    /// attribute that spilled into dense storage is not in a header to be
9628    /// shared at all, and one holding an object reference says an object
9629    /// header address that exists only after the allocation phase. Counting
9630    /// either of them earlier would count a body no header ends up carrying,
9631    /// and leave the header that carries the real one literal — which
9632    /// [`check_header_size`] would then refuse, the block having been
9633    /// reserved at a pointer's width.
9634    ///
9635    /// Once per file: a second finalize (a SWMR session's close) keeps the
9636    /// table the first one published rather than allocating a second one and
9637    /// stranding the first.
9638    fn prepare_shared_messages(&self, datasets: &[usize]) -> IoResult<()> {
9639        let Some(sohm) = self.sohm.as_deref() else {
9640            return Ok(());
9641        };
9642        if sohm.table_addr.lock().is_some() {
9643            return Ok(());
9644        }
9645
9646        // Collect: build every header this finalize will write and throw it
9647        // away, keeping only what its shareable messages were.
9648        *sohm.phase.lock() = SohmPhase::Collect(SohmCollector::new(sohm.indexes.len()));
9649        for &i in datasets {
9650            self.build_dataset_header(i)?;
9651        }
9652        for gi in 0..self.group_count() {
9653            if self.grp(gi).lock().deleted {
9654                continue;
9655            }
9656            self.build_group_header(gi)?;
9657        }
9658        self.build_root_group_header()?;
9659        let SohmPhase::Collect(collector) =
9660            std::mem::replace(&mut *sohm.phase.lock(), SohmPhase::Idle)
9661        else {
9662            return Err(crate::io::IoError::InvalidState(
9663                "the shared-message collect pass did not finish in the collect phase".into(),
9664            ));
9665        };
9666
9667        let indexes: Vec<SohmIndexContent> = sohm
9668            .indexes
9669            .iter()
9670            .zip(collector.messages)
9671            .map(|(&spec, messages)| SohmIndexContent { spec, messages })
9672            .collect();
9673        // The table a reopen found is superseded whole by the one below, and
9674        // every header that pointed into it is in this finalize's rewrite set
9675        // — so its blocks go back immediately before the replacement is laid
9676        // out, and the new table lands in them instead of growing the file on
9677        // every open/close cycle. Taken, not read: a second finalize must not
9678        // free the same blocks twice.
9679        for (addr, len) in std::mem::take(&mut *sohm.superseded.lock()) {
9680            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9681        }
9682        let built = build_shared_messages(&indexes, &self.ctx, &mut |len| {
9683            self.allocator.allocate(len, FreeSpaceClass::Metadata)
9684        })?;
9685        for block in &built.blocks {
9686            self.handle.write_at(block.addr, &block.image)?;
9687        }
9688
9689        // Only now, with every block on disk: from here the header builders
9690        // substitute pointers, and `write_superblock_extension` names the
9691        // table this laid out.
9692        *sohm.phase.lock() = SohmPhase::Resolve {
9693            ids: built.heap_ids,
9694            first: FirstCopies::default(),
9695        };
9696        *sohm.table_addr.lock() = Some(built.table_addr);
9697        Ok(())
9698    }
9699
9700    /// Write the file's free-space managers over the space this close leaves
9701    /// free, and return the file-space info message body naming them.
9702    ///
9703    /// Called from [`write_superblock_extension`](Self::write_superblock_extension)
9704    /// once every other block of the file has an address, which is what makes
9705    /// the allocator's free list the file's *final* free space: a block
9706    /// allocated after this point would land in space a manager still claims.
9707    ///
9708    /// INVARIANT: from the moment this returns, every byte the allocator holds
9709    /// free is a byte some sections block records, and the two blocks each
9710    /// manager itself occupies are held by neither. Nothing may allocate
9711    /// between here and the superblock write; `write_object_headers` writes
9712    /// over blocks reserved in an earlier phase and is the only thing that
9713    /// runs in between.
9714    ///
9715    /// Returns `None` for a file with no message of its own to write — a
9716    /// reopen whose carried message this session must not touch, and a file
9717    /// created at the library defaults — which leaves both byte-identical to
9718    /// what the same close wrote before free space was recorded at all. A file
9719    /// that carries the message but keeps no managers (either non-manager
9720    /// strategy, or `persist: false`) gets the message back with every address
9721    /// undefined, which is what `H5F__super_init` writes for it.
9722    fn write_free_space_managers(&self) -> IoResult<Option<Vec<u8>>> {
9723        let Some(fs) = self.free_space.as_deref() else {
9724            return Ok(None);
9725        };
9726        if !fs.records_free_space() {
9727            return Ok(Some(fs.info.encode(&self.ctx)?));
9728        }
9729        // The managers a reopen found are superseded whole by the ones below,
9730        // so their blocks go back before anything is laid out: the space the
9731        // old manager occupied is free space the new one records, and the new
9732        // one may be laid out in it.
9733        for &(addr, len) in &fs.superseded {
9734            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9735        }
9736
9737        let hdr_size = FreeSpaceHeader::encoded_size(&self.ctx) as u64;
9738        let settled = self.settle_free_space_managers(hdr_size, fs.info.threshold)?;
9739
9740        let mut info = fs.info.clone();
9741        info.fs_addr = vec![UNDEF_ADDR; info.fs_addr.len()];
9742        for placed in &settled {
9743            let mut header = manager_header(&placed.sections);
9744            // The settle loop sized the block; that the encode agrees is the
9745            // invariant that makes `sect_size` a length a reader can trust.
9746            let needed = free_space::sinfo_encoded_size(&header, &placed.sections, &self.ctx);
9747            if needed > placed.sect_size {
9748                return Err(crate::io::IoError::InvalidState(format!(
9749                    "the free-space sections need {needed} bytes, not the {} laid out",
9750                    placed.sect_size
9751                )));
9752            }
9753            header.sect_addr = placed.sect_addr;
9754            header.sect_size = placed.sect_size;
9755            header.alloc_sect_size = placed.sect_size;
9756            self.handle.write_at(
9757                placed.sect_addr,
9758                &free_space::encode_sections(
9759                    &header,
9760                    placed.hdr_addr,
9761                    &placed.sections,
9762                    placed.sect_size as usize,
9763                    &self.ctx,
9764                ),
9765            )?;
9766            self.handle
9767                .write_at(placed.hdr_addr, &header.encode(&self.ctx))?;
9768            // `H5MF__close_delete_fstype` leaves a manager with no sections
9769            // without an address, so only the ones written name themselves.
9770            info.fs_addr[placed.manager.message_slot()] = placed.hdr_addr;
9771        }
9772        // The end of the file *after* the settle above, not before it, which
9773        // the field's name denies: it is 1.10 vintage, where two EOAs were
9774        // kept — one taken before the self-referential managers were placed
9775        // and one after (H5MF.c:3305 and 3382 in 1.10.11) — and the message
9776        // carried the first (1.10.11 H5MF.c:1833, 1999). 1.14 keeps one,
9777        // `f->shared->eoa_fsm_fsalloc`, read once the allocation loop has run
9778        // (H5MF.c:3234-3240) and encoded into this field by both close paths
9779        // (H5MF.c:1759, 1923); H5Fsuper.c:826 names it "the final eoa". A
9780        // 1.10 reader wants that value and not the older one: equal EOAs are
9781        // the case `H5MF_tidy_self_referential_fsm_hack` returns on
9782        // (1.10.11 H5MF.c:3620-3622), which is what leaves the managers this
9783        // close wrote in place.
9784        info.eoa_pre_fsm_fsalloc = self.allocator.eof();
9785        Ok(Some(info.encode(&self.ctx)?))
9786    }
9787
9788    /// The file's free space as each manager will record it: address-ordered
9789    /// per manager, tagged with the section class that manager writes, and
9790    /// with everything below `threshold` left out.
9791    ///
9792    /// The allocator is the single owner of merging — `H5FS__sect_merge`'s
9793    /// rules, per manager and, on a paged file, per page — so nothing merges
9794    /// here; overlap is checked because two overlapping sections would be a
9795    /// manager claiming space another structure holds.
9796    fn free_sections(&self, threshold: u64) -> IoResult<Vec<(FreeSpaceManager, Vec<FreeSection>)>> {
9797        let policy = self.allocator.policy();
9798        let extents = self.allocator.free_extents();
9799        let mut sets = Vec::new();
9800        for manager in FreeSpaceManager::ALL {
9801            let mut sections: Vec<FreeSection> = extents
9802                .iter()
9803                .filter(|b| b.manager == manager)
9804                // `H5FS_sect_add` refuses a section below the file's
9805                // threshold, so a block smaller than it is space the file
9806                // leaks rather than records — the same trade the threshold is
9807                // there to make.
9808                .filter(|b| b.len >= threshold)
9809                .map(|b| FreeSection {
9810                    addr: b.addr,
9811                    len: b.len,
9812                    class: policy.section_class(manager),
9813                })
9814                .collect();
9815            sections.sort_unstable_by_key(|s| s.addr);
9816            if let Some(bad) = sections
9817                .windows(2)
9818                .find(|w| w[0].addr + w[0].len > w[1].addr)
9819            {
9820                return Err(crate::io::IoError::InvalidState(format!(
9821                    "this session freed overlapping blocks: {:#x}+{} overlaps {:#x}",
9822                    bad[0].addr, bad[0].len, bad[1].addr
9823                )));
9824            }
9825            sets.push((manager, sections));
9826        }
9827        Ok(sets)
9828    }
9829
9830    /// Give every manager that records anything its own header and sections
9831    /// blocks, and return what each will write.
9832    ///
9833    /// Self-referential, which is the whole difficulty: a manager's two blocks
9834    /// come out of the free space the managers record, and taking them changes
9835    /// that space, which changes how many bytes the sections block needs.
9836    /// Upstream reruns the allocation pass until no manager allocates anything
9837    /// further — the `do { ... } while (continue_alloc_fsm)` loop in
9838    /// `H5MF_settle_meta_data_fsm` (H5MF.c:3213-3247) around
9839    /// `H5FS_vfd_alloc_hdr_and_section_info_if_needed`, which allocates
9840    /// through `H5MF_alloc` like everything else. So does this: the blocks
9841    /// come out of the same [`FileAllocator`], under the same strategy, so a
9842    /// paged file's manager blocks land in pages and their page remainders are
9843    /// recorded like any others.
9844    ///
9845    /// Two rules make it terminate. A manager, once placed, stays placed: were
9846    /// its blocks released because its sections had been consumed, freeing
9847    /// them would put those sections back and the next round would place it
9848    /// again. And a sections block only ever grows: upstream frees a block
9849    /// that turned out too small and reallocates it next round
9850    /// (H5FSsection.c:2418-2423), and a size that only rises reaches its
9851    /// bound.
9852    fn settle_free_space_managers(
9853        &self,
9854        hdr_size: u64,
9855        threshold: u64,
9856    ) -> IoResult<Vec<PlacedManager>> {
9857        /// Rounds before the layout is called divergent. A round either places
9858        /// a manager or grows one sections block, and there are three
9859        /// managers, so a file that needs more than this is not converging.
9860        const ROUNDS: usize = 16;
9861
9862        // Raw data first and metadata last, in `H5MF_settle_raw_data_fsm`'s
9863        // order (H5C.c:689-696): every manager's own blocks are metadata
9864        // allocations, so the metadata manager funds all of them and is the
9865        // one whose section set the others change.
9866        const ORDER: [FreeSpaceManager; 3] = [
9867            FreeSpaceManager::RawData,
9868            FreeSpaceManager::Large,
9869            FreeSpaceManager::Metadata,
9870        ];
9871
9872        let size_of = |sections: &[FreeSection]| {
9873            let ordered = free_space::serialization_order(sections);
9874            free_space::sinfo_encoded_size(&manager_header(&ordered), &ordered, &self.ctx)
9875        };
9876        let mut placed: Vec<PlacedManager> = Vec::new();
9877        for _ in 0..ROUNDS {
9878            let sets = self.free_sections(threshold)?;
9879            let sections_of = |manager: FreeSpaceManager| {
9880                sets.iter()
9881                    .find(|(m, _)| *m == manager)
9882                    .map(|(_, s)| s.as_slice())
9883                    .unwrap_or_default()
9884            };
9885
9886            let mut changed = false;
9887            for manager in ORDER {
9888                let sections = sections_of(manager);
9889                if sections.is_empty() || placed.iter().any(|p| p.manager == manager) {
9890                    continue;
9891                }
9892                let sect_size = size_of(sections);
9893                let hdr_addr = self.allocator.allocate(hdr_size, FreeSpaceClass::Metadata);
9894                let sect_addr = self.allocator.allocate(sect_size, FreeSpaceClass::Metadata);
9895                placed.push(PlacedManager {
9896                    manager,
9897                    hdr_addr,
9898                    sect_addr,
9899                    sect_size,
9900                    sections: Vec::new(),
9901                });
9902                changed = true;
9903            }
9904            if !changed {
9905                for p in &mut placed {
9906                    let needed = size_of(sections_of(p.manager));
9907                    if needed > p.sect_size {
9908                        self.allocator
9909                            .free(p.sect_addr, p.sect_size, FreeSpaceClass::Metadata);
9910                        p.sect_size = needed;
9911                        p.sect_addr = self.allocator.allocate(needed, FreeSpaceClass::Metadata);
9912                        changed = true;
9913                    }
9914                }
9915            }
9916            if !changed {
9917                for p in &mut placed {
9918                    p.sections = free_space::serialization_order(sections_of(p.manager));
9919                }
9920                return Ok(placed);
9921            }
9922        }
9923        Err(crate::io::IoError::InvalidState(format!(
9924            "the free-space managers did not settle in {ROUNDS} rounds"
9925        )))
9926    }
9927
9928    /// Write the file's superblock extension, and the sole owner of that
9929    /// object header.
9930    ///
9931    /// Runs after [`prepare_shared_messages`](Self::prepare_shared_messages),
9932    /// whose table it names, and before the superblock that names it. What it
9933    /// writes is [`CarriedExtension`] — every message the reopened file's
9934    /// extension held — plus the shared-message table message, which is the
9935    /// one message whose content this session owns: the table moved, so the
9936    /// message read is stale and the message written names the new address.
9937    ///
9938    /// A file with neither carried messages nor shared messages gets no
9939    /// extension, which is what libhdf5 writes for it: `H5F__super_ext_create`
9940    /// is called only when there is a message to put in one.
9941    ///
9942    /// Version 1, holding its messages in one chunk: the extension is created
9943    /// before anything raises the file's object header version
9944    /// (`H5F__super_ext_create` passes `H5O_HDR_STORE_TIMES` off and takes the
9945    /// version-1 path), so an extension of any generation of file looks the
9946    /// same.
9947    fn write_superblock_extension(&self) -> IoResult<()> {
9948        if self.extension.addr.lock().is_some() {
9949            return Ok(());
9950        }
9951        let table = self.sohm.as_deref().and_then(|sohm| {
9952            sohm.table_addr
9953                .lock()
9954                .map(|addr| (sohm.indexes.len(), addr))
9955        });
9956        // A file with file-space properties of its own needs an extension
9957        // too: the message that declares them is the only place they are
9958        // recorded, and a file created with them carries nothing else.
9959        if self.extension.carried.is_empty() && table.is_none() && self.free_space.is_none() {
9960            return Ok(());
9961        }
9962
9963        let mut messages: Vec<crate::io::object_header_io::ExtensionMessage> =
9964            self.extension.carried.clone();
9965        if let Some(fs) = self.free_space.as_deref() {
9966            // The declared message, at exactly the length the one written
9967            // below will have — every field of it is fixed-width, and only
9968            // `persist` and the message version change the count of
9969            // addresses, neither of which the close alters. The image is sized
9970            // and its block allocated before the managers can be laid out, so
9971            // the message has to reach its final *length* here even though its
9972            // content is settled later.
9973            let declared = fs.info.encode(&self.ctx)?;
9974            match messages
9975                .iter_mut()
9976                .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
9977            {
9978                Some(msg) => msg.body = declared,
9979                None => messages.push(crate::io::object_header_io::ExtensionMessage {
9980                    msg_type: MSG_FILE_SPACE_INFO,
9981                    flags: MSG_FLAG_DONTSHARE | MSG_FLAG_MARK_IF_UNKNOWN,
9982                    body: declared,
9983                }),
9984            }
9985        }
9986        if let Some((nindexes, table_addr)) = table {
9987            let nindexes = u8::try_from(nindexes).map_err(|_| {
9988                crate::io::IoError::InvalidState(format!("{nindexes} shared-message indexes"))
9989            })?;
9990            messages.push(crate::io::object_header_io::ExtensionMessage {
9991                msg_type: MSG_SHARED_MESSAGE_TABLE,
9992                flags: MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
9993                body: SharedMessageTableMessage {
9994                    version: 0,
9995                    table_address: table_addr,
9996                    nindexes,
9997                }
9998                .encode(&self.ctx),
9999            });
10000        }
10001        let encode = |messages: &[crate::io::object_header_io::ExtensionMessage]| {
10002            let mut extension = ObjectHeader::new();
10003            for msg in messages {
10004                extension.add_message(msg.msg_type, msg.flags, msg.body.clone());
10005            }
10006            extension.encode_v1(1)
10007        };
10008        let image = encode(&messages)?;
10009        // Freed before the replacement is placed, so a reopen reuses the block
10010        // instead of stranding one per open/close cycle — the rule every other
10011        // superseded structure follows.
10012        for &(addr, len) in &self.extension.superseded {
10013            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
10014        }
10015        let addr = self
10016            .allocator
10017            .allocate(image.len() as u64, FreeSpaceClass::Metadata);
10018
10019        // Every block of this file now has an address, so the allocator holds
10020        // exactly the file's free space: settle the free-space managers over
10021        // it and say in this extension where they went.
10022        let image = match self.write_free_space_managers()? {
10023            None => image,
10024            Some(body) => {
10025                let msg = messages
10026                    .iter_mut()
10027                    .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10028                    .ok_or_else(|| {
10029                        crate::io::IoError::InvalidState(
10030                            "a persisting file lost its file-space info message".into(),
10031                        )
10032                    })?;
10033                // Same length as the declared body put in above, so the
10034                // image measured before the block was allocated still fits.
10035                if body.len() != msg.body.len() {
10036                    return Err(crate::io::IoError::InvalidState(format!(
10037                        "the file-space info message was laid out at {} bytes and \
10038                         written back at {}",
10039                        msg.body.len(),
10040                        body.len()
10041                    )));
10042                }
10043                msg.body = body;
10044                encode(&messages)?
10045            }
10046        };
10047        self.handle.write_at(addr, &image)?;
10048        *self.extension.addr.lock() = Some(addr);
10049        Ok(())
10050    }
10051
10052    /// Define a new contiguous dataset. Returns the dataset index (used with
10053    /// `write_dataset_raw`).
10054    ///
10055    /// The raw-data region is allocated immediately so that
10056    /// `write_dataset_raw` can be called at any time before `close()`.
10057    pub fn create_dataset(
10058        &self,
10059        name: &str,
10060        datatype: DatatypeMessage,
10061        dims: &[u64],
10062    ) -> IoResult<usize> {
10063        let create = self.begin_create(name)?;
10064        let name = create.name.as_str();
10065        let total_elements: u64 = if dims.is_empty() {
10066            1
10067        } else {
10068            dims.iter().product()
10069        };
10070        let element_size = datatype.element_size() as u64;
10071        let data_size = total_elements * element_size;
10072
10073        // Allocate space for the raw data.
10074        let data_addr = if data_size > 0 {
10075            self.allocator.allocate(data_size, FreeSpaceClass::RawData)
10076        } else {
10077            UNDEF_ADDR
10078        };
10079
10080        let dataspace = if dims.is_empty() {
10081            DataspaceMessage::scalar()
10082        } else {
10083            DataspaceMessage::simple(dims)
10084        };
10085
10086        let idx = self.push_dataset(
10087            &create,
10088            DatasetInfo {
10089                name: name.to_string(),
10090                datatype,
10091                committed_type: None,
10092                external: None,
10093                virtual_storage: None,
10094                dataspace,
10095                read_format: None,
10096                obj_header_addr: 0, // set during finalize
10097                data_addr,
10098                data_size,
10099                compact: None,
10100                chunked: None,
10101                fixed_array: None,
10102                implicit: None,
10103                single_chunk: None,
10104                btree_v1: None,
10105                btree_v2: None,
10106                append: None,
10107                attributes: Vec::new(),
10108                obj_header_written_addr: None,
10109                obj_header_blocks: Vec::new(),
10110                filter_pipeline: None,
10111                deleted: false,
10112                extent_dirty: false,
10113                header_dirty: false,
10114                nlink_written: 1,
10115                creation_seq: self.take_creation_seq(),
10116                track_attr_order: self.track_order.attrs,
10117                fill_value: None,
10118                fill_time: FILL_TIME_IFSET,
10119                layout_version: 4,
10120                times: self.created_object_times(),
10121            },
10122        );
10123
10124        Ok(idx)
10125    }
10126
10127    /// Define a new dataset whose raw data lives in files outside this one —
10128    /// `H5Pset_external`, h5py's `external=[(name, offset, size)]`.
10129    ///
10130    /// Each entry names a file, the byte offset in it where that entry's
10131    /// region starts, and how many bytes of the dataset the region holds; the
10132    /// entries concatenate, in order, into the dataset's logical byte range,
10133    /// and together must cover it. Nothing is allocated in this file: the data
10134    /// layout message says contiguous storage at an undefined address, and it
10135    /// is the External File List beside it that says where the bytes are
10136    /// (`H5D__layout_oh_create`).
10137    ///
10138    /// A named file is created on first write and never truncated, so several
10139    /// slots — or several datasets — may own disjoint ranges of one file, the
10140    /// way `H5D__efl_write` opens them.
10141    ///
10142    /// The last slot may take the unlimited size `H5O_EFL_UNLIMITED`, which
10143    /// makes it absorb however many bytes the dataset comes to hold; a
10144    /// dataset whose dataspace is unlimited must have one, since nothing
10145    /// finite could cover it (`H5D__efl_construct`: "unlimited dataspace but
10146    /// finite storage"). Only the first dimension may be extendible, which is
10147    /// the same function's other rule.
10148    pub fn create_external_dataset(
10149        &self,
10150        name: &str,
10151        datatype: DatatypeMessage,
10152        dims: &[u64],
10153        max_dims: Option<&[u64]>,
10154        files: &[(&str, u64, u64)],
10155    ) -> IoResult<usize> {
10156        if files.is_empty() {
10157            return Err(crate::io::IoError::InvalidState(format!(
10158                "external dataset '{name}' names no files; external storage is defined by \
10159                 the files it lives in, so at least one is required"
10160            )));
10161        }
10162        let create = self.begin_create(name)?;
10163        let name = create.name.as_str();
10164        let total_elements: u64 = if dims.is_empty() {
10165            1
10166        } else {
10167            dims.iter().product()
10168        };
10169        let data_size = total_elements * datatype.element_size() as u64;
10170
10171        let mut heap = LocalHeapImage::with_empty_string();
10172        let mut entries = Vec::with_capacity(files.len());
10173        for (i, &(file_name, offset, size)) in files.iter().enumerate() {
10174            if file_name.is_empty() {
10175                return Err(crate::io::IoError::InvalidState(format!(
10176                    "external dataset '{name}' has a slot with an empty file name"
10177                )));
10178            }
10179            // `H5Pset_external` refuses to add a slot behind an unlimited one
10180            // ("previous file size is unlimited"): the unlimited slot already
10181            // owns every byte from its own start onwards, so nothing after it
10182            // could ever be reached.
10183            if size == UNLIMITED && i + 1 != files.len() {
10184                return Err(crate::io::IoError::InvalidState(format!(
10185                    "external dataset '{name}' gives slot {i} ('{file_name}') the unlimited \
10186                     size H5O_EFL_UNLIMITED with {} slot(s) behind it; an unlimited slot \
10187                     absorbs the rest of the dataset, so it can only be the last",
10188                    files.len() - i - 1
10189                )));
10190            }
10191            if offset.checked_add(size).is_none() {
10192                return Err(crate::io::IoError::InvalidState(format!(
10193                    "external dataset '{name}' slot '{file_name}' spans offset {offset} \
10194                     plus {size} bytes, past the end of the 64-bit address space"
10195                )));
10196            }
10197            entries.push(ExternalFile {
10198                name: file_name.to_string(),
10199                name_offset: heap.insert_str(file_name),
10200                offset,
10201                size,
10202            });
10203        }
10204        let external = ExternalStorage {
10205            // Filled in below, once the heap the names went into has an
10206            // address; the names' offsets within it are already final.
10207            heap_addr: UNDEF_ADDR,
10208            files: entries,
10209            // Settled by the open this create hands a handle out for, which
10210            // is `H5D__create` reading the dapl at H5Dint.c:1318.
10211            prefix: EfilePrefix::default(),
10212        };
10213        // `H5D__efl_construct`, over the dataset's *maximum* extent: the
10214        // slots must reserve at least every byte the dataset could come to
10215        // hold, and an unlimited extent can only be covered by an unlimited
10216        // last slot ("unlimited dataspace but finite storage").
10217        let max_dims = max_dims.unwrap_or(dims);
10218        if max_dims.len() != dims.len() {
10219            return Err(crate::io::IoError::InvalidState(format!(
10220                "external dataset '{name}' has {} dimensions but {} maximum ones",
10221                dims.len(),
10222                max_dims.len()
10223            )));
10224        }
10225        for (d, (&max, &cur)) in max_dims.iter().zip(dims).enumerate().skip(1) {
10226            if max > cur {
10227                return Err(crate::io::IoError::InvalidState(format!(
10228                    "external dataset '{name}' makes dimension {d} extendible ({cur} of \
10229                     {max}); only the first dimension can be extendible for external storage"
10230                )));
10231            }
10232        }
10233        let reserved = external.total_size();
10234        if max_dims.contains(&u64::MAX) {
10235            if reserved != UNLIMITED {
10236                return Err(crate::io::IoError::InvalidState(format!(
10237                    "external dataset '{name}' has an unlimited dataspace but its files \
10238                     reserve only {reserved} bytes; the last slot must take the unlimited \
10239                     size H5O_EFL_UNLIMITED"
10240                )));
10241            }
10242        } else {
10243            let max_bytes = max_dims
10244                .iter()
10245                .try_fold(datatype.element_size() as u64, |acc, &d| acc.checked_mul(d))
10246                .ok_or_else(|| {
10247                    crate::io::IoError::InvalidState(format!(
10248                        "external dataset '{name}' maximum extent times its element size \
10249                         overflows 64 bits"
10250                    ))
10251                })?;
10252            if reserved < max_bytes {
10253                return Err(crate::io::IoError::InvalidState(format!(
10254                    "external dataset '{name}' needs {max_bytes} bytes but its files reserve \
10255                     only {reserved}"
10256                )));
10257            }
10258        }
10259
10260        // The names' heap, written now: it is ordinary metadata of this file,
10261        // and the message the header carries is only an address into it.
10262        let sa = self.ctx.sizeof_addr as usize;
10263        let ss = self.ctx.sizeof_size as usize;
10264        let heap_bytes = heap.as_bytes().to_vec();
10265        let heap_addr = self.allocator.allocate(
10266            local_heap_header_size(sa, ss) as u64,
10267            FreeSpaceClass::Metadata,
10268        );
10269        let heap_data_addr = self
10270            .allocator
10271            .allocate(heap_bytes.len() as u64, FreeSpaceClass::Metadata);
10272        let heap_hdr = LocalHeapHeader {
10273            data_size: heap_bytes.len() as u64,
10274            // Sized to hold exactly these names, so no block of it is free.
10275            free_list_offset: LOCAL_HEAP_FREE_NULL,
10276            data_addr: heap_data_addr,
10277        };
10278        self.handle.write_at(heap_addr, &heap_hdr.encode(sa, ss))?;
10279        self.handle.write_at(heap_data_addr, &heap_bytes)?;
10280        let external = ExternalStorage {
10281            heap_addr,
10282            ..external
10283        };
10284
10285        let dataspace = if dims.is_empty() {
10286            DataspaceMessage::scalar()
10287        } else {
10288            let mut ds = DataspaceMessage::simple(dims);
10289            if max_dims != dims {
10290                ds.max_dims = Some(max_dims.to_vec());
10291            }
10292            ds
10293        };
10294
10295        let idx = self.push_dataset(
10296            &create,
10297            DatasetInfo {
10298                name: name.to_string(),
10299                datatype,
10300                committed_type: None,
10301                external: Some(external),
10302                virtual_storage: None,
10303                dataspace,
10304                read_format: None,
10305                obj_header_addr: 0, // set during finalize
10306                // No block of this file's own: the layout message declares
10307                // contiguous storage at an undefined address, which is what
10308                // sends a reader to the external file list instead.
10309                data_addr: UNDEF_ADDR,
10310                data_size,
10311                compact: None,
10312                chunked: None,
10313                fixed_array: None,
10314                btree_v2: None,
10315                implicit: None,
10316                single_chunk: None,
10317                btree_v1: None,
10318                append: None,
10319                attributes: Vec::new(),
10320                obj_header_written_addr: None,
10321                obj_header_blocks: Vec::new(),
10322                filter_pipeline: None,
10323                deleted: false,
10324                extent_dirty: false,
10325                header_dirty: false,
10326                nlink_written: 1,
10327                creation_seq: self.take_creation_seq(),
10328                track_attr_order: self.track_order.attrs,
10329                fill_value: None,
10330                fill_time: FILL_TIME_IFSET,
10331                layout_version: 4,
10332                times: self.created_object_times(),
10333            },
10334        );
10335
10336        Ok(idx)
10337    }
10338
10339    /// Define a new virtual dataset — `H5Pset_virtual`, h5py's
10340    /// `create_virtual_dataset(name, VirtualLayout)`.
10341    ///
10342    /// Each mapping says which elements of this dataset (`virtual_selection`)
10343    /// are read from which elements (`source_selection`) of a dataset in
10344    /// another file; the sources are never opened here, and a mapping naming
10345    /// one that does not exist yet is perfectly legal — libhdf5 resolves each
10346    /// at read time, filling from the fill value where nothing maps.
10347    ///
10348    /// The mappings do not live in the object header: they are serialized
10349    /// into one global heap object and the layout message carries only its
10350    /// address and index (`H5D__virtual_store_layout`), which is why this
10351    /// allocates a heap object and nothing else.
10352    ///
10353    /// An unlimited (`H5S_UNLIMITED`) selection is written as one: the
10354    /// mapping grows with its source, and the virtual dataset's extent in
10355    /// that dimension is whatever the sources reachable at read time supply
10356    /// (`H5D__virtual_set_extent_unlim`). A `printf`-style source name is
10357    /// written as one too: `%b` substitutes the block index, so one mapping
10358    /// stands for the family of source datasets that fill the successive
10359    /// blocks of an unlimited virtual selection.
10360    pub fn create_virtual_dataset(
10361        &self,
10362        name: &str,
10363        datatype: DatatypeMessage,
10364        dims: &[u64],
10365        max_dims: Option<&[u64]>,
10366        mappings: &[VirtualMapping],
10367    ) -> IoResult<usize> {
10368        if mappings.is_empty() {
10369            return Err(crate::io::IoError::InvalidState(format!(
10370                "virtual dataset '{name}' names no mappings; a virtual dataset is defined \
10371                 by the source datasets it maps, so at least one is required"
10372            )));
10373        }
10374        for m in mappings {
10375            check_virtual_mapping(name, m)?;
10376        }
10377
10378        let create = self.begin_create(name)?;
10379        let name = create.name.as_str();
10380
10381        // The mapping list is ordinary file metadata, written now: the header
10382        // built at finalize carries only the heap address and object index it
10383        // lands at.
10384        let block = VirtualMappingList {
10385            mappings: mappings.to_vec(),
10386        }
10387        .encode(&self.ctx)?;
10388        let (heap_addr, heap_index) = self.insert_vlen_objects(&[&block])?[0];
10389
10390        let dataspace = if dims.is_empty() {
10391            DataspaceMessage::scalar()
10392        } else {
10393            let mut ds = DataspaceMessage::simple(dims);
10394            // A caller that named no maximum gets the current dimensions, the
10395            // maximum `simple` already filled in: `H5Screate_simple(rank,
10396            // dims, NULL)` reaches the encoder with `extent.max` set
10397            // (H5S.c:1293-1299), so leaving it absent here would write a
10398            // message no upstream API call can produce.
10399            if let Some(max) = max_dims {
10400                ds.max_dims = Some(max.to_vec());
10401            }
10402            ds
10403        };
10404
10405        let idx = self.push_dataset(
10406            &create,
10407            DatasetInfo {
10408                name: name.to_string(),
10409                datatype,
10410                committed_type: None,
10411                external: None,
10412                virtual_storage: Some(VirtualStorage {
10413                    heap_addr,
10414                    heap_index: heap_index as u32,
10415                    mappings: mappings.to_vec(),
10416                }),
10417                dataspace,
10418                read_format: None,
10419                obj_header_addr: 0, // set during finalize
10420                // Not a block of this file at all: every element is read out
10421                // of a source dataset, so there is nothing here to allocate
10422                // and nothing to free when the dataset is deleted.
10423                data_addr: UNDEF_ADDR,
10424                data_size: 0,
10425                compact: None,
10426                chunked: None,
10427                fixed_array: None,
10428                btree_v2: None,
10429                implicit: None,
10430                single_chunk: None,
10431                btree_v1: None,
10432                append: None,
10433                attributes: Vec::new(),
10434                obj_header_written_addr: None,
10435                obj_header_blocks: Vec::new(),
10436                filter_pipeline: None,
10437                deleted: false,
10438                extent_dirty: false,
10439                header_dirty: false,
10440                nlink_written: 1,
10441                creation_seq: self.take_creation_seq(),
10442                track_attr_order: self.track_order.attrs,
10443                fill_value: None,
10444                fill_time: FILL_TIME_IFSET,
10445                layout_version: 4,
10446                times: self.created_object_times(),
10447            },
10448        );
10449
10450        Ok(idx)
10451    }
10452
10453    /// Define a new compact dataset — `H5Pset_layout(dcpl, H5D_COMPACT)`.
10454    ///
10455    /// The raw data lives inside the data layout message in the dataset's own
10456    /// object header, so it costs no block of its own and no extra seek to
10457    /// read; the price is the ceiling, and that the whole image is rewritten
10458    /// whenever the header is. The buffer is created at its final length and
10459    /// zero-filled, which is what `H5D__compact_fill` does at create time, so
10460    /// a dataset never written still reads back as its fill value.
10461    ///
10462    /// Errors when the image exceeds [`MAX_COMPACT_DATA`].
10463    pub fn create_compact_dataset(
10464        &self,
10465        name: &str,
10466        datatype: DatatypeMessage,
10467        dims: &[u64],
10468    ) -> IoResult<usize> {
10469        let total_elements: u64 = if dims.is_empty() {
10470            1
10471        } else {
10472            dims.iter().product()
10473        };
10474        let data_size = total_elements * datatype.element_size() as u64;
10475        if data_size > MAX_COMPACT_DATA as u64 {
10476            return Err(crate::io::IoError::InvalidState(format!(
10477                "compact dataset '{name}' needs {data_size} bytes, above the \
10478                 {MAX_COMPACT_DATA}-byte ceiling a data layout message can hold; \
10479                 use contiguous or chunked storage"
10480            )));
10481        }
10482
10483        let create = self.begin_create(name)?;
10484        let name = create.name.as_str();
10485        let dataspace = if dims.is_empty() {
10486            DataspaceMessage::scalar()
10487        } else {
10488            DataspaceMessage::simple(dims)
10489        };
10490
10491        let idx = self.push_dataset(
10492            &create,
10493            DatasetInfo {
10494                name: name.to_string(),
10495                datatype,
10496                committed_type: None,
10497                external: None,
10498                virtual_storage: None,
10499                dataspace,
10500                read_format: None,
10501                obj_header_addr: 0, // set during finalize
10502                data_addr: UNDEF_ADDR,
10503                data_size: 0,
10504                compact: Some(vec![0u8; data_size as usize]),
10505                chunked: None,
10506                fixed_array: None,
10507                implicit: None,
10508                single_chunk: None,
10509                btree_v1: None,
10510                btree_v2: None,
10511                append: None,
10512                attributes: Vec::new(),
10513                obj_header_written_addr: None,
10514                obj_header_blocks: Vec::new(),
10515                filter_pipeline: None,
10516                deleted: false,
10517                extent_dirty: false,
10518                header_dirty: false,
10519                nlink_written: 1,
10520                creation_seq: self.take_creation_seq(),
10521                track_attr_order: self.track_order.attrs,
10522                fill_value: None,
10523                fill_time: FILL_TIME_IFSET,
10524                layout_version: 4,
10525                times: self.created_object_times(),
10526            },
10527        );
10528
10529        Ok(idx)
10530    }
10531
10532    /// Define a new dataset with the NULL dataspace: no elements at all.
10533    ///
10534    /// Distinct from a scalar dataset (`create_dataset` with `dims == []`),
10535    /// which holds exactly one element — a NULL dataspace holds zero, so
10536    /// there is no raw image to allocate: `data_addr` stays `UNDEF_ADDR` and
10537    /// `data_size` stays 0 permanently, the same terminal state
10538    /// `create_dataset` already reaches for a zero-length dimension.
10539    pub fn create_null_dataset(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
10540        let create = self.begin_create(name)?;
10541        let name = create.name.as_str();
10542
10543        let idx = self.push_dataset(
10544            &create,
10545            DatasetInfo {
10546                name: name.to_string(),
10547                datatype,
10548                committed_type: None,
10549                external: None,
10550                virtual_storage: None,
10551                dataspace: DataspaceMessage::null(),
10552                read_format: None,
10553                obj_header_addr: 0, // set during finalize
10554                data_addr: UNDEF_ADDR,
10555                data_size: 0,
10556                compact: None,
10557                chunked: None,
10558                fixed_array: None,
10559                implicit: None,
10560                single_chunk: None,
10561                btree_v1: None,
10562                btree_v2: None,
10563                append: None,
10564                attributes: Vec::new(),
10565                obj_header_written_addr: None,
10566                obj_header_blocks: Vec::new(),
10567                filter_pipeline: None,
10568                deleted: false,
10569                extent_dirty: false,
10570                header_dirty: false,
10571                nlink_written: 1,
10572                creation_seq: self.take_creation_seq(),
10573                track_attr_order: self.track_order.attrs,
10574                fill_value: None,
10575                fill_time: FILL_TIME_IFSET,
10576                layout_version: 4,
10577                times: self.created_object_times(),
10578            },
10579        );
10580
10581        Ok(idx)
10582    }
10583
10584    /// Define a new chunked dataset with an extensible array index.
10585    ///
10586    /// Returns the dataset index. The dataset starts empty (dims[0] = 0 if
10587    /// the first dimension is unlimited). Use `write_chunk` and
10588    /// `extend_dataset` to add data.
10589    pub fn create_chunked_dataset(
10590        &self,
10591        name: &str,
10592        datatype: DatatypeMessage,
10593        dims: &[u64],
10594        max_dims: &[u64],
10595        chunk_dims: &[u64],
10596    ) -> IoResult<usize> {
10597        let create = self.begin_create(name)?;
10598        let name = create.name.as_str();
10599        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
10600        ensure_at_most_one_unlimited(max_dims)?;
10601        let chunk_bytes = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
10602        let layout_version = self.chunk_layout_version(false, chunk_bytes);
10603        let earray_params = EarrayParams::default_params();
10604        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
10605        let nsblk_addrs = compute_nsblk_addrs(
10606            earray_params.idx_blk_elmts,
10607            earray_params.data_blk_min_elmts,
10608            earray_params.sup_blk_min_data_ptrs,
10609            earray_params.max_nelmts_bits,
10610        )?;
10611
10612        // Create EA header
10613        let mut ea_header = ExtensibleArrayHeader::new_for_chunks(&self.ctx);
10614        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
10615        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
10616        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
10617        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
10618        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
10619
10620        // Allocate and write EA header (placeholder, will be updated)
10621        let hdr_encoded = ea_header.encode(&self.ctx);
10622        let ea_header_addr = self
10623            .allocator
10624            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
10625
10626        // Create EA index block with pre-allocated super block address slots
10627        let ea_iblk = ExtensibleArrayIndexBlock::new(
10628            ea_header_addr,
10629            earray_params.idx_blk_elmts,
10630            ndblk_addrs,
10631            nsblk_addrs,
10632        );
10633
10634        // Allocate and write EA index block
10635        let iblk_encoded = ea_iblk.encode(&self.ctx);
10636        let ea_iblk_addr = self
10637            .allocator
10638            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
10639
10640        // Update header with index block address
10641        ea_header.idx_blk_addr = ea_iblk_addr;
10642
10643        // Write both to disk
10644        let hdr_encoded = ea_header.encode(&self.ctx);
10645        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
10646        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
10647
10648        // Build dataspace with max dims
10649        let dataspace = DataspaceMessage {
10650            // Chunked storage always requires at least one dimension, so
10651            // this is never Scalar or Null.
10652            class: DataspaceClass::Simple,
10653            dims: dims.to_vec(),
10654            max_dims: Some(max_dims.to_vec()),
10655        };
10656
10657        let idx = self.push_dataset(
10658            &create,
10659            DatasetInfo {
10660                name: name.to_string(),
10661                datatype,
10662                committed_type: None,
10663                external: None,
10664                virtual_storage: None,
10665                dataspace,
10666                read_format: None,
10667                obj_header_addr: 0,
10668                data_addr: UNDEF_ADDR,
10669                data_size: 0,
10670                compact: None,
10671                attributes: Vec::new(),
10672                obj_header_written_addr: None,
10673                obj_header_blocks: Vec::new(),
10674                filter_pipeline: None,
10675                deleted: false,
10676                extent_dirty: false,
10677                header_dirty: false,
10678                nlink_written: 1,
10679                creation_seq: self.take_creation_seq(),
10680                track_attr_order: self.track_order.attrs,
10681                fill_value: None,
10682                fill_time: FILL_TIME_IFSET,
10683                layout_version,
10684                times: self.created_object_times(),
10685                fixed_array: None,
10686                implicit: None,
10687                single_chunk: None,
10688                btree_v1: None,
10689                btree_v2: None,
10690                chunked: Some(ChunkedDatasetInfo {
10691                    chunk_dims: chunk_dims.to_vec(),
10692                    earray_params,
10693                    ea_header_addr,
10694                    ea_iblk_addr,
10695                    ea_header,
10696                    ea_iblk,
10697                    chunks_written: 0,
10698                    filt_iblk: None,
10699                    chunk_size_len: 0,
10700                }),
10701                append: None,
10702            },
10703        );
10704
10705        Ok(idx)
10706    }
10707
10708    /// Write `data` into a contiguous dataset's raw storage at *dataset-
10709    /// relative* byte offset `off`.
10710    ///
10711    /// The single owner of a contiguous raw-data write. Which storage that is
10712    /// — a block of this file, or the files an External File List names — is
10713    /// decided once, by [`DatasetInfo::contiguous_target`], and never at a
10714    /// call site.
10715    fn write_contiguous_bytes(
10716        &self,
10717        target: &ContiguousTarget,
10718        off: u64,
10719        data: &[u8],
10720    ) -> IoResult<()> {
10721        match target {
10722            ContiguousTarget::Local(addr) => Ok(self.handle.write_at(addr + off, data)?),
10723            ContiguousTarget::External { files, prefix } => {
10724                // The prefix the open settled, not one resolved here:
10725                // `H5D__efl_write` joins against `dset->shared->extfile_prefix`
10726                // (H5Defl.c:429-431), the same field `H5D__efl_read` joins
10727                // against, so a relative name lands where a later read looks.
10728                write_external_file_bytes(files, prefix.as_deref(), off, data)
10729            }
10730            ContiguousTarget::Virtual => Err(virtual_write_refused()),
10731        }
10732    }
10733
10734    /// Write raw bytes to a contiguous dataset identified by `index`.
10735    ///
10736    /// The caller is responsible for providing data in the correct byte order
10737    /// and layout. The length must match the total data size declared at
10738    /// creation time.
10739    pub fn write_dataset_raw(&self, index: usize, data: &[u8]) -> IoResult<()> {
10740        let ds = self.ds(index);
10741        let _op = ds.op.lock();
10742        let target = {
10743            let mut g = ds.lock();
10744            if g.is_chunked() {
10745                return Err(crate::io::IoError::InvalidState(
10746                    "use write_chunk for chunked datasets".into(),
10747                ));
10748            }
10749            // A compact dataset's raw image is its layout message, so the
10750            // write lands in the buffer the header is built from rather than
10751            // at a file offset, and the header it is built into is now stale.
10752            if let Some(image) = g.compact.as_mut() {
10753                if data.len() != image.len() {
10754                    return Err(crate::io::IoError::InvalidState(format!(
10755                        "data size mismatch: expected {} bytes, got {}",
10756                        image.len(),
10757                        data.len()
10758                    )));
10759                }
10760                image.copy_from_slice(data);
10761                g.header_dirty = true;
10762                return Ok(());
10763            }
10764            let Some(target) = g.contiguous_target() else {
10765                return Err(crate::io::IoError::InvalidState(
10766                    "dataset has no data allocated".into(),
10767                ));
10768            };
10769            // A dataset that stores nothing of its own has no byte count to
10770            // check a write against — `write_contiguous_bytes` refuses it by
10771            // name below, which is the answer the caller needs.
10772            if target.is_storage() && data.len() as u64 != g.data_size {
10773                return Err(crate::io::IoError::InvalidState(format!(
10774                    "data size mismatch: expected {} bytes, got {}",
10775                    g.data_size,
10776                    data.len()
10777                )));
10778            }
10779            target
10780        };
10781        self.write_contiguous_bytes(&target, 0, data)
10782    }
10783
10784    /// Write a chunk of data to a chunked dataset.
10785    ///
10786    /// `chunk_offset` is the chunk coordinates (e.g., [frame_idx] for a 1D-chunked
10787    /// streaming dataset where chunk_dims = [1, H, W]).
10788    /// Only the first (unlimited) dimension index is used for EA indexing.
10789    ///
10790    /// `data` must be exactly chunk_size bytes (product of chunk_dims * element_size).
10791    pub fn write_chunk(&self, index: usize, chunk_idx: u64, data: &[u8]) -> IoResult<()> {
10792        let ds = self.ds(index);
10793        let _op = ds.op.lock();
10794        self.write_chunk_inner(index, chunk_idx, data)
10795    }
10796
10797    /// [`Self::write_chunk`] body; the caller holds the dataset's op lock or
10798    /// the writer exclusively.
10799    pub(crate) fn write_chunk_inner(
10800        &self,
10801        index: usize,
10802        chunk_idx: u64,
10803        data: &[u8],
10804    ) -> IoResult<()> {
10805        let ds = self.ds(index);
10806        // Read the chunk geometry and filter pipeline under one brief lock,
10807        // then drop it: compression runs *outside* the lock, and
10808        // `record_ea_chunk` re-locks the same slot, so the guard must not be
10809        // held across either.
10810        let (chunk_bytes, pipeline) = {
10811            let g = ds.lock();
10812            let element_size = g.datatype.element_size() as u64;
10813            let chunked = g
10814                .chunked
10815                .as_ref()
10816                .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?;
10817            (
10818                chunked.chunk_dims.iter().product::<u64>() * element_size,
10819                g.filter_pipeline.clone(),
10820            )
10821        };
10822
10823        if data.len() as u64 != chunk_bytes {
10824            return Err(crate::io::IoError::InvalidState(format!(
10825                "chunk data size mismatch: expected {} bytes, got {}",
10826                chunk_bytes,
10827                data.len()
10828            )));
10829        }
10830
10831        // Apply compression if filter pipeline is set
10832        let compressed;
10833        let write_data = if let Some(ref pipeline) = pipeline {
10834            compressed = filter::apply_filters(pipeline, data)?;
10835            &compressed
10836        } else {
10837            data
10838        };
10839        // filter_mask = 0: this path runs the whole pipeline, so no filter is
10840        // skipped for the chunk.
10841        self.record_ea_chunk(index, chunk_idx, write_data, 0)
10842    }
10843
10844    /// Decide where a chunk's bytes belong and put them there, returning the
10845    /// address to record in the index.
10846    ///
10847    /// `old` is the chunk's current `(address, stored length)` if the index
10848    /// already holds an entry for it. This is the single owner of the
10849    /// rewrite-placement rule, mirroring libhdf5's `H5D__chunk_file_alloc`
10850    /// (`H5Dchunk.c`): a chunk whose stored size is unchanged is overwritten
10851    /// where it already lives, and only a chunk that no longer fits moves,
10852    /// releasing its old block. Without this every rewrite would abandon the
10853    /// old block and grow the file.
10854    fn place_chunk(&self, old: Option<(u64, u64)>, new_len: u64) -> u64 {
10855        match old {
10856            // Same stored size: overwrite in place. This is every unfiltered
10857            // rewrite (the stored size is fixed by the chunk shape) and every
10858            // filtered rewrite that compressed to the same length.
10859            Some((addr, len)) if addr != UNDEF_ADDR && len == new_len => addr,
10860            Some((addr, len)) if addr != UNDEF_ADDR => {
10861                // The chunk has to move. Under SWMR a reader may still hold an
10862                // index that points at the old block, so libhdf5 keeps it
10863                // (H5D__chunk_file_alloc skips H5MF_xfree when the file is
10864                // open for SWMR writing); do the same.
10865                if !self.swmr_active {
10866                    self.allocator.free(addr, len, FreeSpaceClass::RawData);
10867                }
10868                self.allocator.allocate(new_len, FreeSpaceClass::RawData)
10869            }
10870            _ => self.allocator.allocate(new_len, FreeSpaceClass::RawData),
10871        }
10872    }
10873
10874    /// Place a chunk's already-final bytes (filtered if the dataset is
10875    /// filtered) in the file and record them in the extensible-array index —
10876    /// in the index block, a data block, or a super block per the EA geometry.
10877    /// Shared by write_chunk and write_compressed_chunk.
10878    ///
10879    /// The index lookup happens *before* the bytes are placed, because the
10880    /// entry it finds is what tells [`place_chunk`](Self::place_chunk) whether
10881    /// this is a rewrite that can stay put.
10882    fn record_ea_chunk(
10883        &self,
10884        index: usize,
10885        chunk_idx: u64,
10886        final_bytes: &[u8],
10887        filter_mask: u32,
10888    ) -> IoResult<()> {
10889        let compressed_size = final_bytes.len() as u64;
10890        let ds = self.ds(index);
10891        // Hold one slot guard for the whole method: every dataset-state access
10892        // below goes through `m`, while `self.handle`/`self.allocator`/`self.ctx`
10893        // are disjoint fields safe to touch with the guard held.
10894        let mut m = ds.lock();
10895        let is_filtered = m.filter_pipeline.is_some();
10896        // For a filtered dataset the chunk's stored size is encoded in the
10897        // `chunk_size_len`-byte field of each filtered EA entry
10898        // (`FilteredChunkEntry::encode` writes `nbytes[..chunk_size_len]`,
10899        // which truncates silently). Reject a size that would not fit, the way
10900        // libhdf5's H5D_CHUNK_ENCODE_SIZE_CHECK does, instead of corrupting the
10901        // index. The compress path never exceeds this (chunk_size_len holds the
10902        // uncompressed chunk size); a direct/raw write with caller-supplied
10903        // bytes can.
10904        if is_filtered {
10905            let chunk_size_len = m.chunked.as_ref().unwrap().chunk_size_len as usize;
10906            if chunk_size_len < 8 && compressed_size >= (1u64 << (chunk_size_len * 8)) {
10907                return Err(crate::io::IoError::InvalidState(format!(
10908                    "filtered chunk size {compressed_size} does not fit in the \
10909                     {chunk_size_len}-byte extensible-array chunk-size field"
10910                )));
10911            }
10912        }
10913        let idx_blk_elmts = {
10914            let c = m.chunked.as_ref().unwrap();
10915            c.earray_params.idx_blk_elmts as u64
10916        };
10917
10918        if chunk_idx < idx_blk_elmts {
10919            let chunked = m.chunked.as_mut().unwrap();
10920            if is_filtered {
10921                if let Some(ref mut fiblk) = chunked.filt_iblk {
10922                    let old = fiblk.elements[chunk_idx as usize];
10923                    let chunk_addr =
10924                        self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
10925                    self.handle.write_at(chunk_addr, final_bytes)?;
10926                    fiblk.elements[chunk_idx as usize] = FilteredChunkEntry {
10927                        addr: chunk_addr,
10928                        nbytes: compressed_size,
10929                        filter_mask,
10930                    };
10931                }
10932            } else {
10933                // An unfiltered chunk's stored size is fixed by the chunk
10934                // shape, so a rewrite always fits where it already is.
10935                let old = chunked.ea_iblk.elements[chunk_idx as usize];
10936                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
10937                self.handle.write_at(chunk_addr, final_bytes)?;
10938                chunked.ea_iblk.elements[chunk_idx as usize] = chunk_addr;
10939            }
10940            chunked.chunks_written += 1;
10941            if chunk_idx + 1 > chunked.ea_header.max_idx_set {
10942                chunked.ea_header.max_idx_set = chunk_idx + 1;
10943            }
10944            if chunked.ea_header.num_elmts_realized < idx_blk_elmts {
10945                chunked.ea_header.num_elmts_realized = idx_blk_elmts;
10946            }
10947        } else {
10948            // chunk_idx >= idx_blk_elmts: place the chunk through the EA
10949            // data-block / super-block hierarchy (libhdf5-compatible geometry).
10950            let (geo, max_nelmts_bits, chunk_size_len, ea_header_addr) = {
10951                let c = m.chunked.as_ref().unwrap();
10952                let p = &c.earray_params;
10953                (
10954                    EaGeometry::new(
10955                        p.idx_blk_elmts,
10956                        p.data_blk_min_elmts,
10957                        p.sup_blk_min_data_ptrs,
10958                        p.max_nelmts_bits,
10959                        p.max_dblk_page_nelmts_bits,
10960                    )?,
10961                    p.max_nelmts_bits,
10962                    c.chunk_size_len,
10963                    c.ea_header_addr,
10964                )
10965            };
10966            let loc = match geo.locate(chunk_idx)? {
10967                EaLoc::Dblk(l) => l,
10968                EaLoc::Index { .. } => unreachable!("chunk_idx >= idx_blk_elmts"),
10969            };
10970            if loc.paged {
10971                return Err(crate::io::IoError::InvalidState(format!(
10972                    "chunk index {} needs a paged extensible-array data block, \
10973                     which is not yet supported",
10974                    chunk_idx
10975                )));
10976            }
10977            let class_id = if is_filtered {
10978                EA_CLS_FILT_CHUNK
10979            } else {
10980                EA_CLS_CHUNK
10981            };
10982            let dblk_nelmts = loc.dblk_nelmts as usize;
10983
10984            // Resolve the data block's current address and its parent slot,
10985            // creating the owning super block on demand.
10986            let parent: DblkParent;
10987            let mut dblk_addr: u64;
10988            match loc.path {
10989                EaDblkPath::Direct { idx: di } => {
10990                    let c = m.chunked.as_ref().unwrap();
10991                    dblk_addr = if is_filtered {
10992                        c.filt_iblk.as_ref().unwrap().dblk_addrs[di]
10993                    } else {
10994                        c.ea_iblk.dblk_addrs[di]
10995                    };
10996                    parent = DblkParent::IndexBlock(di);
10997                }
10998                EaDblkPath::ViaSblk {
10999                    sblk_off,
11000                    local_dblk,
11001                    ndblks_in_sblk,
11002                    sblk_block_offset,
11003                } => {
11004                    let mut sblk_addr = {
11005                        let c = m.chunked.as_ref().unwrap();
11006                        if is_filtered {
11007                            c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
11008                        } else {
11009                            c.ea_iblk.sblk_addrs[sblk_off]
11010                        }
11011                    };
11012                    if sblk_addr == UNDEF_ADDR {
11013                        let sb = ExtensibleArraySuperBlock::new(
11014                            class_id,
11015                            ea_header_addr,
11016                            sblk_block_offset,
11017                            ndblks_in_sblk,
11018                        );
11019                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11020                        sblk_addr = self
11021                            .allocator
11022                            .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11023                        self.handle.write_at(sblk_addr, &enc)?;
11024                        let c = m.chunked.as_mut().unwrap();
11025                        if is_filtered {
11026                            c.filt_iblk.as_mut().unwrap().sblk_addrs[sblk_off] = sblk_addr;
11027                        } else {
11028                            c.ea_iblk.sblk_addrs[sblk_off] = sblk_addr;
11029                        }
11030                        c.ea_header.num_sblks_created += 1;
11031                        c.ea_header.size_sblks_created += enc.len() as u64;
11032                    }
11033                    let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
11034                    // The writer never creates paged super blocks (it errors
11035                    // before the paging threshold), so page_init_total is 0.
11036                    let sb = ExtensibleArraySuperBlock::decode(
11037                        &sb_buf,
11038                        &self.ctx,
11039                        max_nelmts_bits,
11040                        ndblks_in_sblk,
11041                        0,
11042                    )?;
11043                    dblk_addr = sb.dblk_addrs[local_dblk];
11044                    parent = DblkParent::SuperBlock {
11045                        sblk_addr,
11046                        ndblks_in_sblk,
11047                        local_dblk,
11048                    };
11049                }
11050            }
11051
11052            // Create or update the data block holding this chunk's entry.
11053            let created = dblk_addr == UNDEF_ADDR;
11054            if is_filtered {
11055                let mut dblk = if created {
11056                    FilteredDataBlock::new(ea_header_addr, loc.dblk_block_offset, dblk_nelmts)
11057                } else {
11058                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11059                    FilteredDataBlock::decode(
11060                        &buf,
11061                        &self.ctx,
11062                        max_nelmts_bits,
11063                        dblk_nelmts,
11064                        chunk_size_len,
11065                    )?
11066                };
11067                // A freshly created data block holds only undefined addresses,
11068                // so this reads as "no previous chunk" without a special case.
11069                let old = dblk.elements[loc.offset_in_dblk as usize];
11070                let chunk_addr = self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11071                self.handle.write_at(chunk_addr, final_bytes)?;
11072                let entry = FilteredChunkEntry {
11073                    addr: chunk_addr,
11074                    nbytes: compressed_size,
11075                    filter_mask,
11076                };
11077                dblk.elements[loc.offset_in_dblk as usize] = entry;
11078                let enc = dblk.encode(&self.ctx, max_nelmts_bits, chunk_size_len);
11079                if created {
11080                    dblk_addr = self
11081                        .allocator
11082                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11083                }
11084                self.handle.write_at(dblk_addr, &enc)?;
11085                if created {
11086                    let c = m.chunked.as_mut().unwrap();
11087                    c.ea_header.num_dblks_created += 1;
11088                    c.ea_header.size_dblks_created += enc.len() as u64;
11089                }
11090            } else {
11091                let mut dblk = if created {
11092                    ExtensibleArrayDataBlock::new(
11093                        ea_header_addr,
11094                        loc.dblk_block_offset,
11095                        dblk_nelmts,
11096                    )
11097                } else {
11098                    let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11099                    ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, dblk_nelmts)?
11100                };
11101                // Unfiltered: the stored size is fixed by the chunk shape, so
11102                // a rewrite always fits its old block. A freshly created data
11103                // block holds undefined addresses and falls through to a new
11104                // allocation.
11105                let old = dblk.elements[loc.offset_in_dblk as usize];
11106                let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11107                self.handle.write_at(chunk_addr, final_bytes)?;
11108                dblk.elements[loc.offset_in_dblk as usize] = chunk_addr;
11109                let enc = dblk.encode(&self.ctx, max_nelmts_bits);
11110                if created {
11111                    dblk_addr = self
11112                        .allocator
11113                        .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11114                }
11115                self.handle.write_at(dblk_addr, &enc)?;
11116                if created {
11117                    let c = m.chunked.as_mut().unwrap();
11118                    c.ea_header.num_dblks_created += 1;
11119                    c.ea_header.size_dblks_created += enc.len() as u64;
11120                }
11121            }
11122
11123            // Record a newly-created data block's address in its parent.
11124            if created {
11125                match parent {
11126                    DblkParent::IndexBlock(di) => {
11127                        let c = m.chunked.as_mut().unwrap();
11128                        if is_filtered {
11129                            c.filt_iblk.as_mut().unwrap().dblk_addrs[di] = dblk_addr;
11130                        } else {
11131                            c.ea_iblk.dblk_addrs[di] = dblk_addr;
11132                        }
11133                    }
11134                    DblkParent::SuperBlock {
11135                        sblk_addr,
11136                        ndblks_in_sblk,
11137                        local_dblk,
11138                    } => {
11139                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
11140                        let mut sb = ExtensibleArraySuperBlock::decode(
11141                            &buf,
11142                            &self.ctx,
11143                            max_nelmts_bits,
11144                            ndblks_in_sblk,
11145                            0,
11146                        )?;
11147                        sb.dblk_addrs[local_dblk] = dblk_addr;
11148                        let enc = sb.encode(&self.ctx, max_nelmts_bits);
11149                        self.handle.write_at(sblk_addr, &enc)?;
11150                    }
11151                }
11152            }
11153
11154            // Statistics.
11155            let c = m.chunked.as_mut().unwrap();
11156            c.chunks_written += 1;
11157            if chunk_idx + 1 > c.ea_header.max_idx_set {
11158                c.ea_header.max_idx_set = chunk_idx + 1;
11159            }
11160            if created {
11161                c.ea_header.num_elmts_realized += loc.dblk_nelmts;
11162            }
11163        }
11164        Ok(())
11165    }
11166
11167    /// Write a slice (hyperslab) of data to a dataset, contiguous or chunked.
11168    ///
11169    /// `starts` and `counts` define the N-dimensional selection.
11170    /// `data` must be exactly `product(counts) * element_size` bytes.
11171    ///
11172    /// The selection is validated once here and then handed to the layout's
11173    /// own writer, so a caller never has to know which storage the dataset
11174    /// uses.
11175    pub fn write_slice(
11176        &self,
11177        index: usize,
11178        starts: &[u64],
11179        counts: &[u64],
11180        data: &[u8],
11181    ) -> IoResult<()> {
11182        let ds = self.ds(index);
11183        let _op = ds.op.lock();
11184        self.write_slice_inner(index, starts, counts, data)
11185    }
11186
11187    /// [`Self::write_slice`] body; the caller holds the dataset's op lock or
11188    /// the writer exclusively.
11189    pub(crate) fn write_slice_inner(
11190        &self,
11191        index: usize,
11192        starts: &[u64],
11193        counts: &[u64],
11194        data: &[u8],
11195    ) -> IoResult<()> {
11196        let ds_ref = self.ds(index);
11197        let ds = ds_ref.lock();
11198        let is_chunked = ds.is_chunked();
11199
11200        let dims = &ds.dataspace.dims;
11201        let element_size = ds.datatype.element_size() as u64;
11202        let ndims = dims.len();
11203
11204        if starts.len() != ndims || counts.len() != ndims {
11205            return Err(crate::io::IoError::InvalidState(
11206                "starts/counts length must match dataset rank".into(),
11207            ));
11208        }
11209        if ndims == 0 {
11210            return Err(crate::io::IoError::InvalidState(
11211                "write_slice does not support scalar datasets; use write_dataset_raw".into(),
11212            ));
11213        }
11214
11215        // Every hyperslab edge must stay inside the dataset; without this an
11216        // out-of-bounds selection writes raw bytes over neighbouring data.
11217        for d in 0..ndims {
11218            let end = starts[d]
11219                .checked_add(counts[d])
11220                .ok_or_else(|| crate::io::IoError::InvalidState("slice extent overflow".into()))?;
11221            if end > dims[d] {
11222                return Err(crate::io::IoError::InvalidState(format!(
11223                    "slice out of bounds in dimension {}: start {} + count {} exceeds extent {}",
11224                    d, starts[d], counts[d], dims[d]
11225                )));
11226            }
11227        }
11228
11229        let out_elems: u64 = counts.iter().product();
11230        if data.len() as u64 != out_elems * element_size {
11231            return Err(crate::io::IoError::InvalidState(format!(
11232                "data size mismatch: expected {} bytes, got {}",
11233                out_elems * element_size,
11234                data.len()
11235            )));
11236        }
11237
11238        // `dims` borrows the dataset slot; collect what the writers below need
11239        // so the guard can be dropped before they re-lock it.
11240        let dims = dims.clone();
11241        let target = ds.contiguous_target();
11242        drop(ds);
11243
11244        if is_chunked {
11245            // Rows the append buffer holds are not in the chunks yet; writing
11246            // them there anyway would be undone when the buffer flushes at
11247            // close. Hand them to the chunks first.
11248            self.flush_append_buffer_if_intersecting(index, starts[0], starts[0] + counts[0])?;
11249            return self.write_slice_chunked(index, starts, counts, data);
11250        }
11251        let Some(target) = target else {
11252            return Err(crate::io::IoError::InvalidState(
11253                "dataset has no data allocated".into(),
11254            ));
11255        };
11256
11257        // Write each maximal contiguous run in one write. Trailing
11258        // full-selected dimensions coalesce, mirroring the read path: a slice
11259        // with a full last axis becomes one write per outer index instead of
11260        // one write per last-axis row.
11261        for_each_contiguous_run(
11262            &dims,
11263            starts,
11264            counts,
11265            element_size,
11266            |dst_off, src_off, len| {
11267                self.write_contiguous_bytes(&target, dst_off, &data[src_off..src_off + len])
11268            },
11269        )?;
11270
11271        Ok(())
11272    }
11273
11274    /// Write a hyperslab into a chunked dataset, one chunk at a time.
11275    ///
11276    /// The selection is already validated by [`write_slice`](Self::write_slice).
11277    /// For each chunk the selection touches, the chunk's share of `data` is
11278    /// scattered into a whole-chunk buffer and the chunk is rewritten:
11279    ///
11280    /// - a chunk the selection covers completely is built from `data` alone —
11281    ///   nothing needs reading back (libhdf5 takes the same shortcut with the
11282    ///   `relax` flag of `H5D__chunk_lock`);
11283    /// - a chunk covered only in part starts from what is already stored, or
11284    ///   from a fill-value buffer when the chunk has never been written, so
11285    ///   neighbouring elements survive and untouched ones read as fill.
11286    ///
11287    /// An edge chunk that hangs past the dataset extent is always the partial
11288    /// case, so the region beyond the extent keeps its fill value.
11289    fn write_slice_chunked(
11290        &self,
11291        index: usize,
11292        starts: &[u64],
11293        counts: &[u64],
11294        data: &[u8],
11295    ) -> IoResult<()> {
11296        if counts.contains(&0) {
11297            return Ok(());
11298        }
11299        let geo = self.chunk_geometry(index)?;
11300        let ndims = geo.dims.len();
11301        if geo.chunk_dims.len() != ndims {
11302            return Err(crate::io::IoError::InvalidState(format!(
11303                "dataset chunk shape has {} dimensions but the dataspace has {}",
11304                geo.chunk_dims.len(),
11305                ndims
11306            )));
11307        }
11308        if geo.chunk_dims.contains(&0) {
11309            return Err(crate::io::IoError::InvalidState(
11310                "chunk shape has a zero-length dimension".into(),
11311            ));
11312        }
11313        let chunk_bytes = geo.chunk_bytes() as usize;
11314
11315        // Grid range the selection touches, inclusive on both ends.
11316        let first: Vec<u64> = (0..ndims).map(|d| starts[d] / geo.chunk_dims[d]).collect();
11317        let last: Vec<u64> = (0..ndims)
11318            .map(|d| (starts[d] + counts[d] - 1) / geo.chunk_dims[d])
11319            .collect();
11320
11321        let mut coords = first.clone();
11322        loop {
11323            // Intersect the selection with this chunk. `in_chunk` is the
11324            // region's origin inside the chunk, `in_data` its origin inside
11325            // the caller's counts-shaped buffer, `extent` its size.
11326            let mut in_chunk = vec![0u64; ndims];
11327            let mut in_data = vec![0u64; ndims];
11328            let mut extent = vec![0u64; ndims];
11329            let mut covers_whole_chunk = true;
11330            for d in 0..ndims {
11331                let chunk_origin = coords[d] * geo.chunk_dims[d];
11332                let lo = starts[d].max(chunk_origin);
11333                let hi = (starts[d] + counts[d]).min(chunk_origin + geo.chunk_dims[d]);
11334                in_chunk[d] = lo - chunk_origin;
11335                in_data[d] = lo - starts[d];
11336                extent[d] = hi - lo;
11337                if in_chunk[d] != 0 || extent[d] != geo.chunk_dims[d] {
11338                    covers_whole_chunk = false;
11339                }
11340            }
11341
11342            let mut buf = if covers_whole_chunk {
11343                // Every byte is overwritten below.
11344                vec![0u8; chunk_bytes]
11345            } else {
11346                match self.read_chunk_at_coords(index, &coords)? {
11347                    Some(existing) => {
11348                        if existing.len() != chunk_bytes {
11349                            return Err(crate::io::IoError::InvalidState(format!(
11350                                "stored chunk at {coords:?} is {} bytes but the chunk shape \
11351                                 needs {chunk_bytes}",
11352                                existing.len()
11353                            )));
11354                        }
11355                        existing
11356                    }
11357                    None => self.new_write_chunk_buffer(index, chunk_bytes),
11358                }
11359            };
11360
11361            for_each_dual_run(
11362                &geo.chunk_dims,
11363                &in_chunk,
11364                counts,
11365                &in_data,
11366                &extent,
11367                geo.element_size,
11368                |dst_off, src_off, len| {
11369                    let dst = dst_off as usize;
11370                    let src = src_off as usize;
11371                    buf[dst..dst + len].copy_from_slice(&data[src..src + len]);
11372                    Ok(())
11373                },
11374            )?;
11375            self.write_chunk_at_coords(index, &coords, &buf)?;
11376
11377            // Odometer over the touched grid range.
11378            let mut d = ndims;
11379            loop {
11380                if d == 0 {
11381                    return Ok(());
11382                }
11383                d -= 1;
11384                if coords[d] < last[d] {
11385                    coords[d] += 1;
11386                    break;
11387                }
11388                coords[d] = first[d];
11389            }
11390        }
11391    }
11392
11393    /// Add an attribute to the root group (file-level attribute), replacing
11394    /// a same-name attribute. See [`set_attribute`](Self::set_attribute).
11395    pub fn add_root_attribute(&self, attr: AttributeMessage) -> IoResult<()> {
11396        self.set_attribute(AttrTarget::Root, attr)
11397    }
11398
11399    /// Insert `attr` into the attribute list `target` names, replacing a
11400    /// same-name attribute.
11401    ///
11402    /// The single owner of attribute-list mutation: an `AttributeMessage`
11403    /// that leaves a list here has its vlen global-heap objects released, so
11404    /// no replacement — vlen over vlen, numeric over vlen — can strand heap
11405    /// space (the attribute counterpart of issue #10's dataset fix).
11406    ///
11407    /// Under SWMR every attribute mutation is refused, matching libhdf5's
11408    /// rule for SWMR writes. Object headers are frozen once streaming
11409    /// starts — a change was committed at close only when the header
11410    /// happened to be rebuilt (group attrs always, dataset attrs only if
11411    /// the dataset also got chunk writes) and silently dropped otherwise —
11412    /// and a replacement's superseded vlen value could never be reclaimed,
11413    /// since a streaming reader may hold its heap references.
11414    pub fn set_attribute(&self, target: AttrTarget<'_>, attr: AttributeMessage) -> IoResult<()> {
11415        self.insert_attribute(target, attr, Created)
11416    }
11417
11418    /// The body of [`set_attribute`](Self::set_attribute), told whether the
11419    /// attribute it is inserting is genuinely new — see [`AttrOrigin`].
11420    fn insert_attribute(
11421        &self,
11422        target: AttrTarget<'_>,
11423        attr: AttributeMessage,
11424        origin: AttrOrigin,
11425    ) -> IoResult<()> {
11426        if self.swmr_active {
11427            return Err(swmr_attr_error(&attr.name));
11428        }
11429        // Whatever this name meant before, it means the incoming message now.
11430        self.forget_attribute_reference(self.attr_scope(target)?, &attr.name);
11431        // No size gate: an attribute whose message is too large for the
11432        // 16-bit size field an object header message has spills the object's
11433        // whole attribute set to dense storage at finalize, exactly as
11434        // `H5O__attr_create` does. See `attributes_need_dense`.
11435        let mut entry = AttributeEntry::from(attr);
11436        let old = self.with_attr_list(target, |attrs| {
11437            if let Some(pos) = attrs.iter().position(|a| a.name() == entry.name()) {
11438                // `H5O__attr_write` replaces an existing attribute's value and
11439                // leaves its `crt_idx` alone: the attribute was not created
11440                // again, so its creation index does not move.
11441                entry.set_creation_index(attrs[pos].creation_index());
11442                Some(std::mem::replace(&mut attrs[pos], entry))
11443            } else {
11444                // `H5O__attr_create` stamps the set's running maximum onto the
11445                // new attribute and post-increments it — but only a create
11446                // reaches for it.
11447                entry.set_creation_index(match origin {
11448                    Created => Some(next_creation_index(attrs)),
11449                    Rewritten(kept) => kept,
11450                });
11451                attrs.push(entry);
11452                None
11453            }
11454        })?;
11455        match old {
11456            Some(old) => self.release_attr_vlen(&old),
11457            None => Ok(()),
11458        }
11459    }
11460
11461    /// Set a variable-length string attribute on `target`, replacing any
11462    /// same-name attribute.
11463    ///
11464    /// Owns the whole replacement sequence: the superseded attribute is
11465    /// removed and its heap objects released *before* the new value's
11466    /// collection is allocated — the free-before-alloc order (issue #10)
11467    /// that lets a reopen-replace loop land in the block it just freed
11468    /// instead of growing the file every session. The cost, as on the
11469    /// dataset path: a failure between the eviction and the insert below
11470    /// loses the attribute rather than leaking its heap space.
11471    pub fn set_vlen_string_attribute(
11472        &self,
11473        target: AttrTarget<'_>,
11474        name: &str,
11475        value: &str,
11476    ) -> IoResult<()> {
11477        let origin = self.evict_attr(target, name)?;
11478        let attr = self.vlen_string_attribute(name, value)?;
11479        self.insert_attribute(target, attr, origin)
11480    }
11481
11482    /// The array counterpart of
11483    /// [`set_vlen_string_attribute`](Self::set_vlen_string_attribute).
11484    pub fn set_vlen_string_array_attribute(
11485        &self,
11486        target: AttrTarget<'_>,
11487        name: &str,
11488        values: &[&str],
11489        dims: &[u64],
11490    ) -> IoResult<()> {
11491        let origin = self.evict_attr(target, name)?;
11492        let attr = self.vlen_string_array_attribute(name, values, dims)?;
11493        self.insert_attribute(target, attr, origin)
11494    }
11495
11496    /// Set an attribute on `target` whose value is the object references
11497    /// naming `paths` — h5py's `obj.attrs['ref'] = f['/target'].ref`.
11498    ///
11499    /// `dims` is the attribute's dataspace: empty for the scalar shape a
11500    /// single reference takes, `&[n]` for an array of them. Each path names a
11501    /// dataset or a group (`/` is the root group) and must already exist. What
11502    /// reaches the file is each target's object header address, which finalize
11503    /// assigns — so the paths are what is stored, and the attribute's message
11504    /// is built from them every time an object header is
11505    /// ([`object_attributes`](Self::object_attributes)). The message carries a
11506    /// zero image of the final width until then.
11507    pub fn set_object_reference_attribute(
11508        &self,
11509        target: AttrTarget<'_>,
11510        name: &str,
11511        paths: &[&str],
11512        dims: &[u64],
11513    ) -> IoResult<()> {
11514        let scope = self.attr_scope(target)?;
11515        // An empty `dims` is the scalar shape, whose one element the empty
11516        // product already reports.
11517        let elements: u64 = dims.iter().product();
11518        if elements != paths.len() as u64 {
11519            return Err(crate::io::IoError::InvalidState(format!(
11520                "attribute '{name}' shape {dims:?} needs {elements} references, got {}",
11521                paths.len()
11522            )));
11523        }
11524        // Resolve now as well as at finalize, so a path that names nothing is
11525        // reported at the call that got it wrong.
11526        for path in paths {
11527            self.object_reference_target(path)?;
11528        }
11529        let datatype = DatatypeMessage::object_reference(&self.ctx);
11530        let image = vec![0u8; paths.len() * datatype.element_size() as usize];
11531        let attr = if dims.is_empty() {
11532            AttributeMessage::scalar_numeric(name, datatype, image)
11533        } else {
11534            AttributeMessage::array_numeric(name, datatype, dims, image)
11535        };
11536        // Through the same owner as every other attribute, which is also what
11537        // drops any value this name carried before.
11538        self.set_attribute(target, attr)?;
11539        self.attribute_references
11540            .lock()
11541            .push(AttributeReferenceValue {
11542                scope,
11543                name: name.to_string(),
11544                targets: paths.iter().map(|p| (*p).to_string()).collect(),
11545            });
11546        Ok(())
11547    }
11548
11549    /// Take the attribute `name` off `target`'s list, releasing its heap
11550    /// objects. No-op when absent. Refused under SWMR — see
11551    /// [`set_attribute`](Self::set_attribute).
11552    ///
11553    /// What it answers is what the insert that follows it must be told: an
11554    /// attribute that was there is being rewritten and keeps its creation
11555    /// index, and one that was not is created.
11556    fn evict_attr(&self, target: AttrTarget<'_>, name: &str) -> IoResult<AttrOrigin> {
11557        if self.swmr_active {
11558            return Err(swmr_attr_error(name));
11559        }
11560        self.forget_attribute_reference(self.attr_scope(target)?, name);
11561        let old = self.with_attr_list(target, |attrs| {
11562            attrs
11563                .iter()
11564                .position(|a| a.name() == name)
11565                .map(|pos| attrs.remove(pos))
11566        })?;
11567        match old {
11568            Some(old) => {
11569                let origin = Rewritten(old.creation_index());
11570                self.release_attr_vlen(&old)?;
11571                Ok(origin)
11572            }
11573            None => Ok(Created),
11574        }
11575    }
11576
11577    /// Release the global-heap objects a superseded attribute owned.
11578    /// Recognizes top-level vlen datatypes only: a *compound* attribute
11579    /// with vlen members — which this crate cannot write, only a foreign
11580    /// file can carry — keeps its members' heap objects when replaced or
11581    /// deleted, the storage cost the foreign writer accepted. Every other
11582    /// class stores its value inline in the message. Per-object removal
11583    /// keeps collections shared with other refs (libhdf5-written files)
11584    /// intact.
11585    fn release_attr_vlen(&self, old: &AttributeEntry) -> IoResult<()> {
11586        use crate::format::messages::datatype::DatatypeMessage;
11587        // An attribute whose message this crate could not decode keeps
11588        // whatever heap space it references: releasing objects named by bytes
11589        // we cannot interpret would free storage that is still live.
11590        let Some(old) = old.readable() else {
11591            return Ok(());
11592        };
11593        if matches!(
11594            old.datatype,
11595            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
11596        ) {
11597            self.release_vlen_references(&old.data)?;
11598        }
11599        Ok(())
11600    }
11601
11602    /// Run `f` on the attribute list `target` names — the accessor every
11603    /// attribute mutation shares.
11604    fn with_attr_list<R>(
11605        &self,
11606        target: AttrTarget<'_>,
11607        f: impl FnOnce(&mut Vec<AttributeEntry>) -> R,
11608    ) -> IoResult<R> {
11609        match target {
11610            AttrTarget::Root => Ok(f(&mut self.root_attributes.lock())),
11611            AttrTarget::Group(path) => {
11612                let path = self.canonical_group_path(path);
11613                for grp in self.group_refs() {
11614                    let mut g = grp.lock();
11615                    if g.name == path && !g.deleted {
11616                        return Ok(f(&mut g.attributes));
11617                    }
11618                }
11619                Err(crate::io::IoError::NotFound(format!(
11620                    "group '{path}' not found"
11621                )))
11622            }
11623            AttrTarget::Dataset(index) => {
11624                let count = self.dataset_count();
11625                if index >= count {
11626                    return Err(crate::io::IoError::InvalidState(format!(
11627                        "dataset index {index} out of range (have {count})"
11628                    )));
11629                }
11630                let ds = self.ds(index);
11631                let mut m = ds.lock();
11632                // Every caller of this mutates the list, and a reopened
11633                // dataset's header is rewritten only when it is marked stale.
11634                m.header_dirty = true;
11635                Ok(f(&mut m.attributes))
11636            }
11637        }
11638    }
11639
11640    /// Store each of `items` as a global heap object and return its
11641    /// placement `(collection address, object index)`, in input order —
11642    /// the writer side of libhdf5's `H5HG_insert`.
11643    ///
11644    /// Placement follows libhdf5: a collection from the CWFS list takes an
11645    /// item when its free space holds the object *and* a residual
11646    /// free-space marker header (`encode_at_size` always emits the
11647    /// marker); what no listed collection can take goes into a fresh
11648    /// collection, spilling into another at the 65535-object index cap.
11649    /// One batch may therefore span several collections — invisible to
11650    /// readers, which resolve each reference's own collection address. An
11651    /// empty batch allocates nothing: an empty collection still encodes
11652    /// to the 4096-byte `H5HG_MINALLOC` minimum, a block nothing would
11653    /// reference. libhdf5 additionally tries to extend a nearly-full
11654    /// collection's block in place (`H5MF_try_extend`); this writer does
11655    /// not — an oversized item always starts a fresh collection.
11656    ///
11657    /// The `cwfs` lock is held across every read-modify-rewrite of a
11658    /// listed collection block: it serializes concurrent inserts (two
11659    /// datasets' writers can pack the same block) and inserts against
11660    /// [`release_vlen_references`](Self::release_vlen_references), which
11661    /// rewrites the same blocks when objects are freed.
11662    ///
11663    /// Under SWMR the CWFS list is neither consulted nor updated and every
11664    /// batch gets fresh collections: packing rewrites a block a streaming
11665    /// reader may be mid-walk on — the same reason `place_chunk` keeps a
11666    /// relocated chunk's old block.
11667    fn insert_vlen_objects(&self, items: &[&[u8]]) -> IoResult<Vec<(u64, u16)>> {
11668        use crate::format::global_heap::{GlobalHeapCollection, GlobalHeapObject};
11669
11670        if items.is_empty() {
11671            return Ok(Vec::new());
11672        }
11673        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
11674        let mut placements = Vec::with_capacity(items.len());
11675        let mut i = 0;
11676
11677        // Pack into listed collections while one can take the next item.
11678        if !self.swmr_active {
11679            let mut cwfs = self.cwfs.lock();
11680            while i < items.len() {
11681                let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
11682                let Some(pos) = cwfs.iter().position(|e| e.free >= need + objhdr) else {
11683                    // Second pass of libhdf5's H5F_cwfs_find_free_heap: no
11684                    // listed collection has room, so try to grow one in
11685                    // place before falling back to a fresh collection.
11686                    if self.extend_listed_collection(&mut cwfs, need + objhdr)? {
11687                        continue;
11688                    }
11689                    break;
11690                };
11691                let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
11692                let image = self.handle.read_at(addr, size)?;
11693                let (mut gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
11694                // The disk is the truth for free space; the entry is a hint.
11695                let Some(mut free) = gcol.free_space_at(&self.ctx, size) else {
11696                    cwfs.remove(pos);
11697                    continue;
11698                };
11699                let mut next_idx = gcol.max_index();
11700                let mut took = false;
11701                while i < items.len() && next_idx < u16::MAX {
11702                    let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
11703                    if free < need + objhdr {
11704                        break;
11705                    }
11706                    next_idx += 1;
11707                    gcol.objects.push(GlobalHeapObject {
11708                        index: next_idx,
11709                        ref_count: 0,
11710                        data: items[i].to_vec(),
11711                    });
11712                    placements.push((addr, next_idx));
11713                    free -= need;
11714                    took = true;
11715                    i += 1;
11716                }
11717                if took {
11718                    let rewritten = gcol.encode_at_size(&self.ctx, size)?;
11719                    self.handle.write_at(addr, &rewritten)?;
11720                    // Correct the entry to the measured free space and move
11721                    // it to the front — libhdf5 keeps `cwfs` in
11722                    // most-recently-used order.
11723                    let mut e = cwfs.remove(pos);
11724                    e.free = free;
11725                    cwfs.insert(0, e);
11726                } else if next_idx == u16::MAX {
11727                    // At the index cap nothing can be inserted no matter the
11728                    // free space; drop the entry or the scan re-picks it
11729                    // forever. (A removal can lower the top index again, and
11730                    // the release side re-lists the collection then.)
11731                    cwfs.remove(pos);
11732                } else {
11733                    // The hint overstated the block's free space — shrink it
11734                    // to the measured value so the scan moves on.
11735                    cwfs[pos].free = free;
11736                }
11737            }
11738        }
11739
11740        // What remains goes into fresh collections.
11741        while i < items.len() {
11742            let mut gcol = GlobalHeapCollection::new();
11743            // Objects are pushed with a running index: `add_object` rescans
11744            // for the max index per call, O(n²) across a spill-sized batch.
11745            let mut next_idx: u16 = 0;
11746            while i < items.len() && next_idx < u16::MAX {
11747                next_idx += 1;
11748                gcol.objects.push(GlobalHeapObject {
11749                    index: next_idx,
11750                    ref_count: 0,
11751                    data: items[i].to_vec(),
11752                });
11753                i += 1;
11754            }
11755            let encoded = gcol.encode(&self.ctx);
11756            let addr = self
11757                .allocator
11758                .allocate(encoded.len() as u64, FreeSpaceClass::RawData);
11759            self.handle.write_at(addr, &encoded)?;
11760            for idx in 1..=next_idx {
11761                placements.push((addr, idx));
11762            }
11763            // List the block's leftover free space for later inserts — the
11764            // minimum-size padding of a small batch is most of 4096 bytes.
11765            // Below two object headers not even an empty object fits.
11766            if !self.swmr_active {
11767                if let Some(free) = gcol.free_space_at(&self.ctx, encoded.len()) {
11768                    if free >= 2 * objhdr {
11769                        cwfs_note(&mut self.cwfs.lock(), addr, encoded.len(), free);
11770                    }
11771                }
11772            }
11773        }
11774        Ok(placements)
11775    }
11776
11777    /// Try to extend one listed collection in place so it can take an
11778    /// object needing `want` bytes of free space — the second pass of
11779    /// libhdf5's `H5F_cwfs_find_free_heap`: grow the file allocation
11780    /// ([`FileAllocator::try_extend`], mirroring `H5MF_try_extend`) and then
11781    /// the collection itself (`H5HG_extend`: a larger declared size and a
11782    /// free-space marker covering the new tail — here by re-encoding at the
11783    /// grown size, which writes exactly those two things).
11784    ///
11785    /// Extension size is `max(collection_size, shortfall)` — at least a
11786    /// doubling — capped so the result stays within [`GCOL_MAX_SIZE`], both
11787    /// as upstream computes them. On success the grown entry moves to the
11788    /// front of the list and the caller's scan re-picks it; the free-space
11789    /// measurement is taken from the block on disk, not the list's hint, so
11790    /// the rewrite and the entry agree.
11791    ///
11792    /// Caller holds the `cwfs` lock (it passes the guarded list), which is
11793    /// what serializes this read-modify-rewrite against concurrent inserts
11794    /// and releases.
11795    fn extend_listed_collection(&self, cwfs: &mut Vec<CwfsEntry>, want: usize) -> IoResult<bool> {
11796        use crate::format::global_heap::{GlobalHeapCollection, GCOL_MAX_SIZE};
11797
11798        let mut pos = 0;
11799        while pos < cwfs.len() {
11800            let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
11801            let image = self.handle.read_at(addr, size)?;
11802            let (gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
11803            // The disk is the truth for free space; the entry is a hint.
11804            let Some(free) = gcol.free_space_at(&self.ctx, size) else {
11805                cwfs.remove(pos);
11806                continue;
11807            };
11808            // A hint can understate the block (upstream's FREE_SIZE is its
11809            // in-memory truth and cannot): if the block already has room,
11810            // correct the hint instead of doubling the collection.
11811            if free >= want {
11812                cwfs[pos].free = free;
11813                return Ok(true);
11814            }
11815            let new_need = size.max(want.saturating_sub(free));
11816            if size + new_need > GCOL_MAX_SIZE
11817                || !self.allocator.try_extend(
11818                    addr,
11819                    size as u64,
11820                    new_need as u64,
11821                    FreeSpaceClass::RawData,
11822                )
11823            {
11824                pos += 1;
11825                continue;
11826            }
11827            let new_size = size + new_need;
11828            let rewritten = gcol.encode_at_size(&self.ctx, new_size)?;
11829            self.handle.write_at(addr, &rewritten)?;
11830            let mut e = cwfs.remove(pos);
11831            e.size = new_size;
11832            e.free = free + new_need;
11833            cwfs.insert(0, e);
11834            return Ok(true);
11835        }
11836        Ok(false)
11837    }
11838
11839    /// Create a variable-length string dataset and write string data.
11840    ///
11841    /// Stores strings in the global heap. The dataset raw data consists of
11842    /// vlen references (collection_addr + object_index pairs).
11843    ///
11844    /// `charset` is the datatype's declared character set (0 = ASCII,
11845    /// 1 = UTF-8); the strings are checked against it before anything is
11846    /// written, so the type never misdescribes the bytes under it.
11847    pub fn create_vlen_string_dataset(
11848        &self,
11849        name: &str,
11850        strings: &[&str],
11851        charset: u8,
11852    ) -> IoResult<usize> {
11853        use crate::format::global_heap::encode_vlen_reference;
11854        use crate::format::messages::datatype::DatatypeMessage;
11855
11856        ensure_vlen_charset(charset, strings)?;
11857
11858        let create = self.begin_create(name)?;
11859        let name = create.name.as_str();
11860        let num_strings = strings.len() as u64;
11861
11862        // Store the strings as heap objects; a batch that fits an earlier
11863        // collection's free space shares its block.
11864        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
11865        let placements = self.insert_vlen_objects(&items)?;
11866
11867        // Build raw data: vlen references
11868        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
11869        let data_size = (num_strings as usize) * ref_size;
11870        let mut raw_data = Vec::with_capacity(data_size);
11871        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
11872            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
11873            raw_data.extend_from_slice(&encode_vlen_reference(
11874                seq_len,
11875                gcol_addr,
11876                obj_idx as u32,
11877                &self.ctx,
11878            ));
11879        }
11880
11881        // Allocate and write raw data
11882        let data_addr = self
11883            .allocator
11884            .allocate(data_size as u64, FreeSpaceClass::RawData);
11885        self.handle.write_at(data_addr, &raw_data)?;
11886
11887        // Create the dataset with vlen string datatype
11888        let datatype = DatatypeMessage::VarLenString {
11889            padding: 0,
11890            charset,
11891        };
11892        let dataspace =
11893            crate::format::messages::dataspace::DataspaceMessage::simple(&[num_strings]);
11894
11895        let idx = self.push_dataset(
11896            &create,
11897            DatasetInfo {
11898                name: name.to_string(),
11899                datatype,
11900                committed_type: None,
11901                external: None,
11902                virtual_storage: None,
11903                dataspace,
11904                read_format: None,
11905                obj_header_addr: 0,
11906                data_addr,
11907                data_size: data_size as u64,
11908                compact: None,
11909                attributes: Vec::new(),
11910                obj_header_written_addr: None,
11911                obj_header_blocks: Vec::new(),
11912                filter_pipeline: None,
11913                deleted: false,
11914                extent_dirty: false,
11915                header_dirty: false,
11916                nlink_written: 1,
11917                creation_seq: self.take_creation_seq(),
11918                track_attr_order: self.track_order.attrs,
11919                fill_value: None,
11920                fill_time: FILL_TIME_IFSET,
11921                layout_version: 4,
11922                times: self.created_object_times(),
11923                chunked: None,
11924                fixed_array: None,
11925                implicit: None,
11926                single_chunk: None,
11927                btree_v1: None,
11928                btree_v2: None,
11929                append: None,
11930            },
11931        );
11932
11933        Ok(idx)
11934    }
11935
11936    /// Create a 1-D variable-length byte-array dataset.
11937    ///
11938    /// The `u8` case of [`create_vlen_sequence_dataset`], where an item's
11939    /// byte image and its element count are the same number.
11940    ///
11941    /// [`create_vlen_sequence_dataset`]: Self::create_vlen_sequence_dataset
11942    ///
11943    /// Superseded in production by [`write_vlen_numeric`](crate::H5File::write_vlen_numeric)
11944    /// (`H5Group::write_vlen_bytes` routes through it, not through here);
11945    /// kept as a direct entry point for this crate's own white-box tests.
11946    #[cfg(test)]
11947    pub fn create_vlen_bytes_dataset(&self, name: &str, items: &[&[u8]]) -> IoResult<usize> {
11948        use crate::format::messages::datatype::DatatypeMessage;
11949
11950        self.create_vlen_sequence_dataset(name, DatatypeMessage::u8_type(), items)
11951    }
11952
11953    /// Create a 1-D variable-length sequence dataset over `base`.
11954    ///
11955    /// Each item is the encoded image of one sequence — `n * base.element_size()`
11956    /// bytes in the base type's own byte order — and is stored as a global-heap
11957    /// object; the dataset holds one vlen reference per item, the same on-disk
11958    /// shape a vlen string dataset has. The `H5T_VLEN` length field counts base
11959    /// elements rather than bytes, so an image whose length is not a whole
11960    /// number of elements is refused here rather than stored under a length
11961    /// that misreads it.
11962    pub fn create_vlen_sequence_dataset(
11963        &self,
11964        name: &str,
11965        base: DatatypeMessage,
11966        items: &[&[u8]],
11967    ) -> IoResult<usize> {
11968        use crate::format::global_heap::encode_vlen_reference;
11969        use crate::format::messages::datatype::DatatypeMessage;
11970
11971        let elem_size = base.element_size() as usize;
11972        if elem_size == 0 {
11973            return Err(crate::io::IoError::InvalidState(format!(
11974                "vlen base datatype {base} has no element size"
11975            )));
11976        }
11977        for (i, item) in items.iter().enumerate() {
11978            if !item.len().is_multiple_of(elem_size) {
11979                return Err(crate::io::IoError::InvalidState(format!(
11980                    "sequence {i} is {} bytes, not a whole number of {elem_size}-byte elements",
11981                    item.len()
11982                )));
11983            }
11984        }
11985
11986        let create = self.begin_create(name)?;
11987        let name = create.name.as_str();
11988        let num_items = items.len() as u64;
11989
11990        // Store the sequence images as heap objects, sharing collection
11991        // blocks as `create_vlen_string_dataset` does.
11992        let placements = self.insert_vlen_objects(items)?;
11993
11994        // Build raw data: one vlen reference per item.
11995        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
11996        let data_size = (num_items as usize) * ref_size;
11997        let mut raw_data = Vec::with_capacity(data_size);
11998        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
11999            let seq_len = crate::format::global_heap::vlen_seq_len(items[i].len() / elem_size)?;
12000            raw_data.extend_from_slice(&encode_vlen_reference(
12001                seq_len,
12002                gcol_addr,
12003                obj_idx as u32,
12004                &self.ctx,
12005            ));
12006        }
12007
12008        // Allocate and write raw data.
12009        let data_addr = self
12010            .allocator
12011            .allocate(data_size as u64, FreeSpaceClass::RawData);
12012        self.handle.write_at(data_addr, &raw_data)?;
12013
12014        let datatype = DatatypeMessage::VarLenSequence {
12015            base: Box::new(base),
12016        };
12017        let dataspace = crate::format::messages::dataspace::DataspaceMessage::simple(&[num_items]);
12018
12019        let idx = self.push_dataset(
12020            &create,
12021            DatasetInfo {
12022                name: name.to_string(),
12023                datatype,
12024                committed_type: None,
12025                external: None,
12026                virtual_storage: None,
12027                dataspace,
12028                read_format: None,
12029                obj_header_addr: 0,
12030                data_addr,
12031                data_size: data_size as u64,
12032                compact: None,
12033                attributes: Vec::new(),
12034                obj_header_written_addr: None,
12035                obj_header_blocks: Vec::new(),
12036                filter_pipeline: None,
12037                deleted: false,
12038                extent_dirty: false,
12039                header_dirty: false,
12040                nlink_written: 1,
12041                creation_seq: self.take_creation_seq(),
12042                track_attr_order: self.track_order.attrs,
12043                fill_value: None,
12044                fill_time: FILL_TIME_IFSET,
12045                layout_version: 4,
12046                times: self.created_object_times(),
12047                chunked: None,
12048                fixed_array: None,
12049                implicit: None,
12050                single_chunk: None,
12051                btree_v1: None,
12052                btree_v2: None,
12053                append: None,
12054            },
12055        );
12056
12057        Ok(idx)
12058    }
12059
12060    /// Create a chunked, compressed variable-length string dataset.
12061    ///
12062    /// Strings are stored in the global heap (same as `create_vlen_string_dataset`),
12063    /// but the vlen references are stored in chunked layout with the given filter
12064    /// pipeline (e.g., deflate, zstd). `chunk_size` is the number of strings per chunk.
12065    pub fn create_vlen_string_dataset_compressed(
12066        &self,
12067        name: &str,
12068        strings: &[&str],
12069        chunk_size: usize,
12070        pipeline: FilterPipeline,
12071    ) -> IoResult<usize> {
12072        use crate::format::global_heap::encode_vlen_reference;
12073        use crate::format::messages::datatype::DatatypeMessage;
12074
12075        let create = self.begin_create(name)?;
12076        let name = create.name.as_str();
12077        let num_strings = strings.len() as u64;
12078        validate_chunk_geometry(&[num_strings], &[num_strings], &[chunk_size as u64])?;
12079
12080        // Store the strings as heap objects; the geometry validation above
12081        // must precede this so a refused call allocates nothing.
12082        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12083        let placements = self.insert_vlen_objects(&items)?;
12084
12085        // Build raw data: vlen references
12086        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12087        let data_size = (num_strings as usize) * ref_size;
12088        let mut raw_data = Vec::with_capacity(data_size);
12089        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12090            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12091            raw_data.extend_from_slice(&encode_vlen_reference(
12092                seq_len,
12093                gcol_addr,
12094                obj_idx as u32,
12095                &self.ctx,
12096            ));
12097        }
12098
12099        // Set up chunked compressed layout
12100        let datatype = DatatypeMessage::vlen_string_utf8();
12101        let element_size = datatype.element_size_ctx(&self.ctx) as u64;
12102        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12103        let dims: Vec<u64> = vec![num_strings];
12104        let max_dims: Vec<u64> = vec![num_strings];
12105        let chunk_bytes = chunk_size as u64 * element_size;
12106        let layout_version = self.chunk_layout_version(true, chunk_bytes);
12107        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
12108
12109        let earray_params = EarrayParams::default_params();
12110        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
12111        let nsblk_addrs = compute_nsblk_addrs(
12112            earray_params.idx_blk_elmts,
12113            earray_params.data_blk_min_elmts,
12114            earray_params.sup_blk_min_data_ptrs,
12115            earray_params.max_nelmts_bits,
12116        )?;
12117
12118        // Create filtered EA header
12119        let mut ea_header =
12120            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
12121        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
12122        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
12123        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
12124        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
12125        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
12126
12127        let hdr_encoded = ea_header.encode(&self.ctx);
12128        let ea_header_addr = self
12129            .allocator
12130            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
12131
12132        // Create filtered index block
12133        let filt_iblk = FilteredIndexBlock::new(
12134            ea_header_addr,
12135            earray_params.idx_blk_elmts,
12136            ndblk_addrs,
12137            nsblk_addrs,
12138        );
12139        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
12140        let ea_iblk_addr = self
12141            .allocator
12142            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
12143
12144        ea_header.idx_blk_addr = ea_iblk_addr;
12145
12146        let hdr_encoded = ea_header.encode(&self.ctx);
12147        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
12148        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
12149
12150        let dataspace = DataspaceMessage {
12151            // Chunked storage always requires at least one dimension, so
12152            // this is never Scalar or Null.
12153            class: DataspaceClass::Simple,
12154            dims: dims.to_vec(),
12155            max_dims: Some(max_dims.to_vec()),
12156        };
12157
12158        let ea_iblk = ExtensibleArrayIndexBlock::new(
12159            ea_header_addr,
12160            earray_params.idx_blk_elmts,
12161            ndblk_addrs,
12162            nsblk_addrs,
12163        );
12164
12165        let idx = self.push_dataset(
12166            &create,
12167            DatasetInfo {
12168                name: name.to_string(),
12169                datatype,
12170                committed_type: None,
12171                external: None,
12172                virtual_storage: None,
12173                dataspace,
12174                read_format: None,
12175                obj_header_addr: 0,
12176                data_addr: UNDEF_ADDR,
12177                data_size: 0,
12178                compact: None,
12179                attributes: Vec::new(),
12180                obj_header_written_addr: None,
12181                obj_header_blocks: Vec::new(),
12182                filter_pipeline: Some(pipeline),
12183                deleted: false,
12184                extent_dirty: false,
12185                header_dirty: false,
12186                nlink_written: 1,
12187                creation_seq: self.take_creation_seq(),
12188                track_attr_order: self.track_order.attrs,
12189                fill_value: None,
12190                fill_time: FILL_TIME_IFSET,
12191                layout_version,
12192                times: self.created_object_times(),
12193                fixed_array: None,
12194                implicit: None,
12195                single_chunk: None,
12196                btree_v1: None,
12197                btree_v2: None,
12198                chunked: Some(ChunkedDatasetInfo {
12199                    chunk_dims: chunk_dims.clone(),
12200                    earray_params,
12201                    ea_header_addr,
12202                    ea_iblk_addr,
12203                    ea_header,
12204                    ea_iblk,
12205                    chunks_written: 0,
12206                    filt_iblk: Some(filt_iblk),
12207                    chunk_size_len,
12208                }),
12209                append: None,
12210            },
12211        );
12212
12213        // Write chunks of vlen references with compression
12214        let chunk_byte_size = chunk_bytes as usize;
12215        let num_chunks = raw_data.len().div_ceil(chunk_byte_size);
12216        for chunk_i in 0..num_chunks {
12217            let start = chunk_i * chunk_byte_size;
12218            let end = (start + chunk_byte_size).min(raw_data.len());
12219            let chunk_data = if end - start < chunk_byte_size {
12220                // Pad last chunk to full size (vlen datasets carry no user
12221                // fill value, so this resolves to zero = null vlen reference).
12222                let mut padded = self.new_chunk_buffer(idx, chunk_byte_size);
12223                padded[..end - start].copy_from_slice(&raw_data[start..end]);
12224                padded
12225            } else {
12226                raw_data[start..end].to_vec()
12227            };
12228            self.write_chunk(idx, chunk_i as u64, &chunk_data)?;
12229        }
12230
12231        Ok(idx)
12232    }
12233
12234    /// Create an empty chunked vlen string dataset ready for incremental appends.
12235    ///
12236    /// The dataset starts with `dims = [0]` and `max_dims = [unlimited]`.
12237    /// Use `append_vlen_strings` to add data.
12238    pub fn create_appendable_vlen_string_dataset(
12239        &self,
12240        name: &str,
12241        chunk_size: usize,
12242        pipeline: Option<FilterPipeline>,
12243    ) -> IoResult<usize> {
12244        let datatype = DatatypeMessage::vlen_string_utf8();
12245        let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12246        let dims: Vec<u64> = vec![0];
12247        let max_dims: Vec<u64> = vec![u64::MAX];
12248
12249        if let Some(ref pl) = pipeline {
12250            self.create_chunked_dataset_with_pipeline(
12251                name,
12252                datatype,
12253                &dims,
12254                &max_dims,
12255                &chunk_dims,
12256                pl.clone(),
12257            )
12258        } else {
12259            self.create_chunked_dataset(name, datatype, &dims, &max_dims, &chunk_dims)
12260        }
12261    }
12262
12263    /// Append variable-length strings to an existing chunked vlen string dataset.
12264    ///
12265    /// Creates a new global heap collection for the strings, builds vlen
12266    /// references, and appends them as new chunks to the dataset.
12267    pub fn append_vlen_strings(&self, ds_index: usize, strings: &[&str]) -> IoResult<()> {
12268        use crate::format::global_heap::encode_vlen_reference;
12269        use crate::format::messages::datatype::DatatypeMessage;
12270
12271        if strings.is_empty() {
12272            return Ok(());
12273        }
12274
12275        // Whole-operation guard: buffer take, frame writes, re-buffer and
12276        // extend below are separate slot acquisitions that a concurrent
12277        // same-dataset append must not interleave with.
12278        let cell = self.ds(ds_index);
12279        let _op = cell.op.lock();
12280
12281        // The elements about to be written are vlen references; any other
12282        // element type would be overwritten with them as raw bytes.
12283        let charset = {
12284            let ds = self.ds(ds_index);
12285            let m = ds.lock();
12286            match m.datatype {
12287                DatatypeMessage::VarLenString { charset, .. } => charset,
12288                _ => {
12289                    return Err(crate::io::IoError::InvalidState(
12290                        "append_vlen_strings is only for variable-length string datasets".into(),
12291                    ))
12292                }
12293            }
12294        };
12295        ensure_vlen_charset(charset, strings)?;
12296
12297        // Every deterministic rejection must precede the heap write below:
12298        // a collection written for a batch the append then refuses (a
12299        // contiguous dataset, or a reopened dataset whose chunk index was
12300        // not reconstructed) is a 4096-byte orphan nothing references.
12301        let chunk_dims = self
12302            .dataset_chunk_dims(ds_index)
12303            .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?
12304            .to_vec();
12305        let dims = self.dataset_dims(ds_index).to_vec();
12306
12307        // Store the batch's strings as heap objects; a batch that fits an
12308        // earlier collection's free space shares its block.
12309        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12310        let placements = self.insert_vlen_objects(&items)?;
12311
12312        // Build raw vlen reference bytes
12313        let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12314        let mut raw = Vec::with_capacity(strings.len() * ref_size);
12315        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12316            let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12317            raw.extend_from_slice(&encode_vlen_reference(
12318                seq_len,
12319                gcol_addr,
12320                obj_idx as u32,
12321                &self.ctx,
12322            ));
12323        }
12324
12325        let n_new_frames = strings.len();
12326        let current_dim0 = dims[0] as usize;
12327        let chunk_dim0 = chunk_dims[0] as usize;
12328        let frame_bytes = ref_size;
12329
12330        // Merge the buffer with the new frames when it is the dataset's tail;
12331        // a buffer left mid-extent (the extent moved past it) keeps its
12332        // recorded place — flush it and start fresh at the current end.
12333        let taken = { self.ds(ds_index).lock().append.take() };
12334        let (base_dim0, buffered_frames, mut combined) = match taken {
12335            Some(b) if b.base + b.frames == current_dim0 as u64 => {
12336                (b.base as usize, b.frames as usize, b.bytes)
12337            }
12338            Some(b) => {
12339                self.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
12340                (current_dim0, 0, Vec::new())
12341            }
12342            None => (current_dim0, 0, Vec::new()),
12343        };
12344        combined.extend_from_slice(&raw);
12345
12346        let total_frames = buffered_frames + n_new_frames;
12347
12348        // Rows up to the last chunk boundary are written now; the tail that
12349        // does not complete a chunk goes back in the buffer for the next
12350        // append (or the flush at close). The boundary can precede
12351        // `base_dim0` — a reopened file's flushed partial chunk leaves the
12352        // base mid-chunk — in which case everything is tail.
12353        let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
12354        let write_frames = last_boundary.saturating_sub(base_dim0);
12355        let tail_frames = total_frames - write_frames;
12356        if write_frames > 0 {
12357            self.write_append_frames(
12358                ds_index,
12359                base_dim0 as u64,
12360                write_frames as u64,
12361                &combined[..write_frames * frame_bytes],
12362            )?;
12363        }
12364        if tail_frames > 0 {
12365            let ds = self.ds(ds_index);
12366            let mut m = ds.lock();
12367            m.append = Some(AppendBuffer {
12368                base: (base_dim0 + write_frames) as u64,
12369                frames: tail_frames as u64,
12370                bytes: combined[write_frames * frame_bytes..].to_vec(),
12371            });
12372        }
12373
12374        // Extend dims
12375        let logical_dim0 = base_dim0 + total_frames;
12376        let mut new_dims = dims;
12377        new_dims[0] = logical_dim0 as u64;
12378        self.extend_dataset_inner(ds_index, &new_dims)?;
12379
12380        Ok(())
12381    }
12382
12383    /// Replace elements `start .. start + strings.len()` of a 1-D
12384    /// variable-length string dataset, leaving its extent and every other
12385    /// element alone.
12386    ///
12387    /// The replacements go into the global heap and only the vlen
12388    /// references of the named elements are rewritten, so the cost is the
12389    /// new strings plus the chunks those references live in — not the column.
12390    /// The objects the old references pointed at are freed *before* the
12391    /// replacement is allocated, so repeated updates reuse space instead of
12392    /// growing the file — including across close/reopen cycles, where the
12393    /// in-memory free list starts empty and only this free-first order lets
12394    /// the session reuse the block it just released. This is what libhdf5
12395    /// does: `H5T__vlen_disk_write` deletes the reference it read into the
12396    /// conversion background buffer before storing the new one.
12397    ///
12398    /// Elements the append buffer still holds are flushed to their chunks
12399    /// first, so the whole range is on disk and one write path covers it.
12400    pub fn write_vlen_strings_slice(
12401        &self,
12402        ds_index: usize,
12403        start: u64,
12404        strings: &[&str],
12405    ) -> IoResult<()> {
12406        use crate::format::global_heap::{encode_vlen_reference, vlen_reference_size};
12407        use crate::format::messages::datatype::DatatypeMessage;
12408
12409        // An empty batch is a no-op: nothing to replace, nothing to free.
12410        if strings.is_empty() {
12411            return Ok(());
12412        }
12413
12414        // Whole-operation guard: the flush, the old-reference reads and the
12415        // slice write below must not interleave with a concurrent
12416        // same-dataset operation.
12417        let cell = self.ds(ds_index);
12418        let _op = cell.op.lock();
12419
12420        // Snapshot what the write needs, then drop the guard: `write_slice`
12421        // below re-locks the same slot.
12422        let (charset, dims, writable) = {
12423            let ds = self.ds(ds_index);
12424            let m = ds.lock();
12425            let charset = match m.datatype {
12426                DatatypeMessage::VarLenString { charset, .. } => charset,
12427                _ => {
12428                    return Err(crate::io::IoError::InvalidState(
12429                        "write_vlen_strings_slice is only for variable-length string datasets"
12430                            .into(),
12431                    ))
12432                }
12433            };
12434            let writable = if m.is_chunked() {
12435                Ok(())
12436            } else {
12437                match m.contiguous_target() {
12438                    Some(ContiguousTarget::Virtual) => Err(virtual_write_refused()),
12439                    Some(_) => Ok(()),
12440                    None => Err(crate::io::IoError::InvalidState(
12441                        "dataset has no data allocated".into(),
12442                    )),
12443                }
12444            };
12445            (charset, m.dataspace.dims.clone(), writable)
12446        };
12447
12448        // `write_slice_inner` rejects a dataset with neither chunk machinery
12449        // nor allocated data (a reopened dataset whose index was not
12450        // reconstructed), and refuses a virtual one outright — those
12451        // rejections must come before the heap write below, or every failed
12452        // call orphans a 4096-byte collection.
12453        writable?;
12454
12455        if dims.len() != 1 {
12456            return Err(crate::io::IoError::InvalidState(format!(
12457                "write_vlen_strings_slice is only for 1-dimension datasets, this one has {}",
12458                dims.len()
12459            )));
12460        }
12461        let end = start + strings.len() as u64;
12462        if end > dims[0] {
12463            return Err(crate::io::IoError::InvalidState(format!(
12464                "elements {start}..{end} are outside the dataset's {} elements",
12465                dims[0]
12466            )));
12467        }
12468        ensure_vlen_charset(charset, strings)?;
12469
12470        let ref_size = vlen_reference_size(&self.ctx);
12471
12472        // Elements the append buffer holds are not in the chunks yet: hand
12473        // them to the chunks first so the whole range is on disk and the one
12474        // write path below covers it.
12475        self.flush_append_buffer_if_intersecting(ds_index, start, end)?;
12476
12477        // The on-disk references about to be overwritten, read before anything
12478        // moves. libhdf5 reads the same bytes into the conversion background
12479        // buffer (`H5D__scatgath_write` gathers the file's current elements
12480        // when `need_bkg` is set) and hands them to `H5T__vlen_disk_write`,
12481        // which deletes them before storing the new reference.
12482        let superseded = self.current_element_bytes(ds_index, start, end - start, ref_size)?;
12483
12484        // Free the superseded objects *before* allocating the replacement,
12485        // the order `H5T__vlen_disk_write` uses. The freed block satisfies
12486        // the allocation below within this same session, so a reopen-and-
12487        // replace loop keeps the file flat — no persisted free-space
12488        // information exists to carry it across sessions (issue #10). The
12489        // cost, shared with libhdf5: a failure between here and the ref
12490        // write below leaves the dataset's old references dangling.
12491        self.release_vlen_references(&superseded)?;
12492
12493        // The insert comes after the release above so the space the release
12494        // recovered — a freed block, or in-collection bytes the release just
12495        // listed in `cwfs` — can satisfy this batch.
12496        let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12497        let placements = self.insert_vlen_objects(&items)?;
12498
12499        let mut refs = Vec::with_capacity(strings.len() * ref_size);
12500        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12501            refs.extend_from_slice(&encode_vlen_reference(
12502                crate::format::global_heap::vlen_seq_len(strings[i].len())?,
12503                gcol_addr,
12504                obj_idx as u32,
12505                &self.ctx,
12506            ));
12507        }
12508
12509        self.write_slice_inner(ds_index, &[start], &[strings.len() as u64], &refs)?;
12510
12511        Ok(())
12512    }
12513
12514    /// The bytes elements `start .. start + count` of a 1-D dataset currently
12515    /// hold, whichever layout stores them.
12516    ///
12517    /// Elements no write has reached yet read as zeros — for a vlen dataset
12518    /// that is the nil reference, which names no heap object.
12519    fn current_element_bytes(
12520        &self,
12521        ds_index: usize,
12522        start: u64,
12523        count: u64,
12524        element_size: usize,
12525    ) -> IoResult<Vec<u8>> {
12526        let mut out = vec![0u8; count as usize * element_size];
12527        if count == 0 {
12528            return Ok(out);
12529        }
12530
12531        let (is_chunked, data_addr) = {
12532            let ds = self.ds(ds_index);
12533            let m = ds.lock();
12534            (m.is_chunked(), m.data_addr)
12535        };
12536
12537        if !is_chunked {
12538            if data_addr != UNDEF_ADDR {
12539                // `read_at_most`, not `read_at`: a contiguous dataset's block is
12540                // reserved when it is created, so the file can still be shorter
12541                // than the block until something writes it. What is missing has
12542                // never been written, which is the zeros above.
12543                let at = data_addr + start * element_size as u64;
12544                let got = self.handle.read_at_most(at, out.len())?;
12545                out[..got.len()].copy_from_slice(&got);
12546            }
12547            return Ok(out);
12548        }
12549
12550        let geo = self.chunk_geometry(ds_index)?;
12551        let per_chunk = geo.chunk_dims[0];
12552        // Only a corrupt or crafted file declares a zero-length chunk
12553        // dimension; the divisions below must reject it the way
12554        // `write_slice` does, not panic.
12555        if per_chunk == 0 {
12556            return Err(crate::io::IoError::InvalidState(
12557                "chunk shape has a zero-length dimension".into(),
12558            ));
12559        }
12560        let end = start + count;
12561        for c in (start / per_chunk)..=((end - 1) / per_chunk) {
12562            let origin = c * per_chunk;
12563            let lo = start.max(origin);
12564            let hi = end.min(origin + per_chunk);
12565            // A chunk with no block yet leaves this span as the zeros above.
12566            let Some(chunk) = self.read_chunk_at_coords(ds_index, &[c])? else {
12567                continue;
12568            };
12569            let src = ((lo - origin) as usize) * element_size;
12570            let dst = ((lo - start) as usize) * element_size;
12571            let len = ((hi - lo) as usize) * element_size;
12572            if src + len > chunk.len() {
12573                return Err(crate::io::IoError::InvalidState(format!(
12574                    "chunk {c} is {} bytes, too short for elements {lo}..{hi}",
12575                    chunk.len()
12576                )));
12577            }
12578            out[dst..dst + len].copy_from_slice(&chunk[src..src + len]);
12579        }
12580        Ok(out)
12581    }
12582
12583    /// Free the global heap objects `refs` names, so replacing a vlen element
12584    /// does not strand what it used to point at.
12585    ///
12586    /// Callers pass refs only for *top-level* vlen datatypes (the
12587    /// `collect_refs` / `is_vlen` decisions at the prune, delete and
12588    /// attribute-release sites all match `VarLenString`/`VarLenSequence`).
12589    /// A compound datatype with vlen members — writable only by a foreign
12590    /// library, never by this crate — keeps its members' heap objects when
12591    /// its storage is pruned, deleted or replaced.
12592    ///
12593    /// This is libhdf5's `H5HG_remove` reached through `H5T__vlen_disk_delete`:
12594    /// the object leaves its collection, the collection is rewritten at its
12595    /// existing size with the recovered bytes given to the free-space marker,
12596    /// and a collection that ends up empty returns its block to the allocator.
12597    /// A rewritten collection's recovered space is listed in `cwfs` for
12598    /// [`insert_vlen_objects`](Self::insert_vlen_objects) to pack into; a
12599    /// freed block leaves the list.
12600    /// A nil reference (address 0 or `UNDEF_ADDR`) names no object. The
12601    /// address decides, not the sequence length: this crate's writers store
12602    /// even the empty string as a real heap object, so a zero-length reference
12603    /// with a defined address still holds one that must be released. libhdf5
12604    /// diverges here against itself — `H5T__vlen_disk_delete` returns before
12605    /// `H5HG_remove` when the sequence length is zero, yet its write path
12606    /// (`H5VL__native_blob_put`) inserts a heap object even for an empty
12607    /// sequence, stranding it forever. The address rule frees those objects.
12608    ///
12609    /// Heap objects carry no reference count on this path, matching libhdf5:
12610    /// its vlen code never calls `H5HG_link` (only the virtual-dataset layer
12611    /// does). Releasing the same reference twice is absorbed by the
12612    /// missing-index check below, but a crafted file in which two elements
12613    /// share one heap object would lose it for the survivor when either is
12614    /// replaced — the same exposure the file has under libhdf5. This crate's
12615    /// writers never share: each element write inserts its own object.
12616    ///
12617    /// Under SWMR nothing is freed and no collection is rewritten: a reader may
12618    /// be following those references, the same reason `place_chunk` keeps a
12619    /// relocated chunk's old block.
12620    fn release_vlen_references(&self, refs: &[u8]) -> IoResult<()> {
12621        use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
12622
12623        let ref_size = vlen_reference_size(&self.ctx);
12624        if ref_size == 0 || refs.len() < ref_size {
12625            return Ok(());
12626        }
12627
12628        // Group by collection so one holding several replaced objects is read,
12629        // rewritten and judged empty exactly once.
12630        let mut per_collection: std::collections::BTreeMap<u64, Vec<u16>> = Default::default();
12631        for r in refs.chunks_exact(ref_size) {
12632            let (_seq_len, addr, obj_idx) = decode_vlen_reference(r, &self.ctx)?;
12633            if addr == 0 || addr == UNDEF_ADDR {
12634                continue;
12635            }
12636            let Ok(idx) = u16::try_from(obj_idx) else {
12637                return Err(crate::io::IoError::InvalidState(format!(
12638                    "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
12639                )));
12640            };
12641            per_collection.entry(addr).or_default().push(idx);
12642        }
12643        self.remove_heap_objects(per_collection)
12644    }
12645
12646    /// Remove global heap objects — `H5HG_remove` — given the object indices
12647    /// grouped by the collection they live in.
12648    ///
12649    /// The single owner of heap-object removal: the vlen release path above
12650    /// reaches it with the objects a replaced element used to name, and
12651    /// [`release_dataset_storage`](Self::release_dataset_storage) with the
12652    /// one mapping-list object a deleted virtual dataset owned, which is what
12653    /// `H5D__virtual_delete` frees the same way.
12654    fn remove_heap_objects(
12655        &self,
12656        per_collection: std::collections::BTreeMap<u64, Vec<u16>>,
12657    ) -> IoResult<()> {
12658        use crate::format::global_heap::GlobalHeapCollection;
12659
12660        if self.swmr_active {
12661            return Ok(());
12662        }
12663
12664        // The `cwfs` lock is held across the sweep: it serializes these
12665        // collection-block rewrites (and frees) against
12666        // `insert_vlen_objects`, which may be packing new objects into the
12667        // same blocks.
12668        let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
12669        let mut cwfs = self.cwfs.lock();
12670        for (addr, indices) in per_collection {
12671            // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
12672            // exactly that, so one read usually covers the whole image; only
12673            // an oversized collection needs a second read at its declared size.
12674            let mut image = self.handle.read_at_most(addr, 4096)?;
12675            let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
12676            if declared > image.len() {
12677                image = self.handle.read_at(addr, declared)?;
12678            }
12679            let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
12680            let mut removed_any = false;
12681            for idx in indices {
12682                removed_any |= gcol.remove_object(idx);
12683            }
12684            // Every index already gone (a stale or duplicate reference):
12685            // leave the image alone. Rewriting is not just wasted I/O — a
12686            // 100%-full collection written by libhdf5 has no free-space
12687            // marker, so re-encoding it at its declared size cannot fit one
12688            // and the whole element update would fail.
12689            if !removed_any {
12690                continue;
12691            }
12692            if gcol.is_empty() {
12693                self.allocator
12694                    .free(addr, declared as u64, FreeSpaceClass::RawData);
12695                // The block is gone; a lingering entry would let an insert
12696                // pack into space the allocator can hand to anything.
12697                cwfs.retain(|e| e.addr != addr);
12698            } else {
12699                let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
12700                self.handle.write_at(addr, &rewritten)?;
12701                // The recovered bytes are packable now — list them, the way
12702                // libhdf5's `H5HG_remove` adds the heap to `cwfs`.
12703                if let Some(free) = gcol.free_space_at(&self.ctx, declared) {
12704                    if free >= 2 * objhdr {
12705                        cwfs_note(&mut cwfs, addr, declared, free);
12706                    }
12707                }
12708            }
12709        }
12710        Ok(())
12711    }
12712
12713    /// Add an attribute to a dataset.
12714    ///
12715    /// The attribute will be written as a message in the dataset's object
12716    /// header when the file is finalized.
12717    pub fn add_dataset_attribute(&self, ds_index: usize, attr: AttributeMessage) -> IoResult<()> {
12718        self.set_attribute(AttrTarget::Dataset(ds_index), attr)
12719    }
12720
12721    /// Build a variable-length UTF-8 string attribute message.
12722    ///
12723    /// The string is stored as one object in a global heap collection and the
12724    /// returned [`AttributeMessage`] carries the vlen reference as its data,
12725    /// with a vlen-string datatype and scalar dataspace. h5py reads the value
12726    /// back as a Python `str` (not `bytes`).
12727    ///
12728    /// This is the single owner of vlen-string-attribute construction: every
12729    /// public string-attribute setter (dataset, group, root, and the SWMR
12730    /// equivalents) routes through it, so a `VarLenUnicode` /
12731    /// `set_attr_string` value is always stored as a true variable-length
12732    /// string rather than the fixed-length string it used to be.
12733    ///
12734    /// The string's heap object is placed by
12735    /// [`insert_vlen_objects`](Self::insert_vlen_objects), so consecutive
12736    /// attributes pack into a shared collection instead of each paying the
12737    /// 4096-byte `H5HG_MINALLOC` minimum for a block that holds one string.
12738    fn vlen_string_attribute(&self, name: &str, value: &str) -> IoResult<AttributeMessage> {
12739        use crate::format::global_heap::encode_vlen_reference;
12740        use crate::format::messages::dataspace::DataspaceMessage;
12741        use crate::format::messages::datatype::DatatypeMessage;
12742
12743        let (gcol_addr, obj_idx) = self.insert_vlen_objects(&[value.as_bytes()])?[0];
12744        let seq_len = crate::format::global_heap::vlen_seq_len(value.len())?;
12745        let data = encode_vlen_reference(seq_len, gcol_addr, obj_idx as u32, &self.ctx);
12746        Ok(AttributeMessage {
12747            name: name.to_string(),
12748            datatype: DatatypeMessage::vlen_string_utf8(),
12749            dataspace: DataspaceMessage::scalar(),
12750            data,
12751        })
12752    }
12753
12754    /// Build a variable-length UTF-8 string **array** attribute message.
12755    ///
12756    /// The N-dimensional counterpart of
12757    /// [`vlen_string_attribute`](Self::vlen_string_attribute): every element
12758    /// string is stored as one object in a single global heap collection, and
12759    /// the attribute data is the row-major concatenation of one vlen reference
12760    /// per element. The datatype is the same vlen-string datatype; the dataspace
12761    /// is the simple dataspace described by `shape` (an empty `shape` is a
12762    /// scalar). h5py reads the value back as a numpy array of Python `str` with
12763    /// that shape.
12764    ///
12765    /// The caller owns the invariant that `values.len()` equals the product of
12766    /// `shape` (the public setters validate it before calling). The element
12767    /// objects are placed by
12768    /// [`insert_vlen_objects`](Self::insert_vlen_objects) — a zero-element
12769    /// array allocates nothing, and each reference carries its element's
12770    /// own collection address.
12771    fn vlen_string_array_attribute(
12772        &self,
12773        name: &str,
12774        values: &[&str],
12775        shape: &[u64],
12776    ) -> IoResult<AttributeMessage> {
12777        use crate::format::global_heap::encode_vlen_reference;
12778        use crate::format::messages::dataspace::DataspaceMessage;
12779        use crate::format::messages::datatype::DatatypeMessage;
12780
12781        debug_assert_eq!(
12782            values.len() as u64,
12783            shape.iter().product::<u64>(),
12784            "vlen_string_array_attribute values.len() must equal product(shape)"
12785        );
12786
12787        let items: Vec<&[u8]> = values.iter().map(|v| v.as_bytes()).collect();
12788        let placements = self.insert_vlen_objects(&items)?;
12789
12790        let mut data = Vec::with_capacity(values.len() * 16);
12791        for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12792            data.extend_from_slice(&encode_vlen_reference(
12793                crate::format::global_heap::vlen_seq_len(values[i].len())?,
12794                gcol_addr,
12795                obj_idx as u32,
12796                &self.ctx,
12797            ));
12798        }
12799        Ok(AttributeMessage {
12800            name: name.to_string(),
12801            datatype: DatatypeMessage::vlen_string_utf8(),
12802            dataspace: DataspaceMessage::simple(shape),
12803            data,
12804        })
12805    }
12806
12807    /// Set a user-defined fill value for a dataset.
12808    ///
12809    /// `bytes` must be exactly one element wide (matching the dataset's
12810    /// datatype). The value is emitted as a `fill_defined = 2` fill-value
12811    /// message in the dataset object header when the file is finalized.
12812    ///
12813    /// IMPORTANT: for a *contiguous* dataset this also immediately writes
12814    /// the tiled fill value across the whole data block, so it must be
12815    /// called BEFORE any `write_dataset_raw` / `write_slice` — otherwise the
12816    /// fill write clobbers data already written. (The high-level builder
12817    /// always calls this right after creating the dataset.)
12818    pub fn set_dataset_fill_value(&self, ds_index: usize, bytes: Vec<u8>) -> IoResult<()> {
12819        let count = self.dataset_count();
12820        if ds_index >= count {
12821            return Err(crate::io::IoError::InvalidState(format!(
12822                "dataset index {} out of range",
12823                ds_index
12824            )));
12825        }
12826        let ds_ref = self.ds(ds_index);
12827        let mut ds = ds_ref.lock();
12828        let es = ds.datatype.element_size() as usize;
12829        if bytes.len() != es {
12830            return Err(crate::io::IoError::InvalidState(format!(
12831                "fill value is {} bytes but dataset element size is {}",
12832                bytes.len(),
12833                es
12834            )));
12835        }
12836        // For a dataset with no per-chunk fill path the fill-value message
12837        // only declares fill-on-allocation — tile the fill value across the
12838        // storage itself now, so unwritten elements read back as the fill
12839        // value. Which storage that is depends on the layout: a compact
12840        // dataset's is the image inside its layout message, a contiguous
12841        // one's is its data block. (The high-level builder calls this
12842        // immediately after create, before any data is written; a subsequent
12843        // write_raw/write_slice overwrites its region.)
12844        // An implicitly indexed dataset is filled here too, and for the same
12845        // reason: that index has no per-chunk fill path because it has no
12846        // per-chunk anything — its whole chunk grid is one run of space,
12847        // allocated and filled at create like a contiguous block. So the test
12848        // is not "is it chunked" but "does something else fill its chunks".
12849        let fills_per_chunk = ds
12850            .chunk_index_kind()
12851            .is_some_and(|k| k != ChunkIndexKind::Implicit);
12852        // `H5D_FILL_TIME_NEVER` means exactly this: the library never writes
12853        // the fill value into allocated storage. Call `set_dataset_fill_time`
12854        // before this method to have it observed here — the storage this
12855        // would otherwise tile keeps whatever zero bytes its allocation
12856        // already gave it.
12857        if !fills_per_chunk && ds.fill_time != FILL_TIME_NEVER {
12858            if let Some(len) = ds.compact.as_ref().map(Vec::len) {
12859                ds.compact = Some(crate::format::messages::fill_value::tiled_fill(
12860                    len,
12861                    Some(&bytes),
12862                ));
12863            } else {
12864                // An implicit index's chunk grid is filled as one run, the
12865                // same way a contiguous block is, and storage this file did
12866                // not allocate is not filled at all; `allocated_storage_run`
12867                // is where both of those are decided.
12868                let run = ds.allocated_storage_run();
12869                if let Some((target, data_size)) = run.filter(|&(_, size)| size > 0) {
12870                    let filled = crate::format::messages::fill_value::tiled_fill(
12871                        data_size as usize,
12872                        Some(&bytes),
12873                    );
12874                    self.write_contiguous_bytes(&target, 0, &filled)?;
12875                }
12876            }
12877        }
12878
12879        ds.fill_value = Some(bytes);
12880        ds.header_dirty = true;
12881        Ok(())
12882    }
12883
12884    /// Set when the fill value is written into allocated storage —
12885    /// `H5Pset_fill_time`. `time` is one of [`FILL_TIME_ALLOC`],
12886    /// [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]; anything else is rejected
12887    /// the way `H5Pset_fill_time` rejects an out-of-range `H5D_fill_time_t`.
12888    ///
12889    /// Call this before [`set_dataset_fill_value`](Self::set_dataset_fill_value)
12890    /// so that a `FILL_TIME_NEVER` policy is in place before that call
12891    /// decides whether to eager-tile the value into storage. (The
12892    /// high-level builder always calls it first.)
12893    pub fn set_dataset_fill_time(&self, ds_index: usize, time: u8) -> IoResult<()> {
12894        if !matches!(time, FILL_TIME_ALLOC | FILL_TIME_NEVER | FILL_TIME_IFSET) {
12895            return Err(crate::io::IoError::InvalidState(format!(
12896                "invalid fill time {time}; must be {FILL_TIME_ALLOC} (alloc), \
12897                 {FILL_TIME_NEVER} (never) or {FILL_TIME_IFSET} (if-set)"
12898            )));
12899        }
12900        let count = self.dataset_count();
12901        if ds_index >= count {
12902            return Err(crate::io::IoError::InvalidState(format!(
12903                "dataset index {} out of range",
12904                ds_index
12905            )));
12906        }
12907        let ds_ref = self.ds(ds_index);
12908        let mut ds = ds_ref.lock();
12909        ds.fill_time = time;
12910        ds.header_dirty = true;
12911        Ok(())
12912    }
12913
12914    /// Allocate a `chunk_bytes`-sized buffer pre-filled with dataset
12915    /// `ds_index`'s fill value (tiled one element wide), or zeros when no
12916    /// user-defined fill value exists.
12917    ///
12918    /// Every partial chunk the writer emits must be built on top of a
12919    /// buffer from this method, so that the unwritten element region of an
12920    /// allocated chunk reads back as the fill value rather than zero.
12921    ///
12922    /// Unconditional: a shrink's straddler refill
12923    /// (`refill_chunk_beyond_extent`) calls this to repair data about to
12924    /// become reachable again, which libhdf5's `H5D__chunk_prune_fill` does
12925    /// regardless of the fill-time policy. [`new_write_chunk_buffer`](Self::new_write_chunk_buffer)
12926    /// is the gated counterpart for a chunk touched for the first time
12927    /// during a write, where the policy does apply.
12928    pub(crate) fn new_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
12929        let ds = self.ds(ds_index);
12930        let m = ds.lock();
12931        let fv = m.fill_value.as_deref();
12932        crate::format::messages::fill_value::tiled_fill(chunk_bytes, fv)
12933    }
12934
12935    /// The buffer a chunk gets the first time a write touches it — this
12936    /// dataset's allocation-time fill gate. `H5D__chunk_lock`'s cache-miss
12937    /// path (H5Dchunk.c:4894) fills such a buffer only for `ALLOC`, or for
12938    /// `IFSET` with a fill value defined; `NEVER` leaves it as the zeros a
12939    /// fresh buffer already has. Everything else about the buffer is
12940    /// [`new_chunk_buffer`](Self::new_chunk_buffer)'s.
12941    fn new_write_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
12942        let never = {
12943            let ds = self.ds(ds_index);
12944            let m = ds.lock();
12945            m.fill_time == FILL_TIME_NEVER
12946        };
12947        if never {
12948            vec![0u8; chunk_bytes]
12949        } else {
12950            self.new_chunk_buffer(ds_index, chunk_bytes)
12951        }
12952    }
12953
12954    /// Write `n_frames` whole frames whose first row is `base_frame`, for
12955    /// whichever chunk index the dataset uses and whatever its chunk shape.
12956    ///
12957    /// The single owner of an append's chunk writes. The frames are one
12958    /// hyperslab — rows `base_frame .. base_frame + n_frames` over the full
12959    /// row shape — so the write goes through
12960    /// [`write_slice_chunked`](Self::write_slice_chunked), the same engine
12961    /// `write_slice` uses: a chunk the span covers completely is written
12962    /// straight through, a partial one is read-modify-write on top of what
12963    /// is stored (or the fill value), and a chunk row narrower or wider
12964    /// than the frame row is scattered at the chunk stride. The previous
12965    /// owner required the extensible-array index and packed rows at the
12966    /// frame stride, so appends to a fixed-array or v2 B-tree dataset
12967    /// failed at close and lost the buffered rows.
12968    ///
12969    /// The caller holds the dataset's op lock or the writer exclusively.
12970    pub(crate) fn write_append_frames(
12971        &self,
12972        ds_index: usize,
12973        base_frame: u64,
12974        n_frames: u64,
12975        frames: &[u8],
12976    ) -> IoResult<()> {
12977        if n_frames == 0 {
12978            return Ok(());
12979        }
12980        let geo = self.chunk_geometry(ds_index)?;
12981        let mut starts = vec![0u64; geo.dims.len()];
12982        starts[0] = base_frame;
12983        let mut counts = geo.dims.clone();
12984        counts[0] = n_frames;
12985        let expected = counts.iter().product::<u64>() * geo.element_size;
12986        if frames.len() as u64 != expected {
12987            return Err(crate::io::IoError::InvalidState(format!(
12988                "{n_frames} frames at rows {base_frame}.. need {expected} bytes, got {}",
12989                frames.len()
12990            )));
12991        }
12992        self.write_slice_chunked(ds_index, &starts, &counts, frames)
12993    }
12994
12995    /// Write the dataset's append buffer (if any) into its chunks and clear
12996    /// it. The single owner of the buffer-to-chunks transition: the flush at
12997    /// close, an append meeting a non-contiguous buffer, and any operation
12998    /// about to write rows the buffer holds all come through here.
12999    ///
13000    /// The caller holds the dataset's op lock or the writer exclusively —
13001    /// the take and the frame writes are separate acquisitions.
13002    pub(crate) fn flush_append_buffer(&self, ds_index: usize) -> IoResult<()> {
13003        let taken = { self.ds(ds_index).lock().append.take() };
13004        match taken {
13005            Some(b) => self.write_append_frames(ds_index, b.base, b.frames, &b.bytes),
13006            None => Ok(()),
13007        }
13008    }
13009
13010    /// Flush the append buffer when rows `start_row .. end_row` intersect
13011    /// the buffered range — those rows' current content is the buffer, and
13012    /// writing them on disk while the buffer still holds them would be
13013    /// undone by the flush at close.
13014    ///
13015    /// The caller holds the dataset's op lock or the writer exclusively.
13016    pub(crate) fn flush_append_buffer_if_intersecting(
13017        &self,
13018        ds_index: usize,
13019        start_row: u64,
13020        end_row: u64,
13021    ) -> IoResult<()> {
13022        let intersects = {
13023            let ds = self.ds(ds_index);
13024            let m = ds.lock();
13025            m.append
13026                .as_ref()
13027                .is_some_and(|b| start_row < b.base + b.frames && end_row > b.base)
13028        };
13029        if intersects {
13030            self.flush_append_buffer(ds_index)
13031        } else {
13032            Ok(())
13033        }
13034    }
13035
13036    /// Read an already-written chunk's *decompressed* bytes when the chunk
13037    /// is allocated and resolvable from the in-memory extensible-array
13038    /// index. Handles index-block and data-block chunks, filtered and
13039    /// unfiltered.
13040    ///
13041    /// Returns `Ok(None)` only when the chunk has never been written
13042    /// (address `UNDEF`) or the index genuinely does not reach it, which for
13043    /// a read-modify-write means the chunk's content is the fill value.
13044    pub(crate) fn read_chunk_if_present(
13045        &self,
13046        ds_index: usize,
13047        chunk_idx: u64,
13048    ) -> IoResult<Option<Vec<u8>>> {
13049        // Phase 1: resolve the chunk's location from the in-memory index.
13050        // Hold the slot guard through Phase 1: `chunked` borrows it, while the
13051        // `self.handle`/`self.ctx` reads below touch disjoint fields.
13052        let ds = self.ds(ds_index);
13053        let m = ds.lock();
13054        let element_size = m.datatype.element_size() as u64;
13055        let pipeline = m.filter_pipeline.clone();
13056        let Some(chunked) = m.chunked.as_ref() else {
13057            return Ok(None);
13058        };
13059        let chunk_bytes = chunked.chunk_dims.iter().product::<u64>() * element_size;
13060        let max_nelmts_bits = chunked.earray_params.max_nelmts_bits;
13061        let chunk_size_len = chunked.chunk_size_len;
13062        let is_filtered = chunked.filt_iblk.is_some();
13063
13064        // The chunk entry is either read straight from an index block, or
13065        // located via a data block that must itself be read from disk.
13066        enum Loc {
13067            Direct(u64, u64, u32),
13068            DataBlock {
13069                dblk_addr: u64,
13070                offset: usize,
13071                nelmts: usize,
13072            },
13073        }
13074
13075        // Resolve the chunk's location with the libhdf5-compatible EA
13076        // geometry (super-block-grouped data blocks), matching `record_ea_chunk`.
13077        let ea_loc = {
13078            let p = &chunked.earray_params;
13079            EaGeometry::new(
13080                p.idx_blk_elmts,
13081                p.data_blk_min_elmts,
13082                p.sup_blk_min_data_ptrs,
13083                p.max_nelmts_bits,
13084                p.max_dblk_page_nelmts_bits,
13085            )?
13086            .locate(chunk_idx)?
13087        };
13088        let loc = match ea_loc {
13089            EaLoc::Index { elem } => {
13090                if is_filtered {
13091                    let e = &chunked.filt_iblk.as_ref().unwrap().elements[elem];
13092                    Loc::Direct(e.addr, e.nbytes, e.filter_mask)
13093                } else {
13094                    Loc::Direct(chunked.ea_iblk.elements[elem], chunk_bytes, 0)
13095                }
13096            }
13097            EaLoc::Dblk(l) => {
13098                if l.paged {
13099                    return Err(crate::io::IoError::InvalidState(format!(
13100                        "chunk index {} lives in a paged extensible-array data \
13101                         block, which is not yet supported for read-modify-write",
13102                        chunk_idx
13103                    )));
13104                }
13105                let dblk_addr = match l.path {
13106                    EaDblkPath::Direct { idx } => {
13107                        if is_filtered {
13108                            chunked.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
13109                        } else {
13110                            chunked.ea_iblk.dblk_addrs[idx]
13111                        }
13112                    }
13113                    EaDblkPath::ViaSblk {
13114                        sblk_off,
13115                        local_dblk,
13116                        ndblks_in_sblk,
13117                        ..
13118                    } => {
13119                        let sblk_addr = if is_filtered {
13120                            chunked.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
13121                        } else {
13122                            chunked.ea_iblk.sblk_addrs[sblk_off]
13123                        };
13124                        if sblk_addr == UNDEF_ADDR {
13125                            return Ok(None);
13126                        }
13127                        let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
13128                        let sb = ExtensibleArraySuperBlock::decode(
13129                            &sb_buf,
13130                            &self.ctx,
13131                            max_nelmts_bits,
13132                            ndblks_in_sblk,
13133                            0,
13134                        )?;
13135                        sb.dblk_addrs[local_dblk]
13136                    }
13137                };
13138                if dblk_addr == UNDEF_ADDR {
13139                    return Ok(None);
13140                }
13141                Loc::DataBlock {
13142                    dblk_addr,
13143                    offset: l.offset_in_dblk as usize,
13144                    nelmts: l.dblk_nelmts as usize,
13145                }
13146            }
13147        };
13148
13149        // Phase 2: resolve through the data block (if needed) and read. The
13150        // mask is the chunk's filter mask (0 for unfiltered), so a chunk
13151        // written via a direct chunk write with a skipped filter is reversed
13152        // correctly during read-modify-write.
13153        let (addr, nbytes, mask) = match loc {
13154            Loc::Direct(a, n, m) => (a, n, m),
13155            Loc::DataBlock {
13156                dblk_addr,
13157                offset,
13158                nelmts,
13159            } => {
13160                let buf = self.handle.read_at_most(dblk_addr, 65536)?;
13161                if is_filtered {
13162                    let dblk = FilteredDataBlock::decode(
13163                        &buf,
13164                        &self.ctx,
13165                        max_nelmts_bits,
13166                        nelmts,
13167                        chunk_size_len,
13168                    )?;
13169                    let e = &dblk.elements[offset];
13170                    (e.addr, e.nbytes, e.filter_mask)
13171                } else {
13172                    let dblk =
13173                        ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, nelmts)?;
13174                    (dblk.elements[offset], chunk_bytes, 0)
13175                }
13176            }
13177        };
13178        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13179    }
13180
13181    /// Read one stored chunk block and undo its filters.
13182    ///
13183    /// `nbytes` is the *stored* length and `mask` the chunk's filter mask, so
13184    /// a chunk written by a direct chunk write with a skipped filter is
13185    /// reversed correctly. `Ok(None)` means the chunk has no block yet — the
13186    /// single place that judgement is made, shared by every chunk index.
13187    fn read_chunk_block(
13188        &self,
13189        pipeline: Option<&FilterPipeline>,
13190        addr: u64,
13191        nbytes: u64,
13192        mask: u32,
13193    ) -> IoResult<Option<Vec<u8>>> {
13194        if addr == UNDEF_ADDR || nbytes == 0 {
13195            return Ok(None);
13196        }
13197        let raw = self.handle.read_at(addr, nbytes as usize)?;
13198        match pipeline {
13199            Some(pl) => Ok(Some(filter::reverse_filters_masked(pl, &raw, mask)?)),
13200            None => Ok(Some(raw)),
13201        }
13202    }
13203
13204    /// Read the *decompressed* bytes of the chunk at `chunk_coords`, whichever
13205    /// chunk index the dataset uses, or `Ok(None)` when that chunk has never
13206    /// been written.
13207    ///
13208    /// This is the read half of a partial-chunk read-modify-write: a hyperslab
13209    /// write that covers only part of a chunk must start from what is already
13210    /// there. Keeping one entry point for all three index types is what lets
13211    /// [`write_slice`](Self::write_slice) stay index-agnostic.
13212    pub(crate) fn read_chunk_at_coords(
13213        &self,
13214        ds_index: usize,
13215        chunk_coords: &[u64],
13216    ) -> IoResult<Option<Vec<u8>>> {
13217        let geo = self.chunk_geometry(ds_index)?;
13218        // Only the linearly-addressed indexes compute a slot; a v2 B-tree is
13219        // keyed by the coordinates themselves (and may hold unlimited inner
13220        // dimensions, which have no linear slot).
13221        match geo.kind {
13222            ChunkIndexKind::ExtensibleArray => {
13223                let linear = geo.linear_index(chunk_coords)?;
13224                self.read_chunk_if_present(ds_index, linear)
13225            }
13226            ChunkIndexKind::FixedArray => {
13227                let linear = geo.linear_index(chunk_coords)?;
13228                let ds = self.ds(ds_index);
13229                let m = ds.lock();
13230                let pipeline = m.filter_pipeline.clone();
13231                let fa = m.fixed_array.as_ref().unwrap();
13232                let lidx = linear as usize;
13233                let (addr, nbytes, mask) = if pipeline.is_some() {
13234                    match fa.fa_dblk.filtered_elements.get(lidx) {
13235                        Some(e) => (e.address, e.chunk_size, e.filter_mask),
13236                        None => return Ok(None),
13237                    }
13238                } else {
13239                    match fa.fa_dblk.elements.get(lidx) {
13240                        Some(&a) => (a, geo.chunk_bytes(), 0),
13241                        None => return Ok(None),
13242                    }
13243                };
13244                drop(m);
13245                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13246            }
13247            ChunkIndexKind::BtreeV2 => {
13248                let ds = self.ds(ds_index);
13249                let m = ds.lock();
13250                let pipeline = m.filter_pipeline.clone();
13251                let bt2 = m.btree_v2.as_ref().unwrap();
13252                // A filtered index records the stored size and mask per chunk;
13253                // an unfiltered one stores whole chunks, so their size is the
13254                // chunk shape and no filter ran.
13255                let found = if bt2.index.filtered {
13256                    bt2.index
13257                        .lookup_filtered(chunk_coords)
13258                        .map(|r| (r.chunk_address, r.chunk_size, r.filter_mask))
13259                } else {
13260                    bt2.index
13261                        .lookup(chunk_coords)
13262                        .map(|r| (r.chunk_address, geo.chunk_bytes(), 0))
13263                };
13264                drop(m);
13265                match found {
13266                    Some((addr, nbytes, mask)) => {
13267                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13268                    }
13269                    None => Ok(None),
13270                }
13271            }
13272            // Every chunk of an implicitly indexed dataset exists from the
13273            // moment the dataset does, so there is no "never written" answer
13274            // to give: an untouched chunk reads back as the fill value the
13275            // create wrote there.
13276            ChunkIndexKind::Implicit => {
13277                let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
13278                self.read_chunk_block(None, grid + offset, geo.chunk_bytes(), 0)
13279            }
13280            // A single-chunk dataset's one chunk is never written until its
13281            // first write (unless the dataset was early-allocated and
13282            // unfiltered, in which case create already gave it an address) —
13283            // unlike Implicit, `UNDEF_ADDR` here is a real "never written".
13284            ChunkIndexKind::SingleChunk => {
13285                let ds = self.ds(ds_index);
13286                let m = ds.lock();
13287                let pipeline = m.filter_pipeline.clone();
13288                let sc = m.single_chunk.as_ref().unwrap();
13289                if sc.data_addr == UNDEF_ADDR {
13290                    return Ok(None);
13291                }
13292                let (addr, nbytes, mask) = if pipeline.is_some() {
13293                    (sc.data_addr, sc.nbytes, sc.filter_mask)
13294                } else {
13295                    (sc.data_addr, geo.chunk_bytes(), 0)
13296                };
13297                drop(m);
13298                self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13299            }
13300            ChunkIndexKind::BtreeV1 => {
13301                let ds = self.ds(ds_index);
13302                let m = ds.lock();
13303                let pipeline = m.filter_pipeline.clone();
13304                let bt1 = m.btree_v1.as_ref().unwrap();
13305                let found = bt1
13306                    .position(chunk_coords)
13307                    .ok()
13308                    .map(|i| &bt1.records[i])
13309                    .map(|r| (r.address, r.nbytes as u64, r.filter_mask));
13310                drop(m);
13311                match found {
13312                    Some((addr, nbytes, mask)) => {
13313                        self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13314                    }
13315                    None => Ok(None),
13316                }
13317            }
13318        }
13319    }
13320
13321    /// The slot one chunk of an implicitly indexed dataset occupies: the
13322    /// address its whole chunk grid starts at, and the chunk's offset within
13323    /// that grid. `data_addr + linear_index * chunk_bytes` is the whole of
13324    /// that index (`H5D__none_idx_get_addr`, H5Dnone.c).
13325    ///
13326    /// The one place a chunk of such a dataset is placed — read and write both
13327    /// come through here, so the bounds check below covers both. The grid it
13328    /// names is [`DatasetInfo::implicit_grid`], which is why the write side
13329    /// can hand [`ContiguousTarget::Local`] to
13330    /// [`write_contiguous_bytes`](Self::write_contiguous_bytes) without asking
13331    /// anything: the external and virtual destinations that owner also knows
13332    /// about are unreachable from a chunked dataset.
13333    fn implicit_chunk_slot(
13334        &self,
13335        ds_index: usize,
13336        geo: &ChunkGeometry,
13337        chunk_coords: &[u64],
13338    ) -> IoResult<(u64, u64)> {
13339        let linear = geo.linear_index(chunk_coords)?;
13340        let ds = self.ds(ds_index);
13341        let m = ds.lock();
13342        let (grid, grid_size) = m.implicit_grid().ok_or_else(|| {
13343            crate::io::IoError::InvalidState("no implicitly indexed chunk grid".into())
13344        })?;
13345        let offset = linear.checked_mul(geo.chunk_bytes()).ok_or_else(|| {
13346            crate::io::IoError::InvalidState("implicit chunk offset overflows u64".into())
13347        })?;
13348        if offset + geo.chunk_bytes() > grid_size {
13349            return Err(crate::io::IoError::InvalidState(format!(
13350                "chunk {chunk_coords:?} lies outside the {grid_size} bytes of chunk space \
13351                 this implicitly indexed dataset was created with"
13352            )));
13353        }
13354        Ok((grid, offset))
13355    }
13356
13357    /// Write one whole chunk addressed by its grid coordinates, whichever
13358    /// chunk index the dataset uses. `data` is the chunk's unfiltered bytes;
13359    /// the dataset's filter pipeline (if any) runs here.
13360    ///
13361    /// The write half of the pair with
13362    /// [`read_chunk_at_coords`](Self::read_chunk_at_coords). Unlike the
13363    /// dataset-level `write_chunk_at`, this never grows the dataspace — a
13364    /// hyperslab write is bounded by the current extent by definition.
13365    ///
13366    /// The caller holds the dataset's op lock or the writer exclusively.
13367    pub(crate) fn write_chunk_at_coords(
13368        &self,
13369        ds_index: usize,
13370        chunk_coords: &[u64],
13371        data: &[u8],
13372    ) -> IoResult<()> {
13373        let geo = self.chunk_geometry(ds_index)?;
13374        match geo.kind {
13375            ChunkIndexKind::ExtensibleArray => {
13376                let linear = geo.linear_index(chunk_coords)?;
13377                self.write_chunk_inner(ds_index, linear, data)
13378            }
13379            ChunkIndexKind::FixedArray => {
13380                self.write_chunk_fixed_array_inner(ds_index, chunk_coords, data)
13381            }
13382            ChunkIndexKind::BtreeV2 => {
13383                self.write_chunk_btree_v2_inner(ds_index, chunk_coords, data)
13384            }
13385            ChunkIndexKind::Implicit => {
13386                self.write_chunk_implicit_inner(ds_index, chunk_coords, data)
13387            }
13388            ChunkIndexKind::SingleChunk => {
13389                self.write_chunk_single_chunk_inner(ds_index, chunk_coords, data)
13390            }
13391            ChunkIndexKind::BtreeV1 => {
13392                self.write_chunk_btree_v1_inner(ds_index, chunk_coords, data)
13393            }
13394        }
13395    }
13396
13397    /// Write one whole chunk to a dataset indexed by a version-1 B-tree.
13398    ///
13399    /// `chunk_coords` is the chunk's grid position. `data` is the chunk's
13400    /// unfiltered bytes; the dataset's filter pipeline runs here if it has
13401    /// one, and the key records the stored size and mask the way libhdf5's
13402    /// does (`H5D__btree_new_node`).
13403    ///
13404    /// The caller holds the dataset's op lock or the writer exclusively.
13405    pub(crate) fn write_chunk_btree_v1_inner(
13406        &self,
13407        ds_index: usize,
13408        chunk_coords: &[u64],
13409        data: &[u8],
13410    ) -> IoResult<()> {
13411        // Read what the write needs under a brief guard, then filter OUTSIDE
13412        // the lock, as every other index's write path does.
13413        let ds = self.ds(ds_index);
13414        let (chunk_bytes, pipeline) = {
13415            let m = ds.lock();
13416            let element_size = m.datatype.element_size() as u64;
13417            let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
13418                crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
13419            })?;
13420            (
13421                bt1.chunk_dims.iter().product::<u64>() * element_size,
13422                m.filter_pipeline.clone(),
13423            )
13424        };
13425        if data.len() as u64 != chunk_bytes {
13426            return Err(crate::io::IoError::InvalidState(format!(
13427                "chunk data size mismatch: expected {} bytes, got {}",
13428                chunk_bytes,
13429                data.len()
13430            )));
13431        }
13432
13433        let filtered;
13434        let stored = match pipeline {
13435            Some(ref pl) => {
13436                filtered = filter::apply_filters(pl, data)?;
13437                &filtered[..]
13438            }
13439            None => data,
13440        };
13441        self.record_btree_v1_chunk(ds_index, chunk_coords, stored, 0)
13442    }
13443
13444    /// Write a pre-filtered chunk verbatim to a version-1 B-tree dataset,
13445    /// recording the caller-supplied `filter_mask` — the classic-index half
13446    /// of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
13447    ///
13448    /// The caller holds the dataset's op lock or the writer exclusively.
13449    pub(crate) fn write_compressed_chunk_btree_v1_inner(
13450        &self,
13451        ds_index: usize,
13452        chunk_coords: &[u64],
13453        data: &[u8],
13454        filter_mask: u32,
13455    ) -> IoResult<()> {
13456        if self.ds(ds_index).lock().filter_pipeline.is_none() {
13457            return Err(crate::io::IoError::InvalidState(
13458                "write_chunk_raw requires a filtered dataset (an unfiltered chunk \
13459                 is stored at its full size, so there is nothing for a stored size \
13460                 or a filter mask to say)"
13461                    .into(),
13462            ));
13463        }
13464        self.record_btree_v1_chunk(ds_index, chunk_coords, data, filter_mask)
13465    }
13466
13467    /// Place a chunk's already-final bytes in the file and record them in the
13468    /// version-1 B-tree under the caller-supplied `filter_mask`.
13469    ///
13470    /// Shared by the two writes above, so both reach the index through one
13471    /// placement rule. The records are kept in key order here — the bulk load
13472    /// at flush walks them in that order and a lookup bisects them.
13473    fn record_btree_v1_chunk(
13474        &self,
13475        ds_index: usize,
13476        chunk_coords: &[u64],
13477        final_bytes: &[u8],
13478        filter_mask: u32,
13479    ) -> IoResult<()> {
13480        let stored_len = final_bytes.len() as u64;
13481        // The key's size field is 32 bits wide (`H5D_btree_key_t::nbytes`),
13482        // which is also libhdf5's limit on a chunk in this index.
13483        let Ok(nbytes) = u32::try_from(stored_len) else {
13484            return Err(crate::io::IoError::InvalidState(format!(
13485                "stored chunk size {stored_len} does not fit in the 32-bit size \
13486                 field of a version-1 B-tree chunk key"
13487            )));
13488        };
13489        let ds = self.ds(ds_index);
13490        let mut m = ds.lock();
13491        let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
13492            crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
13493        })?;
13494        if chunk_coords.len() != bt1.chunk_dims.len() {
13495            return Err(crate::io::IoError::InvalidState(format!(
13496                "chunk_coords has {} entries but the dataset has {} dimensions",
13497                chunk_coords.len(),
13498                bt1.chunk_dims.len()
13499            )));
13500        }
13501        // A coordinate past the maximum extent has no chunk to be: unlike the
13502        // array indexes there is no slot to run out of, so the bound is
13503        // checked here or not at all. An unlimited dimension has none.
13504        for (d, ((&c, &cd), &max)) in chunk_coords
13505            .iter()
13506            .zip(&bt1.chunk_dims)
13507            .zip(&bt1.max_dims)
13508            .enumerate()
13509        {
13510            if max != u64::MAX && c.saturating_mul(cd) >= max {
13511                return Err(crate::io::IoError::InvalidState(format!(
13512                    "chunk coordinate {c} in dimension {d} is outside the maximum \
13513                     extent {max}"
13514                )));
13515            }
13516        }
13517        let slot = bt1.position(chunk_coords);
13518        let old = slot.ok().map(|i| {
13519            let r = &bt1.records[i];
13520            (r.address, r.nbytes as u64)
13521        });
13522        // A rewrite whose stored size is unchanged stays where it is (always
13523        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
13524        let address = self.place_chunk(old, stored_len);
13525        self.handle.write_at(address, final_bytes)?;
13526
13527        let bt1 = m.btree_v1.as_mut().unwrap();
13528        let record = BtreeV1ChunkRecord {
13529            scaled: chunk_coords.to_vec(),
13530            address,
13531            nbytes,
13532            filter_mask,
13533        };
13534        match slot {
13535            Ok(i) => bt1.records[i] = record,
13536            Err(i) => bt1.records.insert(i, record),
13537        }
13538        bt1.chunks_written += 1;
13539        Ok(())
13540    }
13541
13542    /// Write one whole chunk of an implicitly indexed dataset into the slot
13543    /// its coordinates name. There is no index to record anything in — the
13544    /// slot is where it always was — so this is the write in full.
13545    ///
13546    /// The bytes go through [`write_contiguous_bytes`](Self::write_contiguous_bytes),
13547    /// the one owner of a raw-byte write, against the grid
13548    /// [`implicit_chunk_slot`](Self::implicit_chunk_slot) names.
13549    ///
13550    /// The caller holds the dataset's op lock or the writer exclusively.
13551    pub(crate) fn write_chunk_implicit_inner(
13552        &self,
13553        ds_index: usize,
13554        chunk_coords: &[u64],
13555        data: &[u8],
13556    ) -> IoResult<()> {
13557        let geo = self.chunk_geometry(ds_index)?;
13558        let chunk_bytes = geo.chunk_bytes();
13559        if data.len() as u64 != chunk_bytes {
13560            return Err(crate::io::IoError::InvalidState(format!(
13561                "chunk data size mismatch: expected {} bytes, got {}",
13562                chunk_bytes,
13563                data.len()
13564            )));
13565        }
13566        let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
13567        self.write_contiguous_bytes(&ContiguousTarget::Local(grid), offset, data)
13568    }
13569
13570    /// Snapshot the geometry needed to address a chunked dataset's grid.
13571    ///
13572    /// Taken under one brief slot guard so the callers below — which re-lock
13573    /// the slot through `write_chunk`/`read_chunk_*` — never hold it across
13574    /// compression or I/O.
13575    fn chunk_geometry(&self, ds_index: usize) -> IoResult<ChunkGeometry> {
13576        let ds = self.ds(ds_index);
13577        let m = ds.lock();
13578        let Some(kind) = m.chunk_index_kind() else {
13579            return Err(crate::io::IoError::InvalidState(
13580                "not a chunked dataset".into(),
13581            ));
13582        };
13583        let chunk_dims = match kind {
13584            ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
13585            ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
13586            ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
13587            ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
13588            ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
13589            ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
13590        };
13591        Ok(ChunkGeometry {
13592            kind,
13593            dims: m.dataspace.dims.clone(),
13594            max_dims: m.dataspace.max_dims.clone(),
13595            chunk_dims,
13596            element_size: m.datatype.element_size() as u64,
13597        })
13598    }
13599
13600    /// Index-grid slot of the chunk at grid `coords` (see
13601    /// [`crate::io::chunk_grid`]).
13602    pub(crate) fn chunk_slot(&self, ds_index: usize, coords: &[u64]) -> IoResult<u64> {
13603        self.chunk_geometry(ds_index)?.linear_index(coords)
13604    }
13605
13606    /// Grid coordinates of the chunk recorded under index-grid slot `linear`
13607    /// — the inverse of [`Self::chunk_slot`].
13608    pub(crate) fn chunk_coords_from_slot(
13609        &self,
13610        ds_index: usize,
13611        linear: u64,
13612    ) -> IoResult<Vec<u64>> {
13613        let geo = self.chunk_geometry(ds_index)?;
13614        crate::io::chunk_grid::coords_of(
13615            &geo.dims,
13616            geo.max_dims.as_deref(),
13617            &geo.chunk_dims,
13618            linear,
13619        )
13620    }
13621
13622    /// Define a chunked dataset indexed by a fixed array, fixed at its
13623    /// current shape (`max_dims == dims`). `chunk_dims` defines the chunk
13624    /// shape. Returns the dataset index.
13625    pub fn create_fixed_array_dataset(
13626        &self,
13627        name: &str,
13628        datatype: DatatypeMessage,
13629        dims: &[u64],
13630        chunk_dims: &[u64],
13631    ) -> IoResult<usize> {
13632        self.create_fixed_array_dataset_with_max(name, datatype, dims, dims, chunk_dims, None)
13633    }
13634
13635    /// Define a fixed-shape compressed chunked dataset indexed by a
13636    /// *filtered* Fixed Array (`max_dims == dims`).
13637    ///
13638    /// Like `create_fixed_array_dataset`, but the FA header carries the filtered
13639    /// client id and a `chunk_size_len`-wide compressed-size field per chunk
13640    /// (`FixedArrayFilteredChunkElement`), and the dataset gets a filter
13641    /// pipeline. Chunks written via `write_chunk_fixed_array` are compressed and
13642    /// their compressed size + filter mask are recorded in the data block.
13643    ///
13644    /// A convenience over [`create_fixed_array_dataset_with_max`]'s own
13645    /// pipeline argument; production dataset creation calls that directly,
13646    /// so this is kept as a direct entry point for this crate's own
13647    /// white-box tests.
13648    ///
13649    /// [`create_fixed_array_dataset_with_max`]: Self::create_fixed_array_dataset_with_max
13650    #[cfg(test)]
13651    pub fn create_fixed_array_dataset_with_pipeline(
13652        &self,
13653        name: &str,
13654        datatype: DatatypeMessage,
13655        dims: &[u64],
13656        chunk_dims: &[u64],
13657        pipeline: FilterPipeline,
13658    ) -> IoResult<usize> {
13659        self.create_fixed_array_dataset_with_max(
13660            name,
13661            datatype,
13662            dims,
13663            dims,
13664            chunk_dims,
13665            Some(pipeline),
13666        )
13667    }
13668
13669    /// Define a chunked dataset indexed by a fixed array, growable up to
13670    /// `max_dims` (every maximum finite — libhdf5 picks this index exactly
13671    /// when no dimension is unlimited).
13672    ///
13673    /// The array is sized for the chunk grid of the *maximum* extent, the
13674    /// libhdf5 rule (`H5D__farray_idx_create` uses `max_nchunks`), so the
13675    /// dataset can be extended to `max_dims` without re-indexing chunks.
13676    pub fn create_fixed_array_dataset_with_max(
13677        &self,
13678        name: &str,
13679        datatype: DatatypeMessage,
13680        dims: &[u64],
13681        max_dims: &[u64],
13682        chunk_dims: &[u64],
13683        pipeline: Option<FilterPipeline>,
13684    ) -> IoResult<usize> {
13685        let create = self.begin_create(name)?;
13686        let name = create.name.as_str();
13687        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
13688        if max_dims.contains(&u64::MAX) {
13689            return Err(crate::io::IoError::InvalidState(
13690                "a fixed-array index requires a fixed maximum shape (no unlimited dimension)"
13691                    .into(),
13692            ));
13693        }
13694        let mut num_chunks: u64 = 1;
13695        for g in crate::io::chunk_grid::index_grid(dims, Some(max_dims), chunk_dims)? {
13696            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
13697                crate::io::IoError::InvalidState("chunk count overflows u64".into())
13698            })?;
13699        }
13700
13701        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13702        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
13703
13704        // Create the FA header. For a filtered FA, chunk_size_len is sized
13705        // the same way the filtered Extensible Array path computes it:
13706        // derived from the uncompressed chunk byte count under layout v4,
13707        // the fixed `sizeof_size` under layout v5.
13708        let mut fa_header = if pipeline.is_some() {
13709            let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
13710            FixedArrayHeader::new_for_filtered_chunks(&self.ctx, num_chunks, chunk_size_len)
13711        } else {
13712            FixedArrayHeader::new_for_chunks(&self.ctx, num_chunks)
13713        };
13714        let hdr_encoded = fa_header.encode(&self.ctx);
13715        let fa_header_addr = self
13716            .allocator
13717            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
13718
13719        // Create the FA data block. libhdf5 switches to a paged layout once
13720        // num_elmts exceeds dblk_page_nelmts; both layouts allocate space
13721        // for `num_chunks` entries up front, but the paged layout also
13722        // reserves the page-init bitmap and a per-page checksum.
13723        let fa_dblk = if pipeline.is_some() {
13724            FixedArrayDataBlock::new_filtered(fa_header_addr, num_chunks as usize)
13725        } else {
13726            FixedArrayDataBlock::new_unfiltered(fa_header_addr, num_chunks as usize)
13727        };
13728        let dblk_size = fixed_array_dblk_disk_size(&self.ctx, &fa_header);
13729        let fa_dblk_addr = self.allocator.allocate(dblk_size, FreeSpaceClass::Metadata);
13730
13731        // Update header with data block address
13732        fa_header.data_blk_addr = fa_dblk_addr;
13733
13734        // Write both. The data block content is finalized in `flush_dataset`
13735        // once all chunk addresses are known; here we just reserve space and
13736        // write the header so the file is structurally consistent.
13737        let hdr_encoded = fa_header.encode(&self.ctx);
13738        self.handle.write_at(fa_header_addr, &hdr_encoded)?;
13739        let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa_header, &fa_dblk);
13740        debug_assert_eq!(dblk_encoded.len() as u64, dblk_size);
13741        self.handle.write_at(fa_dblk_addr, &dblk_encoded)?;
13742
13743        // The maximum is stored even when it equals the dims: it is what
13744        // `extend_dataset` checks growth against, and the FA capacity above
13745        // is exactly its chunk grid.
13746        let dataspace = DataspaceMessage {
13747            // Chunked storage always requires at least one dimension, so
13748            // this is never Scalar or Null.
13749            class: DataspaceClass::Simple,
13750            dims: dims.to_vec(),
13751            max_dims: Some(max_dims.to_vec()),
13752        };
13753
13754        let idx = self.push_dataset(
13755            &create,
13756            DatasetInfo {
13757                name: name.to_string(),
13758                datatype,
13759                committed_type: None,
13760                external: None,
13761                virtual_storage: None,
13762                dataspace,
13763                read_format: None,
13764                obj_header_addr: 0,
13765                data_addr: UNDEF_ADDR,
13766                data_size: 0,
13767                compact: None,
13768                attributes: Vec::new(),
13769                obj_header_written_addr: None,
13770                obj_header_blocks: Vec::new(),
13771                filter_pipeline: pipeline,
13772                deleted: false,
13773                extent_dirty: false,
13774                header_dirty: false,
13775                nlink_written: 1,
13776                creation_seq: self.take_creation_seq(),
13777                track_attr_order: self.track_order.attrs,
13778                fill_value: None,
13779                fill_time: FILL_TIME_IFSET,
13780                layout_version,
13781                times: self.created_object_times(),
13782                chunked: None,
13783                btree_v2: None,
13784                implicit: None,
13785                single_chunk: None,
13786                btree_v1: None,
13787                fixed_array: Some(FixedArrayDatasetInfo {
13788                    chunk_dims: chunk_dims.to_vec(),
13789                    fa_header_addr,
13790                    fa_dblk_addr,
13791                    fa_header,
13792                    fa_dblk,
13793                    chunks_written: 0,
13794                }),
13795                append: None,
13796            },
13797        );
13798
13799        Ok(idx)
13800    }
13801
13802    /// Define a chunked dataset with the *implicit* index: no index structure
13803    /// at all, every chunk of the grid allocated at create in one contiguous
13804    /// run, addressed by arithmetic (`H5Dnone.c`).
13805    ///
13806    /// libhdf5 picks this index only where that arithmetic is total, and this
13807    /// enforces the same three conditions
13808    /// (`H5D__layout_set_latest_indexing`, H5Dlayout.c): no filter — a
13809    /// filtered chunk is not `chunk_bytes` long, so the run would not be a
13810    /// grid; no unlimited dimension — the run has to have a length; and early
13811    /// allocation, which is what this creator *does* rather than something it
13812    /// checks. The dataset's fill-value message says so
13813    /// (`build_dataset_header`), because a file claiming incremental
13814    /// allocation is one libhdf5 would never have chosen this index for.
13815    pub fn create_implicit_dataset(
13816        &self,
13817        name: &str,
13818        datatype: DatatypeMessage,
13819        dims: &[u64],
13820        chunk_dims: &[u64],
13821    ) -> IoResult<usize> {
13822        let create = self.begin_create(name)?;
13823        let name = create.name.as_str();
13824        validate_chunk_geometry(dims, dims, chunk_dims)?;
13825        let mut num_chunks: u64 = 1;
13826        for g in crate::io::chunk_grid::index_grid(dims, None, chunk_dims)? {
13827            num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
13828                crate::io::IoError::InvalidState("chunk count overflows u64".into())
13829            })?;
13830        }
13831        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13832        let data_size = num_chunks.checked_mul(chunk_bytes).ok_or_else(|| {
13833            crate::io::IoError::InvalidState("implicit chunk storage overflows u64".into())
13834        })?;
13835        let layout_version = self.chunk_layout_version(false, chunk_bytes);
13836
13837        // Early allocation is the whole of this index: the run exists, and
13838        // holds the fill value, before any chunk is written. It is written
13839        // out rather than merely reserved because the file's end-of-file
13840        // address is what libhdf5 checks a file's completeness against — a
13841        // reserved-but-absent tail is a truncated file to it.
13842        let data_addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
13843        self.handle.write_at(
13844            data_addr,
13845            &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
13846        )?;
13847
13848        let dataspace = DataspaceMessage {
13849            // Chunked storage always requires at least one dimension, so
13850            // this is never Scalar or Null.
13851            class: DataspaceClass::Simple,
13852            dims: dims.to_vec(),
13853            max_dims: Some(dims.to_vec()),
13854        };
13855
13856        let idx = self.push_dataset(
13857            &create,
13858            DatasetInfo {
13859                name: name.to_string(),
13860                datatype,
13861                committed_type: None,
13862                external: None,
13863                virtual_storage: None,
13864                dataspace,
13865                read_format: None,
13866                obj_header_addr: 0,
13867                data_addr: UNDEF_ADDR,
13868                data_size: 0,
13869                compact: None,
13870                attributes: Vec::new(),
13871                obj_header_written_addr: None,
13872                obj_header_blocks: Vec::new(),
13873                filter_pipeline: None,
13874                deleted: false,
13875                extent_dirty: false,
13876                header_dirty: false,
13877                nlink_written: 1,
13878                creation_seq: self.take_creation_seq(),
13879                track_attr_order: self.track_order.attrs,
13880                fill_value: None,
13881                fill_time: FILL_TIME_IFSET,
13882                layout_version,
13883                times: self.created_object_times(),
13884                chunked: None,
13885                btree_v2: None,
13886                fixed_array: None,
13887                implicit: Some(ImplicitDatasetInfo {
13888                    chunk_dims: chunk_dims.to_vec(),
13889                    data_addr,
13890                    data_size,
13891                }),
13892                single_chunk: None,
13893                btree_v1: None,
13894                append: None,
13895            },
13896        );
13897
13898        Ok(idx)
13899    }
13900
13901    /// Define a chunked dataset indexed by the single-chunk index: a fixed
13902    /// shape covered by exactly one whole chunk (`chunk_dims == dims`), its
13903    /// address — and, once written, size and filter mask if filtered — held
13904    /// directly in the layout message instead of any index structure
13905    /// (`H5Dsingle.c`). libhdf5 selects this index ahead of both Implicit and
13906    /// Fixed Array whenever the shape qualifies, filtered or not, early
13907    /// allocation or not (`H5D__layout_set_latest_indexing`).
13908    ///
13909    /// `early_alloc` mirrors [`create_implicit_dataset`](Self::create_implicit_dataset):
13910    /// when true, the chunk's storage is allocated and filled with the fill
13911    /// value immediately, matching an early-allocated unfiltered dataset
13912    /// whose one chunk covers the whole shape. When false, the chunk has no
13913    /// address until its first write, the same as an unfiltered Fixed Array
13914    /// element.
13915    pub fn create_single_chunk_dataset(
13916        &self,
13917        name: &str,
13918        datatype: DatatypeMessage,
13919        dims: &[u64],
13920        chunk_dims: &[u64],
13921        early_alloc: bool,
13922    ) -> IoResult<usize> {
13923        let create = self.begin_create(name)?;
13924        let name = create.name.as_str();
13925        validate_chunk_geometry(dims, dims, chunk_dims)?;
13926        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
13927        let layout_version = self.chunk_layout_version(false, data_size);
13928
13929        let data_addr = if early_alloc {
13930            // Same reasoning as `create_implicit_dataset`: the fill-value
13931            // bytes are written now, not merely reserved, because the
13932            // file's end-of-file address is what libhdf5 checks a file's
13933            // completeness against.
13934            let addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
13935            self.handle.write_at(
13936                addr,
13937                &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
13938            )?;
13939            addr
13940        } else {
13941            UNDEF_ADDR
13942        };
13943
13944        let dataspace = DataspaceMessage {
13945            // Chunked storage always requires at least one dimension, so
13946            // this is never Scalar or Null.
13947            class: DataspaceClass::Simple,
13948            dims: dims.to_vec(),
13949            max_dims: Some(dims.to_vec()),
13950        };
13951
13952        let idx = self.push_dataset(
13953            &create,
13954            DatasetInfo {
13955                name: name.to_string(),
13956                datatype,
13957                committed_type: None,
13958                external: None,
13959                virtual_storage: None,
13960                dataspace,
13961                read_format: None,
13962                obj_header_addr: 0,
13963                data_addr: UNDEF_ADDR,
13964                data_size: 0,
13965                compact: None,
13966                attributes: Vec::new(),
13967                obj_header_written_addr: None,
13968                obj_header_blocks: Vec::new(),
13969                filter_pipeline: None,
13970                deleted: false,
13971                extent_dirty: false,
13972                header_dirty: false,
13973                nlink_written: 1,
13974                creation_seq: self.take_creation_seq(),
13975                track_attr_order: self.track_order.attrs,
13976                fill_value: None,
13977                fill_time: FILL_TIME_IFSET,
13978                layout_version,
13979                times: self.created_object_times(),
13980                chunked: None,
13981                btree_v2: None,
13982                fixed_array: None,
13983                implicit: None,
13984                single_chunk: Some(SingleChunkDatasetInfo {
13985                    chunk_dims: chunk_dims.to_vec(),
13986                    data_addr,
13987                    data_size,
13988                    nbytes: if early_alloc { data_size } else { 0 },
13989                    filter_mask: 0,
13990                    chunks_written: 0,
13991                    early_alloc,
13992                }),
13993                btree_v1: None,
13994                append: None,
13995            },
13996        );
13997
13998        Ok(idx)
13999    }
14000
14001    /// Define a fixed-shape compressed chunked dataset — of exactly one
14002    /// whole chunk — indexed by a *filtered* single-chunk index
14003    /// (`H5O_LAYOUT_CHUNK_SINGLE_INDEX_WITH_FILTER`, H5Dsingle.c). The
14004    /// chunk's stored size and filter mask are recorded inline in the
14005    /// layout message once the chunk is written.
14006    ///
14007    /// Like [`create_fixed_array_dataset_with_pipeline`](Self::create_fixed_array_dataset_with_pipeline),
14008    /// there is nothing to allocate ahead of that first write — a filtered
14009    /// chunk's stored length isn't known until it is compressed — so this
14010    /// dataset is always incrementally allocated regardless of the caller's
14011    /// requested allocation time.
14012    pub fn create_single_chunk_dataset_with_pipeline(
14013        &self,
14014        name: &str,
14015        datatype: DatatypeMessage,
14016        dims: &[u64],
14017        chunk_dims: &[u64],
14018        pipeline: FilterPipeline,
14019    ) -> IoResult<usize> {
14020        let create = self.begin_create(name)?;
14021        let name = create.name.as_str();
14022        validate_chunk_geometry(dims, dims, chunk_dims)?;
14023        let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14024        let layout_version = self.chunk_layout_version(true, data_size);
14025
14026        let dataspace = DataspaceMessage {
14027            // Chunked storage always requires at least one dimension, so
14028            // this is never Scalar or Null.
14029            class: DataspaceClass::Simple,
14030            dims: dims.to_vec(),
14031            max_dims: Some(dims.to_vec()),
14032        };
14033
14034        let idx = self.push_dataset(
14035            &create,
14036            DatasetInfo {
14037                name: name.to_string(),
14038                datatype,
14039                committed_type: None,
14040                external: None,
14041                virtual_storage: None,
14042                dataspace,
14043                read_format: None,
14044                obj_header_addr: 0,
14045                data_addr: UNDEF_ADDR,
14046                data_size: 0,
14047                compact: None,
14048                attributes: Vec::new(),
14049                obj_header_written_addr: None,
14050                obj_header_blocks: Vec::new(),
14051                filter_pipeline: Some(pipeline),
14052                deleted: false,
14053                extent_dirty: false,
14054                header_dirty: false,
14055                nlink_written: 1,
14056                creation_seq: self.take_creation_seq(),
14057                track_attr_order: self.track_order.attrs,
14058                fill_value: None,
14059                fill_time: FILL_TIME_IFSET,
14060                layout_version,
14061                times: self.created_object_times(),
14062                chunked: None,
14063                btree_v2: None,
14064                fixed_array: None,
14065                implicit: None,
14066                single_chunk: Some(SingleChunkDatasetInfo {
14067                    chunk_dims: chunk_dims.to_vec(),
14068                    data_addr: UNDEF_ADDR,
14069                    data_size,
14070                    nbytes: 0,
14071                    filter_mask: 0,
14072                    chunks_written: 0,
14073                    early_alloc: false,
14074                }),
14075                btree_v1: None,
14076                append: None,
14077            },
14078        );
14079
14080        Ok(idx)
14081    }
14082
14083    /// Define a chunked dataset indexed by a version-1 B-tree — the classic
14084    /// chunk index, and the only one a version-0/1 superblock file can carry.
14085    ///
14086    /// The tree itself is not created here: libhdf5 leaves the layout
14087    /// message's address undefined until the first chunk is inserted
14088    /// (`H5D__btree_idx_create` runs on that insert), and so does this — the
14089    /// flush that bulk-loads the records is what puts a node in the file.
14090    ///
14091    /// Unlike the array indexes this one has no grid to size, so it takes any
14092    /// number of unlimited dimensions: a key *is* the chunk's position, and
14093    /// the tree is ordered by it.
14094    pub fn create_btree_v1_dataset(
14095        &self,
14096        name: &str,
14097        datatype: DatatypeMessage,
14098        dims: &[u64],
14099        max_dims: &[u64],
14100        chunk_dims: &[u64],
14101        pipeline: Option<FilterPipeline>,
14102    ) -> IoResult<usize> {
14103        let create = self.begin_create(name)?;
14104        let name = create.name.as_str();
14105        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14106        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14107        if chunk_bytes > u32::MAX as u64 {
14108            return Err(crate::io::IoError::InvalidState(format!(
14109                "a {chunk_bytes}-byte chunk does not fit the 32-bit size field of a \
14110                 version-1 B-tree chunk key"
14111            )));
14112        }
14113
14114        let dataspace = DataspaceMessage {
14115            // Chunked storage always requires at least one dimension, so
14116            // this is never Scalar or Null.
14117            class: DataspaceClass::Simple,
14118            dims: dims.to_vec(),
14119            max_dims: Some(max_dims.to_vec()),
14120        };
14121
14122        let idx = self.push_dataset(
14123            &create,
14124            DatasetInfo {
14125                name: name.to_string(),
14126                datatype,
14127                committed_type: None,
14128                external: None,
14129                virtual_storage: None,
14130                dataspace,
14131                read_format: None,
14132                obj_header_addr: 0,
14133                data_addr: UNDEF_ADDR,
14134                data_size: 0,
14135                compact: None,
14136                attributes: Vec::new(),
14137                obj_header_written_addr: None,
14138                obj_header_blocks: Vec::new(),
14139                filter_pipeline: pipeline,
14140                deleted: false,
14141                extent_dirty: false,
14142                header_dirty: false,
14143                nlink_written: 1,
14144                creation_seq: self.take_creation_seq(),
14145                track_attr_order: self.track_order.attrs,
14146                fill_value: None,
14147                fill_time: FILL_TIME_IFSET,
14148                // The version-3 data layout message this index encodes as:
14149                // `H5O_LAYOUT_VERSION_DEFAULT`, which is the floor of
14150                // `H5D__chunk_set_info`'s final MAX and the whole of it below
14151                // the version-4 gate — a bound whose row is lower does not
14152                // push the message down, it only keeps the v1.10 indexes out.
14153                layout_version: LAYOUT_VERSION_DEFAULT,
14154                times: self.created_object_times(),
14155                chunked: None,
14156                fixed_array: None,
14157                btree_v2: None,
14158                implicit: None,
14159                single_chunk: None,
14160                btree_v1: Some(BtreeV1DatasetInfo {
14161                    chunk_dims: chunk_dims.to_vec(),
14162                    max_dims: max_dims.to_vec(),
14163                    config: self.btree_v1_config(),
14164                    records: Vec::new(),
14165                    node_addrs: Vec::new(),
14166                    root_addr: UNDEF_ADDR,
14167                    chunks_written: 0,
14168                }),
14169                append: None,
14170            },
14171        );
14172
14173        Ok(idx)
14174    }
14175
14176    /// Define a chunked dataset indexed by a B-tree v2 (multiple unlimited dimensions).
14177    ///
14178    /// Returns the dataset index.
14179    pub fn create_btree_v2_dataset(
14180        &self,
14181        name: &str,
14182        datatype: DatatypeMessage,
14183        dims: &[u64],
14184        max_dims: &[u64],
14185        chunk_dims: &[u64],
14186    ) -> IoResult<usize> {
14187        self.create_btree_v2_dataset_inner(name, datatype, dims, max_dims, chunk_dims, None)
14188    }
14189
14190    /// Define a *filtered* chunked dataset indexed by a B-tree v2.
14191    ///
14192    /// The v2 B-tree counterpart of
14193    /// [`create_chunked_dataset_with_pipeline`](Self::create_chunked_dataset_with_pipeline):
14194    /// chunks are compressed on write and the index records each chunk's
14195    /// stored size and filter mask (record type 11), the same shape libhdf5
14196    /// builds when a multi-unlimited-dimension dataset has a filter pipeline
14197    /// (`H5Dbtree2.c`, `H5D_BT2_FILT`).
14198    pub fn create_btree_v2_dataset_with_pipeline(
14199        &self,
14200        name: &str,
14201        datatype: DatatypeMessage,
14202        dims: &[u64],
14203        max_dims: &[u64],
14204        chunk_dims: &[u64],
14205        pipeline: FilterPipeline,
14206    ) -> IoResult<usize> {
14207        self.create_btree_v2_dataset_inner(
14208            name,
14209            datatype,
14210            dims,
14211            max_dims,
14212            chunk_dims,
14213            Some(pipeline),
14214        )
14215    }
14216
14217    fn create_btree_v2_dataset_inner(
14218        &self,
14219        name: &str,
14220        datatype: DatatypeMessage,
14221        dims: &[u64],
14222        max_dims: &[u64],
14223        chunk_dims: &[u64],
14224        pipeline: Option<FilterPipeline>,
14225    ) -> IoResult<usize> {
14226        use crate::format::chunk_index::btree_v2::Bt2Header;
14227
14228        let create = self.begin_create(name)?;
14229        let name = create.name.as_str();
14230        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14231        let ndims = dims.len();
14232        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14233        let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
14234
14235        // The filtered record's size field is as wide as libhdf5 will
14236        // recompute it — from the uncompressed chunk size under layout v4,
14237        // the fixed `sizeof_size` under layout v5 — exactly as the
14238        // extensible- and fixed-array filtered paths size theirs.
14239        let bt2_index = match pipeline {
14240            Some(_) => {
14241                let len = self.chunk_size_len_for(layout_version, chunk_bytes);
14242                Bt2ChunkIndex::new_filtered(ndims, len)
14243            }
14244            None => Bt2ChunkIndex::new_unfiltered(ndims),
14245        };
14246
14247        // The bulk loader spreads a level's records evenly over its nodes, one
14248        // separator between adjacent siblings, which needs room for a few
14249        // records per node. HDF5's rank limit of 32 leaves room for seven; a
14250        // wider rank than that has no valid geometry, so reject it here rather
14251        // than emit a tree no reader can walk.
14252        let record_size = bt2_index.record_size(&self.ctx) as usize;
14253        let node_size = bt2_index.node_size as usize;
14254        if node_size < 10 + 3 * record_size {
14255            return Err(crate::io::IoError::InvalidState(format!(
14256                "a {ndims}-dimension v2 B-tree record is {record_size} bytes, too wide \
14257                 for a {node_size}-byte node"
14258            )));
14259        }
14260
14261        // Only the header gets a home now: it names an empty tree, whose root
14262        // is undefined until the first flush bulk-loads the index into nodes.
14263        let hdr = if bt2_index.filtered {
14264            Bt2Header::new_for_filtered_chunks(&self.ctx, ndims, bt2_index.chunk_size_len)
14265        } else {
14266            Bt2Header::new_for_chunks(&self.ctx, ndims)
14267        };
14268        let hdr_encoded = hdr.encode(&self.ctx);
14269        let bt2_header_addr = self
14270            .allocator
14271            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14272        self.handle.write_at(bt2_header_addr, &hdr_encoded)?;
14273
14274        let dataspace = DataspaceMessage {
14275            // Chunked storage always requires at least one dimension, so
14276            // this is never Scalar or Null.
14277            class: DataspaceClass::Simple,
14278            dims: dims.to_vec(),
14279            max_dims: Some(max_dims.to_vec()),
14280        };
14281
14282        let idx = self.push_dataset(
14283            &create,
14284            DatasetInfo {
14285                name: name.to_string(),
14286                datatype,
14287                committed_type: None,
14288                external: None,
14289                virtual_storage: None,
14290                dataspace,
14291                read_format: None,
14292                obj_header_addr: 0,
14293                data_addr: UNDEF_ADDR,
14294                data_size: 0,
14295                compact: None,
14296                attributes: Vec::new(),
14297                obj_header_written_addr: None,
14298                obj_header_blocks: Vec::new(),
14299                filter_pipeline: pipeline,
14300                deleted: false,
14301                extent_dirty: false,
14302                header_dirty: false,
14303                nlink_written: 1,
14304                creation_seq: self.take_creation_seq(),
14305                track_attr_order: self.track_order.attrs,
14306                fill_value: None,
14307                fill_time: FILL_TIME_IFSET,
14308                layout_version,
14309                times: self.created_object_times(),
14310                chunked: None,
14311                fixed_array: None,
14312                implicit: None,
14313                single_chunk: None,
14314                btree_v1: None,
14315                btree_v2: Some(Bt2DatasetInfo {
14316                    chunk_dims: chunk_dims.to_vec(),
14317                    bt2_header_addr,
14318                    node_addrs: Vec::new(),
14319                    index: bt2_index,
14320                    chunks_written: 0,
14321                }),
14322                append: None,
14323            },
14324        );
14325
14326        Ok(idx)
14327    }
14328
14329    /// Create a chunked dataset with a custom filter pipeline.
14330    pub fn create_chunked_dataset_with_pipeline(
14331        &self,
14332        name: &str,
14333        datatype: DatatypeMessage,
14334        dims: &[u64],
14335        max_dims: &[u64],
14336        chunk_dims: &[u64],
14337        pipeline: FilterPipeline,
14338    ) -> IoResult<usize> {
14339        let create = self.begin_create(name)?;
14340        let name = create.name.as_str();
14341        validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14342        ensure_at_most_one_unlimited(max_dims)?;
14343        let element_size = datatype.element_size() as u64;
14344        let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * element_size;
14345        let layout_version = self.chunk_layout_version(true, chunk_bytes);
14346        let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
14347
14348        let earray_params = EarrayParams::default_params();
14349        let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
14350        let nsblk_addrs = compute_nsblk_addrs(
14351            earray_params.idx_blk_elmts,
14352            earray_params.data_blk_min_elmts,
14353            earray_params.sup_blk_min_data_ptrs,
14354            earray_params.max_nelmts_bits,
14355        )?;
14356
14357        let mut ea_header =
14358            ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
14359        ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
14360        ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
14361        ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
14362        ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
14363        ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
14364
14365        let hdr_encoded = ea_header.encode(&self.ctx);
14366        let ea_header_addr = self
14367            .allocator
14368            .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14369
14370        let filt_iblk = FilteredIndexBlock::new(
14371            ea_header_addr,
14372            earray_params.idx_blk_elmts,
14373            ndblk_addrs,
14374            nsblk_addrs,
14375        );
14376        let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
14377        let ea_iblk_addr = self
14378            .allocator
14379            .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
14380
14381        ea_header.idx_blk_addr = ea_iblk_addr;
14382        let hdr_encoded = ea_header.encode(&self.ctx);
14383        self.handle.write_at(ea_header_addr, &hdr_encoded)?;
14384        self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
14385
14386        let dataspace = DataspaceMessage {
14387            // Chunked storage always requires at least one dimension, so
14388            // this is never Scalar or Null.
14389            class: DataspaceClass::Simple,
14390            dims: dims.to_vec(),
14391            max_dims: Some(max_dims.to_vec()),
14392        };
14393        let ea_iblk = ExtensibleArrayIndexBlock::new(
14394            ea_header_addr,
14395            earray_params.idx_blk_elmts,
14396            ndblk_addrs,
14397            nsblk_addrs,
14398        );
14399
14400        let idx = self.push_dataset(
14401            &create,
14402            DatasetInfo {
14403                name: name.to_string(),
14404                datatype,
14405                committed_type: None,
14406                external: None,
14407                virtual_storage: None,
14408                dataspace,
14409                read_format: None,
14410                obj_header_addr: 0,
14411                data_addr: UNDEF_ADDR,
14412                data_size: 0,
14413                compact: None,
14414                attributes: Vec::new(),
14415                obj_header_written_addr: None,
14416                obj_header_blocks: Vec::new(),
14417                filter_pipeline: Some(pipeline),
14418                deleted: false,
14419                extent_dirty: false,
14420                header_dirty: false,
14421                nlink_written: 1,
14422                creation_seq: self.take_creation_seq(),
14423                track_attr_order: self.track_order.attrs,
14424                fill_value: None,
14425                fill_time: FILL_TIME_IFSET,
14426                layout_version,
14427                times: self.created_object_times(),
14428                fixed_array: None,
14429                implicit: None,
14430                single_chunk: None,
14431                btree_v1: None,
14432                btree_v2: None,
14433                chunked: Some(ChunkedDatasetInfo {
14434                    chunk_dims: chunk_dims.to_vec(),
14435                    earray_params,
14436                    ea_header_addr,
14437                    ea_iblk_addr,
14438                    ea_header,
14439                    ea_iblk,
14440                    chunks_written: 0,
14441                    filt_iblk: Some(filt_iblk),
14442                    chunk_size_len,
14443                }),
14444                append: None,
14445            },
14446        );
14447        Ok(idx)
14448    }
14449
14450    /// Write a chunk to a fixed-array-indexed dataset.
14451    ///
14452    /// `chunk_coords` is the multidimensional chunk index (e.g., [row_chunk, col_chunk]).
14453    /// The uncompressed `data` must be exactly one chunk wide; the filter
14454    /// pipeline (if any) runs here before the bytes reach the index.
14455    pub fn write_chunk_fixed_array(
14456        &self,
14457        index: usize,
14458        chunk_coords: &[u64],
14459        data: &[u8],
14460    ) -> IoResult<()> {
14461        let ds = self.ds(index);
14462        let _op = ds.op.lock();
14463        self.write_chunk_fixed_array_inner(index, chunk_coords, data)
14464    }
14465
14466    /// [`Self::write_chunk_fixed_array`] body; the caller holds the dataset's
14467    /// op lock or the writer exclusively.
14468    pub(crate) fn write_chunk_fixed_array_inner(
14469        &self,
14470        index: usize,
14471        chunk_coords: &[u64],
14472        data: &[u8],
14473    ) -> IoResult<()> {
14474        // Read what we need under one brief slot guard, then compress
14475        // OUTSIDE the lock: `record_fixed_array_chunk` re-locks the same slot,
14476        // so the guard must be dropped before it (and before apply_filters).
14477        let ds = self.ds(index);
14478        let (chunk_bytes, pipeline) = {
14479            let m = ds.lock();
14480            let element_size = m.datatype.element_size() as u64;
14481            let fa = m.fixed_array.as_ref().ok_or_else(|| {
14482                crate::io::IoError::InvalidState("not a fixed-array dataset".into())
14483            })?;
14484            (
14485                fa.chunk_dims.iter().product::<u64>() * element_size,
14486                m.filter_pipeline.clone(),
14487            )
14488        };
14489
14490        if data.len() as u64 != chunk_bytes {
14491            return Err(crate::io::IoError::InvalidState(format!(
14492                "chunk data size mismatch: expected {} bytes, got {}",
14493                chunk_bytes,
14494                data.len()
14495            )));
14496        }
14497        let write_data;
14498        let data_to_write = if let Some(ref pipeline) = pipeline {
14499            write_data = filter::apply_filters(pipeline, data)?;
14500            &write_data[..]
14501        } else {
14502            data
14503        };
14504        // filter_mask = 0: the whole pipeline ran (or the dataset is
14505        // unfiltered), so no filter is skipped for this chunk.
14506        self.record_fixed_array_chunk(index, chunk_coords, data_to_write, 0)
14507    }
14508
14509    /// Write a pre-filtered chunk verbatim to a fixed-array dataset, recording
14510    /// the caller-supplied `filter_mask`.
14511    ///
14512    /// The bytes are stored exactly as given (no filter pipeline is run); this
14513    /// is the fixed-array half of the HDF5 "direct chunk write"
14514    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
14515    /// means filter *i* of the pipeline was **not** applied to this chunk and
14516    /// must be skipped on read; pass 0 when the full pipeline was applied
14517    /// upstream.
14518    ///
14519    /// Requires a filtered dataset — only the filtered FA element carries the
14520    /// size+mask slot.
14521    ///
14522    /// The caller holds the dataset's op lock or the writer exclusively.
14523    pub(crate) fn write_compressed_chunk_fixed_array_inner(
14524        &self,
14525        index: usize,
14526        chunk_coords: &[u64],
14527        data: &[u8],
14528        filter_mask: u32,
14529    ) -> IoResult<()> {
14530        if self.ds(index).lock().filter_pipeline.is_none() {
14531            return Err(crate::io::IoError::InvalidState(
14532                "write_compressed_chunk_fixed_array requires a filtered dataset \
14533                 (no slot for a compressed size or filter mask on an unfiltered \
14534                 chunk index)"
14535                    .into(),
14536            ));
14537        }
14538        self.record_fixed_array_chunk(index, chunk_coords, data, filter_mask)
14539    }
14540
14541    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
14542    /// filtered if the dataset is filtered, raw otherwise) into a fixed-array
14543    /// dataset's data block, recording the caller-supplied `filter_mask`.
14544    /// Shared by [`write_chunk_fixed_array`](Self::write_chunk_fixed_array)
14545    /// and [`write_compressed_chunk_fixed_array`](Self::write_compressed_chunk_fixed_array).
14546    fn record_fixed_array_chunk(
14547        &self,
14548        index: usize,
14549        chunk_coords: &[u64],
14550        final_bytes: &[u8],
14551        filter_mask: u32,
14552    ) -> IoResult<()> {
14553        // Hold one slot guard for the whole method; `self.allocator`/`self.handle`/
14554        // `self.ctx` below touch disjoint fields safe to use with the guard held.
14555        let ds = self.ds(index);
14556        let mut m = ds.lock();
14557        let is_filtered = m.filter_pipeline.is_some();
14558        let fa = m
14559            .fixed_array
14560            .as_ref()
14561            .ok_or_else(|| crate::io::IoError::InvalidState("not a fixed-array dataset".into()))?;
14562
14563        // Linear chunk index in the maximum-extent grid — the slot the fixed
14564        // array (sized from that grid at create) records the chunk under.
14565        let linear_idx = crate::io::chunk_grid::linear_index(
14566            &m.dataspace.dims,
14567            m.dataspace.max_dims.as_deref(),
14568            &fa.chunk_dims,
14569            chunk_coords,
14570        )?;
14571
14572        // Update the fixed array data block. The slot is read before the bytes
14573        // are placed so a rewrite can stay where it is (see `place_chunk`).
14574        let fa = m.fixed_array.as_mut().unwrap();
14575        let lidx = linear_idx as usize;
14576        if is_filtered {
14577            // Filtered FA: store address + stored size + filter mask. A
14578            // non-zero mask bit means "filter i was skipped for this chunk".
14579            let stored_size = final_bytes.len();
14580            // The stored size is encoded in the FA header's `chunk_size_len`-byte
14581            // field; libhdf5 errors if it does not fit (H5D_CHUNK_ENCODE_SIZE_CHECK)
14582            // rather than truncating silently. element_size = sizeof_addr +
14583            // chunk_size_len + 4 by construction.
14584            let chunk_size_len = (fa.fa_header.element_size as usize)
14585                .checked_sub(self.ctx.sizeof_addr as usize + 4)
14586                .ok_or_else(|| {
14587                    crate::io::IoError::InvalidState(
14588                        "filtered fixed-array element size is too small".into(),
14589                    )
14590                })?;
14591            if chunk_size_len < 8 && stored_size >= (1usize << (chunk_size_len * 8)) {
14592                return Err(crate::io::IoError::InvalidState(format!(
14593                    "compressed chunk size {stored_size} does not fit in the \
14594                     {chunk_size_len}-byte fixed-array chunk-size field"
14595                )));
14596            }
14597            if lidx < fa.fa_dblk.filtered_elements.len() {
14598                let old = &fa.fa_dblk.filtered_elements[lidx];
14599                let chunk_addr =
14600                    self.place_chunk(Some((old.address, old.chunk_size)), stored_size as u64);
14601                self.handle.write_at(chunk_addr, final_bytes)?;
14602                fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
14603                    address: chunk_addr,
14604                    chunk_size: stored_size as u64,
14605                    filter_mask,
14606                };
14607                fa.chunks_written += 1;
14608            } else {
14609                return Err(crate::io::IoError::InvalidState(format!(
14610                    "chunk index {} out of range (max {})",
14611                    linear_idx,
14612                    fa.fa_dblk.filtered_elements.len()
14613                )));
14614            }
14615        } else {
14616            // An unfiltered fixed array stores only addresses — there is no
14617            // slot for a filter mask, so a non-zero mask cannot be honored.
14618            if filter_mask != 0 {
14619                return Err(crate::io::IoError::InvalidState(
14620                    "filter_mask is non-zero but the dataset is unfiltered".into(),
14621                ));
14622            }
14623            if lidx < fa.fa_dblk.elements.len() {
14624                // Unfiltered: the stored size is fixed by the chunk shape, so
14625                // a rewrite always fits its old block.
14626                let old = fa.fa_dblk.elements[lidx];
14627                let len = final_bytes.len() as u64;
14628                let chunk_addr = self.place_chunk(Some((old, len)), len);
14629                self.handle.write_at(chunk_addr, final_bytes)?;
14630                fa.fa_dblk.elements[lidx] = chunk_addr;
14631                fa.chunks_written += 1;
14632            } else {
14633                return Err(crate::io::IoError::InvalidState(format!(
14634                    "chunk index {} out of range (max {})",
14635                    linear_idx,
14636                    fa.fa_dblk.elements.len()
14637                )));
14638            }
14639        }
14640
14641        Ok(())
14642    }
14643
14644    /// Write the one chunk of a single-chunk indexed dataset.
14645    ///
14646    /// `chunk_coords` is validated against the grid the same way every other
14647    /// coordinate-addressed index does (`ChunkGeometry::linear_index`), even
14648    /// though the grid holds exactly one slot — this is what rejects an
14649    /// out-of-range coordinate instead of silently writing to that slot.
14650    /// `data` is the chunk's unfiltered bytes; the dataset's filter pipeline
14651    /// runs here if it has one.
14652    ///
14653    /// The caller holds the dataset's op lock or the writer exclusively.
14654    pub(crate) fn write_chunk_single_chunk_inner(
14655        &self,
14656        index: usize,
14657        chunk_coords: &[u64],
14658        data: &[u8],
14659    ) -> IoResult<()> {
14660        let geo = self.chunk_geometry(index)?;
14661        geo.linear_index(chunk_coords)?;
14662        let chunk_bytes = geo.chunk_bytes();
14663        if data.len() as u64 != chunk_bytes {
14664            return Err(crate::io::IoError::InvalidState(format!(
14665                "chunk data size mismatch: expected {} bytes, got {}",
14666                chunk_bytes,
14667                data.len()
14668            )));
14669        }
14670        let pipeline = self.ds(index).lock().filter_pipeline.clone();
14671        let write_data;
14672        let data_to_write = if let Some(ref pipeline) = pipeline {
14673            write_data = filter::apply_filters(pipeline, data)?;
14674            &write_data[..]
14675        } else {
14676            data
14677        };
14678        // filter_mask = 0: the whole pipeline ran (or the dataset is
14679        // unfiltered), so no filter is skipped for this chunk.
14680        self.record_single_chunk(index, data_to_write, 0)
14681    }
14682
14683    /// Write a pre-filtered chunk verbatim to a single-chunk dataset,
14684    /// recording the caller-supplied `filter_mask`.
14685    ///
14686    /// The bytes are stored exactly as given (no filter pipeline is run); this
14687    /// is the single-chunk half of the HDF5 "direct chunk write"
14688    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
14689    /// means filter *i* of the pipeline was **not** applied to this chunk and
14690    /// must be skipped on read; pass 0 when the full pipeline was applied
14691    /// upstream.
14692    ///
14693    /// Requires a filtered dataset — only the filtered single-chunk layout
14694    /// carries a size+mask slot.
14695    ///
14696    /// The caller holds the dataset's op lock or the writer exclusively.
14697    pub(crate) fn write_compressed_chunk_single_chunk_inner(
14698        &self,
14699        index: usize,
14700        chunk_coords: &[u64],
14701        data: &[u8],
14702        filter_mask: u32,
14703    ) -> IoResult<()> {
14704        if self.ds(index).lock().filter_pipeline.is_none() {
14705            return Err(crate::io::IoError::InvalidState(
14706                "write_compressed_chunk_single_chunk requires a filtered dataset \
14707                 (no slot for a compressed size or filter mask on an unfiltered \
14708                 chunk index)"
14709                    .into(),
14710            ));
14711        }
14712        let geo = self.chunk_geometry(index)?;
14713        geo.linear_index(chunk_coords)?;
14714        self.record_single_chunk(index, data, filter_mask)
14715    }
14716
14717    /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
14718    /// filtered if the dataset is filtered, raw otherwise) into a single-chunk
14719    /// dataset's layout message fields, recording the caller-supplied
14720    /// `filter_mask`. Shared by
14721    /// [`write_chunk_single_chunk_inner`](Self::write_chunk_single_chunk_inner)
14722    /// and
14723    /// [`write_compressed_chunk_single_chunk_inner`](Self::write_compressed_chunk_single_chunk_inner).
14724    ///
14725    /// Unlike the array indexes there is no per-chunk slot to look up — the
14726    /// dataset has exactly one chunk, and its address/size/mask live directly
14727    /// in the layout message (`H5Dsingle.c`) — so this only ever rewrites the
14728    /// one chunk in place, via [`place_chunk`](Self::place_chunk) the same as
14729    /// every other index's rewrite path.
14730    fn record_single_chunk(
14731        &self,
14732        index: usize,
14733        final_bytes: &[u8],
14734        filter_mask: u32,
14735    ) -> IoResult<()> {
14736        let ds = self.ds(index);
14737        let mut m = ds.lock();
14738        let is_filtered = m.filter_pipeline.is_some();
14739        if !is_filtered && filter_mask != 0 {
14740            return Err(crate::io::IoError::InvalidState(
14741                "filter_mask is non-zero but the dataset is unfiltered".into(),
14742            ));
14743        }
14744        let sc = m
14745            .single_chunk
14746            .as_ref()
14747            .ok_or_else(|| crate::io::IoError::InvalidState("not a single-chunk dataset".into()))?;
14748
14749        // A rewrite whose stored size is unchanged stays where it is (always
14750        // so when unfiltered), one that no longer fits moves. See `place_chunk`.
14751        let old = if sc.data_addr == UNDEF_ADDR {
14752            None
14753        } else {
14754            Some((
14755                sc.data_addr,
14756                if is_filtered { sc.nbytes } else { sc.data_size },
14757            ))
14758        };
14759        let stored_size = final_bytes.len() as u64;
14760        let addr = self.place_chunk(old, stored_size);
14761        self.handle.write_at(addr, final_bytes)?;
14762
14763        let sc = m.single_chunk.as_mut().unwrap();
14764        sc.data_addr = addr;
14765        sc.nbytes = stored_size;
14766        sc.filter_mask = filter_mask;
14767        sc.chunks_written = 1;
14768        Ok(())
14769    }
14770
14771    /// Write a chunk to a B-tree v2 indexed dataset.
14772    ///
14773    /// `chunk_coords` is the scaled chunk coordinates (one per dimension).
14774    /// `data` is the chunk's unfiltered bytes; if the dataset has a filter
14775    /// pipeline it runs here and the index records the stored size and mask.
14776    ///
14777    /// Production writes call [`write_chunk_btree_v2_inner`](Self::write_chunk_btree_v2_inner)
14778    /// directly (they already hold the dataset's op lock); this self-locking
14779    /// form is kept as a direct entry point for this crate's own white-box
14780    /// tests.
14781    #[cfg(test)]
14782    pub fn write_chunk_btree_v2(
14783        &self,
14784        index: usize,
14785        chunk_coords: &[u64],
14786        data: &[u8],
14787    ) -> IoResult<()> {
14788        let ds = self.ds(index);
14789        let _op = ds.op.lock();
14790        self.write_chunk_btree_v2_inner(index, chunk_coords, data)
14791    }
14792
14793    /// [`Self::write_chunk_btree_v2`] body; the caller holds the dataset's op
14794    /// lock or the writer exclusively.
14795    pub(crate) fn write_chunk_btree_v2_inner(
14796        &self,
14797        index: usize,
14798        chunk_coords: &[u64],
14799        data: &[u8],
14800    ) -> IoResult<()> {
14801        // Read what the write needs under a brief guard, then compress OUTSIDE
14802        // the lock — filtering a chunk must not hold the dataset slot.
14803        let ds = self.ds(index);
14804        let (chunk_bytes, pipeline) = {
14805            let m = ds.lock();
14806            let element_size = m.datatype.element_size() as u64;
14807            let bt2 = m.btree_v2.as_ref().ok_or_else(|| {
14808                crate::io::IoError::InvalidState("not a B-tree v2 dataset".into())
14809            })?;
14810            (
14811                bt2.chunk_dims.iter().product::<u64>() * element_size,
14812                m.filter_pipeline.clone(),
14813            )
14814        };
14815
14816        if data.len() as u64 != chunk_bytes {
14817            return Err(crate::io::IoError::InvalidState(format!(
14818                "chunk data size mismatch: expected {} bytes, got {}",
14819                chunk_bytes,
14820                data.len()
14821            )));
14822        }
14823
14824        let filtered;
14825        let stored = match pipeline {
14826            Some(ref pl) => {
14827                filtered = filter::apply_filters(pl, data)?;
14828                &filtered[..]
14829            }
14830            None => data,
14831        };
14832
14833        // filter_mask = 0: the whole pipeline ran (or the dataset is
14834        // unfiltered), so no filter is skipped.
14835        self.record_btree_v2_chunk(index, chunk_coords, stored, 0)
14836    }
14837
14838    /// Write a pre-filtered chunk verbatim to a BT2-indexed dataset, recording
14839    /// the caller-supplied `filter_mask`.
14840    ///
14841    /// The v2-B-tree half of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
14842    /// The bytes are stored exactly as given; `filter_mask` bit *i* set means
14843    /// filter *i* of the pipeline was **not** applied and must be skipped on
14844    /// read. Requires a filtered dataset — only a type-11 record has a slot for
14845    /// a stored size and mask.
14846    ///
14847    /// The caller holds the dataset's op lock or the writer exclusively.
14848    pub(crate) fn write_compressed_chunk_btree_v2_inner(
14849        &self,
14850        index: usize,
14851        chunk_coords: &[u64],
14852        data: &[u8],
14853        filter_mask: u32,
14854    ) -> IoResult<()> {
14855        if self.ds(index).lock().filter_pipeline.is_none() {
14856            return Err(crate::io::IoError::InvalidState(
14857                "write_compressed_chunk_btree_v2 requires a filtered dataset (no \
14858                 slot for a compressed size or filter mask on an unfiltered chunk \
14859                 index)"
14860                    .into(),
14861            ));
14862        }
14863        self.record_btree_v2_chunk(index, chunk_coords, data, filter_mask)
14864    }
14865
14866    /// Place a chunk's already-final bytes (filtered if the dataset is
14867    /// filtered, raw otherwise) in the file and record them in the v2 B-tree,
14868    /// under the caller-supplied `filter_mask`.
14869    ///
14870    /// Shared by [`write_chunk_btree_v2`](Self::write_chunk_btree_v2) and
14871    /// [`write_compressed_chunk_btree_v2`](Self::write_compressed_chunk_btree_v2),
14872    /// so both reach the index through one placement rule.
14873    fn record_btree_v2_chunk(
14874        &self,
14875        index: usize,
14876        chunk_coords: &[u64],
14877        final_bytes: &[u8],
14878        filter_mask: u32,
14879    ) -> IoResult<()> {
14880        let stored_len = final_bytes.len() as u64;
14881        let ds = self.ds(index);
14882        let mut m = ds.lock();
14883        let element_size = m.datatype.element_size() as u64;
14884        let bt2 = m
14885            .btree_v2
14886            .as_ref()
14887            .ok_or_else(|| crate::io::IoError::InvalidState("not a B-tree v2 dataset".into()))?;
14888        let chunk_bytes = bt2.chunk_dims.iter().product::<u64>() * element_size;
14889        // A filtered record encodes the stored size in a `chunk_size_len`-byte
14890        // field that truncates silently. Reject a size that would not fit, as
14891        // the extensible-array path does — the compress path never exceeds it,
14892        // but a direct write with caller-supplied bytes can.
14893        if bt2.index.filtered {
14894            let chunk_size_len = bt2.index.chunk_size_len as usize;
14895            if chunk_size_len < 8 && stored_len >= (1u64 << (chunk_size_len * 8)) {
14896                return Err(crate::io::IoError::InvalidState(format!(
14897                    "filtered chunk size {stored_len} does not fit in the \
14898                     {chunk_size_len}-byte v2 B-tree chunk-size field"
14899                )));
14900            }
14901        }
14902        // Place the bytes: a rewrite whose stored size is unchanged stays
14903        // where it is (always so when unfiltered — the size is fixed by the
14904        // chunk shape), and one that no longer fits moves, releasing its old
14905        // block. See `place_chunk`.
14906        let old = if bt2.index.filtered {
14907            bt2.index
14908                .lookup_filtered(chunk_coords)
14909                .map(|r| (r.chunk_address, r.chunk_size))
14910        } else {
14911            bt2.index
14912                .lookup(chunk_coords)
14913                .map(|r| (r.chunk_address, chunk_bytes))
14914        };
14915        let chunk_addr = self.place_chunk(old, stored_len);
14916        self.handle.write_at(chunk_addr, final_bytes)?;
14917
14918        let bt2 = m.btree_v2.as_mut().unwrap();
14919        if bt2.index.filtered {
14920            bt2.index
14921                .insert_filtered(chunk_coords.to_vec(), chunk_addr, stored_len, filter_mask);
14922        } else {
14923            bt2.index.insert(chunk_coords.to_vec(), chunk_addr);
14924        }
14925        bt2.chunks_written += 1;
14926
14927        Ok(())
14928    }
14929
14930    /// Write multiple chunks in a batch, optionally compressing in parallel.
14931    ///
14932    /// `chunks` is a list of (chunk_idx, data) pairs for an EA-indexed dataset.
14933    pub fn write_chunks_batch(&self, ds_index: usize, chunks: &[(u64, &[u8])]) -> IoResult<()> {
14934        let ds = self.ds(ds_index);
14935        let _op = ds.op.lock();
14936        self.write_chunks_batch_inner(ds_index, chunks)
14937    }
14938
14939    /// [`Self::write_chunks_batch`] body; the caller holds the dataset's op
14940    /// lock or the writer exclusively.
14941    pub(crate) fn write_chunks_batch_inner(
14942        &self,
14943        ds_index: usize,
14944        chunks: &[(u64, &[u8])],
14945    ) -> IoResult<()> {
14946        #[cfg(feature = "parallel")]
14947        {
14948            // If filter pipeline is set, compress all chunks in parallel.
14949            // Clone the pipeline out under a brief slot guard so the parallel
14950            // compression below runs off the lock.
14951            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
14952            if let Some(ref pipeline) = pipeline {
14953                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
14954                // Propagate a filter error rather than storing raw bytes under a
14955                // filter_mask that claims the pipeline ran (see
14956                // apply_filters_parallel). Ok reaching here means every chunk
14957                // compressed fully, so filter_mask = 0 is truthful.
14958                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
14959                for ((idx, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
14960                    self.write_compressed_chunk_inner(ds_index, *idx, compressed_data, 0)?;
14961                }
14962                return Ok(());
14963            }
14964        }
14965        // Fallback: sequential
14966        for (idx, data) in chunks {
14967            self.write_chunk_inner(ds_index, *idx, data)?;
14968        }
14969        Ok(())
14970    }
14971
14972    /// Write multiple fixed-array chunks in a batch, compressing them in
14973    /// parallel when a filter pipeline is set and the `parallel` feature is on.
14974    ///
14975    /// The fixed-array analogue of [`write_chunks_batch`](Self::write_chunks_batch):
14976    /// chunks are addressed by grid coordinates rather than a linear index.
14977    /// `record_fixed_array_chunk` writes already-compressed bytes verbatim, so
14978    /// the parallel compressor is the only place a filter runs. Falls back to
14979    /// per-chunk [`write_chunk_fixed_array`](Self::write_chunk_fixed_array) when
14980    /// unfiltered or when `parallel` is off.
14981    ///
14982    /// The caller holds the dataset's op lock or the writer exclusively.
14983    pub(crate) fn write_chunks_fixed_array_batch_inner(
14984        &self,
14985        ds_index: usize,
14986        chunks: &[(&[u64], &[u8])],
14987    ) -> IoResult<()> {
14988        #[cfg(feature = "parallel")]
14989        {
14990            // Clone the pipeline out under a brief slot guard so the parallel
14991            // compression below runs off the lock.
14992            let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
14993            if let Some(ref pipeline) = pipeline {
14994                let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
14995                // Same single owner as the EA batch: apply_filters_parallel
14996                // propagates a filter error instead of storing raw bytes under a
14997                // filter_mask that claims the pipeline ran. Ok here means every
14998                // chunk compressed fully, so filter_mask = 0 is truthful.
14999                let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15000                for ((coords, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15001                    self.record_fixed_array_chunk(ds_index, coords, compressed_data, 0)?;
15002                }
15003                return Ok(());
15004            }
15005        }
15006        // Fallback: sequential (write_chunk_fixed_array_inner compresses per
15007        // chunk).
15008        for (coords, data) in chunks {
15009            self.write_chunk_fixed_array_inner(ds_index, coords, data)?;
15010        }
15011        Ok(())
15012    }
15013
15014    /// Write a pre-filtered chunk verbatim to an EA-indexed dataset, recording
15015    /// the caller-supplied `filter_mask`.
15016    ///
15017    /// The bytes are stored exactly as given (no filter pipeline is run); this
15018    /// is the extensible-array half of the HDF5 "direct chunk write"
15019    /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15020    /// means filter *i* of the pipeline was **not** applied to this chunk and
15021    /// must be skipped on read; pass 0 when the full pipeline was applied
15022    /// upstream.
15023    ///
15024    /// Requires a filtered dataset — only the filtered EA entry carries the
15025    /// size+mask slot. An unfiltered dataset has nowhere to record either.
15026    ///
15027    /// The caller holds the dataset's op lock or the writer exclusively.
15028    pub(crate) fn write_compressed_chunk_inner(
15029        &self,
15030        index: usize,
15031        chunk_idx: u64,
15032        compressed_data: &[u8],
15033        filter_mask: u32,
15034    ) -> IoResult<()> {
15035        if self.ds(index).lock().filter_pipeline.is_none() {
15036            return Err(crate::io::IoError::InvalidState(
15037                "write_compressed_chunk requires a filtered dataset (no slot for \
15038                 a compressed size or filter mask on an unfiltered chunk index)"
15039                    .into(),
15040            ));
15041        }
15042        self.record_ea_chunk(index, chunk_idx, compressed_data, filter_mask)
15043    }
15044
15045    /// Extend the dimensions of a chunked dataset.
15046    pub fn extend_dataset(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15047        let ds = self.ds(index);
15048        let _op = ds.op.lock();
15049        self.extend_dataset_inner(index, new_dims)
15050    }
15051
15052    /// [`Self::extend_dataset`] body; the caller holds the dataset's op lock
15053    /// or the writer exclusively.
15054    pub(crate) fn extend_dataset_inner(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15055        let ds = self.ds(index);
15056        let mut m = ds.lock();
15057        if !m.is_chunked() {
15058            return Err(crate::io::IoError::InvalidState(
15059                "can only extend chunked datasets".into(),
15060            ));
15061        }
15062        if new_dims.len() != m.dataspace.dims.len() {
15063            return Err(crate::io::IoError::InvalidState(format!(
15064                "extend_dataset rank mismatch: dataset has {} dimensions, got {}",
15065                m.dataspace.dims.len(),
15066                new_dims.len()
15067            )));
15068        }
15069        // The chunk index and append buffers assume the logical size only
15070        // grows; shrinking below already-written data desynchronizes them.
15071        for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15072            if new < cur {
15073                return Err(crate::io::IoError::InvalidState(format!(
15074                    "extend_dataset cannot shrink dimension {d} from {cur} to {new}"
15075                )));
15076            }
15077            // An absent maximum shape means the shape is fixed (libhdf5
15078            // defaults maxdims to dims at creation), so any growth exceeds it.
15079            match m.dataspace.max_dims {
15080                Some(ref max) if new > max[d] => {
15081                    return Err(crate::io::IoError::InvalidState(format!(
15082                        "extend_dataset dimension {d} ({new}) exceeds the maximum {}",
15083                        max[d]
15084                    )));
15085                }
15086                None if new > cur => {
15087                    return Err(crate::io::IoError::InvalidState(format!(
15088                        "extend_dataset dimension {d} ({new}) exceeds the maximum {cur}: \
15089                         a dataset without a stored maximum shape is fixed at its extent"
15090                    )));
15091                }
15092                _ => {}
15093            }
15094        }
15095        if m.dataspace.dims != new_dims {
15096            m.dataspace.dims = new_dims.to_vec();
15097            m.extent_dirty = true;
15098        }
15099        Ok(())
15100    }
15101
15102    /// Set the logical extent of a chunked dataset, growing **or shrinking**
15103    /// any dimension (unlike [`extend_dataset`](Self::extend_dataset), which
15104    /// only grows).
15105    ///
15106    /// A shrink prunes the stored chunks the way libhdf5's
15107    /// `H5D__chunk_prune_by_extent` (H5Dchunk.c) does: a chunk entirely
15108    /// beyond the new extent leaves the chunk index and its block is freed
15109    /// for reuse (kept under SWMR, where a live reader may still hold its
15110    /// address — the rule `H5Dearray.c` applies in `idx_remove`), and a
15111    /// chunk the new extent cuts through has its out-of-extent region
15112    /// overwritten with the fill value, so growing the extent back exposes
15113    /// fill values rather than the stale data.
15114    pub fn set_dataset_extent(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15115        let ds = self.ds(index);
15116        let _op = ds.op.lock();
15117        let old_dims = {
15118            let m = ds.lock();
15119            if !m.is_chunked() {
15120                return Err(crate::io::IoError::InvalidState(
15121                    "can only set the extent of chunked datasets".into(),
15122                ));
15123            }
15124            if new_dims.len() != m.dataspace.dims.len() {
15125                return Err(crate::io::IoError::InvalidState(format!(
15126                    "set_extent rank mismatch: dataset has {} dimensions, got {}",
15127                    m.dataspace.dims.len(),
15128                    new_dims.len()
15129                )));
15130            }
15131            // A shrink can cut into buffered rows, whose recorded base would
15132            // then point past the extent; refuse rather than reconcile.
15133            if m.append.is_some() {
15134                return Err(crate::io::IoError::InvalidState(
15135                    "set_extent cannot run while the dataset has buffered appends; \
15136                     flush them first"
15137                        .into(),
15138                ));
15139            }
15140            // An absent maximum shape means the shape is fixed (libhdf5
15141            // defaults maxdims to dims at creation), so growth is bounded by
15142            // the extent.
15143            match m.dataspace.max_dims {
15144                Some(ref max) => {
15145                    for (d, (&new, &mx)) in new_dims.iter().zip(max).enumerate() {
15146                        if new > mx {
15147                            return Err(crate::io::IoError::InvalidState(format!(
15148                                "set_extent dimension {d} ({new}) exceeds the maximum {mx}"
15149                            )));
15150                        }
15151                    }
15152                }
15153                None => {
15154                    for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15155                        if new > cur {
15156                            return Err(crate::io::IoError::InvalidState(format!(
15157                                "set_extent dimension {d} ({new}) exceeds the maximum {cur}: \
15158                                 a dataset without a stored maximum shape is fixed at its extent"
15159                            )));
15160                        }
15161                    }
15162                }
15163            }
15164            m.dataspace.dims.clone()
15165        };
15166        // A shrink strands chunks; prune them (and refill the straddlers)
15167        // *before* the dims update — chunk addressing uses the
15168        // maximum-extent grid, which the update does not change, and the
15169        // helpers re-lock the slot themselves.
15170        if new_dims.iter().zip(&old_dims).any(|(&n, &o)| n < o) {
15171            self.prune_chunks_beyond(index, new_dims)?;
15172        }
15173        let mut m = ds.lock();
15174        if m.dataspace.dims != new_dims {
15175            m.dataspace.dims = new_dims.to_vec();
15176            m.extent_dirty = true;
15177        }
15178        Ok(())
15179    }
15180
15181    /// Remove and refill the chunks a shrink to `new_dims` strands — the
15182    /// libhdf5 `H5D__chunk_prune_by_extent` behavior. A chunk entirely
15183    /// beyond the new extent leaves the index and its block is freed (kept
15184    /// under SWMR, where a live reader may still hold its address); a chunk
15185    /// the extent cuts through gets its out-of-extent region refilled with
15186    /// the fill value, so a later regrow reads fill, not stale elements.
15187    ///
15188    /// Runs *before* the dims update: the index grid chunks are addressed in
15189    /// comes from the maximum extent, which a shrink never changes, so every
15190    /// stored entry still resolves. The caller holds the dataset's op lock.
15191    fn prune_chunks_beyond(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15192        let geo = self.chunk_geometry(index)?;
15193        // A vlen dataset's elements are global-heap IDs: the pruned chunks
15194        // still reference live heap objects, so the walkers read each dead
15195        // chunk's bytes before freeing its block and the heap objects are
15196        // released here — otherwise every shrink strands its strings in the
15197        // file. `release_vlen_references` is a SWMR no-op, so the reads are
15198        // skipped under SWMR too.
15199        let collect_refs = !self.swmr_active && {
15200            let ds = self.ds(index);
15201            let m = ds.lock();
15202            matches!(
15203                m.datatype,
15204                DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
15205            )
15206        };
15207        let (straddlers, dead_refs) = match geo.kind {
15208            ChunkIndexKind::ExtensibleArray => {
15209                self.prune_ea_chunks(index, &geo, new_dims, collect_refs)?
15210            }
15211            ChunkIndexKind::FixedArray => {
15212                self.prune_fa_chunks(index, &geo, new_dims, collect_refs)?
15213            }
15214            ChunkIndexKind::BtreeV2 => {
15215                self.prune_bt2_chunks(index, &geo, new_dims, collect_refs)?
15216            }
15217            // Removing a chunk from the implicit index is
15218            // `H5D__none_idx_remove`: a no-op, because the chunk's space is
15219            // the dataset's space and stays allocated either way. Only the
15220            // straddlers matter, and they are refilled by the caller.
15221            ChunkIndexKind::Implicit => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15222            // A single-chunk index has no per-chunk remove either — its one
15223            // chunk's address lives in the layout message, not an index
15224            // structure, and stays exactly where it is; a shrink only ever
15225            // straddles that one chunk (`H5D__single_idx_remove` is likewise
15226            // a no-op).
15227            ChunkIndexKind::SingleChunk => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15228            ChunkIndexKind::BtreeV1 => {
15229                self.prune_btree_v1_chunks(index, &geo, new_dims, collect_refs)?
15230            }
15231        };
15232        if !dead_refs.is_empty() {
15233            self.release_vlen_references(&dead_refs)?;
15234        }
15235        // Whole-chunk read-modify-write per straddler: an unfiltered chunk
15236        // rewrites in place, a filtered one re-places through `place_chunk`.
15237        let chunk_bytes = geo.chunk_bytes() as usize;
15238        for coords in straddlers {
15239            let Some(mut data) = self.read_chunk_at_coords(index, &coords)? else {
15240                continue;
15241            };
15242            let fill = self.new_chunk_buffer(index, chunk_bytes);
15243            let replaced = refill_chunk_beyond_extent(
15244                &mut data,
15245                &fill,
15246                &coords,
15247                &geo.chunk_dims,
15248                new_dims,
15249                geo.element_size as usize,
15250            );
15251            // Release before the write-back: a filtered straddler re-places
15252            // its block, and freed heap space must be visible to that
15253            // allocation (free-before-alloc, as everywhere else).
15254            if collect_refs && !replaced.is_empty() {
15255                self.release_vlen_references(&replaced)?;
15256            }
15257            self.write_chunk_at_coords(index, &coords, &data)?;
15258        }
15259        Ok(())
15260    }
15261
15262    /// Extensible-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15263    /// walk every slot the array has ever set, free and clear the entries of
15264    /// chunks entirely beyond `new_dims`, and return the grid coordinates of
15265    /// the chunks that straddle it, plus — when `collect_refs` — the dead
15266    /// chunks' element bytes so the caller can release their heap objects.
15267    fn prune_ea_chunks(
15268        &self,
15269        index: usize,
15270        geo: &ChunkGeometry,
15271        new_dims: &[u64],
15272        collect_refs: bool,
15273    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15274        let ds = self.ds(index);
15275        // One slot guard for the whole walk, the `record_ea_chunk` pattern:
15276        // `self.handle`/`self.allocator`/`self.ctx` are disjoint fields.
15277        let mut m = ds.lock();
15278        let is_filtered = m.filter_pipeline.is_some();
15279        let pipeline = m.filter_pipeline.clone();
15280        let chunk_bytes = geo.chunk_bytes();
15281        let (ea_geo, max_nelmts_bits, chunk_size_len, max_idx) = {
15282            let c = m.chunked.as_ref().unwrap();
15283            let p = &c.earray_params;
15284            (
15285                EaGeometry::new(
15286                    p.idx_blk_elmts,
15287                    p.data_blk_min_elmts,
15288                    p.sup_blk_min_data_ptrs,
15289                    p.max_nelmts_bits,
15290                    p.max_dblk_page_nelmts_bits,
15291                )?,
15292                p.max_nelmts_bits,
15293                c.chunk_size_len,
15294                c.ea_header.max_idx_set,
15295            )
15296        };
15297
15298        let mut straddlers = Vec::new();
15299        let mut dead_refs = Vec::new();
15300
15301        // The decoded data block the walk is currently inside, written back
15302        // when the walk leaves it (or ends) having cleared an entry.
15303        enum Dblk {
15304            Unfiltered(ExtensibleArrayDataBlock),
15305            Filtered(FilteredDataBlock),
15306        }
15307        let mut cache: Option<(u64, Dblk, bool)> = None;
15308        let flush = |cache: &mut Option<(u64, Dblk, bool)>| -> IoResult<()> {
15309            if let Some((addr, blk, dirty)) = cache.take() {
15310                if dirty {
15311                    let enc = match &blk {
15312                        Dblk::Unfiltered(d) => d.encode(&self.ctx, max_nelmts_bits),
15313                        Dblk::Filtered(d) => d.encode(&self.ctx, max_nelmts_bits, chunk_size_len),
15314                    };
15315                    self.handle.write_at(addr, &enc)?;
15316                }
15317            }
15318            Ok(())
15319        };
15320        // Consecutive slots resolve through the same super block, so keep
15321        // the last decode. Super blocks are only read here — clearing a
15322        // data-block element never moves the block — so it never dirties.
15323        let mut sblk_cache: Option<(usize, ExtensibleArraySuperBlock)> = None;
15324
15325        let mut slot = 0u64;
15326        while slot < max_idx {
15327            let coords = crate::io::chunk_grid::coords_of(
15328                &geo.dims,
15329                geo.max_dims.as_deref(),
15330                &geo.chunk_dims,
15331                slot,
15332            )?;
15333            if !chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
15334                if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15335                    straddlers.push(coords);
15336                }
15337                slot += 1;
15338                continue;
15339            }
15340            match ea_geo.locate(slot)? {
15341                EaLoc::Index { elem } => {
15342                    let c = m.chunked.as_mut().unwrap();
15343                    if is_filtered {
15344                        let fiblk = c.filt_iblk.as_mut().unwrap();
15345                        let e = fiblk.elements[elem];
15346                        if e.addr != UNDEF_ADDR {
15347                            if collect_refs {
15348                                if let Some(bytes) = self.read_chunk_block(
15349                                    pipeline.as_ref(),
15350                                    e.addr,
15351                                    e.nbytes,
15352                                    e.filter_mask,
15353                                )? {
15354                                    dead_refs.extend_from_slice(&bytes);
15355                                }
15356                            }
15357                            if !self.swmr_active {
15358                                self.allocator
15359                                    .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
15360                            }
15361                            fiblk.elements[elem] = FilteredChunkEntry {
15362                                addr: UNDEF_ADDR,
15363                                nbytes: 0,
15364                                filter_mask: 0,
15365                            };
15366                        }
15367                    } else {
15368                        let a = c.ea_iblk.elements[elem];
15369                        if a != UNDEF_ADDR {
15370                            if collect_refs {
15371                                if let Some(bytes) =
15372                                    self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
15373                                {
15374                                    dead_refs.extend_from_slice(&bytes);
15375                                }
15376                            }
15377                            if !self.swmr_active {
15378                                self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
15379                            }
15380                            c.ea_iblk.elements[elem] = UNDEF_ADDR;
15381                        }
15382                    }
15383                    slot += 1;
15384                }
15385                EaLoc::Dblk(l) => {
15386                    if l.paged {
15387                        return Err(crate::io::IoError::InvalidState(format!(
15388                            "chunk index {slot} lives in a paged extensible-array \
15389                             data block, which is not yet supported"
15390                        )));
15391                    }
15392                    let dblk_start = slot - l.offset_in_dblk;
15393                    let dblk_end = dblk_start + l.dblk_nelmts;
15394                    // Resolve the data block's address; an undefined super or
15395                    // data block means nothing in its whole element range was
15396                    // ever written, so the walk skips the range.
15397                    let dblk_addr = {
15398                        let c = m.chunked.as_ref().unwrap();
15399                        match l.path {
15400                            EaDblkPath::Direct { idx } => {
15401                                if is_filtered {
15402                                    c.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
15403                                } else {
15404                                    c.ea_iblk.dblk_addrs[idx]
15405                                }
15406                            }
15407                            EaDblkPath::ViaSblk {
15408                                sblk_off,
15409                                local_dblk,
15410                                ndblks_in_sblk,
15411                                ..
15412                            } => {
15413                                let sblk_addr = if is_filtered {
15414                                    c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
15415                                } else {
15416                                    c.ea_iblk.sblk_addrs[sblk_off]
15417                                };
15418                                if sblk_addr == UNDEF_ADDR {
15419                                    UNDEF_ADDR
15420                                } else {
15421                                    if sblk_cache.as_ref().map(|&(o, _)| o) != Some(sblk_off) {
15422                                        let buf = self.handle.read_at_most(sblk_addr, 65536)?;
15423                                        let sb = ExtensibleArraySuperBlock::decode(
15424                                            &buf,
15425                                            &self.ctx,
15426                                            max_nelmts_bits,
15427                                            ndblks_in_sblk,
15428                                            0,
15429                                        )?;
15430                                        sblk_cache = Some((sblk_off, sb));
15431                                    }
15432                                    sblk_cache.as_ref().unwrap().1.dblk_addrs[local_dblk]
15433                                }
15434                            }
15435                        }
15436                    };
15437                    if dblk_addr == UNDEF_ADDR {
15438                        slot = dblk_end;
15439                        continue;
15440                    }
15441                    if cache.as_ref().map(|&(a, _, _)| a) != Some(dblk_addr) {
15442                        flush(&mut cache)?;
15443                        let buf = self.handle.read_at_most(dblk_addr, 65536)?;
15444                        let blk = if is_filtered {
15445                            Dblk::Filtered(FilteredDataBlock::decode(
15446                                &buf,
15447                                &self.ctx,
15448                                max_nelmts_bits,
15449                                l.dblk_nelmts as usize,
15450                                chunk_size_len,
15451                            )?)
15452                        } else {
15453                            Dblk::Unfiltered(ExtensibleArrayDataBlock::decode(
15454                                &buf,
15455                                &self.ctx,
15456                                max_nelmts_bits,
15457                                l.dblk_nelmts as usize,
15458                            )?)
15459                        };
15460                        cache = Some((dblk_addr, blk, false));
15461                    }
15462                    let (_, blk, dirty) = cache.as_mut().unwrap();
15463                    match blk {
15464                        Dblk::Filtered(d) => {
15465                            let e = d.elements[l.offset_in_dblk as usize];
15466                            if e.addr != UNDEF_ADDR {
15467                                if collect_refs {
15468                                    if let Some(bytes) = self.read_chunk_block(
15469                                        pipeline.as_ref(),
15470                                        e.addr,
15471                                        e.nbytes,
15472                                        e.filter_mask,
15473                                    )? {
15474                                        dead_refs.extend_from_slice(&bytes);
15475                                    }
15476                                }
15477                                if !self.swmr_active {
15478                                    self.allocator
15479                                        .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
15480                                }
15481                                d.elements[l.offset_in_dblk as usize] = FilteredChunkEntry {
15482                                    addr: UNDEF_ADDR,
15483                                    nbytes: 0,
15484                                    filter_mask: 0,
15485                                };
15486                                *dirty = true;
15487                            }
15488                        }
15489                        Dblk::Unfiltered(d) => {
15490                            let a = d.elements[l.offset_in_dblk as usize];
15491                            if a != UNDEF_ADDR {
15492                                if collect_refs {
15493                                    if let Some(bytes) =
15494                                        self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
15495                                    {
15496                                        dead_refs.extend_from_slice(&bytes);
15497                                    }
15498                                }
15499                                if !self.swmr_active {
15500                                    self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
15501                                }
15502                                d.elements[l.offset_in_dblk as usize] = UNDEF_ADDR;
15503                                *dirty = true;
15504                            }
15505                        }
15506                    }
15507                    slot += 1;
15508                }
15509            }
15510        }
15511        flush(&mut cache)?;
15512        Ok((straddlers, dead_refs))
15513    }
15514
15515    /// Fixed-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15516    /// the whole element array is in memory and flushed at close, so
15517    /// clearing an entry is pure bookkeeping.
15518    fn prune_fa_chunks(
15519        &self,
15520        index: usize,
15521        geo: &ChunkGeometry,
15522        new_dims: &[u64],
15523        collect_refs: bool,
15524    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15525        let ds = self.ds(index);
15526        let mut m = ds.lock();
15527        let is_filtered = m.filter_pipeline.is_some();
15528        let pipeline = m.filter_pipeline.clone();
15529        let chunk_bytes = geo.chunk_bytes();
15530        let mut straddlers = Vec::new();
15531        let mut dead_refs = Vec::new();
15532        let fa = m.fixed_array.as_mut().unwrap();
15533        let nslots = if is_filtered {
15534            fa.fa_dblk.filtered_elements.len()
15535        } else {
15536            fa.fa_dblk.elements.len()
15537        };
15538        for lidx in 0..nslots {
15539            let (addr, stored, mask) = if is_filtered {
15540                let e = &fa.fa_dblk.filtered_elements[lidx];
15541                (e.address, e.chunk_size, e.filter_mask)
15542            } else {
15543                (fa.fa_dblk.elements[lidx], chunk_bytes, 0)
15544            };
15545            if addr == UNDEF_ADDR {
15546                continue;
15547            }
15548            let coords = crate::io::chunk_grid::coords_of(
15549                &geo.dims,
15550                geo.max_dims.as_deref(),
15551                &geo.chunk_dims,
15552                lidx as u64,
15553            )?;
15554            if chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
15555                if collect_refs {
15556                    if let Some(bytes) =
15557                        self.read_chunk_block(pipeline.as_ref(), addr, stored, mask)?
15558                    {
15559                        dead_refs.extend_from_slice(&bytes);
15560                    }
15561                }
15562                if !self.swmr_active {
15563                    self.allocator.free(addr, stored, FreeSpaceClass::RawData);
15564                }
15565                if is_filtered {
15566                    fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
15567                        address: UNDEF_ADDR,
15568                        chunk_size: 0,
15569                        filter_mask: 0,
15570                    };
15571                } else {
15572                    fa.fa_dblk.elements[lidx] = UNDEF_ADDR;
15573                }
15574            } else if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15575                straddlers.push(coords);
15576            }
15577        }
15578        Ok((straddlers, dead_refs))
15579    }
15580
15581    /// Implicit half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15582    /// the grid coordinates of the chunks a shrink to `new_dims` cuts
15583    /// through. Nothing is freed or cleared — this index has no per-chunk
15584    /// state to clear and no per-chunk block to free — so the chunks wholly
15585    /// beyond the extent keep their bytes, exactly as `H5D__none_idx_remove`
15586    /// leaves them. That also means their elements stay reachable, so a
15587    /// variable-length dataset's heap objects must *not* be released here.
15588    fn implicit_straddlers(
15589        &self,
15590        geo: &ChunkGeometry,
15591        new_dims: &[u64],
15592    ) -> IoResult<Vec<Vec<u64>>> {
15593        let mut nchunks: u64 = 1;
15594        for g in
15595            crate::io::chunk_grid::index_grid(&geo.dims, geo.max_dims.as_deref(), &geo.chunk_dims)?
15596        {
15597            nchunks = nchunks.checked_mul(g).ok_or_else(|| {
15598                crate::io::IoError::InvalidState("chunk count overflows u64".into())
15599            })?;
15600        }
15601        let mut straddlers = Vec::new();
15602        for lidx in 0..nchunks {
15603            let coords = crate::io::chunk_grid::coords_of(
15604                &geo.dims,
15605                geo.max_dims.as_deref(),
15606                &geo.chunk_dims,
15607                lidx,
15608            )?;
15609            if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
15610                straddlers.push(coords);
15611            }
15612        }
15613        Ok(straddlers)
15614    }
15615
15616    /// V2-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15617    /// drop the records of chunks beyond the extent — the next flush
15618    /// re-serializes the smaller tree over the node pool and releases the
15619    /// surplus node blocks.
15620    fn prune_bt2_chunks(
15621        &self,
15622        index: usize,
15623        geo: &ChunkGeometry,
15624        new_dims: &[u64],
15625        collect_refs: bool,
15626    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15627        let ds = self.ds(index);
15628        let mut m = ds.lock();
15629        let pipeline = m.filter_pipeline.clone();
15630        let chunk_bytes = geo.chunk_bytes();
15631        let swmr = self.swmr_active;
15632        let mut straddlers = Vec::new();
15633        let mut dead_refs = Vec::new();
15634        let bt2 = m.btree_v2.as_mut().unwrap();
15635        if bt2.index.filtered {
15636            let records = std::mem::take(&mut bt2.index.filtered_records);
15637            let mut kept = Vec::with_capacity(records.len());
15638            for r in records {
15639                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15640                    if collect_refs {
15641                        if let Some(bytes) = self.read_chunk_block(
15642                            pipeline.as_ref(),
15643                            r.chunk_address,
15644                            r.chunk_size,
15645                            r.filter_mask,
15646                        )? {
15647                            dead_refs.extend_from_slice(&bytes);
15648                        }
15649                    }
15650                    if !swmr {
15651                        self.allocator
15652                            .free(r.chunk_address, r.chunk_size, FreeSpaceClass::RawData);
15653                    }
15654                } else {
15655                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15656                        straddlers.push(r.scaled_offsets.clone());
15657                    }
15658                    kept.push(r);
15659                }
15660            }
15661            bt2.index.filtered_records = kept;
15662        } else {
15663            let records = std::mem::take(&mut bt2.index.records);
15664            let mut kept = Vec::with_capacity(records.len());
15665            for r in records {
15666                if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15667                    if collect_refs {
15668                        if let Some(bytes) = self.read_chunk_block(
15669                            pipeline.as_ref(),
15670                            r.chunk_address,
15671                            chunk_bytes,
15672                            0,
15673                        )? {
15674                            dead_refs.extend_from_slice(&bytes);
15675                        }
15676                    }
15677                    if !swmr {
15678                        self.allocator
15679                            .free(r.chunk_address, chunk_bytes, FreeSpaceClass::RawData);
15680                    }
15681                } else {
15682                    if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
15683                        straddlers.push(r.scaled_offsets.clone());
15684                    }
15685                    kept.push(r);
15686                }
15687            }
15688            bt2.index.records = kept;
15689        }
15690        Ok((straddlers, dead_refs))
15691    }
15692
15693    /// Version-1-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15694    /// drop the records of chunks beyond the extent — the next flush
15695    /// re-serializes the smaller tree over the node pool and releases the
15696    /// surplus node blocks.
15697    fn prune_btree_v1_chunks(
15698        &self,
15699        index: usize,
15700        geo: &ChunkGeometry,
15701        new_dims: &[u64],
15702        collect_refs: bool,
15703    ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15704        let ds = self.ds(index);
15705        let mut m = ds.lock();
15706        let pipeline = m.filter_pipeline.clone();
15707        let swmr = self.swmr_active;
15708        let mut straddlers = Vec::new();
15709        let mut dead_refs = Vec::new();
15710        let bt1 = m.btree_v1.as_mut().unwrap();
15711        let records = std::mem::take(&mut bt1.records);
15712        let mut kept = Vec::with_capacity(records.len());
15713        for r in records {
15714            if chunk_outside_extent(&r.scaled, &geo.chunk_dims, new_dims) {
15715                if collect_refs {
15716                    if let Some(bytes) = self.read_chunk_block(
15717                        pipeline.as_ref(),
15718                        r.address,
15719                        r.nbytes as u64,
15720                        r.filter_mask,
15721                    )? {
15722                        dead_refs.extend_from_slice(&bytes);
15723                    }
15724                }
15725                if !swmr {
15726                    self.allocator
15727                        .free(r.address, r.nbytes as u64, FreeSpaceClass::RawData);
15728                }
15729            } else {
15730                if chunk_straddles_extent(&r.scaled, &geo.chunk_dims, new_dims) {
15731                    straddlers.push(r.scaled.clone());
15732                }
15733                kept.push(r);
15734            }
15735        }
15736        m.btree_v1.as_mut().unwrap().records = kept;
15737        Ok((straddlers, dead_refs))
15738    }
15739
15740    /// Flush a chunked dataset's index structures to disk (durable).
15741    ///
15742    /// Writes the index blocks and issues an `fdatasync` so the data is
15743    /// durable — the guarantee SWMR readers and standalone callers rely on.
15744    pub fn flush_dataset(&self, index: usize) -> IoResult<()> {
15745        let ds = self.ds(index);
15746        let _op = ds.op.lock();
15747        self.flush_dataset_synced(index, true)
15748    }
15749
15750    /// Flush a chunked dataset's index structures, syncing only if `sync`.
15751    ///
15752    /// `finalize` threads its own durability choice here so that a
15753    /// [`close_no_sync`](Self::close_no_sync) skips this per-dataset
15754    /// `sync_data` too — otherwise gating only the final `sync_all` would
15755    /// leave one `fdatasync` per indexed dataset and defeat the fast close.
15756    fn flush_dataset_synced(&self, index: usize, sync: bool) -> IoResult<()> {
15757        // Hold one slot guard for the whole method; `self.handle`/`self.ctx`/
15758        // `self.allocator` below touch disjoint fields.
15759        let ds = self.ds(index);
15760        let mut m = ds.lock();
15761
15762        // EA-indexed dataset
15763        if let Some(ref chunked) = m.chunked {
15764            if let Some(ref fiblk) = chunked.filt_iblk {
15765                // Filtered EA
15766                let iblk_encoded = fiblk.encode(&self.ctx, chunked.chunk_size_len);
15767                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
15768            } else {
15769                // Unfiltered EA
15770                let iblk_encoded = chunked.ea_iblk.encode(&self.ctx);
15771                self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
15772            }
15773            let hdr_encoded = chunked.ea_header.encode(&self.ctx);
15774            self.handle.write_at(chunked.ea_header_addr, &hdr_encoded)?;
15775            if sync {
15776                self.handle.sync_data()?;
15777            }
15778            return Ok(());
15779        }
15780
15781        // Fixed-array-indexed dataset
15782        if let Some(ref fa) = m.fixed_array {
15783            let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa.fa_header, &fa.fa_dblk);
15784            self.handle.write_at(fa.fa_dblk_addr, &dblk_encoded)?;
15785            let hdr_encoded = fa.fa_header.encode(&self.ctx);
15786            self.handle.write_at(fa.fa_header_addr, &hdr_encoded)?;
15787            if sync {
15788                self.handle.sync_data()?;
15789            }
15790            return Ok(());
15791        }
15792
15793        // BT2-indexed dataset
15794        if let Some(ref bt2) = m.btree_v2 {
15795            // Bulk-load the index into fixed-size nodes and lay them over the
15796            // dataset's block pool. Because every node is the same size, the
15797            // blocks already on disk are reused in place and only the shortfall
15798            // is allocated — the pool is the single owner of these addresses,
15799            // so no flush leaves a block behind. The addresses a reader already
15800            // holds stay valid, which is also what SWMR needs.
15801            let tree = bt2.index.build_tree(&self.ctx);
15802            let mut node_addrs = bt2.node_addrs.clone();
15803            while node_addrs.len() < tree.nodes.len() {
15804                node_addrs.push(
15805                    self.allocator
15806                        .allocate(tree.node_size as u64, FreeSpaceClass::Metadata),
15807                );
15808            }
15809            // A tree with fewer nodes than last flush releases the surplus
15810            // rather than leaving it recorded and unreachable, so the pool is
15811            // exactly one block per node whichever way the count moved. Under
15812            // SWMR a reader may still hold a header naming those blocks, so
15813            // keep them out of the free list — the same rule `place_chunk`
15814            // applies to a relocated chunk.
15815            for addr in node_addrs.split_off(tree.nodes.len()) {
15816                if !self.swmr_active {
15817                    self.allocator
15818                        .free(addr, tree.node_size as u64, FreeSpaceClass::Metadata);
15819                }
15820            }
15821
15822            for (image, &addr) in tree.encode(&self.ctx, &node_addrs).iter().zip(&node_addrs) {
15823                self.handle.write_at(addr, image)?;
15824            }
15825
15826            // The root is the last node the bulk load emits.
15827            let root_addr = match tree.nodes.len() {
15828                0 => UNDEF_ADDR,
15829                n => node_addrs[n - 1],
15830            };
15831            let hdr_encoded = tree.header(root_addr).encode(&self.ctx);
15832            self.handle.write_at(bt2.bt2_header_addr, &hdr_encoded)?;
15833
15834            m.btree_v2.as_mut().unwrap().node_addrs = node_addrs;
15835
15836            if sync {
15837                self.handle.sync_data()?;
15838            }
15839            return Ok(());
15840        }
15841
15842        // Version-1-B-tree-indexed dataset
15843        if let Some(ref bt1) = m.btree_v1 {
15844            // Bulk-loaded over the same block pool the v2 B-tree above uses,
15845            // and for the same reason: every node of a v1 tree is the width
15846            // its "K" value gives, so a block stays usable however the tree
15847            // reshapes, and only the shortfall is ever allocated.
15848            let element_size = m.datatype.element_size() as u64;
15849            let tree = bt1.build_tree(element_size, self.ctx.sizeof_addr as usize);
15850            let node_size = tree.node_size() as u64;
15851            let mut node_addrs = bt1.node_addrs.clone();
15852            while node_addrs.len() < tree.node_count() {
15853                node_addrs.push(self.allocator.allocate(node_size, FreeSpaceClass::Metadata));
15854            }
15855            // A tree with fewer nodes than last flush releases the surplus
15856            // straight away, where the v2 B-tree has to keep it out of the
15857            // free list for a live SWMR reader: this index lives only in a
15858            // classic file, which `start_swmr` refuses outright (and upstream
15859            // says the same in `H5D_COPS_BTREE`).
15860            for addr in node_addrs.split_off(tree.node_count()) {
15861                self.allocator
15862                    .free(addr, node_size, FreeSpaceClass::Metadata);
15863            }
15864            for (image, &addr) in tree.encode(&node_addrs)?.iter().zip(&node_addrs) {
15865                self.handle.write_at(addr, image)?;
15866            }
15867            // The root is the last node the bulk load emits, and is undefined
15868            // while the dataset has no chunks — what the version-3 data
15869            // layout message then carries, exactly as libhdf5 leaves it.
15870            let root_addr = tree.root_address(&node_addrs);
15871            let bt1 = m.btree_v1.as_mut().unwrap();
15872            bt1.node_addrs = node_addrs;
15873            bt1.root_addr = root_addr;
15874
15875            if sync {
15876                self.handle.sync_data()?;
15877            }
15878            return Ok(());
15879        }
15880
15881        Ok(())
15882    }
15883
15884    /// Finalize and close the file.
15885    ///
15886    /// Writes the dataset object headers, root group object header, and
15887    /// superblock. After this call the file is a valid HDF5 file.
15888    pub fn close(mut self) -> IoResult<()> {
15889        // Mark closed BEFORE finalizing: finalize writes external truth
15890        // (object headers + superblock) and must run exactly once. If we
15891        // finalized first and it failed, the `?` would return with `closed`
15892        // still false, and dropping `self` would re-run `finalize` a second
15893        // time over a half-written file (and print the "call close()" notice
15894        // the caller already heeded). Committing to the close path first makes
15895        // `Drop` (the only other finalize site) a no-op regardless of outcome,
15896        // so the error is reported exactly once via this `Result`.
15897        self.closed = true;
15898        self.finalize(true)
15899    }
15900
15901    /// Finalize and close the file without a final `fsync`.
15902    ///
15903    /// Identical to [`close`](Self::close) — the same object headers and
15904    /// superblock are written, so on return the file is a complete, valid HDF5
15905    /// file readable by any process — except that the trailing `sync_all`
15906    /// (fsync) is skipped. The bytes are handed to the OS but are not
15907    /// guaranteed durable against power loss or an OS crash until the OS
15908    /// flushes its page cache; a normal process exit or a same-machine reader
15909    /// sees the full file regardless.
15910    ///
15911    /// This trades durability for speed: `sync_all` typically dominates close
15912    /// latency, so bulk writers that do not need crash durability (the file can
15913    /// be regenerated) can use this to avoid that cost. Use [`close`](Self::close)
15914    /// when durability matters. `Drop` always finalizes durably, so a writer
15915    /// finalized this way must reach `close_no_sync` explicitly.
15916    pub fn close_no_sync(mut self) -> IoResult<()> {
15917        // Same close-once discipline as `close`: commit to the close path
15918        // before finalizing so `Drop` cannot re-run `finalize` on failure.
15919        self.closed = true;
15920        self.finalize(false)
15921    }
15922
15923    /// Provide mutable access to the underlying file handle.
15924    pub fn handle(&mut self) -> &mut FileHandle {
15925        &mut self.handle
15926    }
15927
15928    /// The superblock version this file will be written with.
15929    ///
15930    /// `H5F__super_init` takes the oldest version that can describe the file
15931    /// and raises it to the one the file's library-version low bound implies:
15932    /// `super_vers = MAX(super_vers, HDF5_superblock_ver_bounds[low_bound])`,
15933    /// with the bounds table reading 0, 2, 3, 3, 3, 3, 3 for EARLIEST, V18,
15934    /// V110, V112, V114, V200, LATEST (H5Fsuper.c:68, :1128-1154). A file
15935    /// created at `H5F_LIBVER_EARLIEST` takes that bound's entry directly
15936    /// ([`SuperblockVersion::Chosen`], and the classic branch below) — version
15937    /// 0, or version 2 when the file carries shared messages, whose master
15938    /// table needs the superblock extension only a version-2 superblock has
15939    /// (H5Fsuper.c:1135). For every other file the bound is read back from
15940    /// what this crate writes:
15941    ///
15942    /// * The floor is `H5F_LIBVER_V18`, hence version 2. Every group such a
15943    ///   file holds is a link-message group, which libhdf5 only writes at a
15944    ///   low bound of V18 or newer (`use_at_least_v18`, H5Gobj.c:179), and
15945    ///   every object header in it is version 2, which `H5O_obj_ver_bounds`
15946    ///   likewise puts at V18 (H5Oint.c:125). A version-0 superblock over
15947    ///   this content would claim a file libhdf5 1.6 can read, and no libhdf5
15948    ///   writes that combination.
15949    /// * A chunked dataset — extensible array, fixed array or version-2
15950    ///   B-tree, all reached through a version-4 or -5 data layout message —
15951    ///   reads back as V110 (`H5O_layout_ver_bounds`, H5Dlayout.c:44), hence
15952    ///   version 3.
15953    /// * SWMR writes version 3 outright (H5Fsuper.c:1129).
15954    ///
15955    /// A file whose caller *named* a bound skips the read-back and takes that
15956    /// bound's row directly, so `V18` stays at version 2 however its chunked
15957    /// datasets are indexed — which is what libhdf5 does, the layout version
15958    /// being no input to `H5F__super_init` at all.
15959    ///
15960    /// None of that applies to a reopened file. `H5F__super_read` validates
15961    /// the version it finds and never recomputes one, so the version written
15962    /// back is the version read — see [`SuperblockVersion`], which is also
15963    /// where the other half of that rule lives: the version floors the bound
15964    /// the appended structures are written at, which is why nothing this
15965    /// session adds can need a newer one.
15966    fn superblock_version_for(&self, flags: u8) -> u8 {
15967        let chosen = match self.superblock_version {
15968            SuperblockVersion::Existing(version) => return version,
15969            SuperblockVersion::Chosen(version) => version,
15970        };
15971        if self.is_legacy() {
15972            // A classic file keeps the version it was created at — 0, or 2
15973            // when its shared messages needed the extension. Nothing a session
15974            // can add reaches past that: its objects get symbol-table links,
15975            // its chunked datasets the version-1 B-tree behind a version-3
15976            // layout message, and the two features that would raise the bound
15977            // — SWMR and the 2.0 format — are refused where the caller asks
15978            // for them.
15979            return chosen;
15980        }
15981        let mut version = chosen
15982            .max(SUPERBLOCK_V2)
15983            .max(self.effective_libver().superblock_version());
15984        if self.swmr_active || flags & FLAG_SWMR_WRITE != 0 {
15985            version = version.max(SUPERBLOCK_V3);
15986        }
15987        version
15988    }
15989
15990    /// The low bound a modern file this writer *created* is effectively
15991    /// written at: the one the caller named, or — with none named — the one
15992    /// its content reads back as. A reopened file never reaches here; its
15993    /// superblock version is not derived from its content at all.
15994    ///
15995    /// The read-back is what `superblock_version_for` needs and the field
15996    /// alone cannot give: this crate's default file names no bound, and the
15997    /// generation it writes is not one bound but two rows (see the `libver`
15998    /// field). The floor is `V18`, the oldest bound under which libhdf5 writes
15999    /// link-message groups (`use_at_least_v18`, H5Gobj.c:179) and version-2
16000    /// object headers (`H5O_obj_ver_bounds`, H5Oint.c:125), which is all such
16001    /// a file holds; a v1.10 chunk index in it raises that to `V110`, the
16002    /// oldest bound whose `H5O_layout_ver_bounds` row reaches the version-4
16003    /// layout message that index is written behind.
16004    fn effective_libver(&self) -> LibverBound {
16005        self.libver.unwrap_or_else(|| {
16006            if self.has_v110_chunk_index() {
16007                LibverBound::V110
16008            } else {
16009                LibverBound::V18
16010            }
16011        })
16012    }
16013
16014    /// Whether any dataset still in the file is indexed by a v1.10 chunk
16015    /// index — the markers `build_dataset_header` turns into a version-4/5
16016    /// data layout message, and nothing else it can emit reaches that
16017    /// version.
16018    ///
16019    /// Not "is any dataset chunked": the version-1 B-tree is a chunk index
16020    /// that encodes as a *version-3* layout message, the version
16021    /// `H5O_layout_ver_bounds` gives the earliest bound, so a dataset using
16022    /// it asks nothing of the superblock.
16023    fn has_v110_chunk_index(&self) -> bool {
16024        self.dataset_refs().iter().any(|d| {
16025            let m = d.lock();
16026            !m.deleted
16027                && m.chunk_index_kind()
16028                    .is_some_and(|k| k != ChunkIndexKind::BtreeV1)
16029        })
16030    }
16031
16032    /// Write the superblock at offset 0 with the given flags.
16033    ///
16034    /// Requires that the root group has already been written (via `finalize`
16035    /// or `finalize_for_swmr`).
16036    pub fn write_superblock(&mut self, flags: u8) -> IoResult<()> {
16037        let root_addr = self
16038            .root_group_addr
16039            .ok_or_else(|| crate::io::IoError::InvalidState("root group not yet written".into()))?;
16040        // The userblock this file was opened with. `H5F__super_read` prefers
16041        // the located address over this field, but `H5Pget_userblock` reports
16042        // it, so a rewrite that zeroed it would hide the block from every
16043        // reader that asks for its size.
16044        let base = self.handle.base();
16045        // The end of file is the one address in the superblock measured from
16046        // the start of the *file* rather than from the base: `H5F__super_read`
16047        // sets the EOA to `stored_eof - base_addr` (H5Fsuper.c:635) and calls
16048        // the file truncated when `eof + base_addr < stored_eof` (:573). The
16049        // allocator counts in the based space, so the userblock is added back.
16050        let eof = self.allocator.eof() + base;
16051        let version = self.superblock_version_for(flags);
16052        // Which of the two images is written follows the version, not the
16053        // generation: a classic file carrying shared messages is a version-2
16054        // superblock over version-1 messages and symbol-table groups
16055        // (H5Fsuper.c:1135), and only the version-2/3 image has the extension
16056        // address that table is reached through. Below version 2 the file is
16057        // always a classic one — the other branch floors at 2.
16058        if let Some(legacy) = self.legacy.as_deref().filter(|_| version < SUPERBLOCK_V2) {
16059            // Re-emitted, not rebuilt: the "K" ranks, the userblock size and
16060            // the driver info address are recorded nowhere else in the file,
16061            // and every node width in it is derived from the ranks. Only the
16062            // three things this session can have changed are recomputed.
16063            let root_stab = self
16064                .symbol_tables
16065                .written
16066                .lock()
16067                .get(&LinkScope::Root)
16068                .copied();
16069            let mut sb = legacy.superblock.clone();
16070            sb.version = version;
16071            sb.file_consistency_flags = flags as u32;
16072            sb.end_of_file_address = eof;
16073            sb.root_symbol_table_entry.obj_header_addr = root_addr;
16074            // `H5G__stab_valid` (H5Groot.c) reads this pair back and compares
16075            // it against the root header's Symbol Table message, repairing the
16076            // superblock when they disagree. Writing the pair that message now
16077            // names is what keeps the file from needing that repair. A root
16078            // that keeps its links in messages has no such pair and no entry
16079            // in `written`, and gets `H5G_NOTHING_CACHED` — what libhdf5
16080            // writes for the same root.
16081            sb.root_symbol_table_entry.cache = match root_stab {
16082                Some(s) => SymbolTableCache::SymbolTable {
16083                    btree_addr: s.btree_addr,
16084                    heap_addr: s.heap_addr,
16085                },
16086                None => SymbolTableCache::Nothing,
16087            };
16088            self.handle.write_at(0, &sb.encode())?;
16089            return Ok(());
16090        }
16091        let sb = SuperblockV2V3 {
16092            version,
16093            sizeof_offsets: self.ctx.sizeof_addr,
16094            sizeof_lengths: self.ctx.sizeof_size,
16095            file_consistency_flags: flags,
16096            base_address: base,
16097            // Whatever `write_superblock_extension` put there, which is the
16098            // only place an extension is written.
16099            superblock_extension_address: self.extension.addr.lock().unwrap_or(UNDEF_ADDR),
16100            end_of_file_address: eof,
16101            root_group_object_header_address: root_addr,
16102        };
16103        self.handle.write_at(0, &sb.encode())?;
16104        Ok(())
16105    }
16106
16107    /// Re-write a dataset's object header in place (SWMR update).
16108    ///
16109    /// The header must have been previously written via `finalize_for_swmr`.
16110    /// Only the dataspace dimensions change; the encoded size must not exceed
16111    /// the originally allocated space.
16112    pub fn write_dataset_header_inplace(&mut self, index: usize) -> IoResult<()> {
16113        // Scope the slot guard: `build_dataset_header` re-locks the same slot.
16114        let (addr, original_size) = {
16115            let ds = self.ds(index);
16116            let m = ds.lock();
16117            // One block, because a finalize writes every header as one
16118            // chunk: an in-place rewrite has that block's room and no more.
16119            match m.obj_header_blocks.as_slice() {
16120                [(addr, size)] => (*addr, *size as usize),
16121                _ => {
16122                    return Err(crate::io::IoError::InvalidState(
16123                        "dataset header not yet written as a single chunk".into(),
16124                    ))
16125                }
16126            }
16127        };
16128
16129        let header = self.build_dataset_header(index)?;
16130        let nlink = self.object_link_count(HardLinkTarget::Dataset(index));
16131        let encoded =
16132            self.encode_header_at(&header, nlink, self.dataset_header_format(index), addr)?;
16133
16134        if encoded.len() > original_size {
16135            return Err(crate::io::IoError::InvalidState(format!(
16136                "dataset header grew from {} to {} bytes; cannot rewrite in place",
16137                original_size,
16138                encoded.len()
16139            )));
16140        }
16141
16142        // Pad to original size with zeros (the trailing zeros after the
16143        // checksum won't be parsed by readers since chunk0_data_size is fixed).
16144        let mut padded = encoded;
16145        padded.resize(original_size, 0);
16146
16147        self.handle.write_at(addr, &padded)?;
16148        // Only after the bytes are down: a failed write leaves the registry
16149        // describing the header the file still holds.
16150        self.ds(index).lock().header_written(nlink);
16151        Ok(())
16152    }
16153
16154    /// Perform a full finalize for SWMR mode.
16155    ///
16156    /// This writes all dataset object headers, the root group header, and the
16157    /// superblock with SWMR flags. After this call, the file is valid for
16158    /// SWMR readers. Subsequent writes use in-place updates.
16159    pub fn finalize_for_swmr(&mut self) -> IoResult<()> {
16160        self.reject_swmr()?;
16161        // 0. Flush all chunked dataset index structures.
16162        for i in 0..self.dataset_count() {
16163            let is_indexed = {
16164                let ds = self.ds(i);
16165                let m = ds.lock();
16166                !m.deleted && m.is_chunked()
16167            };
16168            if is_indexed {
16169                self.flush_dataset(i)?;
16170            }
16171        }
16172
16173        // 1. Allocate every object header (none for a dataset deleted before
16174        // start_swmr — its storage was freed at delete time). Same three
16175        // phases as the full finalize, and for the same reason: nothing a
16176        // header names can be laid out until every object has an address.
16177        let live: Vec<usize> = (0..self.dataset_count())
16178            .filter(|&i| !self.ds(i).lock().deleted)
16179            .collect();
16180        // Before any dataset header: a sharing dataset's header names the
16181        // committed type's address.
16182        self.write_committed_datatype_headers()?;
16183        let layout = self.allocate_object_headers(&live)?;
16184
16185        // 2. Build content against those addresses.
16186        self.prepare_dense_attributes(&live)?;
16187        self.prepare_link_storage()?;
16188        self.write_reference_values()?;
16189
16190        // 3. Write every object header.
16191        self.write_object_headers(&layout)?;
16192        // What SWMR alone needs to know afterwards: where each dataset's
16193        // header is published and how much room it has, which is what
16194        // `write_dataset_header_inplace` rewrites within.
16195        for &(i, addr, size) in &layout.datasets {
16196            let ds = self.ds(i);
16197            let mut m = ds.lock();
16198            m.obj_header_written_addr = Some(addr);
16199            m.obj_header_blocks = vec![(addr, size as u64)];
16200        }
16201        self.root_group_encoded_size = layout.root.1;
16202
16203        // 4. Write superblock with SWMR flags.
16204        self.write_superblock(FLAG_WRITE_ACCESS | FLAG_SWMR_WRITE)?;
16205        self.handle.set_eof(self.allocator.eof())?;
16206
16207        self.handle.sync_all()?;
16208        // Readers can now be following this file, so a chunk that moves must
16209        // leave its old block intact for whoever is still holding the previous
16210        // index (see `swmr_active`).
16211        self.swmr_active = true;
16212        Ok(())
16213    }
16214
16215    // ------------------------------------------------------------------
16216    // Internal helpers
16217    // ------------------------------------------------------------------
16218
16219    /// Flush every dataset's append buffer into the chunks it belongs to,
16220    /// through [`flush_append_buffer`](Self::flush_append_buffer): frames
16221    /// already in the chunk survive, and the rest of it reads back as the
16222    /// dataset's fill value (zeros when none is defined).
16223    fn flush_append_buffers(&mut self) -> IoResult<()> {
16224        for i in 0..self.dataset_count() {
16225            if self.ds(i).lock().deleted {
16226                continue;
16227            }
16228            self.flush_append_buffer(i)?;
16229        }
16230        Ok(())
16231    }
16232
16233    /// Write all object headers and the superblock, producing a complete,
16234    /// valid HDF5 file.
16235    ///
16236    /// `sync == true` issues a final `sync_all` (fsync) so the bytes are
16237    /// durable against power loss / OS crash before returning. `sync == false`
16238    /// skips that fsync: the file is still fully written to the OS and readable
16239    /// by any process, but durability is left to the OS page-cache flush. This
16240    /// is the only difference between [`close`](Self::close) (durable) and
16241    /// [`close_no_sync`](Self::close_no_sync) (fast).
16242    fn finalize(&mut self, sync: bool) -> IoResult<()> {
16243        // Flush any partial append buffers before finalizing
16244        self.flush_append_buffers()?;
16245
16246        // A SWMR session (`finalize_for_swmr` already ran, so
16247        // `root_group_addr` is `Some`) is closed by the same full finalize as
16248        // a fresh write: every object header is rebuilt at a fresh address and
16249        // the superblock is written with clean-close flags. A full rebuild —
16250        // rather than the in-place header rewrite used by the live
16251        // `SwmrWriter::flush` path — is required so any structural change made
16252        // after `start_swmr` is committed to the final file. A hard link, in
16253        // particular, both grows its target's header with an object
16254        // reference-count message and adds a `MSG_LINK` record to a group
16255        // header; an in-place rewrite cannot accommodate the grown header and
16256        // never re-emits group/root headers. The fall-through below already
16257        // handles datasets whose header was written by `finalize_for_swmr`
16258        // (`obj_header_written_addr.is_some()`).
16259
16260        // 0. Flush chunked dataset index structures (only modified datasets).
16261        for i in 0..self.dataset_count() {
16262            let ds = self.ds(i);
16263            {
16264                let m = ds.lock();
16265                if m.deleted {
16266                    continue;
16267                }
16268                if m.obj_header_written_addr.is_some() && !m.storage_dirty() {
16269                    continue;
16270                }
16271                let is_indexed = m.is_chunked();
16272                if !is_indexed {
16273                    continue;
16274                }
16275            }
16276            self.flush_dataset_synced(i, sync)?;
16277        }
16278
16279        // Every header block this finalize supersedes — a reopened root or
16280        // group header, a modified dataset's reopened header — is freed
16281        // before its replacement is allocated, so the rewrite reuses the
16282        // block instead of growing the file on every open/close cycle.
16283        // Never under SWMR: a live reader may be walking the old headers,
16284        // the same rule `release_vlen_references` and `place_chunk` follow.
16285        // Hard links can alias one header under several names; the set keeps
16286        // an aliased block from entering the free list twice.
16287        let mut freed_headers = std::collections::HashSet::new();
16288
16289        // 1. Plan. Which datasets get a header (deleted datasets get none —
16290        // their storage was already freed at delete time) is settled first,
16291        // because everything the next phases lay out is laid out only for the
16292        // headers this finalize actually rewrites; and every header block
16293        // those phases supersede is returned here, before the first
16294        // allocation, so a rewrite can land in it.
16295        let mut rewritten: Vec<usize> = Vec::new();
16296        // A finalize that lays the shared-message table out afresh reassigns
16297        // every heap ID in the file, so no existing header can keep its bytes:
16298        // the pointers in them name heap objects the new table does not have.
16299        let table_replaced = self.rebuilds_shared_messages();
16300        for i in 0..self.dataset_count() {
16301            // Before the slot guard: `object_link_count` re-locks every
16302            // dataset and group slot, this one included.
16303            let nlink = self.object_link_count(HardLinkTarget::Dataset(i));
16304            let ds = self.ds(i);
16305            let mut m = ds.lock();
16306            if m.deleted {
16307                continue;
16308            }
16309            if m.obj_header_written_addr.is_some() {
16310                // An existing dataset from append mode keeps its header — and
16311                // everything that header names — unless this session changed
16312                // what the header says.
16313                if !table_replaced && !m.header_stale_with(nlink) {
16314                    // Keep the original object header address for the root group link.
16315                    m.obj_header_addr = m.obj_header_written_addr.unwrap();
16316                    continue;
16317                }
16318                if !self.swmr_active && !m.obj_header_blocks.is_empty() {
16319                    let old = m.obj_header_written_addr.take().unwrap();
16320                    let blocks = std::mem::take(&mut m.obj_header_blocks);
16321                    if freed_headers.insert(old) {
16322                        for (addr, len) in blocks {
16323                            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16324                        }
16325                    }
16326                }
16327            }
16328            rewritten.push(i);
16329        }
16330        if !self.swmr_active {
16331            for gi in 0..self.group_count() {
16332                let grp = self.grp(gi);
16333                let mut g = grp.lock();
16334                if let Some(old) = g
16335                    .obj_header_written_addr
16336                    .take()
16337                    .filter(|_| !g.obj_header_blocks.is_empty())
16338                {
16339                    let blocks = std::mem::take(&mut g.obj_header_blocks);
16340                    if freed_headers.insert(old) {
16341                        for (addr, len) in blocks {
16342                            self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16343                        }
16344                    }
16345                }
16346            }
16347            let root_blocks = std::mem::take(&mut self.superseded_root_header);
16348            if root_blocks
16349                .first()
16350                .is_some_and(|&(addr, _)| freed_headers.insert(addr))
16351            {
16352                for (addr, len) in root_blocks {
16353                    self.allocator.free(addr, len, FreeSpaceClass::Metadata);
16354                }
16355            }
16356        }
16357
16358        // 2. Allocate. Committed datatype headers go down whole: a header of
16359        // theirs holds a datatype and a reference count, so it waits on
16360        // nothing, while a dataset sharing the type and the group naming it
16361        // both store its address. They are written before the shared-message
16362        // phase opens, so a committed type reaches the file as itself.
16363        self.write_committed_datatype_headers()?;
16364        self.begin_shared_message_layout();
16365        let layout = self.allocate_object_headers(&rewritten)?;
16366
16367        // 3. Build content, with every object header's address known. Dense
16368        // attribute storage holds the attribute messages themselves — an
16369        // object reference among them is a header address; dense links and
16370        // symbol tables name header addresses; a reference dataset's elements
16371        // are header addresses. Nothing here is a fixup: each is written once,
16372        // with the value the file keeps. The shared-message table comes last:
16373        // it counts the bodies the headers will hold, and the three above are
16374        // what settle them.
16375        self.prepare_dense_attributes(&rewritten)?;
16376        self.prepare_link_storage()?;
16377        self.write_reference_values()?;
16378        self.prepare_shared_messages(&rewritten)?;
16379        self.write_superblock_extension()?;
16380
16381        // 4. Write every object header over the block phase 2 reserved for it.
16382        self.write_object_headers(&layout)?;
16383
16384        // 5. Write superblock at offset 0.
16385        self.write_superblock(0)?;
16386
16387        // 6. End the file where its address space ends (`H5FD_truncate`, which
16388        // `H5F__dest` calls on every close). Allocated-but-unwritten space at
16389        // the end would otherwise leave the file shorter than the end-of-file
16390        // address the superblock just recorded, which libhdf5 reads as a
16391        // truncated file.
16392        self.handle.set_eof(self.allocator.eof())?;
16393
16394        // Durability is opt-in per call: `close` passes `true`, `close_no_sync`
16395        // passes `false`, and `Drop` passes `true` so an un-`close`d writer is
16396        // still finalized durably by default.
16397        if sync {
16398            self.handle.sync_all()?;
16399        }
16400        Ok(())
16401    }
16402
16403    /// Give every object header this finalize writes an address, before
16404    /// anything that names one is built.
16405    ///
16406    /// INVARIANT: from the moment this returns until the file is closed, every
16407    /// object in it has the object header address it will be found at. That is
16408    /// what lets the phase after this one say an address wherever the format
16409    /// wants one — in a link message, in a symbol table entry, in a reference
16410    /// dataset's elements, and in an attribute's value, which is the one of the
16411    /// four that cannot be revisited after its header is written.
16412    ///
16413    /// Measuring a header before its content is final is sound because no
16414    /// address changes its length: every address is a fixed-width field, and an
16415    /// object that has none yet reads as zero, which is the same width. The
16416    /// storage a header names is laid out between the two passes for the same
16417    /// reason and answers the same way — `emit_attributes` and `emit_links`
16418    /// each fall back to a size-equal placeholder message. It is
16419    /// [`write_object_headers`](Self::write_object_headers) that checks this
16420    /// held, rather than either pass assuming it.
16421    fn allocate_object_headers(&mut self, datasets: &[usize]) -> IoResult<HeaderLayout> {
16422        let mut layout = HeaderLayout {
16423            datasets: Vec::with_capacity(datasets.len()),
16424            groups: Vec::new(),
16425            root: (0, 0),
16426        };
16427        for &i in datasets {
16428            let rc = self.object_link_count(HardLinkTarget::Dataset(i));
16429            let header = self.build_dataset_header(i)?;
16430            let size = self.header_encoded_size(&header, rc, self.dataset_header_format(i))?;
16431            let addr = self
16432                .allocator
16433                .allocate(size as u64, FreeSpaceClass::Metadata);
16434            self.ds(i).lock().obj_header_addr = addr;
16435            layout.datasets.push((i, addr, size));
16436        }
16437        for gi in 0..self.group_count() {
16438            if self.grp(gi).lock().deleted {
16439                continue;
16440            }
16441            let rc = self.object_link_count(HardLinkTarget::Group(gi));
16442            let header = self.build_group_header(gi)?;
16443            let size = self.header_encoded_size(&header, rc, self.group_header_format(gi))?;
16444            let addr = self
16445                .allocator
16446                .allocate(size as u64, FreeSpaceClass::Metadata);
16447            self.grp(gi).lock().obj_header_addr = addr;
16448            layout.groups.push((gi, addr, size));
16449        }
16450        let header = self.build_root_group_header()?;
16451        let size =
16452            self.header_encoded_size(&header, 1, self.header_format(self.root_track_order))?;
16453        let addr = self
16454            .allocator
16455            .allocate(size as u64, FreeSpaceClass::Metadata);
16456        self.root_group_addr = Some(addr);
16457        layout.root = (addr, size);
16458        Ok(layout)
16459    }
16460
16461    /// Write every object header over the block
16462    /// [`allocate_object_headers`](Self::allocate_object_headers) reserved for
16463    /// it.
16464    ///
16465    /// The single owner of object header writing in both finalize paths, and
16466    /// the only place a header's body meets its block: a body that does not
16467    /// fill its measurement exactly fails the finalize here rather than
16468    /// overrunning the next object or leaving a tail of the previous one, which
16469    /// is how a message whose length turns out to depend on an address would
16470    /// show up.
16471    fn write_object_headers(&mut self, layout: &HeaderLayout) -> IoResult<()> {
16472        for &(i, addr, size) in &layout.datasets {
16473            let rc = self.object_link_count(HardLinkTarget::Dataset(i));
16474            let header = self.build_dataset_header(i)?;
16475            let encoded =
16476                self.encode_header_at(&header, rc, self.dataset_header_format(i), addr)?;
16477            check_header_size(&encoded, size, || {
16478                format!("dataset '{}'", self.ds(i).lock().name)
16479            })?;
16480            self.handle.write_at(addr, &encoded)?;
16481            // Only after the bytes are down: a failed write leaves the registry
16482            // describing the header the file still holds.
16483            self.ds(i).lock().header_written(rc);
16484        }
16485        for &(gi, addr, size) in &layout.groups {
16486            let rc = self.object_link_count(HardLinkTarget::Group(gi));
16487            let header = self.build_group_header(gi)?;
16488            let encoded = self.encode_header_at(&header, rc, self.group_header_format(gi), addr)?;
16489            check_header_size(&encoded, size, || {
16490                format!("group '{}'", self.grp(gi).lock().name)
16491            })?;
16492            self.handle.write_at(addr, &encoded)?;
16493        }
16494        let (addr, size) = layout.root;
16495        let header = self.build_root_group_header()?;
16496        let encoded =
16497            self.encode_header_at(&header, 1, self.header_format(self.root_track_order), addr)?;
16498        check_header_size(&encoded, size, || "the root group".to_string())?;
16499        self.handle.write_at(addr, &encoded)?;
16500        Ok(())
16501    }
16502
16503    fn build_dataset_header(&self, index: usize) -> IoResult<ObjectHeader> {
16504        // Compute the link count first: object_link_count re-locks dataset and
16505        // group slots (including this one), so it must run before we take this
16506        // dataset's slot guard — otherwise it would deadlock on the same slot.
16507        let rc = self.object_link_count(HardLinkTarget::Dataset(index));
16508        // Same reason: reading the committed type's address locks the
16509        // committed-datatype registry, which the slot guard below must not be
16510        // held across.
16511        let committed = self.ds(index).lock().committed_type;
16512        let committed_addr = committed.map(|r| match r {
16513            CommittedTypeRef::Session(ci) => self.committed_datatypes.lock()[ci].obj_header_addr,
16514            CommittedTypeRef::Preserved(addr) => addr,
16515        });
16516        // And again: an attribute holding an object reference is said in the
16517        // target's header address, which is read off that object's slot.
16518        let attributes = self.object_attributes(AttrScope::Dataset(index))?;
16519
16520        // Hold one slot guard for the whole header build.
16521        let ds = self.ds(index);
16522        let m = ds.lock();
16523        let mut header = ObjectHeader::new();
16524
16525        // Every message below is written in the format this dataset already
16526        // has, not the one this session would pick. libhdf5 grows a header in
16527        // place and never re-encodes a message it did not touch, so reopening
16528        // a superblock-v2 file — which raises the low bound to V18
16529        // (hdf5_1.14.6 H5Fsuper.c:460-462) — leaves the version-1 dataspaces
16530        // an EARLIEST-bound creating session wrote exactly as they are. This
16531        // writer has to lay the whole header out again whenever the
16532        // shared-message heap moves, so preserving the encoding is the only
16533        // way to land on the same bytes.
16534        let format = m.read_format.unwrap_or_else(|| self.message_format());
16535        let libver = match format {
16536            ObjectFormat::Legacy => LibverBound::Earliest,
16537            ObjectFormat::Modern => self.encoding_libver(),
16538        };
16539
16540        // Dataspace message (type 0x01)
16541        let ds_msg = m.dataspace.encode_for(&self.ctx, format);
16542        let owner = ShareOwner::Header(m.obj_header_addr);
16543        let (flags, ds_msg) = self.share_message(owner, MSG_DATASPACE, 0x00, ds_msg);
16544        header.add_message(MSG_DATASPACE, flags, ds_msg);
16545
16546        // Datatype message (type 0x03). A dataset built on a committed type
16547        // stores a pointer to that object header in place of the message, and
16548        // the shared flag is what says the body is a pointer — the two are one
16549        // statement, so they are written together.
16550        match committed_addr {
16551            Some(addr) => header.add_message(
16552                MSG_DATATYPE,
16553                MSG_FLAG_CONSTANT | MSG_FLAG_SHARED,
16554                SharedMessagePointer::encode_committed(addr, &self.ctx),
16555            ),
16556            None => {
16557                let body = m.datatype.encode_at(&self.ctx, libver);
16558                let (flags, body) = if self.dataset_datatype_shareable(&m.datatype, libver) {
16559                    self.share_message(owner, MSG_DATATYPE, MSG_FLAG_CONSTANT, body)
16560                } else {
16561                    (MSG_FLAG_CONSTANT, body)
16562                };
16563                header.add_message(MSG_DATATYPE, flags, body)
16564            }
16565        }
16566
16567        // Fill Value message (type 0x05)
16568        let is_chunked = m.is_chunked();
16569        // `H5P__init_def_layout` gives each storage class its own default
16570        // allocation time: incremental for chunked and for virtual (whose
16571        // source datasets are allocated as they are written), early for
16572        // compact (the space is the header, so it exists as soon as the
16573        // dataset does), late for contiguous. An implicitly indexed dataset is
16574        // the one chunked exception, and not by default but by definition:
16575        // early allocation is a *condition* of that index
16576        // (`H5D__layout_set_latest_indexing`), so a header claiming
16577        // incremental would describe a file libhdf5 would never have chosen
16578        // this index for. A single-chunk dataset can go either way — unlike
16579        // Implicit, early allocation is not one of its selection conditions
16580        // — so its `early_alloc` flag (set only for an unfiltered dataset
16581        // created that way) is what this checks instead.
16582        let alloc_time = if m.compact.is_some()
16583            || m.implicit.is_some()
16584            || m.single_chunk.as_ref().is_some_and(|s| s.early_alloc)
16585        {
16586            1 // early
16587        } else if is_chunked || m.virtual_storage.is_some() {
16588            3 // incremental
16589        } else {
16590            2 // late
16591        };
16592        // `H5D__update_oh_info` (H5Dint.c:927-943): a variable-length
16593        // datatype with no explicit fill value forces ALLOC regardless of
16594        // the declared policy — its heap-reference encoding has no safe
16595        // all-zero "no fill" representation, so libhdf5 always writes the
16596        // (empty) fill value at allocation for such a dataset. `IFSET` is
16597        // the only declared policy this touches: an explicit `ALLOC` is
16598        // already what it forces, and upstream rejects `NEVER` for a
16599        // VL-typed dataset at `H5Dcreate` outright — this crate's
16600        // VL-typed datasets have no builder path to declare `NEVER` in the
16601        // first place, so that branch cannot be reached here.
16602        let is_vlen = matches!(
16603            m.datatype,
16604            DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
16605        );
16606        let fill_write_time = if is_vlen && m.fill_value.is_none() && m.fill_time == FILL_TIME_IFSET
16607        {
16608            FILL_TIME_ALLOC
16609        } else {
16610            m.fill_time
16611        };
16612        let fv = if let Some(ref bytes) = m.fill_value {
16613            // User-defined fill value (fill_defined = 2).
16614            FillValueMessage {
16615                alloc_time,
16616                fill_write_time,
16617                fill_defined: 2,
16618                fill_value: Some(bytes.clone()),
16619            }
16620        } else {
16621            // No fill value of the dataset's own (fill_defined = 1, the
16622            // implicit default zero fill) — `alloc_time` above already
16623            // carries the per-layout-class default (`H5P__set_layout`,
16624            // H5Pdcpl.c:1864-1877), so this branch must use it too instead
16625            // of `FillValueMessage::default()`'s hardcoded LATE: that was
16626            // wrong for a compact (EARLY) or virtual (INCR) dataset with no
16627            // fill value, only coincidentally right for contiguous.
16628            FillValueMessage {
16629                alloc_time,
16630                fill_write_time,
16631                fill_defined: 1, // default value (zeros)
16632                fill_value: None,
16633            }
16634        };
16635        // `H5O_MSG_FLAG_CONSTANT`, as `H5D__update_oh_info` appends it
16636        // (H5Dint.c:965) — the same flag the datatype message beside it
16637        // carries (H5Dint.c:961) and the old fill value below (H5Dint.c:981).
16638        // A dataset's fill value is fixed at creation: `H5Pset_fill_value` is
16639        // a creation property, so nothing can rewrite the message in place and
16640        // libhdf5 tells the header so.
16641        let fv_msg = fv.encode_for(format);
16642        let (flags, fv_msg) = self.share_message(owner, MSG_FILL_VALUE, MSG_FLAG_CONSTANT, fv_msg);
16643        header.add_message(MSG_FILL_VALUE, flags, fv_msg);
16644
16645        // The "fill value (old)" message (type 0x04) beside the new one, for a
16646        // user-defined fill value below the v1.8 bound. `H5D__update_oh_info`
16647        // (H5Dint.c:1024-1035) appends `H5O_FILL_ID` whenever `fill_prop->buf`
16648        // is set and `use_at_least_v18` — `H5F_LOW_BOUND(file) >= V18`, which
16649        // here is exactly a non-`Legacy` message format — is false, so that a
16650        // reader that predates the new message still finds the value. The body
16651        // is the size and the bytes and nothing else: no allocation time, no
16652        // write time, no defined flag (`H5O__fill_old_encode`, H5Ofill.c:512).
16653        if matches!(format, ObjectFormat::Legacy) {
16654            if let Some(ref bytes) = m.fill_value {
16655                let mut old = Vec::with_capacity(4 + bytes.len());
16656                old.extend_from_slice(&(bytes.len() as u32).to_le_bytes());
16657                old.extend_from_slice(bytes);
16658                let (flags, old) =
16659                    self.share_message(owner, MSG_FILL_VALUE_OLD, MSG_FLAG_CONSTANT, old);
16660                header.add_message(MSG_FILL_VALUE_OLD, flags, old);
16661            }
16662        }
16663
16664        // External Data Files message (type 0x07), before the layout message
16665        // and marked constant, exactly where `H5D__layout_oh_create` puts it.
16666        // It is what makes a reader route the dataset's I/O through the files
16667        // it names rather than through the undefined address the layout
16668        // message below still declares.
16669        if let Some(ref ext) = m.external {
16670            header.add_message(
16671                MSG_EXTERNAL_FILE_LIST,
16672                MSG_FLAG_CONSTANT,
16673                ext.message().encode(&self.ctx),
16674            );
16675        }
16676
16677        // Data Layout message (type 0x08)
16678        let layout = if let Some(ref chunked) = m.chunked {
16679            let mut layout_dims = chunked.chunk_dims.clone();
16680            layout_dims.push(m.datatype.element_size() as u64);
16681            DataLayoutMessage::chunked_v4_earray(
16682                m.layout_version,
16683                layout_dims,
16684                chunked.earray_params.clone(),
16685                chunked.ea_header_addr,
16686            )
16687        } else if let Some(ref fa) = m.fixed_array {
16688            let mut layout_dims = fa.chunk_dims.clone();
16689            layout_dims.push(m.datatype.element_size() as u64);
16690            DataLayoutMessage::chunked_v4_farray(
16691                m.layout_version,
16692                layout_dims,
16693                FixedArrayParams::default_params(),
16694                fa.fa_header_addr,
16695            )
16696        } else if let Some(ref bt2) = m.btree_v2 {
16697            let mut layout_dims = bt2.chunk_dims.clone();
16698            layout_dims.push(m.datatype.element_size() as u64);
16699            DataLayoutMessage::chunked_v4_btree_v2(
16700                m.layout_version,
16701                layout_dims,
16702                crate::format::messages::data_layout::Bt2Params {
16703                    node_size: bt2.index.node_size,
16704                    split_percent: bt2.index.split_percent,
16705                    merge_percent: bt2.index.merge_percent,
16706                },
16707                bt2.bt2_header_addr,
16708            )
16709        } else if let Some(ref imp) = m.implicit {
16710            let mut layout_dims = imp.chunk_dims.clone();
16711            layout_dims.push(m.datatype.element_size() as u64);
16712            DataLayoutMessage::chunked_v4_implicit(m.layout_version, layout_dims, imp.data_addr)
16713        } else if let Some(ref sc) = m.single_chunk {
16714            let mut layout_dims = sc.chunk_dims.clone();
16715            layout_dims.push(m.datatype.element_size() as u64);
16716            if m.filter_pipeline.is_some() {
16717                DataLayoutMessage::chunked_v4_single_filtered(
16718                    layout_dims,
16719                    sc.data_addr,
16720                    sc.nbytes,
16721                    sc.filter_mask,
16722                )
16723            } else {
16724                DataLayoutMessage::chunked_v4_single(layout_dims, sc.data_addr)
16725            }
16726        } else if let Some(ref bt1) = m.btree_v1 {
16727            // The classic index: a version-3 layout message carrying the
16728            // address of the tree's root node, which is undefined until a
16729            // chunk is written.
16730            let mut layout_dims = bt1.chunk_dims.clone();
16731            layout_dims.push(m.datatype.element_size() as u64);
16732            DataLayoutMessage::chunked_v3_btree_v1(layout_dims, bt1.root_addr)
16733        } else if let Some(ref image) = m.compact {
16734            DataLayoutMessage::compact(image.clone())
16735        } else if let Some(ref virt) = m.virtual_storage {
16736            // Version 4 always: the virtual layout class did not exist before
16737            // it, so the default virtual layout is created at version 4 and
16738            // `H5Pset_virtual` raises any lower one to it (H5Pdcpl.c),
16739            // whatever the file's library-version bounds say — which is why a
16740            // v0-superblock file can still hold one.
16741            DataLayoutMessage::virtual_layout(4, virt.heap_addr, virt.heap_index)
16742        } else {
16743            DataLayoutMessage::contiguous(m.data_addr, m.data_size)
16744        };
16745        // `H5D__layout_oh_create` (H5Dlayout.c:530-536) marks the layout
16746        // message constant only where the storage it names is certain to be
16747        // there already: allocation time is early, the class is not compact,
16748        // no filter can change a chunk's size, and the dataspace holds at
16749        // least one element. Anything else leaves the address undefined at
16750        // creation and rewrites the message when the space is allocated, so
16751        // the flag would be a lie. `H5S_GET_EXTENT_NPOINTS` is zero for a
16752        // NULL dataspace and for any extent with a zero-length dimension.
16753        let npoints: u64 = if m.dataspace.is_null() {
16754            0
16755        } else {
16756            m.dataspace.dims.iter().product()
16757        };
16758        let filtered = m
16759            .filter_pipeline
16760            .as_ref()
16761            .is_some_and(|p| !p.filters.is_empty());
16762        let layout_flags = if alloc_time == 1 && m.compact.is_none() && !filtered && npoints != 0 {
16763            MSG_FLAG_CONSTANT
16764        } else {
16765            0x00
16766        };
16767        let layout_msg = layout.encode(&self.ctx);
16768        header.add_message(MSG_DATA_LAYOUT, layout_flags, layout_msg);
16769
16770        // Filter Pipeline message (type 0x0B) -- only if filters are
16771        // configured. `H5D__layout_oh_create` appends it with
16772        // `H5O_MSG_FLAG_CONSTANT` (H5Dlayout.c:462), as does the group
16773        // pipeline for dense links (H5Gobj.c:264): the pipeline is a creation
16774        // property, and every chunk already written was filtered through it,
16775        // so it can never be rewritten in place.
16776        if let Some(ref pipeline) = m.filter_pipeline {
16777            if !pipeline.filters.is_empty() {
16778                let (flags, filter_msg) = self.share_message(
16779                    owner,
16780                    MSG_FILTER_PIPELINE,
16781                    MSG_FLAG_CONSTANT,
16782                    pipeline.encode_for(format),
16783                );
16784                header.add_message(MSG_FILTER_PIPELINE, flags, filter_msg);
16785            }
16786        }
16787
16788        // A dataset has no links, so only attribute creation order can raise
16789        // its header past version 1 (`H5O__set_version`).
16790        let format = self.header_format(TrackOrder {
16791            links: CreationOrder::default(),
16792            attrs: m.track_attr_order,
16793        });
16794
16795        // Modification time, here and not earlier: `H5D__update_oh_info` makes
16796        // this the last message it writes (H5Dint.c:1022-1026), and the
16797        // attributes below it are added by `H5A` calls that come after the
16798        // dataset exists.
16799        touch_oh(&mut header, format, m.times, true);
16800
16801        // Attribute Info (type 0x15) + attribute messages (type 0x0C).
16802        self.emit_attributes(
16803            &mut header,
16804            AttrScope::Dataset(index),
16805            &attributes,
16806            m.track_attr_order,
16807            format,
16808            owner,
16809        );
16810
16811        self.emit_refcount(&mut header, rc, format);
16812
16813        Ok(header)
16814    }
16815
16816    /// Write the object header of every committed datatype something still
16817    /// reaches, recording the address each one landed at.
16818    ///
16819    /// Runs before the dataset and group headers because both name these
16820    /// addresses — a sharing dataset in its datatype message, the parent
16821    /// group in the link. One pass is enough: the header holds a datatype
16822    /// message and at most a reference count, neither of which depends on an
16823    /// address.
16824    fn write_committed_datatype_headers(&mut self) -> IoResult<()> {
16825        // The count is bound first: a lock guard in the `for` iterator
16826        // expression would live for the whole loop body, which locks the same
16827        // registry again.
16828        let count = self.committed_datatypes.lock().len();
16829        for i in 0..count {
16830            let rc = self.committed_datatype_refcount(i);
16831            if rc == 0 {
16832                // Its name's group was deleted and no dataset shares it, so
16833                // nothing in the file could reach the header.
16834                continue;
16835            }
16836            let format = self.committed_datatype_header_format();
16837            let encoded = self
16838                .build_committed_datatype_header(i, rc, format)
16839                .encode_for(format, rc)?;
16840            let addr = self
16841                .allocator
16842                .allocate(encoded.len() as u64, FreeSpaceClass::Metadata);
16843            self.handle.write_at(addr, &encoded)?;
16844            self.committed_datatypes.lock()[i].obj_header_addr = addr;
16845        }
16846        Ok(())
16847    }
16848
16849    /// The header format a committed datatype gets.
16850    ///
16851    /// `H5T__commit` creates the header from the datatype creation property
16852    /// list (H5Tcommit.c:468), which carries no link order and, by default, no
16853    /// attribute order — so the version is the file's floor exactly as
16854    /// `H5O__set_version` computes it, and a committed datatype in a classic
16855    /// file is a version-1 header like every other object in it.
16856    fn committed_datatype_header_format(&self) -> ObjectFormat {
16857        self.header_format(TrackOrder::default())
16858    }
16859
16860    /// Build the object header for a committed datatype: the type, and the
16861    /// reference count when more than one name reaches it.
16862    fn build_committed_datatype_header(
16863        &self,
16864        index: usize,
16865        rc: u32,
16866        format: ObjectFormat,
16867    ) -> ObjectHeader {
16868        let (datatype, times) = {
16869            let reg = self.committed_datatypes.lock();
16870            (reg[index].datatype.clone(), reg[index].times)
16871        };
16872        let mut header = ObjectHeader::new();
16873        // No attributes to emit, so nothing else would apply the file-wide
16874        // floor to this header. `store_msg_crt_idx` is a property of the file,
16875        // not of the object: every header created under it records creation
16876        // indices, a committed datatype's included.
16877        header.set_attribute_creation_order(self.header_attr_order(CreationOrder::default()));
16878        // `H5T__commit` marks the message constant and unshareable: this
16879        // header is where shared datatype bodies are read *from*, so its own
16880        // message must never become a pointer into the shared-message heap.
16881        header.add_message(
16882            MSG_DATATYPE,
16883            MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
16884            datatype.encode_at(&self.ctx, self.encoding_libver()),
16885        );
16886        touch_oh(&mut header, format, times, false);
16887        // Through the same owner as every other object's count: a dataset
16888        // sharing this type raises it (`H5O__shared_link_adj`, H5Oshared.c:249)
16889        // just as a second name does, and where that count is recorded is the
16890        // header version's business, not the caller's.
16891        self.emit_refcount(&mut header, rc, format);
16892        header
16893    }
16894
16895    /// Build the object header for a subgroup.
16896    fn build_group_header(&self, group_idx: usize) -> IoResult<ObjectHeader> {
16897        let mut header = ObjectHeader::new();
16898
16899        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
16900        // themselves, compact or dense.
16901        // Snapshot what the header needs, then drop the slot guard: the calls
16902        // below re-lock group slots (including this one).
16903        let (track_order, times, owner) = {
16904            let grp = self.grp(group_idx);
16905            let g = grp.lock();
16906            (
16907                g.track_order,
16908                g.times,
16909                ShareOwner::Header(g.obj_header_addr),
16910            )
16911        };
16912        let attributes = self.object_attributes(AttrScope::Group(group_idx))?;
16913        touch_oh(&mut header, self.header_format(track_order), times, false);
16914
16915        let links = self.group_links(LinkScope::Group(group_idx), track_order.links);
16916        self.emit_links(
16917            &mut header,
16918            LinkScope::Group(group_idx),
16919            &links,
16920            track_order.links,
16921        );
16922
16923        // Attribute Info (type 0x15) + attributes (type 0x0C) -- e.g. NeXus
16924        // `NX_class`.
16925        let format = self.header_format(track_order);
16926        self.emit_attributes(
16927            &mut header,
16928            AttrScope::Group(group_idx),
16929            &attributes,
16930            track_order.attrs,
16931            format,
16932            owner,
16933        );
16934
16935        self.emit_refcount(
16936            &mut header,
16937            self.object_link_count(HardLinkTarget::Group(group_idx)),
16938            format,
16939        );
16940
16941        Ok(header)
16942    }
16943
16944    fn build_root_group_header(&self) -> IoResult<ObjectHeader> {
16945        let mut header = ObjectHeader::new();
16946        touch_oh(
16947            &mut header,
16948            self.header_format(self.root_track_order),
16949            self.root_times,
16950            false,
16951        );
16952
16953        // Link Info (type 0x02) + Group Info (type 0x0A) + the links
16954        // themselves, compact or dense.
16955        let links = self.group_links(LinkScope::Root, self.root_track_order.links);
16956        self.emit_links(
16957            &mut header,
16958            LinkScope::Root,
16959            &links,
16960            self.root_track_order.links,
16961        );
16962
16963        // Root-level attributes
16964        let root_attributes = self.object_attributes(AttrScope::Root)?;
16965        self.emit_attributes(
16966            &mut header,
16967            AttrScope::Root,
16968            &root_attributes,
16969            self.root_track_order.attrs,
16970            self.header_format(self.root_track_order),
16971            ShareOwner::Header(self.root_group_addr.unwrap_or(0)),
16972        );
16973
16974        Ok(header)
16975    }
16976}
16977
16978impl Drop for Hdf5Writer {
16979    fn drop(&mut self) {
16980        if !self.closed {
16981            // Best-effort finalize on drop. Drop cannot return a Result, so a
16982            // failure here is otherwise invisible: it would leave a truncated
16983            // or unflushed file on disk while the caller believes the write
16984            // succeeded. Surface it on stderr instead of swallowing it.
16985            // Callers that need to handle the error must call
16986            // `H5File::close()` explicitly, which returns the Result.
16987            if let Err(e) = self.finalize(true) {
16988                eprintln!(
16989                    "rust-hdf5: failed to finalize HDF5 file on drop: {e}. \
16990                     The file may be incomplete or corrupt; call \
16991                     H5File::close() to handle this error explicitly."
16992                );
16993            }
16994        }
16995    }
16996}
16997
16998#[cfg(test)]
16999mod tests {
17000    use super::*;
17001    use crate::format::messages::datatype::DatatypeMessage;
17002    use crate::io::reader::Hdf5Reader;
17003
17004    fn fixture(name: &str) -> std::path::PathBuf {
17005        std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
17006            .join("tests/fixtures")
17007            .join(name)
17008    }
17009
17010    /// Copy a fixture so a test that appends does not edit the checked-in file.
17011    fn fixture_copy(name: &str, tag: &str) -> std::path::PathBuf {
17012        let path = temp_path(tag);
17013        std::fs::copy(fixture(name), &path).unwrap();
17014        path
17015    }
17016
17017    fn temp_path(tag: &str) -> std::path::PathBuf {
17018        use std::sync::atomic::{AtomicU64, Ordering};
17019        static COUNTER: AtomicU64 = AtomicU64::new(0);
17020        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
17021        std::env::temp_dir().join(format!(
17022            "rust_hdf5_w_{}_{}_{}.h5",
17023            std::process::id(),
17024            tag,
17025            n
17026        ))
17027    }
17028
17029    /// A group past the link phase change keeps its links in a fractal heap
17030    /// with a v2 B-tree name index. The reopen that rewrites that group's
17031    /// header lays a fresh pair out, so both blocks the old header named must
17032    /// come back to the allocator — every block of the heap, and the index
17033    /// header with its nodes.
17034    ///
17035    /// Asserted on the free list rather than on the file size: a reopen does
17036    /// not yet carry dense links forward, so the rewritten group's links (and
17037    /// the datasets they name) are dropped, and the file size that follows
17038    /// says more about that than about this.
17039    #[test]
17040    fn a_reopen_frees_the_dense_link_storage_its_rewrite_supersedes() {
17041        let path = temp_path("dense_link_reclaim");
17042
17043        let writer = Hdf5Writer::create(&path).unwrap();
17044        writer.create_group("/", "run").unwrap();
17045        for i in 0..12 {
17046            writer
17047                .create_dataset(&format!("run/d{i:02}"), DatatypeMessage::i32_type(), &[2])
17048                .unwrap();
17049        }
17050        writer.close().unwrap();
17051
17052        let writer = Hdf5Writer::open_append(&path).unwrap();
17053        let gidx = (0..writer.group_count())
17054            .find(|&g| writer.grp(g).lock().name == "/run")
17055            .expect("the reopen registered the group");
17056        let linfo = writer
17057            .superseded_dense
17058            .lock()
17059            .as_ref()
17060            .and_then(|s| s.links.get(&LinkScope::Group(gidx)).cloned())
17061            .expect("the reopen recorded the group's dense link storage");
17062        assert_ne!(linfo.fractal_heap_address, UNDEF_ADDR);
17063        assert_ne!(linfo.name_btree_address, UNDEF_ADDR);
17064
17065        writer
17066            .release_superseded_dense_links(LinkScope::Group(gidx))
17067            .unwrap();
17068        let freed = writer.allocator.free_blocks();
17069        let covers = |addr: u64| {
17070            freed
17071                .iter()
17072                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17073        };
17074        assert!(covers(linfo.fractal_heap_address), "heap header: {freed:?}");
17075        assert!(covers(linfo.name_btree_address), "name index: {freed:?}");
17076
17077        // And exactly once: the entry is gone, so the finalize that follows
17078        // cannot hand the same blocks back a second time.
17079        assert!(writer
17080            .superseded_dense
17081            .lock()
17082            .as_ref()
17083            .is_none_or(|s| s.links.is_empty()));
17084        writer
17085            .release_superseded_dense_links(LinkScope::Group(gidx))
17086            .unwrap();
17087        assert_eq!(writer.allocator.free_blocks(), freed);
17088
17089        writer.close().unwrap();
17090        std::fs::remove_file(&path).ok();
17091    }
17092
17093    /// The rewrite frees what it supersedes even when the replacement is not
17094    /// dense at all. An attribute set that drops back under `max_compact`
17095    /// goes into the object header, so nothing names the old heap any more —
17096    /// and a free driven by "the new set needs dense storage" would never
17097    /// reach this one.
17098    #[test]
17099    fn a_rewrite_that_drops_out_of_dense_storage_still_frees_it() {
17100        let path = temp_path("dense_attr_to_compact");
17101        let numeric = |name: &str| {
17102            AttributeMessage::scalar_numeric(
17103                name,
17104                DatatypeMessage::i32_type(),
17105                7i32.to_le_bytes().to_vec(),
17106            )
17107        };
17108
17109        let writer = Hdf5Writer::create(&path).unwrap();
17110        for i in 0..12 {
17111            writer
17112                .add_root_attribute(numeric(&format!("a{i:02}")))
17113                .unwrap();
17114        }
17115        writer.close().unwrap();
17116
17117        let writer = Hdf5Writer::open_append(&path).unwrap();
17118        let ainfo = writer
17119            .superseded_dense
17120            .lock()
17121            .as_ref()
17122            .and_then(|s| s.attrs.get(&AttrScope::Root).cloned())
17123            .expect("the reopen recorded the root's dense attribute storage");
17124        for i in 0..10 {
17125            writer
17126                .evict_attr(AttrTarget::Root, &format!("a{i:02}"))
17127                .unwrap();
17128        }
17129        assert!(!writer.attributes_need_dense(&writer.root_attributes.lock(), ObjectFormat::Modern));
17130
17131        writer.prepare_dense_attributes(&[]).unwrap();
17132        let freed = writer.allocator.free_blocks();
17133        let covers = |addr: u64| {
17134            freed
17135                .iter()
17136                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17137        };
17138        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17139        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17140        assert!(writer
17141            .superseded_dense
17142            .lock()
17143            .as_ref()
17144            .is_none_or(|s| s.attrs.is_empty()));
17145
17146        writer.close().unwrap();
17147        std::fs::remove_file(&path).ok();
17148    }
17149
17150    /// Deleting a reopened object supersedes its dense storage as surely as
17151    /// rewriting one does: nothing in the finalized file names the heap, so
17152    /// the delete owner frees it through the same entry.
17153    #[test]
17154    fn deleting_a_reopened_group_frees_its_dense_attribute_storage() {
17155        let path = temp_path("dense_attr_delete");
17156        let numeric = |name: &str| {
17157            AttributeMessage::scalar_numeric(
17158                name,
17159                DatatypeMessage::i32_type(),
17160                7i32.to_le_bytes().to_vec(),
17161            )
17162        };
17163
17164        let writer = Hdf5Writer::create(&path).unwrap();
17165        writer.create_group("/", "run").unwrap();
17166        for i in 0..12 {
17167            writer
17168                .set_attribute(AttrTarget::Group("/run"), numeric(&format!("a{i:02}")))
17169                .unwrap();
17170        }
17171        writer.close().unwrap();
17172
17173        let writer = Hdf5Writer::open_append(&path).unwrap();
17174        let gidx = (0..writer.group_count())
17175            .find(|&g| writer.grp(g).lock().name == "/run")
17176            .expect("the reopen registered the group");
17177        let ainfo = writer
17178            .superseded_dense
17179            .lock()
17180            .as_ref()
17181            .and_then(|s| s.attrs.get(&AttrScope::Group(gidx)).cloned())
17182            .expect("the reopen recorded the group's dense attribute storage");
17183
17184        writer.delete_group("/run").unwrap();
17185        let freed = writer.allocator.free_blocks();
17186        let covers = |addr: u64| {
17187            freed
17188                .iter()
17189                .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17190        };
17191        assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17192        assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17193        assert!(writer
17194            .superseded_dense
17195            .lock()
17196            .as_ref()
17197            .is_none_or(|s| s.attrs.is_empty()));
17198
17199        writer.close().unwrap();
17200        std::fs::remove_file(&path).ok();
17201    }
17202
17203    /// The charset rule is one owner shared by every vlen string writer:
17204    /// appends into an ASCII-declared dataset reject non-ASCII strings the
17205    /// same way the slice writer does, and a dataset whose elements are not
17206    /// vlen references at all is refused instead of overwritten with them.
17207    #[test]
17208    fn append_vlen_strings_checks_the_datatype_and_charset() {
17209        let path = temp_path("append_vlen_charset");
17210
17211        let writer = Hdf5Writer::create(&path).unwrap();
17212        let idx = writer
17213            .create_appendable_vlen_string_dataset("d", 4, None)
17214            .unwrap();
17215        writer.ds(idx).lock().datatype = DatatypeMessage::vlen_string_ascii();
17216        let err = writer
17217            .append_vlen_strings(idx, &["ok", "안녕"])
17218            .unwrap_err();
17219        assert!(
17220            err.to_string().contains("is not ASCII"),
17221            "unexpected error: {err}"
17222        );
17223        writer.append_vlen_strings(idx, &["ok", "fine"]).unwrap();
17224
17225        let nums = writer
17226            .create_chunked_dataset("n", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
17227            .unwrap();
17228        let err = writer.append_vlen_strings(nums, &["x"]).unwrap_err();
17229        assert!(
17230            err.to_string()
17231                .contains("only for variable-length string datasets"),
17232            "unexpected error: {err}"
17233        );
17234
17235        writer.close().unwrap();
17236        std::fs::remove_file(&path).ok();
17237    }
17238
17239    /// `create_chunked_dataset` builds an extensible-array index unconditionally
17240    /// (the caller — the high-level dataset API — is the one that decides when
17241    /// two-or-more unlimited dimensions should go to a v2 B-tree instead), so
17242    /// its own guard is the last line of defense against a shape that index
17243    /// can't represent at all.
17244    #[test]
17245    fn create_chunked_dataset_rejects_two_unlimited_dimensions() {
17246        let path = temp_path("earray_two_unlimited");
17247        let writer = Hdf5Writer::create(&path).unwrap();
17248        let err = writer
17249            .create_chunked_dataset(
17250                "d",
17251                DatatypeMessage::i32_type(),
17252                &[4, 4],
17253                &[u64::MAX, u64::MAX],
17254                &[2, 2],
17255            )
17256            .unwrap_err();
17257        assert!(err.to_string().contains("at most one unlimited"), "{err}");
17258        writer.close().unwrap();
17259        std::fs::remove_file(&path).ok();
17260    }
17261
17262    /// Every creator must enter through `begin_create`; the four that used
17263    /// to bypass it could push a second dataset under an existing name and
17264    /// emit an invalid file with two same-named links.
17265    #[test]
17266    fn every_creator_rejects_an_existing_dataset_name() {
17267        let path = temp_path("create_gate");
17268
17269        let writer = Hdf5Writer::create(&path).unwrap();
17270        writer
17271            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
17272            .unwrap();
17273
17274        let attempts: [(&str, IoResult<usize>); 4] = [
17275            (
17276                "vlen_string",
17277                writer.create_vlen_string_dataset("d", &["x"], 1),
17278            ),
17279            ("vlen_bytes", writer.create_vlen_bytes_dataset("d", &[b"x"])),
17280            (
17281                "vlen_string_compressed",
17282                writer.create_vlen_string_dataset_compressed(
17283                    "d",
17284                    &["x"],
17285                    1,
17286                    FilterPipeline::deflate(6),
17287                ),
17288            ),
17289            (
17290                "chunked_with_pipeline",
17291                writer.create_chunked_dataset_with_pipeline(
17292                    "d",
17293                    DatatypeMessage::i32_type(),
17294                    &[0],
17295                    &[u64::MAX],
17296                    &[4],
17297                    FilterPipeline::deflate(6),
17298                ),
17299            ),
17300        ];
17301        for (which, res) in attempts {
17302            match res {
17303                Ok(_) => panic!("{which} accepted a duplicate name"),
17304                Err(e) => assert!(
17305                    e.to_string().contains("already exists"),
17306                    "{which}: unexpected error: {e}"
17307                ),
17308            }
17309        }
17310
17311        writer.close().unwrap();
17312        std::fs::remove_file(&path).ok();
17313    }
17314
17315    /// Every creator and every kind of name meet at `ensure_name_free`.
17316    ///
17317    /// The gate's whole value is that it is one list: a creator must be
17318    /// blind neither to a name kind it does not itself make nor to one added
17319    /// after it. This crosses the two — six names, one of each kind the
17320    /// writer can put in a group, against every creator — so a creator that
17321    /// grows its own check, or a name kind that stops being on the list,
17322    /// fails here rather than in a file holding two links of one name.
17323    #[test]
17324    fn every_creator_refuses_every_kind_of_taken_name() {
17325        let path = temp_path("create_gate_matrix");
17326        let writer = Hdf5Writer::create(&path).unwrap();
17327
17328        let i32t = || DatatypeMessage::i32_type();
17329        writer.create_dataset("d", i32t(), &[2]).unwrap();
17330        writer.create_compact_dataset("c", i32t(), &[2]).unwrap();
17331        writer.create_group("/", "g").unwrap();
17332        writer.commit_datatype("t", i32t()).unwrap();
17333        writer.create_hard_link("/", "h", "d").unwrap();
17334        writer
17335            .create_symbolic_link(
17336                "/",
17337                "s",
17338                LinkTarget::Soft {
17339                    target: "/d".into(),
17340                },
17341            )
17342            .unwrap();
17343        writer
17344            .create_symbolic_link(
17345                "/",
17346                "e",
17347                LinkTarget::External {
17348                    file: "other.h5".into(),
17349                    path: "/x".into(),
17350                },
17351            )
17352            .unwrap();
17353
17354        for taken in ["d", "c", "g", "t", "h", "s", "e"] {
17355            let attempts: [(&str, IoResult<()>); 8] = [
17356                (
17357                    "dataset",
17358                    writer.create_dataset(taken, i32t(), &[2]).map(|_| ()),
17359                ),
17360                (
17361                    "compact",
17362                    writer
17363                        .create_compact_dataset(taken, i32t(), &[2])
17364                        .map(|_| ()),
17365                ),
17366                (
17367                    "chunked",
17368                    writer
17369                        .create_chunked_dataset(taken, i32t(), &[0], &[u64::MAX], &[4])
17370                        .map(|_| ()),
17371                ),
17372                (
17373                    "vlen_string",
17374                    writer
17375                        .create_vlen_string_dataset(taken, &["x"], 1)
17376                        .map(|_| ()),
17377                ),
17378                (
17379                    "committed datatype",
17380                    writer.commit_datatype(taken, i32t()).map(|_| ()),
17381                ),
17382                ("group", writer.create_group("/", taken).map(|_| ())),
17383                ("hard link", writer.create_hard_link("/", taken, "d")),
17384                (
17385                    "soft link",
17386                    writer.create_symbolic_link(
17387                        "/",
17388                        taken,
17389                        LinkTarget::Soft {
17390                            target: "/d".into(),
17391                        },
17392                    ),
17393                ),
17394            ];
17395            for (which, res) in attempts {
17396                match res {
17397                    Ok(()) => panic!("{which} accepted the taken name '{taken}'"),
17398                    Err(e) => assert!(
17399                        e.to_string().contains("already exists"),
17400                        "{which} on '{taken}': unexpected error: {e}"
17401                    ),
17402                }
17403            }
17404        }
17405
17406        writer.close().unwrap();
17407        std::fs::remove_file(&path).ok();
17408    }
17409
17410    /// The `H5T_VLEN` length field counts base elements, so an image that is
17411    /// not a whole number of them has no length that reads back as what was
17412    /// handed over; it is refused at the call rather than stored truncated.
17413    #[test]
17414    fn vlen_sequence_refuses_a_partial_element() {
17415        let path = temp_path("vlen_partial_element");
17416
17417        let writer = Hdf5Writer::create(&path).unwrap();
17418        let err = writer
17419            .create_vlen_sequence_dataset("d", DatatypeMessage::i32_type(), &[&[1u8, 2, 3, 4, 5]])
17420            .unwrap_err()
17421            .to_string();
17422        assert!(err.contains("5 bytes"), "unexpected error: {err}");
17423        assert!(err.contains("4-byte elements"), "unexpected error: {err}");
17424
17425        // The refusal is the length rule alone: the same base takes a whole
17426        // number of elements, and an empty sequence is a legal one.
17427        writer
17428            .create_vlen_sequence_dataset(
17429                "d",
17430                DatatypeMessage::i32_type(),
17431                &[&[1u8, 2, 3, 4], &[][..]],
17432            )
17433            .unwrap();
17434
17435        writer.close().unwrap();
17436        std::fs::remove_file(&path).ok();
17437    }
17438
17439    /// A corrupt file can declare a zero-length chunk dimension; the
17440    /// superseded-reference read must reject it the way `write_slice` does,
17441    /// not divide by it.
17442    #[test]
17443    fn vlen_slice_rejects_a_zero_chunk_dimension() {
17444        let path = temp_path("vlen_slice_zero_chunk");
17445
17446        let writer = Hdf5Writer::create(&path).unwrap();
17447        let idx = writer
17448            .create_appendable_vlen_string_dataset("d", 2, None)
17449            .unwrap();
17450        writer.append_vlen_strings(idx, &["a", "b"]).unwrap();
17451        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 0;
17452        let err = writer.write_vlen_strings_slice(idx, 0, &["x"]).unwrap_err();
17453        assert!(
17454            err.to_string().contains("zero-length dimension"),
17455            "unexpected error: {err}"
17456        );
17457
17458        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 2;
17459        writer.close().unwrap();
17460        std::fs::remove_file(&path).ok();
17461    }
17462
17463    /// A libhdf5-written collection can be 100% full — no free-space marker,
17464    /// content exactly the declared size. When a stale reference names an
17465    /// index that is not there, nothing is removed, and the collection must
17466    /// be left alone: re-encoding it at its declared size cannot fit the
17467    /// free-space marker and would fail the whole update.
17468    #[test]
17469    fn release_leaves_a_full_collection_it_removed_nothing_from() {
17470        use crate::format::global_heap::encode_vlen_reference;
17471
17472        let path = temp_path("release_full_gcol");
17473        let writer = Hdf5Writer::create(&path).unwrap();
17474
17475        // Hand-built full collection: 16-byte header + one 16+8-byte object,
17476        // declared size exactly 40, no free-space marker.
17477        let mut img = Vec::new();
17478        img.extend_from_slice(b"GCOL");
17479        img.push(1);
17480        img.extend_from_slice(&[0u8; 3]);
17481        img.extend_from_slice(&40u64.to_le_bytes());
17482        img.extend_from_slice(&1u16.to_le_bytes()); // object index 1
17483        img.extend_from_slice(&1u16.to_le_bytes()); // ref_count
17484        img.extend_from_slice(&0u32.to_le_bytes()); // reserved
17485        img.extend_from_slice(&8u64.to_le_bytes()); // data size
17486        img.extend_from_slice(b"deadbeef");
17487        assert_eq!(img.len(), 40);
17488        let addr = writer
17489            .allocator
17490            .allocate(img.len() as u64, FreeSpaceClass::RawData);
17491        writer.handle.write_at(addr, &img).unwrap();
17492
17493        // The superseded reference names index 2, which the collection does
17494        // not hold — a no-op removal.
17495        let refs = encode_vlen_reference(3, addr, 2, &writer.ctx);
17496        writer.release_vlen_references(&refs).unwrap();
17497        assert_eq!(writer.handle.read_at(addr, 40).unwrap(), img);
17498
17499        writer.close().unwrap();
17500        std::fs::remove_file(&path).ok();
17501    }
17502
17503    /// The CWFS second pass (`H5F_cwfs_find_free_heap`): an object too big
17504    /// for the listed collection's remaining free space extends the
17505    /// collection in place — the file allocation grows off the end of the
17506    /// file (`H5MF_try_extend`) and the collection's declared size and
17507    /// free-space marker grow with it (`H5HG_extend`) — instead of opening
17508    /// a second collection.
17509    #[test]
17510    fn an_oversized_vlen_insert_extends_the_listed_collection() {
17511        use crate::format::global_heap::GlobalHeapCollection;
17512
17513        let path = temp_path("cwfs_extend_tail");
17514        let writer = Hdf5Writer::create(&path).unwrap();
17515        // A small object opens a minimum-size (4096) listed collection —
17516        // the file's last allocation, so the extension grows the file end.
17517        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
17518        let big = vec![0x41u8; 5000]; // more than the ~4 KiB remaining
17519        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
17520        assert_eq!(
17521            p2[0].0, p1[0].0,
17522            "the big object opened a second collection"
17523        );
17524
17525        // The block on disk is one grown collection holding both objects.
17526        let img = writer.handle.read_at_most(p1[0].0, 65536).unwrap();
17527        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
17528        assert!(csize > 4096, "declared size did not grow: {csize}");
17529        assert_eq!(gcol.objects.len(), 2);
17530        assert_eq!(gcol.objects[1].data, big);
17531
17532        writer.close().unwrap();
17533        let bytes = std::fs::read(&path).unwrap();
17534        assert_eq!(
17535            bytes.windows(4).filter(|w| *w == b"GCOL").count(),
17536            1,
17537            "a second collection signature is in the file"
17538        );
17539        std::fs::remove_file(&path).ok();
17540    }
17541
17542    /// The non-tail counterpart: the collection is pinned away from the end
17543    /// of the file, but a released block starts right after it, so the
17544    /// extension consumes the front of that block (`H5MF_try_extend`'s
17545    /// free-section path) and the remainder stays reusable.
17546    #[test]
17547    fn extension_consumes_a_freed_block_after_the_collection() {
17548        use crate::format::global_heap::GlobalHeapCollection;
17549
17550        let path = temp_path("cwfs_extend_freed");
17551        let writer = Hdf5Writer::create(&path).unwrap();
17552        let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
17553        let addr = p1[0].0;
17554        // Land a block right after the collection, pin the file end past
17555        // it, then release it: extension must use the released space.
17556        let spacer = writer.allocator.allocate(8192, FreeSpaceClass::RawData);
17557        assert_eq!(spacer, addr + 4096, "spacer not adjacent; layout changed");
17558        writer.allocator.allocate(8, FreeSpaceClass::RawData);
17559        writer.allocator.free(spacer, 8192, FreeSpaceClass::RawData);
17560
17561        let big = vec![0x42u8; 5000];
17562        let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
17563        assert_eq!(p2[0].0, addr, "the big object opened a second collection");
17564
17565        let img = writer.handle.read_at_most(addr, 65536).unwrap();
17566        let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
17567        assert_eq!(csize, 8192, "grew by max(size, shortfall) = 4096");
17568        assert_eq!(gcol.objects.len(), 2);
17569
17570        // The remainder of the released block is still allocatable.
17571        assert_eq!(
17572            writer.allocator.allocate(4096, FreeSpaceClass::RawData),
17573            addr + 8192,
17574            "the freed block's tail was lost"
17575        );
17576        writer.close().unwrap();
17577        std::fs::remove_file(&path).ok();
17578    }
17579
17580    /// Issue #10: a reopen-and-replace loop on a vlen string must not grow
17581    /// the file. The superseded heap objects are freed *before* the
17582    /// replacement is allocated, so each session reuses the block it just
17583    /// released even though the free list starts empty on reopen. The old
17584    /// free-after-alloc order failed this by one collection per session.
17585    #[test]
17586    fn vlen_replace_across_reopen_keeps_the_file_flat() {
17587        let path = temp_path("vlen_reopen_flat");
17588        let payload_a = "a".repeat(64 * 1024);
17589        let payload_b = "b".repeat(64 * 1024);
17590
17591        let writer = Hdf5Writer::create(&path).unwrap();
17592        writer
17593            .create_vlen_string_dataset("notes", &["initial"], 1)
17594            .unwrap();
17595        writer.close().unwrap();
17596
17597        let mut sizes = Vec::new();
17598        for i in 0..8 {
17599            let writer = Hdf5Writer::open_append(&path).unwrap();
17600            let payload = if i % 2 == 0 { &payload_a } else { &payload_b };
17601            writer
17602                .write_vlen_strings_slice(0, 0, &[payload.as_str()])
17603                .unwrap();
17604            writer.close().unwrap();
17605            sizes.push(std::fs::metadata(&path).unwrap().len());
17606        }
17607        // The first replacement grows the file once (the initial collection
17608        // cannot hold 64 KiB); every later equal-size replacement must land
17609        // in the block its own session just freed.
17610        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
17611
17612        // The reused blocks still form a valid file holding the last value.
17613        let mut reader = Hdf5Reader::open(&path).unwrap();
17614        assert_eq!(
17615            reader.read_vlen_strings("notes").unwrap(),
17616            vec![payload_b.clone()]
17617        );
17618
17619        std::fs::remove_file(&path).ok();
17620    }
17621
17622    /// Replacing a vlen string attribute must release the superseded
17623    /// global-heap collection *before* the replacement's collection is
17624    /// allocated, so a reopen-replace loop lands each new value in the block
17625    /// it just freed instead of growing the file by one collection per
17626    /// session — the attribute counterpart of
17627    /// [`vlen_replace_across_reopen_keeps_the_file_flat`].
17628    #[test]
17629    fn vlen_attr_replace_across_reopen_keeps_the_file_flat() {
17630        let path = temp_path("vlen_attr_reopen_flat");
17631        let payload_a = "a".repeat(8 * 1024);
17632        let payload_b = "b".repeat(8 * 1024);
17633
17634        let writer = Hdf5Writer::create(&path).unwrap();
17635        writer
17636            .set_vlen_string_attribute(AttrTarget::Root, "note", &payload_a)
17637            .unwrap();
17638        writer.close().unwrap();
17639
17640        let mut sizes = Vec::new();
17641        for i in 0..8 {
17642            let writer = Hdf5Writer::open_append(&path).unwrap();
17643            let payload = if i % 2 == 0 { &payload_b } else { &payload_a };
17644            writer
17645                .set_vlen_string_attribute(AttrTarget::Root, "note", payload)
17646                .unwrap();
17647            writer.close().unwrap();
17648            sizes.push(std::fs::metadata(&path).unwrap().len());
17649        }
17650        assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
17651
17652        // The reused blocks still hold the last value.
17653        let reader = Hdf5Reader::open(&path).unwrap();
17654        let attr = reader.root_attr("note").unwrap().clone();
17655        let mut reader = reader;
17656        assert_eq!(reader.attr_string_value(&attr).unwrap(), payload_a);
17657
17658        std::fs::remove_file(&path).ok();
17659    }
17660
17661    /// A numeric attribute replacing a vlen one goes through the same list
17662    /// owner, so the superseded collection is released even though the new
17663    /// value holds no heap reference: a later same-size vlen attribute must
17664    /// land in the freed block, making the file exactly as large as one that
17665    /// never stored the replaced value.
17666    #[test]
17667    fn numeric_replacing_a_vlen_attr_releases_its_collection() {
17668        let payload = "x".repeat(8 * 1024);
17669        let numeric = || {
17670            AttributeMessage::scalar_numeric(
17671                "x",
17672                DatatypeMessage::i32_type(),
17673                7i32.to_le_bytes().to_vec(),
17674            )
17675        };
17676
17677        let path_a = temp_path("vlen_attr_cross_a");
17678        let writer = Hdf5Writer::create(&path_a).unwrap();
17679        writer
17680            .set_vlen_string_attribute(AttrTarget::Root, "x", &payload)
17681            .unwrap();
17682        writer.add_root_attribute(numeric()).unwrap();
17683        writer
17684            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
17685            .unwrap();
17686        writer.close().unwrap();
17687
17688        // The same end state written without the replaced vlen value.
17689        let path_b = temp_path("vlen_attr_cross_b");
17690        let writer = Hdf5Writer::create(&path_b).unwrap();
17691        writer.add_root_attribute(numeric()).unwrap();
17692        writer
17693            .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
17694            .unwrap();
17695        writer.close().unwrap();
17696
17697        assert_eq!(
17698            std::fs::metadata(&path_a).unwrap().len(),
17699            std::fs::metadata(&path_b).unwrap().len()
17700        );
17701
17702        let reader = Hdf5Reader::open(&path_a).unwrap();
17703        let y = reader.root_attr("y").unwrap().clone();
17704        let mut reader = reader;
17705        assert_eq!(reader.attr_string_value(&y).unwrap(), payload);
17706
17707        std::fs::remove_file(&path_a).ok();
17708        std::fs::remove_file(&path_b).ok();
17709    }
17710
17711    /// Reopen/write/close cycles must not leak the object-header blocks
17712    /// finalize rewrites: the reopened root header, the reopened group
17713    /// header, and the modified chunked dataset's header are each freed
17714    /// before their replacements are allocated. The chunk rewrite itself is
17715    /// in place (unfiltered chunks never move), so a leak of any header
17716    /// block shows up as monotonic growth here.
17717    #[test]
17718    fn reopen_cycles_reuse_superseded_header_blocks() {
17719        let path = temp_path("header_reuse");
17720        {
17721            let writer = Hdf5Writer::create(&path).unwrap();
17722            writer.create_group("/", "g").unwrap();
17723            let idx = writer
17724                .create_chunked_dataset(
17725                    "g/data",
17726                    DatatypeMessage::i32_type(),
17727                    &[4],
17728                    &[u64::MAX],
17729                    &[4],
17730                )
17731                .unwrap();
17732            let seed: Vec<u8> = [1i32, 2, 3, 4]
17733                .iter()
17734                .flat_map(|v| v.to_le_bytes())
17735                .collect();
17736            writer.write_chunk(idx, 0, &seed).unwrap();
17737            writer.close().unwrap();
17738        }
17739
17740        let mut sizes = Vec::new();
17741        for i in 0..6i32 {
17742            let writer = Hdf5Writer::open_append(&path).unwrap();
17743            let data: Vec<u8> = [i; 4].iter().flat_map(|v| v.to_le_bytes()).collect();
17744            writer.write_chunk(0, 0, &data).unwrap();
17745            writer.close().unwrap();
17746            sizes.push(std::fs::metadata(&path).unwrap().len());
17747        }
17748        assert_eq!(&sizes[1..], &vec![sizes[0]; 5][..], "sizes: {sizes:?}");
17749
17750        // The reused header blocks still form a valid file.
17751        let mut reader = Hdf5Reader::open(&path).unwrap();
17752        let raw = reader.read_dataset_raw("g/data").unwrap();
17753        let values: Vec<i32> = raw
17754            .chunks(4)
17755            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
17756            .collect();
17757        assert_eq!(values, vec![5, 5, 5, 5]);
17758
17759        std::fs::remove_file(&path).ok();
17760    }
17761
17762    #[test]
17763    fn create_empty_file() {
17764        let path = temp_path("empty");
17765
17766        let writer = Hdf5Writer::create(&path).unwrap();
17767        writer.close().unwrap();
17768
17769        // Verify we can read it back
17770        let reader = Hdf5Reader::open(&path).unwrap();
17771        assert!(reader.dataset_names().is_empty());
17772
17773        std::fs::remove_file(&path).ok();
17774    }
17775
17776    #[test]
17777    fn create_single_dataset() {
17778        let path = temp_path("single");
17779
17780        let writer = Hdf5Writer::create(&path).unwrap();
17781        let idx = writer
17782            .create_dataset("data", DatatypeMessage::f64_type(), &[4])
17783            .unwrap();
17784        let values: Vec<f64> = vec![1.0, 2.0, 3.0, 4.0];
17785        let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17786        writer.write_dataset_raw(idx, &raw).unwrap();
17787        writer.close().unwrap();
17788
17789        // Read back
17790        let mut reader = Hdf5Reader::open(&path).unwrap();
17791        assert_eq!(reader.dataset_names(), vec!["data"]);
17792        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4]);
17793        let readback = reader.read_dataset_raw("data").unwrap();
17794        assert_eq!(readback, raw);
17795
17796        std::fs::remove_file(&path).ok();
17797    }
17798
17799    #[test]
17800    fn create_multiple_datasets() {
17801        let path = temp_path("multi");
17802
17803        let writer = Hdf5Writer::create(&path).unwrap();
17804
17805        let idx0 = writer
17806            .create_dataset("ints", DatatypeMessage::i32_type(), &[3])
17807            .unwrap();
17808        let i_data: Vec<u8> = [10i32, 20, 30]
17809            .iter()
17810            .flat_map(|v| v.to_le_bytes())
17811            .collect();
17812        writer.write_dataset_raw(idx0, &i_data).unwrap();
17813
17814        let idx1 = writer
17815            .create_dataset("floats", DatatypeMessage::f32_type(), &[2, 2])
17816            .unwrap();
17817        let f_data: Vec<u8> = [1.0f32, 2.0, 3.0, 4.0]
17818            .iter()
17819            .flat_map(|v| v.to_le_bytes())
17820            .collect();
17821        writer.write_dataset_raw(idx1, &f_data).unwrap();
17822
17823        writer.close().unwrap();
17824
17825        let mut reader = Hdf5Reader::open(&path).unwrap();
17826        let names = reader.dataset_names();
17827        assert!(names.contains(&"ints"));
17828        assert!(names.contains(&"floats"));
17829        assert_eq!(reader.dataset_shape("ints").unwrap(), vec![3]);
17830        assert_eq!(reader.dataset_shape("floats").unwrap(), vec![2, 2]);
17831        assert_eq!(reader.read_dataset_raw("ints").unwrap(), i_data);
17832        assert_eq!(reader.read_dataset_raw("floats").unwrap(), f_data);
17833
17834        std::fs::remove_file(&path).ok();
17835    }
17836
17837    #[test]
17838    fn data_size_mismatch() {
17839        let path = temp_path("mismatch");
17840
17841        let writer = Hdf5Writer::create(&path).unwrap();
17842        let idx = writer
17843            .create_dataset("x", DatatypeMessage::u8_type(), &[4])
17844            .unwrap();
17845        let err = writer.write_dataset_raw(idx, &[1, 2, 3]); // 3 bytes instead of 4
17846        assert!(err.is_err());
17847
17848        std::fs::remove_file(&path).ok();
17849    }
17850
17851    #[test]
17852    fn create_chunked_dataset_simple() {
17853        let path = temp_path("chunked_simple");
17854
17855        let writer = Hdf5Writer::create(&path).unwrap();
17856        let idx = writer
17857            .create_chunked_dataset(
17858                "data",
17859                DatatypeMessage::f64_type(),
17860                &[0, 4],        // start empty
17861                &[u64::MAX, 4], // unlimited first dim
17862                &[1, 4],        // chunk = [1, 4]
17863            )
17864            .unwrap();
17865
17866        // Write 3 frames (chunks)
17867        for frame in 0..3u64 {
17868            let values: Vec<f64> = (0..4).map(|i| (frame * 4 + i) as f64).collect();
17869            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17870            writer.write_chunk(idx, frame, &raw).unwrap();
17871        }
17872
17873        // Extend dimensions
17874        writer.extend_dataset(idx, &[3, 4]).unwrap();
17875
17876        writer.close().unwrap();
17877
17878        // Read back
17879        let mut reader = Hdf5Reader::open(&path).unwrap();
17880        assert_eq!(reader.dataset_names(), vec!["data"]);
17881        assert_eq!(reader.dataset_shape("data").unwrap(), vec![3, 4]);
17882
17883        let raw = reader.read_dataset_raw("data").unwrap();
17884        let values: Vec<f64> = raw
17885            .chunks(8)
17886            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
17887            .collect();
17888        assert_eq!(values.len(), 12);
17889        for (i, val) in values.iter().enumerate() {
17890            assert_eq!(*val, i as f64);
17891        }
17892
17893        std::fs::remove_file(&path).ok();
17894    }
17895
17896    #[test]
17897    fn chunked_dataset_many_frames() {
17898        let path = temp_path("chunked_many");
17899
17900        let writer = Hdf5Writer::create(&path).unwrap();
17901        let idx = writer
17902            .create_chunked_dataset(
17903                "frames",
17904                DatatypeMessage::i32_type(),
17905                &[0, 2],
17906                &[u64::MAX, 2],
17907                &[1, 2],
17908            )
17909            .unwrap();
17910
17911        let n_frames = 10u64;
17912        for frame in 0..n_frames {
17913            let values = [(frame * 2) as i32, (frame * 2 + 1) as i32];
17914            let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
17915            writer.write_chunk(idx, frame, &raw).unwrap();
17916        }
17917
17918        writer.extend_dataset(idx, &[n_frames, 2]).unwrap();
17919        writer.close().unwrap();
17920
17921        // Read back
17922        let mut reader = Hdf5Reader::open(&path).unwrap();
17923        assert_eq!(reader.dataset_shape("frames").unwrap(), vec![10, 2]);
17924
17925        let raw = reader.read_dataset_raw("frames").unwrap();
17926        let values: Vec<i32> = raw
17927            .chunks(4)
17928            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
17929            .collect();
17930        assert_eq!(values.len(), 20);
17931        for (i, val) in values.iter().enumerate() {
17932            assert_eq!(*val, i as i32);
17933        }
17934
17935        std::fs::remove_file(&path).ok();
17936    }
17937
17938    #[test]
17939    fn create_fixed_array_dataset_roundtrip() {
17940        let path = temp_path("fixed_array");
17941
17942        let writer = Hdf5Writer::create(&path).unwrap();
17943        let idx = writer
17944            .create_fixed_array_dataset(
17945                "grid",
17946                DatatypeMessage::i32_type(),
17947                &[4, 6], // 4x6 grid
17948                &[2, 3], // chunk = 2x3
17949            )
17950            .unwrap();
17951
17952        // Write all chunks: 2x2 = 4 chunks
17953        // chunk (0,0): rows 0-1, cols 0-2
17954        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
17955            .iter()
17956            .flat_map(|v| v.to_le_bytes())
17957            .collect();
17958        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
17959
17960        // chunk (0,1): rows 0-1, cols 3-5
17961        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
17962            .iter()
17963            .flat_map(|v| v.to_le_bytes())
17964            .collect();
17965        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
17966
17967        // chunk (1,0): rows 2-3, cols 0-2
17968        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
17969            .iter()
17970            .flat_map(|v| v.to_le_bytes())
17971            .collect();
17972        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
17973
17974        // chunk (1,1): rows 2-3, cols 3-5
17975        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
17976            .iter()
17977            .flat_map(|v| v.to_le_bytes())
17978            .collect();
17979        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
17980
17981        writer.close().unwrap();
17982
17983        // Read back
17984        let mut reader = Hdf5Reader::open(&path).unwrap();
17985        assert_eq!(reader.dataset_names(), vec!["grid"]);
17986        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
17987
17988        let raw = reader.read_dataset_raw("grid").unwrap();
17989        let values: Vec<i32> = raw
17990            .chunks(4)
17991            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
17992            .collect();
17993        assert_eq!(values.len(), 24);
17994        for (i, val) in values.iter().enumerate() {
17995            assert_eq!(*val, i as i32);
17996        }
17997
17998        std::fs::remove_file(&path).ok();
17999    }
18000
18001    #[test]
18002    fn fixed_array_paged_dblk_disk_size() {
18003        let ctx = FormatContext {
18004            sizeof_addr: 8,
18005            sizeof_size: 8,
18006        };
18007        // 1024 elements per page (bits=10). 3000 chunks => 3 pages.
18008        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 3000);
18009        assert!(hdr.is_paged());
18010        assert_eq!(hdr.npages(), 3);
18011        // prefix: 4+1+1+8 + bitmap(1) + cksum(4) = 19
18012        // elements: 3000 * 8 = 24000 ; per-page cksum: 3 * 4 = 12
18013        assert_eq!(fixed_array_dblk_disk_size(&ctx, &hdr), 19 + 24000 + 12);
18014
18015        // Non-paged: 1000 elements. prefix(14) + 1000*8 + cksum(4).
18016        let small = FixedArrayHeader::new_for_chunks(&ctx, 1000);
18017        assert!(!small.is_paged());
18018        assert_eq!(fixed_array_dblk_disk_size(&ctx, &small), 14 + 8000 + 4);
18019    }
18020
18021    #[test]
18022    fn fixed_array_paged_encode_matches_reader_layout() {
18023        let ctx = FormatContext {
18024            sizeof_addr: 8,
18025            sizeof_size: 8,
18026        };
18027        let mut hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18028        hdr.data_blk_addr = 0x9000;
18029        let npages = hdr.npages() as usize; // ceil(2500/1024) = 3
18030
18031        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18032        for (i, e) in dblk.elements.iter_mut().enumerate() {
18033            *e = 0x10000 + (i as u64) * 0x100;
18034        }
18035
18036        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18037        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18038
18039        // Decode the prefix and pages exactly as the reader does.
18040        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18041        assert_eq!(prefix.header_addr, 0x1000);
18042        for p in 0..npages {
18043            assert!(prefix.page_initialized(p), "page {p} should be initialized");
18044        }
18045
18046        let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
18047        let page_stride = dblk_page_nelmts * 8 + 4;
18048        let mut recovered = Vec::new();
18049        for p in 0..npages {
18050            let page_nelmts = if p + 1 == npages {
18051                2500 - p * dblk_page_nelmts
18052            } else {
18053                dblk_page_nelmts
18054            };
18055            let off = prefix.prefix_size + p * page_stride;
18056            let page_buf = &encoded[off..];
18057            let addrs = crate::format::chunk_index::fixed_array::decode_unfiltered_page(
18058                page_buf,
18059                &ctx,
18060                page_nelmts,
18061            )
18062            .unwrap();
18063            recovered.extend(addrs);
18064        }
18065        assert_eq!(recovered, dblk.elements);
18066    }
18067
18068    #[test]
18069    fn fixed_array_paged_decode_roundtrip_with_uninitialized_page() {
18070        let ctx = FormatContext {
18071            sizeof_addr: 8,
18072            sizeof_size: 8,
18073        };
18074        let hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18075        let npages = hdr.npages() as usize; // 3
18076        let page = hdr.dblk_page_nelmts() as usize; // 1024
18077
18078        // Populate pages 0 and 2; leave page 1 entirely undefined so its
18079        // bitmap bit stays clear on encode.
18080        let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18081        for i in (0..page).chain(2 * page..2500) {
18082            dblk.elements[i] = 0x10000 + (i as u64) * 0x100;
18083        }
18084
18085        let mut encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18086        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18087        assert!(prefix.page_initialized(0));
18088        assert!(!prefix.page_initialized(1));
18089        assert!(prefix.page_initialized(2));
18090
18091        // Corrupt the uninitialized page's bytes the way libhdf5 leaves
18092        // them: arbitrary, no valid checksum. Decode must not look at it.
18093        let page_stride = page * 8 + 4;
18094        let p1 = prefix.prefix_size + page_stride;
18095        for b in &mut encoded[p1..p1 + page_stride] {
18096            *b = 0x5A;
18097        }
18098
18099        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, 0).unwrap();
18100        assert_eq!(decoded.elements, dblk.elements);
18101        assert_eq!(decoded.header_addr, 0x1000);
18102    }
18103
18104    #[test]
18105    fn fixed_array_paged_decode_filtered_roundtrip() {
18106        let ctx = FormatContext {
18107            sizeof_addr: 8,
18108            sizeof_size: 8,
18109        };
18110        let chunk_size_len = 4usize;
18111        let hdr = FixedArrayHeader::new_for_filtered_chunks(&ctx, 1500, chunk_size_len as u8);
18112        assert!(hdr.is_paged());
18113
18114        let mut dblk = FixedArrayDataBlock::new_filtered(0x2000, 1500);
18115        for (i, e) in dblk.filtered_elements.iter_mut().enumerate() {
18116            e.address = 0x8000 + (i as u64) * 0x40;
18117            e.chunk_size = 100 + i as u64;
18118            e.filter_mask = (i % 3) as u32;
18119        }
18120
18121        let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18122        assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18123        let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, chunk_size_len).unwrap();
18124        assert_eq!(decoded.filtered_elements, dblk.filtered_elements);
18125        assert_eq!(decoded.client_id, FA_CLIENT_FILT_CHUNK);
18126    }
18127
18128    #[test]
18129    fn create_fixed_array_paged_dataset_roundtrip() {
18130        let path = temp_path("fixed_array_paged");
18131
18132        // 1D dataset of 3000 elements, chunk size 1 => 3000 chunks.
18133        // 3000 > 1024 (one page) => the FA data block must be paged.
18134        let n: usize = 3000;
18135        let writer = Hdf5Writer::create(&path).unwrap();
18136        let idx = writer
18137            .create_fixed_array_dataset("paged", DatatypeMessage::i32_type(), &[n as u64], &[1])
18138            .unwrap();
18139
18140        for i in 0..n {
18141            let v = (i as i32).to_le_bytes();
18142            writer
18143                .write_chunk_fixed_array(idx, &[i as u64], &v)
18144                .unwrap();
18145        }
18146        writer.close().unwrap();
18147
18148        let mut reader = Hdf5Reader::open(&path).unwrap();
18149        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18150        let raw = reader.read_dataset_raw("paged").unwrap();
18151        let values: Vec<i32> = raw
18152            .chunks(4)
18153            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18154            .collect();
18155        assert_eq!(values.len(), n);
18156        for (i, v) in values.iter().enumerate() {
18157            assert_eq!(*v, i as i32, "element {i}");
18158        }
18159
18160        std::fs::remove_file(&path).ok();
18161    }
18162
18163    #[cfg(feature = "deflate")]
18164    #[test]
18165    fn create_filtered_fixed_array_dataset_roundtrip() {
18166        // Small compressed fixed-shape chunked dataset: flat filtered FA.
18167        let path = temp_path("fixed_array_filt");
18168
18169        let writer = Hdf5Writer::create(&path).unwrap();
18170        let idx = writer
18171            .create_fixed_array_dataset_with_pipeline(
18172                "grid",
18173                DatatypeMessage::i32_type(),
18174                &[4, 6], // 4x6 grid
18175                &[2, 3], // chunk = 2x3 => 2x2 = 4 chunks
18176                FilterPipeline::deflate(6),
18177            )
18178            .unwrap();
18179
18180        let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18181            .iter()
18182            .flat_map(|v| v.to_le_bytes())
18183            .collect();
18184        writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18185        let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18186            .iter()
18187            .flat_map(|v| v.to_le_bytes())
18188            .collect();
18189        writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
18190        let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
18191            .iter()
18192            .flat_map(|v| v.to_le_bytes())
18193            .collect();
18194        writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
18195        let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
18196            .iter()
18197            .flat_map(|v| v.to_le_bytes())
18198            .collect();
18199        writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
18200
18201        writer.close().unwrap();
18202
18203        let mut reader = Hdf5Reader::open(&path).unwrap();
18204        assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18205        let raw = reader.read_dataset_raw("grid").unwrap();
18206        let values: Vec<i32> = raw
18207            .chunks(4)
18208            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18209            .collect();
18210        assert_eq!(values.len(), 24);
18211        for (i, v) in values.iter().enumerate() {
18212            assert_eq!(*v, i as i32, "element {i}");
18213        }
18214
18215        std::fs::remove_file(&path).ok();
18216    }
18217
18218    #[cfg(feature = "deflate")]
18219    #[test]
18220    fn create_filtered_fixed_array_paged_dataset_roundtrip() {
18221        // Large compressed fixed-shape chunked dataset (>1024 chunks): the
18222        // filtered FA data block must be paged.
18223        let path = temp_path("fixed_array_filt_paged");
18224
18225        let n: usize = 3000;
18226        let writer = Hdf5Writer::create(&path).unwrap();
18227        let idx = writer
18228            .create_fixed_array_dataset_with_pipeline(
18229                "paged",
18230                DatatypeMessage::i32_type(),
18231                &[n as u64],
18232                &[1],
18233                FilterPipeline::deflate(6),
18234            )
18235            .unwrap();
18236
18237        for i in 0..n {
18238            let v = (i as i32).to_le_bytes();
18239            writer
18240                .write_chunk_fixed_array(idx, &[i as u64], &v)
18241                .unwrap();
18242        }
18243        writer.close().unwrap();
18244
18245        let mut reader = Hdf5Reader::open(&path).unwrap();
18246        assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18247        let raw = reader.read_dataset_raw("paged").unwrap();
18248        let values: Vec<i32> = raw
18249            .chunks(4)
18250            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18251            .collect();
18252        assert_eq!(values.len(), n);
18253        for (i, v) in values.iter().enumerate() {
18254            assert_eq!(*v, i as i32, "element {i}");
18255        }
18256
18257        std::fs::remove_file(&path).ok();
18258    }
18259
18260    #[test]
18261    fn filtered_fixed_array_dblk_disk_size_and_encode() {
18262        // Cross-check filtered FA data-block sizing against the encoded length,
18263        // for both flat and paged layouts.
18264        let ctx = FormatContext {
18265            sizeof_addr: 8,
18266            sizeof_size: 8,
18267        };
18268        let csl = 3u8; // chunk_size_len
18269        let elem_size = 8 + csl as usize + 4; // addr + size + filter_mask
18270
18271        // Flat: 100 chunks. prefix(14) + 100*elem_size + cksum(4).
18272        let mut flat = FixedArrayHeader::new_for_filtered_chunks(&ctx, 100, csl);
18273        flat.data_blk_addr = 0x4000;
18274        assert!(!flat.is_paged());
18275        assert_eq!(
18276            fixed_array_dblk_disk_size(&ctx, &flat),
18277            (14 + 100 * elem_size + 4) as u64
18278        );
18279        let flat_dblk = FixedArrayDataBlock::new_filtered(0x1000, 100);
18280        assert_eq!(
18281            encode_fixed_array_dblk(&ctx, &flat, &flat_dblk).len() as u64,
18282            fixed_array_dblk_disk_size(&ctx, &flat)
18283        );
18284
18285        // Paged: 2500 chunks => 3 pages. prefix(4+1+1+8+1+4=19)
18286        // + 2500*elem_size + 3*cksum(4).
18287        let mut paged = FixedArrayHeader::new_for_filtered_chunks(&ctx, 2500, csl);
18288        paged.data_blk_addr = 0x9000;
18289        assert!(paged.is_paged());
18290        assert_eq!(paged.npages(), 3);
18291        assert_eq!(
18292            fixed_array_dblk_disk_size(&ctx, &paged),
18293            (19 + 2500 * elem_size + 12) as u64
18294        );
18295        let mut paged_dblk = FixedArrayDataBlock::new_filtered(0x1000, 2500);
18296        for (i, e) in paged_dblk.filtered_elements.iter_mut().enumerate() {
18297            e.address = 0x10000 + (i as u64) * 0x100;
18298            e.chunk_size = (i % 200) as u64;
18299        }
18300        let encoded = encode_fixed_array_dblk(&ctx, &paged, &paged_dblk);
18301        assert_eq!(
18302            encoded.len() as u64,
18303            fixed_array_dblk_disk_size(&ctx, &paged)
18304        );
18305
18306        // Decode the paged prefix + pages as the reader does.
18307        let npages = paged.npages() as usize;
18308        let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18309        for p in 0..npages {
18310            assert!(prefix.page_initialized(p), "page {p}");
18311        }
18312        let dblk_page_nelmts = paged.dblk_page_nelmts() as usize;
18313        let page_stride = dblk_page_nelmts * elem_size + 4;
18314        let mut recovered = Vec::new();
18315        for p in 0..npages {
18316            let page_nelmts = if p + 1 == npages {
18317                2500 - p * dblk_page_nelmts
18318            } else {
18319                dblk_page_nelmts
18320            };
18321            let off = prefix.prefix_size + p * page_stride;
18322            let elems = crate::format::chunk_index::fixed_array::decode_filtered_page(
18323                &encoded[off..],
18324                &ctx,
18325                page_nelmts,
18326                csl as usize,
18327            )
18328            .unwrap();
18329            recovered.extend(elems);
18330        }
18331        assert_eq!(recovered, paged_dblk.filtered_elements);
18332    }
18333
18334    #[test]
18335    fn create_btree_v2_dataset_roundtrip() {
18336        let path = temp_path("btree_v2");
18337
18338        let writer = Hdf5Writer::create(&path).unwrap();
18339        let idx = writer
18340            .create_btree_v2_dataset(
18341                "data",
18342                DatatypeMessage::f64_type(),
18343                &[0, 0],               // start empty
18344                &[u64::MAX, u64::MAX], // both dims unlimited
18345                &[2, 3],               // chunk = 2x3
18346            )
18347            .unwrap();
18348
18349        // Write chunks for a 4x6 dataset
18350        // chunk (0,0)
18351        let c00: Vec<u8> = [0.0f64, 1.0, 2.0, 6.0, 7.0, 8.0]
18352            .iter()
18353            .flat_map(|v| v.to_le_bytes())
18354            .collect();
18355        writer.write_chunk_btree_v2(idx, &[0, 0], &c00).unwrap();
18356
18357        // chunk (0,1)
18358        let c01: Vec<u8> = [3.0f64, 4.0, 5.0, 9.0, 10.0, 11.0]
18359            .iter()
18360            .flat_map(|v| v.to_le_bytes())
18361            .collect();
18362        writer.write_chunk_btree_v2(idx, &[0, 1], &c01).unwrap();
18363
18364        // chunk (1,0)
18365        let c10: Vec<u8> = [12.0f64, 13.0, 14.0, 18.0, 19.0, 20.0]
18366            .iter()
18367            .flat_map(|v| v.to_le_bytes())
18368            .collect();
18369        writer.write_chunk_btree_v2(idx, &[1, 0], &c10).unwrap();
18370
18371        // chunk (1,1)
18372        let c11: Vec<u8> = [15.0f64, 16.0, 17.0, 21.0, 22.0, 23.0]
18373            .iter()
18374            .flat_map(|v| v.to_le_bytes())
18375            .collect();
18376        writer.write_chunk_btree_v2(idx, &[1, 1], &c11).unwrap();
18377
18378        writer.extend_dataset(idx, &[4, 6]).unwrap();
18379        writer.close().unwrap();
18380
18381        // Read back
18382        let mut reader = Hdf5Reader::open(&path).unwrap();
18383        assert_eq!(reader.dataset_names(), vec!["data"]);
18384        assert_eq!(reader.dataset_shape("data").unwrap(), vec![4, 6]);
18385
18386        let raw = reader.read_dataset_raw("data").unwrap();
18387        let values: Vec<f64> = raw
18388            .chunks(8)
18389            .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
18390            .collect();
18391        assert_eq!(values.len(), 24);
18392        for (i, val) in values.iter().enumerate() {
18393            assert_eq!(*val, i as f64);
18394        }
18395
18396        std::fs::remove_file(&path).ok();
18397    }
18398
18399    /// Bytes one chunk of [`btree_v2_flush_probe`]'s dataset occupies — an
18400    /// f64 element, so the allocator's alignment neither pads nor merges it and
18401    /// the file's growth is exactly the bytes asked for.
18402    const BT2_PROBE_CHUNK: u64 = 8;
18403
18404    /// Write chunks of a 1x1-chunked 2-D BT2 dataset, flushing at each batch
18405    /// boundary, and report `(node addresses, file length)` after every flush.
18406    /// Chunks are addressed down column 0 so the record count — and hence the
18407    /// tree's shape — grows one record at a time.
18408    fn btree_v2_flush_probe(path: &std::path::Path, batches: &[u64]) -> Vec<(Vec<u64>, u64)> {
18409        let writer = Hdf5Writer::create(path).unwrap();
18410        let idx = writer
18411            .create_btree_v2_dataset(
18412                "data",
18413                DatatypeMessage::f64_type(),
18414                &[0, 0],
18415                &[u64::MAX, u64::MAX],
18416                &[1, 1],
18417            )
18418            .unwrap();
18419        let mut written = 0u64;
18420        let mut out = Vec::new();
18421        for &upto in batches {
18422            while written < upto {
18423                writer
18424                    .write_chunk_btree_v2(idx, &[written, 0], &(written as f64).to_le_bytes())
18425                    .unwrap();
18426                written += 1;
18427            }
18428            writer.flush_dataset(idx).unwrap();
18429            let addrs = writer
18430                .ds(idx)
18431                .lock()
18432                .btree_v2
18433                .as_ref()
18434                .unwrap()
18435                .node_addrs
18436                .clone();
18437            out.push((addrs, std::fs::metadata(path).unwrap().len()));
18438        }
18439        writer.extend_dataset(idx, &[written.max(1), 1]).unwrap();
18440        writer.close().unwrap();
18441        out
18442    }
18443
18444    /// The node pool tracks the tree in both directions. Dropping records is
18445    /// what a removal path would do — [`Bt2ChunkIndex`] has none today, so the
18446    /// test drops them itself — and the flush that follows must hand the blocks
18447    /// its smaller tree no longer needs back to the allocator instead of
18448    /// leaving them recorded and unreachable.
18449    #[test]
18450    fn a_btree_v2_flush_frees_the_node_blocks_its_tree_gave_up() {
18451        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
18452
18453        let path = temp_path("bt2_node_shrink");
18454        let writer = Hdf5Writer::create(&path).unwrap();
18455        let idx = writer
18456            .create_btree_v2_dataset(
18457                "data",
18458                DatatypeMessage::f64_type(),
18459                &[0, 0],
18460                &[u64::MAX, u64::MAX],
18461                &[1, 1],
18462            )
18463            .unwrap();
18464        // 85 records is one past a leaf, so the tree is two leaves and a root.
18465        for i in 0..85u64 {
18466            writer
18467                .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18468                .unwrap();
18469        }
18470        writer.flush_dataset(idx).unwrap();
18471        let grown = writer
18472            .ds(idx)
18473            .lock()
18474            .btree_v2
18475            .as_ref()
18476            .unwrap()
18477            .node_addrs
18478            .clone();
18479        assert_eq!(grown.len(), 3, "expected two leaves and a root");
18480
18481        // Back to 84 records: one leaf, so two of the three blocks are surplus.
18482        writer
18483            .ds(idx)
18484            .lock()
18485            .btree_v2
18486            .as_mut()
18487            .unwrap()
18488            .index
18489            .records
18490            .truncate(84);
18491        writer.flush_dataset(idx).unwrap();
18492        let shrunk = writer
18493            .ds(idx)
18494            .lock()
18495            .btree_v2
18496            .as_ref()
18497            .unwrap()
18498            .node_addrs
18499            .clone();
18500        assert_eq!(
18501            shrunk,
18502            grown[..1],
18503            "the pool still records the surplus blocks"
18504        );
18505
18506        // The surplus went back to the allocator, not on the floor: the next
18507        // node-sized allocation lands inside the region the two blocks covered.
18508        let reused = writer
18509            .allocator
18510            .allocate(BT2_NODE_SIZE as u64, FreeSpaceClass::Metadata);
18511        assert!(
18512            (grown[1]..grown[1] + 2 * BT2_NODE_SIZE as u64).contains(&reused),
18513            "a node block allocated at {reused:#x}, outside the freed \
18514             [{:#x}, {:#x}) the flush gave up",
18515            grown[1],
18516            grown[1] + 2 * BT2_NODE_SIZE as u64
18517        );
18518
18519        writer.extend_dataset(idx, &[85, 1]).unwrap();
18520        writer.close().unwrap();
18521        std::fs::remove_file(&path).ok();
18522    }
18523
18524    /// A v2 B-tree whose header declares a non-default node size — libhdf5
18525    /// built with a different `H5D_BT2_NODE_SIZE`, or any other writer —
18526    /// reopens for append: the reconstruction adopts the header's node_size,
18527    /// split and merge instead of refusing everything but 2048, and the next
18528    /// flush re-serializes at that size (upstream allocates every node at
18529    /// `hdr->node_size`, H5B2leaf.c / H5B2internal.c).
18530    #[test]
18531    fn a_btree_v2_with_a_foreign_node_size_reopens_and_grows() {
18532        let path = temp_path("bt2_foreign_node_size");
18533        {
18534            let writer = Hdf5Writer::create(&path).unwrap();
18535            let idx = writer
18536                .create_btree_v2_dataset(
18537                    "data",
18538                    DatatypeMessage::f64_type(),
18539                    &[0, 0],
18540                    &[u64::MAX, u64::MAX],
18541                    &[1, 1],
18542                )
18543                .unwrap();
18544            // Act as a foreign writer: 512-byte nodes, non-default tuning.
18545            // record_size 24 => a 512-byte leaf holds 20 records, so 85
18546            // records make a depth-1 tree of 512-byte blocks.
18547            {
18548                let ds = writer.ds(idx);
18549                let mut m = ds.lock();
18550                let index = &mut m.btree_v2.as_mut().unwrap().index;
18551                index.node_size = 512;
18552                index.split_percent = 90;
18553                index.merge_percent = 30;
18554            }
18555            for i in 0..85u64 {
18556                writer
18557                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18558                    .unwrap();
18559            }
18560            writer.extend_dataset(idx, &[85, 1]).unwrap();
18561            writer.close().unwrap();
18562        }
18563        {
18564            let writer = Hdf5Writer::open_append(&path).unwrap();
18565            let idx = writer.dataset_index("data").unwrap();
18566            {
18567                let ds = writer.ds(idx);
18568                let m = ds.lock();
18569                let index = &m.btree_v2.as_ref().unwrap().index;
18570                assert_eq!(index.node_size, 512, "header node_size not adopted");
18571                assert_eq!(index.split_percent, 90);
18572                assert_eq!(index.merge_percent, 30);
18573                assert_eq!(index.records.len(), 85, "records not walked back");
18574            }
18575            for i in 85..115u64 {
18576                writer
18577                    .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
18578                    .unwrap();
18579            }
18580            writer.extend_dataset(idx, &[115, 1]).unwrap();
18581            writer.close().unwrap();
18582        }
18583
18584        let mut reader = Hdf5Reader::open(&path).unwrap();
18585        let raw = reader.read_dataset_raw("data").unwrap();
18586        let values: Vec<f64> = raw
18587            .chunks(8)
18588            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
18589            .collect();
18590        assert_eq!(values.len(), 115);
18591        for (i, v) in values.iter().enumerate() {
18592            assert_eq!(*v, i as f64, "element {i}");
18593        }
18594        std::fs::remove_file(&path).ok();
18595    }
18596
18597    /// A node's record count falls as well as rises: the tree's first leaf goes
18598    /// from a full 84 records to 42 when 85 records force it to split. The node
18599    /// image is padded to the whole block so re-serializing overwrites the
18600    /// block, not a prefix of it — otherwise that leaf keeps the tail of its
18601    /// 84-record self, stale records sitting in a live node block.
18602    #[test]
18603    fn a_shrinking_btree_v2_node_leaves_no_stale_records_behind() {
18604        use crate::format::chunk_index::btree_v2::{Bt2ChunkIndex, BT2_NODE_SIZE};
18605
18606        let path = temp_path("bt2_node_blocks");
18607        let probe = btree_v2_flush_probe(&path, &[84, 85]);
18608        let node0 = probe.last().unwrap().0[0];
18609
18610        // What the first leaf holds once the tree has split.
18611        let ctx = FormatContext {
18612            sizeof_addr: 8,
18613            sizeof_size: 8,
18614        };
18615        let mut index = Bt2ChunkIndex::new_unfiltered(2);
18616        for i in 0..85u64 {
18617            index.insert(vec![i, 0], 0);
18618        }
18619        let tree = index.build_tree(&ctx);
18620        assert!(
18621            tree.nodes[0].num_records < 84,
18622            "this test needs the first leaf to shrink, got {}",
18623            tree.nodes[0].num_records
18624        );
18625        // signature(4) + version(1) + type(1) + records + checksum(4)
18626        let used = 10 + tree.nodes[0].num_records as usize * tree.record_size as usize;
18627
18628        let bytes = std::fs::read(&path).unwrap();
18629        let block = &bytes[node0 as usize..node0 as usize + BT2_NODE_SIZE as usize];
18630        assert!(
18631            block[used..].iter().all(|&b| b == 0),
18632            "leaf block at {node0:#x} still holds {} bytes of its previous, larger image",
18633            block[used..].iter().rposition(|&b| b != 0).unwrap_or(0) + 1
18634        );
18635        std::fs::remove_file(&path).ok();
18636    }
18637
18638    /// The node pool is the single owner of the tree's block addresses: a flush
18639    /// reuses every block already in it and allocates only the shortfall. So
18640    /// re-flushing an unchanged index must cost nothing, and a flush that grows
18641    /// the tree must cost exactly the blocks it added — anything more means a
18642    /// block was stranded.
18643    #[test]
18644    fn a_btree_v2_flush_allocates_only_the_node_blocks_it_adds() {
18645        use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
18646
18647        let path = temp_path("bt2_pool_growth");
18648        // Re-flush at 84 (still one leaf), then cross into a three-node depth-1
18649        // tree, then keep growing.
18650        let batches = [84u64, 84, 85, 200, 200];
18651        let probe = btree_v2_flush_probe(&path, &batches);
18652        for i in 1..probe.len() {
18653            let (prev_addrs, prev_len) = &probe[i - 1];
18654            let (addrs, len) = &probe[i];
18655            assert!(
18656                addrs.starts_with(prev_addrs),
18657                "flush {i} moved a node block instead of reusing it"
18658            );
18659            let new_blocks = (addrs.len() - prev_addrs.len()) as u64 * BT2_NODE_SIZE as u64;
18660            let new_chunks = (batches[i] - batches[i - 1]) * BT2_PROBE_CHUNK;
18661            assert_eq!(
18662                len - prev_len,
18663                new_blocks + new_chunks,
18664                "flush {i} grew the file by more than the blocks it added"
18665            );
18666        }
18667        // The unchanged re-flushes must be free.
18668        assert_eq!(probe[1].1, probe[0].1);
18669        assert_eq!(probe[4].1, probe[3].1);
18670        std::fs::remove_file(&path).ok();
18671    }
18672
18673    #[cfg(feature = "parallel")]
18674    #[test]
18675    fn parallel_batch_write_roundtrip() {
18676        let path = temp_path("parallel_batch");
18677
18678        let writer = Hdf5Writer::create(&path).unwrap();
18679        let idx = writer
18680            .create_chunked_dataset(
18681                "data",
18682                DatatypeMessage::i32_type(),
18683                &[0, 4],
18684                &[u64::MAX, 4],
18685                &[1, 4],
18686            )
18687            .unwrap();
18688
18689        // Prepare chunks
18690        let chunks_data: Vec<(u64, Vec<u8>)> = (0..8u64)
18691            .map(|frame| {
18692                let values: Vec<i32> = (0..4).map(|i| (frame * 4 + i) as i32).collect();
18693                let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18694                (frame, raw)
18695            })
18696            .collect();
18697
18698        let batch: Vec<(u64, &[u8])> = chunks_data
18699            .iter()
18700            .map(|(idx, data)| (*idx, data.as_slice()))
18701            .collect();
18702
18703        writer.write_chunks_batch(idx, &batch).unwrap();
18704        writer.extend_dataset(idx, &[8, 4]).unwrap();
18705        writer.close().unwrap();
18706
18707        // Read back
18708        let mut reader = Hdf5Reader::open(&path).unwrap();
18709        assert_eq!(reader.dataset_shape("data").unwrap(), vec![8, 4]);
18710        let raw = reader.read_dataset_raw("data").unwrap();
18711        let values: Vec<i32> = raw
18712            .chunks(4)
18713            .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18714            .collect();
18715        assert_eq!(values.len(), 32);
18716        for (i, val) in values.iter().enumerate() {
18717            assert_eq!(*val, i as i32);
18718        }
18719
18720        std::fs::remove_file(&path).ok();
18721    }
18722
18723    #[test]
18724    fn swmr_writer_append_frames() {
18725        use crate::io::swmr::SwmrWriter;
18726
18727        // Per-call unique path so concurrent cargo invocations and
18728        // kernel-side flock release races cannot collide.
18729        use std::sync::atomic::{AtomicU64, Ordering};
18730        static COUNTER: AtomicU64 = AtomicU64::new(0);
18731        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18732        let path = std::env::temp_dir().join(format!(
18733            "rust_hdf5_swmr_append_{}_{}.h5",
18734            std::process::id(),
18735            n
18736        ));
18737
18738        let mut swmr = SwmrWriter::create(&path).unwrap();
18739        let idx = swmr
18740            .create_streaming_dataset("detector", DatatypeMessage::u16_type(), &[4, 4])
18741            .unwrap();
18742
18743        swmr.start_swmr().unwrap();
18744
18745        // Append 5 frames
18746        for frame in 0..5u16 {
18747            let data: Vec<u16> = (0..16).map(|i| frame * 16 + i).collect();
18748            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18749            swmr.append_frame(idx, &raw).unwrap();
18750        }
18751
18752        swmr.flush().unwrap();
18753        swmr.close().unwrap();
18754
18755        // Read back
18756        let mut reader = Hdf5Reader::open(&path).unwrap();
18757        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![5, 4, 4]);
18758
18759        let raw = reader.read_dataset_raw("detector").unwrap();
18760        let values: Vec<u16> = raw
18761            .chunks(2)
18762            .map(|chunk| u16::from_le_bytes(chunk.try_into().unwrap()))
18763            .collect();
18764        assert_eq!(values.len(), 80); // 5 * 4 * 4
18765                                      // Verify first frame
18766        for (i, val) in values.iter().enumerate().take(16) {
18767            assert_eq!(*val, i as u16);
18768        }
18769        // Verify last frame
18770        for (i, val) in values[64..80].iter().enumerate() {
18771            assert_eq!(*val, 4 * 16 + i as u16);
18772        }
18773
18774        std::fs::remove_file(&path).ok();
18775    }
18776
18777    #[test]
18778    fn swmr_writer_tiled_frames() {
18779        use crate::io::swmr::SwmrWriter;
18780        use std::sync::atomic::{AtomicU64, Ordering};
18781        static COUNTER: AtomicU64 = AtomicU64::new(0);
18782        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18783        let path = std::env::temp_dir().join(format!(
18784            "rust_hdf5_swmr_tiled_{}_{}.h5",
18785            std::process::id(),
18786            n
18787        ));
18788
18789        let mut swmr = SwmrWriter::create(&path).unwrap();
18790        // 4x4 frames, tiled into 2x2 chunks -> 4 chunks per frame.
18791        let idx = swmr
18792            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[4, 4], &[2, 2])
18793            .unwrap();
18794        swmr.start_swmr().unwrap();
18795
18796        for frame in 0..3u16 {
18797            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
18798            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18799            swmr.append_frame(idx, &raw).unwrap();
18800        }
18801        swmr.flush().unwrap();
18802        swmr.close().unwrap();
18803
18804        let mut reader = Hdf5Reader::open(&path).unwrap();
18805        assert_eq!(reader.dataset_shape("det").unwrap(), vec![3, 4, 4]);
18806        let raw = reader.read_dataset_raw("det").unwrap();
18807        let values: Vec<u16> = raw
18808            .chunks(2)
18809            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18810            .collect();
18811        assert_eq!(values.len(), 48);
18812        // Every element must survive the frame -> tile split and the
18813        // tile -> frame reassembly on read.
18814        for frame in 0..3u16 {
18815            for i in 0..16usize {
18816                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
18817            }
18818        }
18819        std::fs::remove_file(&path).ok();
18820    }
18821
18822    /// A chunk tile larger than the frame is geometry libhdf5 refuses to
18823    /// create (`H5D__chunk_construct`: chunk must not exceed a fixed maximum
18824    /// dimension), so no libhdf5-based writer — including the NDFileHDF5
18825    /// tiling controls this API mirrors — can produce such a file. Until
18826    /// 0.4.1 we accepted it and zero-padded the frame up to the tile; now
18827    /// the create is rejected like every other creator's.
18828    #[test]
18829    fn swmr_writer_tiled_chunk_larger_than_frame_is_rejected() {
18830        use crate::io::swmr::SwmrWriter;
18831        use std::sync::atomic::{AtomicU64, Ordering};
18832        static COUNTER: AtomicU64 = AtomicU64::new(0);
18833        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18834        let path = std::env::temp_dir().join(format!(
18835            "rust_hdf5_swmr_bigchunk_{}_{}.h5",
18836            std::process::id(),
18837            n
18838        ));
18839
18840        let mut swmr = SwmrWriter::create(&path).unwrap();
18841        let err = swmr
18842            .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[3, 3], &[8, 8])
18843            .unwrap_err();
18844        assert!(
18845            err.to_string().contains("maximum dimension size"),
18846            "unexpected error: {err}"
18847        );
18848        swmr.close().unwrap();
18849        std::fs::remove_file(&path).ok();
18850    }
18851
18852    #[test]
18853    fn swmr_writer_multi_frame_chunks() {
18854        use crate::io::swmr::SwmrWriter;
18855        use std::sync::atomic::{AtomicU64, Ordering};
18856        static COUNTER: AtomicU64 = AtomicU64::new(0);
18857        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18858        let path = std::env::temp_dir().join(format!(
18859            "rust_hdf5_swmr_mfc_{}_{}.h5",
18860            std::process::id(),
18861            n
18862        ));
18863
18864        // 3x3 frames, chunk = 4 frames x full frame. 10 frames -> 3 bands
18865        // of 4, 4, 2 (the last band partial).
18866        let mut swmr = SwmrWriter::create(&path).unwrap();
18867        let idx = swmr
18868            .create_streaming_dataset_chunked(
18869                "det",
18870                DatatypeMessage::u16_type(),
18871                &[3, 3],
18872                &[4, 3, 3],
18873            )
18874            .unwrap();
18875        swmr.start_swmr().unwrap();
18876        for frame in 0..10u16 {
18877            let data: Vec<u16> = (0..9).map(|i| frame * 100 + i).collect();
18878            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18879            swmr.append_frame(idx, &raw).unwrap();
18880        }
18881        swmr.flush().unwrap();
18882        swmr.close().unwrap();
18883
18884        let mut reader = Hdf5Reader::open(&path).unwrap();
18885        // The partial last band must not over-extend the frame count.
18886        assert_eq!(reader.dataset_shape("det").unwrap(), vec![10, 3, 3]);
18887        let raw = reader.read_dataset_raw("det").unwrap();
18888        let values: Vec<u16> = raw
18889            .chunks(2)
18890            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18891            .collect();
18892        assert_eq!(values.len(), 90);
18893        for frame in 0..10u16 {
18894            for i in 0..9usize {
18895                assert_eq!(values[frame as usize * 9 + i], frame * 100 + i as u16);
18896            }
18897        }
18898        std::fs::remove_file(&path).ok();
18899    }
18900
18901    #[test]
18902    fn swmr_writer_multi_frame_tiled_chunks() {
18903        use crate::io::swmr::SwmrWriter;
18904        use std::sync::atomic::{AtomicU64, Ordering};
18905        static COUNTER: AtomicU64 = AtomicU64::new(0);
18906        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18907        let path = std::env::temp_dir().join(format!(
18908            "rust_hdf5_swmr_mftc_{}_{}.h5",
18909            std::process::id(),
18910            n
18911        ));
18912
18913        // 4x4 frames, chunk = 2 frames x 2x2 tiles. 5 frames -> bands of
18914        // 2, 2, 1; every frame is also split into a 2x2 tile grid.
18915        let mut swmr = SwmrWriter::create(&path).unwrap();
18916        let idx = swmr
18917            .create_streaming_dataset_chunked(
18918                "det",
18919                DatatypeMessage::u16_type(),
18920                &[4, 4],
18921                &[2, 2, 2],
18922            )
18923            .unwrap();
18924        swmr.start_swmr().unwrap();
18925        for frame in 0..5u16 {
18926            let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
18927            let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
18928            swmr.append_frame(idx, &raw).unwrap();
18929        }
18930        swmr.flush().unwrap();
18931        swmr.close().unwrap();
18932
18933        let mut reader = Hdf5Reader::open(&path).unwrap();
18934        assert_eq!(reader.dataset_shape("det").unwrap(), vec![5, 4, 4]);
18935        let raw = reader.read_dataset_raw("det").unwrap();
18936        let values: Vec<u16> = raw
18937            .chunks(2)
18938            .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
18939            .collect();
18940        assert_eq!(values.len(), 80);
18941        for frame in 0..5u16 {
18942            for i in 0..16usize {
18943                assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
18944            }
18945        }
18946        std::fs::remove_file(&path).ok();
18947    }
18948
18949    #[cfg(feature = "deflate")]
18950    #[test]
18951    fn swmr_writer_compressed_frames() {
18952        use crate::io::swmr::SwmrWriter;
18953        use std::sync::atomic::{AtomicU64, Ordering};
18954        static COUNTER: AtomicU64 = AtomicU64::new(0);
18955        let n = COUNTER.fetch_add(1, Ordering::Relaxed);
18956        let path = std::env::temp_dir().join(format!(
18957            "rust_hdf5_swmr_comp_{}_{}.h5",
18958            std::process::id(),
18959            n
18960        ));
18961
18962        let mut swmr = SwmrWriter::create(&path).unwrap();
18963        let pipeline = crate::format::messages::filter::FilterPipeline::deflate(4);
18964        let idx = swmr
18965            .create_streaming_dataset_compressed(
18966                "detector",
18967                DatatypeMessage::i32_type(),
18968                &[8],
18969                pipeline,
18970            )
18971            .unwrap();
18972        swmr.start_swmr().unwrap();
18973
18974        for frame in 0..40i32 {
18975            let raw: Vec<u8> = (0..8).flat_map(|i| (frame * 8 + i).to_le_bytes()).collect();
18976            swmr.append_frame(idx, &raw).unwrap();
18977            if frame % 7 == 0 {
18978                swmr.flush().unwrap();
18979            }
18980        }
18981        swmr.flush().unwrap();
18982        swmr.close().unwrap();
18983
18984        let mut reader = Hdf5Reader::open(&path).unwrap();
18985        assert_eq!(reader.dataset_shape("detector").unwrap(), vec![40, 8]);
18986        let raw = reader.read_dataset_raw("detector").unwrap();
18987        let values: Vec<i32> = raw
18988            .chunks(4)
18989            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18990            .collect();
18991        assert_eq!(values, (0..320).collect::<Vec<i32>>());
18992
18993        std::fs::remove_file(&path).ok();
18994    }
18995
18996    #[test]
18997    fn group_hierarchy_writer_reader() {
18998        let path = temp_path("group_hierarchy");
18999
19000        let writer = Hdf5Writer::create(&path).unwrap();
19001
19002        // Create groups
19003        let g0 = writer.create_group("/", "group1").unwrap();
19004        let g1 = writer.create_group("/group1", "sub").unwrap();
19005        assert_eq!(g0, 0);
19006        assert_eq!(g1, 1);
19007
19008        // Create datasets
19009        let ds_root = writer
19010            .create_dataset("root_data", DatatypeMessage::f64_type(), &[2])
19011            .unwrap();
19012        let raw_root: Vec<u8> = [1.0f64, 2.0].iter().flat_map(|v| v.to_le_bytes()).collect();
19013        writer.write_dataset_raw(ds_root, &raw_root).unwrap();
19014
19015        let ds_g0 = writer
19016            .create_dataset("group1/data", DatatypeMessage::i32_type(), &[3])
19017            .unwrap();
19018        let raw_g0: Vec<u8> = [10i32, 20, 30]
19019            .iter()
19020            .flat_map(|v| v.to_le_bytes())
19021            .collect();
19022        writer.write_dataset_raw(ds_g0, &raw_g0).unwrap();
19023
19024        let ds_g1 = writer
19025            .create_dataset("group1/sub/values", DatatypeMessage::u8_type(), &[4])
19026            .unwrap();
19027        writer.write_dataset_raw(ds_g1, &[1u8, 2, 3, 4]).unwrap();
19028
19029        writer.close().unwrap();
19030
19031        // Read back
19032        let mut reader = Hdf5Reader::open(&path).unwrap();
19033        let names = reader.dataset_names();
19034        assert!(names.contains(&"root_data"), "names: {:?}", names);
19035        assert!(names.contains(&"group1/data"), "names: {:?}", names);
19036        assert!(names.contains(&"group1/sub/values"), "names: {:?}", names);
19037
19038        let raw = reader.read_dataset_raw("root_data").unwrap();
19039        let vals: Vec<f64> = raw
19040            .chunks(8)
19041            .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19042            .collect();
19043        assert_eq!(vals, vec![1.0, 2.0]);
19044
19045        let raw = reader.read_dataset_raw("group1/data").unwrap();
19046        let vals: Vec<i32> = raw
19047            .chunks(4)
19048            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19049            .collect();
19050        assert_eq!(vals, vec![10, 20, 30]);
19051
19052        let raw = reader.read_dataset_raw("group1/sub/values").unwrap();
19053        assert_eq!(raw, vec![1, 2, 3, 4]);
19054
19055        std::fs::remove_file(&path).ok();
19056    }
19057
19058    /// libhdf5 (`H5D__chunk_construct`) rejects a chunk dimension that
19059    /// exceeds a fixed maximum dimension. Before this check, such a dataset
19060    /// was created and appends landed rows at the chunk stride instead of
19061    /// the row stride, reading back [1, 2, 0, 0] for [1, 2, 3, 4].
19062    #[test]
19063    fn create_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
19064        let path = temp_path("chunk_wider_than_max");
19065
19066        let writer = Hdf5Writer::create(&path).unwrap();
19067        let err = writer
19068            .create_chunked_dataset(
19069                "data",
19070                DatatypeMessage::f64_type(),
19071                &[0, 2],
19072                &[u64::MAX, 2],
19073                &[2, 4],
19074            )
19075            .unwrap_err();
19076        assert!(
19077            err.to_string().contains("maximum dimension size"),
19078            "unexpected error: {err}"
19079        );
19080
19081        // The fixed-array creators derive the maximum from the fixed dims.
19082        let err = writer
19083            .create_fixed_array_dataset("fa", DatatypeMessage::f64_type(), &[3], &[5])
19084            .unwrap_err();
19085        assert!(
19086            err.to_string().contains("maximum dimension size"),
19087            "unexpected error: {err}"
19088        );
19089
19090        writer.close().unwrap();
19091        std::fs::remove_file(&path).ok();
19092    }
19093
19094    /// libhdf5 exempts a dimension whose *current* size is zero from the
19095    /// chunk-vs-maximum check (`curr_dims[u] &&` in `H5D__chunk_construct`),
19096    /// and rejects a zero chunk dimension on every path.
19097    #[test]
19098    fn create_mirrors_the_libhdf5_chunk_geometry_exemptions() {
19099        let path = temp_path("chunk_geometry_exemptions");
19100
19101        let writer = Hdf5Writer::create(&path).unwrap();
19102        // dims[1] == 0: chunk 4 > max 2 is allowed, as libhdf5 allows it.
19103        writer
19104            .create_chunked_dataset(
19105                "exempt",
19106                DatatypeMessage::f64_type(),
19107                &[0, 0],
19108                &[u64::MAX, 2],
19109                &[2, 4],
19110            )
19111            .unwrap();
19112
19113        let err = writer
19114            .create_chunked_dataset("zero", DatatypeMessage::f64_type(), &[0], &[u64::MAX], &[0])
19115            .unwrap_err();
19116        assert!(
19117            err.to_string().contains("chunk dimension 0 is zero"),
19118            "unexpected error: {err}"
19119        );
19120
19121        writer.close().unwrap();
19122        std::fs::remove_file(&path).ok();
19123    }
19124
19125    /// A file written by 0.4.0 can carry a chunk row wider than the frame
19126    /// row — create now rejects that geometry, but reopened files keep it.
19127    /// Appends must scatter frames at the chunk stride, not pack them at
19128    /// the frame stride (which read back `[1, 2, 0, 0]` for `[1, 2, 3, 4]`).
19129    /// The wide shape is simulated by widening the registered chunk dims
19130    /// after create, which also lands in the layout message at close.
19131    #[test]
19132    fn append_scatters_into_a_legacy_wider_than_row_chunk() {
19133        let path = temp_path("legacy_wide_chunk_append");
19134
19135        let writer = Hdf5Writer::create(&path).unwrap();
19136        let idx = writer
19137            .create_chunked_dataset(
19138                "data",
19139                DatatypeMessage::i32_type(),
19140                &[0, 2],
19141                &[u64::MAX, 2],
19142                &[2, 2],
19143            )
19144            .unwrap();
19145        writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims = vec![2, 4];
19146
19147        let frames: Vec<u8> = [1i32, 2, 3, 4]
19148            .iter()
19149            .flat_map(|v| v.to_le_bytes())
19150            .collect();
19151        writer.write_append_frames(idx, 0, 2, &frames).unwrap();
19152        writer.extend_dataset(idx, &[2, 2]).unwrap();
19153        writer.close().unwrap();
19154
19155        let mut reader = Hdf5Reader::open(&path).unwrap();
19156        assert_eq!(reader.dataset_shape("data").unwrap(), vec![2, 2]);
19157        let raw = reader.read_dataset_raw("data").unwrap();
19158        let values: Vec<i32> = raw
19159            .chunks(4)
19160            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19161            .collect();
19162        assert_eq!(values, vec![1, 2, 3, 4]);
19163        std::fs::remove_file(&path).ok();
19164    }
19165
19166    /// The compressed vlen creator sizes its chunked layout from a
19167    /// caller-supplied chunk size; it goes through the same geometry
19168    /// validation as every other creator (empty inputs are exempt because
19169    /// their current size is zero).
19170    #[test]
19171    #[cfg(feature = "deflate")]
19172    fn compressed_vlen_create_validates_its_chunk_size() {
19173        use crate::format::messages::filter::FilterPipeline;
19174        let path = temp_path("vlen_compressed_chunk");
19175
19176        let writer = Hdf5Writer::create(&path).unwrap();
19177        let err = writer
19178            .create_vlen_string_dataset_compressed(
19179                "texts",
19180                &["a", "b", "c"],
19181                100,
19182                FilterPipeline::deflate(6),
19183            )
19184            .unwrap_err();
19185        assert!(
19186            err.to_string().contains("maximum dimension size"),
19187            "unexpected error: {err}"
19188        );
19189
19190        writer
19191            .create_vlen_string_dataset_compressed("empty", &[], 16, FilterPipeline::deflate(6))
19192            .unwrap();
19193
19194        writer.close().unwrap();
19195        std::fs::remove_file(&path).ok();
19196    }
19197
19198    /// `set_libver_latest` moves *filtered* chunked datasets to layout v5 with
19199    /// fixed 8-byte chunk-size fields; unfiltered chunked and pre-opt-in
19200    /// datasets keep v4 with the derived width, matching libhdf5's
19201    /// `version_perf` rule (only the filtered index arms bump to 5).
19202    #[cfg(feature = "deflate")]
19203    #[test]
19204    fn libver_latest_selects_v5_for_filtered_chunks_only() {
19205        let path = temp_path("libver_v5_select");
19206
19207        let mut writer = Hdf5Writer::create(&path).unwrap();
19208        let before = writer
19209            .create_chunked_dataset_with_pipeline(
19210                "d4",
19211                DatatypeMessage::i32_type(),
19212                &[0],
19213                &[u64::MAX],
19214                &[16],
19215                FilterPipeline::deflate(4),
19216            )
19217            .unwrap();
19218        writer.set_libver_latest(true).unwrap();
19219        let ea5 = writer
19220            .create_chunked_dataset_with_pipeline(
19221                "ea5",
19222                DatatypeMessage::i32_type(),
19223                &[0],
19224                &[u64::MAX],
19225                &[16],
19226                FilterPipeline::deflate(4),
19227            )
19228            .unwrap();
19229        let plain = writer
19230            .create_chunked_dataset(
19231                "plain",
19232                DatatypeMessage::i32_type(),
19233                &[0],
19234                &[u64::MAX],
19235                &[16],
19236            )
19237            .unwrap();
19238        let fa5 = writer
19239            .create_fixed_array_dataset_with_pipeline(
19240                "fa5",
19241                DatatypeMessage::i32_type(),
19242                &[4, 6],
19243                &[2, 3],
19244                FilterPipeline::deflate(6),
19245            )
19246            .unwrap();
19247        let bt5 = writer
19248            .create_btree_v2_dataset_with_pipeline(
19249                "bt5",
19250                DatatypeMessage::i32_type(),
19251                &[0, 0],
19252                &[u64::MAX, u64::MAX],
19253                &[2, 3],
19254                FilterPipeline::deflate(6),
19255            )
19256            .unwrap();
19257
19258        {
19259            let d4 = writer.ds(before);
19260            let d4 = d4.lock();
19261            assert_eq!(d4.layout_version, 4);
19262            assert_eq!(
19263                d4.chunked.as_ref().unwrap().chunk_size_len,
19264                compute_chunk_size_len(16 * 4)
19265            );
19266            let e5 = writer.ds(ea5);
19267            let e5 = e5.lock();
19268            assert_eq!(e5.layout_version, 5);
19269            assert_eq!(e5.chunked.as_ref().unwrap().chunk_size_len, 8);
19270            assert_eq!(writer.ds(plain).lock().layout_version, 4);
19271            assert_eq!(writer.ds(fa5).lock().layout_version, 5);
19272            assert_eq!(writer.ds(bt5).lock().layout_version, 5);
19273        }
19274
19275        // Write through the FA and BT2 v5 indexes so their 8-byte chunk-size
19276        // fields are exercised end to end, not just selected.
19277        for (coords, vals) in [
19278            ([0u64, 0], [0i32, 1, 2, 6, 7, 8]),
19279            ([0, 1], [3, 4, 5, 9, 10, 11]),
19280            ([1, 0], [12, 13, 14, 18, 19, 20]),
19281            ([1, 1], [15, 16, 17, 21, 22, 23]),
19282        ] {
19283            let bytes: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
19284            writer
19285                .write_chunk_fixed_array(fa5, &coords, &bytes)
19286                .unwrap();
19287            writer.write_chunk_btree_v2(bt5, &coords, &bytes).unwrap();
19288        }
19289        writer.extend_dataset(bt5, &[4, 6]).unwrap();
19290        writer.close().unwrap();
19291
19292        let mut reader = Hdf5Reader::open(&path).unwrap();
19293        for name in ["fa5", "bt5"] {
19294            let raw = reader.read_dataset_raw(name).unwrap();
19295            let values: Vec<i32> = raw
19296                .chunks(4)
19297                .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19298                .collect();
19299            assert_eq!(values, (0..24).collect::<Vec<i32>>(), "dataset {name}");
19300        }
19301
19302        std::fs::remove_file(&path).ok();
19303    }
19304
19305    /// A v5 file reopened for append must stay v5: the decode → `DatasetInfo`
19306    /// → finalize path carries the version through, so the re-encoded layout
19307    /// message matches the 8-byte size fields the filtered index was built
19308    /// with. A silent v4 downgrade here would make libhdf5 derive a narrower
19309    /// field width than the index uses.
19310    #[cfg(feature = "deflate")]
19311    #[test]
19312    fn v5_layout_survives_reopen_and_append() {
19313        let path = temp_path("libver_v5_reopen");
19314        let chunk: usize = 8;
19315
19316        let mut writer = Hdf5Writer::create(&path).unwrap();
19317        writer.set_libver_latest(true).unwrap();
19318        let idx = writer
19319            .create_chunked_dataset_with_pipeline(
19320                "d",
19321                DatatypeMessage::i32_type(),
19322                &[0],
19323                &[u64::MAX],
19324                &[chunk as u64],
19325                FilterPipeline::deflate(4),
19326            )
19327            .unwrap();
19328        for c in 0..2u64 {
19329            let data: Vec<u8> = (0..chunk as i32)
19330                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
19331                .collect();
19332            writer.write_chunk(idx, c, &data).unwrap();
19333        }
19334        writer.extend_dataset(idx, &[2 * chunk as u64]).unwrap();
19335        writer.close().unwrap();
19336
19337        // Reopen: the decoded layout version must be preserved, and appends
19338        // must keep working against the 8-byte-size-field index.
19339        let writer = Hdf5Writer::open_append(&path).unwrap();
19340        assert_eq!(writer.ds(0).lock().layout_version, 5);
19341        for c in 2..4u64 {
19342            let data: Vec<u8> = (0..chunk as i32)
19343                .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
19344                .collect();
19345            writer.write_chunk(0, c, &data).unwrap();
19346        }
19347        writer.extend_dataset(0, &[4 * chunk as u64]).unwrap();
19348        writer.close().unwrap();
19349
19350        // Still v5 after the second finalize, and fully readable.
19351        let writer = Hdf5Writer::open_append(&path).unwrap();
19352        assert_eq!(writer.ds(0).lock().layout_version, 5);
19353        writer.close().unwrap();
19354
19355        let mut reader = Hdf5Reader::open(&path).unwrap();
19356        let raw = reader.read_dataset_raw("d").unwrap();
19357        let values: Vec<i32> = raw
19358            .chunks(4)
19359            .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19360            .collect();
19361        assert_eq!(values, (0..4 * chunk as i32).collect::<Vec<i32>>());
19362
19363        std::fs::remove_file(&path).ok();
19364    }
19365
19366    /// A chunk strictly larger than `u32::MAX` bytes forces layout v5 with no
19367    /// opt-in — v4's size field cannot represent it — while a chunk of exactly
19368    /// `u32::MAX` bytes stays v4, matching libhdf5's `version_req` boundary
19369    /// (`> 0xffffffff`, filtered or not).
19370    #[test]
19371    fn oversized_chunk_forces_v5_without_opt_in() {
19372        let path = temp_path("libver_4gib_force");
19373
19374        let writer = Hdf5Writer::create(&path).unwrap();
19375        let at_limit = writer
19376            .create_chunked_dataset_with_pipeline(
19377                "at_limit",
19378                DatatypeMessage::u8_type(),
19379                &[0],
19380                &[u64::MAX],
19381                &[u32::MAX as u64],
19382                FilterPipeline::deflate(4),
19383            )
19384            .unwrap();
19385        let over = writer
19386            .create_chunked_dataset_with_pipeline(
19387                "over",
19388                DatatypeMessage::u8_type(),
19389                &[0],
19390                &[u64::MAX],
19391                &[u32::MAX as u64 + 1],
19392                FilterPipeline::deflate(4),
19393            )
19394            .unwrap();
19395        let over_unfiltered = writer
19396            .create_chunked_dataset(
19397                "over_plain",
19398                DatatypeMessage::u8_type(),
19399                &[0],
19400                &[u64::MAX],
19401                &[u32::MAX as u64 + 1],
19402            )
19403            .unwrap();
19404
19405        assert_eq!(writer.ds(at_limit).lock().layout_version, 4);
19406        {
19407            let ds = writer.ds(over);
19408            let ds = ds.lock();
19409            assert_eq!(ds.layout_version, 5);
19410            assert_eq!(ds.chunked.as_ref().unwrap().chunk_size_len, 8);
19411        }
19412        assert_eq!(writer.ds(over_unfiltered).lock().layout_version, 5);
19413        writer.close().unwrap();
19414        std::fs::remove_file(&path).ok();
19415    }
19416
19417    /// SWMR reaches version 3 on its own, without a chunked dataset to raise
19418    /// the bound — through the flags `finalize_for_swmr` passes, and then
19419    /// through `swmr_active` for every superblock written after it. Only a
19420    /// file with nothing else newer in it can tell the two arms apart, and
19421    /// the public SWMR API always creates a chunked streaming dataset.
19422    #[test]
19423    fn swmr_reaches_version_3_with_no_chunked_dataset_in_the_file() {
19424        let path = temp_path("swmr_superblock");
19425
19426        let mut writer = Hdf5Writer::create(&path).unwrap();
19427        writer
19428            .create_dataset("d", DatatypeMessage::i32_type(), &[2])
19429            .unwrap();
19430        assert_eq!(writer.superblock_version_for(0), SUPERBLOCK_V2);
19431
19432        writer.finalize_for_swmr().unwrap();
19433        // What `start_swmr` does after finalizing, and what lets a second
19434        // handle read the file while this writer lives — the writer's
19435        // exclusive lock is mandatory on Windows.
19436        writer.handle().release_lock().unwrap();
19437        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
19438
19439        // The close-time finalize carries no SWMR flag; the file is still an
19440        // SWMR file and must not be handed back a version older than the one
19441        // its readers attached to.
19442        writer.close().unwrap();
19443        assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
19444        std::fs::remove_file(&path).ok();
19445    }
19446
19447    /// A named bound below `H5F_LIBVER_V110` refuses the session instead —
19448    /// the two checks `H5F__start_swmr_write` opens with, a version-3
19449    /// superblock (H5Fint.c:3814) and a low bound of at least V110
19450    /// (H5Fint.c:3818). Naming no bound at all is what the test above does,
19451    /// and that file is free to become version 3.
19452    #[test]
19453    fn a_named_bound_below_v110_refuses_an_swmr_session() {
19454        for bound in [LibverBound::Earliest, LibverBound::V18] {
19455            let path = temp_path(&format!("swmr_refused_{bound:?}"));
19456            let mut writer = Hdf5Writer::create_with_options(
19457                &path,
19458                FileCreateOptions {
19459                    libver: Some(bound),
19460                    ..Default::default()
19461                },
19462            )
19463            .unwrap();
19464            writer
19465                .create_dataset("d", DatatypeMessage::i32_type(), &[2])
19466                .unwrap();
19467
19468            let err = writer.finalize_for_swmr().unwrap_err().to_string();
19469            assert!(err.contains("SWMR"), "{bound:?}: {err}");
19470            assert!(err.contains("H5F_LIBVER_V110"), "{bound:?}: {err}");
19471
19472            // Refused, not half-done: nothing was published, and the close
19473            // writes the file the bound asked for.
19474            writer.close().unwrap();
19475            let version = std::fs::read(&path).unwrap()[8];
19476            assert_eq!(version, bound.superblock_version(), "{bound:?}");
19477            std::fs::remove_file(&path).ok();
19478        }
19479    }
19480
19481    /// After every writer of a dataset object header, `nlink_written` is the
19482    /// count that writer encoded.
19483    ///
19484    /// `header_stale_with` is the one authority for "does the on-disk header
19485    /// still describe this dataset?", and it reads `nlink_written`; the three
19486    /// writers — `finalize`, `finalize_for_swmr` and
19487    /// `write_dataset_header_inplace` — therefore all record through
19488    /// `DatasetInfo::header_written`. This walks the SWMR sequence, where the
19489    /// in-place writer is the one that could drift, and pins why it does not:
19490    /// a name added after the publish grows the header past the block it was
19491    /// published into, so the rewrite is refused rather than half-applied and
19492    /// the count on disk stays the one the registry names.
19493    #[test]
19494    fn every_dataset_header_write_records_its_link_count() {
19495        let path = temp_path("header_write_records_nlink");
19496        let writer = Hdf5Writer::create(&path).unwrap();
19497        let idx = writer
19498            .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
19499            .unwrap();
19500        let mut writer = writer;
19501        writer.finalize_for_swmr().unwrap();
19502        assert_eq!(
19503            writer.ds(idx).lock().nlink_written,
19504            1,
19505            "the SWMR publish put one name in the header"
19506        );
19507        writer.write_dataset_header_inplace(idx).unwrap();
19508        assert_eq!(writer.ds(idx).lock().nlink_written, 1);
19509
19510        // A second name after the publish: the reference-count message it
19511        // adds does not fit the published block.
19512        writer.create_hard_link("/", "alias", "d").unwrap();
19513        assert_eq!(writer.object_link_count(HardLinkTarget::Dataset(idx)), 2);
19514        let grew = writer
19515            .write_dataset_header_inplace(idx)
19516            .unwrap_err()
19517            .to_string();
19518        assert!(
19519            grew.contains("cannot rewrite in place"),
19520            "a header that outgrew its block must be refused: {grew}"
19521        );
19522        assert_eq!(
19523            writer.ds(idx).lock().nlink_written,
19524            1,
19525            "a refused rewrite leaves the registry describing the header the file holds"
19526        );
19527
19528        // The close-time finalize is the writer that commits the second name,
19529        // and a reopen reads the same count back off the link graph.
19530        writer.close().unwrap();
19531        let writer = Hdf5Writer::open_append(&path).unwrap();
19532        assert_eq!(
19533            writer.ds(0).lock().nlink_written,
19534            2,
19535            "finalize wrote two names and the reopen reads two"
19536        );
19537        writer.close().unwrap();
19538        std::fs::remove_file(&path).ok();
19539    }
19540
19541    /// `H5D__chunk_set_info`'s `version_req` (H5Dchunk.c:909, :936): version 5
19542    /// is required for a chunk over 4 GiB — the version-4 layout message's
19543    /// stored-size field is 32 bits and cannot record one — and
19544    /// `LAYOUT_VERSION_DEFAULT` (3, `H5O_LAYOUT_VERSION_DEFAULT`) is the floor
19545    /// for everything at or under that limit. Pure arithmetic on the byte
19546    /// count: no chunk is ever allocated.
19547    #[test]
19548    fn required_chunk_layout_version_pins_5_past_4_gib() {
19549        assert_eq!(
19550            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64),
19551            LAYOUT_VERSION_DEFAULT
19552        );
19553        assert_eq!(
19554            Hdf5Writer::required_chunk_layout_version(u32::MAX as u64 + 1),
19555            5
19556        );
19557    }
19558
19559    /// `H5D__chunk_set_info`'s index-selection gate (H5Dchunk.c:936): a chunk
19560    /// over 4 GiB reaches the v1.10 chunk indexes even under a bound whose
19561    /// `H5O_layout_ver_bounds` row (`LibverBound::layout_version`) is below
19562    /// 4 — `V18` (row 3) and `Earliest` (row 1) both normally keep an
19563    /// ordinary chunk on the version-1 B-tree, but
19564    /// `required_chunk_layout_version`'s own escape to 5 overrides that row
19565    /// for this one chunk. The default bound (`V110`, row 4) already crosses
19566    /// the threshold on its own, so it is asserted only as the baseline, not
19567    /// as a distinguishing case for the escape.
19568    #[test]
19569    fn uses_v110_chunk_indexing_escapes_past_4_gib_at_every_bound() {
19570        let over_4gib = u32::MAX as u64 + 1;
19571        let small = 1024u64;
19572
19573        let path = temp_path("uses_v110_default");
19574        let writer = Hdf5Writer::create(&path).unwrap();
19575        assert!(writer.uses_v110_chunk_indexing(small));
19576        assert!(writer.uses_v110_chunk_indexing(over_4gib));
19577        writer.close().unwrap();
19578        std::fs::remove_file(&path).ok();
19579
19580        let path = temp_path("uses_v110_v18");
19581        let mut writer = Hdf5Writer::create(&path).unwrap();
19582        writer.set_libver_bound(LibverBound::V18).unwrap();
19583        assert!(
19584            !writer.uses_v110_chunk_indexing(small),
19585            "V18's layout row (3) stays below the v1.10 gate for an ordinary chunk"
19586        );
19587        assert!(
19588            writer.uses_v110_chunk_indexing(over_4gib),
19589            "the >4 GiB escape reaches v1.10 indexing despite V18's row"
19590        );
19591        writer.close().unwrap();
19592        std::fs::remove_file(&path).ok();
19593
19594        let path = temp_path("uses_v110_earliest");
19595        let mut writer = Hdf5Writer::create(&path).unwrap();
19596        writer.set_libver_bound(LibverBound::Earliest).unwrap();
19597        assert!(
19598            !writer.uses_v110_chunk_indexing(small),
19599            "Earliest's layout row (1) stays below the v1.10 gate for an ordinary chunk"
19600        );
19601        assert!(
19602            writer.uses_v110_chunk_indexing(over_4gib),
19603            "the >4 GiB escape reaches v1.10 indexing despite Earliest's row"
19604        );
19605        writer.close().unwrap();
19606        std::fs::remove_file(&path).ok();
19607    }
19608
19609    /// `H5D__chunk_set_info`'s closing `MAX3` (H5Dchunk.c:1046): the same
19610    /// escape pins the layout message itself at version 5 for a chunk over
19611    /// 4 GiB regardless of bound — `required_chunk_layout_version` dominates
19612    /// the max chain ahead of both the bound-derived preference and
19613    /// `LAYOUT_VERSION_DEFAULT`.
19614    #[test]
19615    fn chunk_layout_version_pins_5_past_4_gib_at_every_bound() {
19616        let over_4gib = u32::MAX as u64 + 1;
19617        let small = 1024u64;
19618
19619        let path = temp_path("chunk_ver_default");
19620        let writer = Hdf5Writer::create(&path).unwrap();
19621        assert_eq!(writer.chunk_layout_version(false, small), 4);
19622        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19623        writer.close().unwrap();
19624        std::fs::remove_file(&path).ok();
19625
19626        let path = temp_path("chunk_ver_v18");
19627        let mut writer = Hdf5Writer::create(&path).unwrap();
19628        writer.set_libver_bound(LibverBound::V18).unwrap();
19629        assert_eq!(writer.chunk_layout_version(false, small), 3);
19630        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19631        writer.close().unwrap();
19632        std::fs::remove_file(&path).ok();
19633
19634        let path = temp_path("chunk_ver_earliest");
19635        let mut writer = Hdf5Writer::create(&path).unwrap();
19636        writer.set_libver_bound(LibverBound::Earliest).unwrap();
19637        assert_eq!(
19638            writer.chunk_layout_version(false, small),
19639            LAYOUT_VERSION_DEFAULT
19640        );
19641        assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
19642        writer.close().unwrap();
19643        std::fs::remove_file(&path).ok();
19644    }
19645    /// `fsm_persist.h5` persists two managers — metadata and raw data. The
19646    /// reopen reads both, hands their merged sections to the allocator, and
19647    /// claims the four blocks the managers themselves occupy.
19648    #[test]
19649    fn a_persisting_file_reopens_with_its_free_sections() {
19650        let path = fixture_copy("fsm_persist.h5", "fsm_read");
19651        let writer = Hdf5Writer::open_append(&path).unwrap();
19652        let fs = writer.free_space.as_deref().expect("managers were read");
19653
19654        assert!(fs.info.persist);
19655        assert_eq!(fs.info.strategy, FileSpaceStrategy::FsmAggr);
19656        assert_eq!(fs.info.threshold, 1);
19657
19658        let sections = writer.allocator.free_blocks();
19659        // h5stat -S reports 1910 bytes of tracked free space for this file.
19660        assert_eq!(sections.iter().map(|s| s.1).sum::<u64>(), 1910);
19661        // Address-ordered, and no two sections touch: what the two managers
19662        // held separately came out coalesced.
19663        for w in sections.windows(2) {
19664            assert!(w[0].0 + w[0].1 < w[1].0, "{sections:?}");
19665        }
19666        // Two headers plus the two sections blocks they name.
19667        assert_eq!(fs.superseded.len(), 4);
19668        for &(addr, len) in &fs.superseded {
19669            assert!(len > 0);
19670            assert!(
19671                !sections
19672                    .iter()
19673                    .any(|&(a, l)| addr < a + l && a < addr + len),
19674                "manager block {addr:#x}+{len} sits in a free section"
19675            );
19676        }
19677        drop(writer);
19678        let _ = std::fs::remove_file(&path);
19679    }
19680
19681    /// A file created with non-default file-space properties carries the
19682    /// message that declares them, and one created to persist gets real
19683    /// managers as soon as anything is freed.
19684    #[test]
19685    fn a_created_file_declares_the_strategy_it_was_made_with() {
19686        let path = temp_path("fsm_create");
19687        {
19688            let w = Hdf5Writer::create_with_options(
19689                &path,
19690                FileCreateOptions {
19691                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
19692                    ..Default::default()
19693                },
19694            )
19695            .unwrap();
19696            let i = w
19697                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19698                .unwrap();
19699            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19700            w.close().unwrap();
19701        }
19702
19703        let info = read_only_append(&path)
19704            .free_space
19705            .as_deref()
19706            .expect("the created file declares a strategy")
19707            .info
19708            .clone();
19709        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
19710        assert!(info.persist);
19711        assert_eq!(info.threshold, 1);
19712        assert_eq!(info.page_size, 4096);
19713        // The alignment fragments the creation left behind are the file's
19714        // first free space, so the metadata manager already has an address
19715        // and the raw-data one, which nothing freed into, does not.
19716        assert_ne!(info.fs_addr[0], UNDEF_ADDR);
19717        assert!(info.fs_addr.iter().skip(1).all(|&a| a == UNDEF_ADDR));
19718
19719        // An append supersedes the root header and the extension, and that
19720        // freed space is what the managers now record.
19721        append_one(&path, "added", false);
19722        assert!(
19723            tracked_free_space(&path) > 0,
19724            "the append recorded no free space"
19725        );
19726        let _ = std::fs::remove_file(&path);
19727    }
19728
19729    /// The two strategies without managers, and the default. All three are
19730    /// `H5Pset_file_space_strategy` settings; only the default leaves the file
19731    /// without the message.
19732    #[test]
19733    fn a_strategy_without_managers_still_declares_itself() {
19734        for (strategy, persist) in [
19735            (FileSpaceStrategy::Aggr, true),
19736            (FileSpaceStrategy::None, false),
19737        ] {
19738            let path = temp_path("fsm_nomgr");
19739            {
19740                let w = Hdf5Writer::create_with_options(
19741                    &path,
19742                    FileCreateOptions {
19743                        file_space: FileSpaceConfig::new(strategy, persist, 7),
19744                        ..Default::default()
19745                    },
19746                )
19747                .unwrap();
19748                w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19749                    .unwrap();
19750                w.close().unwrap();
19751            }
19752            // Read through the reader, not the writer: a reopen only builds
19753            // free-space state for a file it will rewrite managers for, and
19754            // these two have none.
19755            let info = declared_file_space(&path).expect("the strategy is declared");
19756            assert_eq!(info.strategy, strategy);
19757            // `H5P__set_file_space_strategy` stores neither for a strategy
19758            // that has no managers, so both keep the library defaults.
19759            assert!(!info.persist);
19760            assert_eq!(info.threshold, 1);
19761            let _ = std::fs::remove_file(&path);
19762        }
19763    }
19764
19765    /// The library defaults are what a file says by saying nothing.
19766    #[test]
19767    fn the_default_strategy_writes_no_message() {
19768        let path = temp_path("fsm_default");
19769        {
19770            let w = Hdf5Writer::create_with_options(
19771                &path,
19772                FileCreateOptions {
19773                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, false, 1),
19774                    ..Default::default()
19775                },
19776            )
19777            .unwrap();
19778            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19779                .unwrap();
19780            w.close().unwrap();
19781        }
19782        assert!(declared_file_space(&path).is_none());
19783        let _ = std::fs::remove_file(&path);
19784    }
19785
19786    /// The file-space info message a file carries, read back the way any
19787    /// reader sees it.
19788    fn declared_file_space(path: &std::path::Path) -> Option<FileSpaceInfoMessage> {
19789        crate::io::reader::Hdf5Reader::open(path)
19790            .unwrap()
19791            .superblock_extension()
19792            .file_space_info
19793            .clone()
19794    }
19795
19796    /// A created paged file is laid out on its page grid: the superblock takes
19797    /// the whole of page zero and the rest of that page is the metadata
19798    /// manager's first section, which is what `H5MF__alloc_pagefs` gives
19799    /// `H5F__super_init`'s `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`.
19800    #[test]
19801    fn a_created_paged_file_lays_its_pages_out() {
19802        let path = temp_path("fsm_paged_created");
19803        {
19804            let w = Hdf5Writer::create_with_options(
19805                &path,
19806                FileCreateOptions {
19807                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
19808                    ..Default::default()
19809                },
19810            )
19811            .unwrap();
19812            let i = w
19813                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19814                .unwrap();
19815            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19816            w.close().unwrap();
19817        }
19818        let info = read_only_append(&path)
19819            .free_space
19820            .as_deref()
19821            .expect("the created file declares a strategy")
19822            .info
19823            .clone();
19824        assert_eq!(info.strategy, FileSpaceStrategy::Page);
19825        assert!(info.persist);
19826        assert_eq!(info.page_size, 4096);
19827        assert_eq!(
19828            std::fs::metadata(&path).unwrap().len() % info.page_size,
19829            0,
19830            "a paged file ends on a page boundary"
19831        );
19832        let _ = std::fs::remove_file(&path);
19833    }
19834
19835    /// A userblock has to be a whole number of pages, or every page boundary
19836    /// after it is off the file's own grid — `H5F__super_init` refuses one
19837    /// that is not (H5Fsuper.c:1182-1192).
19838    #[test]
19839    fn a_paged_file_refuses_a_userblock_smaller_than_its_page() {
19840        let path = temp_path("fsm_paged_userblock");
19841        let Err(err) = Hdf5Writer::create_with_options(
19842            &path,
19843            FileCreateOptions {
19844                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
19845                userblock: 512,
19846                ..Default::default()
19847            },
19848        ) else {
19849            panic!("a 512-byte userblock was accepted on a 4096-byte page");
19850        };
19851        assert!(
19852            format!("{err}").contains("multiple of its 4096-byte"),
19853            "{err}"
19854        );
19855        let _ = std::fs::remove_file(&path);
19856    }
19857
19858    /// A page size the builder names is the page the file is actually laid
19859    /// out in, not just a number the message repeats: every allocation is
19860    /// shaped by it and the file ends on one of its boundaries.
19861    #[test]
19862    fn a_file_created_at_a_non_default_page_size_allocates_by_it() {
19863        let path = temp_path("fsm_page_size_8k");
19864        {
19865            let w = Hdf5Writer::create_with_options(
19866                &path,
19867                FileCreateOptions {
19868                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19869                        .with_page_size(8192),
19870                    ..Default::default()
19871                },
19872            )
19873            .unwrap();
19874            let i = w
19875                .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
19876                .unwrap();
19877            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
19878            w.close().unwrap();
19879        }
19880        let info = read_only_append(&path)
19881            .free_space
19882            .as_deref()
19883            .expect("the created file declares a strategy")
19884            .info
19885            .clone();
19886        assert_eq!(info.page_size, 8192);
19887        assert_eq!(
19888            std::fs::metadata(&path).unwrap().len() % 8192,
19889            0,
19890            "the file ends on one of the pages it was created with"
19891        );
19892        let _ = std::fs::remove_file(&path);
19893    }
19894
19895    /// The page size is the fourth of the four properties `H5F__super_init`
19896    /// compares against the library defaults (H5Fsuper.c:1092-1097), so
19897    /// naming it is on its own enough to give a file the message — under the
19898    /// default strategy, which allocates without it.
19899    #[test]
19900    fn a_non_default_page_size_alone_gives_the_file_a_message() {
19901        let path = temp_path("fsm_page_size_only");
19902        {
19903            let w = Hdf5Writer::create_with_options(
19904                &path,
19905                FileCreateOptions {
19906                    file_space: FileSpaceConfig::default().with_page_size(1024),
19907                    ..Default::default()
19908                },
19909            )
19910            .unwrap();
19911            w.close().unwrap();
19912        }
19913        let info = declared_file_space(&path)
19914            .expect("a file naming only a page size still carries the message");
19915        assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
19916        assert!(!info.persist);
19917        assert_eq!(info.page_size, 1024);
19918        let _ = std::fs::remove_file(&path);
19919    }
19920
19921    /// `H5Pset_file_space_page_size` refuses anything below 512 or above
19922    /// 1 GiB (H5Pfcpl.c:1389-1393), and nothing between: no power of two is
19923    /// required, so a size the bounds admit is one the file may carry.
19924    #[test]
19925    fn a_page_size_outside_the_library_bounds_is_refused() {
19926        for size in [0, 1, 511, PAGE_SIZE_MAX + 1] {
19927            let path = temp_path(&format!("fsm_page_size_bad_{size}"));
19928            let Err(err) = Hdf5Writer::create_with_options(
19929                &path,
19930                FileCreateOptions {
19931                    file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19932                        .with_page_size(size),
19933                    ..Default::default()
19934                },
19935            ) else {
19936                panic!("a {size}-byte file-space page was accepted");
19937            };
19938            assert!(
19939                format!("{err}").contains("between 512 bytes and 1073741824"),
19940                "{err}"
19941            );
19942            let _ = std::fs::remove_file(&path);
19943        }
19944        let path = temp_path("fsm_page_size_odd");
19945        let w = Hdf5Writer::create_with_options(
19946            &path,
19947            FileCreateOptions {
19948                file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
19949                    .with_page_size(513),
19950                ..Default::default()
19951            },
19952        )
19953        .expect("513 is inside the bounds, and no power of two is required");
19954        w.close().unwrap();
19955        let _ = std::fs::remove_file(&path);
19956    }
19957
19958    /// A paged file's managers are read on reopen, the same as any other
19959    /// file's: paged aggregation changes which manager a request maps to, not
19960    /// whether the file has managers to rewrite.
19961    #[test]
19962    fn a_paged_file_reports_the_managers_it_persists() {
19963        let path = fixture_copy("fsm_persist_page.h5", "fsm_read_paged");
19964        let writer = Hdf5Writer::open_append(&path).unwrap();
19965        let fs = writer.free_space.as_deref().expect("no managers read");
19966        assert_eq!(fs.info.strategy, FileSpaceStrategy::Page);
19967        assert!(
19968            !writer.allocator.free_extents().is_empty(),
19969            "the sections the file records were not put back in circulation"
19970        );
19971        drop(writer);
19972        let _ = std::fs::remove_file(&path);
19973    }
19974
19975    /// A file with no file-space info message at all — every file this crate
19976    /// creates — has nothing to read and nothing to write back.
19977    #[test]
19978    fn a_file_without_a_strategy_has_no_managers() {
19979        let path = temp_path("fsm_none");
19980        {
19981            let w = Hdf5Writer::create(&path).unwrap();
19982            w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
19983                .unwrap();
19984            w.close().unwrap();
19985        }
19986        let writer = Hdf5Writer::open_append(&path).unwrap();
19987        assert!(writer.free_space.is_none());
19988        drop(writer);
19989        let _ = std::fs::remove_file(&path);
19990    }
19991    /// Sum of the sections the managers a file names actually hold — what
19992    /// `h5stat -S` prints as "Amount of tracked free space", read back through
19993    /// this crate's own decoder so a test can assert on it. A reopen seeds the
19994    /// allocator with exactly those sections, so its free list is the number.
19995    fn tracked_free_space(path: &std::path::Path) -> u64 {
19996        read_only_append(path)
19997            .allocator
19998            .free_blocks()
19999            .iter()
20000            .map(|b| b.1)
20001            .sum()
20002    }
20003
20004    /// Open for append and mark the writer closed, so dropping it releases the
20005    /// file lock instead of finalizing and rewriting what is being inspected.
20006    fn read_only_append(path: &std::path::Path) -> Hdf5Writer {
20007        let mut w = Hdf5Writer::open_append(path).unwrap();
20008        w.closed = true;
20009        w
20010    }
20011
20012    /// Add one small dataset, the smallest append that still rewrites the root
20013    /// header, the superblock extension and — on a persisting file — the
20014    /// free-space manager.
20015    fn append_one(path: &std::path::Path, name: &str, disable_managers: bool) {
20016        let mut w = Hdf5Writer::open_append(path).unwrap();
20017        if disable_managers {
20018            // Both halves of the change, so the control is the file as this
20019            // crate wrote it before: the session neither allocates from the
20020            // recorded sections nor writes any back.
20021            w.free_space = None;
20022            w.allocator.reset_free_list(&[]);
20023        }
20024        let i = w
20025            .create_dataset(name, DatatypeMessage::i32_type(), &[8])
20026            .unwrap();
20027        w.write_dataset_raw(
20028            i,
20029            &(0..8i32).flat_map(|v| v.to_le_bytes()).collect::<Vec<u8>>(),
20030        )
20031        .unwrap();
20032        w.close().unwrap();
20033    }
20034
20035    /// The block list a reopen carries for the superblock extension covers
20036    /// every chunk of the header, not just the first. The fixture's extension
20037    /// is a two-chunk header — libhdf5 put the file-space info message in a
20038    /// continuation — and freeing chunk zero alone left the continuation
20039    /// allocated with nothing naming it.
20040    #[test]
20041    fn a_reopen_carries_every_chunk_of_the_superblock_extension() {
20042        let path = fixture_copy("fsm_persist.h5", "fsm_ext_chunks");
20043        let blocks = read_only_append(&path).extension.superseded.clone();
20044        assert!(
20045            blocks.len() > 1,
20046            "the fixture's extension is one chunk, so this proves nothing: {blocks:?}"
20047        );
20048        let _ = std::fs::remove_file(&path);
20049    }
20050
20051    /// An append on a persisting file both spends and records the space its
20052    /// managers track: the new dataset comes out of the sections the file
20053    /// already had, and what the rewrite frees goes back into them.
20054    #[test]
20055    fn an_append_reuses_and_records_the_space_the_managers_track() {
20056        let path = fixture_copy("fsm_persist.h5", "fsm_write");
20057        let original = std::fs::metadata(&path).unwrap().len();
20058        let before = tracked_free_space(&path);
20059        assert_eq!(before, 1910, "the fixture's own managers");
20060
20061        append_one(&path, "added", false);
20062        let size = std::fs::metadata(&path).unwrap().len();
20063        let tracked = tracked_free_space(&path);
20064
20065        // Negative control: the same append with both halves of this off — no
20066        // allocating out of the recorded sections and no writing any back —
20067        // which is what this crate did before it read free space at all.
20068        let control = fixture_copy("fsm_persist.h5", "fsm_write_control");
20069        append_one(&control, "added", true);
20070        let control_size = std::fs::metadata(&control).unwrap().len();
20071        assert_eq!(
20072            tracked_free_space(&control),
20073            before,
20074            "with the manager rewrite disabled the number must not move"
20075        );
20076
20077        // The new dataset's raw data comes out of the raw-data sections the
20078        // file already recorded, so the append grows the file by less than the
20079        // same append with the reuse off. It does not stop the growth:
20080        // `H5MF_alloc` asks one manager and no other, and of this fixture's
20081        // 1910 free bytes 1848 are raw-data ones, so the metadata the append
20082        // writes still comes from the end of the file.
20083        assert!(
20084            size < control_size,
20085            "the append took nothing from the {before} bytes free: \
20086             {original} grew to {size}, the control to {control_size}"
20087        );
20088        assert!(
20089            control_size > original,
20090            "the control has to grow or it proves nothing"
20091        );
20092        // Space no manager and no object claims — `h5stat -S`'s "unaccounted
20093        // space" — is what the leak was, and it is smaller now.
20094        assert!(
20095            size - tracked < control_size - before,
20096            "unaccounted space went from {} to {}",
20097            control_size - before,
20098            size - tracked
20099        );
20100
20101        for p in [&path, &control] {
20102            let _ = std::fs::remove_file(p);
20103        }
20104    }
20105
20106    /// The set the writer holds free when it finishes is exactly the set the
20107    /// manager it just wrote records — the invariant that makes the on-disk
20108    /// managers a faithful account of the file's free space.
20109    #[test]
20110    fn the_manager_records_the_free_list_the_close_ends_with() {
20111        let path = fixture_copy("fsm_persist.h5", "fsm_roundtrip");
20112        let internal = {
20113            let mut w = Hdf5Writer::open_append(&path).unwrap();
20114            let i = w
20115                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20116                .unwrap();
20117            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20118            w.finalize(true).unwrap();
20119            let blocks = w.allocator.free_extents();
20120            w.closed = true;
20121            blocks
20122        };
20123        assert!(!internal.is_empty(), "the append freed nothing");
20124
20125        // Classes included: a section read back out of the wrong manager is a
20126        // section libhdf5 would offer to the wrong kind of allocation.
20127        let reread = {
20128            let w = read_only_append(&path);
20129            assert!(w.free_space.is_some(), "managers were written");
20130            w.allocator.free_extents()
20131        };
20132        assert_eq!(internal, reread);
20133        let _ = std::fs::remove_file(&path);
20134    }
20135
20136    /// The paged half of
20137    /// [`the_manager_records_the_free_list_the_close_ends_with`]: a paged
20138    /// file's sections carry a page and a class as well as an address, and a
20139    /// section written into the wrong manager or split across a page boundary
20140    /// would come back different.
20141    #[test]
20142    fn the_manager_records_the_free_list_a_paged_close_ends_with() {
20143        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_roundtrip");
20144        let internal = {
20145            let mut w = Hdf5Writer::open_append(&path).unwrap();
20146            let i = w
20147                .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20148                .unwrap();
20149            w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20150            w.finalize(true).unwrap();
20151            let blocks = w.allocator.free_extents();
20152            w.closed = true;
20153            blocks
20154        };
20155        assert!(!internal.is_empty(), "the append freed nothing");
20156
20157        let reread = {
20158            let w = read_only_append(&path);
20159            assert!(w.free_space.is_some(), "managers were written");
20160            w.allocator.free_extents()
20161        };
20162        assert_eq!(internal, reread);
20163        let _ = std::fs::remove_file(&path);
20164    }
20165
20166    /// Negative control for the paged managers: with the read and the rewrite
20167    /// both off — the file as this crate handled a paged file before — the
20168    /// space the append frees is recorded nowhere, and the number this crate
20169    /// reads back is the fixture's own.
20170    #[test]
20171    fn a_paged_append_records_nothing_without_the_manager_rewrite() {
20172        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_measured");
20173        let control = fixture_copy("fsm_persist_page.h5", "fsm_paged_control");
20174        let before = tracked_free_space(&path);
20175        let original = std::fs::metadata(&path).unwrap().len();
20176
20177        append_one(&path, "added", false);
20178        append_one(&control, "added", true);
20179
20180        assert_eq!(
20181            tracked_free_space(&control),
20182            before,
20183            "the control moved the number it is there to hold still"
20184        );
20185        assert_eq!(
20186            std::fs::metadata(&path).unwrap().len(),
20187            original,
20188            "the append grew a paged file with {before} bytes recorded free"
20189        );
20190        assert!(
20191            std::fs::metadata(&control).unwrap().len() > original,
20192            "the control has to grow or it proves nothing"
20193        );
20194        assert_ne!(
20195            tracked_free_space(&path),
20196            before,
20197            "the managers came back holding what the fixture wrote"
20198        );
20199        for p in [&path, &control] {
20200            let _ = std::fs::remove_file(p);
20201        }
20202    }
20203
20204    /// A block released from a dataset's raw data is recorded by the manager
20205    /// `H5MF_ALLOC_TO_FS_AGGR_TYPE` maps `H5FD_MEM_DRAW` to, and nothing else
20206    /// is: the dichotomy the sec2 driver installs is what decides, and the two
20207    /// managers it collapses to are the file-space info message's slots 0 and
20208    /// 2.
20209    #[test]
20210    fn a_released_raw_block_lands_in_the_raw_data_manager() {
20211        let path = temp_path("fsm_dichotomy");
20212        {
20213            let w = Hdf5Writer::create_with_options(
20214                &path,
20215                FileCreateOptions {
20216                    file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
20217                    ..Default::default()
20218                },
20219            )
20220            .unwrap();
20221            let i = w
20222                .create_dataset("bulk", DatatypeMessage::i32_type(), &[256])
20223                .unwrap();
20224            w.write_dataset_raw(i, &vec![0u8; 1024]).unwrap();
20225            w.create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20226                .unwrap();
20227            w.close().unwrap();
20228        }
20229        let (raw_addr, raw_len) = {
20230            let w = read_only_append(&path);
20231            let i = w.dataset_index("bulk").unwrap();
20232            let ds = w.ds(i);
20233            let m = ds.lock();
20234            (m.data_addr, m.data_size)
20235        };
20236        assert!(raw_len >= 1024, "the raw block is {raw_len} bytes");
20237        {
20238            let w = Hdf5Writer::open_append(&path).unwrap();
20239            w.delete_dataset("bulk").unwrap();
20240            w.close().unwrap();
20241        }
20242
20243        let mut w = read_only_append(&path);
20244        let info = w
20245            .free_space
20246            .as_deref()
20247            .expect("the file persists managers")
20248            .info
20249            .clone();
20250        assert_ne!(info.fs_addr[0], UNDEF_ADDR, "no metadata manager");
20251        assert_ne!(info.fs_addr[2], UNDEF_ADDR, "no raw-data manager");
20252        for (slot, &addr) in info.fs_addr.iter().enumerate() {
20253            if slot != 0 && slot != 2 {
20254                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
20255            }
20256        }
20257
20258        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20259        let inside = |b: &FreeBlock| b.addr >= raw_addr && b.addr + b.len <= raw_addr + raw_len;
20260        let raw: Vec<&FreeBlock> = found
20261            .sections
20262            .iter()
20263            .filter(|b| b.manager == FreeSpaceManager::RawData)
20264            .collect();
20265        assert!(
20266            !raw.is_empty(),
20267            "the deleted dataset's bytes were not recorded"
20268        );
20269        assert!(
20270            raw.iter().all(|b| inside(b)),
20271            "a raw-data section is outside the deleted dataset's block: {raw:?}"
20272        );
20273        assert!(
20274            found
20275                .sections
20276                .iter()
20277                .filter(|b| b.manager == FreeSpaceManager::Metadata)
20278                .all(|b| !inside(b)),
20279            "raw-data bytes were recorded by the metadata manager"
20280        );
20281        drop(w);
20282        let _ = std::fs::remove_file(&path);
20283    }
20284
20285    /// A reopened paged file's managers are this writer's to rewrite, and the
20286    /// three the sec2 driver can reach are the only ones it names.
20287    ///
20288    /// `H5MF__alloc_to_fs_type` (H5MF.c:265) sends a request of at least one
20289    /// page to `H5F_MEM_PAGE_GENERIC` unless the driver declares
20290    /// `H5FD_FEAT_PAGED_AGGR`, which only the multi and split drivers do, so a
20291    /// sec2 file has the dichotomy's two small managers and that one large
20292    /// one: message slots 0, 2 and 6.
20293    #[test]
20294    fn a_paged_file_names_only_the_managers_sec2_can_reach() {
20295        let path = fixture_copy("fsm_persist_page.h5", "fsm_write_paged");
20296        assert!(
20297            read_only_append(&path).free_space.is_some(),
20298            "the paged fixture's managers were not read"
20299        );
20300        append_one(&path, "added", false);
20301
20302        let mut w = read_only_append(&path);
20303        let info = w
20304            .free_space
20305            .as_deref()
20306            .expect("the file persists managers")
20307            .info
20308            .clone();
20309        assert_eq!(info.strategy, FileSpaceStrategy::Page);
20310        for (slot, &addr) in info.fs_addr.iter().enumerate() {
20311            if !matches!(slot, 0 | 2 | 6) {
20312                assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
20313            }
20314        }
20315        assert!(
20316            info.fs_addr.iter().any(|&a| a != UNDEF_ADDR),
20317            "the rewritten file records nothing free"
20318        );
20319        crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20320        drop(w);
20321        let _ = std::fs::remove_file(&path);
20322    }
20323
20324    /// Every section a paged file records sits inside one page, and the pages
20325    /// its small managers use are pages of their own kind — the invariant
20326    /// `H5MF__alloc_pagefs` maintains by giving each small request a whole
20327    /// page of its class and recording the rest of it in that class's manager.
20328    #[test]
20329    fn a_paged_files_small_sections_stay_inside_one_page_of_one_kind() {
20330        let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_pages");
20331        append_one(&path, "added", false);
20332
20333        let mut w = read_only_append(&path);
20334        let info = w
20335            .free_space
20336            .as_deref()
20337            .expect("the file persists managers")
20338            .info
20339            .clone();
20340        let page = info.page_size;
20341        let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
20342        let mut kind_of_page: std::collections::HashMap<u64, FreeSpaceManager> =
20343            std::collections::HashMap::new();
20344        for section in &found.sections {
20345            if section.manager == FreeSpaceManager::Large {
20346                continue;
20347            }
20348            assert_eq!(
20349                section.addr / page,
20350                (section.addr + section.len - 1) / page,
20351                "the section at {:#x} crosses a page boundary",
20352                section.addr
20353            );
20354            let owner = kind_of_page
20355                .entry(section.addr / page)
20356                .or_insert(section.manager);
20357            assert_eq!(
20358                *owner,
20359                section.manager,
20360                "page {} holds sections of two kinds",
20361                section.addr / page
20362            );
20363        }
20364        drop(w);
20365        let _ = std::fs::remove_file(&path);
20366    }
20367}