rust_hdf5/io/writer.rs
1//! HDF5 file writer.
2//!
3//! Produces a valid HDF5 file with superblock v3, a root group object header,
4//! and datasets with contiguous or chunked storage. The output is readable by `h5dump`.
5
6use std::collections::{HashMap, HashSet};
7use std::path::{Path, PathBuf};
8
9use crate::dataset::DatasetAccess;
10use crate::format::btree_v1::{BTreeV1Config, ChunkBTreeV1Node, ChunkBTreeV1Tree, ChunkKey};
11use crate::format::chunk_index::btree_v2::Bt2ChunkIndex;
12use crate::format::chunk_index::extensible_array::{
13 compute_chunk_size_len, compute_ndblk_addrs, compute_nsblk_addrs, EaDblkPath, EaGeometry,
14 EaLoc, ExtensibleArrayDataBlock, ExtensibleArrayHeader, ExtensibleArrayIndexBlock,
15 ExtensibleArraySuperBlock, FilteredChunkEntry, FilteredDataBlock, FilteredIndexBlock,
16 EA_CLS_CHUNK, EA_CLS_FILT_CHUNK,
17};
18use crate::format::chunk_index::fixed_array::{
19 decode_filtered_page, decode_unfiltered_page, encode_filtered_page, encode_unfiltered_page,
20 FixedArrayDataBlock, FixedArrayFilteredChunkElement, FixedArrayHeader, FixedArrayPagedPrefix,
21 FA_CLIENT_FILT_CHUNK,
22};
23use crate::format::creation_order::CreationOrder;
24use crate::format::dense_attr::build_dense_attributes;
25use crate::format::dense_link::build_dense_links;
26use crate::format::free_space::{
27 self, FreeSection, FreeSpaceClass, FreeSpaceHeader, FreeSpaceManager,
28};
29use crate::format::local_heap::{
30 local_heap_header_size, LocalHeapHeader, LocalHeapImage, LOCAL_HEAP_FREE_NULL,
31};
32use crate::format::messages::attr_info::{next_creation_index, AttributeInfoMessage};
33use crate::format::messages::attribute::{
34 AttributeEntry, AttributeMessage, ATTR_FLAG_SPACE_SHARED, ATTR_FLAG_TYPE_SHARED,
35};
36use crate::format::messages::data_layout::{
37 DataLayoutMessage, EarrayParams, FixedArrayParams, LAYOUT_VERSION_DEFAULT,
38};
39use crate::format::messages::dataspace::{DataspaceClass, DataspaceMessage};
40use crate::format::messages::datatype::{ByteOrder, DatatypeMessage, ReferenceKind};
41use crate::format::messages::external_file_list::{ExternalFileListMessage, UNLIMITED};
42use crate::format::messages::fill_value::{
43 FillValueMessage, FILL_TIME_ALLOC, FILL_TIME_IFSET, FILL_TIME_NEVER,
44};
45use crate::format::messages::filter::{self, FilterPipeline};
46use crate::format::messages::group_info::GroupInfoMessage;
47use crate::format::messages::link::{CharacterSet, LinkMessage, LinkTarget};
48use crate::format::messages::link_info::LinkInfoMessage;
49use crate::format::messages::mod_time::ModificationTime;
50use crate::format::messages::superblock_ext::{
51 FileSpaceInfoMessage, FileSpaceStrategy, SharedMessageTableMessage,
52 DEFAULT_FILE_SPACE_PAGE_SIZE, FS_ADDR_COUNT_V1, PAGE_SIZE_MAX, PAGE_SIZE_MIN,
53};
54use crate::format::messages::virtual_mapping::{
55 parse_source_name, VirtualMapping, VirtualMappingList,
56};
57use crate::format::messages::*;
58use crate::format::object_header::{ObjectHeader, ObjectTimes, MAX_MESSAGE_SIZE};
59use crate::format::reference::{
60 encode_reference_element, encode_revised_blob, ReferenceElementImage, ReferenceTarget,
61 REVISED_BLOB_TOKEN_OFFSET,
62};
63use crate::format::selection::Selection;
64use crate::format::sohm::{
65 type_flag, SharedMessagePointer, MAX_SOHM_INDEXES, SOHM_HEAP_ID_LEN, SOHM_POINTER_HEAP_ID_AT,
66};
67use crate::format::sohm_write::{
68 build_shared_messages, NestedShare, SharedMessage, SohmIndexContent, SohmIndexSpec,
69};
70use crate::format::superblock::*;
71use crate::format::{FormatContext, LibverBound, ObjectFormat, UNDEF_ADDR};
72
73use crate::format::selection::check_hyperslab;
74use crate::io::allocator::{FileAllocator, FreeBlock};
75use crate::io::file_handle::FileHandle;
76use crate::io::hyperslab::{for_each_contiguous_run, for_each_dual_run};
77use crate::io::symbol_table_io::{free_stab, write_stab, Stab, StabExtents, StabLink, StabTarget};
78use crate::io::{FileMeta, IoResult};
79
80/// On-disk size in bytes of a fixed-array data block, for the layout (paged or
81/// flat) implied by `hdr`.
82///
83/// Mirrors `H5FA_DBLOCK_SIZE` (`H5FApkg.h`):
84/// - non-paged: `prefix + nelmts * raw_elmt_size + checksum`
85/// - paged: `prefix + page_init_bitmap + nelmts * raw_elmt_size
86/// + npages * checksum`, where the prefix checksum covers the bitmap.
87///
88/// `raw_elmt_size` is `sizeof_addr` for an unfiltered array, and
89/// `sizeof_addr + chunk_size_len + 4` (the filtered element: address +
90/// compressed size + filter mask) for a filtered array. libhdf5 carries this
91/// value as `hdr->cparam.raw_elmt_size`, i.e. exactly `hdr.element_size`.
92fn fixed_array_dblk_disk_size(ctx: &FormatContext, hdr: &FixedArrayHeader) -> u64 {
93 let elem_size = hdr.element_size as u64;
94 let sa = ctx.sizeof_addr as u64;
95 let nelmts = hdr.num_elmts;
96 // Common metadata prefix: signature(4) + version(1) + client_id(1) + header_addr(sa).
97 let meta_prefix = 4 + 1 + 1 + sa;
98 if hdr.is_paged() {
99 let npages = hdr.npages();
100 let bitmap_size = npages.div_ceil(8);
101 // prefix (incl. its own 4-byte checksum) + elements + per-page checksums.
102 (meta_prefix + bitmap_size + 4) + nelmts * elem_size + npages * 4
103 } else {
104 // prefix + elements + single 4-byte checksum.
105 meta_prefix + nelmts * elem_size + 4
106 }
107}
108
109/// A walk of a v2 B-tree: the file and node geometry the descent reads
110/// through, and the two collections it fills — every node's raw record
111/// bytes and every node block's address, the latter because `open_append`
112/// needs it so the reconstructed [`Bt2DatasetInfo::node_addrs`] pool owns
113/// the on-disk nodes (the next flush re-serializes the tree over them, and
114/// a delete frees them).
115///
116/// `record_size`, `node_size` and `geo` are constant for the whole walk, so
117/// [`descend`](Self::descend) takes only what changes per level: the node's
118/// address, its depth, and how many records it holds.
119struct Bt2Walk<'a> {
120 handle: &'a FileHandle,
121 ctx: &'a FormatContext,
122 record_size: u16,
123 node_size: u32,
124 geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
125 records: Vec<u8>,
126 node_addrs: Vec<u64>,
127}
128
129impl<'a> Bt2Walk<'a> {
130 fn new(
131 handle: &'a FileHandle,
132 ctx: &'a FormatContext,
133 record_size: u16,
134 node_size: u32,
135 geo: &'a crate::format::chunk_index::btree_v2::Bt2Geometry,
136 ) -> Self {
137 Self {
138 handle,
139 ctx,
140 record_size,
141 node_size,
142 geo,
143 records: Vec::new(),
144 node_addrs: Vec::new(),
145 }
146 }
147
148 /// Walk the subtree rooted at `addr`, at depth `depth` with `nrec`
149 /// records, collecting every node's raw record bytes and every node
150 /// block's address.
151 fn descend(&mut self, addr: u64, depth: u16, nrec: u16) -> IoResult<()> {
152 use crate::format::chunk_index::btree_v2::{Bt2InternalNode, Bt2LeafNode};
153
154 self.node_addrs.push(addr);
155 let buf = self.handle.read_at_most(addr, self.node_size as usize)?;
156 if depth == 0 {
157 let leaf = Bt2LeafNode::decode(&buf, nrec, self.record_size)?;
158 self.records.extend_from_slice(&leaf.record_data);
159 } else {
160 let node = Bt2InternalNode::decode(
161 &buf,
162 self.ctx,
163 depth,
164 nrec,
165 self.record_size,
166 self.geo.max_nrec_size,
167 self.geo.child_total_size(depth),
168 )?;
169 // In-order: an internal node's records separate its children, so each
170 // one belongs between the subtrees on either side of it.
171 let children: Vec<(u64, u16)> = node
172 .child_addrs
173 .iter()
174 .zip(node.child_nrecords.iter())
175 .map(|(&a, &n)| (a, n))
176 .collect();
177 let rec = self.record_size as usize;
178 for (i, (child_addr, child_nrec)) in children.into_iter().enumerate() {
179 self.descend(child_addr, depth - 1, child_nrec)?;
180 if let Some(record) = node.record_data.get(i * rec..(i + 1) * rec) {
181 self.records.extend_from_slice(record);
182 }
183 }
184 }
185 Ok(())
186 }
187}
188
189/// A walk of a version-1 raw-data-chunk B-tree: the file and geometry the
190/// descent reads through, and the two collections it fills.
191///
192/// The v1 counterpart of [`Bt2Walk`], and for the same reason: the
193/// records are what [`BtreeV1DatasetInfo::build_tree`] bulk-loads on the next
194/// flush, and the addresses are the block pool that flush re-serializes over,
195/// so a reopened tree owns the nodes it found instead of leaking them and
196/// allocating a second set beside them.
197///
198/// One value rather than nine parameters threaded through the recursion: only
199/// `addr` and `depth` change between one level and the next, so they are what
200/// [`descend`](Self::descend) takes and everything else lives here.
201struct BtreeV1Walk<'a> {
202 handle: &'a FileHandle,
203 ctx: &'a FormatContext,
204 config: &'a BTreeV1Config,
205 /// The chunk edge lengths, *without* the trailing element-size dimension,
206 /// so `chunk_dims.len()` is the rank the node keys are decoded at.
207 chunk_dims: &'a [u64],
208 file_size: u64,
209 records: Vec<BtreeV1ChunkRecord>,
210 node_addrs: Vec<u64>,
211}
212
213impl<'a> BtreeV1Walk<'a> {
214 fn new(
215 handle: &'a FileHandle,
216 ctx: &'a FormatContext,
217 config: &'a BTreeV1Config,
218 chunk_dims: &'a [u64],
219 file_size: u64,
220 ) -> Self {
221 Self {
222 handle,
223 ctx,
224 config,
225 chunk_dims,
226 file_size,
227 records: Vec::new(),
228 node_addrs: Vec::new(),
229 }
230 }
231
232 /// Walk the subtree rooted at `addr`, collecting every leaf entry as a
233 /// [`BtreeV1ChunkRecord`] and every node block's address.
234 ///
235 /// Records come out in key order because a v1 B-tree's leaves are in key
236 /// order and this descends left to right, which is what
237 /// [`BtreeV1DatasetInfo::position`]'s binary search needs. The keys store
238 /// element offsets (`scaled * chunk_dim`, `H5D__btree_encode_key`), so the
239 /// grid position this records is the quotient.
240 fn descend(&mut self, addr: u64, depth: u32) -> IoResult<()> {
241 // The same bound the reader's walk uses: a node's level is one byte, so
242 // no honest tree is deeper than that, and a cyclic index stops here.
243 if depth > 256 {
244 return Err(crate::io::IoError::InvalidState(
245 "chunk B-tree v1 exceeds maximum depth".into(),
246 ));
247 }
248 if addr == UNDEF_ADDR || addr >= self.file_size {
249 return Ok(());
250 }
251 let rank = self.chunk_dims.len();
252 let sa = self.ctx.sizeof_addr as usize;
253 let node_size = self.config.chunk_btree_node_size(sa, rank);
254 let buf = self.handle.read_at_most(addr, node_size)?;
255 let node = ChunkBTreeV1Node::decode(&buf, sa, rank, self.config.chunk_max_entries())?;
256 self.node_addrs.push(addr);
257
258 if node.level == 0 {
259 for (i, &child_addr) in node.children.iter().enumerate() {
260 let key = &node.keys[i];
261 let scaled: Vec<u64> = key.offsets[..rank]
262 .iter()
263 .zip(self.chunk_dims)
264 .map(|(&offset, &dim)| offset.checked_div(dim).unwrap_or(0))
265 .collect();
266 self.records.push(BtreeV1ChunkRecord {
267 scaled,
268 address: child_addr,
269 nbytes: key.chunk_size,
270 filter_mask: key.filter_mask,
271 });
272 }
273 } else {
274 for &child_addr in &node.children {
275 self.descend(child_addr, depth + 1)?;
276 }
277 }
278 Ok(())
279 }
280}
281
282/// Encode a fixed-array data block for the layout implied by `hdr`, using the
283/// chunk addresses held in `dblk.elements` (unfiltered) or the filtered chunk
284/// entries in `dblk.filtered_elements` (filtered, `client_id == 1`).
285///
286/// For the paged layout (`hdr.is_paged()`), emits the `FADB` prefix with a
287/// page-init bitmap followed by `npages` checksummed element pages. A page is
288/// marked initialized iff at least one of its chunk addresses is defined,
289/// mirroring libhdf5's lazy `H5FA__dblk_page_create`. Uninitialized pages are
290/// still written (all `UNDEF_ADDR`, valid checksum) so the file contains no
291/// uninitialized bytes; the reader skips them via the bitmap.
292fn encode_fixed_array_dblk(
293 ctx: &FormatContext,
294 hdr: &FixedArrayHeader,
295 dblk: &FixedArrayDataBlock,
296) -> Vec<u8> {
297 let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
298 let sa = ctx.sizeof_addr as usize;
299 // chunk_size_len for filtered entries = element_size - sizeof_addr - 4.
300 // libhdf5 carries element_size = sizeof_addr + chunk_size_len + 4.
301 let chunk_size_len = (hdr.element_size as usize).saturating_sub(sa + 4);
302
303 if !hdr.is_paged() {
304 return if is_filtered {
305 dblk.encode_filtered(ctx, chunk_size_len)
306 } else {
307 dblk.encode_unfiltered(ctx)
308 };
309 }
310
311 let npages = hdr.npages() as usize;
312 let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
313
314 // Build the page-init bitmap (MSB-first): a page is initialized iff any of
315 // its elements points at a defined address.
316 let mut bitmap = vec![0u8; npages.div_ceil(8)];
317 let nelmts = if is_filtered {
318 dblk.filtered_elements.len()
319 } else {
320 dblk.elements.len()
321 };
322 for p in 0..npages {
323 let start = p * dblk_page_nelmts;
324 let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
325 let initialized = if is_filtered {
326 dblk.filtered_elements[start..end]
327 .iter()
328 .any(|e| e.address != UNDEF_ADDR)
329 } else {
330 dblk.elements[start..end].iter().any(|&a| a != UNDEF_ADDR)
331 };
332 if initialized {
333 bitmap[p / 8] |= 0x80u8 >> (p % 8);
334 }
335 }
336
337 let prefix = FixedArrayPagedPrefix {
338 client_id: hdr.client_id,
339 header_addr: dblk.header_addr,
340 page_init_bitmap: bitmap,
341 prefix_size: 4 + 1 + 1 + sa + npages.div_ceil(8) + 4,
342 };
343
344 let mut buf = prefix.encode(ctx);
345 debug_assert_eq!(buf.len(), prefix.prefix_size);
346
347 // Append each page: all pages use the full `dblk_page_nelmts` stride;
348 // only the last page holds fewer elements (libhdf5 H5FA.c).
349 for p in 0..npages {
350 let start = p * dblk_page_nelmts;
351 let end = ((p + 1) * dblk_page_nelmts).min(nelmts);
352 if is_filtered {
353 buf.extend_from_slice(&encode_filtered_page(
354 &dblk.filtered_elements[start..end],
355 ctx,
356 chunk_size_len,
357 ));
358 } else {
359 buf.extend_from_slice(&encode_unfiltered_page(&dblk.elements[start..end], ctx));
360 }
361 }
362 buf
363}
364
365/// Decode a fixed-array data block for the layout implied by `hdr` — the
366/// inverse of [`encode_fixed_array_dblk`], and the single decode dispatch
367/// over non-paged/paged × unfiltered/filtered.
368///
369/// For the paged layout, pages whose bitmap bit is clear are skipped, not
370/// decoded: libhdf5 never writes an uninitialized page, so its bytes are
371/// arbitrary and carry no valid checksum. Their elements stay at the
372/// undefined-address defaults, which is exactly what the bitmap means.
373fn decode_fixed_array_dblk(
374 ctx: &FormatContext,
375 hdr: &FixedArrayHeader,
376 buf: &[u8],
377 chunk_size_len: usize,
378) -> crate::format::FormatResult<FixedArrayDataBlock> {
379 let is_filtered = hdr.client_id == FA_CLIENT_FILT_CHUNK;
380 let num_elmts = hdr.num_elmts as usize;
381
382 if !hdr.is_paged() {
383 return if is_filtered {
384 FixedArrayDataBlock::decode_filtered(buf, ctx, num_elmts, chunk_size_len)
385 } else {
386 FixedArrayDataBlock::decode_unfiltered(buf, ctx, num_elmts)
387 };
388 }
389
390 let npages = hdr.npages() as usize;
391 let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
392 let prefix = FixedArrayPagedPrefix::decode(buf, ctx, npages as u64)?;
393
394 let mut dblk = if is_filtered {
395 FixedArrayDataBlock::new_filtered(prefix.header_addr, num_elmts)
396 } else {
397 FixedArrayDataBlock::new_unfiltered(prefix.header_addr, num_elmts)
398 };
399 dblk.client_id = hdr.client_id;
400
401 // Pages follow the prefix back to back; every page spans the full
402 // `dblk_page_nelmts` stride except the last, which holds the remainder.
403 let mut pos = prefix.prefix_size;
404 for p in 0..npages {
405 let start = p * dblk_page_nelmts;
406 let end = ((p + 1) * dblk_page_nelmts).min(num_elmts);
407 let nelmts = end - start;
408 if prefix.page_initialized(p) {
409 let page_buf = buf.get(pos..).unwrap_or(&[]);
410 if is_filtered {
411 let elems = decode_filtered_page(page_buf, ctx, nelmts, chunk_size_len)?;
412 dblk.filtered_elements[start..end].clone_from_slice(&elems);
413 } else {
414 let addrs = decode_unfiltered_page(page_buf, ctx, nelmts)?;
415 dblk.elements[start..end].copy_from_slice(&addrs);
416 }
417 }
418 pos += nelmts * hdr.element_size as usize + 4;
419 }
420 Ok(dblk)
421}
422
423/// Interior-mutability cell for per-dataset write state, selected by feature.
424///
425/// This is the §5-B "cfg-selected interior types" from
426/// `docs/threadsafe-fine-grained-locking.md`: the single-threaded build uses a
427/// `RefCell` (zero overhead, no atomics), while the `threadsafe` build uses a
428/// `Mutex` so two threads can write *different* datasets concurrently while the
429/// same dataset's writes serialize. Call sites are identical across both via
430/// [`Slot::lock`].
431#[cfg(not(feature = "threadsafe"))]
432pub(crate) struct Slot<T>(std::cell::RefCell<T>);
433
434#[cfg(not(feature = "threadsafe"))]
435impl<T> Slot<T> {
436 pub(crate) fn new(value: T) -> Self {
437 Slot(std::cell::RefCell::new(value))
438 }
439 /// Borrow the contents mutably (an uncontended `RefCell` borrow).
440 pub(crate) fn lock(&self) -> std::cell::RefMut<'_, T> {
441 self.0.borrow_mut()
442 }
443}
444
445#[cfg(feature = "threadsafe")]
446pub(crate) struct Slot<T>(std::sync::Mutex<T>);
447
448#[cfg(feature = "threadsafe")]
449impl<T> Slot<T> {
450 pub(crate) fn new(value: T) -> Self {
451 Slot(std::sync::Mutex::new(value))
452 }
453 /// Lock the contents. Different datasets hold different slots, so this
454 /// only contends when two threads write the *same* dataset.
455 pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, T> {
456 self.0.lock().unwrap()
457 }
458}
459
460/// Proof that the create gate (`create_lock`) is held and the new dataset's
461/// name passed the uniqueness check. Only [`Hdf5Writer::begin_create`]
462/// constructs one and [`Hdf5Writer::push_dataset`] demands one, so a creator
463/// cannot reach the dataset registry while skipping either step. Carries
464/// the canonical (link-resolved) name the creator must store, so the
465/// registry only ever holds tree paths.
466pub(crate) struct CreateGuard<'a> {
467 #[cfg(not(feature = "threadsafe"))]
468 _gate: std::cell::RefMut<'a, ()>,
469 #[cfg(feature = "threadsafe")]
470 _gate: std::sync::MutexGuard<'a, ()>,
471 /// The dataset name with every group hard link in it resolved.
472 pub(crate) name: String,
473 /// The group that will hold the new dataset's link, resolved from the
474 /// path components of `name`; `None` is the root group. Carried here so
475 /// [`Hdf5Writer::push_dataset`] registers the child itself and no creator
476 /// can leave a dataset whose name says one thing and whose parent group
477 /// says another.
478 pub(crate) parent: Option<usize>,
479}
480
481/// Reference-counted shared pointer, feature-selected. The single-thread
482/// build uses `Rc` (no atomics); the `threadsafe` build uses `Arc` so a
483/// dataset/group slot can be cloned out of the registry and locked on its
484/// own — letting writes to *different* datasets proceed concurrently without
485/// holding the registry lock. See `docs/threadsafe-fine-grained-locking.md`
486/// (Stage 3).
487#[cfg(not(feature = "threadsafe"))]
488pub(crate) type Shared<T> = std::rc::Rc<T>;
489#[cfg(feature = "threadsafe")]
490pub(crate) type Shared<T> = std::sync::Arc<T>;
491
492/// One dataset's cell in the registry: its metadata slot plus the operation
493/// lock that serializes whole logical operations on it. Both live in one
494/// allocation so they cannot fall out of step — every dataset has its op
495/// lock by construction.
496pub(crate) struct DatasetCell {
497 /// Serializes one *whole* logical operation on this dataset.
498 ///
499 /// The metadata slot below serializes each individual acquisition, but a
500 /// multi-acquisition operation — take the append buffer → write chunks →
501 /// re-buffer the tail → extend, or flush-then-overwrite in a slice write
502 /// — would interleave with a concurrent same-dataset operation *between*
503 /// its acquisitions under `threadsafe`. Public write entries take this
504 /// lock and delegate to `_inner` variants; `_inner` variants and the
505 /// `pub(crate)` write helpers require the caller to hold it (or to hold
506 /// the writer exclusively via `&mut`, as close and the SWMR wrapper do).
507 ///
508 /// Not reentrant: the single-thread build's `RefCell` panics instantly
509 /// on a nested acquisition, so a missed entry/inner split fails loudly
510 /// in every test run rather than deadlocking only under `threadsafe`.
511 ///
512 /// Lock order: `create_lock → op → registry spine → metadata slot`. An
513 /// op lock is never held across another dataset's op lock, and no
514 /// op-lock holder takes `create_lock`, so the order is acyclic.
515 pub(crate) op: Slot<()>,
516 info: Slot<DatasetInfo>,
517}
518
519impl DatasetCell {
520 pub(crate) fn new(info: DatasetInfo) -> Self {
521 DatasetCell {
522 op: Slot::new(()),
523 info: Slot::new(info),
524 }
525 }
526
527 /// Borrow the metadata slot (a single acquisition; see [`Self::op`] for
528 /// whole-operation serialization).
529 #[cfg(not(feature = "threadsafe"))]
530 pub(crate) fn lock(&self) -> std::cell::RefMut<'_, DatasetInfo> {
531 self.info.lock()
532 }
533
534 /// Lock the metadata slot (a single acquisition; see [`Self::op`] for
535 /// whole-operation serialization).
536 #[cfg(feature = "threadsafe")]
537 pub(crate) fn lock(&self) -> std::sync::MutexGuard<'_, DatasetInfo> {
538 self.info.lock()
539 }
540}
541
542/// A single dataset's [`DatasetCell`], reference-counted so a writer can
543/// clone it out of the registry (releasing the registry lock) and then lock
544/// just this one dataset. Two threads writing different datasets take
545/// different `DatasetRef` locks and never contend; the same dataset's writes
546/// serialize, which is required because one chunk index is not concurrently
547/// mutable.
548pub(crate) type DatasetRef = Shared<DatasetCell>;
549
550/// A single group's metadata behind its own [`Slot`], reference-counted like
551/// [`DatasetRef`].
552pub(crate) type GroupRef = Shared<Slot<GroupInfo>>;
553
554/// Appended frames held back until they complete a chunk.
555///
556/// The buffer is the sole authority for rows `base .. base + frames`: the
557/// file's chunks do not hold them yet, and any operation that writes those
558/// rows must go through [`Hdf5Writer::flush_append_buffer`] first. `base` is
559/// recorded when the frames are buffered — never derived from the current
560/// extent, which an `extend_dataset` can move independently.
561pub struct AppendBuffer {
562 /// Absolute row of the first buffered frame.
563 pub base: u64,
564 /// Number of buffered frames.
565 pub frames: u64,
566 /// The frames' bytes, `frames` whole rows, row-major.
567 pub bytes: Vec<u8>,
568}
569
570/// One file a dataset's raw data lives in, as the writer holds it: the name
571/// the I/O path opens, together with the local-heap offset the External File
572/// List message stores that name as.
573///
574/// The two halves are one entry rather than two parallel lists because they
575/// describe one slot — the message encodes `name_offset`, and every read or
576/// write of the slot's bytes opens `name`; splitting them is what lets a
577/// rewrite pair a name with another slot's offset.
578#[derive(Debug, Clone, PartialEq, Eq)]
579pub struct ExternalFile {
580 /// The file name exactly as the heap stores it. Resolved against
581 /// `HDF5_EXTFILE_PREFIX` at I/O time, never here — the same rule the read
582 /// side follows.
583 pub name: String,
584 /// Where `name` sits in the local heap at [`ExternalStorage::heap_addr`].
585 pub name_offset: u64,
586 /// Byte offset within `name` where this slot's region begins.
587 pub offset: u64,
588 /// Bytes of the dataset's raw data this slot holds.
589 pub size: u64,
590}
591
592/// A dataset whose contiguous raw data lives outside this file — the External
593/// File List message (`H5O_EFL_ID`) and the local heap its names are in.
594///
595/// The data layout message of such a dataset still says `Contiguous`, with
596/// its address left undefined: it is this message's presence that makes
597/// libhdf5 route the dataset's I/O through `H5D_LOPS_EFL` (H5Dlayout.c).
598#[derive(Debug, Clone)]
599pub struct ExternalStorage {
600 /// Address of the local heap header holding every slot's name.
601 pub heap_addr: u64,
602 /// The files, in the order their regions concatenate into the dataset's
603 /// logical byte range.
604 pub files: Vec<ExternalFile>,
605 /// The prefix every one of those names is joined against, and the open
606 /// that settled it. Lives here rather than on [`DatasetInfo`] so a
607 /// dataset with no external storage cannot carry a prefix and a dataset
608 /// with external storage cannot lack one.
609 prefix: EfilePrefix,
610}
611
612/// The expanded external file prefix in force for one dataset, and the open
613/// that decided it — libhdf5's `dset->shared->extfile_prefix`.
614///
615/// `H5D__build_file_prefix` runs it once per open of the shared info, from
616/// the dapl of `H5D__create` (H5Dint.c:1318) or of the `H5D__open` that
617/// found no shared info yet (:1537), and both `H5D__efl_read` and
618/// `H5D__efl_write` then join against that one answer (H5Defl.c:315-317,
619/// :429-431). Measured under libhdf5 1.14.6 and 2.0.0: `H5Dcreate2` with a
620/// dapl naming a directory creates the raw data file there at `H5Dwrite`,
621/// and `HDF5_EXTFILE_PREFIX` shadows that property on the write path exactly
622/// as it does on the read path.
623#[derive(Debug, Clone, Default)]
624struct EfilePrefix {
625 /// The expansion itself; `None` is "no prefix", which leaves a stored
626 /// name to resolve against the process's current directory.
627 expanded: Option<PathBuf>,
628 /// The open that decided [`expanded`](Self::expanded). An expired handle
629 /// means no open is holding the answer any more, so the next one settles
630 /// it afresh — which is the state a dataset this session reopened starts
631 /// in, `H5Fopen` opening no dataset of its own.
632 open: std::sync::Weak<()>,
633}
634
635impl ExternalStorage {
636 /// The message this storage encodes to (`H5O_efl_t`).
637 fn message(&self) -> ExternalFileListMessage {
638 ExternalFileListMessage {
639 heap_addr: self.heap_addr,
640 slots: self
641 .files
642 .iter()
643 .map(
644 |f| crate::format::messages::external_file_list::ExternalFileSlot {
645 name_offset: f.name_offset,
646 offset: f.offset,
647 size: f.size,
648 },
649 )
650 .collect(),
651 }
652 }
653
654 /// Bytes the slots reserve in total (`H5O_efl_total_size`), saturating
655 /// rather than wrapping so an overflowing list reads as "as large as it
656 /// gets" and passes any size check instead of failing one.
657 fn total_size(&self) -> u64 {
658 self.files
659 .iter()
660 .fold(0u64, |acc, f| acc.saturating_add(f.size))
661 }
662}
663
664/// A dataset whose elements are read out of other datasets — the virtual
665/// layout message (`H5D_VIRTUAL`) and the mapping list it points at.
666///
667/// The mappings live in one global heap object rather than in the header
668/// (`H5D__virtual_store_layout`), so the layout message carries only its
669/// address and index; the list itself is kept here so a rewrite of the header
670/// can re-emit the message pointing at the same object.
671#[derive(Debug, Clone, PartialEq, Eq)]
672pub struct VirtualStorage {
673 /// Address of the global heap collection holding the mapping list.
674 pub heap_addr: u64,
675 /// Index of the mapping-list object within that collection.
676 pub heap_index: u32,
677 /// The mappings themselves, in the order they were declared — which is
678 /// the order libhdf5 resolves overlapping ones in.
679 pub mappings: Vec<VirtualMapping>,
680}
681
682/// Where a contiguous dataset's raw bytes live, read off its registry entry
683/// so the write itself can run with the slot unlocked.
684///
685/// The one place the local-versus-external-versus-nowhere choice is made; see
686/// [`DatasetInfo::contiguous_target`].
687enum ContiguousTarget {
688 /// A block in this file, starting at this address.
689 Local(u64),
690 /// The files an External File List names, in dataset order, and the
691 /// prefix in force for the open doing the writing — carried together
692 /// because a slot name means nothing without it.
693 External {
694 files: Vec<ExternalFile>,
695 prefix: Option<PathBuf>,
696 },
697 /// Nowhere: the dataset is virtual, and every element of it is stored in
698 /// whichever source dataset its mappings send that element to.
699 Virtual,
700}
701
702/// What a writer-mode `H5Dataset` handle is built from — the shape and
703/// element width it answers questions with, the chunk index it writes
704/// through, and the open it holds.
705pub(crate) struct DatasetHandleParts {
706 pub(crate) shape: Vec<usize>,
707 pub(crate) element_size: usize,
708 /// `None` for storage that is not chunked.
709 pub(crate) chunk_index: Option<ChunkIndexKind>,
710 /// Keeps this open alive; see [`Hdf5Writer::bind_efile_prefix`].
711 pub(crate) open: Option<crate::io::reader::DatasetOpenToken>,
712}
713
714impl ContiguousTarget {
715 /// Whether this target is storage bytes can be written into at all —
716 /// false only for [`ContiguousTarget::Virtual`], which names sources
717 /// rather than storage.
718 fn is_storage(&self) -> bool {
719 !matches!(self, Self::Virtual)
720 }
721}
722
723/// The one refusal of a write into a virtual dataset, so the two paths that
724/// can reach one — [`Hdf5Writer::write_contiguous_bytes`] and the pre-insert
725/// gate of [`Hdf5Writer::write_vlen_strings_slice`] — say the same thing.
726///
727/// libhdf5 does take this write, pushing each element through the mapping
728/// that covers it into the source dataset holding it (`H5D__virtual_write`);
729/// this writer never opens a source file, so it refuses rather than dropping
730/// the bytes somewhere they cannot be read back from.
731/// The legality checks `H5Pset_virtual` runs over one mapping —
732/// `H5D_virtual_check_mapping_pre` and `H5D_virtual_check_mapping_post`
733/// (H5Dvirtual.c).
734///
735/// The two upstream checks that need the *source dataset's* own extent (the
736/// limited/limited element-count match, and a printf mapping's single-block
737/// match) are not run here for the same reason upstream skips them when the
738/// source space status is `H5O_VIRTUAL_STATUS_INVALID`: a mapping may name a
739/// source that does not exist yet, and nothing here opens one.
740fn check_virtual_mapping(dataset: &str, m: &VirtualMapping) -> IoResult<()> {
741 for (which, sel) in [
742 ("virtual", &m.virtual_selection),
743 ("source", &m.source_selection),
744 ] {
745 if matches!(sel, Selection::Points(_)) {
746 return Err(crate::io::IoError::Unsupported(format!(
747 "virtual dataset '{dataset}' has a point {which} selection, which \
748 H5D_virtual_check_mapping_pre refuses for every virtual dataset mapping \
749 (\"point selections not currently supported with virtual datasets\")"
750 )));
751 }
752 }
753
754 let unlim_virtual = m.virtual_selection.unlim_dim().is_some();
755 let unlim_source = m.source_selection.unlim_dim().is_some();
756
757 // Both sides unbounded: the mapping grows with its source, so the slices
758 // they exchange must be the same shape whatever either extent becomes.
759 if unlim_virtual && unlim_source {
760 if let (Some(v), Some(sr)) = (
761 regular_hyperslab(&m.virtual_selection),
762 regular_hyperslab(&m.source_selection),
763 ) {
764 let (nv, ns) = (v.num_elem_non_unlim(), sr.num_elem_non_unlim());
765 if nv != ns {
766 return Err(crate::io::IoError::InvalidState(format!(
767 "virtual dataset '{dataset}' maps an unlimited source selection onto an \
768 unlimited virtual selection, but a slice of the non-unlimited \
769 dimensions holds {ns:?} source elements and {nv:?} virtual ones"
770 )));
771 }
772 }
773 }
774
775 // `H5D_virtual_check_mapping_post`: an unlimited virtual selection over a
776 // limited source selection is the printf shape, where each block of the
777 // virtual selection is filled by a *different* source dataset named by
778 // substituting that block's index. It needs a `%b` to name them, and a
779 // hyperslab virtual selection to have blocks at all; every other shape
780 // needs the opposite, since a substitution with only one block to fill
781 // has nothing to vary over.
782 let nsubs = parse_source_name(&m.source_file_name)
783 .and_then(|f| Ok(f.nsubs() + parse_source_name(&m.source_dset_name)?.nsubs()))
784 .map_err(|e| {
785 crate::io::IoError::InvalidState(format!(
786 "virtual dataset '{dataset}' source name: {e}"
787 ))
788 })?;
789 if unlim_virtual && !unlim_source {
790 if nsubs == 0 {
791 return Err(crate::io::IoError::InvalidState(format!(
792 "virtual dataset '{dataset}' has an unlimited virtual selection, a limited \
793 source selection, and no printf specifiers in source names"
794 )));
795 }
796 if !matches!(m.virtual_selection, Selection::Hyperslab { .. }) {
797 return Err(crate::io::IoError::InvalidState(format!(
798 "virtual dataset '{dataset}' has a printf mapping whose virtual selection is \
799 not a hyperslab; the substitution runs over the blocks of that hyperslab"
800 )));
801 }
802 } else if nsubs > 0 {
803 return Err(crate::io::IoError::InvalidState(format!(
804 "virtual dataset '{dataset}' has printf specifier(s) in source name(s) without \
805 an unlimited virtual selection and limited source selection"
806 )));
807 }
808 Ok(())
809}
810
811/// The regular (start, stride, count, block) form behind a selection, or
812/// `None` — the only form that can carry `H5S_UNLIMITED`, so every unlimited
813/// check goes through it.
814fn regular_hyperslab(sel: &Selection) -> Option<&crate::format::selection::RegularHyperslab> {
815 match sel {
816 Selection::Hyperslab {
817 form: crate::format::selection::Hyperslab::Regular(r),
818 ..
819 } => Some(r),
820 _ => None,
821 }
822}
823
824fn virtual_write_refused() -> crate::io::IoError {
825 crate::io::IoError::Unsupported(
826 "cannot write into a virtual dataset: its elements live in the source datasets \
827 its mappings name, and this writer does not write through to them — write the \
828 source datasets themselves"
829 .into(),
830 )
831}
832
833/// Metadata for a dataset being written.
834///
835/// The whole struct lives behind a per-dataset [`Slot`] (via [`DatasetRef`]).
836/// The streaming write path locks it only briefly — compression runs *outside*
837/// the lock — so writes to different datasets do not contend, and a structural
838/// op (create/delete) that scans names only momentarily touches a sibling
839/// slot.
840pub struct DatasetInfo {
841 /// Link name within the root group.
842 pub name: String,
843 /// Element datatype.
844 pub datatype: DatatypeMessage,
845 /// The committed datatype this dataset shares, when it was created from
846 /// one. The type itself stays in [`datatype`](Self::datatype) — the
847 /// dataspace, the element width and every payload check need it — and
848 /// this says the header must store a pointer to that object instead of a
849 /// datatype message of its own.
850 pub committed_type: Option<CommittedTypeRef>,
851 /// Dataspace (dimensionality).
852 pub dataspace: DataspaceMessage,
853 /// The object format the reopen found this dataset's messages written in,
854 /// `None` for a dataset this session created.
855 ///
856 /// A rewrite re-encodes the whole header — the shared-message table is
857 /// laid out whole, so every heap ID moves and every header naming one has
858 /// to be written again. Re-deriving the message format from the reopened
859 /// session's bounds would upgrade messages the file already has, which
860 /// libhdf5 never does: it grows a header in place and leaves every
861 /// message it did not touch alone. The same rule the reopen already
862 /// applies to a group it found in a symbol table
863 /// ([`uses_symbol_table`](Hdf5Writer::uses_symbol_table)) — what the file
864 /// says governs, not what this session's bound would have chosen.
865 pub read_format: Option<ObjectFormat>,
866 /// File offset of the dataset's object header (set during finalize).
867 pub obj_header_addr: u64,
868 /// File offset of the raw data block (contiguous only).
869 pub data_addr: u64,
870 /// Size of the raw data in bytes (contiguous only).
871 pub data_size: u64,
872 /// The raw data itself, for a compact dataset — the whole image, which
873 /// [`build_dataset_header`](Hdf5Writer::build_dataset_header) puts inside
874 /// the data layout message rather than in a block of its own. `Some` is
875 /// what makes a dataset compact, and the buffer is created at its final
876 /// length (filled, as `H5D__compact_fill` does, before any write), so it
877 /// is also the dataset's byte count; `data_addr`/`data_size` stay at the
878 /// "no block in the file" values a compact dataset shares with a NULL one.
879 pub compact: Option<Vec<u8>>,
880 /// The files this dataset's contiguous raw data lives in, when it lives
881 /// outside this HDF5 file. `Some` is what makes a contiguous dataset
882 /// externally stored: its `data_addr` stays [`UNDEF_ADDR`] and every byte
883 /// goes to the files named here instead of to a block of this file's own.
884 pub external: Option<ExternalStorage>,
885 /// The source datasets this dataset's elements are read from, when it is
886 /// virtual. `Some` is what makes it virtual, and it stores nothing of its
887 /// own: `data_addr`/`data_size` keep the "no block in this file" values a
888 /// compact dataset also has.
889 pub virtual_storage: Option<VirtualStorage>,
890 /// Chunked storage info (None for contiguous).
891 pub chunked: Option<ChunkedDatasetInfo>,
892 /// Fixed array chunked storage info.
893 pub fixed_array: Option<FixedArrayDatasetInfo>,
894 /// B-tree v2 chunked storage info.
895 pub btree_v2: Option<Bt2DatasetInfo>,
896 /// Implicit (no structure) chunked storage info.
897 pub implicit: Option<ImplicitDatasetInfo>,
898 /// Single-chunk chunked storage info: the whole (fixed) dataspace is
899 /// exactly one chunk.
900 pub single_chunk: Option<SingleChunkDatasetInfo>,
901 /// Version-1 B-tree chunked storage info — the classic-format index.
902 pub btree_v1: Option<BtreeV1DatasetInfo>,
903 /// Appended frames not yet written to chunks, `None` when empty.
904 pub append: Option<AppendBuffer>,
905 /// Attributes attached to this dataset.
906 pub attributes: Vec<AttributeEntry>,
907 /// File offset where the dataset object header was written (for SWMR in-place rewrites).
908 pub obj_header_written_addr: Option<u64>,
909 /// Encoded size of the dataset object header (for verifying in-place rewrites fit).
910 /// Every block the object's on-disk header occupies, chunk 0 first, or
911 /// empty when it has none yet. A rewrite keeps chunk 0's block — its
912 /// address is what every reference to the object holds — and frees the
913 /// rest, so a continuation block left behind is space no free-space
914 /// manager records.
915 pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
916 /// Filter pipeline for compressed chunks.
917 pub filter_pipeline: Option<FilterPipeline>,
918 /// Soft-deleted: excluded from finalize output.
919 pub deleted: bool,
920 /// The dataspace extent changed this session (`extend_dataset` /
921 /// `set_dataset_extent`). On a reopened dataset the finalize gate
922 /// otherwise infers "modified" from `chunks_written` alone, and a
923 /// session that only changed the extent would keep the old on-disk
924 /// header — silently dropping the new shape.
925 pub extent_dirty: bool,
926 /// Something the object header encodes changed this session without
927 /// touching the dataset's storage — an attribute set or removed, a fill
928 /// value defined. See [`header_stale`](DatasetInfo::header_stale).
929 pub header_dirty: bool,
930 /// The hard link count the on-disk header was written with, so finalize
931 /// can tell that this session changed it.
932 ///
933 /// A count, not a flag, because the count is what the header records and
934 /// the ways to change it are many: creating a link, unlinking one,
935 /// deleting a link's parent group, promoting a link to a primary name.
936 /// Comparing the value closes all of them at once, where a dirty flag
937 /// would have to be set at each and would be forgotten at the next one
938 /// added.
939 pub nlink_written: u32,
940 /// When the link naming this dataset was created; see
941 /// [`GroupInfo::creation_seq`].
942 pub creation_seq: u64,
943 /// How this dataset records creation order for its attributes — the
944 /// file's creation-order policy captured when the dataset was created,
945 /// the way libhdf5 captures the DCPL. A dataset holds no links, so only
946 /// the attribute half of [`TrackOrder`] applies to it.
947 pub track_attr_order: CreationOrder,
948 /// User-defined fill value bytes (exactly one element wide). `None`
949 /// means default zero-fill; `Some` is emitted as a `fill_defined = 2`
950 /// fill-value message in the dataset object header.
951 pub fill_value: Option<Vec<u8>>,
952 /// Fill value write time (`H5Pset_fill_time`'s `H5D_fill_time_t`, one of
953 /// [`FILL_TIME_ALLOC`], [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]),
954 /// emitted verbatim into the fill-value message's write-time field.
955 /// Defaults to `FILL_TIME_IFSET`, `H5D_CRT_FILL_TIME_DEF` — what a fresh
956 /// dataset creation property list carries until `set_dataset_fill_time`
957 /// says otherwise.
958 pub fill_time: u8,
959 /// Layout message version for chunked storage: 4, or 5 when the chunk
960 /// index encodes stored chunk sizes in a fixed `sizeof_size` field
961 /// (libhdf5 2.0). Chosen at create by `Hdf5Writer::chunk_layout_version`,
962 /// preserved from the file on reopen, and emitted verbatim at finalize.
963 /// Contiguous datasets ignore it.
964 pub layout_version: u8,
965 /// The times this object tracks: `Some` exactly when it was created with
966 /// `H5Pset_obj_track_times(true)`, `None` when it was not.
967 ///
968 /// One meaning on both header versions, which store them differently and
969 /// store different amounts of them: a version-2 header keeps all four in
970 /// its prefix, and a version-1 dataset keeps one, in an `H5O_MTIME_NEW`
971 /// message. [`touch_oh`] is the single place that turns this into either
972 /// of those, so the four fields are here whichever version the object
973 /// has, exactly as `H5O_t` carries `atime`/`mtime`/`ctime`/`btime` for a
974 /// version-1 header it never serialises them from.
975 pub times: Option<ObjectTimes>,
976}
977
978impl DatasetInfo {
979 /// Which chunk index this dataset uses, `None` for storage that is not
980 /// chunked — the one place the index-carrying fields are turned into an
981 /// answer.
982 ///
983 /// INVARIANT: a chunk index added to this struct is added here. A site
984 /// that spells the disjunction out itself is what classifies a new index
985 /// as contiguous storage, and contiguous storage is read and written at
986 /// [`data_addr`](Self::data_addr) — which a chunked dataset leaves
987 /// undefined, so the misclassification is a read or a write at
988 /// `UNDEF_ADDR` rather than an error.
989 pub(crate) fn chunk_index_kind(&self) -> Option<ChunkIndexKind> {
990 if self.chunked.is_some() {
991 Some(ChunkIndexKind::ExtensibleArray)
992 } else if self.fixed_array.is_some() {
993 Some(ChunkIndexKind::FixedArray)
994 } else if self.btree_v2.is_some() {
995 Some(ChunkIndexKind::BtreeV2)
996 } else if self.implicit.is_some() {
997 Some(ChunkIndexKind::Implicit)
998 } else if self.single_chunk.is_some() {
999 Some(ChunkIndexKind::SingleChunk)
1000 } else if self.btree_v1.is_some() {
1001 Some(ChunkIndexKind::BtreeV1)
1002 } else {
1003 None
1004 }
1005 }
1006
1007 /// Whether this dataset's raw data is stored in chunks — the question
1008 /// every storage-form test asks, asked in one place.
1009 pub(crate) fn is_chunked(&self) -> bool {
1010 self.chunk_index_kind().is_some()
1011 }
1012
1013 /// Where this dataset's contiguous raw bytes live, or `None` when it has
1014 /// no contiguous storage to write into at all — a chunked dataset, a
1015 /// compact one (whose bytes *are* the layout message), or one whose block
1016 /// was never allocated.
1017 ///
1018 /// INVARIANT: every write of a contiguous dataset's raw bytes picks its
1019 /// destination here and reaches it through
1020 /// [`Hdf5Writer::write_contiguous_bytes`]. A site that read `data_addr`
1021 /// itself would write an externally-stored dataset's data into this file
1022 /// — at [`UNDEF_ADDR`], the far end of the address space — instead of into
1023 /// the files its header names, and would do the same to a virtual one,
1024 /// whose bytes are not this file's to write at all.
1025 ///
1026 /// Chunked storage is excluded through
1027 /// [`chunk_index_kind`](Self::chunk_index_kind) rather than by naming the
1028 /// index-carrying fields, so an index added to this struct cannot arrive
1029 /// here as contiguous storage: an implicit-indexed dataset reads
1030 /// `data_addr` as the base of its chunk grid, which as a contiguous
1031 /// destination would take a raw write meant for one chunk and lay it over
1032 /// the whole grid.
1033 fn contiguous_target(&self) -> Option<ContiguousTarget> {
1034 if self.is_chunked() || self.compact.is_some() {
1035 return None;
1036 }
1037 if self.virtual_storage.is_some() {
1038 return Some(ContiguousTarget::Virtual);
1039 }
1040 match &self.external {
1041 Some(ext) => Some(ContiguousTarget::External {
1042 files: ext.files.clone(),
1043 prefix: ext.prefix.expanded.clone(),
1044 }),
1045 None => {
1046 (self.data_addr != UNDEF_ADDR).then_some(ContiguousTarget::Local(self.data_addr))
1047 }
1048 }
1049 }
1050
1051 /// The one run of file bytes an implicitly indexed dataset's chunk grid
1052 /// is — its start and its length — or `None` when the dataset is indexed
1053 /// some other way or its space is not allocated yet.
1054 ///
1055 /// That index has no per-chunk structure to hold an address in: every
1056 /// chunk sits at `data_addr + linear_index * chunk_bytes` and the grid is
1057 /// allocated whole at create (`H5D__none_idx_get_addr`, H5Dnone.c). So the
1058 /// run is file space this writer allocated, and it is the *only* storage a
1059 /// chunk of such a dataset can occupy — the builder refuses external and
1060 /// virtual storage together with chunked storage, which is why
1061 /// [`allocated_storage_run`](Self::allocated_storage_run) can name it
1062 /// [`ContiguousTarget::Local`] and no chunk write can reach the other two.
1063 fn implicit_grid(&self) -> Option<(u64, u64)> {
1064 let imp = self.implicit.as_ref()?;
1065 (imp.data_addr != UNDEF_ADDR).then_some((imp.data_addr, imp.data_size))
1066 }
1067
1068 /// The run of raw storage this writer *allocated* for the dataset — the
1069 /// target to initialise it through and its size — or `None` when it
1070 /// allocated none.
1071 ///
1072 /// The two storage forms that are one run of bytes: a contiguous
1073 /// dataset's data block, and an implicitly indexed dataset's chunk grid.
1074 /// A compact dataset is excluded (its bytes are its layout message) and so
1075 /// is every other chunk index, whose chunks are placed one at a time.
1076 ///
1077 /// External storage is excluded because this writer does not allocate it:
1078 /// `H5D__alloc_storage` skips its whole body — the space reservation and
1079 /// the `H5D__init_storage` that would tile the fill value into it — for a
1080 /// dataset with an external file list or an empty extent, "we assume that
1081 /// external storage is already allocated by the caller, or at least will
1082 /// be before I/O is performed" (H5Dint.c:2270-2274). Measured under
1083 /// libhdf5 1.14.6 and 2.0.0: a user fill value, `H5D_FILL_TIME_ALLOC` and
1084 /// `H5D_ALLOC_TIME_EARLY` together leave the raw data file uncreated at
1085 /// `H5Dcreate2`, and a read before any write fails with "unable to open
1086 /// external raw data file" rather than reporting the fill.
1087 ///
1088 /// INVARIANT: only storage whose bytes this file owns is initialised as
1089 /// one run, so the allocate-time fill cannot reach the files an external
1090 /// file list names or the sources a virtual dataset maps.
1091 fn allocated_storage_run(&self) -> Option<(ContiguousTarget, u64)> {
1092 match self.implicit_grid() {
1093 Some((addr, size)) => Some((ContiguousTarget::Local(addr), size)),
1094 // Not a fallthrough for an unallocated implicit grid:
1095 // `contiguous_target` answers `None` for every chunked dataset.
1096 None => match self.contiguous_target() {
1097 Some(t @ ContiguousTarget::Local(_)) => Some((t, self.data_size)),
1098 _ => None,
1099 },
1100 }
1101 }
1102
1103 /// Whether this session wrote chunk data or changed the extent, so the
1104 /// dataset's index structures have to be re-flushed.
1105 fn storage_dirty(&self) -> bool {
1106 self.chunked.as_ref().is_some_and(|c| c.chunks_written > 0)
1107 || self
1108 .fixed_array
1109 .as_ref()
1110 .is_some_and(|f| f.chunks_written > 0)
1111 || self.btree_v2.as_ref().is_some_and(|b| b.chunks_written > 0)
1112 || self.btree_v1.as_ref().is_some_and(|b| b.chunks_written > 0)
1113 || self
1114 .single_chunk
1115 .as_ref()
1116 .is_some_and(|s| s.chunks_written > 0)
1117 || self.extent_dirty
1118 }
1119
1120 /// Whether a reopened dataset's on-disk object header no longer describes
1121 /// it.
1122 ///
1123 /// INVARIANT: every mutation of something `build_dataset_header` encodes
1124 /// must show up here. Finalize keeps the original header when this is
1125 /// false, so a change this misses is not deferred — it is discarded, with
1126 /// no error to say so. Attributes were the case that proved it: they are
1127 /// invisible to the chunk-write counters, so an attribute set on a
1128 /// reopened dataset vanished at close.
1129 fn header_stale(&self) -> bool {
1130 self.storage_dirty() || self.header_dirty
1131 }
1132
1133 /// The same question for the one thing the dataset itself cannot see: how
1134 /// many hard links resolve to it. That count lives in the header — an
1135 /// Object Reference Count message in a version-2 header, the `nlink`
1136 /// prefix field of a version-1 one — but it is a property of the file's
1137 /// link graph, so the caller supplies today's value.
1138 fn header_stale_with(&self, nlink: u32) -> bool {
1139 self.header_stale() || nlink != self.nlink_written
1140 }
1141
1142 /// Record that this dataset's on-disk object header was just written with
1143 /// `nlink` in it.
1144 ///
1145 /// INVARIANT: every write of a dataset object header passes through here.
1146 /// [`header_stale_with`](Self::header_stale_with) is the one authority for
1147 /// "does what is on disk still describe this dataset?", and it answers by
1148 /// comparing against [`nlink_written`](Self::nlink_written) — so a site
1149 /// that writes a header without saying so leaves that answer describing an
1150 /// older write. There are three writers: `finalize`, `finalize_for_swmr`
1151 /// and `write_dataset_header_inplace`. The last recorded nothing; it could
1152 /// not drift today only because a count it could write is a count that
1153 /// makes the header outgrow its block, which it refuses. That is a
1154 /// property of the reference-count message's size, not a rule anything
1155 /// states, and it is not what the field's definition rests on.
1156 fn header_written(&mut self, nlink: u32) {
1157 self.nlink_written = nlink;
1158 }
1159}
1160
1161/// Runtime metadata for a chunked dataset.
1162pub struct ChunkedDatasetInfo {
1163 /// Chunk dimension sizes.
1164 pub chunk_dims: Vec<u64>,
1165 /// Extensible array parameters.
1166 pub earray_params: EarrayParams,
1167 /// File offset of the EA header.
1168 pub ea_header_addr: u64,
1169 /// File offset of the EA index block.
1170 pub ea_iblk_addr: u64,
1171 /// In-memory copy of the EA header (for updating statistics).
1172 pub ea_header: ExtensibleArrayHeader,
1173 /// In-memory copy of the EA index block (for unfiltered datasets).
1174 pub ea_iblk: ExtensibleArrayIndexBlock,
1175 /// Number of chunks written so far.
1176 pub chunks_written: u64,
1177 /// Filtered index block (for compressed datasets).
1178 pub filt_iblk: Option<FilteredIndexBlock>,
1179 /// chunk_size_len for filtered entries.
1180 pub chunk_size_len: u8,
1181}
1182
1183/// Where a newly-created EA data block's address must be recorded.
1184enum DblkParent {
1185 /// Slot `index_block.dblk_addrs[idx]`.
1186 IndexBlock(usize),
1187 /// Slot `super_block.dblk_addrs[local_dblk]` of the super block at `sblk_addr`.
1188 SuperBlock {
1189 sblk_addr: u64,
1190 ndblks_in_sblk: usize,
1191 local_dblk: usize,
1192 },
1193}
1194
1195/// Which attribute list an attribute operation targets: the root group's,
1196/// a group's (by full path), or a dataset's (by writer index).
1197#[derive(Clone, Copy)]
1198pub enum AttrTarget<'a> {
1199 /// The root group's (file-level) attributes.
1200 Root,
1201 /// A group's attributes, by full path.
1202 Group(&'a str),
1203 /// A dataset's attributes, by writer index.
1204 Dataset(usize),
1205}
1206
1207/// Which chunk index a dataset uses.
1208///
1209/// The five above the line are what `H5D__layout_set_latest_indexing`
1210/// (H5Dlayout.c) picks between once the file format allows a version-4 data
1211/// layout message, in this precedence: a v2 B-tree for two or more unlimited
1212/// dimensions, an extensible array for exactly one, and — for a fixed shape —
1213/// the single-chunk index whenever exactly one chunk covers the whole
1214/// dataspace (`dims == max_dims == chunk_dims`, checked before either
1215/// alternative below and taken regardless of filter or allocation-time), else
1216/// the implicit index when nothing has to be recorded per chunk (no filter,
1217/// early allocation), else a fixed array. [`BtreeV1`](Self::BtreeV1) is not
1218/// one of them: it belongs to the version-3 layout message, and a file whose
1219/// superblock is older than version 2 can carry no other.
1220#[derive(Clone, Copy, PartialEq, Eq, Debug)]
1221pub(crate) enum ChunkIndexKind {
1222 ExtensibleArray,
1223 FixedArray,
1224 BtreeV2,
1225 Implicit,
1226 SingleChunk,
1227 BtreeV1,
1228}
1229
1230/// A chunked dataset's grid geometry, snapshotted out of its slot.
1231///
1232/// The single owner of chunk-grid arithmetic: how many chunks span each
1233/// dimension, where a coordinate sits in the row-major order the array
1234/// indices record, and how many bytes one chunk holds.
1235struct ChunkGeometry {
1236 kind: ChunkIndexKind,
1237 dims: Vec<u64>,
1238 max_dims: Option<Vec<u64>>,
1239 chunk_dims: Vec<u64>,
1240 element_size: u64,
1241}
1242
1243impl ChunkGeometry {
1244 /// Unfiltered byte size of one whole chunk.
1245 fn chunk_bytes(&self) -> u64 {
1246 self.chunk_dims.iter().product::<u64>() * self.element_size
1247 }
1248
1249 /// Row-major position of `coords` in the chunk grid — the linear index an
1250 /// extensible or fixed array records the chunk under, computed against
1251 /// the maximum-extent grid by [`crate::io::chunk_grid::linear_index`].
1252 fn linear_index(&self, coords: &[u64]) -> IoResult<u64> {
1253 crate::io::chunk_grid::linear_index(
1254 &self.dims,
1255 self.max_dims.as_deref(),
1256 &self.chunk_dims,
1257 coords,
1258 )
1259 }
1260}
1261
1262/// The refusal every attribute mutation gets while SWMR streaming is
1263/// active, from the two owners of attribute-list change
1264/// ([`Hdf5Writer::set_attribute`] and `evict_attr`).
1265fn swmr_attr_error(name: &str) -> crate::io::IoError {
1266 crate::io::IoError::InvalidState(format!(
1267 "cannot add or modify attribute '{name}' during SWMR streaming: object \
1268 headers are frozen while readers stream, and a superseded variable-length \
1269 value's heap storage could never be reclaimed; set attributes before \
1270 start_swmr (libhdf5 forbids attribute changes during SWMR writes too)"
1271 ))
1272}
1273
1274/// Where an attribute arriving at [`Hdf5Writer::insert_attribute`] came from.
1275///
1276/// The variable-length setters have to evict before they allocate — the
1277/// free-before-alloc order — so by the time the replacement is inserted the
1278/// list no longer holds the entry it replaces, and the ordinary "already
1279/// present, so keep its index" test cannot see it. `H5A__attr_write` does not
1280/// create the attribute again, so the index travels with the eviction rather
1281/// than being stamped afresh; without it a rewritten attribute takes the set's
1282/// running maximum and moves to the end of the creation order.
1283#[derive(Debug, Clone, Copy)]
1284enum AttrOrigin {
1285 /// A new attribute, which takes the set's next creation index.
1286 Created,
1287 /// A value written over an attribute this writer has just evicted, which
1288 /// keeps that attribute's creation index — `None` when the object tracks
1289 /// no order, and so records none. An eviction that found nothing to remove
1290 /// answers `Created`: what follows it is a create like any other.
1291 Rewritten(Option<u16>),
1292}
1293use AttrOrigin::{Created, Rewritten};
1294
1295/// Take an object's attributes into the append session, or refuse the reopen.
1296///
1297/// Append mode rebuilds every object header it touches out of the attributes
1298/// read from it, so what this returns is what the object will still have when
1299/// the session finalizes. An attribute set that could not be read whole —
1300/// `ObjectAttributes::into_complete` refuses it — would come back as the part
1301/// that did read, silently deleting the rest.
1302///
1303/// Left to surface at `finalize`, that failure would land after this session's
1304/// chunk data and indices had already been written past the allocation point
1305/// the superblock still records, leaving a file libhdf5 reads as truncated.
1306/// Refusing the open leaves it untouched.
1307///
1308/// Size is no longer a reason to refuse: an attribute too large for a header
1309/// message goes back out through dense storage, the form libhdf5 read it from.
1310///
1311/// The set comes back in creation-index order, which is the order the registry
1312/// holds attributes in for an object made in this session too. A dense set is
1313/// read through the name index, so the order it arrives in is the order a hash
1314/// walk took; sorting here is what makes "the list is in creation order" true
1315/// of a reopened object as well, without any later stage having to know which
1316/// storage form the attributes came out of. Attributes of an untracked object
1317/// carry no index and keep the order they were read in.
1318fn take_reopened_attributes(
1319 attrs: crate::io::reader::ObjectAttributes,
1320 owner: &str,
1321) -> IoResult<Vec<AttributeEntry>> {
1322 let mut attrs = attrs.into_complete(owner)?;
1323 attrs.sort_by_key(|a| a.creation_index());
1324 Ok(attrs)
1325}
1326
1327/// The creation-order policy an on-disk object header declares — the single
1328/// owner of the recovery rule, used for the root group, every reopened group
1329/// and (through its attribute half) every reopened dataset.
1330///
1331/// The two halves come from two different places, and reading one for both is
1332/// how a file that sets only one of them came back with both or neither:
1333///
1334/// * links — the `Link Info` message's flag bits, which is what
1335/// `H5Pget_link_creation_order` reads (`H5G__get_create_plist`). A group
1336/// with no such message (or one this crate cannot decode) tracks nothing;
1337/// so does every dataset, which has no links to order.
1338/// * attributes — the object header's own flag bits, which is what
1339/// `H5Pget_attr_creation_order` reads (`H5Pocpl.c`). The `Attribute Info`
1340/// message carries the same two bits, but the header is the authority
1341/// libhdf5 consults, and it is present even when the object has no
1342/// attributes yet.
1343fn recover_track_order(
1344 header: &crate::format::object_header::ObjectHeader,
1345 ctx: &FormatContext,
1346) -> TrackOrder {
1347 let links = header
1348 .messages
1349 .iter()
1350 .find(|m| m.msg_type == crate::format::messages::MSG_LINK_INFO)
1351 .and_then(|m| LinkInfoMessage::decode(&m.data, ctx).ok())
1352 .map(|(info, _)| info.creation_order())
1353 .unwrap_or_default();
1354 TrackOrder {
1355 links,
1356 attrs: header.attribute_creation_order(),
1357 }
1358}
1359
1360/// `H5O_touch_oh` (H5Oint.c:1273): put an object's tracked times where its
1361/// header version keeps them.
1362///
1363/// INVARIANT: every object header this writer builds passes its times through
1364/// here. The version decides the storage and nothing else does — a caller that
1365/// set `ObjectHeader::times` itself would hand a version-1 encode a prefix
1366/// field that version has no room for, and one that added the message itself
1367/// would put a second copy in a version-2 header.
1368///
1369/// `force` is upstream's own parameter, and it is what splits datasets from
1370/// everything else: it creates the version-1 `H5O_MTIME_NEW` message when the
1371/// header has none, and only `H5D__update_oh_info` passes it true
1372/// (H5Dint.c:1022-1026). Every other caller passes false and so creates no
1373/// message at all, which is why a version-1 group or committed datatype
1374/// records no time even when it is tracking them. A version-2 header keeps all
1375/// four times in its prefix whatever `force` says.
1376fn touch_oh(
1377 header: &mut ObjectHeader,
1378 format: ObjectFormat,
1379 times: Option<ObjectTimes>,
1380 force: bool,
1381) {
1382 let Some(times) = touched_times(times) else {
1383 return;
1384 };
1385 match format {
1386 ObjectFormat::Modern => header.times = Some(times),
1387 ObjectFormat::Legacy if force => header.add_message(
1388 crate::format::messages::MSG_MOD_TIME,
1389 0x00,
1390 ModificationTime(times.change).encode(),
1391 ),
1392 ObjectFormat::Legacy => {}
1393 }
1394}
1395
1396/// The times a header being (re)written carries, given what the object had.
1397///
1398/// Every object header this writer emits is one it is writing *now*, which is
1399/// what `H5O_touch_oh` is called for: an object that stores times gets its
1400/// access and change time moved to now, and one that does not store them stays
1401/// that way — the flag belongs to the object's creation property list, and a
1402/// rewrite is not a creation.
1403fn touched_times(times: Option<ObjectTimes>) -> Option<ObjectTimes> {
1404 times.map(|t| t.touched(now_seconds()))
1405}
1406
1407/// Seconds since the epoch, as an object header stores them (`H5_now`).
1408///
1409/// Saturates rather than wrapping: the field is a 32-bit count, and a clock
1410/// past 2106 is better reported as the largest time the format can express
1411/// than as a time in 1970. A clock before the epoch yields 0, which is what
1412/// libhdf5 writes for "no time recorded".
1413fn now_seconds() -> u32 {
1414 std::time::SystemTime::now()
1415 .duration_since(std::time::UNIX_EPOCH)
1416 .map_or(0, |d| u32::try_from(d.as_secs()).unwrap_or(u32::MAX))
1417}
1418
1419/// The dense storage an on-disk object header names: the fractal heap and the
1420/// indices its `Attribute Info` and `Link Info` messages point at.
1421///
1422/// A rewrite of that header lays fresh storage out and stops naming this, so
1423/// what this returns is exactly what the rewrite supersedes and must free.
1424/// Compact storage names no heap and yields `None` — there is nothing to free
1425/// and nothing that could be freed twice.
1426fn superseded_dense(
1427 header: &crate::format::object_header::ObjectHeader,
1428 ctx: &FormatContext,
1429) -> (Option<AttributeInfoMessage>, Option<LinkInfoMessage>) {
1430 let decode = |msg_type: u8| {
1431 header
1432 .messages
1433 .iter()
1434 .find(|m| m.msg_type == msg_type)
1435 .map(|m| m.data.as_slice())
1436 };
1437 let attrs = decode(crate::format::messages::MSG_ATTR_INFO)
1438 .and_then(|d| AttributeInfoMessage::decode(d, ctx).ok())
1439 .map(|(info, _)| info)
1440 .filter(|info| info.is_dense());
1441 let links = decode(crate::format::messages::MSG_LINK_INFO)
1442 .and_then(|d| LinkInfoMessage::decode(d, ctx).ok())
1443 .map(|(info, _)| info)
1444 .filter(|info| info.is_dense());
1445 (attrs, links)
1446}
1447
1448/// One collection block with free space that a later vlen insert may
1449/// fill — an entry in the writer's CWFS list (libhdf5 `f->shared->cwfs`).
1450struct CwfsEntry {
1451 /// Block address of the collection.
1452 addr: u64,
1453 /// Declared block size; never changes after allocation.
1454 size: usize,
1455 /// Bytes its free-space marker owns, per
1456 /// [`GlobalHeapCollection::free_space_at`](crate::format::global_heap::GlobalHeapCollection::free_space_at).
1457 free: usize,
1458}
1459
1460/// Maximum CWFS entries tracked — libhdf5's `H5HG_NCWFS` (H5HGpkg.h).
1461const H5HG_NCWFS: usize = 16;
1462
1463/// Record a collection with `free` bytes in the CWFS list: update its
1464/// entry if present, append while the list is short, and otherwise
1465/// replace the entry with the least free space when this one has more —
1466/// the retention rule of libhdf5's `H5HG_insert`.
1467fn cwfs_note(cwfs: &mut Vec<CwfsEntry>, addr: u64, size: usize, free: usize) {
1468 if let Some(p) = cwfs.iter().position(|e| e.addr == addr) {
1469 cwfs[p].free = free;
1470 return;
1471 }
1472 if cwfs.len() < H5HG_NCWFS {
1473 cwfs.insert(0, CwfsEntry { addr, size, free });
1474 return;
1475 }
1476 if let Some(p) = (0..cwfs.len()).min_by_key(|&p| cwfs[p].free) {
1477 if free > cwfs[p].free {
1478 cwfs[p] = CwfsEntry { addr, size, free };
1479 }
1480 }
1481}
1482
1483/// The uniform rejection for `delete_dataset` / `delete_group` while SWMR
1484/// streaming is active: deleting frees the object's blocks, and a live
1485/// reader may hold any of their addresses.
1486fn swmr_delete_error(name: &str) -> crate::io::IoError {
1487 crate::io::IoError::InvalidState(format!(
1488 "cannot delete '{name}' during SWMR streaming: a reader may hold the \
1489 object's header and storage addresses (libhdf5 forbids link deletion \
1490 during SWMR writes too)"
1491 ))
1492}
1493
1494/// Whether the chunk at grid `coords` lies entirely at or beyond `extent` in
1495/// some dimension — no element of it would survive a shrink to that extent.
1496fn chunk_outside_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1497 coords
1498 .iter()
1499 .zip(chunk_dims)
1500 .zip(extent)
1501 .any(|((&c, &cd), &e)| c.saturating_mul(cd) >= e)
1502}
1503
1504/// Whether the chunk at grid `coords` keeps elements under `extent` but
1505/// extends past it in some dimension — a shrink must refill its
1506/// out-of-extent region with the fill value.
1507fn chunk_straddles_extent(coords: &[u64], chunk_dims: &[u64], extent: &[u64]) -> bool {
1508 !chunk_outside_extent(coords, chunk_dims, extent)
1509 && coords
1510 .iter()
1511 .zip(chunk_dims)
1512 .zip(extent)
1513 .any(|((&c, &cd), &e)| (c + 1).saturating_mul(cd) > e)
1514}
1515
1516/// Overwrite, in `data` (one whole chunk, unfiltered, row-major), every
1517/// element at or beyond `extent` with the matching bytes of `fill` — a
1518/// same-sized buffer tiled with the fill value. The caller guarantees the
1519/// chunk at `coords` straddles `extent`, so every dimension keeps at least
1520/// one element. Returns the replaced bytes, so a vlen dataset's dead
1521/// heap references can be released rather than stranded.
1522fn refill_chunk_beyond_extent(
1523 data: &mut [u8],
1524 fill: &[u8],
1525 coords: &[u64],
1526 chunk_dims: &[u64],
1527 extent: &[u64],
1528 element_size: usize,
1529) -> Vec<u8> {
1530 let ndims = chunk_dims.len();
1531 let keep: Vec<usize> = (0..ndims)
1532 .map(|d| {
1533 let origin = coords[d] * chunk_dims[d];
1534 chunk_dims[d].min(extent[d].saturating_sub(origin)) as usize
1535 })
1536 .collect();
1537 // Row-major walk: for every row (all dimensions but the last),
1538 // overwrite the whole row when its prefix is outside the keep box,
1539 // else only the row's out-of-extent tail.
1540 let row_elems = chunk_dims[ndims - 1] as usize;
1541 let keep_last = keep[ndims - 1];
1542 let nrows: u64 = chunk_dims[..ndims - 1].iter().product();
1543 let mut replaced = Vec::new();
1544 for r in 0..nrows {
1545 let mut rem = r;
1546 let mut in_keep = true;
1547 for d in (0..ndims - 1).rev() {
1548 let c = rem % chunk_dims[d];
1549 rem /= chunk_dims[d];
1550 if c as usize >= keep[d] {
1551 in_keep = false;
1552 }
1553 }
1554 let start = if in_keep { keep_last } else { 0 };
1555 if start == row_elems {
1556 continue;
1557 }
1558 let a = (r as usize * row_elems + start) * element_size;
1559 let b = (r as usize + 1) * row_elems * element_size;
1560 replaced.extend_from_slice(&data[a..b]);
1561 data[a..b].copy_from_slice(&fill[a..b]);
1562 }
1563 replaced
1564}
1565
1566/// Validate caller-supplied chunk geometry at dataset definition, the rule
1567/// libhdf5 applies in `H5D__chunk_construct` (H5Dchunk.c): the chunk rank
1568/// must match the dataspace rank, no chunk dimension may be zero, and a
1569/// chunk dimension may not exceed a fixed maximum dimension — except in a
1570/// dimension whose current size is zero, which libhdf5 exempts.
1571fn validate_chunk_geometry(dims: &[u64], max_dims: &[u64], chunk_dims: &[u64]) -> IoResult<()> {
1572 let ndims = dims.len();
1573 if chunk_dims.len() != ndims {
1574 return Err(crate::io::IoError::InvalidState(format!(
1575 "chunk shape has {} dimensions but the dataspace has {}",
1576 chunk_dims.len(),
1577 ndims
1578 )));
1579 }
1580 if max_dims.len() != ndims {
1581 return Err(crate::io::IoError::InvalidState(format!(
1582 "maximum shape has {} dimensions but the dataspace has {}",
1583 max_dims.len(),
1584 ndims
1585 )));
1586 }
1587 for d in 0..ndims {
1588 if chunk_dims[d] == 0 {
1589 return Err(crate::io::IoError::InvalidState(format!(
1590 "chunk dimension {d} is zero"
1591 )));
1592 }
1593 if dims[d] != 0 && max_dims[d] != u64::MAX && max_dims[d] < chunk_dims[d] {
1594 return Err(crate::io::IoError::InvalidState(format!(
1595 "chunk dimension {} is {} but the maximum dimension size is {}",
1596 d, chunk_dims[d], max_dims[d]
1597 )));
1598 }
1599 }
1600 Ok(())
1601}
1602
1603/// An extensible-array index requires at most one unlimited dimension —
1604/// `H5D__chunk_construct` (H5Dchunk.c) only selects this index for exactly
1605/// one — at any position: `chunk_grid::linear_index` seeds the unlimited
1606/// dimension into the slot no down-chunks multiplier touches, the same
1607/// address libhdf5 reaches by swizzling it to the slowest position
1608/// (`H5VM_swizzle_coords`, H5Dearray.c). Two or more unlimited dimensions
1609/// have no finite grid at all; that shape needs a v2 B-tree index instead.
1610fn ensure_at_most_one_unlimited(max_dims: &[u64]) -> IoResult<()> {
1611 let unlimited: Vec<usize> = max_dims
1612 .iter()
1613 .enumerate()
1614 .filter(|&(_, &m)| m == u64::MAX)
1615 .map(|(d, _)| d)
1616 .collect();
1617 if unlimited.len() > 1 {
1618 return Err(crate::io::IoError::InvalidState(format!(
1619 "an extensible-array index supports at most one unlimited dimension, \
1620 but dimensions {unlimited:?} are all unlimited; a v2 B-tree index \
1621 handles two or more"
1622 )));
1623 }
1624 Ok(())
1625}
1626
1627/// Reject strings the dataset's declared character set cannot label.
1628///
1629/// A Rust `&str` is always UTF-8, so only an ASCII declaration (charset 0)
1630/// can be violated. libhdf5 stores the bytes unvalidated — its vlen write
1631/// path has no cset check anywhere — which mislabels them for every reader
1632/// that trusts the declaration (h5py raises on the same mismatch).
1633fn ensure_vlen_charset(charset: u8, strings: &[&str]) -> IoResult<()> {
1634 if charset == 0 {
1635 if let Some((i, s)) = strings.iter().enumerate().find(|(_, s)| !s.is_ascii()) {
1636 return Err(crate::io::IoError::InvalidState(format!(
1637 "string {i} ({s:?}) is not ASCII, but the dataset's character set is"
1638 )));
1639 }
1640 }
1641 Ok(())
1642}
1643
1644/// Runtime metadata for a fixed-array-indexed chunked dataset.
1645pub struct FixedArrayDatasetInfo {
1646 /// Chunk dimension sizes.
1647 pub chunk_dims: Vec<u64>,
1648 /// File offset of the FA header.
1649 pub fa_header_addr: u64,
1650 /// File offset of the FA data block.
1651 pub fa_dblk_addr: u64,
1652 /// In-memory copy of the FA header.
1653 pub fa_header: FixedArrayHeader,
1654 /// In-memory copy of the FA data block.
1655 pub fa_dblk: FixedArrayDataBlock,
1656 /// Number of chunks written so far.
1657 pub chunks_written: u64,
1658}
1659
1660/// Runtime metadata for an implicitly indexed chunked dataset — the index
1661/// that is no structure at all (`H5Dnone.c`).
1662///
1663/// Every chunk of the maximum-extent grid is allocated at create in one
1664/// contiguous run, in the row-major order [`crate::io::chunk_grid`] defines,
1665/// so a chunk's address is `data_addr + linear_index * chunk_bytes` and
1666/// nothing has to be recorded when one is written. libhdf5 picks this index
1667/// only when that arithmetic is total: no filter (every chunk is exactly
1668/// `chunk_bytes` long), no unlimited dimension (the run has a finite length),
1669/// and early allocation (the run exists before any write).
1670pub struct ImplicitDatasetInfo {
1671 /// Chunk dimension sizes.
1672 pub chunk_dims: Vec<u64>,
1673 /// File offset of the first chunk — the layout message's index address.
1674 pub data_addr: u64,
1675 /// Byte length of the whole chunk run: `nchunks * chunk_bytes`.
1676 pub data_size: u64,
1677}
1678
1679/// Runtime metadata for a single-chunk indexed dataset (`H5Dsingle.c`): a
1680/// fixed dataspace exactly one chunk wide in every dimension
1681/// (`dims == max_dims == chunk_dims`), so there is exactly one chunk and its
1682/// address — and, when filtered, its stored size and filter mask — are held
1683/// directly in the layout message rather than in any index structure.
1684///
1685/// libhdf5 selects this index ahead of the implicit and fixed-array indexes
1686/// whenever the shape qualifies, whether or not the dataset is filtered or
1687/// early-allocated (`H5D__layout_set_latest_indexing`, H5Dlayout.c).
1688pub struct SingleChunkDatasetInfo {
1689 /// Chunk dimension sizes (equal to the dataspace's `dims`).
1690 pub chunk_dims: Vec<u64>,
1691 /// File offset of the chunk, [`UNDEF_ADDR`] until the chunk is written
1692 /// (or immediately, for an unfiltered dataset created with early
1693 /// allocation).
1694 pub data_addr: u64,
1695 /// The chunk's full unfiltered byte length — `chunk_dims.product() *
1696 /// element_size`, fixed for the dataset's lifetime.
1697 pub data_size: u64,
1698 /// Stored (on-disk) byte length: equal to `data_size` when the dataset
1699 /// carries no filter pipeline; the filtered length once the chunk has
1700 /// been written, 0 before then.
1701 pub nbytes: u64,
1702 /// Filter mask recorded for the stored chunk (bit *i* set means filter
1703 /// *i* was skipped); meaningful only when the dataset is filtered.
1704 pub filter_mask: u32,
1705 /// Chunks written this session (0 or 1) — `storage_dirty`'s signal that
1706 /// the layout message's address/size/mask fields must be re-flushed.
1707 pub chunks_written: u64,
1708 /// Whether this dataset was created with early allocation
1709 /// (`H5D_ALLOC_TIME_EARLY`) — distinct from `data_addr` being defined,
1710 /// which also becomes true the moment an incrementally allocated
1711 /// dataset's one chunk is written; `build_dataset_header` needs this to
1712 /// tell the two apart when it reports the fill-value message's
1713 /// allocation time. Only ever set for an unfiltered dataset: a filtered
1714 /// chunk's stored length is not known until it is compressed, so there
1715 /// is nothing to allocate ahead of that write regardless of alloc time
1716 /// (the same gap `create_fixed_array_dataset_with_pipeline` has).
1717 pub early_alloc: bool,
1718}
1719
1720/// One chunk as the version-1 B-tree records it — the key libhdf5 stores
1721/// (`H5D_btree_key_t`) plus the address it keys.
1722pub struct BtreeV1ChunkRecord {
1723 /// Grid position of the chunk. The key's element offsets are derived from
1724 /// it at encode time (`scaled * chunk_dim`), so this is the one place the
1725 /// position is stored and the sort order is over these coordinates.
1726 pub scaled: Vec<u64>,
1727 /// File offset of the chunk's bytes.
1728 pub address: u64,
1729 /// Stored byte length — the filtered length when the dataset is filtered,
1730 /// the full chunk otherwise. `u32` because the key's field is.
1731 pub nbytes: u32,
1732 /// Filter mask: bit `i` set means filter `i` was skipped for this chunk.
1733 pub filter_mask: u32,
1734}
1735
1736/// Runtime metadata for a chunked dataset indexed by a version-1 B-tree —
1737/// the classic-format chunk index (`H5Dbtree.c`), and the only one a
1738/// version-0/1 superblock file can carry.
1739pub struct BtreeV1DatasetInfo {
1740 /// Chunk dimension sizes.
1741 pub chunk_dims: Vec<u64>,
1742 /// Maximum dimensions (u64::MAX = unlimited).
1743 pub max_dims: Vec<u64>,
1744 /// The file's v1-B-tree "K" ranks. Every node's width is derived from
1745 /// them, and they are recorded only in the superblock this file was
1746 /// opened with — so they are carried rather than re-derived.
1747 pub config: BTreeV1Config,
1748 /// The chunks, in key order (`scaled` ascending, lexicographically).
1749 pub records: Vec<BtreeV1ChunkRecord>,
1750 /// Pool of node-size blocks holding the tree's nodes, on the same terms
1751 /// as [`Bt2DatasetInfo::node_addrs`]: a flush re-serializes the whole
1752 /// bulk-loaded tree over them and allocates only the shortfall, so no
1753 /// flush can orphan a block it replaced.
1754 pub node_addrs: Vec<u64>,
1755 /// Address of the tree's root node — what the version-3 data layout
1756 /// message carries. `UNDEF_ADDR` until a flush puts a node in the file,
1757 /// which is the state libhdf5 leaves a chunked dataset in until its first
1758 /// chunk is written.
1759 pub root_addr: u64,
1760 /// Number of chunks written so far.
1761 pub chunks_written: u64,
1762}
1763
1764impl BtreeV1DatasetInfo {
1765 /// The chunk shape a key's offsets are scaled by: the chunk dimensions
1766 /// with the element size appended, which is also what the layout message
1767 /// stores.
1768 fn key_dims(&self, element_size: u64) -> Vec<u64> {
1769 let mut dims = self.chunk_dims.clone();
1770 dims.push(element_size);
1771 dims
1772 }
1773
1774 /// Bulk-load the tree this index's records describe.
1775 fn build_tree(&self, element_size: u64, sizeof_addr: usize) -> ChunkBTreeV1Tree {
1776 let dims = self.key_dims(element_size);
1777 let entries: Vec<(ChunkKey, u64)> = self
1778 .records
1779 .iter()
1780 .map(|r| {
1781 (
1782 ChunkKey::for_chunk(&r.scaled, &dims, r.nbytes, r.filter_mask),
1783 r.address,
1784 )
1785 })
1786 .collect();
1787 // The right boundary closes the tree past its greatest key, which is
1788 // the last record's — the records are kept in key order.
1789 let last = self
1790 .records
1791 .last()
1792 .map_or_else(|| vec![0; self.chunk_dims.len()], |r| r.scaled.clone());
1793 ChunkBTreeV1Tree::build(
1794 &entries,
1795 ChunkKey::right_bound(&last, &dims),
1796 &self.config,
1797 sizeof_addr,
1798 )
1799 }
1800
1801 /// Where `scaled` sits in [`records`](Self::records): `Ok` at its record,
1802 /// `Err` at the position one would be inserted at.
1803 fn position(&self, scaled: &[u64]) -> Result<usize, usize> {
1804 self.records
1805 .binary_search_by(|r| r.scaled.as_slice().cmp(scaled))
1806 }
1807}
1808
1809/// Runtime metadata for a B-tree v2 indexed chunked dataset.
1810pub struct Bt2DatasetInfo {
1811 /// Chunk dimension sizes.
1812 pub chunk_dims: Vec<u64>,
1813 /// File offset of the BT2 header.
1814 pub bt2_header_addr: u64,
1815 /// Pool of node-size blocks (the index's
1816 /// [`node_size`](Bt2ChunkIndex::node_size) bytes each) holding the tree's
1817 /// nodes, in the order [`Bt2Tree::encode`] emits them.
1818 ///
1819 /// The single owner of the tree's node addresses: a flush re-serializes the
1820 /// whole tree over these blocks and allocates only the shortfall, so no
1821 /// flush can orphan a block it replaced. Every node is the same size, so a
1822 /// block stays usable however the tree reshapes.
1823 ///
1824 /// The pool holds exactly one block per node after every flush, in both
1825 /// directions: a taller tree allocates the shortfall, a smaller one frees
1826 /// the surplus. Nothing here depends on the record count only ever rising,
1827 /// so a record-removal path can be added to [`Bt2ChunkIndex`] without the
1828 /// blocks it drops going unreachable.
1829 pub node_addrs: Vec<u64>,
1830 /// In-memory chunk index.
1831 pub index: Bt2ChunkIndex,
1832 /// Number of chunks written so far.
1833 pub chunks_written: u64,
1834}
1835
1836/// Metadata for a group being written.
1837pub struct GroupInfo {
1838 /// Full path of this group (e.g. "/detector" or "/detector/raw").
1839 pub name: String,
1840 /// Index of the parent group in the groups vec, or None for root-level groups.
1841 pub parent: Option<usize>,
1842 /// Indices of child datasets (into `datasets` vec).
1843 pub child_datasets: Vec<usize>,
1844 /// Indices of child groups (into `groups` vec).
1845 pub child_groups: Vec<usize>,
1846 /// File offset of this group's object header (set during finalize).
1847 pub obj_header_addr: u64,
1848 /// File offset of the on-disk header a reopen found for this group, so
1849 /// finalize can free the block it supersedes.
1850 pub obj_header_written_addr: Option<u64>,
1851 /// Encoded size of that on-disk header (first block).
1852 /// Every block the object's on-disk header occupies, chunk 0 first, or
1853 /// empty when it has none yet. A rewrite keeps chunk 0's block — its
1854 /// address is what every reference to the object holds — and frees the
1855 /// rest, so a continuation block left behind is space no free-space
1856 /// manager records.
1857 pub obj_header_blocks: crate::io::object_header_io::HeaderBlocks,
1858 /// Soft-deleted: excluded from finalize output.
1859 pub deleted: bool,
1860 /// Attributes attached to this group (e.g. NeXus `NX_class`).
1861 pub attributes: Vec<AttributeEntry>,
1862 /// When the link naming this group was created, on the writer's single
1863 /// monotonic sequence. Groups, datasets and hard links share it, so a
1864 /// parent can order its links the way they were actually made.
1865 pub creation_seq: u64,
1866 /// How this group records creation order for its links and, separately,
1867 /// for its attributes. Creation-order tracking is a property of the
1868 /// object's creation property list in libhdf5, so it is captured here
1869 /// when the group is created rather than read from the writer at
1870 /// finalize: a later change of policy must not rewrite an object already
1871 /// made.
1872 pub track_order: TrackOrder,
1873 /// The times this group tracks, on the same terms as
1874 /// [`DatasetInfo::times`]. A version-1 group header records none of them:
1875 /// nothing calls `H5O_touch_oh` with `force` for a group, so the message a
1876 /// version-1 dataset gets is never created for one.
1877 pub times: Option<ObjectTimes>,
1878}
1879
1880/// One object's creation-order policy, with the two subsystems libhdf5 keeps
1881/// apart kept apart here too.
1882///
1883/// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` are separate
1884/// calls reading back out of separate places on disk — the Link Info message
1885/// and the object header's own flag bits — and a file may set either alone.
1886/// Carrying them as one flag made a reopen give a one-of-two file both or
1887/// neither.
1888#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
1889pub struct TrackOrder {
1890 /// Creation order of the links this group holds. Meaningless for a
1891 /// dataset, which is why `DatasetInfo` keeps only the attribute half.
1892 pub links: CreationOrder,
1893 /// Creation order of the attributes attached to this object.
1894 pub attrs: CreationOrder,
1895}
1896
1897impl TrackOrder {
1898 /// The policy the crate's single `track_order` knob selects: both
1899 /// subsystems tracked *and* indexed, or neither — the pair h5py's
1900 /// `File(track_order=True)` writes.
1901 pub fn uniform(track: bool) -> Self {
1902 let order = if track {
1903 CreationOrder::Indexed
1904 } else {
1905 CreationOrder::Untracked
1906 };
1907 Self {
1908 links: order,
1909 attrs: order,
1910 }
1911 }
1912}
1913
1914/// The object a [`HardLink`] resolves to.
1915#[derive(Clone, Copy)]
1916pub enum HardLinkTarget {
1917 /// Index into the writer's `datasets` vec.
1918 Dataset(usize),
1919 /// Index into the writer's `groups` vec.
1920 Group(usize),
1921}
1922
1923/// A user-created hard link: an additional name, in some group, for an
1924/// object that already exists under its own name.
1925///
1926/// The HDF5 file format makes every group entry a `name -> object header
1927/// address` mapping, so a hard link is just a second such entry pointing at
1928/// an already-written object. No data is copied.
1929#[derive(Clone)]
1930pub struct HardLink {
1931 /// Parent group index (`None` = the root group).
1932 pub parent: Option<usize>,
1933 /// Leaf name of the link within the parent group.
1934 pub name: String,
1935 /// Object this link resolves to.
1936 pub target: HardLinkTarget,
1937 /// When this link was created; see [`GroupInfo::creation_seq`].
1938 pub creation_seq: u64,
1939}
1940
1941/// A user-created symbolic link: a name in a group whose value is a path
1942/// rather than an object header address.
1943///
1944/// A soft link holds a path within this file; an external link holds a file
1945/// name and a path within that file. Neither names an object this writer
1946/// owns, so — unlike [`HardLink`] — nothing about it is resolved: the link is
1947/// stored as written and answered at traversal time, exactly as `H5Lcreate_soft`
1948/// and `H5Lcreate_external` store theirs.
1949#[derive(Clone)]
1950pub struct SymbolicLink {
1951 /// Parent group index (`None` = the root group).
1952 pub parent: Option<usize>,
1953 /// Leaf name of the link within the parent group.
1954 pub name: String,
1955 /// The path (and, for an external link, the file) this link names.
1956 pub target: LinkTarget,
1957 /// When this link was created; see [`GroupInfo::creation_seq`].
1958 pub creation_seq: u64,
1959}
1960
1961/// A committed (named) datatype: an object header holding one datatype
1962/// message and nothing else, reached by a link like any other object.
1963///
1964/// `H5Tcommit2` makes the type an object in its own right so several datasets
1965/// can declare they share it; each of those datasets then stores a pointer to
1966/// this object header in place of its own datatype message. The object's
1967/// reference count is therefore the links naming it *plus* the datasets
1968/// sharing it — `H5O__shared_link_adj` counts a share as a link — and an
1969/// object no link and no dataset reaches is not written at all.
1970#[derive(Clone)]
1971pub struct CommittedDatatype {
1972 /// Full path with no leading `/`, the form dataset names take.
1973 pub name: String,
1974 /// Parent group index (`None` = the root group).
1975 pub parent: Option<usize>,
1976 /// The committed type.
1977 pub datatype: DatatypeMessage,
1978 /// When the link naming it was created; see [`GroupInfo::creation_seq`].
1979 pub creation_seq: u64,
1980 /// The times it tracks, on the same terms as [`DatasetInfo::times`]. A
1981 /// version-1 committed datatype header records none of them, for the same
1982 /// reason a version-1 group's does not.
1983 pub times: Option<ObjectTimes>,
1984 /// File offset of its object header (set during finalize).
1985 pub obj_header_addr: u64,
1986}
1987
1988/// Where the object header a dataset's shared datatype pointer must name
1989/// comes from.
1990///
1991/// A dataset built on a committed type stores no datatype message: it stores
1992/// the address of the type's object header. Only the address matters at
1993/// encode time, but it is knowable at two different moments — a type this
1994/// session commits has no address until finalize lays the file out, while one
1995/// a reopen found is already at an address this session will not move. Naming
1996/// both here keeps [`build_dataset_header`](Hdf5Writer::build_dataset_header)
1997/// the one place that turns a share into a pointer, whichever way the share
1998/// arrived.
1999#[derive(Clone, Copy, Debug, PartialEq, Eq)]
2000pub enum CommittedTypeRef {
2001 /// A type committed in this session, by its index in
2002 /// [`committed_datatypes`](Hdf5Writer::committed_datatypes); its address
2003 /// is read from that registry once finalize has stamped one.
2004 Session(usize),
2005 /// A committed datatype a reopen kept by its bytes, at the object header
2006 /// address it already occupies.
2007 Preserved(u64),
2008}
2009
2010/// A link a reopened file already held that this writer cannot express.
2011///
2012/// Soft, external and user-defined links have no creation, retarget or delete
2013/// operation here — only hard links do — so a header rewrite that emits what
2014/// the registry models would erase them. Their encoded `Link` message rides
2015/// along instead and is written back byte for byte, which preserves every
2016/// field (name character set, creation order, the link value) without this
2017/// writer having to model any of them.
2018///
2019/// A *hard* link is preserved the same way when the object it names is one
2020/// the reopen could not model: writing the link back unchanged leaves that
2021/// object's header exactly where it is, which is the only way the rewrite can
2022/// keep what it cannot rebuild.
2023#[derive(Clone)]
2024pub struct PreservedLink {
2025 /// Parent group index (`None` = the root group).
2026 pub parent: Option<usize>,
2027 /// Leaf name of the link within the parent group.
2028 pub name: String,
2029 /// The link's class, decoded once at collection so listings can report
2030 /// it. Never the source of what gets written — `encoded` is.
2031 pub class: crate::io::reader::LinkClass,
2032 /// The encoded `Link` message body, exactly as read from the file.
2033 pub encoded: Vec<u8>,
2034 /// Why the object this link names could not be modelled, for the callers
2035 /// that ask for it by name. `None` when the link's own class — not its
2036 /// target — is what this writer cannot express.
2037 pub reason: Option<String>,
2038 /// What the object this link names is, when the walk could tell. A
2039 /// listing asks this; `reason` is prose for the caller that asks why.
2040 pub kind: PreservedKind,
2041}
2042
2043/// Every link a reopen walk met, split by what the writer can do with it.
2044/// A header rewrite emits both halves, so a link in neither half is a link
2045/// the close would destroy.
2046#[derive(Default)]
2047struct CollectedLinks {
2048 /// Hard links whose target the reopen modelled, with the plan that says
2049 /// how to rebuild it.
2050 hard: Vec<(HardEntry, CollectedObject)>,
2051 /// Links written back unchanged: the class this writer cannot express,
2052 /// and the hard links whose object it cannot model.
2053 preserved: Vec<PreservedEntry>,
2054}
2055
2056/// One hard link the reopen walk met: what it names, and the exact message
2057/// that names it.
2058#[derive(Clone)]
2059struct HardEntry {
2060 /// Full link path, in the no-leading-`/` form the registry uses.
2061 path: String,
2062 /// Object header address the link names.
2063 address: u64,
2064 /// The encoded `Link` message body, exactly as read from the file.
2065 encoded: Vec<u8>,
2066}
2067
2068/// A link the rewrite writes back exactly as it read it.
2069struct PreservedEntry {
2070 path: String,
2071 class: crate::io::reader::LinkClass,
2072 encoded: Vec<u8>,
2073 /// Why the object it names could not be modelled; `None` when the link's
2074 /// own class is what this writer cannot express.
2075 reason: Option<String>,
2076 /// What the object is, when the walk could tell.
2077 kind: PreservedKind,
2078}
2079
2080/// What a reopen can do with one object it reached.
2081///
2082/// A header rewrite emits a modelled object out of the registry, so the
2083/// registry may hold an object only when *every* message the model consumes
2084/// decoded. A partial read is not a smaller object, it is a different one:
2085/// before this rule a dataset whose datatype message did not decode was
2086/// registered as a group, and the close rewrote its header as one.
2087enum ObjectPlan {
2088 /// A dataset the rewrite can rebuild.
2089 Dataset(Box<DatasetParts>),
2090 /// A group the rewrite can rebuild, and the links it holds.
2091 Group(GroupParts),
2092 /// An object this writer cannot model, and why. Its header is never
2093 /// rewritten and never freed; the link naming it is written back byte for
2094 /// byte, so the object stays exactly as the file already had it — what
2095 /// libhdf5 does with the parts of a file it does not understand.
2096 ///
2097 /// `kind` is what the walk could still tell about the object it is
2098 /// keeping. Not modelling an object is not the same as not knowing what
2099 /// it is, and answering the second question with the first is what made
2100 /// `named_datatype_names` deny, in write mode, a datatype the same file
2101 /// reports in read mode.
2102 Preserve { why: String, kind: PreservedKind },
2103}
2104
2105/// What a preserved object is, as far as the reopen walk could tell.
2106///
2107/// Deliberately not a copy of the reader's `ObjectKind`: that one carries the
2108/// decoded object, and a preserved object is precisely the one whose contents
2109/// the writer does not decode. This says only what a listing needs.
2110#[derive(Clone, Copy, PartialEq, Eq, Debug)]
2111pub enum PreservedKind {
2112 /// The walk did not classify it — or the link's own class, not its
2113 /// target, is what could not be expressed.
2114 Unclassified,
2115 /// A committed (named) datatype, by
2116 /// [`header_is_committed_datatype`](crate::io::reader::header_is_committed_datatype).
2117 NamedDatatype,
2118}
2119
2120impl ObjectPlan {
2121 /// An object kept by its bytes, of a kind the walk did not classify.
2122 ///
2123 /// Every reason that is a *failure* to read reaches this: a message that
2124 /// did not decode says nothing about what the object was.
2125 fn preserve(why: impl Into<String>) -> Self {
2126 ObjectPlan::Preserve {
2127 why: why.into(),
2128 kind: PreservedKind::Unclassified,
2129 }
2130 }
2131}
2132
2133/// The messages a dataset's rewrite is built from, all decoded.
2134struct DatasetParts {
2135 /// Every block the header chain occupies, chunk 0 first. All of them are
2136 /// superseded: the rewrite re-encodes the whole chain into one fresh
2137 /// chunk, so a continuation left unfreed is space nothing claims.
2138 header_blocks: crate::io::object_header_io::HeaderBlocks,
2139 datatype: DatatypeMessage,
2140 /// The committed datatype object header `datatype` was read *through*,
2141 /// when the header stores a pointer instead of a message of its own.
2142 ///
2143 /// The literal type is in `datatype` either way, because the read resolves
2144 /// the pointer before anything decodes it; this is what a rewrite needs to
2145 /// put the pointer back rather than inline a copy of the named type and
2146 /// leave `H5Tcommitted` false.
2147 committed_type: Option<u64>,
2148 dataspace: crate::format::messages::dataspace::DataspaceMessage,
2149 /// The object format the reopen found this dataset's messages written in,
2150 /// read from the dataspace message's own version byte.
2151 ///
2152 /// A version-2 superblock does not settle it: `H5F__super_init` raises the
2153 /// superblock for a shared-message table or non-default file-space
2154 /// properties without touching `H5F_LOW_BOUND` (H5Fsuper.c:1135, :1144), so
2155 /// a file created at the earliest bound with either can hold version-1
2156 /// messages under a version-2 superblock — which is what
2157 /// `tests/fixtures/sohm_*.h5` are.
2158 read_format: ObjectFormat,
2159 layout: crate::format::messages::data_layout::DataLayoutMessage,
2160 filter_pipeline: Option<FilterPipeline>,
2161 fill_value: Option<Vec<u8>>,
2162 /// The fill-value message's write-time byte, preserved across a
2163 /// rewrite the same way `fill_value` is — an appended-to dataset must
2164 /// keep the policy libhdf5 (or this writer) declared for it, not fall
2165 /// back to the `H5D_CRT_FILL_TIME_DEF` a fresh dataset gets.
2166 fill_write_time: u8,
2167 attributes: Vec<AttributeEntry>,
2168 /// The creation-order policy the on-disk header declares; a rewrite that
2169 /// read it from the writer instead would stamp this session's policy onto
2170 /// an object libhdf5 created under another.
2171 track_order: TrackOrder,
2172 /// The times the on-disk header records, for the same reason: whether an
2173 /// object tracks them is settled when it is created, not when it is
2174 /// rewritten. Recovered by [`ObjectHeader::recorded_times`].
2175 times: Option<ObjectTimes>,
2176 /// The dense storage the rewrite supersedes and must free.
2177 dense: DenseCarry,
2178 /// The External File List the header carries, with each slot's name
2179 /// already read back out of the local heap the message points at. `None`
2180 /// for a dataset whose raw data is in this file.
2181 ///
2182 /// Carried rather than re-derived because the rewrite has to re-emit the
2183 /// message: a contiguous layout with an undefined address and no EFL
2184 /// beside it is a dataset with no data at all, so dropping this on a
2185 /// header rewrite would silently unlink every external byte.
2186 external: Option<ExternalStorage>,
2187}
2188
2189/// The same for a group, plus the links it holds — decoded once, with the
2190/// bytes they came from, so the walk and the rewrite agree on its contents.
2191struct GroupParts {
2192 header_blocks: crate::io::object_header_io::HeaderBlocks,
2193 attributes: Vec<AttributeEntry>,
2194 links: Vec<(crate::format::messages::link::LinkMessage, Vec<u8>)>,
2195 track_order: TrackOrder,
2196 times: Option<ObjectTimes>,
2197 dense: DenseCarry,
2198 /// The symbol-table storage a classic group's header names — the blocks
2199 /// the rewrite supersedes. `None` for a link-message group, which has
2200 /// none. Its links are already in `links`: the walk turns each symbol
2201 /// table entry into the link message it stands for, so nothing downstream
2202 /// has to know which of the two forms the group was in.
2203 stab: Option<StabExtents>,
2204}
2205
2206/// The dense storage one reopened object's header names, which the rewrite of
2207/// that header stops naming and therefore has to free. Both halves are read
2208/// back before this is built — a heap that could not be read makes the object
2209/// [`ObjectPlan::Preserve`], so nothing here describes storage whose contents
2210/// were lost.
2211#[derive(Default)]
2212struct DenseCarry {
2213 attrs: Option<AttributeInfoMessage>,
2214 links: Option<LinkInfoMessage>,
2215}
2216
2217/// A modelled object, as the walk hands it to the registry rebuild. A group's
2218/// links are not here: the walk followed them, and each child is an entry of
2219/// its own.
2220enum CollectedObject {
2221 Dataset(Box<DatasetParts>),
2222 Group {
2223 header_blocks: crate::io::object_header_io::HeaderBlocks,
2224 attributes: Vec<AttributeEntry>,
2225 track_order: TrackOrder,
2226 times: Option<ObjectTimes>,
2227 dense: DenseCarry,
2228 stab: Option<StabExtents>,
2229 },
2230}
2231
2232/// The reopen's discovery pass: one walk that classifies every object it
2233/// reaches and descends into the groups among them.
2234///
2235/// Every object the close will touch is decided here and nowhere else, so
2236/// "modelled or preserved" is a property of the walk rather than of whatever
2237/// each later stage happened to be able to decode.
2238struct ReopenWalk<'a> {
2239 handle: &'a mut FileHandle,
2240 meta: &'a crate::io::FileMeta,
2241 out: CollectedLinks,
2242 /// Object headers already descended into, so hard-link cycles end.
2243 visited: std::collections::HashSet<u64>,
2244}
2245
2246impl<'a> ReopenWalk<'a> {
2247 fn new(handle: &'a mut FileHandle, meta: &'a crate::io::FileMeta) -> Self {
2248 Self {
2249 handle,
2250 meta,
2251 out: CollectedLinks::default(),
2252 visited: std::collections::HashSet::new(),
2253 }
2254 }
2255
2256 /// Everything the walk found.
2257 fn finish(self) -> CollectedLinks {
2258 self.out
2259 }
2260
2261 /// Decide what the reopen can do with the object at `addr`.
2262 ///
2263 /// The single gate: every object the rewrite touches is classified here,
2264 /// and an object is modelled only when each message the model consumes
2265 /// decoded. See [`ObjectPlan`] for why anything else must keep its bytes.
2266 fn plan(&mut self, addr: u64) -> IoResult<ObjectPlan> {
2267 let (handle, meta) = (&mut *self.handle, self.meta);
2268 let ctx = &meta.ctx;
2269 use crate::format::messages::data_layout::DataLayoutMessage;
2270 use crate::format::messages::dataspace::DataspaceMessage;
2271 use crate::format::messages::link::{CharacterSet, LinkMessage};
2272 use crate::format::messages::link_info::LinkInfoMessage;
2273 use crate::format::messages::shared::MSG_FLAG_SHARED;
2274 use crate::format::messages::{
2275 MSG_ATTRIBUTE, MSG_DATASPACE, MSG_DATATYPE, MSG_DATA_LAYOUT, MSG_EXTERNAL_FILE_LIST,
2276 MSG_FILL_VALUE, MSG_FILTER_PIPELINE, MSG_LINK, MSG_LINK_INFO, MSG_SYMBOL_TABLE,
2277 };
2278
2279 // The whole chain, messages and blocks alike: a filter pipeline or an
2280 // attribute that spilled into a continuation is one the rewrite would
2281 // otherwise drop, and a continuation block it does not know about is
2282 // one the rewrite would orphan.
2283 let (header, header_blocks) =
2284 match crate::io::object_header_io::read_object_header_with_blocks(handle, meta, addr) {
2285 Ok(h) => h,
2286 Err(e) => {
2287 return Ok(ObjectPlan::preserve(format!(
2288 "its object header chain does not read: {e}"
2289 )))
2290 }
2291 };
2292
2293 // The policy, the times and the storage the header declares, read once
2294 // from the whole chain: all three are properties of the object, not of
2295 // any one message the loop below happens to reach.
2296 let track_order = recover_track_order(&header, ctx);
2297 let times = header.recorded_times();
2298 let (dense_attrs, dense_links) = superseded_dense(&header, ctx);
2299
2300 // Attributes come from the reader's collector rather than from the
2301 // loop below, so compact, dense and shared attributes all reach the
2302 // rewrite by the one path that knows how to read each of them. An
2303 // object whose set did not read whole is preserved: a short set here
2304 // would be a rewrite deleting the attributes it could not read.
2305 let attributes = match take_reopened_attributes(
2306 crate::io::reader::collect_object_attributes(handle, ctx, &header),
2307 &format!("the object at {addr:#x}"),
2308 ) {
2309 Ok(a) => a,
2310 Err(e) => {
2311 return Ok(ObjectPlan::preserve(format!(
2312 "its attributes do not read back whole: {e}"
2313 )))
2314 }
2315 };
2316
2317 let mut datatype = None;
2318 let mut dataspace = None;
2319 let mut layout = None;
2320 let mut filter_pipeline = None;
2321 let mut fill_value = None;
2322 // No fill-value message at all is the library default, the same
2323 // convention the reader-side decode (`Hdf5Reader::dataset_info`)
2324 // uses for `fill_defined`.
2325 let mut fill_write_time: u8 = FILL_TIME_IFSET;
2326 let mut external = None;
2327 let mut links = Vec::new();
2328 let mut stab = None;
2329 // A datatype, dataspace or layout message says the object is not a
2330 // group, whether or not the three a dataset needs are all there.
2331 let mut dataset_shaped = false;
2332
2333 for msg in &header.messages {
2334 let consumed = matches!(
2335 msg.msg_type,
2336 MSG_DATATYPE
2337 | MSG_DATASPACE
2338 | MSG_DATA_LAYOUT
2339 | MSG_FILTER_PIPELINE
2340 | MSG_FILL_VALUE
2341 | MSG_EXTERNAL_FILE_LIST
2342 | MSG_ATTRIBUTE
2343 | MSG_LINK
2344 | MSG_LINK_INFO
2345 | MSG_SYMBOL_TABLE
2346 );
2347 // A shared message holds a reference to where its body lives, not
2348 // the body. Decoding those bytes as one does not fail loudly — the
2349 // reference's version byte reads as a version and a class of its
2350 // own — so the guard is the only thing between a shared datatype
2351 // and a rewrite that invents a type for it.
2352 if consumed && msg.flags & MSG_FLAG_SHARED != 0 {
2353 return Ok(ObjectPlan::preserve(format!(
2354 "its message of type {:#04x} is a shared-message reference, which this \
2355 writer does not resolve",
2356 msg.msg_type
2357 )));
2358 }
2359 macro_rules! consume {
2360 ($decode:expr, $what:literal) => {
2361 match $decode {
2362 Ok(v) => v,
2363 Err(e) => {
2364 return Ok(ObjectPlan::preserve(format!(
2365 "its {} message does not decode: {e}",
2366 $what
2367 )))
2368 }
2369 }
2370 };
2371 }
2372 match msg.msg_type {
2373 // The pre-1.6 modification time, a formatted date string
2374 // (`H5O_MTIME`, type 0x0E). `recorded_times` reads only the
2375 // modern form, and a rewrite emits only that, so an object
2376 // carrying this one would come back out with the time it
2377 // recorded gone. Keeping its bytes is the same answer an
2378 // undecodable message already gets.
2379 crate::format::messages::MSG_MOD_TIME_OLD => {
2380 return Ok(ObjectPlan::preserve(
2381 "it carries a pre-1.6 modification time message, which this writer \
2382 reads but does not write",
2383 ))
2384 }
2385 MSG_DATATYPE => {
2386 dataset_shaped = true;
2387 let (dt, _) = consume!(DatatypeMessage::decode(&msg.data, ctx), "datatype");
2388 datatype = Some(dt);
2389 }
2390 MSG_DATASPACE => {
2391 dataset_shaped = true;
2392 let version = msg.data.first().copied().unwrap_or(1);
2393 let (ds, _) = consume!(DataspaceMessage::decode(&msg.data, ctx), "dataspace");
2394 dataspace = Some((ds, version));
2395 }
2396 MSG_DATA_LAYOUT => {
2397 dataset_shaped = true;
2398 let (dl, _) =
2399 consume!(DataLayoutMessage::decode(&msg.data, ctx), "data layout");
2400 layout = Some(dl);
2401 }
2402 MSG_FILTER_PIPELINE => {
2403 let (p, _) = consume!(FilterPipeline::decode(&msg.data), "filter pipeline");
2404 if !p.filters.is_empty() {
2405 filter_pipeline = Some(p);
2406 }
2407 }
2408 MSG_FILL_VALUE => {
2409 let (fv, _) = consume!(FillValueMessage::decode(&msg.data), "fill value");
2410 if fv.fill_defined == 2 {
2411 fill_value = fv.fill_value;
2412 }
2413 fill_write_time = fv.fill_write_time;
2414 }
2415 MSG_EXTERNAL_FILE_LIST => {
2416 dataset_shaped = true;
2417 let (efl, _) = consume!(
2418 ExternalFileListMessage::decode(&msg.data, ctx),
2419 "external file list"
2420 );
2421 // The names live in a local heap of their own, so the
2422 // rewrite cannot re-emit the message from its bytes alone
2423 // — it has to be able to point at the same strings. A heap
2424 // that does not read back leaves the object preserved,
2425 // which is what keeps its data reachable.
2426 let resolved = match crate::io::reader::Hdf5Reader::resolve_external_file_slots(
2427 handle, ctx, &efl,
2428 ) {
2429 Ok(r) => r,
2430 Err(e) => {
2431 return Ok(ObjectPlan::preserve(format!(
2432 "its external file list names do not read back: {e}"
2433 )))
2434 }
2435 };
2436 external = Some(ExternalStorage {
2437 heap_addr: efl.heap_addr,
2438 // `H5Fopen` opens no dataset, so nothing has read a
2439 // dapl for this one yet; the first handle it hands
2440 // out settles the prefix.
2441 prefix: EfilePrefix::default(),
2442 files: efl
2443 .slots
2444 .iter()
2445 .zip(resolved)
2446 .map(|(slot, seg)| ExternalFile {
2447 name: seg.name,
2448 name_offset: slot.name_offset,
2449 offset: slot.offset,
2450 size: slot.size,
2451 })
2452 .collect(),
2453 });
2454 }
2455 MSG_LINK => {
2456 let (l, _) = consume!(LinkMessage::decode(&msg.data, ctx), "link");
2457 links.push((l, msg.data.clone()));
2458 }
2459 MSG_LINK_INFO => {
2460 let (li, _) = consume!(LinkInfoMessage::decode(&msg.data, ctx), "link info");
2461 // Once a group holds enough links libhdf5 moves them into
2462 // the fractal heap this message names and writes no `Link`
2463 // messages at all. Reading them back is what makes the
2464 // rewrite emit the group with its children; a rewrite from
2465 // the header messages alone emitted it empty, orphaning
2466 // every object below it.
2467 if li.fractal_heap_address != UNDEF_ADDR {
2468 let dense = match crate::io::reader::Hdf5Reader::read_dense_links(
2469 handle,
2470 ctx,
2471 li.fractal_heap_address,
2472 ) {
2473 Ok(l) => l,
2474 Err(e) => {
2475 return Ok(ObjectPlan::preserve(format!(
2476 "its dense link storage does not read: {e}"
2477 )))
2478 }
2479 };
2480 // Re-encoded rather than carried as bytes: a heap
2481 // object is not a header message, so there are no
2482 // message bytes to carry. The encoding round-trips
2483 // through the same decoder that just read it.
2484 links.extend(dense.into_iter().map(|l| {
2485 let bytes = l.encode(ctx);
2486 (l, bytes)
2487 }));
2488 }
2489 }
2490 MSG_SYMBOL_TABLE => {
2491 // A classic group keeps no link message at all: its links
2492 // are symbol table entries in the B-tree this message
2493 // names. Turning each into the link message it stands for
2494 // is what lets the rest of the reopen — the walk, the
2495 // registry, the preserve path — work on one link model
2496 // whichever form the group is in.
2497 let Some(s) = Stab::decode(&msg.data, ctx) else {
2498 return Ok(ObjectPlan::preserve(
2499 "its symbol table message is shorter than the two addresses it \
2500 must carry",
2501 ));
2502 };
2503 let contents = match crate::io::symbol_table_io::read_stab(handle, meta, s) {
2504 Ok(c) => c,
2505 Err(e) => {
2506 return Ok(ObjectPlan::preserve(format!(
2507 "its symbol table does not read: {e}"
2508 )))
2509 }
2510 };
2511 stab = Some(contents.extents);
2512 links.extend(contents.links.into_iter().map(|l| {
2513 let msg = match l.target {
2514 StabTarget::Hard { addr, .. } => LinkMessage::hard(&l.name, addr),
2515 StabTarget::Soft { value } => LinkMessage::soft(&l.name, &value),
2516 };
2517 // An entry carries no character set field, so the link
2518 // it stands for has the file default whatever its name
2519 // looks like (`H5G__ent_to_link`, H5Gent.c:372).
2520 // Deriving one from the name would take a group
2521 // libhdf5 wrote with a high-byte ASCII name out of its
2522 // symbol table on the rewrite.
2523 let msg = msg.with_cset(CharacterSet::Ascii);
2524 let bytes = msg.encode(ctx);
2525 (msg, bytes)
2526 }));
2527 }
2528 _ => {}
2529 }
2530 }
2531
2532 // libhdf5 refuses a layout that disagrees with its sibling dataspace
2533 // and datatype as the dataset opens (`H5O__layout_decode` for the
2534 // chunk rank, `H5D__compact_init` for the compact size); modelled
2535 // anyway, the disagreement would be read at the wrong rank or past
2536 // the compact payload, so the dataset keeps its bytes, exactly as
2537 // unreadable as the file already had it.
2538 if let (Some((ds, _)), Some(dt), Some(dl)) = (&dataspace, &datatype, &layout) {
2539 if let Err(e) = dl.check_against_dataset(ds, dt, ctx) {
2540 return Ok(ObjectPlan::preserve(format!(
2541 "its layout doesn't fit its dataspace and datatype: {e}"
2542 )));
2543 }
2544 }
2545
2546 match (datatype, dataspace, layout) {
2547 // A layout `rebuild_dataset` has no arm for leaves the registry
2548 // entry with an undefined data address, and the close then rewrites
2549 // the header as a contiguous, unallocated dataset — every element
2550 // gone, silently. Only the layouts that rebuild are modelled; the
2551 // rest keep their bytes, as an undecodable message already does.
2552 // The virtual layout is this.
2553 (Some(_), Some(_), Some(layout)) if !layout_rebuilds(&layout) => {
2554 Ok(ObjectPlan::preserve(format!(
2555 "its data layout is {}, which this writer reads but does not build",
2556 layout.describe()
2557 )))
2558 }
2559 (Some(datatype), Some((dataspace, dataspace_version)), Some(layout)) => {
2560 // Asked of the raw chain, not of `header`: the read above has
2561 // already put the named type's message in place of the pointer.
2562 let committed_type = match crate::io::object_header_io::committed_datatype_address(
2563 handle, meta, addr,
2564 ) {
2565 Ok(c) => c,
2566 Err(e) => {
2567 return Ok(ObjectPlan::preserve(format!(
2568 "its shared datatype pointer does not decode: {e}"
2569 )))
2570 }
2571 };
2572 Ok(ObjectPlan::Dataset(Box::new(DatasetParts {
2573 header_blocks,
2574 datatype,
2575 committed_type,
2576 dataspace,
2577 read_format: if dataspace_version <= 1 {
2578 ObjectFormat::Legacy
2579 } else {
2580 ObjectFormat::Modern
2581 },
2582 layout,
2583 filter_pipeline,
2584 fill_value,
2585 fill_write_time,
2586 attributes,
2587 track_order,
2588 times,
2589 dense: DenseCarry {
2590 attrs: dense_attrs,
2591 links: dense_links,
2592 },
2593 external,
2594 })))
2595 }
2596 // A committed (named) datatype has a datatype message and neither
2597 // of the other two; so does a dataset whose header this crate only
2598 // half understands. Neither is a group, and modelling either as
2599 // one is what rewrote them into empty groups. They part company
2600 // here and nowhere else: the datatype is kept by its bytes like
2601 // the other, but a listing can still name it.
2602 _ if crate::io::reader::header_is_committed_datatype(&header) => {
2603 Ok(ObjectPlan::Preserve {
2604 why: "it is a committed (named) datatype, which this writer carries by \
2605 its bytes rather than re-encoding"
2606 .into(),
2607 kind: PreservedKind::NamedDatatype,
2608 })
2609 }
2610 _ if dataset_shaped => Ok(ObjectPlan::preserve(
2611 "it carries a datatype, dataspace or layout message but not the three a \
2612 dataset is built from; this writer models only groups and datasets",
2613 )),
2614 _ => Ok(ObjectPlan::Group(GroupParts {
2615 header_blocks,
2616 attributes,
2617 links,
2618 track_order,
2619 times,
2620 dense: DenseCarry {
2621 attrs: dense_attrs,
2622 links: dense_links,
2623 },
2624 stab,
2625 })),
2626 }
2627 }
2628
2629 /// Walk `links` (one group's, already decoded), classifying every object
2630 /// they name and descending into the groups among them.
2631 fn group(
2632 &mut self,
2633 links: &[(crate::format::messages::link::LinkMessage, Vec<u8>)],
2634 prefix: &str,
2635 depth: usize,
2636 ) -> IoResult<()> {
2637 // Bound nesting depth so a pathologically deep group chain cannot
2638 // overflow the stack (the `visited` set bounds total work but not
2639 // recursion depth).
2640 if depth > 256 {
2641 return Ok(());
2642 }
2643 use crate::format::messages::link::LinkTarget;
2644 for (link, encoded) in links {
2645 let full_name = if prefix.is_empty() {
2646 link.name.clone()
2647 } else {
2648 format!("{}/{}", prefix, link.name)
2649 };
2650
2651 // Only a hard link names an object this writer can rebuild. Every
2652 // other class is kept by its bytes, because a close that emitted
2653 // only what the registry models would drop it from the file.
2654 let LinkTarget::Hard { address } = &link.target else {
2655 self.out.preserved.push(PreservedEntry {
2656 path: full_name,
2657 class: crate::io::reader::LinkClass::from_target(&link.target),
2658 encoded: encoded.clone(),
2659 reason: None,
2660 kind: PreservedKind::Unclassified,
2661 });
2662 continue;
2663 };
2664 let entry = HardEntry {
2665 path: full_name.clone(),
2666 address: *address,
2667 encoded: encoded.clone(),
2668 };
2669
2670 match self.plan(*address)? {
2671 // Kept by its bytes, exactly as a link class this writer
2672 // cannot express is: writing the link back unchanged is what
2673 // leaves the object's header where the file already has it.
2674 ObjectPlan::Preserve { why, kind } => self.out.preserved.push(PreservedEntry {
2675 path: full_name,
2676 class: crate::io::reader::LinkClass::Hard,
2677 encoded: entry.encoded,
2678 reason: Some(why),
2679 kind,
2680 }),
2681 ObjectPlan::Dataset(parts) => {
2682 self.out.hard.push((entry, CollectedObject::Dataset(parts)));
2683 }
2684 ObjectPlan::Group(parts) => {
2685 self.out.hard.push((
2686 entry,
2687 CollectedObject::Group {
2688 header_blocks: parts.header_blocks,
2689 attributes: parts.attributes,
2690 track_order: parts.track_order,
2691 times: parts.times,
2692 dense: parts.dense,
2693 stab: parts.stab,
2694 },
2695 ));
2696 // Recurse only into a group's header we have not entered
2697 // before — breaks hard-link cycles.
2698 if self.visited.insert(*address) {
2699 self.group(&parts.links, &full_name, depth + 1)?;
2700 }
2701 }
2702 }
2703 }
2704 Ok(())
2705 }
2706}
2707
2708/// Rebuild one reopened dataset's in-memory registry entry, storage and
2709/// all, from the header messages the walk decoded.
2710///
2711/// Fails when the chunk index the file names does not read back. The
2712/// caller answers that by preserving the object rather than registering
2713/// a dataset whose index has forgotten where its chunks are: the close
2714/// rewrites what the registry holds, so an index rebuilt from the part of
2715/// it that decoded would strand every chunk it could not read.
2716fn rebuild_dataset(
2717 handle: &mut FileHandle,
2718 meta: &FileMeta,
2719 file_size: u64,
2720 name: String,
2721 obj_addr: u64,
2722 parts: DatasetParts,
2723) -> IoResult<DatasetInfo> {
2724 let ctx = &meta.ctx;
2725 let DatasetParts {
2726 header_blocks,
2727 datatype: dt,
2728 committed_type,
2729 dataspace: ds,
2730 read_format,
2731 layout: dl,
2732 filter_pipeline: fp,
2733 fill_value,
2734 fill_write_time,
2735 attributes: attrs,
2736 track_order,
2737 times,
2738 dense: _,
2739 external,
2740 } = parts;
2741
2742 let mut info = DatasetInfo {
2743 name,
2744 datatype: dt,
2745 // The named type's own object is preserved by its bytes, so the
2746 // address the walk read the pointer from is the address it will still
2747 // be at when this header is written back.
2748 committed_type: committed_type.map(CommittedTypeRef::Preserved),
2749 read_format: Some(read_format),
2750 external,
2751 virtual_storage: None,
2752 dataspace: ds,
2753 obj_header_addr: obj_addr,
2754 data_addr: UNDEF_ADDR,
2755 data_size: 0,
2756 compact: None,
2757 chunked: None,
2758 fixed_array: None,
2759 implicit: None,
2760 single_chunk: None,
2761 btree_v1: None,
2762 btree_v2: None,
2763 append: None,
2764 attributes: attrs,
2765 obj_header_written_addr: Some(obj_addr),
2766 obj_header_blocks: header_blocks,
2767 filter_pipeline: fp,
2768 deleted: false,
2769 extent_dirty: false,
2770 header_dirty: false,
2771 // Stamped by the caller once the whole link graph is registered: it
2772 // is the count of links reaching this object, which one dataset's
2773 // parts cannot see.
2774 nlink_written: 1,
2775 // Stamped by the caller, which knows the order the walk met each
2776 // object; the rebuild sees one dataset at a time.
2777 creation_seq: 0,
2778 track_attr_order: track_order.attrs,
2779 fill_value,
2780 fill_time: fill_write_time,
2781 // Preserve the on-disk layout version so finalize re-encodes
2782 // what it read: a v5 file reopened and appended to must not be
2783 // silently downgraded to v4 (the filtered indexes keep their
2784 // 8-byte size fields, which v4 readers would mis-derive).
2785 layout_version: match &dl {
2786 DataLayoutMessage::ChunkedV4 { version, .. } => *version,
2787 // The classic index has no version above its own: a version-3
2788 // message is the whole of `H5D__chunk_set_info`'s MAX below the
2789 // version-4 gate, and re-encoding it any higher would name an
2790 // index the message cannot carry.
2791 DataLayoutMessage::ChunkedV3 { .. } => LAYOUT_VERSION_DEFAULT,
2792 _ => 4,
2793 },
2794 times,
2795 };
2796
2797 // Reconstruct storage-specific metadata
2798 debug_assert!(
2799 layout_rebuilds(&dl),
2800 "ReopenWalk::plan must preserve a layout this has no arm for"
2801 );
2802 match &dl {
2803 DataLayoutMessage::Contiguous { address, size } => {
2804 info.data_addr = *address;
2805 info.data_size = *size;
2806 }
2807 // The image is the layout message, so the rebuild carries it out of
2808 // the header it came from: anything that makes this dataset's header
2809 // stale rewrites the layout message from `compact`, and a rebuild
2810 // that left it empty would rewrite the dataset as an unallocated
2811 // contiguous one — dropping every byte.
2812 DataLayoutMessage::Compact { data } => {
2813 info.compact = Some(data.clone());
2814 }
2815 // The classic chunk index, reconstructed into the same
2816 // `BtreeV1DatasetInfo` a chunked dataset *created* in this format
2817 // gets, so the one set of machinery — `build_tree`, the flush's block
2818 // pool, `write_chunk`, `extend_dataset`, the prune a delete runs —
2819 // drives a reopened dataset and a fresh one alike. `root_addr` is what
2820 // the layout message carries and stays undefined for a dataset whose
2821 // chunks were never written, exactly as libhdf5 leaves it.
2822 DataLayoutMessage::ChunkedV3 {
2823 chunk_dims,
2824 b_tree_address,
2825 } => {
2826 let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2827 let mut walk = BtreeV1Walk::new(handle, ctx, &meta.btree, &real_chunk_dims, file_size);
2828 walk.descend(*b_tree_address, 0)?;
2829 let BtreeV1Walk {
2830 records,
2831 node_addrs,
2832 ..
2833 } = walk;
2834 let max_dims = info
2835 .dataspace
2836 .max_dims
2837 .clone()
2838 .unwrap_or_else(|| info.dataspace.dims.clone());
2839 info.btree_v1 = Some(BtreeV1DatasetInfo {
2840 chunk_dims: real_chunk_dims,
2841 max_dims,
2842 // The file's own "K" ranks, not this session's defaults: they
2843 // set every node's width, so a tree bulk-loaded under the
2844 // wrong ones would re-serialize over blocks of the wrong size.
2845 config: meta.btree,
2846 records,
2847 node_addrs,
2848 root_addr: *b_tree_address,
2849 chunks_written: 0,
2850 });
2851 }
2852 DataLayoutMessage::ChunkedV4 {
2853 chunk_dims,
2854 index_address,
2855 index_type,
2856 earray_params,
2857 single_chunk_filter,
2858 ..
2859 } => {
2860 let real_chunk_dims: Vec<u64> = chunk_dims[..chunk_dims.len() - 1].to_vec();
2861
2862 if *index_type == crate::format::messages::data_layout::ChunkIndexType::ExtensibleArray
2863 {
2864 if let Some(params) = earray_params {
2865 let ep = EarrayParams {
2866 max_nelmts_bits: params.max_nelmts_bits,
2867 idx_blk_elmts: params.idx_blk_elmts,
2868 sup_blk_min_data_ptrs: params.sup_blk_min_data_ptrs,
2869 data_blk_min_elmts: params.data_blk_min_elmts,
2870 max_dblk_page_nelmts_bits: params.max_dblk_page_nelmts_bits,
2871 };
2872 let ndblk_addrs = compute_ndblk_addrs(ep.sup_blk_min_data_ptrs)?;
2873 let nsblk_addrs = compute_nsblk_addrs(
2874 ep.idx_blk_elmts,
2875 ep.data_blk_min_elmts,
2876 ep.sup_blk_min_data_ptrs,
2877 ep.max_nelmts_bits,
2878 )?;
2879
2880 // Read EA header
2881 let hdr_buf = handle.read_at_most(*index_address, 256)?;
2882 let ea_header = ExtensibleArrayHeader::decode(&hdr_buf, ctx)?;
2883
2884 let is_filtered = ea_header.class_id
2885 == crate::format::chunk_index::extensible_array::EA_CLS_FILT_CHUNK;
2886 let chunk_size_len = if is_filtered {
2887 ea_header.raw_elmt_size - ctx.sizeof_addr - 4
2888 } else {
2889 0
2890 };
2891
2892 // Read the EA index block. Filtered datasets
2893 // store a `FilteredIndexBlock`; unfiltered ones a
2894 // plain `ExtensibleArrayIndexBlock`. Both must be
2895 // reconstructed so a reopened dataset can append
2896 // (write_chunk consults whichever applies).
2897 let ea_iblk_addr = ea_header.idx_blk_addr;
2898 let (ea_iblk, filt_iblk) = if is_filtered {
2899 let placeholder = ExtensibleArrayIndexBlock::new(
2900 *index_address,
2901 ep.idx_blk_elmts,
2902 ndblk_addrs,
2903 nsblk_addrs,
2904 );
2905 let fib = if ea_iblk_addr != UNDEF_ADDR {
2906 let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2907 FilteredIndexBlock::decode(
2908 &iblk_buf,
2909 ctx,
2910 ep.idx_blk_elmts as usize,
2911 ndblk_addrs,
2912 nsblk_addrs,
2913 chunk_size_len,
2914 )?
2915 } else {
2916 FilteredIndexBlock::new(
2917 *index_address,
2918 ep.idx_blk_elmts,
2919 ndblk_addrs,
2920 nsblk_addrs,
2921 )
2922 };
2923 (placeholder, Some(fib))
2924 } else {
2925 let eib = if ea_iblk_addr != UNDEF_ADDR {
2926 let iblk_buf = handle.read_at_most(ea_iblk_addr, 65536)?;
2927 ExtensibleArrayIndexBlock::decode(
2928 &iblk_buf,
2929 ctx,
2930 ep.idx_blk_elmts as usize,
2931 ndblk_addrs,
2932 nsblk_addrs,
2933 )?
2934 } else {
2935 ExtensibleArrayIndexBlock::new(
2936 *index_address,
2937 ep.idx_blk_elmts,
2938 ndblk_addrs,
2939 nsblk_addrs,
2940 )
2941 };
2942 (eib, None)
2943 };
2944
2945 info.chunked = Some(ChunkedDatasetInfo {
2946 chunk_dims: real_chunk_dims,
2947 earray_params: ep,
2948 ea_header_addr: *index_address,
2949 ea_iblk_addr,
2950 ea_header,
2951 ea_iblk,
2952 chunks_written: 0,
2953 filt_iblk,
2954 chunk_size_len,
2955 });
2956 }
2957 } else if *index_type
2958 == crate::format::messages::data_layout::ChunkIndexType::FixedArray
2959 {
2960 // Read the FA header and data block back so a
2961 // reopened dataset is writable and deletable, not
2962 // re-link only — a placeholder made a delete free
2963 // just the header and leak every chunk plus the
2964 // index. Paged data blocks (any FA with more than
2965 // dblk_page_nelmts chunks, libhdf5 default 1024)
2966 // reconstruct through the same decode owner; only
2967 // pages the bitmap marks initialized are decoded.
2968 let hdr_buf = handle.read_at_most(*index_address, 256)?;
2969 let fa_header = FixedArrayHeader::decode(&hdr_buf, ctx)?;
2970 let is_filtered = fa_header.client_id == FA_CLIENT_FILT_CHUNK;
2971 let chunk_size_len = if is_filtered {
2972 (fa_header.element_size as usize)
2973 .checked_sub(ctx.sizeof_addr as usize + 4)
2974 .ok_or_else(|| {
2975 crate::io::IoError::InvalidState(
2976 "fixed array filtered element_size too small".into(),
2977 )
2978 })?
2979 } else {
2980 0
2981 };
2982 if fa_header.data_blk_addr != UNDEF_ADDR && chunk_size_len <= 8 {
2983 let dblk_size = fixed_array_dblk_disk_size(ctx, &fa_header) as usize;
2984 let dblk_buf = handle.read_at_most(fa_header.data_blk_addr, dblk_size)?;
2985 let fa_dblk =
2986 decode_fixed_array_dblk(ctx, &fa_header, &dblk_buf, chunk_size_len)?;
2987 info.fixed_array = Some(FixedArrayDatasetInfo {
2988 chunk_dims: real_chunk_dims,
2989 fa_header_addr: *index_address,
2990 fa_dblk_addr: fa_header.data_blk_addr,
2991 fa_header,
2992 fa_dblk,
2993 // Chunks written this session, matching the
2994 // EA reconstruction above.
2995 chunks_written: 0,
2996 });
2997 }
2998 } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::BTreeV2 {
2999 use crate::format::chunk_index::btree_v2::{
3000 Bt2Geometry, Bt2Header, BT2_TYPE_CHUNK_FILT, BT2_TYPE_CHUNK_UNFILT,
3001 };
3002
3003 // Walk the tree back into the in-memory index and
3004 // adopt its node blocks as the flush pool. The pool
3005 // re-serializes at the header's node_size, whatever
3006 // it is — libhdf5 sizes every node from
3007 // hdr->node_size (H5B2leaf.c, H5B2internal.c) — so
3008 // a foreign size reopens too. Only a record type
3009 // that is not a chunk record, or a node size below
3010 // the bulk loader's few-records-per-node floor
3011 // (the same bound creation enforces), stays
3012 // re-link only.
3013 let hdr_buf = handle.read_at_most(*index_address, 256)?;
3014 let bt2_hdr = Bt2Header::decode(&hdr_buf, ctx)?;
3015 let ndims = real_chunk_dims.len();
3016 let is_filt = match bt2_hdr.record_type {
3017 BT2_TYPE_CHUNK_UNFILT => Some(false),
3018 BT2_TYPE_CHUNK_FILT => Some(true),
3019 _ => None,
3020 };
3021 if let (Some(is_filt), true) = (
3022 is_filt,
3023 bt2_hdr.node_size as usize >= 10 + 3 * bt2_hdr.record_size as usize,
3024 ) {
3025 let mut index = if is_filt {
3026 let csl = (bt2_hdr.record_size as usize)
3027 .checked_sub(ctx.sizeof_addr as usize + 4 + ndims * 8)
3028 .filter(|&c| c <= 8)
3029 .ok_or_else(|| {
3030 crate::io::IoError::InvalidState(
3031 "v2 B-tree filtered record size does not fit \
3032 its rank and address width"
3033 .into(),
3034 )
3035 })?;
3036 Bt2ChunkIndex::new_filtered(ndims, csl as u8)
3037 } else {
3038 Bt2ChunkIndex::new_unfiltered(ndims)
3039 };
3040 // Re-serialize with the creator's parameters:
3041 // node blocks keep their size and the rewritten
3042 // header keeps its declared split/merge.
3043 index.node_size = bt2_hdr.node_size;
3044 index.split_percent = bt2_hdr.split_percent;
3045 index.merge_percent = bt2_hdr.merge_percent;
3046 let mut node_addrs = Vec::new();
3047 if bt2_hdr.root_node_addr != UNDEF_ADDR && bt2_hdr.total_num_records > 0 {
3048 let geo = Bt2Geometry::new(
3049 bt2_hdr.node_size,
3050 bt2_hdr.record_size,
3051 bt2_hdr.depth,
3052 ctx.sizeof_addr,
3053 );
3054 let mut walk =
3055 Bt2Walk::new(handle, ctx, bt2_hdr.record_size, bt2_hdr.node_size, &geo);
3056 walk.descend(
3057 bt2_hdr.root_node_addr,
3058 bt2_hdr.depth,
3059 bt2_hdr.num_records_in_root,
3060 )?;
3061 node_addrs = walk.node_addrs;
3062 let record_bytes = walk.records;
3063 let total = if bt2_hdr.record_size > 0 {
3064 record_bytes.len() / bt2_hdr.record_size as usize
3065 } else {
3066 0
3067 };
3068 if is_filt {
3069 for r in Bt2ChunkIndex::decode_filtered_records(
3070 &record_bytes,
3071 total,
3072 ndims,
3073 bt2_hdr.record_size,
3074 ctx,
3075 )? {
3076 index.insert_filtered(
3077 r.scaled_offsets,
3078 r.chunk_address,
3079 r.chunk_size,
3080 r.filter_mask,
3081 );
3082 }
3083 } else {
3084 for r in Bt2ChunkIndex::decode_unfiltered_records(
3085 &record_bytes,
3086 total,
3087 ndims,
3088 ctx,
3089 )? {
3090 index.insert(r.scaled_offsets, r.chunk_address);
3091 }
3092 }
3093 }
3094 info.btree_v2 = Some(Bt2DatasetInfo {
3095 chunk_dims: real_chunk_dims,
3096 bt2_header_addr: *index_address,
3097 node_addrs,
3098 index,
3099 chunks_written: 0,
3100 });
3101 }
3102 } else if *index_type == crate::format::messages::data_layout::ChunkIndexType::Implicit
3103 {
3104 // Nothing to read back: the index *is* the run of chunk space
3105 // at `index_address`, and its length is the chunk grid times
3106 // the chunk size. Reconstructing that length is what lets a
3107 // delete free the storage and a write address it — a rebuild
3108 // that left this empty would rewrite the dataset as an
3109 // unallocated contiguous one, dropping every byte.
3110 let mut nchunks: u64 = 1;
3111 for g in crate::io::chunk_grid::index_grid(
3112 &info.dataspace.dims,
3113 info.dataspace.max_dims.as_deref(),
3114 &real_chunk_dims,
3115 )? {
3116 nchunks = nchunks.checked_mul(g).ok_or_else(|| {
3117 crate::io::IoError::InvalidState("chunk count overflows u64".into())
3118 })?;
3119 }
3120 let data_size = nchunks
3121 .checked_mul(chunk_dims.iter().product::<u64>())
3122 .ok_or_else(|| {
3123 crate::io::IoError::InvalidState(
3124 "implicit chunk storage overflows u64".into(),
3125 )
3126 })?;
3127 info.implicit = Some(ImplicitDatasetInfo {
3128 chunk_dims: real_chunk_dims,
3129 data_addr: *index_address,
3130 data_size,
3131 });
3132 } else if *index_type
3133 == crate::format::messages::data_layout::ChunkIndexType::SingleChunk
3134 {
3135 // No index structure to read back either: the one chunk's
3136 // address, and its stored size and filter mask if the
3137 // layout's filtered flag is set, are the whole of the
3138 // layout message. `chunk_dims` already includes the
3139 // trailing element-size dimension, so its product is the
3140 // chunk's unfiltered byte length directly (see `data_size`
3141 // in the Implicit arm above).
3142 let data_size = chunk_dims.iter().product::<u64>();
3143 let (nbytes, filter_mask) = match single_chunk_filter {
3144 Some(scf) => (scf.nbytes, scf.filter_mask),
3145 None => (data_size, 0),
3146 };
3147 info.single_chunk = Some(SingleChunkDatasetInfo {
3148 chunk_dims: real_chunk_dims,
3149 data_addr: *index_address,
3150 data_size,
3151 nbytes,
3152 filter_mask,
3153 chunks_written: 0,
3154 // Whether this was created with early allocation isn't
3155 // recoverable here: `fill_value` above is only the
3156 // decoded fill bytes, not the fill-value message's
3157 // `alloc_time` byte the layout was chosen under. A
3158 // reopened dataset that later gets a header rewrite
3159 // therefore reports incremental allocation regardless
3160 // of how it was actually created — the same
3161 // imprecision a reopened `fixed_array`/`btree_v2`
3162 // dataset already has, for the same reason.
3163 early_alloc: false,
3164 });
3165 }
3166 }
3167 // Unreachable by `layout_rebuilds`, which is the gate
3168 // `ReopenWalk::plan` consults before it ever calls this.
3169 _ => {}
3170 }
3171
3172 Ok(info)
3173}
3174
3175/// Write `data` at *dataset-relative* byte offset `skip` into an external file
3176/// list, walking slots by cumulative declared size exactly like
3177/// `H5D__efl_write` (H5Defl.c).
3178///
3179/// Each slot's file is opened create-if-missing and never truncated, so a
3180/// write touches only the byte range that slot owns. A write past the *total*
3181/// declared size of the list is an error, matching upstream's "write past
3182/// logical end of file" check.
3183fn write_external_file_bytes(
3184 files: &[ExternalFile],
3185 extfile_prefix: Option<&Path>,
3186 mut skip: u64,
3187 data: &[u8],
3188) -> IoResult<()> {
3189 // `H5D__efl_write`'s slot walk: an `H5O_EFL_UNLIMITED` slot matches every
3190 // remaining offset (`skip >= u64::MAX` is never true), so the search stops
3191 // there and the write below takes the whole rest of the data.
3192 let mut slot_idx = 0usize;
3193 while slot_idx < files.len() && skip >= files[slot_idx].size {
3194 skip -= files[slot_idx].size;
3195 slot_idx += 1;
3196 }
3197
3198 let mut written = 0usize;
3199 while written < data.len() {
3200 let Some(slot) = files.get(slot_idx) else {
3201 return Err(crate::io::IoError::InvalidState(
3202 "write past the logical end of the external file list".into(),
3203 ));
3204 };
3205 let full_path = crate::io::reader::combine_prefixed_path(extfile_prefix, &slot.name);
3206 let ext_handle = FileHandle::open_or_create_readwrite_with_locking(
3207 &full_path,
3208 crate::io::locking::FileLocking::Disabled,
3209 )
3210 .map_err(|e| {
3211 crate::io::IoError::InvalidState(format!(
3212 "unable to open external raw data file {} for writing: {e}",
3213 full_path.display()
3214 ))
3215 })?;
3216 let this_write = (slot.size - skip).min((data.len() - written) as u64) as usize;
3217 let at = slot.offset.checked_add(skip).ok_or_else(|| {
3218 crate::io::IoError::InvalidState(format!(
3219 "external file '{}' slot offset {} overflows {skip} bytes into the slot",
3220 slot.name, slot.offset
3221 ))
3222 })?;
3223 ext_handle.write_at(at, &data[written..written + this_write])?;
3224 // This handle is dropped at the end of the iteration, and `Drop` can
3225 // only print a flush failure. Empty the accumulator here instead, so a
3226 // full disk on an external raw-data file reaches the caller.
3227 ext_handle.flush()?;
3228
3229 written += this_write;
3230 skip = 0;
3231 slot_idx += 1;
3232 }
3233 Ok(())
3234}
3235
3236/// The directory the HDF5 file at `path` sits in — libhdf5's `H5F_t::extpath`,
3237/// which `H5D__build_file_prefix` expands `${ORIGIN}` to.
3238///
3239/// Canonicalized, so the value survives the process changing directory and so
3240/// a writer and a reader of the same file agree on it. Called once per open,
3241/// never per I/O, for exactly that reason.
3242fn source_dir_of(path: &Path) -> IoResult<PathBuf> {
3243 let canonical = std::fs::canonicalize(path)?;
3244 Ok(canonical
3245 .parent()
3246 .map(Path::to_path_buf)
3247 .unwrap_or_default())
3248}
3249
3250/// Whether [`rebuild_dataset`] has an arm that reconstructs this layout.
3251///
3252/// The single list: `ReopenWalk::plan` preserves an object whose layout this
3253/// says no to, so a layout added to one side and not the other cannot happen.
3254/// Keeping two lists is what would rewrite a modelled dataset as unallocated
3255/// contiguous storage, or preserve one the writer can now build.
3256fn layout_rebuilds(layout: &DataLayoutMessage) -> bool {
3257 matches!(
3258 layout,
3259 DataLayoutMessage::Contiguous { .. }
3260 | DataLayoutMessage::Compact { .. }
3261 | DataLayoutMessage::ChunkedV3 { .. }
3262 | DataLayoutMessage::ChunkedV4 { .. }
3263 )
3264}
3265
3266/// Encode an Object Reference Count message (type 0x16) body: a version
3267/// byte (`H5O_REFCOUNT_VERSION` = 0) followed by the little-endian u32
3268/// count. Emitted on objects reached by more than one hard link.
3269fn encode_refcount(refcount: u32) -> Vec<u8> {
3270 let mut v = Vec::with_capacity(5);
3271 v.push(0u8);
3272 v.extend_from_slice(&refcount.to_le_bytes());
3273 v
3274}
3275
3276/// The symbol-table storage of every group that has one, and the single owner
3277/// of which groups those are.
3278///
3279/// A group stores its links in a symbol table because the file was *made* that
3280/// way — `H5F_LIBVER_EARLIEST` is the one bound `H5G__obj_create_real`
3281/// (H5Gobj.c:179) writes them at — or because it already had one when the file
3282/// was reopened. The second is not the first: `H5G_obj_insert` inserts into
3283/// whatever storage the group is in and converts only when a link will not fit
3284/// an entry (H5Gobj.c:512), so a symbol table survives a reopen at any bound.
3285/// A file with shared messages is where the two come apart, because its
3286/// superblock extension forces a version-2 superblock over symbol-table groups
3287/// (H5Fsuper.c:1135) — a group the session adds is made as the session's
3288/// bound says while the groups already there stay symbol tables.
3289struct SymbolTables {
3290 /// The scopes the reopen found a Symbol Table message on. Fixed for the
3291 /// session: a group already in that storage stays in it, whatever bound
3292 /// the objects added beside it are written at.
3293 found: HashSet<LinkScope>,
3294 /// The symbol-table storage each group's header already names, by the
3295 /// scope whose rewrite supersedes it.
3296 ///
3297 /// INVARIANT: every entry is freed exactly once, by
3298 /// [`Hdf5Writer::prepare_symbol_tables`], which removes it as it frees.
3299 superseded: Slot<HashMap<LinkScope, StabExtents>>,
3300 /// The storage that same pass laid out, read by the header builders.
3301 ///
3302 /// INVARIANT: an entry exists here only after every block of that group's
3303 /// heap and B-tree is on disk. `build_group_header` reads it and never
3304 /// builds — a header is sized and then written by two separate calls, so a
3305 /// build that allocated would allocate twice.
3306 written: Slot<HashMap<LinkScope, Stab>>,
3307}
3308
3309impl SymbolTables {
3310 /// What a file being created starts from: no group found in a symbol table
3311 /// because none was read, and nothing on disk to free.
3312 fn none_found() -> Self {
3313 Self {
3314 found: HashSet::new(),
3315 superseded: Slot::new(HashMap::new()),
3316 written: Slot::new(HashMap::new()),
3317 }
3318 }
3319}
3320
3321/// Everything a version-0/1 (symbol-table) file carries that a version-2/3 one
3322/// does not.
3323///
3324/// Its presence *is* the generation switch — [`Hdf5Writer::message_format`]
3325/// reads nothing else: libhdf5 at `H5F_LIBVER_EARLIEST` writes a version-0/1
3326/// superblock over version-1 object headers over symbol-table groups. Which
3327/// groups are symbol tables is the separate question [`SymbolTables`] answers,
3328/// because a reopen at a newer bound keeps the ones it finds.
3329///
3330/// Two things put one here, and only two: reopening a file that already is in
3331/// that format, and creating one at that bound
3332/// ([`LegacyFile::created`]). Neither is distinguished afterwards — a file is
3333/// classic or it is not, and every encoder asks only that.
3334struct LegacyFile {
3335 /// The superblock as it was read, or as [`LegacyFile::created`] built it.
3336 /// The close re-emits it with only the end of file and the root symbol
3337 /// table entry recomputed: the "K" ranks in particular are recorded
3338 /// nowhere else, and every node width in the file is derived from them.
3339 superblock: SuperblockV0V1,
3340}
3341
3342impl LegacyFile {
3343 /// The classic-format state a file created at `H5F_LIBVER_EARLIEST`
3344 /// starts from.
3345 ///
3346 /// A new file has no symbol table on disk to free and none laid out, so
3347 /// its [`SymbolTables`] starts empty and every group it makes takes that
3348 /// storage from the bound rather than from what was found.
3349 ///
3350 /// The superblock is the one `H5F__super_init` writes at that bound: the
3351 /// library-default "K" ranks (`H5F_CRT_SYM_LEAF_DEF`,
3352 /// `HDF5_BTREE_SNODE_IK_DEF`), no free-space info and no driver info. The
3353 /// root entry's object header address and cached symbol table are stamped
3354 /// in by [`Hdf5Writer::write_superblock`] once the root group has one;
3355 /// its name offset is the empty string at the front of every local heap.
3356 ///
3357 /// Version 0, not 1: a version-1 superblock exists only to carry a
3358 /// non-default chunked-storage "K" value (H5Fsuper.c:1150), and this
3359 /// writer has no property to set one.
3360 fn created(ctx: FormatContext, base_address: u64) -> Self {
3361 let btree = BTreeV1Config::default();
3362 Self {
3363 superblock: SuperblockV0V1 {
3364 version: SUPERBLOCK_V0,
3365 sizeof_offsets: ctx.sizeof_addr,
3366 sizeof_lengths: ctx.sizeof_size,
3367 file_consistency_flags: 0,
3368 sym_leaf_k: btree.sym_leaf_k,
3369 btree_internal_k: btree.snode_internal_k,
3370 indexed_storage_k: None,
3371 base_address,
3372 superblock_extension_address: UNDEF_ADDR,
3373 end_of_file_address: 0,
3374 driver_info_address: UNDEF_ADDR,
3375 root_symbol_table_entry: SymbolTableEntry {
3376 name_offset: 0,
3377 obj_header_addr: UNDEF_ADDR,
3378 cache: SymbolTableCache::Nothing,
3379 },
3380 },
3381 }
3382 }
3383}
3384
3385/// The superblock extension a reopen found, and the single owner of the one
3386/// this file's close writes back.
3387///
3388/// The extension is external truth: it is where a file records the things its
3389/// superblock has no field for — non-default v1 B-tree "K" ranks, a driver's
3390/// settings, the file space strategy and its persisted free-space managers,
3391/// and the shared object header message table. `H5F__super_ext_write_msg`
3392/// modifies one message of it and leaves the rest alone, so a close that lays
3393/// a fresh extension out from what *this writer* models drops everything it
3394/// does not — and the K ranks are not decoration: a chunked dataset's version-1
3395/// B-tree nodes are sized from `chunk_internal_k`, so a reader that has lost
3396/// the message reads the tree at the default rank and fails outright.
3397///
3398/// INVARIANT: every message of the extension read is re-emitted by
3399/// [`Hdf5Writer::write_superblock_extension`], byte for byte, except the
3400/// shared-message table — the one message naming storage this session lays out
3401/// afresh, which [`SohmState`] recomputes. Nothing else here is interpreted,
3402/// so a message this crate does not model survives exactly as a modelled one
3403/// does.
3404struct CarriedExtension {
3405 /// Every block the extension header occupied — chunk 0 and each
3406 /// continuation it named — freed once the replacement is laid out. Empty
3407 /// for a file with no extension, and for one whose extension this session
3408 /// is the first to write. A rewrite re-encodes the whole chain into one
3409 /// chunk, so freeing only the first would leave the rest as space no
3410 /// free-space manager records and no object claims.
3411 superseded: crate::io::object_header_io::HeaderBlocks,
3412 /// Every message that header held — the shared-message table,
3413 /// continuations and null padding excepted. The first two are structure
3414 /// rather than content; the third is free space.
3415 carried: Vec<crate::io::object_header_io::ExtensionMessage>,
3416 /// Where [`Hdf5Writer::write_superblock_extension`] put the replacement,
3417 /// and the only value the superblock's extension address is read from.
3418 /// `None` until that pass runs, and for a file that needs no extension.
3419 addr: Slot<Option<u64>>,
3420}
3421
3422/// What a reopen learns from a file's free-space managers, split by who owns
3423/// it: the sections go to the allocator and the rest stays with the writer.
3424struct ReopenedFreeSpace {
3425 /// `None` for a file this writer records no free space for.
3426 state: Option<Box<FileSpaceState>>,
3427 /// Every section the managers held, each tagged with the manager it came
3428 /// out of and merged only within it, address-ordered. Empty whenever
3429 /// `state` is `None`.
3430 sections: Vec<FreeBlock>,
3431}
3432
3433/// The file-space info message this session is responsible for, and the
3434/// manager blocks it supersedes.
3435///
3436/// A file whose message says `persist` records the space its own edits
3437/// released in one free-space manager per allocation type: a header block
3438/// (`FSHD`) naming a sections block (`FSSE`) that lists every free region.
3439/// Nothing else in the file says those regions are free, so a session that
3440/// rewrites the file without reading them either leaks the space it frees or
3441/// hands out space a manager still claims.
3442///
3443/// Present for a file this writer *created* with non-default file-space
3444/// properties as well, where there is nothing to read and the message is this
3445/// session's to write. `None` — the field, not this struct — is the third
3446/// case: a reopened file whose message this session must not touch, which the
3447/// carried extension re-emits byte for byte.
3448///
3449/// INVARIANT: the sections read are handed to [`FileAllocator`] and tracked
3450/// there alone, so there is one account of the file's free space and not two.
3451/// What stays here is only what the allocator has no place for: the message to
3452/// write, and the managers' own blocks, which are not free space until the
3453/// close that replaces them frees them.
3454struct FileSpaceState {
3455 /// The message, as read or as the creation options declared it. It is the
3456 /// only place the manager addresses are recorded, so the close that moves
3457 /// them rewrites this message.
3458 info: FileSpaceInfoMessage,
3459 /// The manager blocks themselves — one header, and one sections block per
3460 /// manager that had any sections. Freed by the close that lays their
3461 /// replacements out, the rule every other superseded structure follows.
3462 /// Empty for a created file, which supersedes nothing.
3463 superseded: Vec<(u64, u64)>,
3464}
3465
3466impl FileSpaceState {
3467 /// Whether this file keeps free-space managers on disk. Both strategies
3468 /// that have managers do — paged aggregation has the same managers plus a
3469 /// large one — while the two aggregator-only strategies and
3470 /// `persist: false` still carry the message with nothing to write into it.
3471 fn records_free_space(&self) -> bool {
3472 self.info.persist
3473 && matches!(
3474 self.info.strategy,
3475 FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
3476 )
3477 }
3478}
3479
3480/// One free-space manager that has been given its own two blocks, and the
3481/// sections it will write into them.
3482///
3483/// Produced by
3484/// [`settle_free_space_managers`](Hdf5Writer::settle_free_space_managers).
3485/// Both blocks are ordinary allocations out of the same [`FileAllocator`] the
3486/// rest of the file uses, because upstream's are too:
3487/// `H5FS_vfd_alloc_hdr_and_section_info_if_needed` calls `H5MF_alloc`
3488/// (H5FSsection.c:2352, 2406).
3489struct PlacedManager {
3490 /// Which of the file's managers this is; its message slot names it in the
3491 /// file-space info message.
3492 manager: FreeSpaceManager,
3493 /// Header block address.
3494 hdr_addr: u64,
3495 /// Sections block address.
3496 sect_addr: u64,
3497 /// Bytes the sections block occupies. What the header records as both
3498 /// `sect_size` and `alloc_sect_size`, so an image shorter than the block
3499 /// is padded rather than reported short.
3500 sect_size: u64,
3501 /// The sections this manager records, in serialization order. Filled on
3502 /// the settling round, once no allocation can change them.
3503 sections: Vec<FreeSection>,
3504}
3505
3506/// The manager header for `sections`, before its own blocks have addresses.
3507///
3508/// Every width the section encoding uses comes from here, and the only one
3509/// that varies with the content is `serial_sections` — it decides how many
3510/// bytes a per-size run count takes — so sizing a layout and encoding it must
3511/// go through this one function or the two disagree.
3512fn manager_header(sections: &[FreeSection]) -> FreeSpaceHeader {
3513 FreeSpaceHeader {
3514 client: free_space::CLIENT_FILE,
3515 total_space: sections.iter().map(|s| s.len).sum(),
3516 total_sections: sections.len() as u64,
3517 // Every class the file client registers is serializable; only a
3518 // fractal heap's manager has ghost sections.
3519 serial_sections: sections.len() as u64,
3520 ghost_sections: 0,
3521 nclasses: free_space::FILE_SECT_CLASSES,
3522 shrink_percent: free_space::SHRINK_PERCENT,
3523 expand_percent: free_space::EXPAND_PERCENT,
3524 max_sect_addr: free_space::SEC2_MAX_SECT_ADDR,
3525 max_sect_size: free_space::SEC2_MAXADDR,
3526 sect_addr: UNDEF_ADDR,
3527 sect_size: 0,
3528 alloc_sect_size: 0,
3529 }
3530}
3531
3532impl Default for CarriedExtension {
3533 /// What a file with no extension carries: nothing to free, nothing to
3534 /// re-emit, and no address until a shared-message table gives it one.
3535 fn default() -> Self {
3536 Self {
3537 superseded: Vec::new(),
3538 carried: Vec::new(),
3539 addr: Slot::new(None),
3540 }
3541 }
3542}
3543
3544/// Where a file's superblock version comes from — the two cases libhdf5 keeps
3545/// strictly apart, and this writer's single source for the version it writes
3546/// back.
3547///
3548/// INVARIANT: reopening a file never changes its superblock version.
3549///
3550/// libhdf5 splits the same way. `H5F__super_init` is the only place a version
3551/// is *decided* — content first, then `MAX(super_vers,
3552/// HDF5_superblock_ver_bounds[low_bound])` (H5Fsuper.c:1128-1154).
3553/// `H5F__super_read` never recomputes one. Nor does it bound anything by it:
3554/// the structures a session appends are written at the bound the caller
3555/// named, or the writer's default, whatever version the superblock has.
3556/// libhdf5 1.14 raised a reopened file's low bound to the row its superblock
3557/// version belongs to; libhdf5 2.0 dropped that (HDFGroup/hdf5#4939), and the
3558/// one raise left is SWMR write access, to `H5F_LIBVER_V110`
3559/// (H5Fsuper.c:453), which [`reject_swmr`](Hdf5Writer::reject_swmr) asks of
3560/// the caller instead. One direction only: the bound decides a created
3561/// file's version, the version never decides the bound.
3562///
3563/// Two variants rather than one number with a rule attached, because the
3564/// number means different things on the two paths — a floor to raise on the
3565/// create path, a fixed value on the reopen path — and a single field would
3566/// have every reader re-derive which.
3567#[derive(Debug, Clone, Copy)]
3568enum SuperblockVersion {
3569 /// A file this writer created. The version its creation options start
3570 /// from, which [`superblock_version_for`](Hdf5Writer::superblock_version_for)
3571 /// raises to what the content and the named bound need.
3572 Chosen(u8),
3573 /// A file this writer reopened: the version already in the file, written
3574 /// back unchanged. `Existing(0..=1)` and `Hdf5Writer::legacy` say the same
3575 /// thing from two directions and cannot disagree: `open_append_with_locking`
3576 /// builds the `LegacyFile` from exactly those versions.
3577 Existing(u8),
3578}
3579
3580/// A registry entry that has held some name.
3581///
3582/// Datasets, groups and committed datatypes keep stable indices — their
3583/// registries only grow, deletion being a flag — so the index can name the
3584/// exact entry. The link registries shrink as links are unlinked, and a
3585/// link's path is derived from its parent group's current name, so for those
3586/// the index records only that the kind once claimed the name and the (short)
3587/// list itself answers.
3588#[derive(Clone, Copy, PartialEq, Eq)]
3589enum NameHit {
3590 Dataset(usize),
3591 Group(usize),
3592 Datatype(usize),
3593 HardLink,
3594 SymbolicLink,
3595 PreservedLink,
3596}
3597
3598/// Which names the file model already holds, so creating an object does not
3599/// have to walk every registry to find out.
3600///
3601/// INVARIANT: while `map` is `Some`, every name a registry entry currently
3602/// holds has an entry in `map` covering that entry. The converse is not
3603/// required: a hit whose object was since deleted, or whose name has since
3604/// changed, stays in the map and is filtered out by
3605/// [`Hdf5Writer::name_holder`], which re-runs the very predicates the linear
3606/// scan used. The index may therefore answer "maybe", never "free" for a name
3607/// that is taken.
3608///
3609/// MUST NOT: no code may give a registry entry a name, or move the path a
3610/// link is emitted under, without either registering the new name through
3611/// [`Hdf5Writer::register_name`] or dropping the index through
3612/// [`Hdf5Writer::forget_name_index`]. State a constructor puts straight into
3613/// the registries needs neither — `map` starts `None`, and the first query
3614/// builds it from the registries as they then stand.
3615struct NameIndex {
3616 map: Option<HashMap<String, Vec<NameHit>>>,
3617 /// Bumped whenever the registries move under a build in flight, so that
3618 /// build's result is discarded instead of being installed stale.
3619 epoch: u64,
3620}
3621
3622impl NameIndex {
3623 fn new() -> Self {
3624 NameIndex {
3625 map: None,
3626 epoch: 0,
3627 }
3628 }
3629
3630 /// Record that `hit` holds `name`. With no map built there is nothing to
3631 /// record, but the registries have moved, so any build in flight is
3632 /// invalidated rather than trusted.
3633 fn insert(&mut self, name: &str, hit: NameHit) {
3634 match self.map.as_mut() {
3635 None => self.epoch += 1,
3636 Some(map) => {
3637 let hits = map.entry(name.to_string()).or_default();
3638 if !hits.contains(&hit) {
3639 hits.push(hit);
3640 }
3641 }
3642 }
3643 }
3644
3645 /// Throw the index away: the next query rebuilds it from the registries.
3646 fn forget(&mut self) {
3647 self.map = None;
3648 self.epoch += 1;
3649 }
3650}
3651
3652/// HDF5 file writer.
3653///
3654/// Usage:
3655/// 1. `Hdf5Writer::create(path)` to create a new file.
3656/// 2. `create_dataset(name, datatype, dims)` to define datasets.
3657/// 3. `write_dataset_raw(index, data)` to write raw data.
3658/// 4. `close()` to finalize the file (writes superblock, headers, etc.).
3659pub struct Hdf5Writer {
3660 handle: FileHandle,
3661 allocator: FileAllocator,
3662 ctx: FormatContext,
3663 /// Dataset registry. The outer [`Slot`] guards the spine (push on create,
3664 /// index/clone on access) and is held only briefly; each [`DatasetRef`]
3665 /// carries one dataset's metadata behind its own lock. A writer clones
3666 /// the `DatasetRef` out (releasing this lock) before doing the long
3667 /// per-dataset work, so a create never blocks an in-flight write.
3668 pub(crate) datasets: Slot<Vec<DatasetRef>>,
3669 /// Group registry, same shape as [`Self::datasets`].
3670 pub(crate) groups: Slot<Vec<GroupRef>>,
3671 /// User-created hard links (additional names for existing objects),
3672 /// resolved and emitted during finalize.
3673 pub(crate) hard_links: Slot<Vec<HardLink>>,
3674 /// User-created soft and external links. Held apart from
3675 /// [`Self::hard_links`] because they name a path rather than an object:
3676 /// nothing resolves them, and no object's reference count counts them.
3677 pub(crate) symbolic_links: Slot<Vec<SymbolicLink>>,
3678 /// Datatypes committed this session, each an object of its own; see
3679 /// [`CommittedDatatype`].
3680 pub(crate) committed_datatypes: Slot<Vec<CommittedDatatype>>,
3681 /// Links a reopened file held that this writer cannot express, carried
3682 /// through every header rewrite by their encoded bytes. Always empty for
3683 /// a freshly created file; see [`PreservedLink`].
3684 pub(crate) preserved_links: Slot<Vec<PreservedLink>>,
3685 /// Which names the registries above already hold; see [`NameIndex`].
3686 /// Boxed so this side table costs the writer one pointer: inline, its
3687 /// map shifted every field after it and cost the attribute path ~5%.
3688 name_index: Slot<Box<NameIndex>>,
3689 /// Attributes attached to the root group (file-level attributes).
3690 pub(crate) root_attributes: Slot<Vec<crate::format::messages::attribute::AttributeEntry>>,
3691 /// Serializes object creation so name-uniqueness check and registry insert
3692 /// happen atomically.
3693 ///
3694 /// INVARIANT: no two emitted links share a full-path name. Under
3695 /// `threadsafe`, create methods run on the shared read guard, so without
3696 /// this gate two threads could both pass the duplicate-name check (which
3697 /// snapshots a registry and drops its lock) and both push, writing an
3698 /// invalid HDF5 file with two same-named links. A create holds this lock
3699 /// across its check *and* its push; the streaming write path never takes
3700 /// it, so writes to existing datasets stay fully concurrent. It is the
3701 /// outermost lock a create acquires (create_lock → spine → slot), and no
3702 /// write path takes it, so it cannot deadlock with the registry locks.
3703 pub(crate) create_lock: Slot<()>,
3704 /// The low `H5Pset_libver_bounds` bound the *caller named*, or `None`
3705 /// when none was: the oldest libhdf5 the objects this writer creates must
3706 /// stay readable by. It is the one switch the version-bearing messages
3707 /// read — the datatype message version (`H5O_dtype_ver_bounds`), the data
3708 /// layout message version (`H5O_layout_ver_bounds`) and with it the chunk
3709 /// index, and the superblock floor (`HDF5_superblock_ver_bounds`) of a
3710 /// file this writer creates.
3711 ///
3712 /// `None` is not `Some(Earliest)`. No single libhdf5 bound describes this
3713 /// crate's default file: it takes the earliest row of the datatype and
3714 /// superblock tables (version-1 datatypes, a version-2 superblock raised
3715 /// to 3 only by what the content needs) over the v1.10 chunk indexes,
3716 /// which is the `H5F_LIBVER_V110` row of the layout table. Naming a bound
3717 /// asks for one whole libhdf5 generation instead, so the two cannot share
3718 /// a field.
3719 ///
3720 /// Nothing reads this directly:
3721 /// [`session_libver`](Hdf5Writer::session_libver) is the only reader, and
3722 /// it is where `None` becomes the default of the family asking, the same
3723 /// on a created file and a reopened one: the superblock a reopened file
3724 /// already has says nothing about the bound (see [`SuperblockVersion`]).
3725 libver: Option<LibverBound>,
3726 closed: bool,
3727 /// Set once `finalize_for_swmr` has published a readable file.
3728 ///
3729 /// A SWMR reader may hold a chunk index that still points at a block this
3730 /// writer has since replaced, so from that point on a relocated chunk's
3731 /// old block is kept rather than released for reuse — the same rule as
3732 /// libhdf5's `H5D__chunk_file_alloc`, which skips `H5MF_xfree` under
3733 /// `H5F_ACC_SWMR_WRITE`.
3734 swmr_active: bool,
3735 /// Collections with free space — libhdf5's `f->shared->cwfs` list. A
3736 /// vlen insert fills these partially-filled collection blocks before
3737 /// creating a new one, so many small writes share 4096-byte blocks
3738 /// instead of each taking their own. Entries hold `(addr, block size,
3739 /// free bytes)` hints; the block on disk stays the single truth for
3740 /// contents, and only the two functions that rewrite collection blocks
3741 /// ([`insert_vlen_objects`](Self::insert_vlen_objects) and
3742 /// [`release_vlen_references`](Self::release_vlen_references)) may
3743 /// update this list. In-memory only, like the allocator's free list:
3744 /// a reopened file's free space is rediscovered as releases touch its
3745 /// collections. Capped at [`H5HG_NCWFS`] entries.
3746 cwfs: Slot<Vec<CwfsEntry>>,
3747 /// Address of the root group object header (set after first finalize).
3748 root_group_addr: Option<u64>,
3749 /// Size of the encoded root group object header (for in-place rewrites).
3750 /// The on-disk root header block a reopen found, `(addr, len)`, so
3751 /// finalize can free the block its rewrite supersedes.
3752 superseded_root_header: crate::io::object_header_io::HeaderBlocks,
3753 /// Where this file's superblock version comes from. The single owner of
3754 /// the reopen invariant — see [`SuperblockVersion`] and
3755 /// [`superblock_version_for`](Self::superblock_version_for).
3756 superblock_version: SuperblockVersion,
3757 /// Objects whose attributes this finalize spilled to dense storage, and
3758 /// the `Attribute Info` message naming what was written for each.
3759 ///
3760 /// INVARIANT: an entry exists here only after every block of that
3761 /// object's heap and name index is on disk, and only
3762 /// [`prepare_dense_attributes`](Self::prepare_dense_attributes) may add
3763 /// one. `emit_attributes` reads it and never builds — a header is sized
3764 /// and then written by two separate `build_*_header` calls, so a build
3765 /// that allocated would allocate twice and leave the sized-for blocks
3766 /// stranded.
3767 dense_attributes: Slot<HashMap<AttrScope, AttributeInfoMessage>>,
3768 /// Groups whose links this finalize spilled to dense storage, and the
3769 /// `Link Info` message naming what was written for each.
3770 ///
3771 /// INVARIANT: an entry exists here only after every block of that group's
3772 /// heap and name index is on disk, and only
3773 /// [`prepare_dense_links`](Self::prepare_dense_links) may add one.
3774 dense_links: Slot<HashMap<LinkScope, LinkInfoMessage>>,
3775 /// The dense storage the reopened object headers already name — the heaps
3776 /// and indices this session's rewrites and deletes supersede.
3777 ///
3778 /// `None` for a file this session created: every block such a file will
3779 /// hold was allocated here, so there is nothing on disk to supersede and
3780 /// nothing to allocate for the bookkeeping either.
3781 ///
3782 /// INVARIANT: every entry is freed exactly once, by
3783 /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs)
3784 /// or [`release_superseded_dense_links`](Self::release_superseded_dense_links),
3785 /// which remove it as they free. Nothing else may remove one: an entry
3786 /// that leaves without reaching the allocator is a leaked heap, and one
3787 /// that reaches it twice hands the same blocks to two objects.
3788 superseded_dense: Slot<Option<Box<SupersededDense>>>,
3789 /// The creation-order policy in force: whether an object created from
3790 /// now on records creation order for its links and its attributes. The
3791 /// h5py `track_order` analogue; see
3792 /// [`set_track_order`](Self::set_track_order). Each object captures this
3793 /// at creation, so changing it never rewrites an object already made.
3794 track_order: TrackOrder,
3795 /// Whether an object created from now on records the times its header can
3796 /// hold — `H5Pset_obj_track_times`, whose default is on
3797 /// (`H5O_CRT_OHDR_FLAGS_DEF` is `H5O_HDR_STORE_TIMES`, H5Opkg.h:74).
3798 /// Captured by each object at creation for the same reason
3799 /// [`track_order`](Self::track_order) is: it belongs to the creation
3800 /// property list, so a later change must not rewrite an object already
3801 /// made.
3802 track_times: bool,
3803 /// The root group's own captured policy. The root is created with the
3804 /// file, so its value comes from
3805 /// [`create_with_options`](Self::create_with_options) — or, on reopen,
3806 /// from the header already on disk.
3807 root_track_order: TrackOrder,
3808 /// The root group's stored times, on the same terms as
3809 /// [`GroupInfo::times`]: whatever a reopened file's root header had, and
3810 /// `None` for a file this writer created.
3811 root_times: Option<ObjectTimes>,
3812 /// Hands out the creation sequence numbers that order a group's links.
3813 next_creation_seq: Slot<u64>,
3814 /// Object-reference elements waiting for their target's object header
3815 /// address, which only exists once finalize has placed every header.
3816 pending_object_references: Slot<Vec<PendingObjectReference>>,
3817 /// Heap-backed reference objects waiting for the same address — the
3818 /// pre-1.12 region form and every 1.12 form whose element is a blob id.
3819 pending_heap_references: Slot<Vec<PendingHeapReference>>,
3820 /// What each object-reference attribute's value *means*, so
3821 /// [`object_attributes`](Hdf5Writer::object_attributes) can say it in
3822 /// addresses every time an object header is built.
3823 attribute_references: Slot<Vec<AttributeReferenceValue>>,
3824 /// Set when this file is in the classic (version-0/1 superblock) format,
3825 /// whether it was reopened in it or created at `H5F_LIBVER_EARLIEST`.
3826 /// See [`LegacyFile`]; [`is_legacy`](Self::is_legacy) is the only reader
3827 /// of whether it is there.
3828 legacy: Option<Box<LegacyFile>>,
3829 /// Which groups keep their links in a symbol table, and the storage each
3830 /// of them has. Empty for a file whose groups all store links in messages;
3831 /// see [`SymbolTables`], which owns the question.
3832 symbol_tables: SymbolTables,
3833 /// The v1 B-tree "K" ranks every node width in this file is derived from,
3834 /// after the superblock extension has had its say. A property of the file
3835 /// rather than of its generation: a version-2 superblock records no ranks
3836 /// of its own but its extension may, and a rewrite that used the library
3837 /// defaults there would write nodes of the wrong width.
3838 /// [`btree_v1_config`](Hdf5Writer::btree_v1_config) is the only reader.
3839 btree: BTreeV1Config,
3840 /// The superblock extension this file carries, and where the replacement
3841 /// went; see [`CarriedExtension`].
3842 extension: Box<CarriedExtension>,
3843 /// The free-space managers a reopened `persist: true` file carries; see
3844 /// [`FileSpaceState`]. `None` for every other file — one with no
3845 /// file-space info message, one that does not persist, one under paged
3846 /// aggregation, and every file this session created — and those files get
3847 /// no free-space manager written either.
3848 free_space: Option<Box<FileSpaceState>>,
3849 /// The file's shared-message indexes, when it was created with any.
3850 /// `None` — the default — is a file with no shared-message table, where
3851 /// [`share_message`](Self::share_message) is the identity.
3852 sohm: Option<Box<SohmState>>,
3853 /// The directory holding this HDF5 file, resolved once when it was opened
3854 /// — libhdf5's `H5F_t::extpath`, and the same value the read side keeps.
3855 /// External raw-data file names are joined against it when
3856 /// `HDF5_EXTFILE_PREFIX` names `${ORIGIN}`, so a write and a later read of
3857 /// the same dataset must resolve a relative name identically; capturing it
3858 /// at open time rather than reading the process's current directory per
3859 /// write is what makes that hold.
3860 source_dir: PathBuf,
3861}
3862
3863/// A file's shared object header messages, from creation to the table on disk.
3864///
3865/// INVARIANT: a message body reaches the file either literally or as a pointer
3866/// to exactly one heap object, never both, and the reference count of that
3867/// object is the number of headers that hold the pointer.
3868/// [`share_message`](Hdf5Writer::share_message) is the only place a body is
3869/// offered to an index, and
3870/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) is the
3871/// only place the phase changes — so counting and substituting are two passes
3872/// over the same call site rather than two pieces of logic that must agree.
3873struct SohmState {
3874 /// The indexes the file was created with, in table order.
3875 indexes: Vec<SohmIndexSpec>,
3876 /// What `share_message` does to an eligible message right now.
3877 phase: Slot<SohmPhase>,
3878 /// Address of the master table this session laid out, once it has one.
3879 /// Also the once-only latch on the layout: a second finalize keeps the
3880 /// table the first one published, and
3881 /// [`Hdf5Writer::write_superblock_extension`] reads it to name that table
3882 /// in the extension.
3883 table_addr: Slot<Option<u64>>,
3884 /// The blocks the table a reopen found occupies — the master table and
3885 /// each index's heap and index structure — taken by the finalize that
3886 /// replaces them. Empty for a file this session created.
3887 ///
3888 /// The table is laid out whole from the whole message set, so a reopen
3889 /// replaces it rather than inserting into it, and every header holding a
3890 /// pointer into the old one is rewritten in the same finalize
3891 /// ([`Hdf5Writer::rebuilds_shared_messages`]).
3892 superseded: Slot<Vec<(u64, u64)>>,
3893}
3894
3895/// The passes `share_message` runs in, and the state between them.
3896enum SohmPhase {
3897 /// Outside a finalize: every message stays literal.
3898 Idle,
3899 /// Measuring headers, before the bodies they will hold are final. A
3900 /// shareable message answers at the width of a heap pointer over a heap
3901 /// object that does not exist yet, which is the width the one it ends up
3902 /// pointing at has: a `H5O_shared_t` in heap form is the same size
3903 /// whatever it names. Nothing this pass produces is written — it exists so
3904 /// [`allocate_object_headers`](Hdf5Writer::allocate_object_headers) can
3905 /// reserve a block for a header whose messages are shared before the
3906 /// content phase has decided which heap object each one shares.
3907 ///
3908 /// The set is [`FirstCopies`], and it is why this pass has state at all:
3909 /// a message left literal is *wider* than a pointer, so a header can only
3910 /// be measured by making the same first-copy decision the substituting
3911 /// pass will make.
3912 Predict(FirstCopies),
3913 /// Counting the bodies the file will share. Messages still go in
3914 /// literally, so nothing this pass builds is written.
3915 Collect(SohmCollector),
3916 /// Substituting. A body the collect pass never saw stays literal, which
3917 /// is a valid file: the record it would have shared simply keeps a
3918 /// reference count one higher than the pointers that reach it.
3919 Resolve {
3920 /// Heap ID per body, from the table this finalize laid out.
3921 ids: HashMap<(u8, Vec<u8>), [u8; SOHM_HEAP_ID_LEN]>,
3922 /// The first copies this pass has already handed out; see
3923 /// [`FirstCopies`].
3924 first: FirstCopies,
3925 },
3926}
3927
3928/// The bodies a pass has already left literal in the header that offered them
3929/// first (`H5SM_IN_OH`, H5SM.c:1400-1417).
3930///
3931/// INVARIANT: the three passes walk the same object headers in the same order
3932/// — [`allocate_object_headers`](Hdf5Writer::allocate_object_headers),
3933/// [`prepare_shared_messages`](Hdf5Writer::prepare_shared_messages) and
3934/// [`write_object_headers`](Hdf5Writer::write_object_headers) each build every
3935/// dataset in `datasets` order, then every group, then the root — so "the
3936/// header that offered this body first" is the same header in all three. Each
3937/// pass keeps its own set rather than sharing one, so a pass that does not run
3938/// cannot leave a stale decision behind for the next one. A divergence would
3939/// make a header wider than the block reserved for it, which
3940/// [`check_header_size`] refuses rather than writing.
3941type FirstCopies = std::collections::HashSet<(u8, Vec<u8>)>;
3942
3943/// The object header a message is being written into — `H5SM_try_share`'s
3944/// `open_oh` argument, which is what decides whether a first copy has a header
3945/// to stay literal in at all.
3946#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3947enum ShareOwner {
3948 /// `H5SM_try_share(f, NULL, ...)`: the message belongs to no object header
3949 /// of its own. An attribute's datatype and dataspace are offered this way
3950 /// (H5Aint.c:375-377) — they live inside the attribute's body, so there is
3951 /// no header message for a record to name and the body goes to the heap on
3952 /// first use however shareable its class is.
3953 Detached,
3954 /// `H5SM_try_share(f, oh, ...)`: the message is a message of the object
3955 /// header at this address (`H5O__msg_alloc`, H5Omessage.c:1735).
3956 Header(u64),
3957}
3958
3959impl SohmState {
3960 /// A file's indexes, plus the blocks of the table they were read out of
3961 /// when the file was reopened (empty when it was created this session).
3962 fn new(indexes: Vec<SohmIndexSpec>, superseded: Vec<(u64, u64)>) -> Self {
3963 Self {
3964 indexes,
3965 phase: Slot::new(SohmPhase::Idle),
3966 table_addr: Slot::new(None),
3967 superseded: Slot::new(superseded),
3968 }
3969 }
3970
3971 /// The index that would take a `msg_type` message of `body_len` bytes,
3972 /// as `H5SM_try_share` resolves one: the first index whose type mask
3973 /// covers the class, and then only if the message reaches that index's
3974 /// minimum. A message too small for its index is not offered to another —
3975 /// `H5SM__get_index` picks by type alone and the size check comes after.
3976 fn index_for(&self, msg_type: u8, body_len: usize) -> Option<usize> {
3977 let flag = type_flag(msg_type)?;
3978 let (at, spec) = self
3979 .indexes
3980 .iter()
3981 .enumerate()
3982 .find(|(_, spec)| spec.mesg_types & flag != 0)?;
3983 (body_len as u64 >= u64::from(spec.min_mesg_size)).then_some(at)
3984 }
3985
3986 /// Whether any index takes attribute messages, which is what makes the
3987 /// file record message creation indices — `H5SM_init` sets
3988 /// `store_msg_crt_idx` on exactly this condition (H5SM.c:220).
3989 fn shares_attributes(&self) -> bool {
3990 let Some(flag) = type_flag(MSG_ATTRIBUTE) else {
3991 return false;
3992 };
3993 self.indexes.iter().any(|spec| spec.mesg_types & flag != 0)
3994 }
3995}
3996
3997/// What decides whether two offers are the same shared message: the class,
3998/// the bytes, and the messages the bytes will end up pointing at.
3999type CollectedKey = (u8, Vec<u8>, Vec<NestedShare>);
4000
4001/// The shareable message bodies of one collect pass, in first-seen order.
4002struct SohmCollector {
4003 /// Per index, its bodies with the number of headers holding each.
4004 messages: Vec<Vec<SharedMessage>>,
4005 /// Where a body sits: `(index, position in that index's messages)`, keyed
4006 /// by everything that decides what will be stored — the class, the bytes,
4007 /// and the messages the bytes will end up pointing at.
4008 seen: HashMap<CollectedKey, (usize, usize)>,
4009}
4010
4011impl SohmCollector {
4012 fn new(nindexes: usize) -> Self {
4013 Self {
4014 messages: vec![Vec::new(); nindexes],
4015 seen: HashMap::new(),
4016 }
4017 }
4018
4019 /// Count one message against `index`, adding the body the first time it
4020 /// is seen, and say whether that body is new.
4021 ///
4022 /// `ohdr` is the header this offer would leave the body literal in when it
4023 /// is the first — `None` when the class cannot be shared in an object
4024 /// header or the offer names none. It is recorded only for a first copy:
4025 /// once a body is in the heap, later offers of it are pointers whatever
4026 /// header they come from.
4027 ///
4028 /// Two bodies are the same message only if their nesting agrees as well:
4029 /// the heap IDs a nesting body will hold are still zero here, so two
4030 /// attributes that differ only in their datatype are the same bytes at
4031 /// this point and different bytes on disk.
4032 fn record(
4033 &mut self,
4034 index: usize,
4035 msg_type: u8,
4036 body: &[u8],
4037 nested: &[NestedShare],
4038 ohdr: Option<u64>,
4039 ) -> bool {
4040 let key = (msg_type, body.to_vec(), nested.to_vec());
4041 match self.seen.get(&key) {
4042 Some(&(at, pos)) => {
4043 self.messages[at][pos].ref_count += 1;
4044 false
4045 }
4046 None => {
4047 let pos = self.messages[index].len();
4048 self.messages[index].push(SharedMessage {
4049 msg_type,
4050 body: body.to_vec(),
4051 nested: nested.to_vec(),
4052 ref_count: 1,
4053 ohdr_addr: ohdr,
4054 });
4055 self.seen.insert(key, (index, pos));
4056 true
4057 }
4058 }
4059 }
4060
4061 /// Give back the reference [`record`](Self::record) took for a body whose
4062 /// container turned out to be a copy of one already here.
4063 ///
4064 /// A body reached only through a shared container is referenced once per
4065 /// container *record*, not once per object that has one: the pointer to
4066 /// it lives in the container's heap object, which exists once however
4067 /// many headers name it. `H5O__attr_create` reaches the same count from
4068 /// the other side, by building each attribute's components shared and
4069 /// then calling `H5O__attr_delete` — which decrements exactly the
4070 /// datatype and dataspace (H5Oattr.c:568-585) — whenever the attribute it
4071 /// built was not the first copy (H5Oattribute.c:331-366).
4072 fn release(&mut self, msg_type: u8, body: &[u8]) {
4073 if let Some(&(at, pos)) = self.seen.get(&(msg_type, body.to_vec(), Vec::new())) {
4074 let count = &mut self.messages[at][pos].ref_count;
4075 *count = count.saturating_sub(1);
4076 }
4077 }
4078}
4079
4080/// The file-creation properties a brand-new file is made with.
4081///
4082/// libhdf5 splits these across the file creation and file access property
4083/// lists (`H5Pset_userblock`, `H5Pset_link_creation_order`,
4084/// `H5Pset_libver_bounds`, the locking property); what they have in common is
4085/// that they are read once, when the file is created, and cannot be changed
4086/// afterwards without rewriting it. Options that *can* change mid-session —
4087/// the bound for objects created later, the creation-order policy for later
4088/// objects — have their own setters.
4089#[derive(Debug, Clone, Copy, Default)]
4090pub struct FileCreateOptions {
4091 /// OS-level locking policy for the new file.
4092 pub locking: crate::io::locking::FileLocking,
4093 /// Creation-order policy for the root group, and the default for every
4094 /// object created afterwards; see [`Hdf5Writer::set_track_order`].
4095 pub track_order: bool,
4096 /// Time-tracking policy for the root group, and the default for every
4097 /// object created afterwards; see [`Hdf5Writer::set_track_times`].
4098 pub track_times: bool,
4099 /// The file's low library-version bound (`H5Pset_libver_bounds`'s `low`),
4100 /// or `None` when the caller named none.
4101 ///
4102 /// The distinction is not decoration. `Some(LibverBound::Earliest)` is a
4103 /// request for the format libhdf5 writes at `H5F_LIBVER_EARLIEST` — a
4104 /// version-0 superblock over symbol-table groups and version-1 object
4105 /// headers, which is what [`ObjectFormat::Legacy`] encodes. `None` keeps
4106 /// what this crate has always written for a file whose creator said
4107 /// nothing: the version-2 superblock and link-message groups of the v1.8
4108 /// format, with the earliest bound's message versions where they can
4109 /// express the content. That combination is this crate's own, not one
4110 /// libhdf5 writes, so it cannot be spelled as a bound.
4111 pub libver: Option<LibverBound>,
4112 /// Bytes reserved in front of the superblock for the application's own
4113 /// use (`H5Pset_userblock`). Zero, the default, places the superblock at
4114 /// offset 0; otherwise a power of two of at least
4115 /// [`MIN_USERBLOCK`] bytes, since a reader finds the
4116 /// superblock by doubling its search offset from there.
4117 pub userblock: u64,
4118 /// Shared object header message indexes; see [`SharedMessageConfig`].
4119 pub shared_messages: SharedMessageConfig,
4120 /// How the file manages its own space; see [`FileSpaceConfig`].
4121 pub file_space: FileSpaceConfig,
4122}
4123
4124/// The file-space handling properties a new file is created with — the three
4125/// arguments of `H5Pset_file_space_strategy` and the one of
4126/// `H5Pset_file_space_page_size`.
4127///
4128/// The four together are what `H5F__super_init` compares against the library
4129/// defaults to decide whether the file needs a file-space info message at all
4130/// (H5Fsuper.c:1092-1097), which is why the page size belongs here even though
4131/// only paged aggregation allocates by it: a file that names a page size and
4132/// nothing else still carries the message.
4133#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4134pub struct FileSpaceConfig {
4135 /// `H5F_fspace_strategy_t`.
4136 pub strategy: FileSpaceStrategy,
4137 /// Whether the free-space managers are written to the file on close.
4138 pub persist: bool,
4139 /// The smallest section a manager records; a block freed below it is
4140 /// space the file leaks rather than tracks.
4141 pub threshold: u64,
4142 /// `H5Pset_file_space_page_size`: the file-space page every allocation of
4143 /// a paged file is shaped by, and the value the message carries whatever
4144 /// the strategy.
4145 pub page_size: u64,
4146}
4147
4148impl Default for FileSpaceConfig {
4149 /// `H5F_FILE_SPACE_STRATEGY_DEF`, `H5F_FREE_SPACE_PERSIST_DEF`,
4150 /// `H5F_FREE_SPACE_THRESHOLD_DEF` and `H5F_FILE_SPACE_PAGE_SIZE_DEF`
4151 /// (H5Fprivate.h:326-336).
4152 fn default() -> Self {
4153 Self {
4154 strategy: FileSpaceStrategy::FsmAggr,
4155 persist: false,
4156 threshold: 1,
4157 page_size: DEFAULT_FILE_SPACE_PAGE_SIZE,
4158 }
4159 }
4160}
4161
4162impl FileSpaceConfig {
4163 /// The properties as `H5P__set_file_space_strategy` (H5Pfcpl.c:1176)
4164 /// stores them: `persist` and `threshold` are set only for the two
4165 /// strategies that have free-space managers to persist, and keep their
4166 /// defaults for the two that do not.
4167 pub fn new(strategy: FileSpaceStrategy, persist: bool, threshold: u64) -> Self {
4168 let uses_managers = matches!(
4169 strategy,
4170 FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
4171 );
4172 Self {
4173 strategy,
4174 persist: uses_managers && persist,
4175 threshold: if uses_managers {
4176 threshold
4177 } else {
4178 Self::default().threshold
4179 },
4180 ..Self::default()
4181 }
4182 }
4183
4184 /// `H5Pset_file_space_page_size`, the fourth file-space property and the
4185 /// one libhdf5 sets on its own call.
4186 ///
4187 /// Independent of the strategy, as upstream is: the value reaches the
4188 /// file-space info message whatever the strategy is, and only paged
4189 /// aggregation allocates by it. Out-of-range sizes are refused where the
4190 /// file is created ([`validate`](Self::validate)) rather than here, so a
4191 /// builder chain stays a builder chain.
4192 pub fn with_page_size(mut self, page_size: u64) -> Self {
4193 self.page_size = page_size;
4194 self
4195 }
4196
4197 /// Whether the file has to say any of this on disk. `H5F__super_init`
4198 /// writes the file-space info message only for a file that differs from
4199 /// the library defaults in one of the four properties (H5Fsuper.c:1092),
4200 /// and raises such a file's superblock to version 2 so it has an
4201 /// extension to write it into (H5Fsuper.c:1144).
4202 pub fn is_default(&self) -> bool {
4203 *self == Self::default()
4204 }
4205
4206 /// Refuse what this writer cannot make. `H5Pset_file_space_strategy`
4207 /// itself only refuses a strategy outside the enum (H5Pfcpl.c:1223), and
4208 /// `H5Pset_file_space_page_size` a page size outside `[512, 1 GiB]`
4209 /// (H5Pfcpl.c:1389-1393) — no power of two required, only the bounds.
4210 fn validate(&self) -> IoResult<()> {
4211 if !(PAGE_SIZE_MIN..=PAGE_SIZE_MAX).contains(&self.page_size) {
4212 return Err(crate::io::IoError::InvalidState(format!(
4213 "a file-space page size is between {PAGE_SIZE_MIN} bytes and \
4214 {PAGE_SIZE_MAX}, not {}",
4215 self.page_size
4216 )));
4217 }
4218 match self.strategy {
4219 FileSpaceStrategy::FsmAggr
4220 | FileSpaceStrategy::Aggr
4221 | FileSpaceStrategy::None
4222 | FileSpaceStrategy::Page => Ok(()),
4223 FileSpaceStrategy::Unknown(b) => Err(crate::io::IoError::InvalidState(format!(
4224 "invalid file-space strategy {b}"
4225 ))),
4226 }
4227 }
4228
4229 /// The message a created file carries, before anything is allocated:
4230 /// every manager address undefined and no end-of-allocation recorded,
4231 /// which is what `H5F__super_init` writes (H5Fsuper.c:1369-1382).
4232 fn message(&self) -> FileSpaceInfoMessage {
4233 FileSpaceInfoMessage {
4234 // `H5O_fsinfo_set_version` starts at version 1 and only ever
4235 // raises it, so a created file never carries the version-0 form
4236 // however low its version bounds are.
4237 version: 1,
4238 strategy: self.strategy,
4239 persist: self.persist,
4240 threshold: self.threshold,
4241 page_size: self.page_size,
4242 pgend_meta_thres: 0,
4243 eoa_pre_fsm_fsalloc: UNDEF_ADDR,
4244 fs_addr: vec![UNDEF_ADDR; FS_ADDR_COUNT_V1],
4245 }
4246 }
4247}
4248
4249/// The shared object header message indexes a new file is created with.
4250///
4251/// libhdf5 sets these with three calls on the file creation property list:
4252/// `H5Pset_shared_mesg_nindexes` fixes how many indexes there are,
4253/// `H5Pset_shared_mesg_index` gives each one the message types it covers and
4254/// the smallest message it will take, and `H5Pset_shared_mesg_phase_change`
4255/// sets the list/B-tree thresholds for all of them at once. The default —
4256/// no indexes — is a file with no shared-message table, which is what every
4257/// file this crate wrote before the option existed.
4258#[derive(Debug, Clone, Copy, PartialEq)]
4259pub struct SharedMessageConfig {
4260 /// Indexes in table order; only the first `count` are in use.
4261 indexes: [SohmIndexSpec; MAX_SOHM_INDEXES],
4262 /// How many indexes the caller asked for. Kept even when it is more than
4263 /// the array holds, so file creation can refuse the count the way
4264 /// `H5Pset_shared_mesg_nindexes` does rather than silently drop indexes.
4265 count: usize,
4266}
4267
4268impl Default for SharedMessageConfig {
4269 fn default() -> Self {
4270 Self {
4271 indexes: [SohmIndexSpec {
4272 mesg_types: 0,
4273 min_mesg_size: 0,
4274 list_max: DEFAULT_SOHM_LIST_MAX,
4275 btree_min: DEFAULT_SOHM_BTREE_MIN,
4276 }; MAX_SOHM_INDEXES],
4277 count: 0,
4278 }
4279 }
4280}
4281
4282impl SharedMessageConfig {
4283 /// One index per `(mesg_types, min_mesg_size)` pair — the arguments
4284 /// `H5Pset_shared_mesg_index` takes, where `mesg_types` is the bit mask
4285 /// [`type_flag`](crate::format::sohm::type_flag) builds — with the
4286 /// file-wide phase change `H5Pset_shared_mesg_phase_change` sets: above
4287 /// `list_max` an index is a v2 B-tree, below `btree_min` it is a list
4288 /// again, and `list_max == 0` makes it a B-tree from its first message.
4289 ///
4290 /// Nothing is validated here; [`Hdf5Writer::create_with_options`] refuses
4291 /// a configuration libhdf5 would refuse, so an invalid one is reported
4292 /// where the file is made rather than where the value is typed.
4293 pub fn new(indexes: &[(u16, u32)], list_max: u16, btree_min: u16) -> Self {
4294 let mut config = Self {
4295 count: indexes.len(),
4296 ..Self::default()
4297 };
4298 for (slot, &(mesg_types, min_mesg_size)) in config.indexes.iter_mut().zip(indexes) {
4299 *slot = SohmIndexSpec {
4300 mesg_types,
4301 min_mesg_size,
4302 list_max,
4303 btree_min,
4304 };
4305 }
4306 config
4307 }
4308
4309 /// The indexes in use, in table order.
4310 pub(crate) fn specs(&self) -> &[SohmIndexSpec] {
4311 &self.indexes[..self.count.min(MAX_SOHM_INDEXES)]
4312 }
4313
4314 /// Refuse a configuration `H5Pset_shared_mesg_nindexes` or
4315 /// `H5Pset_shared_mesg_phase_change` would refuse.
4316 fn validate(&self) -> IoResult<()> {
4317 if self.count > MAX_SOHM_INDEXES {
4318 return Err(crate::io::IoError::InvalidState(format!(
4319 "a file may declare at most {MAX_SOHM_INDEXES} shared-message \
4320 indexes, not {}",
4321 self.count
4322 )));
4323 }
4324 for spec in self.specs() {
4325 // The two thresholds must not overlap, or an index would convert
4326 // back and forth on every insert.
4327 if u32::from(spec.btree_min) > u32::from(spec.list_max) + 1 {
4328 return Err(crate::io::IoError::InvalidState(format!(
4329 "shared-message phase change needs btree_min ({}) at most one \
4330 past list_max ({}), or an index converts on every insert",
4331 spec.btree_min, spec.list_max
4332 )));
4333 }
4334 if spec.mesg_types == 0 {
4335 return Err(crate::io::IoError::InvalidState(
4336 "a shared-message index covering no message type would never \
4337 be used; give it a type mask or drop it"
4338 .into(),
4339 ));
4340 }
4341 }
4342 Ok(())
4343 }
4344}
4345
4346/// One object-reference element written before its value could be known.
4347///
4348/// An `H5R_OBJECT1` element is the target's object header address, and
4349/// addresses are assigned during finalize, so a write records the target by
4350/// path here and [`Hdf5Writer::write_object_reference_values`] puts the address
4351/// down once every header has one.
4352pub(crate) struct PendingObjectReference {
4353 /// Dataset holding the element.
4354 dataset: usize,
4355 /// Element index within that dataset.
4356 element: u64,
4357 /// Path of the object the element names; `/` is the root group.
4358 target: String,
4359}
4360
4361/// One heap-backed reference object written before its target's address could
4362/// be known.
4363///
4364/// The *element* of a `H5R_DATASET_REGION1`, and of every 1.12 reference whose
4365/// encoding does not fit inline, is final at write time — it is the global-heap
4366/// id of the object the write inserted. What waits is the `sizeof_addr` bytes
4367/// of that heap object holding the target's object header address, which
4368/// [`Hdf5Writer::write_heap_reference_values`] stamps in.
4369pub(crate) struct PendingHeapReference {
4370 /// Address of the global-heap collection holding the object.
4371 collection: u64,
4372 /// The object's index within that collection.
4373 index: u16,
4374 /// Where the target's token sits inside that object. The pre-1.12 region
4375 /// form leads with it (`H5R__encode_token_region_compat`); every 1.12 form
4376 /// puts the token's length byte first (`H5R__encode_obj_token`).
4377 token_offset: usize,
4378 /// What the reference names, and how strictly its path must resolve.
4379 target: PendingHeapTarget,
4380}
4381
4382/// What the path of a heap-backed reference must resolve to.
4383///
4384/// The two rules `H5R` applies: a region reference names a *dataset*, since
4385/// `H5Rcreate_region` takes one dataset's dataspace and every reader
4386/// dereferences it as one, while an attribute reference names the attribute's
4387/// owner, which `H5Rcreate_attr` lets be any object.
4388#[derive(Debug, Clone)]
4389pub(crate) enum PendingHeapTarget {
4390 Dataset(String),
4391 Object(String),
4392}
4393
4394/// The value of an attribute whose elements are object references, kept as
4395/// what it means rather than as what it encodes to.
4396///
4397/// An attribute's value is part of its object header message, so it cannot be
4398/// stamped after the fact the way a dataset element can — the header is one
4399/// block, written once. What is stored instead is the paths, and
4400/// [`Hdf5Writer::object_attributes`] turns them into addresses every time the
4401/// attribute set is built: the measuring pass reads the zeros of objects that
4402/// have no address yet, the content pass reads the addresses the file will
4403/// have, and the two agree in length because an address is a fixed-width
4404/// field. The entry in the object's attribute list carries an image with
4405/// zeros where the addresses go and is never itself written.
4406///
4407/// The address of `targets[i]` lands at byte `i * stride` of that image: the
4408/// whole element when the attribute is an array of references, the leading
4409/// member when each element is a compound that carries other fields beside
4410/// the reference (`REFERENCE_LIST`'s `dimension`), which the stored image
4411/// already holds.
4412pub(crate) struct AttributeReferenceValue {
4413 /// The object the attribute hangs on.
4414 scope: AttrScope,
4415 /// The attribute's name within that object.
4416 name: String,
4417 /// Paths of the objects the elements name, in element order; `/` is the
4418 /// root group.
4419 targets: Vec<String>,
4420 /// Bytes from one element's address to the next: the element size.
4421 stride: usize,
4422}
4423
4424/// The attribute naming the scales attached to each axis of a dataset.
4425pub(crate) const DIMENSION_LIST: &str = "DIMENSION_LIST";
4426/// The attribute naming every (dataset, axis) a dimension scale is attached to.
4427pub(crate) const REFERENCE_LIST: &str = "REFERENCE_LIST";
4428/// The `CLASS` a dimension scale carries.
4429const DIMENSION_SCALE_CLASS: &str = "DIMENSION_SCALE";
4430
4431/// A dataset's `CLASS` attribute as `H5DS` reads it.
4432enum ClassAttr {
4433 /// A fixed-length string, with what `H5DSis_scale` checks beside the text.
4434 Fixed {
4435 size: u32,
4436 null_terminated: bool,
4437 text: String,
4438 },
4439 /// A variable-length string.
4440 VarLen(String),
4441 /// Not a string at all.
4442 NotString,
4443}
4444
4445/// `bytes` read as a C string: everything before the first NUL.
4446fn c_string(bytes: &[u8]) -> String {
4447 let end = bytes.iter().position(|&b| b == 0).unwrap_or(bytes.len());
4448 String::from_utf8_lossy(&bytes[..end]).into_owned()
4449}
4450
4451/// Refuse an object header body that is not the length its block was reserved
4452/// at.
4453///
4454/// The one check standing behind
4455/// [`HeaderLayout`]'s premise that measuring a header before its content is
4456/// final gives the same length as encoding it after. `what` names the object
4457/// only when the check fails, so the caller pays for the lookup only then.
4458fn check_header_size(
4459 encoded: &[u8],
4460 reserved: usize,
4461 what: impl FnOnce() -> String,
4462) -> IoResult<()> {
4463 if encoded.len() == reserved {
4464 return Ok(());
4465 }
4466 Err(crate::io::IoError::InvalidState(format!(
4467 "the object header of {} encodes to {} bytes but was measured at {}; \
4468 a message in it changed length once the addresses it names were known",
4469 what(),
4470 encoded.len(),
4471 reserved
4472 )))
4473}
4474
4475/// Where one object header goes: chunk 0's block and, when the header does
4476/// not fit it, a continuation block of its own.
4477///
4478/// Produced by [`Hdf5Writer::place_header`] and consumed by
4479/// [`Hdf5Writer::encode_header_in`]; between the two, everything the header
4480/// names is built against the address it records. The sizes travel with the
4481/// addresses because they are what the blocks were reserved at: the writing
4482/// pass checks each image against them rather than trusting that the two
4483/// passes agreed.
4484#[derive(Debug, Clone, Copy)]
4485struct HeaderPlacement {
4486 /// Chunk 0's address.
4487 addr: u64,
4488 /// Bytes reserved at `addr`. For a fresh header that is the whole image,
4489 /// a continuation chunk included, since one is laid directly behind
4490 /// chunk 0 in the same block.
4491 size: usize,
4492 /// Whether the block is one the object's existing header already
4493 /// occupied, which chunk 0 is then held to the size of; a fresh block is
4494 /// an exact fit.
4495 kept: bool,
4496 /// A continuation block of its own, `(address, size)`: what a kept block
4497 /// too small for every message spills into.
4498 continuation: Option<(u64, usize)>,
4499}
4500
4501impl HeaderPlacement {
4502 /// A block of `size` bytes at `addr` holding the whole header.
4503 fn fresh(addr: u64, size: usize) -> Self {
4504 Self {
4505 addr,
4506 size,
4507 kept: false,
4508 continuation: None,
4509 }
4510 }
4511
4512 /// The placement as the registry records a written header: chunk 0's
4513 /// block, then the continuation block when there is one.
4514 fn blocks(&self) -> crate::io::object_header_io::HeaderBlocks {
4515 std::iter::once((self.addr, self.size as u64))
4516 .chain(self.continuation.map(|(a, s)| (a, s as u64)))
4517 .collect()
4518 }
4519
4520 /// The placement a written header's recorded blocks describe, to write
4521 /// it back over: chunk 0 held to its block, and the continuation chunk,
4522 /// if it has one, to its own.
4523 fn over(blocks: &[(u64, u64)]) -> Option<Self> {
4524 match blocks {
4525 [(addr, size)] => Some(Self {
4526 addr: *addr,
4527 size: *size as usize,
4528 kept: true,
4529 continuation: None,
4530 }),
4531 [(addr, size), (cont, cont_size)] => Some(Self {
4532 addr: *addr,
4533 size: *size as usize,
4534 kept: true,
4535 continuation: Some((*cont, *cont_size as usize)),
4536 }),
4537 _ => None,
4538 }
4539 }
4540}
4541
4542/// Where every object header this finalize writes goes.
4543///
4544/// Produced by [`Hdf5Writer::allocate_object_headers`] and consumed by
4545/// [`Hdf5Writer::write_object_headers`].
4546struct HeaderLayout {
4547 /// `(dataset index, placement)`, in write order.
4548 datasets: Vec<(usize, HeaderPlacement)>,
4549 /// `(group index, placement)`, in write order.
4550 groups: Vec<(usize, HeaderPlacement)>,
4551 /// The root group's placement.
4552 root: HeaderPlacement,
4553}
4554
4555/// The chunk-0 blocks existing object headers keep across a rewrite, by
4556/// object: `(address, length)` of each, as the open-time walk read it.
4557///
4558/// Filled by [`Hdf5Writer::supersede_headers`] from the registry's
4559/// `obj_header_blocks` and consumed by
4560/// [`Hdf5Writer::allocate_object_headers`].
4561#[derive(Default)]
4562struct KeptChunks {
4563 datasets: std::collections::HashMap<usize, (u64, u64)>,
4564 groups: std::collections::HashMap<usize, (u64, u64)>,
4565 root: Option<(u64, u64)>,
4566}
4567
4568/// Refuse a region-reference selection the target dataset's extent does not
4569/// admit — libhdf5's `H5S_select_valid`, which `H5Rcreate` applies before it
4570/// serializes anything.
4571///
4572/// The rank check comes from [`Selection::to_boxes`], which also refuses a
4573/// regular hyperslab with an unlimited count or block; a region reference has
4574/// no growable extent to resolve one against.
4575fn validate_region_selection(selection: &Selection, dims: &[u64], path: &str) -> IoResult<()> {
4576 let boxes = selection.to_boxes(dims).map_err(|e| {
4577 crate::io::IoError::InvalidState(format!("region reference over '{path}': {e}"))
4578 })?;
4579 for (start, count) in boxes {
4580 for (d, (&s, &c)) in start.iter().zip(&count).enumerate() {
4581 if s.checked_add(c).is_none_or(|end| end > dims[d]) {
4582 return Err(crate::io::IoError::InvalidState(format!(
4583 "region reference over '{path}' selects {s}..{} in dimension {d}, \
4584 outside the dataset's extent of {}",
4585 s.saturating_add(c),
4586 dims[d]
4587 )));
4588 }
4589 }
4590 }
4591 Ok(())
4592}
4593
4594/// What a reopen found already on disk in dense form, by the scope whose
4595/// header names it.
4596///
4597/// Both halves together because they are found together — one walk of the
4598/// reopened headers fills both — and released together only in the delete
4599/// path; a finalize supersedes attribute storage before it lays object
4600/// headers out and link storage after, so each half has its own owner.
4601#[derive(Debug, Default)]
4602struct SupersededDense {
4603 attrs: HashMap<AttrScope, AttributeInfoMessage>,
4604 links: HashMap<LinkScope, LinkInfoMessage>,
4605}
4606
4607/// Which object's attribute list a prepared dense layout belongs to.
4608#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4609pub(crate) enum AttrScope {
4610 Root,
4611 Group(usize),
4612 Dataset(usize),
4613}
4614
4615/// Which group's link list a prepared dense layout belongs to.
4616#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
4617pub(crate) enum LinkScope {
4618 Root,
4619 Group(usize),
4620}
4621
4622/// Attributes an object header keeps before libhdf5 spills the whole set to
4623/// dense storage (`H5O_CRT_ATTR_MAX_COMPACT_DEF`).
4624const MAX_COMPACT_ATTRS: usize = 8;
4625
4626/// Links `H5G__obj_create_real` sizes a new group's object header for
4627/// (`H5G_CRT_GINFO_EST_NUM_ENTRIES`), and the name length it assumes for each
4628/// (`H5G_CRT_GINFO_EST_NAME_LEN`). Together with the link info and group info
4629/// messages they are the whole of chunk 0 — see
4630/// [`chunk0_capacity`](Hdf5Writer::chunk0_capacity).
4631const EST_LINK_COUNT: usize = 4;
4632/// See [`EST_LINK_COUNT`].
4633const EST_LINK_NAME_LEN: usize = 8;
4634
4635/// Messages a shared-message index keeps in list form before it becomes a v2
4636/// B-tree (`H5F_CRT_SHMSG_LIST_MAX_DEF`).
4637const DEFAULT_SOHM_LIST_MAX: u16 = 50;
4638
4639/// Messages a shared-message B-tree index drops to before it reverts to a
4640/// list (`H5F_CRT_SHMSG_BTREE_MIN_DEF`).
4641const DEFAULT_SOHM_BTREE_MIN: u16 = 40;
4642
4643/// Links a group header keeps before libhdf5 spills the whole set to dense
4644/// storage (`H5G_CRT_GINFO_MAX_COMPACT`). This writer emits no phase-change
4645/// values in the Group Info message, so the default is what applies.
4646const MAX_COMPACT_LINKS: usize = 8;
4647
4648/// Bytes a compact dataset's raw image may occupy.
4649///
4650/// `H5D__compact_construct` bounds it by `H5O_MESG_MAX_SIZE` less the layout
4651/// message's own four bytes (version, class, and the 2-byte data length).
4652/// The constant it subtracts from is 65536, one past what the object header's
4653/// 2-byte message size field can express, so the ceiling here is taken from
4654/// [`MAX_MESSAGE_SIZE`] — the largest message that actually encodes — and is
4655/// one byte below libhdf5's.
4656pub const MAX_COMPACT_DATA: usize = MAX_MESSAGE_SIZE - 4;
4657
4658/// Smallest userblock a file can be created with, and the granularity of
4659/// every larger one: `H5Pset_userblock` takes 0 or a power of two from here
4660/// up, because `H5FD_locate_signature` looks for the superblock at 0 and then
4661/// at this offset doubled repeatedly.
4662pub const MIN_USERBLOCK: u64 = 512;
4663
4664impl Hdf5Writer {
4665 /// Create a new HDF5 file at `path` using the env-var-derived locking
4666 /// policy (controlled by `HDF5_USE_FILE_LOCKING`).
4667 ///
4668 /// The superblock (48 bytes for v3 with 8-byte offsets) is reserved at
4669 /// offset 0 and written during `close()`.
4670 pub fn create(path: &Path) -> IoResult<Self> {
4671 Self::create_with_locking(
4672 path,
4673 crate::io::locking::FileLocking::from_env_or(Default::default()),
4674 )
4675 }
4676
4677 /// Create a new HDF5 file at `path` with an explicit locking policy.
4678 pub fn create_with_locking(
4679 path: &Path,
4680 locking: crate::io::locking::FileLocking,
4681 ) -> IoResult<Self> {
4682 Self::create_with_options(
4683 path,
4684 FileCreateOptions {
4685 locking,
4686 ..Default::default()
4687 },
4688 )
4689 }
4690
4691 /// Create a new HDF5 file at `path` with explicit file-creation options.
4692 pub fn create_with_options(path: &Path, options: FileCreateOptions) -> IoResult<Self> {
4693 let FileCreateOptions {
4694 locking,
4695 track_order,
4696 track_times,
4697 libver,
4698 userblock,
4699 shared_messages,
4700 file_space,
4701 } = options;
4702 shared_messages.validate()?;
4703 file_space.validate()?;
4704 if userblock != 0 && (userblock < MIN_USERBLOCK || !userblock.is_power_of_two()) {
4705 return Err(crate::io::IoError::InvalidState(format!(
4706 "a userblock is {MIN_USERBLOCK} bytes or a power of two above it, \
4707 not {userblock}: a reader locates the superblock by doubling its \
4708 search offset from {MIN_USERBLOCK}, so no other size can hold one"
4709 )));
4710 }
4711 let policy = free_space::SpacePolicy::for_message(&file_space.message());
4712 // `H5F__super_init` (H5Fsuper.c:1182-1192) refuses a userblock that is
4713 // not a whole number of allocation units, which for a paged file is
4714 // the file-space page: everything after the userblock is addressed
4715 // from its end, so a userblock that is not a page multiple would put
4716 // every page boundary off the file's own grid.
4717 if let Some(page) = policy.page() {
4718 if userblock != 0 && userblock % page != 0 {
4719 return Err(crate::io::IoError::InvalidState(format!(
4720 "a paged file's userblock is a multiple of its {page}-byte \
4721 file-space page, not {userblock}"
4722 )));
4723 }
4724 }
4725 let mut handle = FileHandle::create_with_locking(path, locking)?;
4726 if userblock != 0 {
4727 // Written while the handle is still unbased, so offset 0 is the
4728 // start of the file: the block belongs to the application, not to
4729 // the HDF5 address space that begins where it ends. libhdf5 zeroes
4730 // it the same way (`H5F__super_init`), leaving a file whose first
4731 // `userblock` bytes are the application's to overwrite.
4732 handle.write_at(0, &vec![0u8; userblock as usize])?;
4733 handle.set_base(userblock);
4734 }
4735 let ctx = FormatContext::default_v3();
4736
4737 // `H5F_LIBVER_EARLIEST` is the one bound under which libhdf5 writes
4738 // the classic generation — the version-1 rows of every
4739 // message-version table, the symbol-table group form
4740 // (`H5G__obj_create_real`, H5Gobj.c:179) and the version-0 superblock
4741 // row of `HDF5_superblock_ver_bounds`.
4742 //
4743 // Shared object header messages move the last of those three and
4744 // nothing else. Their master table lives in a superblock extension,
4745 // which only a version-2 superblock has, so `H5F__super_init` raises
4746 // the superblock to version 2 whatever the low bound says
4747 // (H5Fsuper.c:1135) — but it does not touch `H5F_LOW_BOUND`, which is
4748 // what every other rule reads. So such a file is a version-2
4749 // superblock over symbol-table groups and version-1 messages, which
4750 // is what the `tests/fixtures/sohm_*.h5` files libhdf5 itself wrote
4751 // are.
4752 let classic = libver == Some(LibverBound::Earliest);
4753 let legacy = classic.then(|| Box::new(LegacyFile::created(ctx, userblock)));
4754 // Non-default file-space properties raise the superblock the same way
4755 // a shared-message table does, and for the same reason: the message
4756 // that declares them lives in an extension, and only a version-2
4757 // superblock has one (H5Fsuper.c:1144).
4758 let superblock_version = SuperblockVersion::Chosen(
4759 if classic && shared_messages.specs().is_empty() && file_space.is_default() {
4760 SUPERBLOCK_V0
4761 } else {
4762 SUPERBLOCK_V2
4763 },
4764 );
4765
4766 // Reserve the superblock at offset 0. Which version it gets is only
4767 // known once the file's content is (see `superblock_version_for`),
4768 // but the two a version-2 file can reach — 2 and 3 — encode to the
4769 // same size, so the reservation follows the base version alone.
4770 let superblock_size = match legacy.as_deref() {
4771 Some(l) if matches!(superblock_version, SuperblockVersion::Chosen(v) if v < SUPERBLOCK_V2) => {
4772 l.superblock.encoded_size()
4773 }
4774 _ => SuperblockV2V3::size_for(ctx.sizeof_addr),
4775 };
4776 // The superblock is an ordinary allocation, not a reservation: under
4777 // paged aggregation it takes the whole of page zero and leaves the
4778 // rest of that page as a section of the metadata manager, which is
4779 // what `H5F__super_init` gets from `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`
4780 // going through `H5MF__alloc_pagefs`. Unpaged it returns offset zero
4781 // and moves the end of the file to `superblock_size`, which is what
4782 // reserving it did.
4783 let allocator = FileAllocator::with_policy(0, policy);
4784 allocator.allocate(superblock_size as u64, FreeSpaceClass::Metadata);
4785
4786 Ok(Self {
4787 handle,
4788 allocator,
4789 ctx,
4790 datasets: Slot::new(Vec::new()),
4791 groups: Slot::new(Vec::new()),
4792 hard_links: Slot::new(Vec::new()),
4793 symbolic_links: Slot::new(Vec::new()),
4794 committed_datatypes: Slot::new(Vec::new()),
4795 preserved_links: Slot::new(Vec::new()),
4796 name_index: Slot::new(Box::new(NameIndex::new())),
4797 root_attributes: Slot::new(Vec::new()),
4798 create_lock: Slot::new(()),
4799 libver,
4800 closed: false,
4801 swmr_active: false,
4802 cwfs: Slot::new(Vec::new()),
4803 root_group_addr: None,
4804 superseded_root_header: Vec::new(),
4805 // A new file starts at the oldest superblock the generation it was
4806 // created in allows, and finalize raises it if the content needs a
4807 // newer one.
4808 superblock_version,
4809 dense_attributes: Slot::new(HashMap::new()),
4810 dense_links: Slot::new(HashMap::new()),
4811 superseded_dense: Slot::new(None),
4812 track_order: TrackOrder::uniform(track_order),
4813 track_times,
4814 root_track_order: TrackOrder::uniform(track_order),
4815 // The root group is created with the file, so it captures the
4816 // policy the same instant every other field of it is settled.
4817 root_times: track_times.then(|| ObjectTimes::created_at(now_seconds())),
4818 next_creation_seq: Slot::new(0),
4819 pending_object_references: Slot::new(Vec::new()),
4820 pending_heap_references: Slot::new(Vec::new()),
4821 attribute_references: Slot::new(Vec::new()),
4822 legacy,
4823 symbol_tables: SymbolTables::none_found(),
4824 // A created file has no extension to carry and no ranks but the
4825 // library defaults: `H5Pset_sym_k`/`H5Pset_istore_k` have no
4826 // equivalent on this writer's creation path.
4827 btree: BTreeV1Config::default(),
4828 extension: Box::default(),
4829 // A file created at the library defaults declares no file-space
4830 // strategy, so it has no message to write and no manager to keep;
4831 // one created with any other properties owns both.
4832 free_space: (!file_space.is_default()).then(|| {
4833 Box::new(FileSpaceState {
4834 info: file_space.message(),
4835 superseded: Vec::new(),
4836 })
4837 }),
4838 sohm: (!shared_messages.specs().is_empty())
4839 .then(|| Box::new(SohmState::new(shared_messages.specs().to_vec(), Vec::new()))),
4840 source_dir: source_dir_of(path)?,
4841 })
4842 }
4843
4844 /// Target the libhdf5 2.0 file format for datasets created after this
4845 /// call: filtered chunked datasets get layout message version 5, whose
4846 /// chunk indexes store chunk sizes in a fixed `sizeof_size`-byte field
4847 /// with no overflow limit (see [`Self::chunk_layout_version`]). Off by
4848 /// default, because readers older than libhdf5 2.0 — including the
4849 /// 1.14-based h5py wheels — reject version 5.
4850 ///
4851 /// `false` names `H5F_LIBVER_EARLIEST`, the far end of the same table,
4852 /// rather than un-naming the bound: it is `set_libver_bound`'s contract
4853 /// that applies, chunk index included.
4854 pub fn set_libver_latest(&mut self, latest: bool) -> IoResult<()> {
4855 self.set_libver_bound(if latest {
4856 LibverBound::V200
4857 } else {
4858 LibverBound::Earliest
4859 })
4860 }
4861
4862 /// Bytes this file reserves in front of its superblock
4863 /// (`H5Pget_userblock`).
4864 ///
4865 /// The same value for a file created with one and for a file reopened
4866 /// through [`open_append_with_locking`](Self::open_append_with_locking),
4867 /// which takes it from where the signature turned up: it is the base of
4868 /// the handle's address space either way.
4869 pub fn userblock_size(&self) -> u64 {
4870 self.handle.base()
4871 }
4872
4873 /// Set the file's low libver bound, the equivalent of
4874 /// `H5Pset_libver_bounds`'s `low` argument. Objects created after this
4875 /// call encode their messages at the versions that bound calls for.
4876 ///
4877 /// On a reopened file the bound is taken as named, above or below the row
4878 /// the file's superblock version belongs to, as `H5Fopen` takes a fapl's
4879 /// since libhdf5 2.0 (HDFGroup/hdf5#4939): a version-3 superblock opened
4880 /// at `Earliest` gets version-1 B-tree chunk indexes appended, as h5py on
4881 /// libhdf5 2.x appends them. Only a bound the file's format cannot express
4882 /// at all is refused.
4883 pub fn set_libver_bound(&mut self, libver: LibverBound) -> IoResult<()> {
4884 // A classic file cannot honour a newer bound: every encoder in it
4885 // reads `H5F_LOW_BOUND`, and raising that is what makes libhdf5 write
4886 // the version-2/3 superblock this file does not have. Refused rather
4887 // than pinned silently, so the caller learns the bound did not take.
4888 if libver != LibverBound::Earliest && self.is_legacy() {
4889 return Err(crate::io::IoError::Unsupported(format!(
4890 "cannot set the library-version bound to {libver:?} on this file: it is in the classic (version-0/1 superblock) format, which libhdf5 writes only at H5F_LIBVER_EARLIEST"
4891 )));
4892 }
4893 self.libver = Some(libver);
4894 Ok(())
4895 }
4896
4897 /// The generation the *message* encoders follow — dataspace, datatype,
4898 /// fill value, attribute.
4899 ///
4900 /// A property of the file, not of the object: `H5S__set_version`,
4901 /// `H5O__fill_set_version`, `H5A__set_version` and `H5T_set_version` all
4902 /// read `H5F_LOW_BOUND(f)` and nothing about the object they are encoding
4903 /// for. So a creation-order-tracking group in a classic file still gets
4904 /// version-1 dataspaces and version-1 attribute messages, even though its
4905 /// own header is version 2.
4906 fn message_format(&self) -> ObjectFormat {
4907 match self.legacy {
4908 Some(_) => ObjectFormat::Legacy,
4909 None => ObjectFormat::Modern,
4910 }
4911 }
4912
4913 /// The object header version an object with this creation-order policy
4914 /// gets — `H5O__set_version` (H5Oint.c:251).
4915 ///
4916 /// Version 1 is the floor a classic file's low bound sets, but tracking
4917 /// creation order of *either* kind raises the object past it: the link
4918 /// creation index lives in the message envelope and the attribute tracking
4919 /// flags live in the header prefix, and version 1 has neither. This is a
4920 /// per-object question in a classic file, which is why the format is not
4921 /// one switch for the whole file — libhdf5 writes version-2 headers inside
4922 /// a version-0 superblock whenever the creation property list asks for
4923 /// creation order.
4924 fn header_format(&self, track: TrackOrder) -> ObjectFormat {
4925 let attrs = self.header_attr_order(track.attrs);
4926 if self.legacy.is_some() && !track.links.is_tracked() && !attrs.is_tracked() {
4927 ObjectFormat::Legacy
4928 } else {
4929 ObjectFormat::Modern
4930 }
4931 }
4932
4933 /// The attribute creation-order policy an object header records, given
4934 /// what the object's creation property list asked for.
4935 ///
4936 /// A file whose shared-message configuration covers attributes records a
4937 /// creation index on every object header message: a shared attribute is
4938 /// found again through it, so `H5SM_init` sets `store_msg_crt_idx`
4939 /// (H5SM.c:220) and `H5O__create_ohdr` then raises every header it creates
4940 /// to version 2 and ORs `H5O_HDR_ATTR_CRT_ORDER_TRACKED` into its flags
4941 /// (H5Oint.c:364, H5Oint.c:442) whatever the property list says. So on
4942 /// such a file the floor is `Tracked` — this is the only place that floor
4943 /// is applied, and both the header version and the header flags come
4944 /// through here.
4945 fn header_attr_order(&self, requested: CreationOrder) -> CreationOrder {
4946 if requested.is_tracked() || !self.tracks_message_creation_index() {
4947 return requested;
4948 }
4949 CreationOrder::Tracked
4950 }
4951
4952 /// Whether this finalize replaces the file's shared-message table.
4953 ///
4954 /// It does whenever the file has indexes and no table has been published
4955 /// this session — every finalize of a file created with them, and the
4956 /// first finalize after a reopen. `build_shared_messages` lays a table out
4957 /// whole from the whole message set rather than inserting into an existing
4958 /// one, so a reopen's table is a *replacement*: every heap ID in the file
4959 /// is reassigned, which makes every object header that holds one stale
4960 /// however little else about it changed. A second finalize (a SWMR close)
4961 /// keeps the table the first published and answers `false`.
4962 fn rebuilds_shared_messages(&self) -> bool {
4963 self.sohm
4964 .as_deref()
4965 .is_some_and(|s| s.table_addr.lock().is_none())
4966 }
4967
4968 /// Whether every object header this writer emits records message creation
4969 /// indices.
4970 fn tracks_message_creation_index(&self) -> bool {
4971 self.sohm
4972 .as_deref()
4973 .is_some_and(SohmState::shares_attributes)
4974 }
4975
4976 /// Whether the group at `scope` stores its links in a symbol table —
4977 /// `H5G__obj_create_real` (H5Gobj.c:129) and the conversion
4978 /// `H5G_obj_insert` performs (H5Gobj.c:512).
4979 ///
4980 /// The new group format is used unconditionally from `H5F_LIBVER_V18` up
4981 /// *for a group being created*, and below it only when the group tracks
4982 /// link creation order: a symbol table entry has no room for a creation
4983 /// index. The two axes are independent — a group that tracks only
4984 /// *attribute* creation order gets a version-2 header over a symbol table,
4985 /// which is what libhdf5 writes for it.
4986 ///
4987 /// A group the reopen found in a symbol table is not being created, and
4988 /// `H5G_obj_insert` never moves an existing group to the new format for
4989 /// the bound's sake. So [`SymbolTables::found`] answers for it whatever
4990 /// generation the rest of this session writes at.
4991 ///
4992 /// The content of the group is the third axis. A symbol table entry has
4993 /// three cache types and no room for a fourth, so an external or
4994 /// user-defined link cannot go in one; libhdf5 answers by converting that
4995 /// one group to link messages the moment such a link is inserted, leaving
4996 /// the superblock version, the object header version and every other group
4997 /// in the file alone. This writer builds each group's storage once at
4998 /// finalize rather than link by link, so the same rule reads as a question
4999 /// about the finished set.
5000 fn uses_symbol_table(&self, scope: LinkScope, links: CreationOrder) -> bool {
5001 (self.legacy.is_some() || self.symbol_tables.found.contains(&scope))
5002 && !links.is_tracked()
5003 && self.links_fit_symbol_table(scope, links)
5004 }
5005
5006 /// Whether every link `scope` holds is one a symbol table entry can
5007 /// express — `H5G_obj_insert`'s `obj_lnk->cset != H5T_CSET_ASCII ||
5008 /// obj_lnk->type > H5L_TYPE_BUILTIN_MAX` test (H5Gobj.c:514), asked of the
5009 /// whole set.
5010 ///
5011 /// A link a reopen carried through verbatim counts too, and one this
5012 /// writer cannot even decode counts as not fitting: the entry would have
5013 /// to be built from the decoded form, while a link message is re-emitted
5014 /// byte for byte.
5015 fn links_fit_symbol_table(&self, scope: LinkScope, order: CreationOrder) -> bool {
5016 self.group_links(scope, order)
5017 .iter()
5018 .all(LinkMessage::fits_symbol_table)
5019 && self.preserved_links_for(scope).iter().all(|encoded| {
5020 LinkMessage::decode(encoded, &self.ctx)
5021 .is_ok_and(|(link, _)| link.fits_symbol_table())
5022 })
5023 }
5024
5025 /// The header format of the registered dataset at `index`.
5026 ///
5027 /// A dataset has no links, so only the attribute half of the policy can
5028 /// raise it past version 1.
5029 fn dataset_header_format(&self, index: usize) -> ObjectFormat {
5030 let ds = self.ds(index);
5031 let attrs = ds.lock().track_attr_order;
5032 self.header_format(TrackOrder {
5033 links: CreationOrder::default(),
5034 attrs,
5035 })
5036 }
5037
5038 /// The header format of the registered group at `index`.
5039 fn group_header_format(&self, index: usize) -> ObjectFormat {
5040 let grp = self.grp(index);
5041 let track = grp.lock().track_order;
5042 self.header_format(track)
5043 }
5044
5045 /// The bound one family of encoders is written at, and the single reader
5046 /// of the `libver` field.
5047 ///
5048 /// A bound the caller named is the fapl's `low`, as it was named: since
5049 /// libhdf5 2.0 `H5F__super_read` raises nothing but an SWMR-write open
5050 /// (HDFGroup/hdf5#4939), so a reopened file's superblock version says
5051 /// which generation the file *is* and nothing about the generation this
5052 /// session appends in. With no bound named the answer is `create_default`,
5053 /// which differs per family because this crate's default file is two rows
5054 /// rather than one bound (see the `libver` field, [`encoding_libver`] and
5055 /// [`layout_version_bound`]) — the same default on a created file, whose
5056 /// superblock is then written to match, and on a reopened one, whose
5057 /// superblock is written back as found.
5058 ///
5059 /// [`encoding_libver`]: Self::encoding_libver
5060 /// [`layout_version_bound`]: Self::layout_version_bound
5061 fn session_libver(&self, create_default: LibverBound) -> LibverBound {
5062 let bound = self.libver.unwrap_or(create_default);
5063 match self.message_format() {
5064 // `H5F_LIBVER_EARLIEST` is the only low bound under which libhdf5
5065 // writes a version-0/1 superblock at all, so a newer structure
5066 // inside one is a combination no libhdf5 produces. Refused where
5067 // the caller asks for it (`set_libver_bound`) rather than silently
5068 // dropped; capping here is what keeps the encoders honest if a
5069 // path ever misses that gate.
5070 ObjectFormat::Legacy => bound.min(LibverBound::Earliest),
5071 ObjectFormat::Modern => bound,
5072 }
5073 }
5074
5075 /// The bound the message encoders see — dataspace, datatype, fill value,
5076 /// attribute.
5077 fn encoding_libver(&self) -> LibverBound {
5078 self.session_libver(LibverBound::Earliest)
5079 }
5080
5081 /// The data layout message version this file's bound calls for —
5082 /// `H5O_layout_ver_bounds[H5F_LOW_BOUND(f)]` (H5Dlayout.c:44), the term
5083 /// `H5D__chunk_set_info` weighs against the version a chunk *requires*
5084 /// (H5Dchunk.c:936, :1046).
5085 ///
5086 /// With no bound named the row is `H5F_LIBVER_V110`'s: this crate's
5087 /// default file uses the v1.10 chunk indexes, which is exactly what that
5088 /// row says and what no other row does (see the `libver` field for why the
5089 /// default is not `Earliest` here even though the datatype and superblock
5090 /// tables read it that way). A reopened file takes the same default: a
5091 /// version-2 superblock gets v1.10 indexes appended unless a bound below
5092 /// `V110` is named, whose layout version of 3 has no index-type field at
5093 /// all and puts the appended chunks on the version-1 B-tree.
5094 fn layout_version_bound(&self) -> u8 {
5095 self.session_libver(LibverBound::V110).layout_version()
5096 }
5097
5098 /// The data layout version a chunk of `chunk_bytes` *requires* whatever
5099 /// the bound says — `version_req` in `H5D__chunk_set_info` (H5Dchunk.c:909).
5100 ///
5101 /// Only one thing raises it: a chunk over 4 GiB does not fit the version-4
5102 /// message's 32-bit stored-size field. The floor is the default the
5103 /// creation property list carries, `H5O_LAYOUT_VERSION_DEFAULT`
5104 /// (H5Oprivate.h:451), which is why a classic file's chunked dataset is a
5105 /// version-3 message rather than the version-1 its bound's row names.
5106 fn required_chunk_layout_version(chunk_bytes: u64) -> u8 {
5107 if chunk_bytes > u32::MAX as u64 {
5108 5
5109 } else {
5110 LAYOUT_VERSION_DEFAULT
5111 }
5112 }
5113
5114 /// Whether a new chunked dataset of this chunk size is indexed by one of
5115 /// the v1.10 indexes — extensible array, fixed array, v2 B-tree, single
5116 /// chunk or implicit — rather than by the version-1 B-tree.
5117 ///
5118 /// The gate `H5D__chunk_set_info` puts in front of the whole
5119 /// index-selection block (H5Dchunk.c:936): the bound's layout version
5120 /// reaches 4, or the chunk requires a version that does. Only inside it
5121 /// does the dataspace get to pick between the five; below it the layout
5122 /// message has no index-type field and the chunks go on the version-1
5123 /// B-tree. So the format decides before the shape does — a fixed shape
5124 /// covered by exactly one chunk takes the single-chunk index only on the
5125 /// near side of this gate.
5126 pub(crate) fn uses_v110_chunk_indexing(&self, chunk_bytes: u64) -> bool {
5127 self.layout_version_bound() >= 4 || Self::required_chunk_layout_version(chunk_bytes) >= 4
5128 }
5129
5130 /// Refuse an SWMR session this file's format cannot record.
5131 ///
5132 /// The two checks `H5F__start_swmr_write` opens with: the superblock must
5133 /// be at least version 3 (H5Fint.c:3814, hdf5_1.14.6 H5Fint.c:3751) — the
5134 /// only version with the status-flags field that says a writer is attached
5135 /// — and the low bound must be at least `H5F_LIBVER_V110` (H5Fint.c:3818),
5136 /// the oldest bound whose `HDF5_superblock_ver_bounds` row reaches version
5137 /// 3.
5138 ///
5139 /// The first is the [`SuperblockVersion`] question: a reopened file
5140 /// already has its version and reopening never rewrites one, so it is
5141 /// checked as found; a file this writer created has no version on disk
5142 /// yet, and nothing in it says the superblock may not be version 3 — SWMR
5143 /// is what makes it one. The second is asked of the bound the caller
5144 /// named on either path, since a reopened file's superblock no longer
5145 /// raises it (HDFGroup/hdf5#4939); a caller who named none passes,
5146 /// because the default's layout row is `V110`'s, the row the v1.10 chunk
5147 /// indexes an SWMR reader follows belong to.
5148 ///
5149 /// Named, not silently upgraded. libhdf5 upgrades in the one case where
5150 /// SWMR is asked for at *create* time (`H5F_ACC_SWMR_WRITE` raises the
5151 /// bound to V110 in `H5F__super_init`, H5Fsuper.c:1131); on the reopen
5152 /// path it refuses instead, and so does this.
5153 fn reject_swmr(&self) -> IoResult<()> {
5154 let below_v110 = self.libver.is_some_and(|b| b < LibverBound::V110);
5155 let why = match self.superblock_version {
5156 SuperblockVersion::Existing(version) if version < SUPERBLOCK_V3 => format!(
5157 "its superblock is version {version}, and reopening a file never \
5158 rewrites that"
5159 ),
5160 SuperblockVersion::Chosen(_) if self.is_legacy() => {
5161 "it is in the classic (version-0/1 superblock) format that \
5162 H5F_LIBVER_EARLIEST selects"
5163 .to_string()
5164 }
5165 _ if below_v110 => "it was asked for at a library-version bound below \
5166 H5F_LIBVER_V110, whose superblock row is version 2"
5167 .to_string(),
5168 _ => return Ok(()),
5169 };
5170 Err(crate::io::IoError::Unsupported(format!(
5171 "cannot start an SWMR session on this file: {why}, and SWMR needs a \
5172 version-3 superblock to record that a writer is attached; create the \
5173 file at H5F_LIBVER_V110 or newer"
5174 )))
5175 }
5176
5177 /// Whether this file is in the classic (version-0/1 superblock) format,
5178 /// whose groups store their links in symbol tables — either because it
5179 /// was reopened in it or because it was created at
5180 /// `H5F_LIBVER_EARLIEST`.
5181 pub(crate) fn is_legacy(&self) -> bool {
5182 self.legacy.is_some()
5183 }
5184
5185 /// The v1-B-tree "K" ranks in force for this file, from which every v1
5186 /// node's width is derived.
5187 ///
5188 /// A version-0/1 superblock records them in a field of its own and a
5189 /// version-2/3 one in a B-tree-K message in its superblock extension, so
5190 /// the file's generation says nothing about whether they are the defaults
5191 /// — `H5F__super_read` reads both into the same `H5F_shared_t`, and so
5192 /// does the reopen, into `btree`.
5193 fn btree_v1_config(&self) -> BTreeV1Config {
5194 self.btree
5195 }
5196
5197 /// Track and index creation order for the links and the attributes of
5198 /// every object created after this call — the equivalent of setting
5199 /// `H5Pset_link_creation_order` and `H5Pset_attr_creation_order` to
5200 /// `H5P_CRT_ORDER_TRACKED | H5P_CRT_ORDER_INDEXED` on the creation
5201 /// property lists those objects are made with.
5202 ///
5203 /// Objects already created keep the policy they were made under, exactly
5204 /// as libhdf5 keeps what their creation property list said. The root
5205 /// group is created with the file, so its policy comes from
5206 /// [`create_with_options`](Self::create_with_options) instead.
5207 pub fn set_track_order(&mut self, track: bool) {
5208 self.track_order = TrackOrder::uniform(track);
5209 }
5210
5211 /// Record the times of every object created after this call —
5212 /// `H5Pset_obj_track_times` on the creation property lists those objects
5213 /// are made with.
5214 ///
5215 /// Off by default, which is h5py's default and not libhdf5's: h5py's
5216 /// high-level API sets `track_times=False` on every object it makes
5217 /// (`_hl/files.py:189`, `_hl/dataset.py:39`, `_hl/group.py:42`), while a
5218 /// bare creation property list leaves it on (`H5O_CRT_OHDR_FLAGS_DEF` is
5219 /// `H5O_HDR_STORE_TIMES`, H5Opkg.h:74). A caller after libhdf5's own
5220 /// bytes turns it on here.
5221 ///
5222 /// Objects already created keep the policy they were made under, and the
5223 /// root group takes its own from
5224 /// [`create_with_options`](Self::create_with_options) — the same split
5225 /// [`set_track_order`](Self::set_track_order) has, and for the same
5226 /// reason: this is a creation property, not a file-wide setting.
5227 pub fn set_track_times(&mut self, track: bool) {
5228 self.track_times = track;
5229 }
5230
5231 /// The times an object created right now records — all four set to the
5232 /// current time, as `H5O_apply_ohdr` initialises them (H5Oint.c:411-414),
5233 /// or `None` when this session is not tracking times.
5234 ///
5235 /// INVARIANT: every object this writer registers takes its `times` from
5236 /// here. The policy belongs to the creation property list, so reading
5237 /// [`track_times`](Self::track_times) at any later moment — a finalize, a
5238 /// header rewrite — would stamp a policy the object was not made under.
5239 fn created_object_times(&self) -> Option<ObjectTimes> {
5240 self.track_times
5241 .then(|| ObjectTimes::created_at(now_seconds()))
5242 }
5243
5244 /// Layout message version for a new chunked dataset on one of the v1.10
5245 /// indexes — `H5D__chunk_set_info`'s closing
5246 /// `MAX3(layout->version, version_req, MIN(bound, version_perf))`
5247 /// (H5Dchunk.c:1046).
5248 ///
5249 /// Version 5 is *required* for a chunk over 4 GiB (pre-2.0 readers cannot
5250 /// handle one even though the v4 wire format could express it) and
5251 /// *preferred* for filtered chunks, which is why it takes the file's
5252 /// bound to get there: the preference is capped by the bound's own row,
5253 /// so only the 2.0 format lets it through. Everything else stays at
5254 /// version 4, which every 1.10+ reader accepts.
5255 fn chunk_layout_version(&self, filtered: bool, chunk_bytes: u64) -> u8 {
5256 // `version_perf`: 4 for the v1.10 indexes as such, 5 when a filter
5257 // can make a chunk expand past what version 4 can record.
5258 let preferred = if filtered { 5 } else { 4 };
5259 Self::required_chunk_layout_version(chunk_bytes)
5260 .max(self.layout_version_bound().min(preferred))
5261 .max(LAYOUT_VERSION_DEFAULT)
5262 }
5263
5264 /// Width of the stored-chunk-size field in a filtered chunk index:
5265 /// version 5 uses the fixed `sizeof_size`; version 4 derives it from the
5266 /// uncompressed chunk byte count (one spare byte included), the
5267 /// `H5D_*_COMPUTE_CHUNK_SIZE_LEN` rule shared by the extensible-array,
5268 /// fixed-array and v2-B-tree indexes.
5269 fn chunk_size_len_for(&self, layout_version: u8, chunk_bytes: u64) -> u8 {
5270 if layout_version >= 5 {
5271 self.ctx.sizeof_size
5272 } else {
5273 compute_chunk_size_len(chunk_bytes)
5274 }
5275 }
5276
5277 /// Provide public access to the format context.
5278 pub fn ctx(&self) -> &FormatContext {
5279 &self.ctx
5280 }
5281
5282 /// Number of dataset slots in the registry (including soft-deleted ones).
5283 pub(crate) fn dataset_count(&self) -> usize {
5284 self.datasets.lock().len()
5285 }
5286
5287 /// Clone out the [`DatasetRef`] for `index`, releasing the registry lock
5288 /// immediately. Lock the returned ref to read or mutate that one dataset.
5289 ///
5290 /// Panics on an out-of-range index, exactly like the `Vec` indexing it
5291 /// replaces; bounds-checking callers consult [`Self::dataset_count`] first.
5292 ///
5293 /// MUST NOT be called while the registry [`Slot`] is already locked (it
5294 /// would deadlock the `threadsafe` mutex / panic the single-thread
5295 /// `RefCell`): collect the refs you need, drop the registry guard, then work.
5296 pub(crate) fn ds(&self, index: usize) -> DatasetRef {
5297 Shared::clone(&self.datasets.lock()[index])
5298 }
5299
5300 /// Number of group slots in the registry (including soft-deleted ones).
5301 pub(crate) fn group_count(&self) -> usize {
5302 self.groups.lock().len()
5303 }
5304
5305 /// Clone out the [`GroupRef`] for `index`. Same contract as [`Self::ds`].
5306 pub(crate) fn grp(&self, index: usize) -> GroupRef {
5307 Shared::clone(&self.groups.lock()[index])
5308 }
5309
5310 /// Enter the create gate: take `create_lock` and check that `name` is not
5311 /// already taken. The returned witness is what [`Self::push_dataset`]
5312 /// requires, so the uniqueness check and the registry push are atomic
5313 /// (see `create_lock`) at every creator by construction.
5314 pub(crate) fn begin_create(&self, name: &str) -> IoResult<CreateGuard<'_>> {
5315 let gate = self.create_lock.lock();
5316 // A creation path through hard links lands in the link's target
5317 // group, as HDF5 traversal does. Canonicalizing here — the one
5318 // entry every creator passes — keeps alias forms out of the
5319 // registry.
5320 let name = self.canonical_dataset_path(name);
5321 // A path that leaves this file, or that runs into an object the
5322 // reopen kept verbatim, is refused here rather than at each creator:
5323 // this is the one gate every creation passes, so a creator added
5324 // later cannot forget the check. Both run before the parent lookup,
5325 // which would otherwise report the group such a path names as absent
5326 // instead of naming what stops the path. Uniqueness comes first among
5327 // them: a name already in the file is taken whatever holds it.
5328 self.reject_external_traversal(&name)?;
5329 self.ensure_name_free(&name)?;
5330 self.reject_preserved_object(&name)?;
5331 let (parent, _leaf) = self.split_parent(&name)?;
5332 Ok(CreateGuard {
5333 _gate: gate,
5334 name,
5335 parent,
5336 })
5337 }
5338
5339 /// Split an object path into the group that will hold its link and the
5340 /// leaf link name, resolving every component through the group registry.
5341 ///
5342 /// `path` is the registry form — no leading `/`, e.g. `"grp/sub/late"`.
5343 /// This is what keeps a `/` out of a link name: HDF5 link names are
5344 /// single path components (`H5G_traverse` splits on `/` before it ever
5345 /// reaches `H5L_link`), so a name that carries a path must name a group
5346 /// that exists, or be refused.
5347 ///
5348 /// A missing component is an error rather than an implicit group: the
5349 /// default link creation property list has `H5Pset_create_intermediate_group`
5350 /// off, and this writer exposes no property list to turn it on with.
5351 fn split_parent(&self, path: &str) -> IoResult<(Option<usize>, String)> {
5352 let (parent_path, leaf) = path.rsplit_once('/').unwrap_or(("", path));
5353 if leaf.is_empty() {
5354 return Err(crate::io::IoError::InvalidState(format!(
5355 "'{path}' does not end in a link name"
5356 )));
5357 }
5358 if parent_path.is_empty() {
5359 return Ok((None, leaf.to_string()));
5360 }
5361 let abs = format!("/{parent_path}");
5362 let groups = self.group_refs();
5363 let idx = groups
5364 .iter()
5365 .position(|g| {
5366 let gg = g.lock();
5367 gg.name == abs && !gg.deleted
5368 })
5369 .ok_or_else(|| {
5370 crate::io::IoError::NotFound(format!(
5371 "cannot create '{path}': group '{abs}' does not exist"
5372 ))
5373 })?;
5374 Ok((Some(idx), leaf.to_string()))
5375 }
5376
5377 /// Push a freshly-built dataset into the registry and return its index.
5378 /// Takes the registry lock only for the push, so it does not block an
5379 /// in-flight write that already cloned its own [`DatasetRef`] out.
5380 /// The [`CreateGuard`] proves the caller entered through
5381 /// [`Self::begin_create`] and still holds the gate.
5382 pub(crate) fn push_dataset(&self, create: &CreateGuard<'_>, info: DatasetInfo) -> usize {
5383 let name = info.name.clone();
5384 let idx = {
5385 let mut reg = self.datasets.lock();
5386 let idx = reg.len();
5387 reg.push(Shared::new(DatasetCell::new(info)));
5388 idx
5389 };
5390 self.register_name(&name, NameHit::Dataset(idx));
5391 // The spine guard is dropped before the group slot is taken: the lock
5392 // order is spine -> slot and never the reverse.
5393 if let Some(pidx) = create.parent {
5394 self.grp(pidx).lock().child_datasets.push(idx);
5395 }
5396 idx
5397 }
5398
5399 /// Push a freshly-built group into the registry and return its index.
5400 pub(crate) fn push_group(&self, info: GroupInfo) -> usize {
5401 let name = info.name.trim_start_matches('/').to_string();
5402 let idx = {
5403 let mut reg = self.groups.lock();
5404 let idx = reg.len();
5405 reg.push(Shared::new(Slot::new(info)));
5406 idx
5407 };
5408 self.register_name(&name, NameHit::Group(idx));
5409 idx
5410 }
5411
5412 /// Snapshot every [`DatasetRef`] (spine lock held only for the clone).
5413 /// Iterate the snapshot to lock each dataset one at a time — this keeps
5414 /// the lock order *spine → slot* and never reacquires the spine while a
5415 /// slot is held, which is what makes the registry deadlock-free.
5416 pub(crate) fn dataset_refs(&self) -> Vec<DatasetRef> {
5417 self.datasets.lock().iter().map(Shared::clone).collect()
5418 }
5419
5420 /// Snapshot every [`GroupRef`]; see [`Self::dataset_refs`].
5421 pub(crate) fn group_refs(&self) -> Vec<GroupRef> {
5422 self.groups.lock().iter().map(Shared::clone).collect()
5423 }
5424
5425 /// Snapshot the hard-link list (the lock is held only for the clone), so
5426 /// callers can resolve each link's target/parent — which locks dataset and
5427 /// group slots — without holding the hard-link lock.
5428 /// The next creation sequence number.
5429 ///
5430 /// One monotonic counter for datasets, groups and hard links alike: a
5431 /// group orders its links by it, so an interleaved run of `create_group`
5432 /// and `create_dataset` comes back out in the order it was made rather
5433 /// than grouped by kind.
5434 fn take_creation_seq(&self) -> u64 {
5435 let mut next = self.next_creation_seq.lock();
5436 let seq = *next;
5437 *next += 1;
5438 seq
5439 }
5440
5441 pub(crate) fn hard_links_vec(&self) -> Vec<HardLink> {
5442 self.hard_links.lock().clone()
5443 }
5444
5445 /// Snapshot the symbolic-link list; see [`Self::hard_links_vec`].
5446 pub(crate) fn symbolic_links_vec(&self) -> Vec<SymbolicLink> {
5447 self.symbolic_links.lock().clone()
5448 }
5449
5450 /// Open an existing HDF5 file for appending new datasets, using the
5451 /// env-var-derived locking policy.
5452 ///
5453 /// Reads existing dataset object headers fully, reconstructing metadata
5454 /// for chunked datasets so that `write_chunk` and `extend_dataset` work
5455 /// on reopened datasets.
5456 pub fn open_append(path: &Path) -> IoResult<Self> {
5457 Self::open_append_with_locking(
5458 path,
5459 crate::io::locking::FileLocking::from_env_or(Default::default()),
5460 )
5461 }
5462
5463 /// Carry a reopened file's shared-message table into the writer's model:
5464 /// the index specifications the file was created with, and every block the
5465 /// table occupies so the finalize that replaces it can give them back.
5466 ///
5467 /// `H5SM_init` fixes the index count, each index's type mask, its minimum
5468 /// message size and the file-wide phase-change pair when the file is
5469 /// created, and nothing afterwards changes any of them — they are file
5470 /// creation properties. So the master table on disk *is* the
5471 /// [`SharedMessageConfig`] the file was made with, read back.
5472 ///
5473 /// Returns `None` for a file with no shared-message table, which is every
5474 /// file libhdf5 writes without `H5Pset_shared_mesg_nindexes`.
5475 /// Read the free-space managers a reopened file persists, if it does.
5476 ///
5477 /// `H5F__super_read` copies the file-space info message's addresses into
5478 /// `f->shared->fs_addr[]` and the library opens each manager lazily; this
5479 /// reads them all at once, because the writer needs the whole section set
5480 /// before it allocates anything.
5481 ///
5482 /// Returns `None` — nothing read, nothing to write back — for a file with
5483 /// no file-space info message, one that does not persist, and one whose
5484 /// strategy keeps no managers at all.
5485 fn reopen_free_space(
5486 handle: &mut FileHandle,
5487 meta: &crate::io::FileMeta,
5488 ext: &crate::io::reader::SuperblockExtension,
5489 ) -> IoResult<ReopenedFreeSpace> {
5490 let none = || ReopenedFreeSpace {
5491 state: None,
5492 sections: Vec::new(),
5493 };
5494 let Some(info) = ext.file_space_info.as_ref().filter(|i| i.persist) else {
5495 return Ok(none());
5496 };
5497 if !matches!(
5498 info.strategy,
5499 FileSpaceStrategy::FsmAggr | FileSpaceStrategy::Page
5500 ) {
5501 return Ok(none());
5502 }
5503 let found = crate::io::free_space_io::read_managers(handle, &meta.ctx, info)?;
5504 Ok(ReopenedFreeSpace {
5505 state: Some(Box::new(FileSpaceState {
5506 info: info.clone(),
5507 superseded: found.blocks,
5508 })),
5509 sections: found.sections,
5510 })
5511 }
5512
5513 fn reopen_shared_messages(
5514 handle: &mut FileHandle,
5515 meta: &crate::io::FileMeta,
5516 ext: &crate::io::reader::SuperblockExtension,
5517 ) -> IoResult<Option<Box<SohmState>>> {
5518 use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
5519 use crate::format::fractal_heap::collect_heap_extents;
5520 use crate::format::sohm::{list_size, SohmMasterTable, SOHM_INDEX_LIST};
5521
5522 let (Some(table), Some(smt)) = (
5523 meta.sohm.as_ref().filter(|t| !t.indexes.is_empty()),
5524 ext.shared_message_table.as_ref(),
5525 ) else {
5526 return Ok(None);
5527 };
5528 let ctx = &meta.ctx;
5529
5530 // The extension header itself is superseded by `CarriedExtension`,
5531 // which owns it whether or not the file has shared messages; what is
5532 // superseded here is only the storage the table message names.
5533 let mut superseded = Vec::new();
5534 superseded.push((
5535 smt.table_address,
5536 SohmMasterTable::encoded_size(ctx, smt.nindexes) as u64,
5537 ));
5538
5539 let mut specs = Vec::with_capacity(table.indexes.len());
5540 for index in &table.indexes {
5541 specs.push(SohmIndexSpec {
5542 mesg_types: index.mesg_types,
5543 min_mesg_size: index.min_mesg_size,
5544 list_max: index.list_max,
5545 btree_min: index.btree_min,
5546 });
5547 let mut reader = crate::io::reader::HandleBlockReader { handle };
5548 if index.heap_addr != UNDEF_ADDR {
5549 superseded.extend(collect_heap_extents(index.heap_addr, ctx, &mut reader)?);
5550 }
5551 if index.index_addr != UNDEF_ADDR {
5552 if index.index_type == SOHM_INDEX_LIST {
5553 // `H5SM_LIST_SIZE`: the block is sized for `list_max`
5554 // records however few are in it.
5555 superseded.push((index.index_addr, list_size(ctx, index.list_max) as u64));
5556 } else {
5557 superseded.extend(collect_btree_v2_extents(
5558 index.index_addr,
5559 ctx,
5560 &mut reader,
5561 )?);
5562 }
5563 }
5564 }
5565 Ok(Some(Box::new(SohmState::new(specs, superseded))))
5566 }
5567
5568 /// Open an existing HDF5 file for appending with an explicit locking
5569 /// policy.
5570 pub fn open_append_with_locking(
5571 path: &Path,
5572 locking: crate::io::locking::FileLocking,
5573 ) -> IoResult<Self> {
5574 let mut handle = FileHandle::open_readwrite_with_locking(path, locking)?;
5575 // The same `H5FD_locate_signature` search the read path makes, through
5576 // the same handle mechanism: the offset it finds is the file's base
5577 // address, so the allocator's end-of-file, every write and the
5578 // superblock rewrite all work in the HDF5 address space, and the
5579 // userblock in `[0, base)` is not addressable from this writer at all.
5580 let super_addr = handle
5581 .locate_signature()?
5582 .ok_or(crate::format::FormatError::InvalidSignature)?;
5583 handle.set_base(super_addr);
5584 let file_size = handle.file_size()?;
5585
5586 let sb_buf = handle.read_at_most(0, 256)?;
5587 // Which generation the file is decides everything the close then
5588 // writes back: version-1 object headers and symbol-table groups over a
5589 // version-0/1 superblock, or version-2 headers and link-message groups
5590 // over a version-2/3 one. libhdf5 writes those two combinations and no
5591 // mixture of them, so the branch is taken once, here, and carried as
5592 // `legacy`.
5593 let version = crate::format::superblock::detect_superblock_version(&sb_buf)?;
5594 let (ctx, sb_btree, root_addr, ext_addr, legacy) = if version <= 1 {
5595 let sb = SuperblockV0V1::decode(&sb_buf)?;
5596 let ctx = FormatContext {
5597 sizeof_addr: sb.sizeof_offsets,
5598 sizeof_size: sb.sizeof_lengths,
5599 };
5600 // Unlike a v2/v3 superblock, a classic one carries the "K" ranks
5601 // itself; every v1-B-tree and symbol-table node width in the file
5602 // comes from them.
5603 let btree = crate::format::btree_v1::BTreeV1Config {
5604 sym_leaf_k: sb.sym_leaf_k,
5605 snode_internal_k: sb.btree_internal_k,
5606 chunk_internal_k: sb.indexed_storage_k.unwrap_or(32),
5607 };
5608 let root = sb.root_symbol_table_entry.obj_header_addr;
5609 let ext = sb.superblock_extension_address;
5610 (ctx, btree, root, ext, Some(sb))
5611 } else {
5612 let sb = SuperblockV2V3::decode(&sb_buf)?;
5613 let ctx = FormatContext {
5614 sizeof_addr: sb.sizeof_offsets,
5615 sizeof_size: sb.sizeof_lengths,
5616 };
5617 (
5618 ctx,
5619 crate::format::btree_v1::BTreeV1Config::default(),
5620 sb.root_group_object_header_address,
5621 sb.superblock_extension_address,
5622 None,
5623 )
5624 };
5625
5626 // The reopen reads object headers exactly as the reader does, so it
5627 // needs the same file-level parameters: a v2/v3 superblock carries no
5628 // B-tree K values, and only the extension can override the defaults.
5629 let (meta, ext) = crate::io::reader::Hdf5Reader::read_extension_and_meta(
5630 &mut handle,
5631 ctx,
5632 sb_btree,
5633 ext_addr,
5634 )?;
5635
5636 // A file with shared object header messages keeps datatypes,
5637 // dataspaces and attributes in a fractal heap per index, and each
5638 // object header holds a heap ID pointing at one. The table is laid out
5639 // whole from the whole message set (`build_shared_messages`), never
5640 // grown insert by insert, so a reopen carries the indexes and the
5641 // bodies forward and the next finalize lays a new table out over the
5642 // old one's blocks — which is sound exactly while no header keeping
5643 // its bytes still points into the old heap. The walk below is what
5644 // settles that.
5645 let sohm = Self::reopen_shared_messages(&mut handle, &meta, &ext)?;
5646
5647 // The extension is external truth this close rewrites, so what it held
5648 // is captured whole here — before anything else reads the file — and
5649 // re-emitted by `write_superblock_extension`. Read from the raw chain
5650 // rather than from `ext`, which keeps only the messages this crate
5651 // models.
5652 let extension = if ext_addr == UNDEF_ADDR || ext_addr == 0 {
5653 Box::<CarriedExtension>::default()
5654 } else {
5655 let (carried, blocks) = crate::io::object_header_io::superblock_extension_messages(
5656 &mut handle,
5657 &meta,
5658 ext_addr,
5659 )?;
5660 Box::new(CarriedExtension {
5661 superseded: blocks,
5662 carried,
5663 addr: Slot::new(None),
5664 })
5665 };
5666
5667 // The managers that extension's file-space info message names, read
5668 // before anything allocates: the sections they hold are file space
5669 // this session may hand out, and the close rewrites them.
5670 let reopened_free_space = Self::reopen_free_space(&mut handle, &meta, &ext)?;
5671
5672 // Discover links from root group (and subgroups recursively). Every
5673 // object is classified before it is registered, and the root is the
5674 // one object with no alternative: its header must be rewritten to
5675 // hold anything new, so an unmodellable root is refused here rather
5676 // than rewritten into whatever this writer could read of it.
5677 let mut walk = ReopenWalk::new(&mut handle, &meta);
5678 let root = match walk.plan(root_addr)? {
5679 ObjectPlan::Group(parts) => parts,
5680 ObjectPlan::Dataset(_) => {
5681 return Err(crate::io::IoError::InvalidState(
5682 "cannot open this file for appending: its root object is a dataset, \
5683 not a group"
5684 .into(),
5685 ))
5686 }
5687 ObjectPlan::Preserve { why, .. } => {
5688 return Err(crate::io::IoError::Unsupported(format!(
5689 "cannot open this file for appending: {why}. Every append rewrites the \
5690 root group's header, and this writer will not rewrite it from the part \
5691 of it that it can read"
5692 )));
5693 }
5694 };
5695 let root_header_blocks = root.header_blocks;
5696 let root_attributes = root.attributes;
5697 let root_track_order = root.track_order;
5698 let root_times = root.times;
5699 let root_dense = root.dense;
5700 let root_stab = root.stab;
5701
5702 walk.group(&root.links, "", 0)?;
5703 let collected = walk.finish();
5704 let mut link_entries = collected.hard;
5705 let mut preserved = collected.preserved;
5706 // Objects the loop below could not rebuild, by header address, so the
5707 // other links to one are preserved with it rather than left pointing
5708 // at a registry entry that is no longer there.
5709 let mut unrebuilt: std::collections::HashMap<u64, String> = Default::default();
5710
5711 // Two link entries can share one object header — hard links. Only
5712 // the first-walked path becomes the object; the rest are rebuilt
5713 // as hard-link registry entries further down. Without this split
5714 // every alias came back as its own DatasetInfo carrying the same
5715 // storage addresses, so deleting (or finalizing) one freed blocks
5716 // the others still referenced.
5717 let mut seen_header_addrs = std::collections::HashSet::new();
5718 let mut alias_entries: Vec<HardEntry> = Vec::new();
5719 link_entries.retain(|(entry, _)| {
5720 if seen_header_addrs.insert(entry.address) {
5721 true
5722 } else {
5723 alias_entries.push(entry.clone());
5724 false
5725 }
5726 });
5727
5728 // The order the walk met each object, kept before the loop below
5729 // consumes the entries: `ensure_groups_for` needs parents to precede
5730 // children.
5731 let walk_order: Vec<String> = link_entries.iter().map(|(e, _)| e.path.clone()).collect();
5732
5733 let mut existing_datasets = Vec::new();
5734 // Non-dataset link targets (groups): the header's chunk-0 address and
5735 // every block its chain occupies, by link path — so finalize can free
5736 // the blocks its rewrite supersedes — plus the attributes the header
5737 // carries, which the group registry below must keep or finalize
5738 // rewrites the group without them.
5739 type GroupHeaderInfo = (
5740 u64,
5741 crate::io::object_header_io::HeaderBlocks,
5742 Vec<AttributeEntry>,
5743 TrackOrder,
5744 Option<ObjectTimes>,
5745 );
5746 let mut group_headers: std::collections::HashMap<String, GroupHeaderInfo> =
5747 Default::default();
5748 // The dense storage each rebuilt dataset's header named, by registry
5749 // index, so finalize frees exactly what its rewrite supersedes. Keyed
5750 // after the rebuild succeeded: a preserved dataset keeps its header,
5751 // and freeing the heap that header still names would strand it.
5752 let mut dataset_dense: Vec<(usize, AttributeInfoMessage)> = Vec::new();
5753 let mut group_dense: Vec<(String, DenseCarry)> = Vec::new();
5754 // The same, for the symbol-table storage a classic group's header
5755 // names: keyed by path here, by registry index once every group has
5756 // one.
5757 let mut group_stabs: Vec<(String, StabExtents)> = Vec::new();
5758 for (entry, object) in link_entries {
5759 let HardEntry {
5760 path: name,
5761 address: obj_addr,
5762 encoded,
5763 } = entry;
5764 let parts = match object {
5765 CollectedObject::Group {
5766 header_blocks,
5767 attributes,
5768 track_order,
5769 times,
5770 dense,
5771 stab,
5772 } => {
5773 group_dense.push((name.clone(), dense));
5774 if let Some(stab) = stab {
5775 group_stabs.push((name.clone(), stab));
5776 }
5777 group_headers.insert(
5778 name,
5779 (obj_addr, header_blocks, attributes, track_order, times),
5780 );
5781 continue;
5782 }
5783 CollectedObject::Dataset(parts) => *parts,
5784 };
5785 let dense_attrs = parts.dense.attrs.clone();
5786 match rebuild_dataset(&mut handle, &meta, file_size, name.clone(), obj_addr, parts) {
5787 Ok(info) => {
5788 if let Some(ainfo) = dense_attrs {
5789 dataset_dense.push((existing_datasets.len(), ainfo));
5790 }
5791 existing_datasets.push(info);
5792 }
5793 // Kept by its bytes for the same reason a header this walk
5794 // could not decode is: the rewrite would otherwise emit an
5795 // object whose chunk index no longer names its chunks.
5796 Err(e) => {
5797 let why = format!("this writer could not rebuild its chunk index: {e}");
5798 unrebuilt.insert(obj_addr, why.clone());
5799 preserved.push(PreservedEntry {
5800 path: name,
5801 class: crate::io::reader::LinkClass::Hard,
5802 encoded,
5803 reason: Some(why),
5804 // A dataset whose chunk index would not rebuild: the
5805 // walk classified it, and it is not a datatype.
5806 kind: PreservedKind::Unclassified,
5807 });
5808 }
5809 }
5810 }
5811
5812 // Reconstruct the group registry. Every group is a link entry of its
5813 // own, whether or not a dataset lives under it, so the registry is
5814 // built from the discovered links — rebuilding it from dataset paths
5815 // alone made attribute-only and empty groups vanish at close, and
5816 // dropped the attributes of the groups that survived.
5817 let mut groups: Vec<GroupInfo> = Vec::new();
5818 let mut group_index_map: std::collections::HashMap<String, usize> =
5819 std::collections::HashMap::new();
5820
5821 // Register the chain of groups "/a", "/a/b", … for the link-style
5822 // path `link_path` ("a/b"), taking each one's on-disk header block
5823 // and attributes out of `group_headers` when the link walk saw it.
5824 fn ensure_groups_for(
5825 link_path: &str,
5826 groups: &mut Vec<GroupInfo>,
5827 group_index_map: &mut std::collections::HashMap<String, usize>,
5828 group_headers: &mut std::collections::HashMap<String, GroupHeaderInfo>,
5829 ) {
5830 let mut path = String::new();
5831 for part in link_path.split('/') {
5832 let parent_path = if path.is_empty() {
5833 "/".to_string()
5834 } else {
5835 path.clone()
5836 };
5837 if path.is_empty() {
5838 path = format!("/{}", part);
5839 } else {
5840 path = format!("{}/{}", path, part);
5841 }
5842 if group_index_map.contains_key(&path) {
5843 continue;
5844 }
5845 let parent = if parent_path == "/" {
5846 None
5847 } else {
5848 group_index_map.get(&parent_path).copied()
5849 };
5850 let gidx = groups.len();
5851 let (obj_header_written_addr, obj_header_blocks, attributes, track_order, times) =
5852 group_headers.remove(path.trim_start_matches('/')).map_or(
5853 (None, Vec::new(), Vec::new(), TrackOrder::default(), None),
5854 |(addr, blocks, attrs, track, times)| {
5855 (Some(addr), blocks, attrs, track, times)
5856 },
5857 );
5858 groups.push(GroupInfo {
5859 name: path.clone(),
5860 parent,
5861 creation_seq: 0,
5862 track_order,
5863 times,
5864 child_datasets: Vec::new(),
5865 child_groups: Vec::new(),
5866 obj_header_addr: 0,
5867 obj_header_written_addr,
5868 obj_header_blocks,
5869 deleted: false,
5870 attributes,
5871 });
5872 if let Some(pidx) = parent {
5873 groups[pidx].child_groups.push(gidx);
5874 }
5875 group_index_map.insert(path.clone(), gidx);
5876 }
5877 }
5878
5879 // Every linked group, in link-walk order (parents precede children).
5880 for name in &walk_order {
5881 if group_headers.contains_key(name.as_str()) {
5882 ensure_groups_for(name, &mut groups, &mut group_index_map, &mut group_headers);
5883 }
5884 }
5885
5886 // Assign each dataset to its immediate parent group, creating any
5887 // group the link walk could not decode (its chain stays placeholder).
5888 for (di, ds) in existing_datasets.iter().enumerate() {
5889 let parts: Vec<&str> = ds.name.split('/').collect();
5890 if parts.len() <= 1 {
5891 continue; // root-level dataset, no group
5892 }
5893 let parent_link_path = parts[..parts.len() - 1].join("/");
5894 ensure_groups_for(
5895 &parent_link_path,
5896 &mut groups,
5897 &mut group_index_map,
5898 &mut group_headers,
5899 );
5900 let gidx = group_index_map[&format!("/{}", parent_link_path)];
5901 groups[gidx].child_datasets.push(di);
5902 }
5903
5904 // An object the rebuild above gave up on is preserved by its bytes,
5905 // so the other links to it are preserved too: there is no registry
5906 // entry for them to name.
5907 alias_entries.retain(|entry| match unrebuilt.get(&entry.address) {
5908 None => true,
5909 Some(why) => {
5910 preserved.push(PreservedEntry {
5911 path: entry.path.clone(),
5912 class: crate::io::reader::LinkClass::Hard,
5913 encoded: entry.encoded.clone(),
5914 reason: Some(why.clone()),
5915 kind: PreservedKind::Unclassified,
5916 });
5917 false
5918 }
5919 });
5920
5921 // The one thing a rebuilt shared-message table can break: an object
5922 // kept by its bytes keeps the heap IDs its header holds, and the
5923 // finalize gives the heap those IDs name back to the allocator. Every
5924 // object the registry holds is rewritten instead
5925 // ([`rebuilds_shared_messages`](Self::rebuilds_shared_messages)), so
5926 // this asks only the preserved ones, and names the object rather than
5927 // the feature — the file is appendable the moment nothing preserved
5928 // holds a heap ID or hides a subtree that might.
5929 if sohm.is_some() {
5930 for entry in &preserved {
5931 if !matches!(entry.class, crate::io::reader::LinkClass::Hard) {
5932 continue;
5933 }
5934 let Ok((link, _)) = LinkMessage::decode(&entry.encoded, &meta.ctx) else {
5935 continue;
5936 };
5937 let LinkTarget::Hard { address } = link.target else {
5938 continue;
5939 };
5940 if let Some(blocks) = crate::io::object_header_io::blocks_shared_message_rebuild(
5941 &mut handle,
5942 &meta,
5943 address,
5944 )? {
5945 let why = entry
5946 .reason
5947 .as_deref()
5948 .unwrap_or("this writer cannot model it");
5949 return Err(crate::io::IoError::Unsupported(format!(
5950 "cannot open this file for appending: '{}' {blocks}, but {why}, so \
5951 its header keeps the bytes it has while the append lays the \
5952 shared-message table out afresh",
5953 entry.path
5954 )));
5955 }
5956 }
5957 }
5958
5959 // Rebuild the hard-link registry from the alias entries set aside
5960 // above, so the H5Ldelete semantics survive a reopen. An alias whose
5961 // target the walk could not model is not here at all: it was
5962 // preserved by its own bytes, exactly as the first link to that
5963 // object was.
5964 let mut hard_links: Vec<HardLink> = Vec::new();
5965 for HardEntry {
5966 path,
5967 address: addr,
5968 ..
5969 } in alias_entries
5970 {
5971 let target = if let Some(di) = existing_datasets
5972 .iter()
5973 .position(|d| d.obj_header_addr == addr)
5974 {
5975 HardLinkTarget::Dataset(di)
5976 } else if let Some(gi) = groups
5977 .iter()
5978 .position(|g| g.obj_header_written_addr == Some(addr))
5979 {
5980 HardLinkTarget::Group(gi)
5981 } else {
5982 continue;
5983 };
5984 let (parent, link_name) = match path.rsplit_once('/') {
5985 None => (None, path),
5986 Some((dir, leaf)) => {
5987 ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
5988 (
5989 group_index_map.get(&format!("/{dir}")).copied(),
5990 leaf.to_string(),
5991 )
5992 }
5993 };
5994 hard_links.push(HardLink {
5995 parent,
5996 name: link_name,
5997 target,
5998 creation_seq: 0,
5999 });
6000 }
6001
6002 // Attach every link the writer cannot express to the group that
6003 // holds it, so the rewrite of that group's header emits it again.
6004 // `ensure_groups_for` registers the parent chain, which matters for
6005 // a group whose only content is such a link: nothing else would put
6006 // it in the registry, and the close would drop group and link alike.
6007 let mut preserved_links: Vec<PreservedLink> = Vec::new();
6008 for PreservedEntry {
6009 path,
6010 class,
6011 encoded,
6012 reason,
6013 kind,
6014 } in preserved
6015 {
6016 let (parent, link_name) = match path.rsplit_once('/') {
6017 None => (None, path),
6018 Some((dir, leaf)) => {
6019 ensure_groups_for(dir, &mut groups, &mut group_index_map, &mut group_headers);
6020 (
6021 group_index_map.get(&format!("/{dir}")).copied(),
6022 leaf.to_string(),
6023 )
6024 }
6025 };
6026 preserved_links.push(PreservedLink {
6027 parent,
6028 name: link_name,
6029 class,
6030 encoded,
6031 reason,
6032 kind,
6033 });
6034 }
6035
6036 // Stamp the creation sequence a reopened file cannot supply. Nothing
6037 // on disk says which link was made first unless the group tracked
6038 // creation order, and this reader does not carry that back out, so
6039 // discovery order is what there is: datasets, then groups, then the
6040 // hard links found beside them — the order the writer emitted links
6041 // in before it ordered them at all.
6042 let mut creation_seq = 0u64;
6043 for d in &mut existing_datasets {
6044 d.creation_seq = creation_seq;
6045 creation_seq += 1;
6046 }
6047 for g in &mut groups {
6048 g.creation_seq = creation_seq;
6049 creation_seq += 1;
6050 }
6051 for l in &mut hard_links {
6052 l.creation_seq = creation_seq;
6053 creation_seq += 1;
6054 }
6055
6056 // The strategy is the file's, not this session's: a paged file
6057 // allocates on its own page grid however it was opened, `persist`
6058 // deciding only whether the managers survive the close.
6059 let allocator = FileAllocator::with_policy(
6060 file_size,
6061 ext.file_space_info
6062 .as_ref()
6063 .map_or(free_space::SpacePolicy::Aggr, |info| {
6064 free_space::SpacePolicy::for_message(info)
6065 }),
6066 );
6067 // The sections the file's own managers recorded are free space, so
6068 // they are what this session allocates from first — `H5MF_alloc` asks
6069 // the free-space manager before it bumps the end of the file, and a
6070 // reopen that skipped this would grow a file that had room.
6071 allocator.reset_free_list(&reopened_free_space.sections);
6072
6073 // Now that every object has its registry index, key the dense storage
6074 // found on disk by the scope that will supersede it. A group the link
6075 // walk saw but never registered is not rewritten either, so leaving it
6076 // out is what keeps its storage referenced.
6077 let mut superseded = SupersededDense {
6078 attrs: dataset_dense
6079 .into_iter()
6080 .map(|(di, ainfo)| (AttrScope::Dataset(di), ainfo))
6081 .collect(),
6082 links: HashMap::new(),
6083 };
6084 superseded
6085 .attrs
6086 .extend(root_dense.attrs.map(|a| (AttrScope::Root, a)));
6087 superseded
6088 .links
6089 .extend(root_dense.links.map(|l| (LinkScope::Root, l)));
6090 for (name, dense) in group_dense {
6091 let Some(&gidx) = group_index_map.get(&format!("/{name}")) else {
6092 continue;
6093 };
6094 superseded
6095 .attrs
6096 .extend(dense.attrs.map(|a| (AttrScope::Group(gidx), a)));
6097 superseded
6098 .links
6099 .extend(dense.links.map(|l| (LinkScope::Group(gidx), l)));
6100 }
6101 let superseded = (!superseded.attrs.is_empty() || !superseded.links.is_empty())
6102 .then(|| Box::new(superseded));
6103
6104 // The same keying for the symbol-table storage. Built from the headers
6105 // alone, not from the superblock version: a group whose header carried
6106 // no Symbol Table message contributes nothing — what happens to a group
6107 // libhdf5 wrote at a newer bound inside an otherwise classic file — and
6108 // one that carried it keeps its storage even where the superblock is
6109 // version 2, which is what a file with shared messages is.
6110 let mut stabs: HashMap<LinkScope, StabExtents> = HashMap::new();
6111 stabs.extend(root_stab.map(|s| (LinkScope::Root, s)));
6112 for (name, extents) in group_stabs {
6113 if let Some(&gidx) = group_index_map.get(&format!("/{name}")) {
6114 stabs.insert(LinkScope::Group(gidx), extents);
6115 }
6116 }
6117 let symbol_tables = SymbolTables {
6118 found: stabs.keys().copied().collect(),
6119 superseded: Slot::new(stabs),
6120 written: Slot::new(HashMap::new()),
6121 };
6122
6123 // The superblock the close re-emits, and the generation every message
6124 // this session encodes belongs to.
6125 let legacy = legacy.map(|superblock| Box::new(LegacyFile { superblock }));
6126
6127 // Wrap the reconstructed plain vecs into the per-slot registry. The
6128 // reconstruction logic above runs single-threaded on local `Vec`s;
6129 // only the final hand-off needs the `Shared<Slot<_>>` shape.
6130 let datasets = existing_datasets
6131 .into_iter()
6132 .map(|i| Shared::new(DatasetCell::new(i)))
6133 .collect();
6134 let groups = groups
6135 .into_iter()
6136 .map(|g| Shared::new(Slot::new(g)))
6137 .collect();
6138
6139 let writer = Self {
6140 handle,
6141 allocator,
6142 ctx,
6143 datasets: Slot::new(datasets),
6144 groups: Slot::new(groups),
6145 hard_links: Slot::new(hard_links),
6146 // A reopen carries the soft and external links it found as
6147 // `preserved_links`, byte for byte; this list holds only the ones
6148 // created in this session.
6149 symbolic_links: Slot::new(Vec::new()),
6150 committed_datatypes: Slot::new(Vec::new()),
6151 preserved_links: Slot::new(preserved_links),
6152 name_index: Slot::new(Box::new(NameIndex::new())),
6153 root_attributes: Slot::new(root_attributes),
6154 create_lock: Slot::new(()),
6155 // A reopen names no bound, so the session appends at the same
6156 // default a create uses. `set_libver_bound` is where a caller
6157 // names one, exactly as `H5Fopen` takes a fapl.
6158 libver: None,
6159 closed: false,
6160 swmr_active: false,
6161 cwfs: Slot::new(Vec::new()),
6162 root_group_addr: None,
6163 superseded_root_header: root_header_blocks,
6164 // The version the file already has, written back unchanged.
6165 superblock_version: SuperblockVersion::Existing(version),
6166 // The reopened file's own policy, so objects added in this
6167 // session are made the way the file already declares.
6168 root_track_order,
6169 root_times,
6170 dense_attributes: Slot::new(HashMap::new()),
6171 dense_links: Slot::new(HashMap::new()),
6172 superseded_dense: Slot::new(superseded),
6173 track_order: root_track_order,
6174 // Not recovered from the file the way the creation-order policy
6175 // is: a version-1 header leaves no trace of whether the object was
6176 // tracking times, so there is nothing on disk to read the policy
6177 // back from. An object added to a reopened file gets this writer's
6178 // own default, the same one a created file starts at.
6179 track_times: false,
6180 next_creation_seq: Slot::new(creation_seq),
6181 pending_object_references: Slot::new(Vec::new()),
6182 pending_heap_references: Slot::new(Vec::new()),
6183 attribute_references: Slot::new(Vec::new()),
6184 legacy,
6185 symbol_tables,
6186 // The ranks the superblock or its extension declared, which every
6187 // v1-B-tree and symbol-table node this session writes is sized by.
6188 btree: meta.btree,
6189 extension,
6190 free_space: reopened_free_space.state,
6191 // The indexes the file was created with, and the blocks its
6192 // current table occupies; the next finalize lays a new table out
6193 // over them from the whole message set.
6194 sohm,
6195 source_dir: source_dir_of(path)?,
6196 };
6197 // The link graph is complete only now, so this is the first point the
6198 // count each on-disk header was written with can be read off it: in a
6199 // well-formed file the links the walk found reaching an object *are*
6200 // that count, so nothing has to be decoded out of the headers.
6201 for i in 0..writer.dataset_count() {
6202 let nlink = writer.object_link_count(HardLinkTarget::Dataset(i));
6203 writer.ds(i).lock().nlink_written = nlink;
6204 }
6205 Ok(writer)
6206 }
6207
6208 /// Return the names of all datasets created so far.
6209 pub fn dataset_names(&self) -> Vec<String> {
6210 self.dataset_refs()
6211 .iter()
6212 .filter_map(|d| {
6213 let g = d.lock();
6214 (!g.deleted).then(|| g.name.clone())
6215 })
6216 .collect()
6217 }
6218
6219 /// Find a dataset index by name. Like `H5Dopen`, the name may be any
6220 /// link path to the dataset: a user hard link's path — or a path
6221 /// whose group components pass through such links — resolves to its
6222 /// target.
6223 pub fn dataset_index(&self, name: &str) -> Option<usize> {
6224 let name = self.canonical_dataset_path(name);
6225 self.dataset_refs()
6226 .iter()
6227 .position(|d| {
6228 let g = d.lock();
6229 g.name == name && !g.deleted
6230 })
6231 .or_else(|| {
6232 self.hard_links_vec().iter().find_map(|l| match l.target {
6233 HardLinkTarget::Dataset(i)
6234 if self.hard_link_emitted(l) && self.hard_link_full_path(l) == name =>
6235 {
6236 Some(i)
6237 }
6238 _ => None,
6239 })
6240 })
6241 }
6242
6243 /// Reconstruct the fields a writer-mode `H5Dataset` handle needs for the
6244 /// dataset at `index`, and open it under `access`. Single owner of this
6245 /// mapping so `H5File::dataset_writer`, `H5Group::dataset_writer`, and
6246 /// the vlen-string helpers all agree — including on
6247 /// [`bind_efile_prefix`](Self::bind_efile_prefix), which no handle site
6248 /// can then forget to run.
6249 pub(crate) fn dataset_handle_parts(
6250 &self,
6251 index: usize,
6252 access: &DatasetAccess,
6253 ) -> IoResult<DatasetHandleParts> {
6254 let open = self.bind_efile_prefix(index, access)?;
6255 let ds = self.ds(index);
6256 let g = ds.lock();
6257 Ok(DatasetHandleParts {
6258 shape: g.dataspace.dims.iter().map(|&d| d as usize).collect(),
6259 element_size: g.datatype.element_size() as usize,
6260 chunk_index: g.chunk_index_kind(),
6261 open,
6262 })
6263 }
6264
6265 /// Put `access`'s external file prefix in force for the dataset at
6266 /// `index`, or join the open that already settled one.
6267 ///
6268 /// INVARIANT: every write of an externally stored dataset's raw bytes
6269 /// joins its slot names against the prefix an *open* settled, and this is
6270 /// the only place that settles one. `write_contiguous_bytes` reads it and
6271 /// nothing else writes it, so a write cannot resolve a prefix of its own
6272 /// and land bytes where a read under the same properties would not look
6273 /// for them.
6274 ///
6275 /// First open wins, and a joining open may not disagree: `H5D__open_name`
6276 /// compares its own expanded prefix against the open dataset's and fails
6277 /// when they differ (H5Dint.c:1533-1545). Measured under libhdf5 1.14.6
6278 /// and 2.0.0, with a dataset created through a dapl naming a directory
6279 /// and its handle still alive: a second open naming another directory is
6280 /// refused, one naming the same directory joins, one naming none is
6281 /// refused too, and with `HDF5_EXTFILE_PREFIX` set — which shadows every
6282 /// property, so all three expand alike — none of them is. Dropping every
6283 /// handle releases the answer and the next open settles it afresh, which
6284 /// the same measurement confirms.
6285 ///
6286 /// Returns the token that keeps the open alive, `None` for a dataset
6287 /// whose raw data is in this file and which therefore has no prefix to
6288 /// agree about.
6289 pub(crate) fn bind_efile_prefix(
6290 &self,
6291 index: usize,
6292 access: &DatasetAccess,
6293 ) -> IoResult<Option<crate::io::reader::DatasetOpenToken>> {
6294 let ds = self.ds(index);
6295 let mut g = ds.lock();
6296 let source_dir = &self.source_dir;
6297 let Some(ext) = g.external.as_mut() else {
6298 return Ok(None);
6299 };
6300 let want =
6301 crate::io::reader::resolve_extfile_prefix(access.efile_prefix_value(), source_dir);
6302 if let Some(open) = ext.prefix.open.upgrade() {
6303 if ext.prefix.expanded != want {
6304 let name = g.name.clone();
6305 return Err(crate::io::IoError::InvalidState(format!(
6306 "dataset {name:?} is already open under a different external file prefix, and libhdf5 refuses to join an open that disagrees about one"
6307 )));
6308 }
6309 return Ok(Some(open));
6310 }
6311 let token: crate::io::reader::DatasetOpenToken = std::sync::Arc::new(());
6312 ext.prefix = EfilePrefix {
6313 expanded: want,
6314 open: std::sync::Arc::downgrade(&token),
6315 };
6316 Ok(Some(token))
6317 }
6318
6319 /// Reject a name some other link in the file already occupies.
6320 ///
6321 /// `name` is the registry's full-path form, with no leading `/`. HDF5
6322 /// requires link names to be unique within their group, and every kind of
6323 /// link this writer can emit competes for the same name: a dataset's own
6324 /// link, a group's, a user hard link, a soft or external link, and a link
6325 /// a reopen is carrying through verbatim. This is the one place that list
6326 /// is written down, so a creator cannot be blind to a kind it does not
6327 /// itself make — nor a kind added after it.
6328 fn ensure_name_free(&self, name: &str) -> IoResult<()> {
6329 let holder = self.name_holder(name);
6330 // The index is a filter over the registries, not a second copy of
6331 // them, so a debug build re-derives the answer on every create: a
6332 // name it failed to record surfaces as a failing assertion in the
6333 // suite rather than as two links of one name in somebody's file.
6334 #[cfg(debug_assertions)]
6335 assert_eq!(
6336 holder,
6337 self.scan_name_holder(name),
6338 "the name index disagrees with the registries for '{name}'"
6339 );
6340 match holder {
6341 None => Ok(()),
6342 Some(kind) => Err(crate::io::IoError::InvalidState(format!(
6343 "a {kind} named '{name}' already exists"
6344 ))),
6345 }
6346 }
6347
6348 /// What already holds `name`, or `None` if it is free.
6349 ///
6350 /// The kinds answer in a fixed order — dataset, group, committed
6351 /// datatype, hard link, symbolic link, preserved link — because the
6352 /// refusal names the first one that holds it. [`NameIndex`] narrows each
6353 /// kind to the entries that ever took this name; every candidate is then
6354 /// put through the same predicate the full scan used, so a hit left
6355 /// behind by a delete or a rename answers exactly as an absent one does.
6356 fn name_holder(&self, name: &str) -> Option<&'static str> {
6357 self.build_name_index();
6358 let hits: Vec<NameHit> = {
6359 let index = self.name_index.lock();
6360 index.map.as_ref().and_then(|m| m.get(name))?.clone()
6361 };
6362 for hit in &hits {
6363 if let NameHit::Dataset(i) = *hit {
6364 let ds = self.ds(i);
6365 let d = ds.lock();
6366 if !d.deleted && d.name == name {
6367 return Some("dataset");
6368 }
6369 }
6370 }
6371 for hit in &hits {
6372 if let NameHit::Group(i) = *hit {
6373 let grp = self.grp(i);
6374 let g = grp.lock();
6375 if !g.deleted && g.name.trim_start_matches('/') == name {
6376 return Some("group");
6377 }
6378 }
6379 }
6380 for hit in &hits {
6381 if let NameHit::Datatype(i) = *hit {
6382 // The registry lock goes before `parent_alive` takes a group
6383 // slot, never across it.
6384 let (parent, held) = {
6385 let reg = self.committed_datatypes.lock();
6386 (reg[i].parent, reg[i].name == name)
6387 };
6388 if held && self.parent_alive(parent) {
6389 return Some("committed datatype");
6390 }
6391 }
6392 }
6393 if hits.contains(&NameHit::HardLink)
6394 && self
6395 .hard_links_vec()
6396 .iter()
6397 .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6398 {
6399 return Some("hard link");
6400 }
6401 if hits.contains(&NameHit::SymbolicLink)
6402 && self
6403 .symbolic_links_vec()
6404 .iter()
6405 .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6406 {
6407 return Some("link");
6408 }
6409 // A preserved link occupies its name in the group just as a modelled
6410 // one does; both are emitted, and two link messages of one name in a
6411 // group is an invalid file.
6412 if hits.contains(&NameHit::PreservedLink)
6413 && self.preserved_link_paths().iter().any(|(p, _)| *p == name)
6414 {
6415 return Some("link");
6416 }
6417 None
6418 }
6419
6420 /// The same answer read straight off the registries, which is what the
6421 /// index is checked against in a debug build.
6422 #[cfg(debug_assertions)]
6423 fn scan_name_holder(&self, name: &str) -> Option<&'static str> {
6424 if self.dataset_refs().iter().any(|d| {
6425 let g = d.lock();
6426 !g.deleted && g.name == name
6427 }) {
6428 return Some("dataset");
6429 }
6430 if self.group_refs().iter().any(|g| {
6431 let gg = g.lock();
6432 !gg.deleted && gg.name.trim_start_matches('/') == name
6433 }) {
6434 return Some("group");
6435 }
6436 if self
6437 .committed_datatypes_vec()
6438 .iter()
6439 .any(|c| self.parent_alive(c.parent) && c.name == name)
6440 {
6441 return Some("committed datatype");
6442 }
6443 if self
6444 .hard_links_vec()
6445 .iter()
6446 .any(|l| self.hard_link_emitted(l) && self.hard_link_full_path(l) == name)
6447 {
6448 return Some("hard link");
6449 }
6450 if self
6451 .symbolic_links_vec()
6452 .iter()
6453 .any(|l| self.symbolic_link_emitted(l) && self.symbolic_link_full_path(l) == name)
6454 {
6455 return Some("link");
6456 }
6457 if self.preserved_link_paths().iter().any(|(p, _)| *p == name) {
6458 return Some("link");
6459 }
6460 None
6461 }
6462
6463 /// Build the name index unless it is already built.
6464 ///
6465 /// The walk takes the registry spines and their slots, so it runs with no
6466 /// index lock held — the writer never holds one lock across another — and
6467 /// the result is kept only if nothing renamed, created or unlinked
6468 /// anything while it ran.
6469 fn build_name_index(&self) {
6470 let epoch = {
6471 let index = self.name_index.lock();
6472 if index.map.is_some() {
6473 return;
6474 }
6475 index.epoch
6476 };
6477 let mut map: HashMap<String, Vec<NameHit>> = HashMap::new();
6478 for (i, ds) in self.dataset_refs().iter().enumerate() {
6479 let d = ds.lock();
6480 if !d.deleted {
6481 map.entry(d.name.clone())
6482 .or_default()
6483 .push(NameHit::Dataset(i));
6484 }
6485 }
6486 for (i, grp) in self.group_refs().iter().enumerate() {
6487 let g = grp.lock();
6488 if !g.deleted {
6489 map.entry(g.name.trim_start_matches('/').to_string())
6490 .or_default()
6491 .push(NameHit::Group(i));
6492 }
6493 }
6494 for (i, c) in self.committed_datatypes_vec().iter().enumerate() {
6495 map.entry(c.name.clone())
6496 .or_default()
6497 .push(NameHit::Datatype(i));
6498 }
6499 for l in self.hard_links_vec().iter() {
6500 map.entry(self.hard_link_full_path(l))
6501 .or_default()
6502 .push(NameHit::HardLink);
6503 }
6504 for l in self.symbolic_links_vec().iter() {
6505 map.entry(self.symbolic_link_full_path(l))
6506 .or_default()
6507 .push(NameHit::SymbolicLink);
6508 }
6509 for (path, _) in self.preserved_link_paths() {
6510 map.entry(path).or_default().push(NameHit::PreservedLink);
6511 }
6512 let mut index = self.name_index.lock();
6513 if index.map.is_none() && index.epoch == epoch {
6514 index.map = Some(map);
6515 }
6516 }
6517
6518 /// Record that `hit` now holds `name` — the one way a new name enters the
6519 /// index, called from every push that gives a registry entry a name.
6520 fn register_name(&self, name: &str, hit: NameHit) {
6521 self.name_index.lock().insert(name, hit);
6522 }
6523
6524 /// Drop the index because something moved names wholesale (a group
6525 /// rename carries its subtree and every link path under it).
6526 fn forget_name_index(&self) {
6527 self.name_index.lock().forget();
6528 }
6529
6530 /// Delete a dataset name, with libhdf5's `H5Ldelete` semantics: a name
6531 /// is only a link. If `name` is a user hard link's path, just that
6532 /// link is removed and the object is untouched. If it is the tree name
6533 /// and a user hard link still names the object, the object survives
6534 /// under it — the link becomes the primary name and nothing is freed.
6535 /// Only deleting the *last* name soft-deletes the object and frees the
6536 /// file space it owned: its chunk blocks and chunk-index structures
6537 /// (or contiguous data block), the global-heap objects of its
6538 /// variable-length data and attributes, and — on a reopened file — the
6539 /// on-disk object header block. The freed space is reused by later
6540 /// allocations in this session; the file does not shrink.
6541 ///
6542 /// Refused while SWMR streaming is active: a live reader may hold any
6543 /// of those addresses (libhdf5 forbids link deletion during SWMR
6544 /// writes too).
6545 pub fn delete_dataset(&self, name: &str) -> IoResult<()> {
6546 if self.swmr_active {
6547 return Err(swmr_delete_error(name));
6548 }
6549 self.reject_external_traversal(name)?;
6550 // The gate keeps the link list and child lists still while this
6551 // delete reads and rewrites them (create_lock → op → slot order,
6552 // the same as every creator).
6553 let _create = self.create_lock.lock();
6554 // `H5Ldelete` resolves the path through links only *up to* the
6555 // leaf — the leaf is what gets deleted, so a leaf naming a user
6556 // link must stay literal and be unlinked, not its target.
6557 let name = match name.rsplit_once('/') {
6558 None => name.to_string(),
6559 Some((dir, leaf)) => format!(
6560 "{}/{leaf}",
6561 self.canonical_group_path(&format!("/{dir}"))
6562 .trim_start_matches('/')
6563 ),
6564 };
6565 let refs = self.dataset_refs();
6566 let idx = match refs.iter().position(|d| {
6567 let g = d.lock();
6568 g.name == name && !g.deleted
6569 }) {
6570 Some(i) => i,
6571 None => {
6572 // Not a tree name — the path may name a user hard link,
6573 // and deleting a link path unlinks just that link (the
6574 // creation collision checks keep the two namespaces
6575 // disjoint, so the order of the lookups cannot matter).
6576 let link = self.hard_links_vec().iter().position(|l| {
6577 self.hard_link_emitted(l)
6578 && matches!(l.target, HardLinkTarget::Dataset(_))
6579 && self.hard_link_full_path(l) == name
6580 });
6581 let Some(pos) = link else {
6582 return Err(crate::io::IoError::NotFound(name));
6583 };
6584 self.hard_links.lock().remove(pos);
6585 return Ok(());
6586 }
6587 };
6588 // A surviving hard link keeps the object: promote the first one to
6589 // the primary name and delete nothing.
6590 let promote = self.hard_links_vec().iter().position(|l| {
6591 self.hard_link_emitted(l) && matches!(l.target, HardLinkTarget::Dataset(i) if i == idx)
6592 });
6593 if let Some(pos) = promote {
6594 self.promote_dataset_to_link(idx, pos);
6595 return Ok(());
6596 }
6597 refs[idx].lock().deleted = true;
6598 // Remove from parent group's child_datasets
6599 for grp in self.group_refs() {
6600 grp.lock().child_datasets.retain(|&di| di != idx);
6601 }
6602 self.purge_dead_links();
6603 let ds = self.ds(idx);
6604 let _op = ds.op.lock();
6605 self.release_dataset_storage(idx)
6606 }
6607
6608 /// Soft-delete a group and all its child datasets and sub-groups,
6609 /// freeing every deleted object's file space the way
6610 /// [`delete_dataset`](Self::delete_dataset) does — with the same
6611 /// `H5Ldelete` semantics: a `name` that is a user hard link's path
6612 /// unlinks just that link, and hard links from *outside* the subtree
6613 /// keep their targets. A dataset or group such a link names survives,
6614 /// re-homed under the link (a group brings its whole subtree with
6615 /// it); a link naming the deleted group itself turns the call into a
6616 /// pure rename and nothing is freed. Refused while SWMR streaming is
6617 /// active, same rule as `delete_dataset`.
6618 pub fn delete_group(&self, name: &str) -> IoResult<()> {
6619 if self.swmr_active {
6620 return Err(swmr_delete_error(name));
6621 }
6622 self.reject_external_traversal(name)?;
6623 // Same gate as `delete_dataset`: the pre-scan below and the
6624 // promotions must see a still link list and child lists.
6625 let _create = self.create_lock.lock();
6626 let name = if name.starts_with('/') {
6627 name.to_string()
6628 } else {
6629 format!("/{}", name)
6630 };
6631 // Leaf stays literal, directory resolves through links — the
6632 // same `H5Ldelete` rule as `delete_dataset`.
6633 let name = match name.rsplit_once('/') {
6634 Some((dir, leaf)) if !dir.is_empty() => {
6635 format!("{}/{leaf}", self.canonical_group_path(dir))
6636 }
6637 _ => name,
6638 };
6639 let groups = self.group_refs();
6640 let gidx = match groups.iter().position(|g| {
6641 let gg = g.lock();
6642 gg.name == name && !gg.deleted
6643 }) {
6644 Some(i) => i,
6645 None => {
6646 // Same `H5Ldelete` rule as `delete_dataset`: a path naming
6647 // a user hard link to a group unlinks just that link.
6648 let trimmed = name.trim_start_matches('/');
6649 let link = self.hard_links_vec().iter().position(|l| {
6650 self.hard_link_emitted(l)
6651 && matches!(l.target, HardLinkTarget::Group(_))
6652 && self.hard_link_full_path(l) == trimmed
6653 });
6654 let Some(pos) = link else {
6655 return Err(crate::io::IoError::NotFound(name.clone()));
6656 };
6657 self.hard_links.lock().remove(pos);
6658 return Ok(());
6659 }
6660 };
6661
6662 // A link is "outside" when its parent group does not die with the
6663 // subtree; only outside links can keep their targets alive.
6664 fn outside(parent: Option<usize>, doomed_gs: &[usize]) -> bool {
6665 match parent {
6666 None => true,
6667 Some(pi) => !doomed_gs.contains(&pi),
6668 }
6669 }
6670 // A group an outside link names survives, re-homed with its whole
6671 // subtree under the link. Each promotion moves that subtree out of
6672 // the doomed set — and can turn a link inside it into an outside
6673 // one — so rescan from scratch until no promotable group is left.
6674 // Promoting `gidx` itself makes the delete a pure rename: return.
6675 let mut doomed_ds = Vec::new();
6676 let mut doomed_gs = Vec::new();
6677 loop {
6678 doomed_ds.clear();
6679 doomed_gs.clear();
6680 self.collect_live_subtree(gidx, &mut doomed_ds, &mut doomed_gs);
6681 let promote = self
6682 .hard_links_vec()
6683 .iter()
6684 .enumerate()
6685 .find_map(|(pos, l)| match l.target {
6686 HardLinkTarget::Group(gi)
6687 if self.hard_link_emitted(l)
6688 && outside(l.parent, &doomed_gs)
6689 && doomed_gs.contains(&gi) =>
6690 {
6691 Some((pos, gi))
6692 }
6693 _ => None,
6694 });
6695 let Some((pos, gi)) = promote else { break };
6696 self.promote_group_to_link(gi, pos);
6697 if gi == gidx {
6698 return Ok(());
6699 }
6700 }
6701 // A dataset an outside link names survives its container: re-home
6702 // it under the link now, so the marking pass below never sees it.
6703 for di in doomed_ds {
6704 let promote = self.hard_links_vec().iter().position(|l| {
6705 self.hard_link_emitted(l)
6706 && outside(l.parent, &doomed_gs)
6707 && matches!(l.target, HardLinkTarget::Dataset(i) if i == di)
6708 });
6709 if let Some(pos) = promote {
6710 self.promote_dataset_to_link(di, pos);
6711 }
6712 }
6713
6714 let mut ds_deleted = Vec::new();
6715 let mut gs_deleted = Vec::new();
6716 self.delete_group_recursive(gidx, &mut ds_deleted, &mut gs_deleted);
6717 // Remove from parent's child_groups
6718 let parent = groups[gidx].lock().parent;
6719 if let Some(pidx) = parent {
6720 groups[pidx].lock().child_groups.retain(|&gi| gi != gidx);
6721 }
6722 self.purge_dead_links();
6723 // Free storage only after the whole subtree is marked: the lists
6724 // hold each object exactly once (the marking pass skips anything
6725 // already deleted), so nothing is freed twice.
6726 for di in ds_deleted {
6727 let ds = self.ds(di);
6728 let _op = ds.op.lock();
6729 self.release_dataset_storage(di)?;
6730 }
6731 for gi in gs_deleted {
6732 self.release_group_storage(gi)?;
6733 }
6734 Ok(())
6735 }
6736
6737 /// Collect the live (not soft-deleted) members of `gidx`'s subtree,
6738 /// each exactly once, without changing anything — the read-only twin
6739 /// of [`delete_group_recursive`](Self::delete_group_recursive), for
6740 /// the pre-scan that must run before any marking.
6741 fn collect_live_subtree(&self, gidx: usize, ds_out: &mut Vec<usize>, gs_out: &mut Vec<usize>) {
6742 if gs_out.contains(&gidx) {
6743 return;
6744 }
6745 let (child_ds, child_gs) = {
6746 let grp = self.grp(gidx);
6747 let g = grp.lock();
6748 if g.deleted {
6749 return;
6750 }
6751 (g.child_datasets.clone(), g.child_groups.clone())
6752 };
6753 gs_out.push(gidx);
6754 for di in child_ds {
6755 if !self.ds(di).lock().deleted && !ds_out.contains(&di) {
6756 ds_out.push(di);
6757 }
6758 }
6759 for gi in child_gs {
6760 self.collect_live_subtree(gi, ds_out, gs_out);
6761 }
6762 }
6763
6764 /// Re-home dataset `idx` under the hard link at `pos` in the link
6765 /// list — the surviving half of `H5Ldelete`: the link leaves the user
6766 /// list and becomes the dataset's primary (tree) name, in the link's
6767 /// parent group. Storage is untouched; any further links to the
6768 /// dataset stay in the list and keep resolving.
6769 fn promote_dataset_to_link(&self, idx: usize, pos: usize) {
6770 let link = self.hard_links.lock().remove(pos);
6771 let new_name = self.hard_link_full_path(&link);
6772 for grp in self.group_refs() {
6773 grp.lock().child_datasets.retain(|&di| di != idx);
6774 }
6775 if let Some(pi) = link.parent {
6776 self.grp(pi).lock().child_datasets.push(idx);
6777 }
6778 self.ds(idx).lock().name = new_name.clone();
6779 self.register_name(&new_name, NameHit::Dataset(idx));
6780 }
6781
6782 /// The group counterpart of
6783 /// [`promote_dataset_to_link`](Self::promote_dataset_to_link): re-home
6784 /// group `gidx` under the hard link at `pos`, bringing its whole
6785 /// subtree with it. Names are stored as full paths, so every live
6786 /// descendant is renamed by prefix.
6787 fn promote_group_to_link(&self, gidx: usize, pos: usize) {
6788 let link = self.hard_links.lock().remove(pos);
6789 let new_name = format!("/{}", self.hard_link_full_path(&link));
6790 let old_name = self.grp(gidx).lock().name.clone();
6791 for grp in self.group_refs() {
6792 grp.lock().child_groups.retain(|&g| g != gidx);
6793 }
6794 {
6795 let grp = self.grp(gidx);
6796 let mut g = grp.lock();
6797 g.parent = link.parent;
6798 g.name = new_name.clone();
6799 }
6800 if let Some(pi) = link.parent {
6801 self.grp(pi).lock().child_groups.push(gidx);
6802 }
6803
6804 let mut ds_in = Vec::new();
6805 let mut gs_in = Vec::new();
6806 self.collect_live_subtree(gidx, &mut ds_in, &mut gs_in);
6807 // Group names carry a leading '/' ("/a/b"), dataset names none
6808 // ("a/b/ds") — two prefix forms of the same rename.
6809 let old_grp_prefix = format!("{old_name}/");
6810 let new_grp_prefix = format!("{new_name}/");
6811 let old_ds_prefix = old_grp_prefix.trim_start_matches('/').to_string();
6812 let new_ds_prefix = new_grp_prefix.trim_start_matches('/').to_string();
6813 for gi in gs_in {
6814 if gi == gidx {
6815 continue;
6816 }
6817 let grp = self.grp(gi);
6818 let mut g = grp.lock();
6819 let renamed = g
6820 .name
6821 .strip_prefix(&old_grp_prefix)
6822 .map(|rest| format!("{new_grp_prefix}{rest}"));
6823 if let Some(n) = renamed {
6824 g.name = n;
6825 }
6826 }
6827 for di in ds_in {
6828 let ds = self.ds(di);
6829 let mut d = ds.lock();
6830 let renamed = d
6831 .name
6832 .strip_prefix(&old_ds_prefix)
6833 .map(|rest| format!("{new_ds_prefix}{rest}"));
6834 if let Some(n) = renamed {
6835 d.name = n;
6836 }
6837 }
6838 // A group carries its subtree and every link path under it, so far
6839 // more names moved than this function can enumerate: start over.
6840 self.forget_name_index();
6841 }
6842
6843 /// Drop link entries that can no longer be emitted — their parent group
6844 /// or, for a hard link, their target object was just deleted — so the
6845 /// lists mirror what the file will hold instead of carrying suppressed
6846 /// zombies. Both kinds are purged here so a delete cannot clear one list
6847 /// and leave the other holding a name in a group that is gone.
6848 fn purge_dead_links(&self) {
6849 let dead: Vec<usize> = self
6850 .hard_links_vec()
6851 .iter()
6852 .enumerate()
6853 .filter(|(_, l)| !self.hard_link_emitted(l))
6854 .map(|(p, _)| p)
6855 .collect();
6856 let mut links = self.hard_links.lock();
6857 for p in dead.into_iter().rev() {
6858 links.remove(p);
6859 }
6860 drop(links);
6861
6862 let dead: Vec<usize> = self
6863 .symbolic_links_vec()
6864 .iter()
6865 .enumerate()
6866 .filter(|(_, l)| !self.symbolic_link_emitted(l))
6867 .map(|(p, _)| p)
6868 .collect();
6869 let mut links = self.symbolic_links.lock();
6870 for p in dead.into_iter().rev() {
6871 links.remove(p);
6872 }
6873 }
6874
6875 /// Mark `gidx` and its subtree deleted, appending each newly-deleted
6876 /// object's index to `ds_out` / `gs_out` exactly once — the caller
6877 /// frees their storage, and an object reachable twice (or a subtree
6878 /// already deleted) must not be freed twice.
6879 fn delete_group_recursive(
6880 &self,
6881 gidx: usize,
6882 ds_out: &mut Vec<usize>,
6883 gs_out: &mut Vec<usize>,
6884 ) {
6885 // Mark deleted and snapshot the child lists, releasing the group lock
6886 // before locking any dataset/child-group slot (spine → slot order).
6887 let (child_ds, child_gs) = {
6888 let grp = self.grp(gidx);
6889 let mut g = grp.lock();
6890 if g.deleted {
6891 return;
6892 }
6893 g.deleted = true;
6894 (g.child_datasets.clone(), g.child_groups.clone())
6895 };
6896 gs_out.push(gidx);
6897 for di in child_ds {
6898 let ds = self.ds(di);
6899 let mut d = ds.lock();
6900 if !d.deleted {
6901 d.deleted = true;
6902 ds_out.push(di);
6903 }
6904 }
6905 for gi in child_gs {
6906 self.delete_group_recursive(gi, ds_out, gs_out);
6907 }
6908 }
6909
6910 /// Free everything a soft-deleted dataset owned. The single owner of
6911 /// delete-time reclamation, called only from the two delete paths with
6912 /// the dataset already marked deleted and its op lock held.
6913 ///
6914 /// A deleted dataset contributes nothing to finalize (the header,
6915 /// index-flush and append-flush loops all skip it), so nothing in the
6916 /// finalized file can reference the blocks freed here. Never runs under
6917 /// SWMR — the delete entry points refuse first.
6918 fn release_dataset_storage(&self, index: usize) -> IoResult<()> {
6919 use crate::format::messages::datatype::DatatypeMessage;
6920 let (indexed, ndims, contiguous, is_vlen, attrs, header_blocks, mapping_list) = {
6921 let ds = self.ds(index);
6922 let mut m = ds.lock();
6923 // Buffered rows were never written to a chunk; they die with
6924 // the dataset instead of being flushed at close.
6925 m.append = None;
6926 let indexed = m.is_chunked();
6927 let contiguous = (!indexed && m.data_addr != UNDEF_ADDR && m.data_size > 0)
6928 .then_some((m.data_addr, m.data_size));
6929 m.data_addr = UNDEF_ADDR;
6930 m.data_size = 0;
6931 // The external files themselves are the application's, not this
6932 // file's, and neither is the name heap freed: `H5O_MSG_EFL`
6933 // installs no file-delete method, so libhdf5 leaves the heap block
6934 // behind too. Dropping the list is what stops a deleted dataset
6935 // still claiming storage.
6936 m.external = None;
6937 // The mapping list is this file's own metadata, so unlike the
6938 // external files above it *is* freed — `H5D__virtual_delete`
6939 // removes the heap object. The source datasets it named are
6940 // another file's and are left alone.
6941 let mapping_list = m
6942 .virtual_storage
6943 .take()
6944 .and_then(|v| u16::try_from(v.heap_index).ok().map(|i| (v.heap_addr, i)));
6945 let is_vlen = matches!(
6946 m.datatype,
6947 DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
6948 );
6949 let attrs = std::mem::take(&mut m.attributes);
6950 m.obj_header_written_addr = None;
6951 let header_blocks = std::mem::take(&mut m.obj_header_blocks);
6952 (
6953 indexed,
6954 m.dataspace.dims.len(),
6955 contiguous,
6956 is_vlen,
6957 attrs,
6958 header_blocks,
6959 mapping_list,
6960 )
6961 };
6962 if let Some((addr, idx)) = mapping_list {
6963 self.remove_heap_objects([(addr, vec![idx])].into_iter().collect())?;
6964 }
6965 if indexed {
6966 // Prune to a zero extent: every stored chunk is entirely beyond
6967 // it, so the walk frees each chunk block and collects the vlen
6968 // references its bytes held (released inside).
6969 self.prune_chunks_beyond(index, &vec![0; ndims])?;
6970 self.free_chunk_index(index)?;
6971 } else if let Some((addr, size)) = contiguous {
6972 if is_vlen {
6973 let data = self.handle.read_at(addr, size as usize)?;
6974 self.release_vlen_references(&data)?;
6975 }
6976 self.allocator.free(addr, size, FreeSpaceClass::RawData);
6977 }
6978 for attr in &attrs {
6979 self.release_attr_vlen(attr)?;
6980 }
6981 self.release_superseded_dense_attrs(AttrScope::Dataset(index))?;
6982 for (addr, size) in header_blocks {
6983 self.allocator.free(addr, size, FreeSpaceClass::Metadata);
6984 }
6985 Ok(())
6986 }
6987
6988 /// Free a deleted group's file space: its attributes' global-heap
6989 /// objects and, on a reopened file, the on-disk header block. The
6990 /// group counterpart of
6991 /// [`release_dataset_storage`](Self::release_dataset_storage).
6992 fn release_group_storage(&self, gidx: usize) -> IoResult<()> {
6993 let (attrs, header_blocks) = {
6994 let grp = self.grp(gidx);
6995 let mut g = grp.lock();
6996 let attrs = std::mem::take(&mut g.attributes);
6997 g.obj_header_written_addr = None;
6998 (attrs, std::mem::take(&mut g.obj_header_blocks))
6999 };
7000 for attr in &attrs {
7001 self.release_attr_vlen(attr)?;
7002 }
7003 self.release_superseded_dense_attrs(AttrScope::Group(gidx))?;
7004 self.release_superseded_dense_links(LinkScope::Group(gidx))?;
7005 for (addr, size) in header_blocks {
7006 self.allocator.free(addr, size, FreeSpaceClass::Metadata);
7007 }
7008 Ok(())
7009 }
7010
7011 /// Free the dense attribute storage a reopened header names, once, when
7012 /// this session stops naming it — because the header is being rewritten
7013 /// around fresh storage, or because the object was deleted.
7014 ///
7015 /// The single owner of that transition: nothing else removes an attribute
7016 /// entry from [`superseded_dense`](Self::superseded_dense), and this
7017 /// removes it as it frees, so no heap is freed twice or left half freed.
7018 /// An object whose storage was compact, or whose header this session
7019 /// keeps, has no entry and nothing happens.
7020 ///
7021 /// Never under SWMR: a live reader may still be walking the storage the
7022 /// published headers name, the same rule the superseded-header and
7023 /// relocated-chunk paths follow. The entry stays in place, unfreed.
7024 fn release_superseded_dense_attrs(&self, scope: AttrScope) -> IoResult<()> {
7025 if self.swmr_active {
7026 return Ok(());
7027 }
7028 let taken = self
7029 .superseded_dense
7030 .lock()
7031 .as_mut()
7032 .and_then(|s| s.attrs.remove(&scope));
7033 let Some(ainfo) = taken else {
7034 return Ok(());
7035 };
7036 self.release_dense_storage(
7037 ainfo.fractal_heap_address,
7038 ainfo.name_btree_address,
7039 ainfo.creation_order_btree_address,
7040 )
7041 }
7042
7043 /// The link counterpart of
7044 /// [`release_superseded_dense_attrs`](Self::release_superseded_dense_attrs),
7045 /// under the same invariant and the same SWMR rule. Split from it because
7046 /// the two are superseded at different points of a finalize: attribute
7047 /// storage before the object headers are laid out, link storage after
7048 /// every one of them has an address.
7049 fn release_superseded_dense_links(&self, scope: LinkScope) -> IoResult<()> {
7050 if self.swmr_active {
7051 return Ok(());
7052 }
7053 let taken = self
7054 .superseded_dense
7055 .lock()
7056 .as_mut()
7057 .and_then(|s| s.links.remove(&scope));
7058 let Some(linfo) = taken else {
7059 return Ok(());
7060 };
7061 self.release_dense_storage(
7062 linfo.fractal_heap_address,
7063 linfo.name_btree_address,
7064 linfo.creation_order_btree_address,
7065 )
7066 }
7067
7068 /// Return one dense storage's file space to the allocator: the fractal
7069 /// heap in full, its name index, and the creation-order index when the
7070 /// object had one.
7071 ///
7072 /// The extents come from walking the structures themselves rather than
7073 /// from re-deriving what a writer would have allocated, so storage
7074 /// libhdf5 laid out is freed as accurately as storage this crate wrote.
7075 /// Every walk here already ran once this session — the reopen read every
7076 /// attribute out of this heap through the same index — so a failure means
7077 /// the file changed underneath us, and surfacing it beats freeing a
7078 /// partial extent list.
7079 fn release_dense_storage(
7080 &self,
7081 heap_addr: u64,
7082 name_bt2_addr: u64,
7083 corder_bt2_addr: Option<u64>,
7084 ) -> IoResult<()> {
7085 use crate::format::chunk_index::btree_v2::collect_btree_v2_extents;
7086 use crate::format::fractal_heap::collect_heap_extents;
7087
7088 let mut reader = crate::io::reader::HandleBlockReader {
7089 handle: &self.handle,
7090 };
7091 let mut extents = Vec::new();
7092 if heap_addr != UNDEF_ADDR {
7093 extents.extend(collect_heap_extents(heap_addr, &self.ctx, &mut reader)?);
7094 }
7095 for addr in [Some(name_bt2_addr), corder_bt2_addr]
7096 .into_iter()
7097 .flatten()
7098 .filter(|&a| a != UNDEF_ADDR)
7099 {
7100 extents.extend(collect_btree_v2_extents(addr, &self.ctx, &mut reader)?);
7101 }
7102 for (addr, len) in extents {
7103 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
7104 }
7105 Ok(())
7106 }
7107
7108 /// Free a deleted dataset's chunk-index structures, after the chunks
7109 /// themselves were freed by a zero-extent prune. Takes the index info
7110 /// out of the slot, so the dataset no longer claims chunked storage.
7111 ///
7112 /// Every block's size is recovered the way its allocation computed it:
7113 /// re-encoding the in-memory copy (EA header and index block, FA
7114 /// header and data block, BT2 header) or sizing a same-shape dummy
7115 /// from the array geometry (EA data blocks, whose element counts come
7116 /// from [`EaGeometry`]; BT2 nodes are all `node_size`).
7117 fn free_chunk_index(&self, index: usize) -> IoResult<()> {
7118 let ds = self.ds(index);
7119 let mut m = ds.lock();
7120 let is_filtered = m.filter_pipeline.is_some();
7121 if let Some(c) = m.chunked.take() {
7122 let p = &c.earray_params;
7123 let bits = p.max_nelmts_bits;
7124 let csl = c.chunk_size_len;
7125 let geo = EaGeometry::new(
7126 p.idx_blk_elmts,
7127 p.data_blk_min_elmts,
7128 p.sup_blk_min_data_ptrs,
7129 bits,
7130 p.max_dblk_page_nelmts_bits,
7131 )?;
7132 let dblk_size = |nelmts: u64| -> u64 {
7133 if is_filtered {
7134 FilteredDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7135 .encode(&self.ctx, bits, csl)
7136 .len() as u64
7137 } else {
7138 ExtensibleArrayDataBlock::new(c.ea_header_addr, 0, nelmts as usize)
7139 .encoded_size(&self.ctx, bits) as u64
7140 }
7141 };
7142 let (dblk_addrs, sblk_addrs, iblk_size) = if is_filtered {
7143 let f = c.filt_iblk.as_ref().unwrap();
7144 (
7145 f.dblk_addrs.clone(),
7146 f.sblk_addrs.clone(),
7147 f.encode(&self.ctx, csl).len() as u64,
7148 )
7149 } else {
7150 (
7151 c.ea_iblk.dblk_addrs.clone(),
7152 c.ea_iblk.sblk_addrs.clone(),
7153 c.ea_iblk.encoded_size(&self.ctx) as u64,
7154 )
7155 };
7156 // Data blocks addressed from the index block belong to the
7157 // first `iblock_nsblks` super blocks; each of those defines the
7158 // element count (and so the disk size) of its data blocks.
7159 let mut g = 0usize;
7160 'direct: for s in geo.sblk.iter().take(geo.iblock_nsblks) {
7161 for _ in 0..s.ndblks {
7162 let Some(&a) = dblk_addrs.get(g) else {
7163 break 'direct;
7164 };
7165 g += 1;
7166 if a == UNDEF_ADDR {
7167 continue;
7168 }
7169 if s.dblk_nelmts > geo.dblk_page_nelmts {
7170 return Err(crate::io::IoError::InvalidState(
7171 "cannot free a paged extensible-array data block, \
7172 which is not yet supported"
7173 .into(),
7174 ));
7175 }
7176 self.allocator
7177 .free(a, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7178 }
7179 }
7180 for (off, &sa) in sblk_addrs.iter().enumerate() {
7181 if sa == UNDEF_ADDR {
7182 continue;
7183 }
7184 let s = geo.sblk[geo.iblock_nsblks + off];
7185 if s.dblk_nelmts > geo.dblk_page_nelmts {
7186 return Err(crate::io::IoError::InvalidState(
7187 "cannot free a paged extensible-array data block, \
7188 which is not yet supported"
7189 .into(),
7190 ));
7191 }
7192 let buf = self.handle.read_at_most(sa, 65536)?;
7193 let sb =
7194 ExtensibleArraySuperBlock::decode(&buf, &self.ctx, bits, s.ndblks as usize, 0)?;
7195 for &da in &sb.dblk_addrs {
7196 if da != UNDEF_ADDR {
7197 self.allocator
7198 .free(da, dblk_size(s.dblk_nelmts), FreeSpaceClass::Metadata);
7199 }
7200 }
7201 self.allocator.free(
7202 sa,
7203 sb.encode(&self.ctx, bits).len() as u64,
7204 FreeSpaceClass::Metadata,
7205 );
7206 }
7207 self.allocator
7208 .free(c.ea_iblk_addr, iblk_size, FreeSpaceClass::Metadata);
7209 self.allocator.free(
7210 c.ea_header_addr,
7211 c.ea_header.encoded_size(&self.ctx) as u64,
7212 FreeSpaceClass::Metadata,
7213 );
7214 return Ok(());
7215 }
7216 if let Some(fa) = m.fixed_array.take() {
7217 self.allocator.free(
7218 fa.fa_dblk_addr,
7219 fixed_array_dblk_disk_size(&self.ctx, &fa.fa_header),
7220 FreeSpaceClass::Metadata,
7221 );
7222 self.allocator.free(
7223 fa.fa_header_addr,
7224 fa.fa_header.encode(&self.ctx).len() as u64,
7225 FreeSpaceClass::Metadata,
7226 );
7227 return Ok(());
7228 }
7229 // The implicit index has no structure to free, only the one run of
7230 // chunk space it was given at create — which is the whole of its
7231 // storage, so nothing else can be leaked or double-freed here.
7232 if let Some(imp) = m.implicit.take() {
7233 self.allocator
7234 .free(imp.data_addr, imp.data_size, FreeSpaceClass::RawData);
7235 return Ok(());
7236 }
7237 // The single-chunk index has no structure of its own either: its one
7238 // chunk is the whole of its storage, addressed directly from the
7239 // layout message rather than any index this function's doc comment's
7240 // "chunks already freed by a zero-extent prune" applies to — so
7241 // freeing it here, if it was ever allocated, is the only place it
7242 // happens.
7243 if let Some(sc) = m.single_chunk.take() {
7244 if sc.data_addr != UNDEF_ADDR {
7245 let len = if is_filtered { sc.nbytes } else { sc.data_size };
7246 self.allocator
7247 .free(sc.data_addr, len, FreeSpaceClass::RawData);
7248 }
7249 return Ok(());
7250 }
7251 // The version-1 B-tree owns nothing but its node blocks: the header
7252 // every other index has is, here, the root pointer inside the layout
7253 // message.
7254 if let Some(bt1) = m.btree_v1.take() {
7255 let element_size = m.datatype.element_size() as u64;
7256 let node_size = bt1
7257 .build_tree(element_size, self.ctx.sizeof_addr as usize)
7258 .node_size();
7259 for &a in &bt1.node_addrs {
7260 self.allocator
7261 .free(a, node_size as u64, FreeSpaceClass::Metadata);
7262 }
7263 return Ok(());
7264 }
7265 if let Some(bt2) = m.btree_v2.take() {
7266 let tree = bt2.index.build_tree(&self.ctx);
7267 for &a in &bt2.node_addrs {
7268 self.allocator
7269 .free(a, tree.node_size as u64, FreeSpaceClass::Metadata);
7270 }
7271 self.allocator.free(
7272 bt2.bt2_header_addr,
7273 tree.header(UNDEF_ADDR).encode(&self.ctx).len() as u64,
7274 FreeSpaceClass::Metadata,
7275 );
7276 }
7277 Ok(())
7278 }
7279
7280 /// Return the chunk dimensions for a dataset, if chunked.
7281 ///
7282 /// Returns an owned `Vec` because the chunk geometry now lives behind the
7283 /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7284 pub fn dataset_chunk_dims(&self, index: usize) -> Option<Vec<u64>> {
7285 let ds = self.ds(index);
7286 let m = ds.lock();
7287 m.chunk_index_kind().map(|kind| match kind {
7288 ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
7289 ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
7290 ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
7291 ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
7292 ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
7293 ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
7294 })
7295 }
7296
7297 /// Return the current dimensions of a dataset.
7298 ///
7299 /// Returns an owned `Vec` because the dataspace now lives behind the
7300 /// per-dataset [`Slot`]; it cannot be borrowed past the guard.
7301 pub fn dataset_dims(&self, index: usize) -> Vec<u64> {
7302 self.ds(index).lock().dataspace.dims.clone()
7303 }
7304
7305 /// Return the maximum extent a dataset declares, per dimension.
7306 ///
7307 /// An absent maximum shape means the shape is fixed at its current extent
7308 /// (libhdf5 defaults maxdims to dims at creation), so the current
7309 /// dimensions are returned; `H5S_UNLIMITED` is `u64::MAX`.
7310 pub fn dataset_max_dims(&self, index: usize) -> Vec<u64> {
7311 let ds = self.ds(index);
7312 let m = ds.lock();
7313 m.dataspace
7314 .max_dims
7315 .clone()
7316 .unwrap_or_else(|| m.dataspace.dims.clone())
7317 }
7318
7319 /// Whether a dataset stores its raw data through a filter pipeline.
7320 ///
7321 /// The write paths ask before choosing how to hand a chunk over: an
7322 /// unfiltered chunk's bytes go to the file exactly as the caller holds
7323 /// them, while a filtered one has to be compressed first.
7324 pub(crate) fn dataset_is_filtered(&self, index: usize) -> bool {
7325 self.ds(index).lock().filter_pipeline.is_some()
7326 }
7327
7328 /// Return the datatype a dataset declares on disk.
7329 ///
7330 /// The typed write paths need it to store bytes in the declared byte
7331 /// order; a reopened dataset handle has no copy of its own, and a cached
7332 /// one could disagree with what the header will say.
7333 pub fn dataset_datatype(&self, index: usize) -> DatatypeMessage {
7334 self.ds(index).lock().datatype.clone()
7335 }
7336
7337 /// Create a group in the file hierarchy.
7338 ///
7339 /// `parent_path` is the full path of the parent group (e.g., "/" for root).
7340 /// `name` is the name of the new group (e.g., "detector").
7341 ///
7342 /// Returns the group index in the writer's group list.
7343 pub fn create_group(&self, parent_path: &str, name: &str) -> IoResult<usize> {
7344 // Hold the create gate across the uniqueness check and the registry
7345 // push so the two are atomic (see `create_lock`).
7346 let _create = self.create_lock.lock();
7347 // A parent path through hard links creates in the link's target,
7348 // as HDF5 traversal does.
7349 let parent_path = self.canonical_group_path(parent_path);
7350 let parent_path = parent_path.as_str();
7351 let full_name = if parent_path == "/" {
7352 format!("/{}", name)
7353 } else {
7354 format!("{}/{}", parent_path, name)
7355 };
7356 // Same rule as dataset creation: a path through a carried external
7357 // link names a group in the other file, which this writer cannot make.
7358 self.reject_external_traversal(&full_name)?;
7359 // `name` may itself carry path components; resolving the whole thing
7360 // is what keeps a '/' out of the link this group will be reached by.
7361 let (parent_idx, _leaf) = self.split_parent(full_name.trim_start_matches('/'))?;
7362
7363 self.ensure_name_free(full_name.trim_start_matches('/'))?;
7364
7365 let group_idx = self.push_group(GroupInfo {
7366 name: full_name,
7367 parent: parent_idx,
7368 creation_seq: self.take_creation_seq(),
7369 track_order: self.track_order,
7370 times: self.created_object_times(),
7371 child_datasets: Vec::new(),
7372 child_groups: Vec::new(),
7373 obj_header_addr: 0,
7374 obj_header_written_addr: None,
7375 obj_header_blocks: Vec::new(),
7376 deleted: false,
7377 attributes: Vec::new(),
7378 });
7379
7380 // Register this group as a child of its parent
7381 if let Some(pidx) = parent_idx {
7382 self.grp(pidx).lock().child_groups.push(group_idx);
7383 }
7384
7385 Ok(group_idx)
7386 }
7387
7388 /// Register a dataset as belonging to a group.
7389 ///
7390 /// `group_path` is the full path of the group (e.g., "/detector").
7391 /// `ds_index` is the dataset index returned by `create_dataset`.
7392 pub fn assign_dataset_to_group(&self, group_path: &str, ds_index: usize) -> IoResult<()> {
7393 let group_path = self.canonical_group_path(group_path);
7394 let group_path = group_path.as_str();
7395 let groups = self.group_refs();
7396 let group_idx = groups
7397 .iter()
7398 .position(|g| {
7399 let gg = g.lock();
7400 gg.name == group_path && !gg.deleted
7401 })
7402 .ok_or_else(|| {
7403 crate::io::IoError::NotFound(format!("group '{}' not found", group_path))
7404 })?;
7405 // A move, not an addition: the create gate has already placed every
7406 // dataset from the path components of its name, so appending here
7407 // would leave one dataset linked from two groups at once.
7408 for g in &groups {
7409 g.lock().child_datasets.retain(|&d| d != ds_index);
7410 }
7411 groups[group_idx].lock().child_datasets.push(ds_index);
7412 Ok(())
7413 }
7414
7415 /// Create a hard link: an additional name for an object that already
7416 /// exists in the file.
7417 ///
7418 /// No data is copied — the link and its target share one object header,
7419 /// exactly as `h5py` / libhdf5 hard links do.
7420 ///
7421 /// * `parent_group_path` — full path of the group that will hold the
7422 /// link (`"/"` for the root group).
7423 /// * `link_name` — leaf name of the new link within that group.
7424 /// * `target_path` — full path of an existing dataset or group, with or
7425 /// without a leading `/`.
7426 pub fn create_hard_link(
7427 &self,
7428 parent_group_path: &str,
7429 link_name: &str,
7430 target_path: &str,
7431 ) -> IoResult<()> {
7432 if link_name.is_empty() || link_name.contains('/') {
7433 return Err(crate::io::IoError::InvalidState(format!(
7434 "hard link name '{link_name}' must be a non-empty leaf name"
7435 )));
7436 }
7437
7438 // Neither end may sit across a carried external link: the target
7439 // would be an object in the other file, and the link itself would be
7440 // a name in a group this writer does not own.
7441 self.reject_external_traversal(target_path)?;
7442 self.reject_external_traversal(&format!(
7443 "{}/{link_name}",
7444 parent_group_path.trim_end_matches('/')
7445 ))?;
7446
7447 // Hold the create gate across the collision check and the hard-link
7448 // push so the two are atomic (see `create_lock`).
7449 let _create = self.create_lock.lock();
7450 // Both paths resolve through hard links, as HDF5 traversal does.
7451 let parent_group_path = self.canonical_group_path(parent_group_path);
7452 let parent_group_path = parent_group_path.as_str();
7453
7454 // Resolve the parent group (None == root).
7455 let parent = if parent_group_path == "/" {
7456 None
7457 } else {
7458 Some(
7459 self.group_refs()
7460 .iter()
7461 .position(|g| {
7462 let gg = g.lock();
7463 gg.name == parent_group_path && !gg.deleted
7464 })
7465 .ok_or_else(|| {
7466 crate::io::IoError::NotFound(format!(
7467 "parent group '{parent_group_path}' not found"
7468 ))
7469 })?,
7470 )
7471 };
7472
7473 // Resolve the target. Dataset names are stored without a leading
7474 // '/', group names with one — compare on the trimmed form. A
7475 // trailing '/' is tolerated too.
7476 let target_rel = self.canonical_dataset_path(target_path.trim_matches('/'));
7477 let target_rel = target_rel.as_str();
7478 if target_rel.is_empty() {
7479 return Err(crate::io::IoError::InvalidState(
7480 "cannot hard-link the root group".into(),
7481 ));
7482 }
7483 let target = self.resolve_object(target_rel).ok_or_else(|| {
7484 crate::io::IoError::NotFound(format!("hard link target '{target_path}' not found"))
7485 })?;
7486
7487 // Reject a name already taken in the parent group.
7488 self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7489
7490 self.hard_links.lock().push(HardLink {
7491 parent,
7492 name: link_name.to_string(),
7493 target,
7494 creation_seq: self.take_creation_seq(),
7495 });
7496 self.register_name(&self.link_full_path(parent, link_name), NameHit::HardLink);
7497 Ok(())
7498 }
7499
7500 /// Whether a hard link will actually be emitted: both its parent group
7501 /// and its target object must still be present (not soft-deleted).
7502 fn hard_link_emitted(&self, link: &HardLink) -> bool {
7503 let parent_ok = self.parent_alive(link.parent);
7504 let target_ok = match link.target {
7505 HardLinkTarget::Dataset(i) => !self.ds(i).lock().deleted,
7506 HardLinkTarget::Group(i) => !self.grp(i).lock().deleted,
7507 };
7508 parent_ok && target_ok
7509 }
7510
7511 /// The full path a link occupies, with no leading `/` — the same form
7512 /// dataset names are stored in. The one place a parent index and a leaf
7513 /// name become a path, so every link kind answers the collision check in
7514 /// the same spelling.
7515 fn link_full_path(&self, parent: Option<usize>, name: &str) -> String {
7516 match parent {
7517 None => name.to_string(),
7518 Some(pi) => format!(
7519 "{}/{name}",
7520 self.grp(pi).lock().name.trim_start_matches('/')
7521 ),
7522 }
7523 }
7524
7525 /// The full path a hard link occupies; see [`Self::link_full_path`].
7526 fn hard_link_full_path(&self, link: &HardLink) -> String {
7527 self.link_full_path(link.parent, &link.name)
7528 }
7529
7530 /// Whether a symbolic link will actually be emitted: its parent group
7531 /// must still be present. There is no target to check — a soft or
7532 /// external link is allowed to dangle, and `H5Lcreate_soft` does not look
7533 /// at the path it stores.
7534 fn symbolic_link_emitted(&self, link: &SymbolicLink) -> bool {
7535 self.parent_alive(link.parent)
7536 }
7537
7538 /// Whether the group that would hold a link still exists; `None` is the
7539 /// root group, which cannot be deleted.
7540 ///
7541 /// A deleted group's header is never written, so nothing it would have
7542 /// held is in the file — and the name is free again. Every registry
7543 /// decides that the same way, through here.
7544 fn parent_alive(&self, parent: Option<usize>) -> bool {
7545 match parent {
7546 None => true,
7547 Some(pi) => !self.grp(pi).lock().deleted,
7548 }
7549 }
7550
7551 /// The full path a symbolic link occupies; see [`Self::link_full_path`].
7552 fn symbolic_link_full_path(&self, link: &SymbolicLink) -> String {
7553 self.link_full_path(link.parent, &link.name)
7554 }
7555
7556 /// Create a soft or external link: a name in a group whose value is a
7557 /// path rather than an object.
7558 ///
7559 /// The single owner of symbolic-link creation — `H5Lcreate_soft` and
7560 /// `H5Lcreate_external` differ only in the value they store, and the
7561 /// name, parent and collision rules they share are all here.
7562 ///
7563 /// * `parent_group_path` — full path of the group that will hold the
7564 /// link (`"/"` for the root group).
7565 /// * `link_name` — leaf name of the new link within that group.
7566 /// * `target` — the path this link names, and for an external link the
7567 /// file holding it. Neither is resolved or required to exist: HDF5
7568 /// answers a symbolic link at traversal time, so a dangling one is a
7569 /// legal file.
7570 pub fn create_symbolic_link(
7571 &self,
7572 parent_group_path: &str,
7573 link_name: &str,
7574 target: LinkTarget,
7575 ) -> IoResult<()> {
7576 if link_name.is_empty() || link_name.contains('/') {
7577 return Err(crate::io::IoError::InvalidState(format!(
7578 "link name '{link_name}' must be a non-empty leaf name"
7579 )));
7580 }
7581 // `H5Lcreate_external` refuses an empty file or object name, and
7582 // stores the object path normalized; a link written here and one
7583 // libhdf5 writes from the same arguments then hold the same bytes.
7584 let target = match target {
7585 LinkTarget::External { file, path } => {
7586 if file.is_empty() || path.is_empty() {
7587 return Err(crate::io::IoError::InvalidState(
7588 "an external link needs both a file name and an object path".into(),
7589 ));
7590 }
7591 LinkTarget::External {
7592 file,
7593 path: crate::format::messages::link::normalize_object_path(&path),
7594 }
7595 }
7596 other => other,
7597 };
7598 // The link itself would be a name in a group that lives in another
7599 // file; its *value* may name anything, including a path this writer
7600 // cannot follow, because nothing follows it here.
7601 self.reject_external_traversal(&format!(
7602 "{}/{link_name}",
7603 parent_group_path.trim_end_matches('/')
7604 ))?;
7605
7606 let _create = self.create_lock.lock();
7607 let parent_group_path = self.canonical_group_path(parent_group_path);
7608 let parent_group_path = parent_group_path.as_str();
7609 let parent = if parent_group_path == "/" {
7610 None
7611 } else {
7612 Some(
7613 self.group_refs()
7614 .iter()
7615 .position(|g| {
7616 let gg = g.lock();
7617 gg.name == parent_group_path && !gg.deleted
7618 })
7619 .ok_or_else(|| {
7620 crate::io::IoError::NotFound(format!(
7621 "parent group '{parent_group_path}' not found"
7622 ))
7623 })?,
7624 )
7625 };
7626
7627 self.ensure_name_free(&self.link_full_path(parent, link_name))?;
7628 self.symbolic_links.lock().push(SymbolicLink {
7629 parent,
7630 name: link_name.to_string(),
7631 target,
7632 creation_seq: self.take_creation_seq(),
7633 });
7634 self.register_name(
7635 &self.link_full_path(parent, link_name),
7636 NameHit::SymbolicLink,
7637 );
7638 Ok(())
7639 }
7640
7641 // ---------------------------------------------------------- committed types
7642
7643 /// Snapshot the committed-datatype list; see [`Self::hard_links_vec`].
7644 pub(crate) fn committed_datatypes_vec(&self) -> Vec<CommittedDatatype> {
7645 self.committed_datatypes.lock().clone()
7646 }
7647
7648 /// The paths of every committed datatype a name still reaches, in
7649 /// creation order. One inside a deleted group is not among them: no link
7650 /// to it is emitted, so the file will not hold that name.
7651 ///
7652 /// Both halves of the file answer. A datatype an earlier session
7653 /// committed is carried by its bytes, not re-encoded, so it lives in the
7654 /// preserved-link list rather than the registry — and listing only the
7655 /// registry is what made this answer `[]` for a file whose every named
7656 /// type was committed before it was opened, while a reader of the same
7657 /// file named them all.
7658 pub(crate) fn committed_datatype_names(&self) -> Vec<String> {
7659 let mut out: Vec<String> = self
7660 .committed_datatypes_vec()
7661 .iter()
7662 .filter(|c| self.parent_alive(c.parent))
7663 .map(|c| c.name.clone())
7664 .collect();
7665 out.extend(
7666 self.preserved_links
7667 .lock()
7668 .iter()
7669 .filter(|l| l.kind == PreservedKind::NamedDatatype)
7670 .map(|l| self.preserved_link_full_path(l)),
7671 );
7672 out
7673 }
7674
7675 /// Commit `datatype` as an object of its own under `name` —
7676 /// `H5Tcommit2`. Returns its index in the committed-datatype registry.
7677 ///
7678 /// The object holds one datatype message and nothing else. It goes
7679 /// through [`begin_create`](Self::begin_create) like a dataset, so its
7680 /// name is resolved to a real parent group, refused if taken, and refused
7681 /// if it would cross a carried external link.
7682 pub fn commit_datatype(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
7683 let create = self.begin_create(name.trim_start_matches('/'))?;
7684 let entry = CommittedDatatype {
7685 name: create.name.clone(),
7686 parent: create.parent,
7687 datatype,
7688 creation_seq: self.take_creation_seq(),
7689 times: self.created_object_times(),
7690 obj_header_addr: 0,
7691 };
7692 let name = entry.name.clone();
7693 let idx = {
7694 let mut reg = self.committed_datatypes.lock();
7695 let idx = reg.len();
7696 reg.push(entry);
7697 idx
7698 };
7699 self.register_name(&name, NameHit::Datatype(idx));
7700 Ok(idx)
7701 }
7702
7703 /// Resolve a committed datatype's path to its registry index and the type
7704 /// it holds — the pair a dataset needs to be built on it.
7705 ///
7706 /// Returned together so the caller cannot pair one committed type's index
7707 /// with another's datatype: the dataset's element width, dataspace and
7708 /// payload checks all come from the type, and its header names the index.
7709 pub(crate) fn committed_datatype_for_share(
7710 &self,
7711 name: &str,
7712 ) -> IoResult<(usize, DatatypeMessage)> {
7713 let name = self.canonical_dataset_path(name.trim_start_matches('/'));
7714 let all = self.committed_datatypes_vec();
7715 all.iter()
7716 .position(|c| self.parent_alive(c.parent) && c.name == name)
7717 .map(|i| (i, all[i].datatype.clone()))
7718 .ok_or_else(|| {
7719 crate::io::IoError::NotFound(format!("no committed datatype named '{name}'"))
7720 })
7721 }
7722
7723 /// Record that dataset `dataset` stores its datatype as a pointer to the
7724 /// committed datatype `committed`.
7725 ///
7726 /// Takes an index [`committed_datatype_for_share`](Self::committed_datatype_for_share)
7727 /// produced, alongside the datatype from the same call, so the two cannot
7728 /// disagree and there is nothing here that can fail after the dataset
7729 /// exists.
7730 pub(crate) fn share_committed_type(&self, dataset: usize, committed: usize) {
7731 debug_assert!(committed < self.committed_datatypes.lock().len());
7732 self.ds(dataset).lock().committed_type = Some(CommittedTypeRef::Session(committed));
7733 }
7734
7735 /// How many names reach the committed datatype `index`: the link that
7736 /// gave it its name, plus every live dataset that shares it.
7737 ///
7738 /// `H5O__shared_link_adj` counts a share as a link, which is why a type
7739 /// h5py commits and then builds one dataset on reports `rc == 2`. Zero
7740 /// means nothing reaches it at all — the group holding its name was
7741 /// deleted and no dataset shares it — and then it is not written.
7742 fn committed_datatype_refcount(&self, index: usize) -> u32 {
7743 let linked = {
7744 let parent = self.committed_datatypes.lock()[index].parent;
7745 u32::from(self.parent_alive(parent))
7746 };
7747 let shares = self
7748 .dataset_refs()
7749 .iter()
7750 .filter(|d| {
7751 let m = d.lock();
7752 !m.deleted && m.committed_type == Some(CommittedTypeRef::Session(index))
7753 })
7754 .count() as u32;
7755 linked + shares
7756 }
7757
7758 /// Append the link naming each committed datatype whose parent group is
7759 /// `parent`. A committed datatype is reached by an ordinary hard link —
7760 /// what makes it a datatype rather than a group or a dataset is the one
7761 /// message in the header it points at.
7762 ///
7763 /// Only a live group's links are collected, and a live parent is itself a
7764 /// reference, so every address named here belongs to a header
7765 /// `write_committed_datatype_headers` wrote.
7766 fn push_committed_datatypes(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
7767 for cd in self.committed_datatypes_vec() {
7768 if cd.parent != parent {
7769 continue;
7770 }
7771 let leaf = cd.name.rsplit('/').next().unwrap_or(&cd.name);
7772 links.push((cd.creation_seq, LinkMessage::hard(leaf, cd.obj_header_addr)));
7773 }
7774 }
7775
7776 /// Rewrite a group path that passes through hard links into the tree
7777 /// path of the group it reaches — HDF5 traversal, where any link in a
7778 /// path component resolves to its target. Group-name form (leading
7779 /// `/`). Repeats because a substituted target's subtree can hold
7780 /// further links; bounded like libhdf5's link-traversal limit, so a
7781 /// link cycle cannot loop forever. A path with no link components
7782 /// (including one naming nothing at all) comes back unchanged.
7783 pub(crate) fn canonical_group_path(&self, path: &str) -> String {
7784 let mut path = path.to_string();
7785 for _ in 0..64 {
7786 // The longest emitted group-link path that is the whole of
7787 // `path` or a '/'-boundary prefix of it.
7788 let mut best: Option<(usize, usize)> = None; // (prefix len, target)
7789 for l in self.hard_links_vec() {
7790 let HardLinkTarget::Group(gi) = l.target else {
7791 continue;
7792 };
7793 if !self.hard_link_emitted(&l) {
7794 continue;
7795 }
7796 let lp = format!("/{}", self.hard_link_full_path(&l));
7797 let covers = path == lp || path.starts_with(&format!("{lp}/"));
7798 if covers && best.is_none_or(|(len, _)| lp.len() > len) {
7799 best = Some((lp.len(), gi));
7800 }
7801 }
7802 let Some((len, gi)) = best else { break };
7803 let target_name = self.grp(gi).lock().name.clone();
7804 path = format!("{}{}", target_name, &path[len..]);
7805 }
7806 path
7807 }
7808
7809 /// [`canonical_group_path`](Self::canonical_group_path) in the
7810 /// dataset-name form (no leading `/`): the leaf is a dataset, so only
7811 /// group links can appear as components and the whole path can go
7812 /// through the group rewrite unchanged.
7813 fn canonical_dataset_path(&self, name: &str) -> String {
7814 self.canonical_group_path(&format!("/{name}"))
7815 .trim_start_matches('/')
7816 .to_string()
7817 }
7818
7819 /// Total number of hard links resolving to an object: its own tree link
7820 /// plus every emitted user-created hard link pointing at it.
7821 fn object_link_count(&self, target: HardLinkTarget) -> u32 {
7822 let same = |a: HardLinkTarget, b: HardLinkTarget| -> bool {
7823 matches!(
7824 (a, b),
7825 (HardLinkTarget::Dataset(x), HardLinkTarget::Dataset(y))
7826 | (HardLinkTarget::Group(x), HardLinkTarget::Group(y))
7827 if x == y
7828 )
7829 };
7830 1 + self
7831 .hard_links_vec()
7832 .iter()
7833 .filter(|l| self.hard_link_emitted(l) && same(l.target, target))
7834 .count() as u32
7835 }
7836
7837 /// The object a path names, or `None` when nothing in the file does.
7838 ///
7839 /// `path` is the trimmed, hard-link-canonical form (no leading or
7840 /// trailing `/`) that dataset and group names compare against. The single
7841 /// owner of path→object resolution on the write side: hard links and
7842 /// object references must agree on what a path means, including that a
7843 /// path may itself be a user hard link — links have no chain (each points
7844 /// straight at the object header, as in libhdf5), so the existing link's
7845 /// target is the answer.
7846 pub(crate) fn resolve_object(&self, path: &str) -> Option<HardLinkTarget> {
7847 if let Some(idx) = self.dataset_refs().iter().position(|d| {
7848 let g = d.lock();
7849 !g.deleted && g.name.trim_start_matches('/') == path
7850 }) {
7851 return Some(HardLinkTarget::Dataset(idx));
7852 }
7853 if let Some(idx) = self.group_refs().iter().position(|g| {
7854 let gg = g.lock();
7855 !gg.deleted && gg.name.trim_start_matches('/') == path
7856 }) {
7857 return Some(HardLinkTarget::Group(idx));
7858 }
7859 self.hard_links_vec().iter().find_map(|l| {
7860 (self.hard_link_emitted(l) && self.hard_link_full_path(l) == path).then_some(l.target)
7861 })
7862 }
7863
7864 /// The address of dataset `index`'s own contiguous block, for the two
7865 /// writers that stamp single elements into it by file offset — object and
7866 /// region references, whose values are only known once finalize has placed
7867 /// every object header.
7868 ///
7869 /// Refuses, rather than handing back an address that is not one, every
7870 /// dataset that has no such block: chunked, compact, unallocated, or with
7871 /// its raw data in files outside this one.
7872 fn local_element_block(&self, index: usize, what: &str) -> IoResult<u64> {
7873 let ds = self.ds(index);
7874 let m = ds.lock();
7875 match m.contiguous_target() {
7876 Some(ContiguousTarget::Local(addr)) => Ok(addr),
7877 Some(ContiguousTarget::External { .. }) => {
7878 Err(crate::io::IoError::InvalidState(format!(
7879 "{what} are stamped into the dataset's own contiguous block, and \
7880 dataset '{}' has none: its raw data lives in external files",
7881 m.name
7882 )))
7883 }
7884 Some(ContiguousTarget::Virtual) => Err(crate::io::IoError::InvalidState(format!(
7885 "{what} are stamped into the dataset's own contiguous block, and \
7886 dataset '{}' has none: it is virtual, and its elements come from \
7887 the source datasets its mappings name",
7888 m.name
7889 ))),
7890 None => Err(crate::io::IoError::InvalidState(format!(
7891 "{what} are stamped into contiguous storage; create the dataset \
7892 without chunking"
7893 ))),
7894 }
7895 }
7896
7897 /// Store object references naming `paths` into the elements of dataset
7898 /// `index` starting at `start`.
7899 ///
7900 /// The value of an `H5R_OBJECT1` element is its target's object header
7901 /// address, which finalize assigns, so what lands here is the target path;
7902 /// [`Self::write_object_reference_values`] writes the addresses. Elements
7903 /// never written keep the zero image libhdf5 reads back as a null
7904 /// reference.
7905 pub fn write_object_references(
7906 &self,
7907 index: usize,
7908 start: u64,
7909 paths: &[&str],
7910 ) -> IoResult<()> {
7911 let elements = {
7912 let ds = self.ds(index);
7913 let m = ds.lock();
7914 match &m.datatype {
7915 // Both generations of object reference: `H5T_STD_REF_OBJ` and
7916 // the 1.12 `H5T_STD_REF`. They differ only in the element
7917 // image, which `encode_reference_element` owns.
7918 DatatypeMessage::Reference {
7919 kind: ReferenceKind::Object1 | ReferenceKind::Object2,
7920 ..
7921 } => {}
7922 other => {
7923 return Err(crate::io::IoError::InvalidState(format!(
7924 "dataset '{}' has datatype {other}, not an object reference",
7925 m.name
7926 )))
7927 }
7928 }
7929 m.dataspace
7930 .dims
7931 .iter()
7932 .fold(1u64, |a, &d| a.saturating_mul(d))
7933 };
7934 // Refused here as well as at fixup time, so a dataset whose storage
7935 // cannot hold stamped elements is reported at the call that chose it.
7936 self.local_element_block(index, "object references")?;
7937 let end = start.saturating_add(paths.len() as u64);
7938 if end > elements {
7939 return Err(crate::io::IoError::InvalidState(format!(
7940 "elements {start}..{end} are outside the dataset's {elements}"
7941 )));
7942 }
7943 // Resolve now as well as at fixup time, so a path that names nothing
7944 // is reported at the call that got it wrong.
7945 for path in paths {
7946 self.object_reference_target(path)?;
7947 }
7948 let mut pending = self.pending_object_references.lock();
7949 for (i, path) in paths.iter().enumerate() {
7950 pending.push(PendingObjectReference {
7951 dataset: index,
7952 element: start + i as u64,
7953 target: (*path).to_string(),
7954 });
7955 }
7956 Ok(())
7957 }
7958
7959 /// Record a hard link count of `rc` in `header`, if this file's format
7960 /// needs a message to carry it.
7961 ///
7962 /// A version-2 header carries the count in an Object Reference Count
7963 /// message, and only when more than one link reaches the object. A
7964 /// version-1 header carries it in its prefix and gets no message at all —
7965 /// `H5O_link_oh` gates every refcount-message operation on
7966 /// `oh->version > H5O_VERSION_1` (H5Oint.c:851), so a version-1 header
7967 /// holding one is a shape libhdf5 never writes.
7968 ///
7969 /// The message carries `H5O_MSG_FLAG_DONTSHARE`, which both refcount
7970 /// operations pass (H5Oint.c:874 append, H5Oint.c:864 write): the count is
7971 /// a property of this one object header, so a shared-message index that
7972 /// pointed several headers at one copy would make every object with the
7973 /// same link count share a single number.
7974 fn emit_refcount(&self, header: &mut ObjectHeader, rc: u32, format: ObjectFormat) {
7975 if rc > 1 && format == ObjectFormat::Modern {
7976 header.add_message(MSG_OBJ_REF_COUNT, MSG_FLAG_DONTSHARE, encode_refcount(rc));
7977 }
7978 }
7979
7980 /// Encode `header` as `placement` lays it out, at the version this file's
7981 /// format calls for and with `rc` as the object's hard link count: every
7982 /// `(address, image)` pair to write, chunk 0 first.
7983 ///
7984 /// The count is passed rather than read off the header because the two
7985 /// versions carry it in different places — the version-1 prefix's `nlink`
7986 /// field, the version-2 Reference Count message
7987 /// [`emit_refcount`](Self::emit_refcount) already added — and only the
7988 /// caller knows it.
7989 ///
7990 /// INVARIANT: an object header's chunk 0 never moves once something in the
7991 /// file has named its address. A written header is rewritten over the
7992 /// chunk-0 block it already has, padded when the messages shrank and
7993 /// spilling into a continuation block of its own when they grew — the way
7994 /// `H5O__alloc_new_chunk` (H5Oalloc.c) grows a header libhdf5 cannot
7995 /// extend in place. That is what keeps every object reference already in
7996 /// the file — in a reference dataset, an attribute, a `REFERENCE_LIST`,
7997 /// whoever wrote them — resolving after this session. The one exception is
7998 /// a block too small to hold even the message naming a continuation, which
7999 /// [`place_header`](Self::place_header) gives up and replaces.
8000 ///
8001 /// A fresh header lives in the one block its address and encoded size
8002 /// describe: one whose messages overflow chunk 0 gets its continuation
8003 /// chunk immediately behind it in that same block, so the address is
8004 /// enough to free or supersede the whole header. libhdf5 would have grown
8005 /// chunk 0 into space that free rather than chaining onto it, but it reads
8006 /// a continuation chunk by the address and length its message states and
8007 /// cares nothing for where that lands.
8008 fn encode_header_in(
8009 &self,
8010 header: &ObjectHeader,
8011 rc: u32,
8012 format: ObjectFormat,
8013 placement: &HeaderPlacement,
8014 ) -> IoResult<Vec<(u64, Vec<u8>)>> {
8015 let plan = if placement.kept {
8016 header.plan_chunks_in(format, placement.size, &self.ctx)?
8017 } else {
8018 header.plan_chunks(format, self.chunk0_capacity(header, format), &self.ctx)?
8019 };
8020 let continuation_addr = match placement.continuation {
8021 Some((addr, _)) => addr,
8022 None => placement.addr + plan.chunk0_size as u64,
8023 };
8024 let (mut chunk0, continuation) =
8025 header.encode_chunked(&plan, format, &self.ctx, continuation_addr, rc)?;
8026 match (placement.continuation, continuation) {
8027 (Some((addr, _)), Some(image)) => Ok(vec![(placement.addr, chunk0), (addr, image)]),
8028 (None, Some(image)) => {
8029 chunk0.extend_from_slice(&image);
8030 Ok(vec![(placement.addr, chunk0)])
8031 }
8032 (None, None) => Ok(vec![(placement.addr, chunk0)]),
8033 (Some((addr, size)), None) => Err(crate::io::IoError::InvalidState(format!(
8034 "an object header was placed with a {size}-byte continuation block at \
8035 {addr:#x} that it no longer needs; a message in it changed length \
8036 once the addresses it names were known"
8037 ))),
8038 }
8039 }
8040
8041 /// Reserve the blocks `header` will be written over, keeping `kept` — the
8042 /// chunk-0 block the object's existing header occupies — when there is
8043 /// one it can be written over.
8044 ///
8045 /// A header's layout does not depend on the addresses it carries, which is
8046 /// what lets the group pass hand every group header an address before it
8047 /// writes any of their content: every address is a fixed-width field.
8048 ///
8049 /// A kept block is given up only when it cannot describe the header at
8050 /// all: too narrow for the message naming a continuation chunk, or not a
8051 /// shape the header's version can pad (see `ObjectHeader::plan_chunks_in`).
8052 /// Then it is freed and the header gets a fresh block, exactly as a new
8053 /// object does — and the references naming it are the caller's to
8054 /// restamp, which the writer does for every one it registered.
8055 fn place_header(
8056 &mut self,
8057 header: &ObjectHeader,
8058 format: ObjectFormat,
8059 kept: Option<(u64, u64)>,
8060 ) -> IoResult<HeaderPlacement> {
8061 if let Some((addr, len)) = kept {
8062 let plan = usize::try_from(len)
8063 .ok()
8064 .and_then(|len| header.plan_chunks_in(format, len, &self.ctx).ok());
8065 match plan {
8066 Some(plan) => {
8067 let continuation = (plan.continuation_size > 0).then(|| {
8068 let size = plan.continuation_size;
8069 let addr = self
8070 .allocator
8071 .allocate(size as u64, FreeSpaceClass::Metadata);
8072 (addr, size)
8073 });
8074 return Ok(HeaderPlacement {
8075 addr,
8076 size: len as usize,
8077 kept: true,
8078 continuation,
8079 });
8080 }
8081 // A block a SWMR reader may be walking stays allocated, as
8082 // everywhere else under `swmr_active`.
8083 None if !self.swmr_active => {
8084 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
8085 }
8086 None => {}
8087 }
8088 }
8089 let plan = header.plan_chunks(format, self.chunk0_capacity(header, format), &self.ctx)?;
8090 let size = plan.chunk0_size + plan.continuation_size;
8091 let addr = self
8092 .allocator
8093 .allocate(size as u64, FreeSpaceClass::Metadata);
8094 Ok(HeaderPlacement::fresh(addr, size))
8095 }
8096
8097 /// How many bytes of messages `header`'s chunk 0 holds before the rest
8098 /// spill into a continuation chunk.
8099 ///
8100 /// libhdf5 sizes chunk 0 once, when the object header is created, and can
8101 /// only grow it while the space behind it is still free — so an object
8102 /// whose creation-time estimate covered every message it would ever hold
8103 /// keeps one chunk, and one whose estimate was a guess does not. A dataset
8104 /// or a committed datatype is created from messages already in hand
8105 /// (`H5D__update_oh_info`, `H5T__commit`), so its estimate is exact and
8106 /// this writer's exact fit is the same answer.
8107 ///
8108 /// A group is the exception: `H5G__obj_create_real` (H5Gobj.c:219) sizes
8109 /// its header for the link info and group info messages plus
8110 /// `H5G_CRT_GINFO_EST_NUM_ENTRIES` links of `H5G_CRT_GINFO_EST_NAME_LEN`
8111 /// characters, and nothing else — attributes above all — is in that
8112 /// estimate. The Link Info message is what identifies one: it is the
8113 /// message that makes an object a new-format group, and
8114 /// `H5G__obj_get_linfo` uses it for exactly this question.
8115 ///
8116 /// A version-1 header is written as one chunk whatever it holds: its
8117 /// groups keep their links in a symbol table, not in the header, so the
8118 /// estimate that makes a version-2 group spill never applies to one.
8119 fn chunk0_capacity(&self, header: &ObjectHeader, format: ObjectFormat) -> usize {
8120 if format == ObjectFormat::Legacy {
8121 return usize::MAX;
8122 }
8123 let envelope = header.message_envelope_size();
8124 let sized = |msg_type: u8| {
8125 header
8126 .messages
8127 .iter()
8128 .find(|m| m.msg_type == msg_type)
8129 .map(|m| envelope + m.data.len())
8130 };
8131 let Some(link_info) = sized(MSG_LINK_INFO) else {
8132 return usize::MAX;
8133 };
8134 // One estimated hard link: version, flags, a one-byte name length for
8135 // a name this short, the name, and the object header address.
8136 let link = envelope + 1 + 1 + 1 + EST_LINK_NAME_LEN + self.ctx.sizeof_addr as usize;
8137 link_info + sized(MSG_GROUP_INFO).unwrap_or(0) + EST_LINK_COUNT * link
8138 }
8139
8140 /// The object an object reference's path names, as a hard-link target;
8141 /// `None` for the root group, which has no registry slot.
8142 fn object_reference_target(&self, path: &str) -> IoResult<Option<HardLinkTarget>> {
8143 let rel = self.canonical_dataset_path(path.trim_matches('/'));
8144 if rel.is_empty() {
8145 return Ok(None);
8146 }
8147 self.resolve_object(&rel)
8148 .map(Some)
8149 .ok_or_else(|| crate::io::IoError::NotFound(format!("reference target '{path}'")))
8150 }
8151
8152 /// The object header address an object reference's `path` names, or zero
8153 /// when that object has not been given one yet.
8154 ///
8155 /// Zero is where the superblock sits, so it is never an object header's
8156 /// address. It is what every object reads as before
8157 /// [`allocate_object_headers`](Self::allocate_object_headers) runs, which
8158 /// is what lets the pass that measures a header stand in for the pass that
8159 /// writes it: an address is a fixed-width field, so the placeholder is the
8160 /// same size as the answer.
8161 fn object_reference_address(&self, path: &str) -> IoResult<u64> {
8162 Ok(match self.object_reference_target(path)? {
8163 Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8164 Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8165 None => self.root_group_addr.unwrap_or(0),
8166 })
8167 }
8168
8169 /// `scope`'s attributes as this finalize will write them: the stored set,
8170 /// with every object-reference attribute's value said in the object header
8171 /// addresses assigned so far.
8172 ///
8173 /// The single owner of a reference attribute's value, and the only source
8174 /// an object header build may take an attribute set from. Nothing stored
8175 /// is mutated, so the pass that measures a header and the pass that writes
8176 /// it cannot disagree about anything but the addresses — which they cannot
8177 /// disagree about in length.
8178 ///
8179 /// INVARIANT: the stored attribute list is what says which attributes
8180 /// exist; a recorded reference value can only give a value to one already
8181 /// in it. So a value left behind by an object whose list was emptied — a
8182 /// deleted group or dataset — cannot put the attribute back, and a value
8183 /// whose attribute was replaced by one of another type is dropped at the
8184 /// replacement instead of reaching it (see
8185 /// [`forget_attribute_reference`](Self::forget_attribute_reference)).
8186 fn object_attributes(&self, scope: AttrScope) -> IoResult<Vec<AttributeEntry>> {
8187 let mut attrs = match scope {
8188 AttrScope::Root => self.root_attributes.lock().clone(),
8189 AttrScope::Group(gi) => self.grp(gi).lock().attributes.clone(),
8190 AttrScope::Dataset(i) => self.ds(i).lock().attributes.clone(),
8191 };
8192 // Snapshot first: resolving a path locks group and dataset slots.
8193 let values: Vec<(String, Vec<String>, usize)> = self
8194 .attribute_references
8195 .lock()
8196 .iter()
8197 .filter(|r| r.scope == scope)
8198 .map(|r| (r.name.clone(), r.targets.clone(), r.stride))
8199 .collect();
8200 let width = self.ctx.sizeof_addr as usize;
8201 for (name, targets, stride) in values {
8202 let Some(pos) = attrs.iter().position(|a| a.name() == name) else {
8203 continue;
8204 };
8205 let Some(msg) = attrs[pos].readable() else {
8206 continue;
8207 };
8208 let mut msg = msg.clone();
8209 for (i, target) in targets.iter().enumerate() {
8210 let at = i * stride;
8211 let held = msg.data.len();
8212 let slot = msg.data.get_mut(at..at + width).ok_or_else(|| {
8213 crate::io::IoError::InvalidState(format!(
8214 "attribute '{name}' holds {held} bytes, too few for reference {i} at {at}"
8215 ))
8216 })?;
8217 slot.copy_from_slice(
8218 &self.object_reference_address(target)?.to_le_bytes()[..width],
8219 );
8220 }
8221 attrs[pos] = AttributeEntry::from(msg).with_creation_index(attrs[pos].creation_index());
8222 }
8223 Ok(attrs)
8224 }
8225
8226 /// The registry scope `target` names — the same object
8227 /// [`with_attr_list`](Self::with_attr_list) reaches, as the key the
8228 /// reference-value registry is indexed by. Refuses what that accessor
8229 /// refuses, and for the same reasons.
8230 fn attr_scope(&self, target: AttrTarget<'_>) -> IoResult<AttrScope> {
8231 match target {
8232 AttrTarget::Root => Ok(AttrScope::Root),
8233 AttrTarget::Group(path) => {
8234 let path = self.canonical_group_path(path);
8235 self.group_refs()
8236 .iter()
8237 .position(|g| {
8238 let gg = g.lock();
8239 gg.name == path && !gg.deleted
8240 })
8241 .map(AttrScope::Group)
8242 .ok_or_else(|| {
8243 crate::io::IoError::NotFound(format!("group '{path}' not found"))
8244 })
8245 }
8246 AttrTarget::Dataset(index) => {
8247 let count = self.dataset_count();
8248 if index >= count {
8249 return Err(crate::io::IoError::InvalidState(format!(
8250 "dataset index {index} out of range (have {count})"
8251 )));
8252 }
8253 Ok(AttrScope::Dataset(index))
8254 }
8255 }
8256 }
8257
8258 /// Drop the reference value recorded for `scope`'s attribute `name`.
8259 ///
8260 /// Called by both owners of attribute-list mutation —
8261 /// [`insert_attribute`](Self::insert_attribute) and
8262 /// [`evict_attr`](Self::evict_attr) — so an attribute that is replaced or
8263 /// removed cannot leave its value behind for whatever takes its name next.
8264 /// A string attribute written over a reference attribute is the case that
8265 /// needs it: without this the string's bytes would be overwritten with
8266 /// addresses at finalize.
8267 fn forget_attribute_reference(&self, scope: AttrScope, name: &str) {
8268 self.attribute_references
8269 .lock()
8270 .retain(|r| !(r.scope == scope && r.name == name));
8271 }
8272
8273 /// Write every pending object reference element as its target's object
8274 /// header address.
8275 ///
8276 /// INVARIANT: a reference element on disk holds its target's header
8277 /// address. Reached through [`write_reference_values`](Self::write_reference_values),
8278 /// which places it after every header has an address; a target that no
8279 /// longer resolves fails the finalize rather than leaving a placeholder
8280 /// behind.
8281 fn write_object_reference_values(&mut self) -> IoResult<()> {
8282 // Snapshot rather than drain: a SWMR session finalizes twice, and the
8283 // close-time finalize rebuilds every header at a fresh address, so the
8284 // elements must be stamped again with the addresses that survive.
8285 let pending: Vec<(usize, u64, String)> = self
8286 .pending_object_references
8287 .lock()
8288 .iter()
8289 .map(|p| (p.dataset, p.element, p.target.clone()))
8290 .collect();
8291 for (dataset, element, target) in &pending {
8292 let addr = match self.object_reference_target(target)? {
8293 Some(HardLinkTarget::Dataset(i)) => self.ds(i).lock().obj_header_addr,
8294 Some(HardLinkTarget::Group(i)) => self.grp(i).lock().obj_header_addr,
8295 None => self.root_group_addr.ok_or_else(|| {
8296 crate::io::IoError::InvalidState(
8297 "root group header address is not assigned yet".into(),
8298 )
8299 })?,
8300 };
8301 // The element image is the dataset's own datatype's business: the
8302 // pre-1.12 and 1.12 forms differ in width and in layout, and the
8303 // dataset says which it holds.
8304 let (kind, width) = {
8305 let ds = self.ds(*dataset);
8306 let m = ds.lock();
8307 let DatatypeMessage::Reference { kind, size } = &m.datatype else {
8308 return Err(crate::io::IoError::InvalidState(format!(
8309 "dataset '{}' is no longer a reference dataset",
8310 m.name
8311 )));
8312 };
8313 (*kind, *size as usize)
8314 };
8315 let image = match kind {
8316 ReferenceKind::Object1 => ReferenceElementImage::Legacy(addr),
8317 ReferenceKind::Object2 => ReferenceElementImage::Inline(addr),
8318 other => {
8319 return Err(crate::io::IoError::InvalidState(format!(
8320 "dataset {dataset} now holds {other:?} elements, not object references"
8321 )))
8322 }
8323 };
8324 let image = encode_reference_element(&image, width, &self.ctx)?;
8325 let data_addr = self.local_element_block(*dataset, "object references")?;
8326 let at = data_addr + element * width as u64;
8327 self.handle.write_at(at, &image)?;
8328 }
8329 Ok(())
8330 }
8331
8332 /// Store region references over `targets` into the elements of dataset
8333 /// `index` starting at `start`.
8334 ///
8335 /// Each target is the path of a dataset and a selection over it. What the
8336 /// element holds is a global-heap id — collection address then object index
8337 /// (`H5R__encode_heap`) — and the heap object it names is the target's
8338 /// object header address followed by the serialized selection
8339 /// (`H5R__encode_token_region_compat`). Both the object and the element are
8340 /// written here; only the address inside the object waits for
8341 /// [`Self::write_heap_reference_values`]. Elements never written keep the
8342 /// zero image libhdf5 reads back as a null reference.
8343 pub fn write_region_references(
8344 &self,
8345 index: usize,
8346 start: u64,
8347 targets: &[(&str, Selection)],
8348 ) -> IoResult<()> {
8349 let elements = {
8350 let ds = self.ds(index);
8351 let m = ds.lock();
8352 match &m.datatype {
8353 DatatypeMessage::Reference {
8354 kind: ReferenceKind::DatasetRegion1,
8355 ..
8356 } => {}
8357 other => {
8358 return Err(crate::io::IoError::InvalidState(format!(
8359 "dataset '{}' has datatype {other}, not a region reference",
8360 m.name
8361 )))
8362 }
8363 }
8364 m.dataspace
8365 .dims
8366 .iter()
8367 .fold(1u64, |a, &d| a.saturating_mul(d))
8368 };
8369 let data_addr = self.local_element_block(index, "region references")?;
8370 let end = start.saturating_add(targets.len() as u64);
8371 if end > elements {
8372 return Err(crate::io::IoError::InvalidState(format!(
8373 "elements {start}..{end} are outside the dataset's {elements}"
8374 )));
8375 }
8376
8377 // Build every heap object before inserting any: a path that names no
8378 // dataset, or a selection its extent does not admit, is reported at the
8379 // call that got it wrong rather than after half the batch is on disk.
8380 let sa = self.ctx.sizeof_addr as usize;
8381 let mut blobs = Vec::with_capacity(targets.len());
8382 for (path, selection) in targets {
8383 let target = self.region_reference_target(path)?;
8384 let dims = self.ds(target).lock().dataspace.dims.clone();
8385 validate_region_selection(selection, &dims, path)?;
8386 let mut blob = vec![0u8; sa];
8387 blob.extend_from_slice(&selection.encode()?);
8388 blobs.push(blob);
8389 }
8390 let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
8391 let placements = self.insert_vlen_objects(&items)?;
8392
8393 let width = (sa + 4) as u64;
8394 let mut pending = self.pending_heap_references.lock();
8395 for (i, &(collection, obj_index)) in placements.iter().enumerate() {
8396 let mut elem = Vec::with_capacity(width as usize);
8397 elem.extend_from_slice(&collection.to_le_bytes()[..sa]);
8398 elem.extend_from_slice(&u32::from(obj_index).to_le_bytes());
8399 self.handle
8400 .write_at(data_addr + (start + i as u64) * width, &elem)?;
8401 pending.push(PendingHeapReference {
8402 collection,
8403 index: obj_index,
8404 token_offset: 0,
8405 target: PendingHeapTarget::Dataset(targets[i].0.to_string()),
8406 });
8407 }
8408 Ok(())
8409 }
8410
8411 /// Store 1.12 references over `targets` into the elements of dataset
8412 /// `index` starting at `start` — the `H5T_STD_REF` trio.
8413 ///
8414 /// One datatype holds all three kinds, because a 1.12 element leads with
8415 /// the kind it holds; which is why this takes a [`ReferenceTarget`] per
8416 /// element rather than a fixed kind. `H5R_OBJECT2` needs nothing but the
8417 /// target's address, so its element is written inline by the same finalize
8418 /// pass every object reference goes through. The other two encode a
8419 /// selection or an attribute name alongside the token, which does not fit
8420 /// an element, so what is stored is a global-heap blob and the element is
8421 /// its id (`H5T__ref_disk_write`). Elements never written keep the zero
8422 /// image `H5T__ref_disk_isnull` reads back as a null reference.
8423 pub fn write_revised_references(
8424 &self,
8425 index: usize,
8426 start: u64,
8427 targets: &[(&str, ReferenceTarget)],
8428 ) -> IoResult<()> {
8429 let (width, elements) = {
8430 let ds = self.ds(index);
8431 let m = ds.lock();
8432 match &m.datatype {
8433 DatatypeMessage::Reference {
8434 kind: ReferenceKind::Object2,
8435 size,
8436 } => (
8437 *size as u64,
8438 m.dataspace
8439 .dims
8440 .iter()
8441 .fold(1u64, |a, &d| a.saturating_mul(d)),
8442 ),
8443 other => {
8444 return Err(crate::io::IoError::InvalidState(format!(
8445 "dataset '{}' has datatype {other}, not the 1.12 H5T_STD_REF",
8446 m.name
8447 )))
8448 }
8449 }
8450 };
8451 let data_addr = self.local_element_block(index, "references")?;
8452 let end = start.saturating_add(targets.len() as u64);
8453 if end > elements {
8454 return Err(crate::io::IoError::InvalidState(format!(
8455 "elements {start}..{end} are outside the dataset's {elements}"
8456 )));
8457 }
8458
8459 // Build every blob before inserting any, so a path that names nothing,
8460 // a selection an extent does not admit or an attribute that does not
8461 // exist is reported at the call that got it wrong rather than after
8462 // half the batch is on disk.
8463 let mut blobs: Vec<(u64, ReferenceKind, PendingHeapTarget, Vec<u8>)> = Vec::new();
8464 let mut inline: Vec<(u64, String)> = Vec::new();
8465 for (i, (path, target)) in targets.iter().enumerate() {
8466 let element = start + i as u64;
8467 // The rank of the extent the selection is over, which only a region
8468 // reference encodes and takes from the target's dataspace.
8469 let mut extent_rank = 0;
8470 let (kind, pending) = match target {
8471 ReferenceTarget::Object => {
8472 self.object_reference_target(path)?;
8473 inline.push((element, (*path).to_string()));
8474 continue;
8475 }
8476 ReferenceTarget::Region(selection) => {
8477 let ds = self.region_reference_target(path)?;
8478 let dims = self.ds(ds).lock().dataspace.dims.clone();
8479 validate_region_selection(selection, &dims, path)?;
8480 extent_rank = dims.len();
8481 (
8482 ReferenceKind::DatasetRegion2,
8483 PendingHeapTarget::Dataset((*path).to_string()),
8484 )
8485 }
8486 ReferenceTarget::Attribute(name) => {
8487 let scope = match self.object_reference_target(path)? {
8488 Some(HardLinkTarget::Dataset(i)) => AttrScope::Dataset(i),
8489 Some(HardLinkTarget::Group(i)) => AttrScope::Group(i),
8490 None => AttrScope::Root,
8491 };
8492 if !self
8493 .object_attributes(scope)?
8494 .iter()
8495 .any(|a| a.name() == name)
8496 {
8497 return Err(crate::io::IoError::NotFound(format!(
8498 "attribute '{name}' of reference target '{path}'"
8499 )));
8500 }
8501 (
8502 ReferenceKind::Attr,
8503 PendingHeapTarget::Object((*path).to_string()),
8504 )
8505 }
8506 };
8507 blobs.push((
8508 element,
8509 kind,
8510 pending,
8511 encode_revised_blob(0, target, extent_rank, &self.ctx)?,
8512 ));
8513 }
8514
8515 let items: Vec<&[u8]> = blobs.iter().map(|(_, _, _, b)| b.as_slice()).collect();
8516 let placements = self.insert_vlen_objects(&items)?;
8517
8518 let mut pending = self.pending_heap_references.lock();
8519 for ((element, kind, target, blob), &(collection, obj_index)) in
8520 blobs.iter().zip(&placements)
8521 {
8522 // The size the element declares is the heap object's own byte
8523 // count: `H5VL__native_blob_get` refuses to read one whose size
8524 // does not match what the element says.
8525 let image = encode_reference_element(
8526 &ReferenceElementImage::Blob {
8527 kind: *kind,
8528 size: blob.len() as u32,
8529 collection,
8530 index: u32::from(obj_index),
8531 },
8532 width as usize,
8533 &self.ctx,
8534 )?;
8535 self.handle.write_at(data_addr + element * width, &image)?;
8536 pending.push(PendingHeapReference {
8537 collection,
8538 index: obj_index,
8539 token_offset: REVISED_BLOB_TOKEN_OFFSET,
8540 target: target.clone(),
8541 });
8542 }
8543 drop(pending);
8544
8545 let mut pending = self.pending_object_references.lock();
8546 for (element, path) in inline {
8547 pending.push(PendingObjectReference {
8548 dataset: index,
8549 element,
8550 target: path,
8551 });
8552 }
8553 Ok(())
8554 }
8555
8556 /// The dataset a region reference's path names.
8557 ///
8558 /// A region reference names a *dataset*: `H5Rcreate` with
8559 /// `H5R_DATASET_REGION` takes the dataspace of one, and every reader
8560 /// dereferences it as one. A path that resolves to a group — or to the root
8561 /// group, which has no registry slot — is refused here rather than stored
8562 /// as a reference nothing can dereference.
8563 fn region_reference_target(&self, path: &str) -> IoResult<usize> {
8564 match self.object_reference_target(path)? {
8565 Some(HardLinkTarget::Dataset(i)) => Ok(i),
8566 _ => Err(crate::io::IoError::InvalidState(format!(
8567 "region reference target '{path}' is not a dataset"
8568 ))),
8569 }
8570 }
8571
8572 /// Stamp every pending heap-backed reference's object with its target's
8573 /// object header address.
8574 ///
8575 /// The references that are still stamped rather than written once: the
8576 /// *element* is a global-heap id, so the heap object has to exist at the
8577 /// call that stores the reference, long before any address does. The object
8578 /// was inserted with its token zeroed, so its size does not change here:
8579 /// each collection is read once, patched, and rewritten at its own declared
8580 /// size, which leaves every element's heap id valid — and leaves the
8581 /// object's byte count equal to the size the 1.12 element declares, which
8582 /// `H5VL__native_blob_get` refuses to read past.
8583 fn write_heap_reference_values(&mut self) -> IoResult<()> {
8584 use crate::format::global_heap::GlobalHeapCollection;
8585
8586 // Snapshot rather than drain, for the same reason the object-reference
8587 // pass does: a SWMR session finalizes twice and the close-time finalize
8588 // rebuilds every header at a fresh address.
8589 let pending: Vec<(u64, u16, usize, PendingHeapTarget)> = self
8590 .pending_heap_references
8591 .lock()
8592 .iter()
8593 .map(|p| (p.collection, p.index, p.token_offset, p.target.clone()))
8594 .collect();
8595 if pending.is_empty() {
8596 return Ok(());
8597 }
8598 let sa = self.ctx.sizeof_addr as usize;
8599 // Group by collection so one holding several references is read and
8600 // rewritten once.
8601 let mut per_collection: std::collections::BTreeMap<u64, Vec<(u16, usize, u64)>> =
8602 Default::default();
8603 for (collection, index, token_offset, target) in &pending {
8604 let addr = match target {
8605 PendingHeapTarget::Dataset(path) => {
8606 let ds = self.region_reference_target(path)?;
8607 self.ds(ds).lock().obj_header_addr
8608 }
8609 PendingHeapTarget::Object(path) => self.object_reference_address(path)?,
8610 };
8611 per_collection
8612 .entry(*collection)
8613 .or_default()
8614 .push((*index, *token_offset, addr));
8615 }
8616 for (collection, patches) in per_collection {
8617 // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
8618 // exactly that, so one read usually covers the whole image.
8619 let mut image = self.handle.read_at_most(collection, 4096)?;
8620 let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
8621 if declared > image.len() {
8622 image = self.handle.read_at(collection, declared)?;
8623 }
8624 let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
8625 for (index, token_offset, addr) in patches {
8626 let token = gcol
8627 .objects
8628 .iter_mut()
8629 .find(|o| o.index == index)
8630 .and_then(|o| o.data.get_mut(token_offset..token_offset + sa))
8631 .ok_or_else(|| {
8632 crate::io::IoError::InvalidState(format!(
8633 "object {index} of global heap collection {collection:#x} is no \
8634 longer the reference written into it"
8635 ))
8636 })?;
8637 token.copy_from_slice(&addr.to_le_bytes()[..sa]);
8638 }
8639 let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
8640 self.handle.write_at(collection, &rewritten)?;
8641 }
8642 Ok(())
8643 }
8644
8645 /// Give every reference written this session its target's object header
8646 /// address.
8647 ///
8648 /// INVARIANT: no file is closed holding a reference whose target address is
8649 /// still the placeholder its write left. Both finalize paths call this in
8650 /// the content phase — after
8651 /// [`allocate_object_headers`](Self::allocate_object_headers), so every
8652 /// address exists, and before any object header is written — and this is
8653 /// the only caller of the per-kind passes, so a reference kind added later
8654 /// is written at both finalize sites or at neither. A target that no longer
8655 /// resolves fails the finalize rather than leaving a placeholder behind.
8656 ///
8657 /// This covers the two reference kinds whose value lives outside an object
8658 /// header. An attribute's value lives *inside* one, so it has no pass here:
8659 /// [`object_attributes`](Self::object_attributes) says it in addresses as
8660 /// the header is built.
8661 fn write_reference_values(&mut self) -> IoResult<()> {
8662 self.write_object_reference_values()?;
8663 self.write_heap_reference_values()
8664 }
8665
8666 /// Append every user-created hard link whose parent group is `parent`
8667 /// (`None` == the root group). Called while collecting a group's links,
8668 /// once every object's header address has been assigned.
8669 fn push_hard_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8670 for link in self.hard_links_vec() {
8671 if link.parent != parent || !self.hard_link_emitted(&link) {
8672 continue;
8673 }
8674 let addr = match link.target {
8675 HardLinkTarget::Dataset(i) => self.ds(i).lock().obj_header_addr,
8676 HardLinkTarget::Group(i) => self.grp(i).lock().obj_header_addr,
8677 };
8678 links.push((link.creation_seq, LinkMessage::hard(&link.name, addr)));
8679 }
8680 }
8681
8682 /// Append every user-created symbolic link whose parent group is `parent`
8683 /// (`None` == the root group).
8684 ///
8685 /// Nothing here waits on the layout pass — the link's value is a path, not
8686 /// an address — but it is collected with the rest so it takes its place in
8687 /// creation order and counts toward the phase change.
8688 fn push_symbolic_links(&self, links: &mut Vec<(u64, LinkMessage)>, parent: Option<usize>) {
8689 for link in self.symbolic_links_vec() {
8690 if link.parent != parent || !self.symbolic_link_emitted(&link) {
8691 continue;
8692 }
8693 links.push((
8694 link.creation_seq,
8695 LinkMessage {
8696 name: link.name.clone(),
8697 target: link.target.clone(),
8698 creation_order: None,
8699 cset: CharacterSet::for_name(&link.name),
8700 },
8701 ));
8702 }
8703 }
8704
8705 /// Refuse a caller path that would have to leave this file through one of
8706 /// the external links a reopened file brought in.
8707 ///
8708 /// The reader follows such a path into the file the link names; the writer
8709 /// cannot, because it models one file and would have to write into
8710 /// another. Saying which link stops the path — rather than reporting the
8711 /// name as absent, or worse, creating a second link of that name beside
8712 /// it — is the whole of what write mode does here.
8713 pub(crate) fn reject_external_traversal(&self, path: &str) -> IoResult<()> {
8714 let path = path.trim_start_matches('/');
8715 let crossing = self.preserved_link_paths().into_iter().find(|(p, class)| {
8716 matches!(class, crate::io::reader::LinkClass::External { .. })
8717 && (path == p || path.starts_with(&format!("{p}/")))
8718 });
8719 match crossing {
8720 None => Ok(()),
8721 Some((link, crate::io::reader::LinkClass::External { file, path: target })) => {
8722 Err(crate::io::IoError::Unsupported(format!(
8723 "'{path}' resolves through the external link '{link}' to '{target}' in \
8724 '{file}'; this writer carries external links through a rewrite but does \
8725 not open the file they name"
8726 )))
8727 }
8728 // `find` matched on the External arm, so no other class reaches here.
8729 Some(_) => Ok(()),
8730 }
8731 }
8732
8733 /// Resolve `name` to a live dataset index, reporting *why* it does not
8734 /// resolve rather than collapsing every cause into absence.
8735 ///
8736 /// The write-mode counterpart of [`Hdf5Reader::open_dataset`]: the single
8737 /// gate every by-name dataset lookup in write mode goes through.
8738 ///
8739 /// [`Hdf5Reader::open_dataset`]: crate::io::reader::Hdf5Reader::open_dataset
8740 pub(crate) fn open_dataset_index(&self, name: &str) -> IoResult<usize> {
8741 self.reject_external_traversal(name)?;
8742 self.reject_preserved_object(name)?;
8743 self.dataset_index(name)
8744 .ok_or_else(|| crate::io::IoError::NotFound(name.to_string()))
8745 }
8746
8747 /// Refuse a caller path that names an object the reopen kept by its bytes
8748 /// rather than modelling.
8749 ///
8750 /// Such an object is in the file and stays in it, but this writer holds
8751 /// none of what it would need to read or rewrite it. Saying so — with the
8752 /// reason the classification recorded — is the difference between an
8753 /// object the writer will not touch and a name the file does not have.
8754 pub(crate) fn reject_preserved_object(&self, path: &str) -> IoResult<()> {
8755 let path = path.trim_start_matches('/');
8756 let objects: Vec<(String, String)> = {
8757 let preserved = self.preserved_links.lock();
8758 preserved
8759 .iter()
8760 .filter_map(|l| {
8761 l.reason
8762 .as_ref()
8763 .map(|why| (self.preserved_link_full_path(l), why.clone()))
8764 })
8765 .collect()
8766 };
8767 match objects
8768 .into_iter()
8769 .find(|(full, _)| path == full || path.starts_with(&format!("{full}/")))
8770 {
8771 None => Ok(()),
8772 Some((link, why)) => Err(crate::io::IoError::Unsupported(format!(
8773 "'{path}' is, or is inside, the object '{link}', which this file's reopen \
8774 kept exactly as it found it because {why}"
8775 ))),
8776 }
8777 }
8778
8779 /// Every link this writer will emit that names a *path* rather than an
8780 /// object, with the class a listing reports for it: the soft and external
8781 /// links created this session, and the ones a reopen is carrying through.
8782 ///
8783 /// The object listings answer for hard links, so a write-mode link
8784 /// listing is this plus those; keeping both sources in one place is what
8785 /// stops a listing from seeing a kind the class lookup does not, or the
8786 /// reverse.
8787 pub(crate) fn path_link_classes(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8788 let mut out: Vec<(String, crate::io::reader::LinkClass)> = self
8789 .symbolic_links_vec()
8790 .iter()
8791 .filter(|l| self.symbolic_link_emitted(l))
8792 .map(|l| {
8793 (
8794 self.symbolic_link_full_path(l),
8795 crate::io::reader::LinkClass::from_target(&l.target),
8796 )
8797 })
8798 .collect();
8799 out.extend(self.preserved_link_paths());
8800 out
8801 }
8802
8803 /// Every link this writer is carrying but cannot express, by full path.
8804 pub(crate) fn preserved_link_paths(&self) -> Vec<(String, crate::io::reader::LinkClass)> {
8805 self.preserved_links
8806 .lock()
8807 .iter()
8808 .map(|l| (self.preserved_link_full_path(l), l.class.clone()))
8809 .collect()
8810 }
8811
8812 /// The full path of a preserved link: its parent group's path plus its
8813 /// leaf name, in the no-leading-`/` form the registry uses.
8814 fn preserved_link_full_path(&self, link: &PreservedLink) -> String {
8815 match link.parent {
8816 None => link.name.clone(),
8817 Some(gi) => {
8818 let group = self.grp(gi).lock().name.clone();
8819 format!("{}/{}", group.trim_start_matches('/'), link.name)
8820 }
8821 }
8822 }
8823
8824 /// The single owner of "which links does this group hold", in the order
8825 /// they were created and, when the file tracks creation order, stamped
8826 /// with it.
8827 ///
8828 /// Both the compact form (one `MSG_LINK` per link) and the dense form (the
8829 /// same messages inside a fractal heap) are built from this one list, so
8830 /// the phase-change decision, the storage it selects and the creation
8831 /// order recorded in either can never disagree about what the group
8832 /// contains.
8833 fn group_links(&self, scope: LinkScope, order: CreationOrder) -> Vec<LinkMessage> {
8834 let mut links: Vec<(u64, LinkMessage)> = Vec::new();
8835 match scope {
8836 LinkScope::Root => {
8837 // Datasets that belong to a subgroup are that group's links,
8838 // not the root's. Each group slot is locked one at a time.
8839 let mut datasets_in_subgroups: std::collections::HashSet<usize> =
8840 std::collections::HashSet::new();
8841 for grp in self.group_refs() {
8842 let g = grp.lock();
8843 if g.deleted {
8844 continue;
8845 }
8846 datasets_in_subgroups.extend(g.child_datasets.iter().copied());
8847 }
8848 // `dataset_refs` preserves registry order, so `enumerate`
8849 // yields each dataset's true index.
8850 for (i, ds) in self.dataset_refs().into_iter().enumerate() {
8851 let m = ds.lock();
8852 if m.deleted || datasets_in_subgroups.contains(&i) {
8853 continue;
8854 }
8855 // The leaf, never the registry path: a link name is one
8856 // path component, and `H5G_traverse` would split a '/'
8857 // in it before `H5L_link` ever saw the name.
8858 let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8859 links.push((
8860 m.creation_seq,
8861 LinkMessage::hard(leaf_name, m.obj_header_addr),
8862 ));
8863 }
8864 for grp in self.group_refs() {
8865 let g = grp.lock();
8866 if g.deleted || g.parent.is_some() {
8867 continue;
8868 }
8869 let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8870 links.push((
8871 g.creation_seq,
8872 LinkMessage::hard(leaf_name, g.obj_header_addr),
8873 ));
8874 }
8875 self.push_hard_links(&mut links, None);
8876 self.push_symbolic_links(&mut links, None);
8877 self.push_committed_datatypes(&mut links, None);
8878 }
8879 LinkScope::Group(group_idx) => {
8880 // Snapshot the child lists, then drop the slot guard: the
8881 // per-child reads below re-lock dataset and group slots
8882 // (including this one).
8883 let (child_datasets, child_groups) = {
8884 let grp = self.grp(group_idx);
8885 let g = grp.lock();
8886 (g.child_datasets.clone(), g.child_groups.clone())
8887 };
8888 for ds_idx in child_datasets {
8889 let ds = self.ds(ds_idx);
8890 let m = ds.lock();
8891 if m.deleted {
8892 continue;
8893 }
8894 let leaf_name = m.name.rsplit('/').next().unwrap_or(&m.name);
8895 links.push((
8896 m.creation_seq,
8897 LinkMessage::hard(leaf_name, m.obj_header_addr),
8898 ));
8899 }
8900 for child_idx in child_groups {
8901 let child_grp = self.grp(child_idx);
8902 let g = child_grp.lock();
8903 if g.deleted {
8904 continue;
8905 }
8906 let leaf_name = g.name.rsplit('/').next().unwrap_or(&g.name);
8907 links.push((
8908 g.creation_seq,
8909 LinkMessage::hard(leaf_name, g.obj_header_addr),
8910 ));
8911 }
8912 self.push_hard_links(&mut links, Some(group_idx));
8913 self.push_symbolic_links(&mut links, Some(group_idx));
8914 self.push_committed_datatypes(&mut links, Some(group_idx));
8915 }
8916 }
8917 // Creation order, not order by kind: a run of create_group and
8918 // create_dataset draws from one counter, so this is the order the
8919 // caller made them in. `H5G_obj_insert` numbers from zero within the
8920 // group, so the rank here is the link's creation order.
8921 links.sort_by_key(|(seq, _)| *seq);
8922 links
8923 .into_iter()
8924 .enumerate()
8925 .map(|(rank, (_, link))| {
8926 if order.is_tracked() {
8927 link.with_creation_order(rank as i64)
8928 } else {
8929 link
8930 }
8931 })
8932 .collect()
8933 }
8934
8935 /// Whether `links` must live in dense storage rather than in the group's
8936 /// object header — the `H5G_obj_insert` phase-change rule, applied to the
8937 /// whole set at once because this writer builds each header from scratch
8938 /// rather than inserting one link at a time.
8939 ///
8940 /// libhdf5 converts when the count *reaches* `max_compact` and another
8941 /// link arrives, so a set of exactly `max_compact` is still compact; and
8942 /// separately when one message would not fit the 16-bit size field an
8943 /// object header message has.
8944 ///
8945 /// The answer depends only on the link names and kinds, never on the
8946 /// addresses they point at, which is what lets a group header be sized
8947 /// before [`prepare_dense_links`](Self::prepare_dense_links) has run.
8948 fn links_need_dense(&self, links: &[LinkMessage]) -> bool {
8949 links.len() > MAX_COMPACT_LINKS
8950 || links
8951 .iter()
8952 .any(|l| l.encode(&self.ctx).len() > MAX_MESSAGE_SIZE)
8953 }
8954
8955 /// The single owner of link emission into a group object header: the Link
8956 /// Info and Group Info messages, and then either one `MSG_LINK` per link
8957 /// or nothing at all when the set has spilled to dense storage.
8958 ///
8959 /// The two storage forms are exclusive (`H5G_obj_insert` moves the whole
8960 /// set at once), and a header carrying both would report every link twice.
8961 ///
8962 /// A group whose links are dense but not yet laid out gets a compact Link
8963 /// Info message here. That is deliberate: the message encodes to the same
8964 /// length either way — two addresses, defined or not — so the sizing pass
8965 /// that runs before `prepare_dense_links` still reserves the right number
8966 /// of bytes, and the write pass that runs after it emits the real heap and
8967 /// index addresses. It is the same two-pass rule the child link addresses
8968 /// already follow.
8969 fn emit_links(
8970 &self,
8971 header: &mut ObjectHeader,
8972 scope: LinkScope,
8973 links: &[LinkMessage],
8974 order: CreationOrder,
8975 ) {
8976 // A symbol-table group holds no link messages at all: its links are the
8977 // entries of the symbol table `prepare_symbol_tables` laid out, and
8978 // the header carries only the two addresses naming it. Link Info and
8979 // Group Info are version-1.8 messages and have no business in a
8980 // version-1 header — `H5G__stab_valid` reads the Symbol Table message
8981 // and nothing else.
8982 if self.uses_symbol_table(scope, order) {
8983 // Sizing runs before the tables are laid out; the message is the
8984 // same two addresses wide either way, so the placeholder reserves
8985 // exactly what the real one needs. Same two-pass rule the child
8986 // link addresses already follow.
8987 let stab = self
8988 .symbol_tables
8989 .written
8990 .lock()
8991 .get(&scope)
8992 .copied()
8993 .unwrap_or(Stab {
8994 btree_addr: UNDEF_ADDR,
8995 heap_addr: UNDEF_ADDR,
8996 });
8997 header.add_message(MSG_SYMBOL_TABLE, 0x00, stab.encode(&self.ctx));
8998 return;
8999 }
9000 // Links a reopen carried through verbatim because this writer cannot
9001 // express them. They are emitted here rather than by a second caller
9002 // so that no header-rewrite path can drop them, and their presence
9003 // pins the group to compact storage: dense storage would have to
9004 // re-encode each link into the heap, which is exactly the byte
9005 // fidelity preserving them is for.
9006 let preserved = self.preserved_links_for(scope);
9007 let dense = preserved.is_empty() && self.links_need_dense(links);
9008 let link_info = self.dense_links.lock().get(&scope).cloned();
9009 let link_info = link_info.unwrap_or_else(|| {
9010 let mut info = LinkInfoMessage::compact();
9011 if order.is_tracked() {
9012 // `H5G__obj_insert` post-increments `max_corder`, so a group
9013 // holding n links reports n.
9014 info.max_creation_order = Some(links.len() as u64);
9015 }
9016 if order.is_indexed() {
9017 // The index address stays undefined while the links live in
9018 // the header, but the message must still carry the field:
9019 // `H5Pget_link_creation_order` reads INDEXED off this flag,
9020 // not off the address.
9021 info.creation_order_btree_address = Some(UNDEF_ADDR);
9022 }
9023 info
9024 });
9025 header.add_message(MSG_LINK_INFO, 0x00, link_info.encode(&self.ctx));
9026 // The link info message takes no flags and the group info message
9027 // takes `H5O_MSG_FLAG_CONSTANT`, exactly as `H5G__obj_create_real`
9028 // creates the pair (H5Gobj.c:255, :259) and as
9029 // `H5G__obj_insert`'s phase change re-creates it (H5Gobj.c:526). The
9030 // asymmetry is real: the link info message records the group's
9031 // storage and its creation-order counter, both of which change as
9032 // links come and go, while the group info message holds the phase
9033 // change and estimated-name-length constants of the creation property
9034 // list, which nothing after creation rewrites.
9035 header.add_message(
9036 MSG_GROUP_INFO,
9037 MSG_FLAG_CONSTANT,
9038 GroupInfoMessage::default().encode(),
9039 );
9040 if dense {
9041 return;
9042 }
9043 for link in links {
9044 header.add_message(MSG_LINK, 0x00, link.encode(&self.ctx));
9045 }
9046 for encoded in preserved {
9047 header.add_message(MSG_LINK, 0x00, encoded);
9048 }
9049 }
9050
9051 /// The verbatim link bodies a reopen carried into `scope`.
9052 fn preserved_links_for(&self, scope: LinkScope) -> Vec<Vec<u8>> {
9053 let parent = match scope {
9054 LinkScope::Root => None,
9055 LinkScope::Group(i) => Some(i),
9056 };
9057 self.preserved_links
9058 .lock()
9059 .iter()
9060 .filter(|l| l.parent == parent)
9061 .map(|l| l.encoded.clone())
9062 .collect()
9063 }
9064
9065 /// Lay out and write dense link storage for every group that needs it,
9066 /// recording the resulting `Link Info` message per group.
9067 ///
9068 /// The sole owner of that transition. It must run after every object
9069 /// header address is assigned — the heap holds encoded link messages, and
9070 /// those name their targets — and before any group header is written.
9071 ///
9072 /// Every group whose header this finalize rewrites passes through here,
9073 /// dense or not: the storage a reopened header named is superseded by the
9074 /// rewrite whichever form the new link set takes, and freeing it first is
9075 /// what lets the replacement reuse those blocks.
9076 fn prepare_dense_links(&self) -> IoResult<()> {
9077 let mut scopes: Vec<(LinkScope, Vec<LinkMessage>, CreationOrder)> = Vec::new();
9078 for gi in 0..self.group_count() {
9079 let (deleted, order) = {
9080 let grp = self.grp(gi);
9081 let g = grp.lock();
9082 (g.deleted, g.track_order.links)
9083 };
9084 // A symbol-table group is `prepare_symbol_tables`' business; it
9085 // has no Link Info message to hold a fractal heap address, and it
9086 // never had dense storage to release.
9087 if deleted || self.uses_symbol_table(LinkScope::Group(gi), order) {
9088 continue;
9089 }
9090 self.release_superseded_dense_links(LinkScope::Group(gi))?;
9091 let links = self.group_links(LinkScope::Group(gi), order);
9092 if self.links_need_dense(&links) {
9093 scopes.push((LinkScope::Group(gi), links, order));
9094 }
9095 }
9096 let root_order = self.root_track_order.links;
9097 if !self.uses_symbol_table(LinkScope::Root, root_order) {
9098 self.release_superseded_dense_links(LinkScope::Root)?;
9099 let root_links = self.group_links(LinkScope::Root, root_order);
9100 if self.links_need_dense(&root_links) {
9101 scopes.push((LinkScope::Root, root_links, root_order));
9102 }
9103 }
9104
9105 for (scope, links, order) in scopes {
9106 // `close` after `start_swmr` finalizes a second time over the same
9107 // groups, so rebuilding here would allocate a whole second heap
9108 // and strand the one the published headers already name.
9109 if self.dense_links.lock().contains_key(&scope) {
9110 continue;
9111 }
9112 let dense = build_dense_links(&links, &self.ctx, order, &mut |len| {
9113 self.allocator.allocate(len, FreeSpaceClass::Metadata)
9114 })?;
9115 for block in &dense.blocks {
9116 self.handle.write_at(block.addr, &block.image)?;
9117 }
9118 self.dense_links.lock().insert(scope, dense.linfo);
9119 }
9120 Ok(())
9121 }
9122
9123 /// Lay out whichever of the two forms of link storage this file uses,
9124 /// before any group header is written.
9125 ///
9126 /// The two are exclusive because the formats are: a classic group has no
9127 /// Link Info message to put a fractal heap address in, and a link-message
9128 /// group has no symbol table.
9129 fn prepare_link_storage(&self) -> IoResult<()> {
9130 self.prepare_dense_links()?;
9131 self.prepare_symbol_tables()
9132 }
9133
9134 /// Lay out and write the symbol table of every classic group, and free the
9135 /// storage each rewrite supersedes. A no-op on a link-message file.
9136 ///
9137 /// The classic counterpart of [`prepare_dense_links`](Self::prepare_dense_links),
9138 /// and the sole owner of that transition. The same two placement rules
9139 /// apply for the same two reasons: it runs after every object header has
9140 /// an address, because a symbol table entry names its target's header, and
9141 /// before any group header is written, because the header carries the
9142 /// Symbol Table message naming what this laid out.
9143 ///
9144 /// Deepest group first, root last. A hard link to a group caches that
9145 /// group's own B-tree and heap in the entry's scratch pad
9146 /// (`H5G__link_to_ent`), so the child's table must exist before the
9147 /// parent's is built; `H5G__stab_valid` checks the root entry's cache
9148 /// against the root header's Symbol Table message, so a stale pair there
9149 /// is not a slow lookup but a file `H5Fopen` rejects.
9150 ///
9151 /// Every classic group is rebuilt on every pass — there is no "already
9152 /// done" short-circuit like the dense one, because the only way this runs
9153 /// twice is a `Drop` retry after a failed `close`, and the entries of the
9154 /// first pass name header addresses the second pass has moved. (A SWMR
9155 /// session, the other double-finalize, cannot reach here: SWMR needs a
9156 /// version-3 superblock, so `start_swmr` refuses a classic file.)
9157 fn prepare_symbol_tables(&self) -> IoResult<()> {
9158 // Depth by parent chain, not by counting separators in the registry
9159 // path: the chain is what actually says which table has to exist first.
9160 let mut scopes: Vec<(usize, LinkScope, CreationOrder)> = Vec::new();
9161 for gi in 0..self.group_count() {
9162 let (deleted, order, mut parent) = {
9163 let grp = self.grp(gi);
9164 let g = grp.lock();
9165 (g.deleted, g.track_order.links, g.parent)
9166 };
9167 if deleted || !self.uses_symbol_table(LinkScope::Group(gi), order) {
9168 continue;
9169 }
9170 let mut depth = 1usize;
9171 while let Some(p) = parent {
9172 depth += 1;
9173 parent = self.grp(p).lock().parent;
9174 }
9175 scopes.push((depth, LinkScope::Group(gi), order));
9176 }
9177 scopes.sort_by_key(|&(depth, ..)| std::cmp::Reverse(depth));
9178 let root_order = self.root_track_order.links;
9179 if self.uses_symbol_table(LinkScope::Root, root_order) {
9180 scopes.push((0, LinkScope::Root, root_order));
9181 }
9182
9183 let meta = self.stab_meta();
9184 for (_, scope, order) in scopes {
9185 // Freed before the replacement is laid out, so a rewrite reuses
9186 // the same blocks instead of growing the file on every open/close
9187 // cycle — the rule `prepare_dense_links` and the header rewrite
9188 // already follow. Removed as it is freed, so no second pass can
9189 // free it twice.
9190 let superseded = self.symbol_tables.superseded.lock().remove(&scope);
9191 if let Some(extents) = superseded {
9192 free_stab(&self.allocator, &extents);
9193 }
9194 let links = self.stab_links_for(scope, order)?;
9195 let stab = write_stab(&self.handle, &self.allocator, &meta, &links)?;
9196 self.symbol_tables.written.lock().insert(scope, stab);
9197 }
9198 Ok(())
9199 }
9200
9201 /// The file-level parameters every symbol-table node width is derived from
9202 /// — the address/length widths and the B-tree "K" ranks. Only a version-0/1
9203 /// superblock records ranks of its own; [`btree_v1_config`] is the one
9204 /// place that decides whether this file has any.
9205 ///
9206 /// [`btree_v1_config`]: Self::btree_v1_config
9207 fn stab_meta(&self) -> FileMeta {
9208 FileMeta {
9209 ctx: self.ctx,
9210 btree: self.btree_v1_config(),
9211 sohm: None,
9212 }
9213 }
9214
9215 /// `scope`'s links as symbol table entries.
9216 ///
9217 /// A link a reopen carried through verbatim is decoded back out of its
9218 /// encoded Link message here, because a classic group has no link message
9219 /// to preserve it into. Nothing is lost in the round trip: the walk built
9220 /// that message from a symbol table entry in the first place, and the two
9221 /// forms carry the same three facts.
9222 fn stab_links_for(&self, scope: LinkScope, order: CreationOrder) -> IoResult<Vec<StabLink>> {
9223 let groups = self.group_header_scopes();
9224 let mut out = Vec::new();
9225 for link in self.group_links(scope, order) {
9226 out.push(self.stab_link(&link, &groups)?);
9227 }
9228 for encoded in self.preserved_links_for(scope) {
9229 let (link, _) = LinkMessage::decode(&encoded, &self.ctx)?;
9230 out.push(self.stab_link(&link, &groups)?);
9231 }
9232 Ok(out)
9233 }
9234
9235 /// Where each group's object header now sits, so a hard link that lands on
9236 /// one can cache that group's symbol table in its scratch pad.
9237 fn group_header_scopes(&self) -> HashMap<u64, LinkScope> {
9238 let mut map = HashMap::new();
9239 for gi in 0..self.group_count() {
9240 let grp = self.grp(gi);
9241 let g = grp.lock();
9242 if !g.deleted {
9243 map.insert(g.obj_header_addr, LinkScope::Group(gi));
9244 }
9245 }
9246 map
9247 }
9248
9249 /// One link as a symbol table entry.
9250 ///
9251 /// The scratch pad caches the target group's B-tree and heap when the
9252 /// target is a group this pass has already laid out — what
9253 /// `H5G__link_to_ent` does, and what lets `H5G__stab_lookup` walk a path
9254 /// without opening each header on the way. For anything else the pad stays
9255 /// `H5G_NOTHING_CACHED`, the value libhdf5 itself writes whenever the
9256 /// target has no Symbol Table message to read.
9257 fn stab_link(
9258 &self,
9259 link: &LinkMessage,
9260 groups: &HashMap<u64, LinkScope>,
9261 ) -> IoResult<StabLink> {
9262 let target = match &link.target {
9263 LinkTarget::Hard { address } => {
9264 let cached = groups
9265 .get(address)
9266 .and_then(|scope| self.symbol_tables.written.lock().get(scope).copied());
9267 StabTarget::Hard {
9268 addr: *address,
9269 cached,
9270 }
9271 }
9272 LinkTarget::Soft { target } => StabTarget::Soft {
9273 value: target.clone(),
9274 },
9275 // Unreachable by construction: a group holding one of these is
9276 // not a symbol-table group at all
9277 // ([`LinkMessage::fits_symbol_table`] is what
9278 // [`Hdf5Writer::uses_symbol_table`] asks), so this pass never
9279 // visits it. Reported rather than panicked so a future caller
9280 // that skips that gate learns which link it lost.
9281 LinkTarget::External { .. } | LinkTarget::UserDefined { .. } => {
9282 return Err(crate::io::IoError::InvalidState(format!(
9283 "cannot store the link {:?} in a symbol table: it holds only \
9284 hard and soft links, and this group was not converted to link \
9285 messages the way `H5G_obj_insert` converts it",
9286 link.name
9287 )))
9288 }
9289 };
9290 Ok(StabLink {
9291 name: link.name.clone(),
9292 target,
9293 })
9294 }
9295
9296 /// The single owner of attribute emission into an object header: appends
9297 /// the Attribute Info message and then one `MSG_ATTRIBUTE` per attribute.
9298 ///
9299 /// On a version-2 object header the two are inseparable.
9300 /// `H5O__attr_count_real` derives `H5Oget_info().num_attrs` from the
9301 /// Attribute Info message alone — with no such message the count reads as
9302 /// zero however many attribute messages follow, which is what made every
9303 /// rust-written file report `num_attrs == 0` to libhdf5 while
9304 /// `H5Aiterate2` still yielded the attributes. The message carries no
9305 /// count of its own: `H5A__get_ainfo` fills `nattrs` from the attribute
9306 /// messages the header loader actually saw, so compact storage needs
9307 /// nothing but the message's presence.
9308 ///
9309 /// When [`prepare_dense_attributes`](Self::prepare_dense_attributes) has
9310 /// spilled `scope`'s attributes to a fractal heap, the same message names
9311 /// that heap instead and *no* attribute message follows: the two storage
9312 /// forms are exclusive (`H5O__attr_create` moves the whole set at once),
9313 /// and a header carrying both would report every attribute twice.
9314 fn emit_attributes(
9315 &self,
9316 header: &mut ObjectHeader,
9317 scope: AttrScope,
9318 attributes: &[AttributeEntry],
9319 order: CreationOrder,
9320 format: ObjectFormat,
9321 owner: ShareOwner,
9322 ) {
9323 // `H5Pget_attr_creation_order` reads the object header's own flags,
9324 // not the Attribute Info message, so this is what makes the object
9325 // report creation-ordered attributes — and tracking widens every
9326 // message envelope by the creation index below.
9327 let order = self.header_attr_order(order);
9328 header.set_attribute_creation_order(order);
9329 if attributes.is_empty() {
9330 return;
9331 }
9332 // A version-1 object header gets the attribute messages alone.
9333 // `H5O__attr_create` gates every mention of the Attribute Info message
9334 // on `oh->version > H5O_VERSION_1` (H5Oattribute.c:218), and so does
9335 // `H5O__attr_count_real`, which is why the count still reads correctly
9336 // without it: on a version-1 header libhdf5 counts the messages.
9337 if format == ObjectFormat::Legacy {
9338 for attr in attributes {
9339 header.add_message(MSG_ATTRIBUTE, 0x00, self.encode_attribute(attr));
9340 }
9341 return;
9342 }
9343 // Whether the set spills is a property of the set alone, so it is the
9344 // same answer in the pass that measures this header and in the pass
9345 // that writes it — even though the storage itself is laid out between
9346 // the two, because it can only be laid out once every object header
9347 // has an address. Sizing therefore falls back to a placeholder message
9348 // of the same width: only the creation-order flags change the
9349 // Attribute Info message's length, so the header measured here holds
9350 // the header written against the storage that replaces it. Same
9351 // two-pass rule `emit_links` follows for dense links and symbol
9352 // tables.
9353 let dense = self.attributes_need_dense(attributes, format);
9354 let stored = self.dense_attributes.lock().get(&scope).cloned();
9355 let ainfo = stored.unwrap_or_else(|| {
9356 let mut ainfo = AttributeInfoMessage::compact();
9357 if order.is_tracked() {
9358 ainfo.max_creation_index = Some(next_creation_index(attributes));
9359 }
9360 if order.is_indexed() {
9361 // Compact storage has no index B-tree, but the message still
9362 // announces one so that its flags match the header's
9363 // (`H5O__attr_create` asserts they agree).
9364 ainfo.creation_order_btree_address = Some(UNDEF_ADDR);
9365 }
9366 ainfo
9367 });
9368 header.add_message(MSG_ATTR_INFO, MSG_FLAG_DONTSHARE, ainfo.encode(&self.ctx));
9369 if dense {
9370 return;
9371 }
9372 // Each attribute states its own creation index — the one it was
9373 // created with here, or the one the file it was read from records. An
9374 // attribute with none belongs to an object that tracks no order, where
9375 // the field is not encoded at all.
9376 for attr in attributes {
9377 let (flags, body) = self.share_attribute(attr, format, owner);
9378 header.add_message_indexed(
9379 MSG_ATTRIBUTE,
9380 flags,
9381 body,
9382 attr.creation_index().unwrap_or(0),
9383 );
9384 }
9385 }
9386
9387 /// One attribute message body, at the version this file's low library
9388 /// bound calls for (`H5A__set_version`, which reads the bound and nothing
9389 /// about the object the attribute hangs on).
9390 fn encode_attribute(&self, attr: &AttributeEntry) -> Vec<u8> {
9391 attr.encode_for(&self.ctx, self.encoding_libver(), self.message_format())
9392 }
9393
9394 /// What a header stores for one attribute: the message flags and the body,
9395 /// with the attribute's own datatype and dataspace shared wherever an
9396 /// index covers them.
9397 ///
9398 /// `H5A__create` offers both to `H5SM_try_share` (H5Aint.c:375-377) before
9399 /// `H5O__attr_create` offers the attribute itself (H5Oattribute.c:726), so
9400 /// the attribute body that reaches the heap already holds their pointers
9401 /// and says which fields they are in its own flags byte
9402 /// (`H5O_ATTR_FLAG_TYPE_SHARED` / `H5O_ATTR_FLAG_SPACE_SHARED`,
9403 /// H5Oattr.c:358-359). Both offers go through
9404 /// [`share_message`](Self::share_message) like any other, so the pass that
9405 /// counts references and the pass that substitutes see the same three
9406 /// messages.
9407 fn share_attribute(
9408 &self,
9409 attr: &AttributeEntry,
9410 format: ObjectFormat,
9411 owner: ShareOwner,
9412 ) -> (u8, Vec<u8>) {
9413 let libver = self.encoding_libver();
9414 // Only a readable attribute has pieces to offer: an unreadable one is
9415 // the bytes it was read from, put back as they were. Version 1 has no
9416 // flags byte to record a shared field in — `H5O__attr_encode` writes a
9417 // reserved zero there — so a classic file shares the attribute whole
9418 // or not at all.
9419 let Some(message) = attr.readable().filter(|_| format.attribute_version() >= 2) else {
9420 return self.share_message(
9421 owner,
9422 MSG_ATTRIBUTE,
9423 0x00,
9424 attr.encode_for(&self.ctx, libver, format),
9425 );
9426 };
9427
9428 let datatype = message.datatype.encode_at(&self.ctx, libver);
9429 let dataspace = message.dataspace.encode_for(&self.ctx, format);
9430 // `H5A__create` passes no open header for either (H5Aint.c:375-377):
9431 // both live inside the attribute's body, so neither has a header
9432 // message a `H5SM_IN_OH` record could name and both reach the heap on
9433 // first use.
9434 let (dt_flags, dt_field) =
9435 self.share_message(ShareOwner::Detached, MSG_DATATYPE, 0x00, datatype.clone());
9436 let (ds_flags, ds_field) =
9437 self.share_message(ShareOwner::Detached, MSG_DATASPACE, 0x00, dataspace.clone());
9438
9439 let mut attr_flags = 0u8;
9440 if dt_flags & MSG_FLAG_SHARED != 0 {
9441 attr_flags |= ATTR_FLAG_TYPE_SHARED;
9442 }
9443 if ds_flags & MSG_FLAG_SHARED != 0 {
9444 attr_flags |= ATTR_FLAG_SPACE_SHARED;
9445 }
9446 let encoded = message.encode_with_fields(attr_flags, &dt_field, &ds_field);
9447
9448 // Each shared field's heap ID sits two bytes into the pointer that
9449 // replaced it; the body offered below carries whatever
9450 // `share_message` just produced, which is a zeroed ID in the pass that
9451 // counts and the real one in the pass that substitutes.
9452 let mut nested = Vec::new();
9453 if attr_flags & ATTR_FLAG_TYPE_SHARED != 0 {
9454 nested.push(NestedShare {
9455 heap_id_at: encoded.datatype_at + SOHM_POINTER_HEAP_ID_AT,
9456 target: (MSG_DATATYPE, datatype),
9457 });
9458 }
9459 if attr_flags & ATTR_FLAG_SPACE_SHARED != 0 {
9460 nested.push(NestedShare {
9461 heap_id_at: encoded.dataspace_at + SOHM_POINTER_HEAP_ID_AT,
9462 target: (MSG_DATASPACE, dataspace),
9463 });
9464 }
9465 self.share_nesting_message(owner, MSG_ATTRIBUTE, 0x00, encoded.body, nested)
9466 }
9467
9468 /// Whether `attributes` must live in dense storage rather than in the
9469 /// object header — the `H5O__attr_create` phase-change rule, applied to
9470 /// the whole set at once because this writer builds each header from
9471 /// scratch rather than inserting one attribute at a time.
9472 ///
9473 /// libhdf5 converts when the count *reaches* `max_compact` and another
9474 /// attribute arrives, so a set of exactly `max_compact` is still compact;
9475 /// and separately when one message would not fit the 16-bit size field an
9476 /// object header message has.
9477 ///
9478 /// Never in a classic file. Dense attribute storage is a fractal heap
9479 /// reached through an Attribute Info message, both introduced in the 1.8
9480 /// format; at `H5F_LIBVER_EARLIEST` libhdf5 keeps every attribute in the
9481 /// header however many there are (`H5O__attr_create` reaches the phase
9482 /// change only when the object header version allows it). An attribute
9483 /// too large for the 16-bit size field is then an error, which
9484 /// `ObjectHeader::encode_v1` raises, rather than a reason to spill.
9485 fn attributes_need_dense(&self, attributes: &[AttributeEntry], format: ObjectFormat) -> bool {
9486 if format == ObjectFormat::Legacy {
9487 return false;
9488 }
9489 attributes.len() > MAX_COMPACT_ATTRS
9490 || attributes
9491 .iter()
9492 .any(|a| self.encode_attribute(a).len() > MAX_MESSAGE_SIZE)
9493 }
9494
9495 /// Every object whose attributes this finalize re-lays-out, with the
9496 /// creation-order policy each one's storage must follow.
9497 ///
9498 /// `datasets` lists the datasets whose headers this finalize will
9499 /// actually write. A reopened dataset that took no writes keeps its
9500 /// original header — and with it whatever storage that header already
9501 /// names — so touching its attribute storage would strand every block of
9502 /// it.
9503 ///
9504 /// The policy is the one the *header* records, not the one the object's
9505 /// creation property list asked for: those differ on a file whose
9506 /// shared-message configuration covers attributes, where
9507 /// [`header_attr_order`](Self::header_attr_order) raises every object to
9508 /// tracked. Storage laid out against the property list would then omit the
9509 /// creation indices the header says are there — and, since the Attribute
9510 /// Info message carries a maximum creation index only when tracked, would
9511 /// be two bytes shorter than the message the sizing pass measured.
9512 fn attribute_scopes(&self, datasets: &[usize]) -> Vec<(AttrScope, CreationOrder)> {
9513 let order_of = |requested| self.header_attr_order(requested);
9514 let mut scopes = vec![(AttrScope::Root, order_of(self.root_track_order.attrs))];
9515 for gi in 0..self.group_count() {
9516 if self.grp(gi).lock().deleted {
9517 continue;
9518 }
9519 let order = self.grp(gi).lock().track_order.attrs;
9520 scopes.push((AttrScope::Group(gi), order_of(order)));
9521 }
9522 for &i in datasets {
9523 let order = self.ds(i).lock().track_attr_order;
9524 scopes.push((AttrScope::Dataset(i), order_of(order)));
9525 }
9526 scopes
9527 }
9528
9529 /// Lay out and write dense attribute storage for every object that needs
9530 /// it, recording the resulting `Attribute Info` message per object.
9531 ///
9532 /// The sole owner of that transition. It runs after every object header
9533 /// has an address — an attribute may hold an object reference, and the
9534 /// heap holds the encoded attribute messages — and before any object
9535 /// header is written, because the header carries the Attribute Info
9536 /// message naming what this laid out. Every block is on disk before the
9537 /// map naming it is populated, so a header written from that map can only
9538 /// point at bytes that exist. The same placement rule, for the same two
9539 /// reasons, as [`prepare_dense_links`](Self::prepare_dense_links).
9540 ///
9541 /// Which objects spill is not decided here: `emit_attributes` asks
9542 /// [`attributes_need_dense`](Self::attributes_need_dense) itself, so the
9543 /// header measured before this ran and the header written after it agree
9544 /// without either consulting the other.
9545 fn prepare_dense_attributes(&self, datasets: &[usize]) -> IoResult<()> {
9546 for (scope, order) in self.attribute_scopes(datasets) {
9547 // Every scope here has its header rewritten, so the storage a
9548 // reopen found on it is superseded whether or not the new set is
9549 // dense again — a free driven by "the new set needs a heap" would
9550 // never reach an object that dropped back to compact. Freed
9551 // immediately before its replacement is laid out, so the rewrite
9552 // lands in the blocks it just gave back instead of growing the
9553 // file on every open/close cycle.
9554 self.release_superseded_dense_attrs(scope)?;
9555 // `close` after `start_swmr` finalizes a second time over the same
9556 // attribute sets — SWMR refuses every attribute mutation — so
9557 // rebuilding here would allocate a whole second heap and strand
9558 // the one the published headers already name.
9559 if self.dense_attributes.lock().contains_key(&scope) {
9560 continue;
9561 }
9562 let attributes = self.object_attributes(scope)?;
9563 if !self.attributes_need_dense(&attributes, self.attr_scope_format(scope)) {
9564 continue;
9565 }
9566 let dense = build_dense_attributes(&attributes, &self.ctx, order, &mut |len| {
9567 self.allocator.allocate(len, FreeSpaceClass::Metadata)
9568 })?;
9569 for block in &dense.blocks {
9570 self.handle.write_at(block.addr, &block.image)?;
9571 }
9572 self.dense_attributes.lock().insert(scope, dense.ainfo);
9573 }
9574 Ok(())
9575 }
9576
9577 /// The object header format `scope`'s owner is written at, which is what
9578 /// decides whether its attributes may spill at all.
9579 fn attr_scope_format(&self, scope: AttrScope) -> ObjectFormat {
9580 match scope {
9581 AttrScope::Root => self.header_format(self.root_track_order),
9582 AttrScope::Group(gi) => self.group_header_format(gi),
9583 AttrScope::Dataset(i) => self.dataset_header_format(i),
9584 }
9585 }
9586
9587 /// Whether a dataset's datatype message may be offered to a
9588 /// shared-message index at all.
9589 ///
9590 /// The datatype is the one message class carrying a `can_share` callback
9591 /// (`H5O__dtype_can_share`, H5Odtype.c:99), and `H5SM__can_share_common`
9592 /// asks it before any index is consulted (H5SM.c:895-899). It refuses an
9593 /// immutable type and a committed one (H5Odtype.c:1893-1901); the
9594 /// committed half is already answered by address at the call site.
9595 ///
9596 /// A dataset's type reaches that predicate still immutable only when
9597 /// `H5D__init_type` kept the caller's own `H5T_t` rather than copying it,
9598 /// which it does exactly when the type is immutable, is not relocatable,
9599 /// and the low bound this dataset's messages are written at is below
9600 /// `H5F_LIBVER_V18` (H5Dint.c:569-572) — the bound the dataset was
9601 /// *created* under, which for a dataset a reopen found is not this
9602 /// session's.
9603 /// Any of the three failing produces an `H5T_COPY_ALL` copy, which is
9604 /// `H5T_STATE_RDONLY` rather than immutable (H5T.c:4461-4462) and so is
9605 /// shareable — which is why `H5Tcopy(H5T_STD_I32LE)` shares where
9606 /// `H5T_STD_I32LE` itself does not (tests/fixtures/gen_sohm.c).
9607 ///
9608 /// An attribute has no such branch: `H5A__create` copies unconditionally
9609 /// (H5Aint.c:341), so its datatype is always eligible and
9610 /// [`share_attribute`](Self::share_attribute) offers it without asking.
9611 fn dataset_datatype_shareable(&self, datatype: &DatatypeMessage, libver: LibverBound) -> bool {
9612 !datatype.is_predefined() || datatype.is_relocatable() || libver >= LibverBound::V18
9613 }
9614
9615 /// Whether the first copy of a `msg_type` message may stay literal in the
9616 /// object header that writes it.
9617 ///
9618 /// `H5O_msg_can_share_in_ohdr` reads the class's `H5O_SHARE_IN_OHDR` flag
9619 /// (H5Omessage.c:1426); the five classes that carry it are datatype
9620 /// (H5Odtype.c:89), dataspace (H5Osdspace.c:61), both fill value messages
9621 /// (H5Ofill.c:106 and :130) and the filter pipeline (H5Opline.c:65). The
9622 /// attribute class does not, which is why an attribute reaches the heap on
9623 /// its first use.
9624 const fn shares_in_ohdr(msg_type: u8) -> bool {
9625 matches!(
9626 msg_type,
9627 MSG_DATASPACE
9628 | MSG_DATATYPE
9629 | MSG_FILL_VALUE
9630 | MSG_FILL_VALUE_OLD
9631 | MSG_FILTER_PIPELINE
9632 )
9633 }
9634
9635 /// What a header stores for a message a shared-message index may cover:
9636 /// the body itself, or a pointer into the shared-message heap.
9637 ///
9638 /// The single point at which a message is offered to an index. Every
9639 /// header builder routes its shareable messages through here, so the pass
9640 /// that counts references and the pass that substitutes pointers walk
9641 /// exactly the same set — the counting and the substituting cannot drift
9642 /// apart, because they are one call site in two phases.
9643 ///
9644 /// `owner` is `H5SM_try_share`'s `open_oh`: the header this message
9645 /// belongs to, or [`ShareOwner::Detached`] for a body that is part of
9646 /// another message rather than a message of a header.
9647 ///
9648 /// Outside a finalize, and in any file created without indexes, this is
9649 /// the identity.
9650 fn share_message(
9651 &self,
9652 owner: ShareOwner,
9653 msg_type: u8,
9654 flags: u8,
9655 body: Vec<u8>,
9656 ) -> (u8, Vec<u8>) {
9657 self.share_nesting_message(owner, msg_type, flags, body, Vec::new())
9658 }
9659
9660 /// [`share_message`](Self::share_message) for a body that itself holds
9661 /// shared-message pointers.
9662 ///
9663 /// `nested` names each heap ID inside `body`, which is zero until the
9664 /// table is laid out. Two bodies that differ only in what they point at
9665 /// are the same bytes here and different bytes on disk, so the count and
9666 /// the substitute are keyed on the pair.
9667 fn share_nesting_message(
9668 &self,
9669 owner: ShareOwner,
9670 msg_type: u8,
9671 flags: u8,
9672 body: Vec<u8>,
9673 nested: Vec<NestedShare>,
9674 ) -> (u8, Vec<u8>) {
9675 let Some(sohm) = self.sohm.as_deref() else {
9676 return (flags, body);
9677 };
9678 // A message already carrying a pointer — a committed datatype — is
9679 // shared by address and must not be shared again, and the message
9680 // classes libhdf5 marks `H5O_MSG_FLAG_DONTSHARE` never reach an index.
9681 if flags & (MSG_FLAG_SHARED | MSG_FLAG_DONTSHARE) != 0 {
9682 return (flags, body);
9683 }
9684 let Some(index) = sohm.index_for(msg_type, body.len()) else {
9685 return (flags, body);
9686 };
9687 // `share_in_ohdr && open_oh` (H5SM.c:1400): the first copy of one of
9688 // these classes stays where it was written, marked shareable, and only
9689 // a second use moves the body to the heap.
9690 let ohdr = match owner {
9691 ShareOwner::Header(addr) if Self::shares_in_ohdr(msg_type) => Some(addr),
9692 _ => None,
9693 };
9694 // What a pointer to this body looks like: a zeroed heap ID until the
9695 // table exists, which is the width the real one has.
9696 let pointer = |id| {
9697 (
9698 flags | MSG_FLAG_SHARED,
9699 SharedMessagePointer::encode_sohm(id),
9700 )
9701 };
9702 match &mut *sohm.phase.lock() {
9703 SohmPhase::Idle => (flags, body),
9704 SohmPhase::Predict(first) => {
9705 if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9706 return (flags | MSG_FLAG_SHAREABLE, body);
9707 }
9708 pointer([0u8; SOHM_HEAP_ID_LEN])
9709 }
9710 // The same substitution `Predict` makes, so that what the collect
9711 // pass builds around a shared message is the width the resolve
9712 // pass will build — which is what lets an attribute body assembled
9713 // in this pass be the body assembled in that one, bar the heap IDs
9714 // it is here recording a need for.
9715 SohmPhase::Collect(collector) => {
9716 let first = collector.record(index, msg_type, &body, &nested, ohdr);
9717 if !first && !nested.is_empty() {
9718 // This body is already here, so the pointers it holds
9719 // already exist in the heap and the offers that built
9720 // this copy of it must not count a second time.
9721 for share in &nested {
9722 collector.release(share.target.0, &share.target.1);
9723 }
9724 }
9725 if ohdr.is_some() && first {
9726 return (flags | MSG_FLAG_SHAREABLE, body);
9727 }
9728 pointer([0u8; SOHM_HEAP_ID_LEN])
9729 }
9730 SohmPhase::Resolve { ids, first } => {
9731 if ohdr.is_some() && first.insert((msg_type, body.clone())) {
9732 return (flags | MSG_FLAG_SHAREABLE, body);
9733 }
9734 let key = (msg_type, body);
9735 match ids.get(&key) {
9736 Some(&id) => pointer(id),
9737 // The collect pass never saw this body — a dataspace a
9738 // SWMR extend changed after the table was laid out, say.
9739 // Left literal, which leaves a heap object counted for one
9740 // reference more than reaches it and nothing else.
9741 None => (flags, key.1),
9742 }
9743 }
9744 }
9745 }
9746
9747 /// Answer every shareable message at a heap pointer's width for the rest
9748 /// of this finalize's allocation phase.
9749 ///
9750 /// Half of the bracket [`prepare_shared_messages`](Self::prepare_shared_messages)
9751 /// closes, and the reason the two can sit on opposite sides of the
9752 /// allocation: a header cannot be measured until it is known which of its
9753 /// messages are pointers, and a body cannot be counted until every address
9754 /// it names exists. Only the width is knowable in the first phase, and the
9755 /// width is all the measurement needs.
9756 ///
9757 /// A finalize that will not lay a table out — a `finalize_for_swmr`, a
9758 /// second finalize over a table already published — leaves the phase where
9759 /// it found it, so what that pass measures is what it writes.
9760 fn begin_shared_message_layout(&self) {
9761 let Some(sohm) = self.sohm.as_deref() else {
9762 return;
9763 };
9764 let mut phase = sohm.phase.lock();
9765 if matches!(*phase, SohmPhase::Idle) && sohm.table_addr.lock().is_none() {
9766 *phase = SohmPhase::Predict(FirstCopies::default());
9767 }
9768 }
9769
9770 /// Lay out the file's shared-message table: count the bodies every header
9771 /// this finalize writes would share, put them in their index's heap, and
9772 /// arm the substitution the header builders then apply.
9773 ///
9774 /// The sole owner of the transition to `Resolve`. It runs last in the
9775 /// content phase, after
9776 /// [`prepare_dense_attributes`](Self::prepare_dense_attributes),
9777 /// [`prepare_link_storage`](Self::prepare_link_storage) and
9778 /// [`write_reference_values`](Self::write_reference_values), because a
9779 /// body is only counted once it is the body the file will hold: an
9780 /// attribute that spilled into dense storage is not in a header to be
9781 /// shared at all, and one holding an object reference says an object
9782 /// header address that exists only after the allocation phase. Counting
9783 /// either of them earlier would count a body no header ends up carrying,
9784 /// and leave the header that carries the real one literal — which
9785 /// [`check_header_size`] would then refuse, the block having been
9786 /// reserved at a pointer's width.
9787 ///
9788 /// Once per file: a second finalize (a SWMR session's close) keeps the
9789 /// table the first one published rather than allocating a second one and
9790 /// stranding the first.
9791 fn prepare_shared_messages(&self, datasets: &[usize]) -> IoResult<()> {
9792 let Some(sohm) = self.sohm.as_deref() else {
9793 return Ok(());
9794 };
9795 if sohm.table_addr.lock().is_some() {
9796 return Ok(());
9797 }
9798
9799 // Collect: build every header this finalize will write and throw it
9800 // away, keeping only what its shareable messages were.
9801 *sohm.phase.lock() = SohmPhase::Collect(SohmCollector::new(sohm.indexes.len()));
9802 for &i in datasets {
9803 self.build_dataset_header(i)?;
9804 }
9805 for gi in 0..self.group_count() {
9806 if self.grp(gi).lock().deleted {
9807 continue;
9808 }
9809 self.build_group_header(gi)?;
9810 }
9811 self.build_root_group_header()?;
9812 let SohmPhase::Collect(collector) =
9813 std::mem::replace(&mut *sohm.phase.lock(), SohmPhase::Idle)
9814 else {
9815 return Err(crate::io::IoError::InvalidState(
9816 "the shared-message collect pass did not finish in the collect phase".into(),
9817 ));
9818 };
9819
9820 let indexes: Vec<SohmIndexContent> = sohm
9821 .indexes
9822 .iter()
9823 .zip(collector.messages)
9824 .map(|(&spec, messages)| SohmIndexContent { spec, messages })
9825 .collect();
9826 // The table a reopen found is superseded whole by the one below, and
9827 // every header that pointed into it is in this finalize's rewrite set
9828 // — so its blocks go back immediately before the replacement is laid
9829 // out, and the new table lands in them instead of growing the file on
9830 // every open/close cycle. Taken, not read: a second finalize must not
9831 // free the same blocks twice.
9832 for (addr, len) in std::mem::take(&mut *sohm.superseded.lock()) {
9833 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9834 }
9835 let built = build_shared_messages(&indexes, &self.ctx, &mut |len| {
9836 self.allocator.allocate(len, FreeSpaceClass::Metadata)
9837 })?;
9838 for block in &built.blocks {
9839 self.handle.write_at(block.addr, &block.image)?;
9840 }
9841
9842 // Only now, with every block on disk: from here the header builders
9843 // substitute pointers, and `write_superblock_extension` names the
9844 // table this laid out.
9845 *sohm.phase.lock() = SohmPhase::Resolve {
9846 ids: built.heap_ids,
9847 first: FirstCopies::default(),
9848 };
9849 *sohm.table_addr.lock() = Some(built.table_addr);
9850 Ok(())
9851 }
9852
9853 /// Write the file's free-space managers over the space this close leaves
9854 /// free, and return the file-space info message body naming them.
9855 ///
9856 /// Called from [`write_superblock_extension`](Self::write_superblock_extension)
9857 /// once every other block of the file has an address, which is what makes
9858 /// the allocator's free list the file's *final* free space: a block
9859 /// allocated after this point would land in space a manager still claims.
9860 ///
9861 /// INVARIANT: from the moment this returns, every byte the allocator holds
9862 /// free is a byte some sections block records, and the two blocks each
9863 /// manager itself occupies are held by neither. Nothing may allocate
9864 /// between here and the superblock write; `write_object_headers` writes
9865 /// over blocks reserved in an earlier phase and is the only thing that
9866 /// runs in between.
9867 ///
9868 /// Returns `None` for a file with no message of its own to write — a
9869 /// reopen whose carried message this session must not touch, and a file
9870 /// created at the library defaults — which leaves both byte-identical to
9871 /// what the same close wrote before free space was recorded at all. A file
9872 /// that carries the message but keeps no managers (either non-manager
9873 /// strategy, or `persist: false`) gets the message back with every address
9874 /// undefined, which is what `H5F__super_init` writes for it.
9875 fn write_free_space_managers(&self) -> IoResult<Option<Vec<u8>>> {
9876 let Some(fs) = self.free_space.as_deref() else {
9877 return Ok(None);
9878 };
9879 if !fs.records_free_space() {
9880 return Ok(Some(fs.info.encode(&self.ctx)?));
9881 }
9882 // The managers a reopen found are superseded whole by the ones below,
9883 // so their blocks go back before anything is laid out: the space the
9884 // old manager occupied is free space the new one records, and the new
9885 // one may be laid out in it.
9886 for &(addr, len) in &fs.superseded {
9887 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
9888 }
9889
9890 let hdr_size = FreeSpaceHeader::encoded_size(&self.ctx) as u64;
9891 let settled = self.settle_free_space_managers(hdr_size, fs.info.threshold)?;
9892
9893 let mut info = fs.info.clone();
9894 info.fs_addr = vec![UNDEF_ADDR; info.fs_addr.len()];
9895 for placed in &settled {
9896 let mut header = manager_header(&placed.sections);
9897 // The settle loop sized the block; that the encode agrees is the
9898 // invariant that makes `sect_size` a length a reader can trust.
9899 let needed = free_space::sinfo_encoded_size(&header, &placed.sections, &self.ctx);
9900 if needed > placed.sect_size {
9901 return Err(crate::io::IoError::InvalidState(format!(
9902 "the free-space sections need {needed} bytes, not the {} laid out",
9903 placed.sect_size
9904 )));
9905 }
9906 header.sect_addr = placed.sect_addr;
9907 header.sect_size = placed.sect_size;
9908 header.alloc_sect_size = placed.sect_size;
9909 self.handle.write_at(
9910 placed.sect_addr,
9911 &free_space::encode_sections(
9912 &header,
9913 placed.hdr_addr,
9914 &placed.sections,
9915 placed.sect_size as usize,
9916 &self.ctx,
9917 ),
9918 )?;
9919 self.handle
9920 .write_at(placed.hdr_addr, &header.encode(&self.ctx))?;
9921 // `H5MF__close_delete_fstype` leaves a manager with no sections
9922 // without an address, so only the ones written name themselves.
9923 info.fs_addr[placed.manager.message_slot()] = placed.hdr_addr;
9924 }
9925 // The end of the file *after* the settle above, not before it, which
9926 // the field's name denies: it is 1.10 vintage, where two EOAs were
9927 // kept — one taken before the self-referential managers were placed
9928 // and one after (H5MF.c:3305 and 3382 in 1.10.11) — and the message
9929 // carried the first (1.10.11 H5MF.c:1833, 1999). 1.14 keeps one,
9930 // `f->shared->eoa_fsm_fsalloc`, read once the allocation loop has run
9931 // (H5MF.c:3234-3240) and encoded into this field by both close paths
9932 // (H5MF.c:1759, 1923); H5Fsuper.c:826 names it "the final eoa". A
9933 // 1.10 reader wants that value and not the older one: equal EOAs are
9934 // the case `H5MF_tidy_self_referential_fsm_hack` returns on
9935 // (1.10.11 H5MF.c:3620-3622), which is what leaves the managers this
9936 // close wrote in place.
9937 info.eoa_pre_fsm_fsalloc = self.allocator.eof();
9938 Ok(Some(info.encode(&self.ctx)?))
9939 }
9940
9941 /// The file's free space as each manager will record it: address-ordered
9942 /// per manager, tagged with the section class that manager writes, and
9943 /// with everything below `threshold` left out.
9944 ///
9945 /// The allocator is the single owner of merging — `H5FS__sect_merge`'s
9946 /// rules, per manager and, on a paged file, per page — so nothing merges
9947 /// here; overlap is checked because two overlapping sections would be a
9948 /// manager claiming space another structure holds.
9949 fn free_sections(&self, threshold: u64) -> IoResult<Vec<(FreeSpaceManager, Vec<FreeSection>)>> {
9950 let policy = self.allocator.policy();
9951 let extents = self.allocator.free_extents();
9952 let mut sets = Vec::new();
9953 for manager in FreeSpaceManager::ALL {
9954 let mut sections: Vec<FreeSection> = extents
9955 .iter()
9956 .filter(|b| b.manager == manager)
9957 // `H5FS_sect_add` refuses a section below the file's
9958 // threshold, so a block smaller than it is space the file
9959 // leaks rather than records — the same trade the threshold is
9960 // there to make.
9961 .filter(|b| b.len >= threshold)
9962 .map(|b| FreeSection {
9963 addr: b.addr,
9964 len: b.len,
9965 class: policy.section_class(manager),
9966 })
9967 .collect();
9968 sections.sort_unstable_by_key(|s| s.addr);
9969 if let Some(bad) = sections
9970 .windows(2)
9971 .find(|w| w[0].addr + w[0].len > w[1].addr)
9972 {
9973 return Err(crate::io::IoError::InvalidState(format!(
9974 "this session freed overlapping blocks: {:#x}+{} overlaps {:#x}",
9975 bad[0].addr, bad[0].len, bad[1].addr
9976 )));
9977 }
9978 sets.push((manager, sections));
9979 }
9980 Ok(sets)
9981 }
9982
9983 /// Give every manager that records anything its own header and sections
9984 /// blocks, and return what each will write.
9985 ///
9986 /// Self-referential, which is the whole difficulty: a manager's two blocks
9987 /// come out of the free space the managers record, and taking them changes
9988 /// that space, which changes how many bytes the sections block needs.
9989 /// Upstream reruns the allocation pass until no manager allocates anything
9990 /// further — the `do { ... } while (continue_alloc_fsm)` loop in
9991 /// `H5MF_settle_meta_data_fsm` (H5MF.c:3213-3247) around
9992 /// `H5FS_vfd_alloc_hdr_and_section_info_if_needed`, which allocates
9993 /// through `H5MF_alloc` like everything else. So does this: the blocks
9994 /// come out of the same [`FileAllocator`], under the same strategy, so a
9995 /// paged file's manager blocks land in pages and their page remainders are
9996 /// recorded like any others.
9997 ///
9998 /// Two rules make it terminate. A manager, once placed, stays placed: were
9999 /// its blocks released because its sections had been consumed, freeing
10000 /// them would put those sections back and the next round would place it
10001 /// again. And a sections block only ever grows: upstream frees a block
10002 /// that turned out too small and reallocates it next round
10003 /// (H5FSsection.c:2418-2423), and a size that only rises reaches its
10004 /// bound.
10005 fn settle_free_space_managers(
10006 &self,
10007 hdr_size: u64,
10008 threshold: u64,
10009 ) -> IoResult<Vec<PlacedManager>> {
10010 /// Rounds before the layout is called divergent. A round either places
10011 /// a manager or grows one sections block, and there are three
10012 /// managers, so a file that needs more than this is not converging.
10013 const ROUNDS: usize = 16;
10014
10015 // Raw data first and metadata last, in `H5MF_settle_raw_data_fsm`'s
10016 // order (H5C.c:689-696): every manager's own blocks are metadata
10017 // allocations, so the metadata manager funds all of them and is the
10018 // one whose section set the others change.
10019 const ORDER: [FreeSpaceManager; 3] = [
10020 FreeSpaceManager::RawData,
10021 FreeSpaceManager::Large,
10022 FreeSpaceManager::Metadata,
10023 ];
10024
10025 let size_of = |sections: &[FreeSection]| {
10026 let ordered = free_space::serialization_order(sections);
10027 free_space::sinfo_encoded_size(&manager_header(&ordered), &ordered, &self.ctx)
10028 };
10029 let mut placed: Vec<PlacedManager> = Vec::new();
10030 for _ in 0..ROUNDS {
10031 let sets = self.free_sections(threshold)?;
10032 let sections_of = |manager: FreeSpaceManager| {
10033 sets.iter()
10034 .find(|(m, _)| *m == manager)
10035 .map(|(_, s)| s.as_slice())
10036 .unwrap_or_default()
10037 };
10038
10039 let mut changed = false;
10040 for manager in ORDER {
10041 let sections = sections_of(manager);
10042 if sections.is_empty() || placed.iter().any(|p| p.manager == manager) {
10043 continue;
10044 }
10045 let sect_size = size_of(sections);
10046 let hdr_addr = self.allocator.allocate(hdr_size, FreeSpaceClass::Metadata);
10047 let sect_addr = self.allocator.allocate(sect_size, FreeSpaceClass::Metadata);
10048 placed.push(PlacedManager {
10049 manager,
10050 hdr_addr,
10051 sect_addr,
10052 sect_size,
10053 sections: Vec::new(),
10054 });
10055 changed = true;
10056 }
10057 if !changed {
10058 for p in &mut placed {
10059 let needed = size_of(sections_of(p.manager));
10060 if needed > p.sect_size {
10061 self.allocator
10062 .free(p.sect_addr, p.sect_size, FreeSpaceClass::Metadata);
10063 p.sect_size = needed;
10064 p.sect_addr = self.allocator.allocate(needed, FreeSpaceClass::Metadata);
10065 changed = true;
10066 }
10067 }
10068 }
10069 if !changed {
10070 for p in &mut placed {
10071 p.sections = free_space::serialization_order(sections_of(p.manager));
10072 }
10073 return Ok(placed);
10074 }
10075 }
10076 Err(crate::io::IoError::InvalidState(format!(
10077 "the free-space managers did not settle in {ROUNDS} rounds"
10078 )))
10079 }
10080
10081 /// Write the file's superblock extension, and the sole owner of that
10082 /// object header.
10083 ///
10084 /// Runs after [`prepare_shared_messages`](Self::prepare_shared_messages),
10085 /// whose table it names, and before the superblock that names it. What it
10086 /// writes is [`CarriedExtension`] — every message the reopened file's
10087 /// extension held — plus the shared-message table message, which is the
10088 /// one message whose content this session owns: the table moved, so the
10089 /// message read is stale and the message written names the new address.
10090 ///
10091 /// A file with neither carried messages nor shared messages gets no
10092 /// extension, which is what libhdf5 writes for it: `H5F__super_ext_create`
10093 /// is called only when there is a message to put in one.
10094 ///
10095 /// Version 1, holding its messages in one chunk: the extension is created
10096 /// before anything raises the file's object header version
10097 /// (`H5F__super_ext_create` passes `H5O_HDR_STORE_TIMES` off and takes the
10098 /// version-1 path), so an extension of any generation of file looks the
10099 /// same.
10100 fn write_superblock_extension(&self) -> IoResult<()> {
10101 if self.extension.addr.lock().is_some() {
10102 return Ok(());
10103 }
10104 let table = self.sohm.as_deref().and_then(|sohm| {
10105 sohm.table_addr
10106 .lock()
10107 .map(|addr| (sohm.indexes.len(), addr))
10108 });
10109 // A file with file-space properties of its own needs an extension
10110 // too: the message that declares them is the only place they are
10111 // recorded, and a file created with them carries nothing else.
10112 if self.extension.carried.is_empty() && table.is_none() && self.free_space.is_none() {
10113 return Ok(());
10114 }
10115
10116 let mut messages: Vec<crate::io::object_header_io::ExtensionMessage> =
10117 self.extension.carried.clone();
10118 if let Some(fs) = self.free_space.as_deref() {
10119 // The declared message, at exactly the length the one written
10120 // below will have — every field of it is fixed-width, and only
10121 // `persist` and the message version change the count of
10122 // addresses, neither of which the close alters. The image is sized
10123 // and its block allocated before the managers can be laid out, so
10124 // the message has to reach its final *length* here even though its
10125 // content is settled later.
10126 let declared = fs.info.encode(&self.ctx)?;
10127 match messages
10128 .iter_mut()
10129 .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10130 {
10131 Some(msg) => msg.body = declared,
10132 None => messages.push(crate::io::object_header_io::ExtensionMessage {
10133 msg_type: MSG_FILE_SPACE_INFO,
10134 flags: MSG_FLAG_DONTSHARE | MSG_FLAG_MARK_IF_UNKNOWN,
10135 body: declared,
10136 }),
10137 }
10138 }
10139 if let Some((nindexes, table_addr)) = table {
10140 let nindexes = u8::try_from(nindexes).map_err(|_| {
10141 crate::io::IoError::InvalidState(format!("{nindexes} shared-message indexes"))
10142 })?;
10143 messages.push(crate::io::object_header_io::ExtensionMessage {
10144 msg_type: MSG_SHARED_MESSAGE_TABLE,
10145 flags: MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
10146 body: SharedMessageTableMessage {
10147 version: 0,
10148 table_address: table_addr,
10149 nindexes,
10150 }
10151 .encode(&self.ctx),
10152 });
10153 }
10154 let encode = |messages: &[crate::io::object_header_io::ExtensionMessage]| {
10155 let mut extension = ObjectHeader::new();
10156 for msg in messages {
10157 extension.add_message(msg.msg_type, msg.flags, msg.body.clone());
10158 }
10159 extension.encode_v1(1)
10160 };
10161 let image = encode(&messages)?;
10162 // Freed before the replacement is placed, so a reopen reuses the block
10163 // instead of stranding one per open/close cycle — the rule every other
10164 // superseded structure follows.
10165 for &(addr, len) in &self.extension.superseded {
10166 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
10167 }
10168 let addr = self
10169 .allocator
10170 .allocate(image.len() as u64, FreeSpaceClass::Metadata);
10171
10172 // Every block of this file now has an address, so the allocator holds
10173 // exactly the file's free space: settle the free-space managers over
10174 // it and say in this extension where they went.
10175 let image = match self.write_free_space_managers()? {
10176 None => image,
10177 Some(body) => {
10178 let msg = messages
10179 .iter_mut()
10180 .find(|m| m.msg_type == MSG_FILE_SPACE_INFO)
10181 .ok_or_else(|| {
10182 crate::io::IoError::InvalidState(
10183 "a persisting file lost its file-space info message".into(),
10184 )
10185 })?;
10186 // Same length as the declared body put in above, so the
10187 // image measured before the block was allocated still fits.
10188 if body.len() != msg.body.len() {
10189 return Err(crate::io::IoError::InvalidState(format!(
10190 "the file-space info message was laid out at {} bytes and \
10191 written back at {}",
10192 msg.body.len(),
10193 body.len()
10194 )));
10195 }
10196 msg.body = body;
10197 encode(&messages)?
10198 }
10199 };
10200 self.handle.write_at(addr, &image)?;
10201 *self.extension.addr.lock() = Some(addr);
10202 Ok(())
10203 }
10204
10205 /// Define a new contiguous dataset. Returns the dataset index (used with
10206 /// `write_dataset_raw`).
10207 ///
10208 /// The raw-data region is allocated immediately so that
10209 /// `write_dataset_raw` can be called at any time before `close()`.
10210 pub fn create_dataset(
10211 &self,
10212 name: &str,
10213 datatype: DatatypeMessage,
10214 dims: &[u64],
10215 ) -> IoResult<usize> {
10216 let create = self.begin_create(name)?;
10217 let name = create.name.as_str();
10218 let total_elements: u64 = if dims.is_empty() {
10219 1
10220 } else {
10221 dims.iter().product()
10222 };
10223 let element_size = datatype.element_size() as u64;
10224 let data_size = total_elements * element_size;
10225
10226 // Allocate space for the raw data.
10227 let data_addr = if data_size > 0 {
10228 self.allocator.allocate(data_size, FreeSpaceClass::RawData)
10229 } else {
10230 UNDEF_ADDR
10231 };
10232
10233 let dataspace = if dims.is_empty() {
10234 DataspaceMessage::scalar()
10235 } else {
10236 DataspaceMessage::simple(dims)
10237 };
10238
10239 let idx = self.push_dataset(
10240 &create,
10241 DatasetInfo {
10242 name: name.to_string(),
10243 datatype,
10244 committed_type: None,
10245 external: None,
10246 virtual_storage: None,
10247 dataspace,
10248 read_format: None,
10249 obj_header_addr: 0, // set during finalize
10250 data_addr,
10251 data_size,
10252 compact: None,
10253 chunked: None,
10254 fixed_array: None,
10255 implicit: None,
10256 single_chunk: None,
10257 btree_v1: None,
10258 btree_v2: None,
10259 append: None,
10260 attributes: Vec::new(),
10261 obj_header_written_addr: None,
10262 obj_header_blocks: Vec::new(),
10263 filter_pipeline: None,
10264 deleted: false,
10265 extent_dirty: false,
10266 header_dirty: false,
10267 nlink_written: 1,
10268 creation_seq: self.take_creation_seq(),
10269 track_attr_order: self.track_order.attrs,
10270 fill_value: None,
10271 fill_time: FILL_TIME_IFSET,
10272 layout_version: 4,
10273 times: self.created_object_times(),
10274 },
10275 );
10276
10277 Ok(idx)
10278 }
10279
10280 /// Define a new dataset whose raw data lives in files outside this one —
10281 /// `H5Pset_external`, h5py's `external=[(name, offset, size)]`.
10282 ///
10283 /// Each entry names a file, the byte offset in it where that entry's
10284 /// region starts, and how many bytes of the dataset the region holds; the
10285 /// entries concatenate, in order, into the dataset's logical byte range,
10286 /// and together must cover it. Nothing is allocated in this file: the data
10287 /// layout message says contiguous storage at an undefined address, and it
10288 /// is the External File List beside it that says where the bytes are
10289 /// (`H5D__layout_oh_create`).
10290 ///
10291 /// A named file is created on first write and never truncated, so several
10292 /// slots — or several datasets — may own disjoint ranges of one file, the
10293 /// way `H5D__efl_write` opens them.
10294 ///
10295 /// The last slot may take the unlimited size `H5O_EFL_UNLIMITED`, which
10296 /// makes it absorb however many bytes the dataset comes to hold; a
10297 /// dataset whose dataspace is unlimited must have one, since nothing
10298 /// finite could cover it (`H5D__efl_construct`: "unlimited dataspace but
10299 /// finite storage"). Only the first dimension may be extendible, which is
10300 /// the same function's other rule.
10301 pub fn create_external_dataset(
10302 &self,
10303 name: &str,
10304 datatype: DatatypeMessage,
10305 dims: &[u64],
10306 max_dims: Option<&[u64]>,
10307 files: &[(&str, u64, u64)],
10308 ) -> IoResult<usize> {
10309 if files.is_empty() {
10310 return Err(crate::io::IoError::InvalidState(format!(
10311 "external dataset '{name}' names no files; external storage is defined by \
10312 the files it lives in, so at least one is required"
10313 )));
10314 }
10315 let create = self.begin_create(name)?;
10316 let name = create.name.as_str();
10317 let total_elements: u64 = if dims.is_empty() {
10318 1
10319 } else {
10320 dims.iter().product()
10321 };
10322 let data_size = total_elements * datatype.element_size() as u64;
10323
10324 let mut heap = LocalHeapImage::with_empty_string();
10325 let mut entries = Vec::with_capacity(files.len());
10326 for (i, &(file_name, offset, size)) in files.iter().enumerate() {
10327 if file_name.is_empty() {
10328 return Err(crate::io::IoError::InvalidState(format!(
10329 "external dataset '{name}' has a slot with an empty file name"
10330 )));
10331 }
10332 // `H5Pset_external` refuses to add a slot behind an unlimited one
10333 // ("previous file size is unlimited"): the unlimited slot already
10334 // owns every byte from its own start onwards, so nothing after it
10335 // could ever be reached.
10336 if size == UNLIMITED && i + 1 != files.len() {
10337 return Err(crate::io::IoError::InvalidState(format!(
10338 "external dataset '{name}' gives slot {i} ('{file_name}') the unlimited \
10339 size H5O_EFL_UNLIMITED with {} slot(s) behind it; an unlimited slot \
10340 absorbs the rest of the dataset, so it can only be the last",
10341 files.len() - i - 1
10342 )));
10343 }
10344 if offset.checked_add(size).is_none() {
10345 return Err(crate::io::IoError::InvalidState(format!(
10346 "external dataset '{name}' slot '{file_name}' spans offset {offset} \
10347 plus {size} bytes, past the end of the 64-bit address space"
10348 )));
10349 }
10350 entries.push(ExternalFile {
10351 name: file_name.to_string(),
10352 name_offset: heap.insert_str(file_name),
10353 offset,
10354 size,
10355 });
10356 }
10357 let external = ExternalStorage {
10358 // Filled in below, once the heap the names went into has an
10359 // address; the names' offsets within it are already final.
10360 heap_addr: UNDEF_ADDR,
10361 files: entries,
10362 // Settled by the open this create hands a handle out for, which
10363 // is `H5D__create` reading the dapl at H5Dint.c:1318.
10364 prefix: EfilePrefix::default(),
10365 };
10366 // `H5D__efl_construct`, over the dataset's *maximum* extent: the
10367 // slots must reserve at least every byte the dataset could come to
10368 // hold, and an unlimited extent can only be covered by an unlimited
10369 // last slot ("unlimited dataspace but finite storage").
10370 let max_dims = max_dims.unwrap_or(dims);
10371 if max_dims.len() != dims.len() {
10372 return Err(crate::io::IoError::InvalidState(format!(
10373 "external dataset '{name}' has {} dimensions but {} maximum ones",
10374 dims.len(),
10375 max_dims.len()
10376 )));
10377 }
10378 for (d, (&max, &cur)) in max_dims.iter().zip(dims).enumerate().skip(1) {
10379 if max > cur {
10380 return Err(crate::io::IoError::InvalidState(format!(
10381 "external dataset '{name}' makes dimension {d} extendible ({cur} of \
10382 {max}); only the first dimension can be extendible for external storage"
10383 )));
10384 }
10385 }
10386 let reserved = external.total_size();
10387 if max_dims.contains(&u64::MAX) {
10388 if reserved != UNLIMITED {
10389 return Err(crate::io::IoError::InvalidState(format!(
10390 "external dataset '{name}' has an unlimited dataspace but its files \
10391 reserve only {reserved} bytes; the last slot must take the unlimited \
10392 size H5O_EFL_UNLIMITED"
10393 )));
10394 }
10395 } else {
10396 let max_bytes = max_dims
10397 .iter()
10398 .try_fold(datatype.element_size() as u64, |acc, &d| acc.checked_mul(d))
10399 .ok_or_else(|| {
10400 crate::io::IoError::InvalidState(format!(
10401 "external dataset '{name}' maximum extent times its element size \
10402 overflows 64 bits"
10403 ))
10404 })?;
10405 if reserved < max_bytes {
10406 return Err(crate::io::IoError::InvalidState(format!(
10407 "external dataset '{name}' needs {max_bytes} bytes but its files reserve \
10408 only {reserved}"
10409 )));
10410 }
10411 }
10412
10413 // The names' heap, written now: it is ordinary metadata of this file,
10414 // and the message the header carries is only an address into it.
10415 let sa = self.ctx.sizeof_addr as usize;
10416 let ss = self.ctx.sizeof_size as usize;
10417 let heap_bytes = heap.as_bytes().to_vec();
10418 let heap_addr = self.allocator.allocate(
10419 local_heap_header_size(sa, ss) as u64,
10420 FreeSpaceClass::Metadata,
10421 );
10422 let heap_data_addr = self
10423 .allocator
10424 .allocate(heap_bytes.len() as u64, FreeSpaceClass::Metadata);
10425 let heap_hdr = LocalHeapHeader {
10426 data_size: heap_bytes.len() as u64,
10427 // Sized to hold exactly these names, so no block of it is free.
10428 free_list_offset: LOCAL_HEAP_FREE_NULL,
10429 data_addr: heap_data_addr,
10430 };
10431 self.handle.write_at(heap_addr, &heap_hdr.encode(sa, ss))?;
10432 self.handle.write_at(heap_data_addr, &heap_bytes)?;
10433 let external = ExternalStorage {
10434 heap_addr,
10435 ..external
10436 };
10437
10438 let dataspace = if dims.is_empty() {
10439 DataspaceMessage::scalar()
10440 } else {
10441 let mut ds = DataspaceMessage::simple(dims);
10442 if max_dims != dims {
10443 ds.max_dims = Some(max_dims.to_vec());
10444 }
10445 ds
10446 };
10447
10448 let idx = self.push_dataset(
10449 &create,
10450 DatasetInfo {
10451 name: name.to_string(),
10452 datatype,
10453 committed_type: None,
10454 external: Some(external),
10455 virtual_storage: None,
10456 dataspace,
10457 read_format: None,
10458 obj_header_addr: 0, // set during finalize
10459 // No block of this file's own: the layout message declares
10460 // contiguous storage at an undefined address, which is what
10461 // sends a reader to the external file list instead.
10462 data_addr: UNDEF_ADDR,
10463 data_size,
10464 compact: None,
10465 chunked: None,
10466 fixed_array: None,
10467 btree_v2: None,
10468 implicit: None,
10469 single_chunk: None,
10470 btree_v1: None,
10471 append: None,
10472 attributes: Vec::new(),
10473 obj_header_written_addr: None,
10474 obj_header_blocks: Vec::new(),
10475 filter_pipeline: None,
10476 deleted: false,
10477 extent_dirty: false,
10478 header_dirty: false,
10479 nlink_written: 1,
10480 creation_seq: self.take_creation_seq(),
10481 track_attr_order: self.track_order.attrs,
10482 fill_value: None,
10483 fill_time: FILL_TIME_IFSET,
10484 layout_version: 4,
10485 times: self.created_object_times(),
10486 },
10487 );
10488
10489 Ok(idx)
10490 }
10491
10492 /// Define a new virtual dataset — `H5Pset_virtual`, h5py's
10493 /// `create_virtual_dataset(name, VirtualLayout)`.
10494 ///
10495 /// Each mapping says which elements of this dataset (`virtual_selection`)
10496 /// are read from which elements (`source_selection`) of a dataset in
10497 /// another file; the sources are never opened here, and a mapping naming
10498 /// one that does not exist yet is perfectly legal — libhdf5 resolves each
10499 /// at read time, filling from the fill value where nothing maps.
10500 ///
10501 /// The mappings do not live in the object header: they are serialized
10502 /// into one global heap object and the layout message carries only its
10503 /// address and index (`H5D__virtual_store_layout`), which is why this
10504 /// allocates a heap object and nothing else.
10505 ///
10506 /// An unlimited (`H5S_UNLIMITED`) selection is written as one: the
10507 /// mapping grows with its source, and the virtual dataset's extent in
10508 /// that dimension is whatever the sources reachable at read time supply
10509 /// (`H5D__virtual_set_extent_unlim`). A `printf`-style source name is
10510 /// written as one too: `%b` substitutes the block index, so one mapping
10511 /// stands for the family of source datasets that fill the successive
10512 /// blocks of an unlimited virtual selection.
10513 pub fn create_virtual_dataset(
10514 &self,
10515 name: &str,
10516 datatype: DatatypeMessage,
10517 dims: &[u64],
10518 max_dims: Option<&[u64]>,
10519 mappings: &[VirtualMapping],
10520 ) -> IoResult<usize> {
10521 if mappings.is_empty() {
10522 return Err(crate::io::IoError::InvalidState(format!(
10523 "virtual dataset '{name}' names no mappings; a virtual dataset is defined \
10524 by the source datasets it maps, so at least one is required"
10525 )));
10526 }
10527 for m in mappings {
10528 check_virtual_mapping(name, m)?;
10529 }
10530
10531 let create = self.begin_create(name)?;
10532 let name = create.name.as_str();
10533
10534 // The mapping list is ordinary file metadata, written now: the header
10535 // built at finalize carries only the heap address and object index it
10536 // lands at.
10537 let block = VirtualMappingList {
10538 mappings: mappings.to_vec(),
10539 }
10540 .encode(&self.ctx)?;
10541 let (heap_addr, heap_index) = self.insert_vlen_objects(&[&block])?[0];
10542
10543 let dataspace = if dims.is_empty() {
10544 DataspaceMessage::scalar()
10545 } else {
10546 let mut ds = DataspaceMessage::simple(dims);
10547 // A caller that named no maximum gets the current dimensions, the
10548 // maximum `simple` already filled in: `H5Screate_simple(rank,
10549 // dims, NULL)` reaches the encoder with `extent.max` set
10550 // (H5S.c:1293-1299), so leaving it absent here would write a
10551 // message no upstream API call can produce.
10552 if let Some(max) = max_dims {
10553 ds.max_dims = Some(max.to_vec());
10554 }
10555 ds
10556 };
10557
10558 let idx = self.push_dataset(
10559 &create,
10560 DatasetInfo {
10561 name: name.to_string(),
10562 datatype,
10563 committed_type: None,
10564 external: None,
10565 virtual_storage: Some(VirtualStorage {
10566 heap_addr,
10567 heap_index: heap_index as u32,
10568 mappings: mappings.to_vec(),
10569 }),
10570 dataspace,
10571 read_format: None,
10572 obj_header_addr: 0, // set during finalize
10573 // Not a block of this file at all: every element is read out
10574 // of a source dataset, so there is nothing here to allocate
10575 // and nothing to free when the dataset is deleted.
10576 data_addr: UNDEF_ADDR,
10577 data_size: 0,
10578 compact: None,
10579 chunked: None,
10580 fixed_array: None,
10581 btree_v2: None,
10582 implicit: None,
10583 single_chunk: None,
10584 btree_v1: None,
10585 append: None,
10586 attributes: Vec::new(),
10587 obj_header_written_addr: None,
10588 obj_header_blocks: Vec::new(),
10589 filter_pipeline: None,
10590 deleted: false,
10591 extent_dirty: false,
10592 header_dirty: false,
10593 nlink_written: 1,
10594 creation_seq: self.take_creation_seq(),
10595 track_attr_order: self.track_order.attrs,
10596 fill_value: None,
10597 fill_time: FILL_TIME_IFSET,
10598 layout_version: 4,
10599 times: self.created_object_times(),
10600 },
10601 );
10602
10603 Ok(idx)
10604 }
10605
10606 /// Define a new compact dataset — `H5Pset_layout(dcpl, H5D_COMPACT)`.
10607 ///
10608 /// The raw data lives inside the data layout message in the dataset's own
10609 /// object header, so it costs no block of its own and no extra seek to
10610 /// read; the price is the ceiling, and that the whole image is rewritten
10611 /// whenever the header is. The buffer is created at its final length and
10612 /// zero-filled, which is what `H5D__compact_fill` does at create time, so
10613 /// a dataset never written still reads back as its fill value.
10614 ///
10615 /// Errors when the image exceeds [`MAX_COMPACT_DATA`].
10616 pub fn create_compact_dataset(
10617 &self,
10618 name: &str,
10619 datatype: DatatypeMessage,
10620 dims: &[u64],
10621 ) -> IoResult<usize> {
10622 let total_elements: u64 = if dims.is_empty() {
10623 1
10624 } else {
10625 dims.iter().product()
10626 };
10627 let data_size = total_elements * datatype.element_size() as u64;
10628 if data_size > MAX_COMPACT_DATA as u64 {
10629 return Err(crate::io::IoError::InvalidState(format!(
10630 "compact dataset '{name}' needs {data_size} bytes, above the \
10631 {MAX_COMPACT_DATA}-byte ceiling a data layout message can hold; \
10632 use contiguous or chunked storage"
10633 )));
10634 }
10635
10636 let create = self.begin_create(name)?;
10637 let name = create.name.as_str();
10638 let dataspace = if dims.is_empty() {
10639 DataspaceMessage::scalar()
10640 } else {
10641 DataspaceMessage::simple(dims)
10642 };
10643
10644 let idx = self.push_dataset(
10645 &create,
10646 DatasetInfo {
10647 name: name.to_string(),
10648 datatype,
10649 committed_type: None,
10650 external: None,
10651 virtual_storage: None,
10652 dataspace,
10653 read_format: None,
10654 obj_header_addr: 0, // set during finalize
10655 data_addr: UNDEF_ADDR,
10656 data_size: 0,
10657 compact: Some(vec![0u8; data_size as usize]),
10658 chunked: None,
10659 fixed_array: None,
10660 implicit: None,
10661 single_chunk: None,
10662 btree_v1: None,
10663 btree_v2: None,
10664 append: None,
10665 attributes: Vec::new(),
10666 obj_header_written_addr: None,
10667 obj_header_blocks: Vec::new(),
10668 filter_pipeline: None,
10669 deleted: false,
10670 extent_dirty: false,
10671 header_dirty: false,
10672 nlink_written: 1,
10673 creation_seq: self.take_creation_seq(),
10674 track_attr_order: self.track_order.attrs,
10675 fill_value: None,
10676 fill_time: FILL_TIME_IFSET,
10677 layout_version: 4,
10678 times: self.created_object_times(),
10679 },
10680 );
10681
10682 Ok(idx)
10683 }
10684
10685 /// Define a new dataset with the NULL dataspace: no elements at all.
10686 ///
10687 /// Distinct from a scalar dataset (`create_dataset` with `dims == []`),
10688 /// which holds exactly one element — a NULL dataspace holds zero, so
10689 /// there is no raw image to allocate: `data_addr` stays `UNDEF_ADDR` and
10690 /// `data_size` stays 0 permanently, the same terminal state
10691 /// `create_dataset` already reaches for a zero-length dimension.
10692 pub fn create_null_dataset(&self, name: &str, datatype: DatatypeMessage) -> IoResult<usize> {
10693 let create = self.begin_create(name)?;
10694 let name = create.name.as_str();
10695
10696 let idx = self.push_dataset(
10697 &create,
10698 DatasetInfo {
10699 name: name.to_string(),
10700 datatype,
10701 committed_type: None,
10702 external: None,
10703 virtual_storage: None,
10704 dataspace: DataspaceMessage::null(),
10705 read_format: None,
10706 obj_header_addr: 0, // set during finalize
10707 data_addr: UNDEF_ADDR,
10708 data_size: 0,
10709 compact: None,
10710 chunked: None,
10711 fixed_array: None,
10712 implicit: None,
10713 single_chunk: None,
10714 btree_v1: None,
10715 btree_v2: None,
10716 append: None,
10717 attributes: Vec::new(),
10718 obj_header_written_addr: None,
10719 obj_header_blocks: Vec::new(),
10720 filter_pipeline: None,
10721 deleted: false,
10722 extent_dirty: false,
10723 header_dirty: false,
10724 nlink_written: 1,
10725 creation_seq: self.take_creation_seq(),
10726 track_attr_order: self.track_order.attrs,
10727 fill_value: None,
10728 fill_time: FILL_TIME_IFSET,
10729 layout_version: 4,
10730 times: self.created_object_times(),
10731 },
10732 );
10733
10734 Ok(idx)
10735 }
10736
10737 /// Define a new chunked dataset with an extensible array index.
10738 ///
10739 /// Returns the dataset index. The dataset starts empty (dims[0] = 0 if
10740 /// the first dimension is unlimited). Use `write_chunk` and
10741 /// `extend_dataset` to add data.
10742 pub fn create_chunked_dataset(
10743 &self,
10744 name: &str,
10745 datatype: DatatypeMessage,
10746 dims: &[u64],
10747 max_dims: &[u64],
10748 chunk_dims: &[u64],
10749 ) -> IoResult<usize> {
10750 let create = self.begin_create(name)?;
10751 let name = create.name.as_str();
10752 validate_chunk_geometry(dims, max_dims, chunk_dims)?;
10753 ensure_at_most_one_unlimited(max_dims)?;
10754 let chunk_bytes = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
10755 let layout_version = self.chunk_layout_version(false, chunk_bytes);
10756 let earray_params = EarrayParams::default_params();
10757 let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
10758 let nsblk_addrs = compute_nsblk_addrs(
10759 earray_params.idx_blk_elmts,
10760 earray_params.data_blk_min_elmts,
10761 earray_params.sup_blk_min_data_ptrs,
10762 earray_params.max_nelmts_bits,
10763 )?;
10764
10765 // Create EA header
10766 let mut ea_header = ExtensibleArrayHeader::new_for_chunks(&self.ctx);
10767 ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
10768 ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
10769 ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
10770 ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
10771 ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
10772
10773 // Allocate and write EA header (placeholder, will be updated)
10774 let hdr_encoded = ea_header.encode(&self.ctx);
10775 let ea_header_addr = self
10776 .allocator
10777 .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
10778
10779 // Create EA index block with pre-allocated super block address slots
10780 let ea_iblk = ExtensibleArrayIndexBlock::new(
10781 ea_header_addr,
10782 earray_params.idx_blk_elmts,
10783 ndblk_addrs,
10784 nsblk_addrs,
10785 );
10786
10787 // Allocate and write EA index block
10788 let iblk_encoded = ea_iblk.encode(&self.ctx);
10789 let ea_iblk_addr = self
10790 .allocator
10791 .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
10792
10793 // Update header with index block address
10794 ea_header.idx_blk_addr = ea_iblk_addr;
10795
10796 // Write both to disk
10797 let hdr_encoded = ea_header.encode(&self.ctx);
10798 self.handle.write_at(ea_header_addr, &hdr_encoded)?;
10799 self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
10800
10801 // Build dataspace with max dims
10802 let dataspace = DataspaceMessage {
10803 // Chunked storage always requires at least one dimension, so
10804 // this is never Scalar or Null.
10805 class: DataspaceClass::Simple,
10806 dims: dims.to_vec(),
10807 max_dims: Some(max_dims.to_vec()),
10808 };
10809
10810 let idx = self.push_dataset(
10811 &create,
10812 DatasetInfo {
10813 name: name.to_string(),
10814 datatype,
10815 committed_type: None,
10816 external: None,
10817 virtual_storage: None,
10818 dataspace,
10819 read_format: None,
10820 obj_header_addr: 0,
10821 data_addr: UNDEF_ADDR,
10822 data_size: 0,
10823 compact: None,
10824 attributes: Vec::new(),
10825 obj_header_written_addr: None,
10826 obj_header_blocks: Vec::new(),
10827 filter_pipeline: None,
10828 deleted: false,
10829 extent_dirty: false,
10830 header_dirty: false,
10831 nlink_written: 1,
10832 creation_seq: self.take_creation_seq(),
10833 track_attr_order: self.track_order.attrs,
10834 fill_value: None,
10835 fill_time: FILL_TIME_IFSET,
10836 layout_version,
10837 times: self.created_object_times(),
10838 fixed_array: None,
10839 implicit: None,
10840 single_chunk: None,
10841 btree_v1: None,
10842 btree_v2: None,
10843 chunked: Some(ChunkedDatasetInfo {
10844 chunk_dims: chunk_dims.to_vec(),
10845 earray_params,
10846 ea_header_addr,
10847 ea_iblk_addr,
10848 ea_header,
10849 ea_iblk,
10850 chunks_written: 0,
10851 filt_iblk: None,
10852 chunk_size_len: 0,
10853 }),
10854 append: None,
10855 },
10856 );
10857
10858 Ok(idx)
10859 }
10860
10861 /// Write `data` into a contiguous dataset's raw storage at *dataset-
10862 /// relative* byte offset `off`.
10863 ///
10864 /// The single owner of a contiguous raw-data write. Which storage that is
10865 /// — a block of this file, or the files an External File List names — is
10866 /// decided once, by [`DatasetInfo::contiguous_target`], and never at a
10867 /// call site.
10868 fn write_contiguous_bytes(
10869 &self,
10870 target: &ContiguousTarget,
10871 off: u64,
10872 data: &[u8],
10873 ) -> IoResult<()> {
10874 match target {
10875 ContiguousTarget::Local(addr) => Ok(self.handle.write_at(addr + off, data)?),
10876 ContiguousTarget::External { files, prefix } => {
10877 // The prefix the open settled, not one resolved here:
10878 // `H5D__efl_write` joins against `dset->shared->extfile_prefix`
10879 // (H5Defl.c:429-431), the same field `H5D__efl_read` joins
10880 // against, so a relative name lands where a later read looks.
10881 write_external_file_bytes(files, prefix.as_deref(), off, data)
10882 }
10883 ContiguousTarget::Virtual => Err(virtual_write_refused()),
10884 }
10885 }
10886
10887 /// Write raw bytes to a contiguous dataset identified by `index`.
10888 ///
10889 /// The caller is responsible for providing data in the correct byte order
10890 /// and layout. The length must match the total data size declared at
10891 /// creation time.
10892 pub fn write_dataset_raw(&self, index: usize, data: &[u8]) -> IoResult<()> {
10893 let ds = self.ds(index);
10894 let _op = ds.op.lock();
10895 let target = {
10896 let mut g = ds.lock();
10897 if g.is_chunked() {
10898 return Err(crate::io::IoError::InvalidState(
10899 "use write_chunk for chunked datasets".into(),
10900 ));
10901 }
10902 // A compact dataset's raw image is its layout message, so the
10903 // write lands in the buffer the header is built from rather than
10904 // at a file offset, and the header it is built into is now stale.
10905 if let Some(image) = g.compact.as_mut() {
10906 if data.len() != image.len() {
10907 return Err(crate::io::IoError::InvalidState(format!(
10908 "data size mismatch: expected {} bytes, got {}",
10909 image.len(),
10910 data.len()
10911 )));
10912 }
10913 image.copy_from_slice(data);
10914 g.header_dirty = true;
10915 return Ok(());
10916 }
10917 let Some(target) = g.contiguous_target() else {
10918 return Err(crate::io::IoError::InvalidState(
10919 "dataset has no data allocated".into(),
10920 ));
10921 };
10922 // A dataset that stores nothing of its own has no byte count to
10923 // check a write against — `write_contiguous_bytes` refuses it by
10924 // name below, which is the answer the caller needs.
10925 if target.is_storage() && data.len() as u64 != g.data_size {
10926 return Err(crate::io::IoError::InvalidState(format!(
10927 "data size mismatch: expected {} bytes, got {}",
10928 g.data_size,
10929 data.len()
10930 )));
10931 }
10932 target
10933 };
10934 self.write_contiguous_bytes(&target, 0, data)
10935 }
10936
10937 /// Write a chunk of data to a chunked dataset.
10938 ///
10939 /// `chunk_offset` is the chunk coordinates (e.g., [frame_idx] for a 1D-chunked
10940 /// streaming dataset where chunk_dims = [1, H, W]).
10941 /// Only the first (unlimited) dimension index is used for EA indexing.
10942 ///
10943 /// `data` must be exactly chunk_size bytes (product of chunk_dims * element_size).
10944 pub fn write_chunk(&self, index: usize, chunk_idx: u64, data: &[u8]) -> IoResult<()> {
10945 let ds = self.ds(index);
10946 let _op = ds.op.lock();
10947 self.write_chunk_inner(index, chunk_idx, data)
10948 }
10949
10950 /// [`Self::write_chunk`] body; the caller holds the dataset's op lock or
10951 /// the writer exclusively.
10952 pub(crate) fn write_chunk_inner(
10953 &self,
10954 index: usize,
10955 chunk_idx: u64,
10956 data: &[u8],
10957 ) -> IoResult<()> {
10958 let ds = self.ds(index);
10959 // Read the chunk geometry and filter pipeline under one brief lock,
10960 // then drop it: compression runs *outside* the lock, and
10961 // `record_ea_chunk` re-locks the same slot, so the guard must not be
10962 // held across either.
10963 let (chunk_bytes, pipeline) = {
10964 let g = ds.lock();
10965 let element_size = g.datatype.element_size() as u64;
10966 let chunked = g
10967 .chunked
10968 .as_ref()
10969 .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?;
10970 (
10971 chunked.chunk_dims.iter().product::<u64>() * element_size,
10972 g.filter_pipeline.clone(),
10973 )
10974 };
10975
10976 if data.len() as u64 != chunk_bytes {
10977 return Err(crate::io::IoError::InvalidState(format!(
10978 "chunk data size mismatch: expected {} bytes, got {}",
10979 chunk_bytes,
10980 data.len()
10981 )));
10982 }
10983
10984 // Apply compression if filter pipeline is set
10985 let compressed;
10986 let write_data = if let Some(ref pipeline) = pipeline {
10987 compressed = filter::apply_filters(pipeline, data)?;
10988 &compressed
10989 } else {
10990 data
10991 };
10992 // filter_mask = 0: this path runs the whole pipeline, so no filter is
10993 // skipped for the chunk.
10994 self.record_ea_chunk(index, chunk_idx, write_data, 0)
10995 }
10996
10997 /// Decide where a chunk's bytes belong and put them there, returning the
10998 /// address to record in the index.
10999 ///
11000 /// `old` is the chunk's current `(address, stored length)` if the index
11001 /// already holds an entry for it. This is the single owner of the
11002 /// rewrite-placement rule, mirroring libhdf5's `H5D__chunk_file_alloc`
11003 /// (`H5Dchunk.c`): a chunk whose stored size is unchanged is overwritten
11004 /// where it already lives, and only a chunk that no longer fits moves,
11005 /// releasing its old block. Without this every rewrite would abandon the
11006 /// old block and grow the file.
11007 fn place_chunk(&self, old: Option<(u64, u64)>, new_len: u64) -> u64 {
11008 match old {
11009 // Same stored size: overwrite in place. This is every unfiltered
11010 // rewrite (the stored size is fixed by the chunk shape) and every
11011 // filtered rewrite that compressed to the same length.
11012 Some((addr, len)) if addr != UNDEF_ADDR && len == new_len => addr,
11013 Some((addr, len)) if addr != UNDEF_ADDR => {
11014 // The chunk has to move. Under SWMR a reader may still hold an
11015 // index that points at the old block, so libhdf5 keeps it
11016 // (H5D__chunk_file_alloc skips H5MF_xfree when the file is
11017 // open for SWMR writing); do the same.
11018 if !self.swmr_active {
11019 self.allocator.free(addr, len, FreeSpaceClass::RawData);
11020 }
11021 self.allocator.allocate(new_len, FreeSpaceClass::RawData)
11022 }
11023 _ => self.allocator.allocate(new_len, FreeSpaceClass::RawData),
11024 }
11025 }
11026
11027 /// Place a chunk's already-final bytes (filtered if the dataset is
11028 /// filtered) in the file and record them in the extensible-array index —
11029 /// in the index block, a data block, or a super block per the EA geometry.
11030 /// Shared by write_chunk and write_compressed_chunk.
11031 ///
11032 /// The index lookup happens *before* the bytes are placed, because the
11033 /// entry it finds is what tells [`place_chunk`](Self::place_chunk) whether
11034 /// this is a rewrite that can stay put.
11035 fn record_ea_chunk(
11036 &self,
11037 index: usize,
11038 chunk_idx: u64,
11039 final_bytes: &[u8],
11040 filter_mask: u32,
11041 ) -> IoResult<()> {
11042 let compressed_size = final_bytes.len() as u64;
11043 let ds = self.ds(index);
11044 // Hold one slot guard for the whole method: every dataset-state access
11045 // below goes through `m`, while `self.handle`/`self.allocator`/`self.ctx`
11046 // are disjoint fields safe to touch with the guard held.
11047 let mut m = ds.lock();
11048 let is_filtered = m.filter_pipeline.is_some();
11049 // For a filtered dataset the chunk's stored size is encoded in the
11050 // `chunk_size_len`-byte field of each filtered EA entry
11051 // (`FilteredChunkEntry::encode` writes `nbytes[..chunk_size_len]`,
11052 // which truncates silently). Reject a size that would not fit, the way
11053 // libhdf5's H5D_CHUNK_ENCODE_SIZE_CHECK does, instead of corrupting the
11054 // index. The compress path never exceeds this (chunk_size_len holds the
11055 // uncompressed chunk size); a direct/raw write with caller-supplied
11056 // bytes can.
11057 if is_filtered {
11058 let chunk_size_len = m.chunked.as_ref().unwrap().chunk_size_len as usize;
11059 if chunk_size_len < 8 && compressed_size >= (1u64 << (chunk_size_len * 8)) {
11060 return Err(crate::io::IoError::InvalidState(format!(
11061 "filtered chunk size {compressed_size} does not fit in the \
11062 {chunk_size_len}-byte extensible-array chunk-size field"
11063 )));
11064 }
11065 }
11066 let idx_blk_elmts = {
11067 let c = m.chunked.as_ref().unwrap();
11068 c.earray_params.idx_blk_elmts as u64
11069 };
11070
11071 if chunk_idx < idx_blk_elmts {
11072 let chunked = m.chunked.as_mut().unwrap();
11073 if is_filtered {
11074 if let Some(ref mut fiblk) = chunked.filt_iblk {
11075 let old = fiblk.elements[chunk_idx as usize];
11076 let chunk_addr =
11077 self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11078 self.handle.write_at(chunk_addr, final_bytes)?;
11079 fiblk.elements[chunk_idx as usize] = FilteredChunkEntry {
11080 addr: chunk_addr,
11081 nbytes: compressed_size,
11082 filter_mask,
11083 };
11084 }
11085 } else {
11086 // An unfiltered chunk's stored size is fixed by the chunk
11087 // shape, so a rewrite always fits where it already is.
11088 let old = chunked.ea_iblk.elements[chunk_idx as usize];
11089 let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11090 self.handle.write_at(chunk_addr, final_bytes)?;
11091 chunked.ea_iblk.elements[chunk_idx as usize] = chunk_addr;
11092 }
11093 chunked.chunks_written += 1;
11094 if chunk_idx + 1 > chunked.ea_header.max_idx_set {
11095 chunked.ea_header.max_idx_set = chunk_idx + 1;
11096 }
11097 if chunked.ea_header.num_elmts_realized < idx_blk_elmts {
11098 chunked.ea_header.num_elmts_realized = idx_blk_elmts;
11099 }
11100 } else {
11101 // chunk_idx >= idx_blk_elmts: place the chunk through the EA
11102 // data-block / super-block hierarchy (libhdf5-compatible geometry).
11103 let (geo, max_nelmts_bits, chunk_size_len, ea_header_addr) = {
11104 let c = m.chunked.as_ref().unwrap();
11105 let p = &c.earray_params;
11106 (
11107 EaGeometry::new(
11108 p.idx_blk_elmts,
11109 p.data_blk_min_elmts,
11110 p.sup_blk_min_data_ptrs,
11111 p.max_nelmts_bits,
11112 p.max_dblk_page_nelmts_bits,
11113 )?,
11114 p.max_nelmts_bits,
11115 c.chunk_size_len,
11116 c.ea_header_addr,
11117 )
11118 };
11119 let loc = match geo.locate(chunk_idx)? {
11120 EaLoc::Dblk(l) => l,
11121 EaLoc::Index { .. } => unreachable!("chunk_idx >= idx_blk_elmts"),
11122 };
11123 if loc.paged {
11124 return Err(crate::io::IoError::InvalidState(format!(
11125 "chunk index {} needs a paged extensible-array data block, \
11126 which is not yet supported",
11127 chunk_idx
11128 )));
11129 }
11130 let class_id = if is_filtered {
11131 EA_CLS_FILT_CHUNK
11132 } else {
11133 EA_CLS_CHUNK
11134 };
11135 let dblk_nelmts = loc.dblk_nelmts as usize;
11136
11137 // Resolve the data block's current address and its parent slot,
11138 // creating the owning super block on demand.
11139 let parent: DblkParent;
11140 let mut dblk_addr: u64;
11141 match loc.path {
11142 EaDblkPath::Direct { idx: di } => {
11143 let c = m.chunked.as_ref().unwrap();
11144 dblk_addr = if is_filtered {
11145 c.filt_iblk.as_ref().unwrap().dblk_addrs[di]
11146 } else {
11147 c.ea_iblk.dblk_addrs[di]
11148 };
11149 parent = DblkParent::IndexBlock(di);
11150 }
11151 EaDblkPath::ViaSblk {
11152 sblk_off,
11153 local_dblk,
11154 ndblks_in_sblk,
11155 sblk_block_offset,
11156 } => {
11157 let mut sblk_addr = {
11158 let c = m.chunked.as_ref().unwrap();
11159 if is_filtered {
11160 c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
11161 } else {
11162 c.ea_iblk.sblk_addrs[sblk_off]
11163 }
11164 };
11165 if sblk_addr == UNDEF_ADDR {
11166 let sb = ExtensibleArraySuperBlock::new(
11167 class_id,
11168 ea_header_addr,
11169 sblk_block_offset,
11170 ndblks_in_sblk,
11171 );
11172 let enc = sb.encode(&self.ctx, max_nelmts_bits);
11173 sblk_addr = self
11174 .allocator
11175 .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11176 self.handle.write_at(sblk_addr, &enc)?;
11177 let c = m.chunked.as_mut().unwrap();
11178 if is_filtered {
11179 c.filt_iblk.as_mut().unwrap().sblk_addrs[sblk_off] = sblk_addr;
11180 } else {
11181 c.ea_iblk.sblk_addrs[sblk_off] = sblk_addr;
11182 }
11183 c.ea_header.num_sblks_created += 1;
11184 c.ea_header.size_sblks_created += enc.len() as u64;
11185 }
11186 let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
11187 // The writer never creates paged super blocks (it errors
11188 // before the paging threshold), so page_init_total is 0.
11189 let sb = ExtensibleArraySuperBlock::decode(
11190 &sb_buf,
11191 &self.ctx,
11192 max_nelmts_bits,
11193 ndblks_in_sblk,
11194 0,
11195 )?;
11196 dblk_addr = sb.dblk_addrs[local_dblk];
11197 parent = DblkParent::SuperBlock {
11198 sblk_addr,
11199 ndblks_in_sblk,
11200 local_dblk,
11201 };
11202 }
11203 }
11204
11205 // Create or update the data block holding this chunk's entry.
11206 let created = dblk_addr == UNDEF_ADDR;
11207 if is_filtered {
11208 let mut dblk = if created {
11209 FilteredDataBlock::new(ea_header_addr, loc.dblk_block_offset, dblk_nelmts)
11210 } else {
11211 let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11212 FilteredDataBlock::decode(
11213 &buf,
11214 &self.ctx,
11215 max_nelmts_bits,
11216 dblk_nelmts,
11217 chunk_size_len,
11218 )?
11219 };
11220 // A freshly created data block holds only undefined addresses,
11221 // so this reads as "no previous chunk" without a special case.
11222 let old = dblk.elements[loc.offset_in_dblk as usize];
11223 let chunk_addr = self.place_chunk(Some((old.addr, old.nbytes)), compressed_size);
11224 self.handle.write_at(chunk_addr, final_bytes)?;
11225 let entry = FilteredChunkEntry {
11226 addr: chunk_addr,
11227 nbytes: compressed_size,
11228 filter_mask,
11229 };
11230 dblk.elements[loc.offset_in_dblk as usize] = entry;
11231 let enc = dblk.encode(&self.ctx, max_nelmts_bits, chunk_size_len);
11232 if created {
11233 dblk_addr = self
11234 .allocator
11235 .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11236 }
11237 self.handle.write_at(dblk_addr, &enc)?;
11238 if created {
11239 let c = m.chunked.as_mut().unwrap();
11240 c.ea_header.num_dblks_created += 1;
11241 c.ea_header.size_dblks_created += enc.len() as u64;
11242 }
11243 } else {
11244 let mut dblk = if created {
11245 ExtensibleArrayDataBlock::new(
11246 ea_header_addr,
11247 loc.dblk_block_offset,
11248 dblk_nelmts,
11249 )
11250 } else {
11251 let buf = self.handle.read_at_most(dblk_addr, 65536)?;
11252 ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, dblk_nelmts)?
11253 };
11254 // Unfiltered: the stored size is fixed by the chunk shape, so
11255 // a rewrite always fits its old block. A freshly created data
11256 // block holds undefined addresses and falls through to a new
11257 // allocation.
11258 let old = dblk.elements[loc.offset_in_dblk as usize];
11259 let chunk_addr = self.place_chunk(Some((old, compressed_size)), compressed_size);
11260 self.handle.write_at(chunk_addr, final_bytes)?;
11261 dblk.elements[loc.offset_in_dblk as usize] = chunk_addr;
11262 let enc = dblk.encode(&self.ctx, max_nelmts_bits);
11263 if created {
11264 dblk_addr = self
11265 .allocator
11266 .allocate(enc.len() as u64, FreeSpaceClass::Metadata);
11267 }
11268 self.handle.write_at(dblk_addr, &enc)?;
11269 if created {
11270 let c = m.chunked.as_mut().unwrap();
11271 c.ea_header.num_dblks_created += 1;
11272 c.ea_header.size_dblks_created += enc.len() as u64;
11273 }
11274 }
11275
11276 // Record a newly-created data block's address in its parent.
11277 if created {
11278 match parent {
11279 DblkParent::IndexBlock(di) => {
11280 let c = m.chunked.as_mut().unwrap();
11281 if is_filtered {
11282 c.filt_iblk.as_mut().unwrap().dblk_addrs[di] = dblk_addr;
11283 } else {
11284 c.ea_iblk.dblk_addrs[di] = dblk_addr;
11285 }
11286 }
11287 DblkParent::SuperBlock {
11288 sblk_addr,
11289 ndblks_in_sblk,
11290 local_dblk,
11291 } => {
11292 let buf = self.handle.read_at_most(sblk_addr, 65536)?;
11293 let mut sb = ExtensibleArraySuperBlock::decode(
11294 &buf,
11295 &self.ctx,
11296 max_nelmts_bits,
11297 ndblks_in_sblk,
11298 0,
11299 )?;
11300 sb.dblk_addrs[local_dblk] = dblk_addr;
11301 let enc = sb.encode(&self.ctx, max_nelmts_bits);
11302 self.handle.write_at(sblk_addr, &enc)?;
11303 }
11304 }
11305 }
11306
11307 // Statistics.
11308 let c = m.chunked.as_mut().unwrap();
11309 c.chunks_written += 1;
11310 if chunk_idx + 1 > c.ea_header.max_idx_set {
11311 c.ea_header.max_idx_set = chunk_idx + 1;
11312 }
11313 if created {
11314 c.ea_header.num_elmts_realized += loc.dblk_nelmts;
11315 }
11316 }
11317 Ok(())
11318 }
11319
11320 /// Write a slice (hyperslab) of data to a dataset, contiguous or chunked.
11321 ///
11322 /// `starts` and `counts` define the N-dimensional selection.
11323 /// `data` must be exactly `product(counts) * element_size` bytes.
11324 ///
11325 /// The selection is validated once here and then handed to the layout's
11326 /// own writer, so a caller never has to know which storage the dataset
11327 /// uses.
11328 pub fn write_slice(
11329 &self,
11330 index: usize,
11331 starts: &[u64],
11332 counts: &[u64],
11333 data: &[u8],
11334 ) -> IoResult<()> {
11335 let ds = self.ds(index);
11336 let _op = ds.op.lock();
11337 self.write_slice_inner(index, starts, counts, data)
11338 }
11339
11340 /// [`Self::write_slice`] body; the caller holds the dataset's op lock or
11341 /// the writer exclusively.
11342 pub(crate) fn write_slice_inner(
11343 &self,
11344 index: usize,
11345 starts: &[u64],
11346 counts: &[u64],
11347 data: &[u8],
11348 ) -> IoResult<()> {
11349 let ds_ref = self.ds(index);
11350 let ds = ds_ref.lock();
11351 let is_chunked = ds.is_chunked();
11352
11353 let dims = &ds.dataspace.dims;
11354 let element_size = ds.datatype.element_size() as u64;
11355 let ndims = dims.len();
11356
11357 // Every hyperslab edge must stay inside the dataset; without this an
11358 // out-of-bounds selection writes raw bytes over neighbouring data.
11359 check_hyperslab(dims, starts, counts)?;
11360 if ndims == 0 {
11361 return Err(crate::io::IoError::InvalidState(
11362 "write_slice does not support scalar datasets; use write_dataset_raw".into(),
11363 ));
11364 }
11365
11366 let out_elems: u64 = counts.iter().product();
11367 if data.len() as u64 != out_elems * element_size {
11368 return Err(crate::io::IoError::InvalidState(format!(
11369 "data size mismatch: expected {} bytes, got {}",
11370 out_elems * element_size,
11371 data.len()
11372 )));
11373 }
11374
11375 // `dims` borrows the dataset slot; collect what the writers below need
11376 // so the guard can be dropped before they re-lock it.
11377 let dims = dims.clone();
11378 let target = ds.contiguous_target();
11379 drop(ds);
11380
11381 if is_chunked {
11382 // Rows the append buffer holds are not in the chunks yet; writing
11383 // them there anyway would be undone when the buffer flushes at
11384 // close. Hand them to the chunks first.
11385 self.flush_append_buffer_if_intersecting(index, starts[0], starts[0] + counts[0])?;
11386 return self.write_slice_chunked(index, starts, counts, data);
11387 }
11388 let Some(target) = target else {
11389 return Err(crate::io::IoError::InvalidState(
11390 "dataset has no data allocated".into(),
11391 ));
11392 };
11393
11394 // Write each maximal contiguous run in one write. Trailing
11395 // full-selected dimensions coalesce, mirroring the read path: a slice
11396 // with a full last axis becomes one write per outer index instead of
11397 // one write per last-axis row.
11398 for_each_contiguous_run(
11399 &dims,
11400 starts,
11401 counts,
11402 element_size,
11403 |dst_off, src_off, len| {
11404 self.write_contiguous_bytes(&target, dst_off, &data[src_off..src_off + len])
11405 },
11406 )?;
11407
11408 Ok(())
11409 }
11410
11411 /// Write a hyperslab into a chunked dataset, one chunk at a time.
11412 ///
11413 /// The selection is already validated by [`write_slice`](Self::write_slice).
11414 /// For each chunk the selection touches, the chunk's share of `data` is
11415 /// scattered into a whole-chunk buffer and the chunk is rewritten:
11416 ///
11417 /// - a chunk the selection covers completely is built from `data` alone —
11418 /// nothing needs reading back (libhdf5 takes the same shortcut with the
11419 /// `relax` flag of `H5D__chunk_lock`);
11420 /// - a chunk covered only in part starts from what is already stored, or
11421 /// from a fill-value buffer when the chunk has never been written, so
11422 /// neighbouring elements survive and untouched ones read as fill.
11423 ///
11424 /// An edge chunk that hangs past the dataset extent is always the partial
11425 /// case, so the region beyond the extent keeps its fill value.
11426 fn write_slice_chunked(
11427 &self,
11428 index: usize,
11429 starts: &[u64],
11430 counts: &[u64],
11431 data: &[u8],
11432 ) -> IoResult<()> {
11433 if counts.contains(&0) {
11434 return Ok(());
11435 }
11436 let geo = self.chunk_geometry(index)?;
11437 let ndims = geo.dims.len();
11438 if geo.chunk_dims.len() != ndims {
11439 return Err(crate::io::IoError::InvalidState(format!(
11440 "dataset chunk shape has {} dimensions but the dataspace has {}",
11441 geo.chunk_dims.len(),
11442 ndims
11443 )));
11444 }
11445 if geo.chunk_dims.contains(&0) {
11446 return Err(crate::io::IoError::InvalidState(
11447 "chunk shape has a zero-length dimension".into(),
11448 ));
11449 }
11450 let chunk_bytes = geo.chunk_bytes() as usize;
11451
11452 // Grid range the selection touches, inclusive on both ends.
11453 let first: Vec<u64> = (0..ndims).map(|d| starts[d] / geo.chunk_dims[d]).collect();
11454 let last: Vec<u64> = (0..ndims)
11455 .map(|d| (starts[d] + counts[d] - 1) / geo.chunk_dims[d])
11456 .collect();
11457
11458 let mut coords = first.clone();
11459 loop {
11460 // Intersect the selection with this chunk. `in_chunk` is the
11461 // region's origin inside the chunk, `in_data` its origin inside
11462 // the caller's counts-shaped buffer, `extent` its size.
11463 let mut in_chunk = vec![0u64; ndims];
11464 let mut in_data = vec![0u64; ndims];
11465 let mut extent = vec![0u64; ndims];
11466 let mut covers_whole_chunk = true;
11467 for d in 0..ndims {
11468 let chunk_origin = coords[d] * geo.chunk_dims[d];
11469 let lo = starts[d].max(chunk_origin);
11470 let hi = (starts[d] + counts[d]).min(chunk_origin + geo.chunk_dims[d]);
11471 in_chunk[d] = lo - chunk_origin;
11472 in_data[d] = lo - starts[d];
11473 extent[d] = hi - lo;
11474 if in_chunk[d] != 0 || extent[d] != geo.chunk_dims[d] {
11475 covers_whole_chunk = false;
11476 }
11477 }
11478
11479 let mut buf = if covers_whole_chunk {
11480 // Every byte is overwritten below.
11481 vec![0u8; chunk_bytes]
11482 } else {
11483 match self.read_chunk_at_coords(index, &coords)? {
11484 Some(existing) => {
11485 if existing.len() != chunk_bytes {
11486 return Err(crate::io::IoError::InvalidState(format!(
11487 "stored chunk at {coords:?} is {} bytes but the chunk shape \
11488 needs {chunk_bytes}",
11489 existing.len()
11490 )));
11491 }
11492 existing
11493 }
11494 None => self.new_write_chunk_buffer(index, chunk_bytes),
11495 }
11496 };
11497
11498 for_each_dual_run(
11499 &geo.chunk_dims,
11500 &in_chunk,
11501 counts,
11502 &in_data,
11503 &extent,
11504 geo.element_size,
11505 |dst_off, src_off, len| {
11506 let dst = dst_off as usize;
11507 let src = src_off as usize;
11508 buf[dst..dst + len].copy_from_slice(&data[src..src + len]);
11509 Ok(())
11510 },
11511 )?;
11512 self.write_chunk_at_coords(index, &coords, &buf)?;
11513
11514 // Odometer over the touched grid range.
11515 let mut d = ndims;
11516 loop {
11517 if d == 0 {
11518 return Ok(());
11519 }
11520 d -= 1;
11521 if coords[d] < last[d] {
11522 coords[d] += 1;
11523 break;
11524 }
11525 coords[d] = first[d];
11526 }
11527 }
11528 }
11529
11530 /// Add an attribute to the root group (file-level attribute), replacing
11531 /// a same-name attribute. See [`set_attribute`](Self::set_attribute).
11532 pub fn add_root_attribute(&self, attr: AttributeMessage) -> IoResult<()> {
11533 self.set_attribute(AttrTarget::Root, attr)
11534 }
11535
11536 /// Insert `attr` into the attribute list `target` names, replacing a
11537 /// same-name attribute.
11538 ///
11539 /// The single owner of attribute-list mutation: an `AttributeMessage`
11540 /// that leaves a list here has its vlen global-heap objects released, so
11541 /// no replacement — vlen over vlen, numeric over vlen — can strand heap
11542 /// space (the attribute counterpart of issue #10's dataset fix).
11543 ///
11544 /// Under SWMR every attribute mutation is refused, matching libhdf5's
11545 /// rule for SWMR writes. Object headers are frozen once streaming
11546 /// starts — a change was committed at close only when the header
11547 /// happened to be rebuilt (group attrs always, dataset attrs only if
11548 /// the dataset also got chunk writes) and silently dropped otherwise —
11549 /// and a replacement's superseded vlen value could never be reclaimed,
11550 /// since a streaming reader may hold its heap references.
11551 pub fn set_attribute(&self, target: AttrTarget<'_>, attr: AttributeMessage) -> IoResult<()> {
11552 self.insert_attribute(target, attr, Created)
11553 }
11554
11555 /// The body of [`set_attribute`](Self::set_attribute), told whether the
11556 /// attribute it is inserting is genuinely new — see [`AttrOrigin`].
11557 fn insert_attribute(
11558 &self,
11559 target: AttrTarget<'_>,
11560 attr: AttributeMessage,
11561 origin: AttrOrigin,
11562 ) -> IoResult<()> {
11563 if self.swmr_active {
11564 return Err(swmr_attr_error(&attr.name));
11565 }
11566 // Whatever this name meant before, it means the incoming message now.
11567 self.forget_attribute_reference(self.attr_scope(target)?, &attr.name);
11568 // No size gate: an attribute whose message is too large for the
11569 // 16-bit size field an object header message has spills the object's
11570 // whole attribute set to dense storage at finalize, exactly as
11571 // `H5O__attr_create` does. See `attributes_need_dense`.
11572 let mut entry = AttributeEntry::from(attr);
11573 let old = self.with_attr_list(target, |attrs| {
11574 if let Some(pos) = attrs.iter().position(|a| a.name() == entry.name()) {
11575 // `H5O__attr_write` replaces an existing attribute's value and
11576 // leaves its `crt_idx` alone: the attribute was not created
11577 // again, so its creation index does not move.
11578 entry.set_creation_index(attrs[pos].creation_index());
11579 Some(std::mem::replace(&mut attrs[pos], entry))
11580 } else {
11581 // `H5O__attr_create` stamps the set's running maximum onto the
11582 // new attribute and post-increments it — but only a create
11583 // reaches for it.
11584 entry.set_creation_index(match origin {
11585 Created => Some(next_creation_index(attrs)),
11586 Rewritten(kept) => kept,
11587 });
11588 attrs.push(entry);
11589 None
11590 }
11591 })?;
11592 match old {
11593 Some(old) => self.release_attr_vlen(&old),
11594 None => Ok(()),
11595 }
11596 }
11597
11598 /// Set a variable-length string attribute on `target`, replacing any
11599 /// same-name attribute.
11600 ///
11601 /// Owns the whole replacement sequence: the superseded attribute is
11602 /// removed and its heap objects released *before* the new value's
11603 /// collection is allocated — the free-before-alloc order (issue #10)
11604 /// that lets a reopen-replace loop land in the block it just freed
11605 /// instead of growing the file every session. The cost, as on the
11606 /// dataset path: a failure between the eviction and the insert below
11607 /// loses the attribute rather than leaking its heap space.
11608 pub fn set_vlen_string_attribute(
11609 &self,
11610 target: AttrTarget<'_>,
11611 name: &str,
11612 value: &str,
11613 ) -> IoResult<()> {
11614 let origin = self.evict_attr(target, name)?;
11615 let attr = self.vlen_string_attribute(name, value)?;
11616 self.insert_attribute(target, attr, origin)
11617 }
11618
11619 /// The array counterpart of
11620 /// [`set_vlen_string_attribute`](Self::set_vlen_string_attribute).
11621 pub fn set_vlen_string_array_attribute(
11622 &self,
11623 target: AttrTarget<'_>,
11624 name: &str,
11625 values: &[&str],
11626 dims: &[u64],
11627 ) -> IoResult<()> {
11628 let origin = self.evict_attr(target, name)?;
11629 let attr = self.vlen_string_array_attribute(name, values, dims)?;
11630 self.insert_attribute(target, attr, origin)
11631 }
11632
11633 /// Set an attribute on `target` whose value is the object references
11634 /// naming `paths` — h5py's `obj.attrs['ref'] = f['/target'].ref`.
11635 ///
11636 /// `dims` is the attribute's dataspace: empty for the scalar shape a
11637 /// single reference takes, `&[n]` for an array of them. Each path names a
11638 /// dataset or a group (`/` is the root group) and must already exist. What
11639 /// reaches the file is each target's object header address, which finalize
11640 /// assigns — so the paths are what is stored, and the attribute's message
11641 /// is built from them every time an object header is
11642 /// ([`object_attributes`](Self::object_attributes)). The message carries a
11643 /// zero image of the final width until then.
11644 pub fn set_object_reference_attribute(
11645 &self,
11646 target: AttrTarget<'_>,
11647 name: &str,
11648 paths: &[&str],
11649 dims: &[u64],
11650 ) -> IoResult<()> {
11651 let scope = self.attr_scope(target)?;
11652 // An empty `dims` is the scalar shape, whose one element the empty
11653 // product already reports.
11654 let elements: u64 = dims.iter().product();
11655 if elements != paths.len() as u64 {
11656 return Err(crate::io::IoError::InvalidState(format!(
11657 "attribute '{name}' shape {dims:?} needs {elements} references, got {}",
11658 paths.len()
11659 )));
11660 }
11661 // Resolve now as well as at finalize, so a path that names nothing is
11662 // reported at the call that got it wrong.
11663 for path in paths {
11664 self.object_reference_target(path)?;
11665 }
11666 let datatype = DatatypeMessage::object_reference(&self.ctx);
11667 let image = vec![0u8; paths.len() * datatype.element_size() as usize];
11668 let attr = if dims.is_empty() {
11669 AttributeMessage::scalar_numeric(name, datatype, image)
11670 } else {
11671 AttributeMessage::array_numeric(name, datatype, dims, image)
11672 };
11673 // Through the same owner as every other attribute, which is also what
11674 // drops any value this name carried before.
11675 self.set_attribute(target, attr)?;
11676 self.attribute_references
11677 .lock()
11678 .push(AttributeReferenceValue {
11679 scope,
11680 name: name.to_string(),
11681 targets: paths.iter().map(|p| (*p).to_string()).collect(),
11682 stride: self.ctx.sizeof_addr as usize,
11683 });
11684 Ok(())
11685 }
11686
11687 // -----------------------------------------------------------------------
11688 // Dimension scales — the H5DS high-level API (hl/src/H5DS.c)
11689 // -----------------------------------------------------------------------
11690
11691 /// Mark dataset `dsid` as a dimension scale — `H5DSset_scale`.
11692 ///
11693 /// Writes `CLASS` as the fixed-length null-terminated ASCII string
11694 /// `DIMENSION_SCALE` and, when `name` is given, `NAME` the same way: the
11695 /// `H5LT_set_attribute_string` form, one byte longer than the text so the
11696 /// terminator is stored, which is what `H5DSis_scale` requires of a scale
11697 /// (a 16-byte null-terminated `CLASS`). Either attribute already there is
11698 /// deleted and created anew, as `H5LT_set_attribute_string` does, so it
11699 /// takes a fresh creation index. A dataset with scales of its own
11700 /// (`DIMENSION_LIST`) is refused, as upstream refuses it.
11701 pub fn set_dimension_scale(&self, dsid: usize, name: Option<&str>) -> IoResult<()> {
11702 let scale_path = self.dataset_name(dsid)?;
11703 if self.dataset_attribute(dsid, DIMENSION_LIST)?.is_some() {
11704 return Err(crate::io::IoError::InvalidState(format!(
11705 "dataset '{scale_path}' has dimension scales attached and cannot become one"
11706 )));
11707 }
11708 self.set_fixed_string_attribute(dsid, "CLASS", DIMENSION_SCALE_CLASS)?;
11709 if let Some(name) = name {
11710 self.set_fixed_string_attribute(dsid, "NAME", name)?;
11711 }
11712 Ok(())
11713 }
11714
11715 /// Attach dataset `dsid` as a dimension scale of axis `idx` of dataset
11716 /// `did` — `H5DSattach_scale`.
11717 ///
11718 /// Two attributes record the attachment: `DIMENSION_LIST` on `did`, one
11719 /// variable-length sequence of object references per axis (a scalar
11720 /// dataset counts as rank 1), and `REFERENCE_LIST` on `dsid`, an array
11721 /// of `{dataset: H5T_STD_REF_OBJ, dimension: uint}` compounds naming
11722 /// every (dataset, axis) the scale is attached to. `dsid` is then made a
11723 /// scale if it is not one already ([`set_dimension_scale`] with no name).
11724 /// Both lists are rewritten whole; what an existing list holds is read
11725 /// back as paths (registered this session, or resolved from the file's
11726 /// addresses), so an attach in an append session keeps earlier
11727 /// attachments and every reference is stamped with the address its
11728 /// target ends up at.
11729 ///
11730 /// Refused, as upstream refuses them: `did == dsid`; a `did` that is a
11731 /// scale or carries a reserved `CLASS` (`IMAGE`, `PALETTE`, `TABLE`); a
11732 /// `dsid` that has scales of its own; an axis beyond `did`'s rank.
11733 ///
11734 /// Attaching a scale already attached to that axis changes nothing. This
11735 /// is stricter than upstream, which leaves `DIMENSION_LIST` as it is but
11736 /// still appends a duplicate `REFERENCE_LIST` entry; a second entry for
11737 /// the same (dataset, axis) tells `H5DSis_attached` nothing the first
11738 /// does not.
11739 ///
11740 /// [`set_dimension_scale`]: Self::set_dimension_scale
11741 pub fn attach_dimension_scale(&self, did: usize, dsid: usize, idx: usize) -> IoResult<()> {
11742 let data_path = self.dataset_name(did)?;
11743 let scale_path = self.dataset_name(dsid)?;
11744 if did == dsid {
11745 return Err(crate::io::IoError::InvalidState(format!(
11746 "dataset '{data_path}' cannot be its own dimension scale"
11747 )));
11748 }
11749 if self.is_dimension_scale(did)? {
11750 return Err(crate::io::IoError::InvalidState(format!(
11751 "dataset '{data_path}' is a dimension scale and cannot have scales attached"
11752 )));
11753 }
11754 if self.dataset_attribute(dsid, DIMENSION_LIST)?.is_some() {
11755 return Err(crate::io::IoError::InvalidState(format!(
11756 "dataset '{scale_path}' has dimension scales attached and cannot be one"
11757 )));
11758 }
11759 if self.has_reserved_class(did)? {
11760 return Err(crate::io::IoError::InvalidState(format!(
11761 "dataset '{data_path}' holds an image, palette or table and cannot have \
11762 dimension scales"
11763 )));
11764 }
11765 let rank = self.ds(did).lock().dataspace.dims.len().max(1);
11766 if idx >= rank {
11767 return Err(crate::io::IoError::InvalidState(format!(
11768 "axis {idx} is out of range for the rank-{rank} dataset '{data_path}'"
11769 )));
11770 }
11771
11772 let mut lists = match self.dimension_list(did)? {
11773 Some(lists) => lists,
11774 None => vec![Vec::new(); rank],
11775 };
11776 if lists.len() != rank {
11777 return Err(crate::io::IoError::InvalidState(format!(
11778 "DIMENSION_LIST of '{data_path}' has {} entries for a rank-{rank} dataset",
11779 lists.len()
11780 )));
11781 }
11782 if lists[idx].contains(&scale_path) {
11783 return Ok(());
11784 }
11785 lists[idx].push(scale_path);
11786 self.write_dimension_list(did, &lists)?;
11787
11788 let mut entries = self.reference_list(dsid)?;
11789 entries.push((data_path, idx as u32));
11790 self.write_reference_list(dsid, &entries)?;
11791
11792 if !self.is_dimension_scale(dsid)? {
11793 self.set_dimension_scale(dsid, None)?;
11794 }
11795 Ok(())
11796 }
11797
11798 /// The registry name of live dataset `index`, or why there is none.
11799 fn dataset_name(&self, index: usize) -> IoResult<String> {
11800 let count = self.dataset_count();
11801 if index >= count {
11802 return Err(crate::io::IoError::InvalidState(format!(
11803 "dataset index {index} out of range (have {count})"
11804 )));
11805 }
11806 let ds = self.ds(index);
11807 let m = ds.lock();
11808 if m.deleted {
11809 return Err(crate::io::IoError::NotFound(format!(
11810 "dataset '{}' has been deleted",
11811 m.name
11812 )));
11813 }
11814 Ok(m.name.clone())
11815 }
11816
11817 /// The stored attribute `name` of dataset `index`, without marking the
11818 /// header dirty the way [`with_attr_list`](Self::with_attr_list) must.
11819 fn dataset_attribute(&self, index: usize, name: &str) -> IoResult<Option<AttributeEntry>> {
11820 self.dataset_name(index)?;
11821 Ok(self
11822 .ds(index)
11823 .lock()
11824 .attributes
11825 .iter()
11826 .find(|a| a.name() == name)
11827 .cloned())
11828 }
11829
11830 /// Write the scalar fixed-length string attribute `name` = `value` on
11831 /// dataset `index` — `H5LT_set_attribute_string`: the string is stored
11832 /// null-terminated in `strlen + 1` bytes, and an attribute of that name
11833 /// is deleted first rather than written over.
11834 fn set_fixed_string_attribute(&self, index: usize, name: &str, value: &str) -> IoResult<()> {
11835 if value.as_bytes().contains(&0) {
11836 return Err(crate::io::IoError::InvalidState(format!(
11837 "attribute '{name}' value holds an interior NUL"
11838 )));
11839 }
11840 let size = u32::try_from(value.len() + 1).map_err(|_| {
11841 crate::io::IoError::InvalidState(format!(
11842 "attribute '{name}' value of {} bytes exceeds the fixed-string width field",
11843 value.len()
11844 ))
11845 })?;
11846 let mut data = value.as_bytes().to_vec();
11847 data.push(0);
11848 let attr =
11849 AttributeMessage::scalar_numeric(name, DatatypeMessage::fixed_string(size), data);
11850 let target = AttrTarget::Dataset(index);
11851 self.evict_attr(target, name)?;
11852 self.insert_attribute(target, attr, Created)
11853 }
11854
11855 /// The `CLASS` attribute of dataset `index`, read the way `H5DS` reads
11856 /// it: as a C string, up to the first NUL.
11857 fn class_attribute(&self, index: usize) -> IoResult<Option<ClassAttr>> {
11858 use crate::format::global_heap::decode_vlen_reference;
11859
11860 let Some(entry) = self.dataset_attribute(index, "CLASS")? else {
11861 return Ok(None);
11862 };
11863 let msg = entry.decoded().map_err(|reason| {
11864 crate::io::IoError::InvalidState(format!(
11865 "CLASS attribute of '{}' cannot be decoded: {reason}",
11866 self.ds(index).lock().name
11867 ))
11868 })?;
11869 Ok(Some(match &msg.datatype {
11870 DatatypeMessage::FixedString { size, padding, .. } => {
11871 let avail = (*size as usize).min(msg.data.len());
11872 ClassAttr::Fixed {
11873 size: *size,
11874 null_terminated: *padding == 0,
11875 text: c_string(&msg.data[..avail]),
11876 }
11877 }
11878 DatatypeMessage::VarLenString { .. } => {
11879 let (_, addr, obj_idx) = decode_vlen_reference(&msg.data, &self.ctx)?;
11880 let bytes = if addr == 0 || addr == UNDEF_ADDR {
11881 Vec::new()
11882 } else {
11883 let obj_idx = u16::try_from(obj_idx).map_err(|_| {
11884 crate::io::IoError::InvalidState(format!(
11885 "global heap object index {obj_idx} does not fit the 16-bit on-disk \
11886 field"
11887 ))
11888 })?;
11889 self.read_heap_object(addr, obj_idx)?
11890 };
11891 ClassAttr::VarLen(c_string(&bytes))
11892 }
11893 _ => ClassAttr::NotString,
11894 }))
11895 }
11896
11897 /// `H5DSis_scale`: a `CLASS` that is a string saying `DIMENSION_SCALE` —
11898 /// and, for a fixed-length string, null-terminated and exactly 16 bytes
11899 /// wide, the width the spec gives the attribute.
11900 fn is_dimension_scale(&self, index: usize) -> IoResult<bool> {
11901 Ok(match self.class_attribute(index)? {
11902 None | Some(ClassAttr::NotString) => false,
11903 Some(ClassAttr::Fixed {
11904 size,
11905 null_terminated,
11906 text,
11907 }) => null_terminated && size == 16 && text == DIMENSION_SCALE_CLASS,
11908 Some(ClassAttr::VarLen(text)) => text == DIMENSION_SCALE_CLASS,
11909 })
11910 }
11911
11912 /// `H5DS_is_reserved`: a `CLASS` naming an image, palette or table — the
11913 /// datasets the other high-level APIs own. A `CLASS` that is not a string
11914 /// is an error here, where [`is_dimension_scale`](Self::is_dimension_scale)
11915 /// reads it as "not a scale", because that is how upstream splits them.
11916 fn has_reserved_class(&self, index: usize) -> IoResult<bool> {
11917 Ok(match self.class_attribute(index)? {
11918 None => false,
11919 Some(ClassAttr::NotString) => {
11920 return Err(crate::io::IoError::InvalidState(format!(
11921 "CLASS attribute of '{}' is not a string",
11922 self.ds(index).lock().name
11923 )))
11924 }
11925 Some(ClassAttr::Fixed { text, .. }) | Some(ClassAttr::VarLen(text)) => {
11926 matches!(text.as_str(), "IMAGE" | "PALETTE" | "TABLE")
11927 }
11928 })
11929 }
11930
11931 /// The bytes of object `index` in the global heap collection at
11932 /// `collection` — `H5HG_read`.
11933 fn read_heap_object(&self, collection: u64, index: u16) -> IoResult<Vec<u8>> {
11934 use crate::format::global_heap::GlobalHeapCollection;
11935
11936 let mut image = self.handle.read_at_most(collection, 4096)?;
11937 let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
11938 if declared > image.len() {
11939 image = self.handle.read_at(collection, declared)?;
11940 }
11941 let (gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
11942 gcol.get_object(index).map(<[u8]>::to_vec).ok_or_else(|| {
11943 crate::io::IoError::InvalidState(format!(
11944 "global heap collection {collection:#x} has no object {index}"
11945 ))
11946 })
11947 }
11948
11949 /// The path of the object whose header is at `addr` in the file as it
11950 /// was opened — what an object reference read back from an append
11951 /// session's existing attributes names.
11952 fn path_of_header_address(&self, addr: u64) -> IoResult<String> {
11953 for ds in self.dataset_refs() {
11954 let m = ds.lock();
11955 if !m.deleted && m.obj_header_written_addr == Some(addr) {
11956 return Ok(m.name.clone());
11957 }
11958 }
11959 for grp in self.group_refs() {
11960 let g = grp.lock();
11961 if !g.deleted && g.obj_header_written_addr == Some(addr) {
11962 return Ok(g.name.clone());
11963 }
11964 }
11965 Err(crate::io::IoError::InvalidState(format!(
11966 "object reference to header {addr:#x} names no dataset or group of this file"
11967 )))
11968 }
11969
11970 /// The path a reference slot inside a global heap object names: the one
11971 /// registered for stamping when this session wrote the slot, else the
11972 /// one the address on disk resolves to.
11973 fn heap_reference_path(
11974 &self,
11975 collection: u64,
11976 index: u16,
11977 token_offset: usize,
11978 on_disk: &[u8],
11979 ) -> IoResult<String> {
11980 let registered = self
11981 .pending_heap_references
11982 .lock()
11983 .iter()
11984 .find(|p| {
11985 p.collection == collection && p.index == index && p.token_offset == token_offset
11986 })
11987 .map(|p| match &p.target {
11988 PendingHeapTarget::Dataset(path) | PendingHeapTarget::Object(path) => path.clone(),
11989 });
11990 if let Some(path) = registered {
11991 return Ok(path);
11992 }
11993 let mut raw = [0u8; 8];
11994 raw[..on_disk.len()].copy_from_slice(on_disk);
11995 self.path_of_header_address(u64::from_le_bytes(raw))
11996 }
11997
11998 /// Dataset `did`'s `DIMENSION_LIST` as the paths of the scales on each
11999 /// axis, or `None` when it has no such attribute.
12000 fn dimension_list(&self, did: usize) -> IoResult<Option<Vec<Vec<String>>>> {
12001 use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
12002
12003 let Some(entry) = self.dataset_attribute(did, DIMENSION_LIST)? else {
12004 return Ok(None);
12005 };
12006 let name = || self.ds(did).lock().name.clone();
12007 let msg = entry.decoded().map_err(|reason| {
12008 crate::io::IoError::InvalidState(format!(
12009 "DIMENSION_LIST of '{}' cannot be decoded: {reason}",
12010 name()
12011 ))
12012 })?;
12013 match &msg.datatype {
12014 DatatypeMessage::VarLenSequence { base }
12015 if matches!(
12016 **base,
12017 DatatypeMessage::Reference {
12018 kind: ReferenceKind::Object1,
12019 ..
12020 }
12021 ) => {}
12022 other => {
12023 return Err(crate::io::IoError::InvalidState(format!(
12024 "DIMENSION_LIST of '{}' is {other}; only a sequence of H5T_STD_REF_OBJ \
12025 references is supported",
12026 name()
12027 )))
12028 }
12029 }
12030 let sa = self.ctx.sizeof_addr as usize;
12031 let ref_size = vlen_reference_size(&self.ctx);
12032 let mut lists = Vec::new();
12033 for elem in msg.data.chunks_exact(ref_size) {
12034 let (seq_len, addr, obj_idx) = decode_vlen_reference(elem, &self.ctx)?;
12035 let seq_len = seq_len as usize;
12036 let mut paths = Vec::with_capacity(seq_len);
12037 if seq_len > 0 {
12038 let index = u16::try_from(obj_idx).map_err(|_| {
12039 crate::io::IoError::InvalidState(format!(
12040 "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
12041 ))
12042 })?;
12043 let bytes = self.read_heap_object(addr, index)?;
12044 if bytes.len() < seq_len * sa {
12045 return Err(crate::io::IoError::InvalidState(format!(
12046 "DIMENSION_LIST of '{}' names {seq_len} scales in a {}-byte heap object",
12047 name(),
12048 bytes.len()
12049 )));
12050 }
12051 for k in 0..seq_len {
12052 paths.push(self.heap_reference_path(
12053 addr,
12054 index,
12055 k * sa,
12056 &bytes[k * sa..(k + 1) * sa],
12057 )?);
12058 }
12059 }
12060 lists.push(paths);
12061 }
12062 Ok(Some(lists))
12063 }
12064
12065 /// Store `lists` — the scales attached to each axis — as dataset `did`'s
12066 /// `DIMENSION_LIST`, replacing the one it has.
12067 ///
12068 /// Each axis is one global heap object of `sizeof_addr` bytes per scale,
12069 /// zero until finalize stamps the scale's header address in through
12070 /// [`write_heap_reference_values`](Self::write_heap_reference_values);
12071 /// an axis with no scale is an empty heap object, as libhdf5's
12072 /// `H5VL__native_blob_put` stores an empty sequence. The attribute's
12073 /// value is the vlen reference to each object, final at write time.
12074 fn write_dimension_list(&self, did: usize, lists: &[Vec<String>]) -> IoResult<()> {
12075 use crate::format::global_heap::{
12076 encode_vlen_reference, vlen_reference_size, vlen_seq_len,
12077 };
12078
12079 let target = AttrTarget::Dataset(did);
12080 let origin = self.evict_attr(target, DIMENSION_LIST)?;
12081 let sa = self.ctx.sizeof_addr as usize;
12082 let blobs: Vec<Vec<u8>> = lists.iter().map(|l| vec![0u8; l.len() * sa]).collect();
12083 let items: Vec<&[u8]> = blobs.iter().map(Vec::as_slice).collect();
12084 let placements = self.insert_vlen_objects(&items)?;
12085
12086 let mut data = Vec::with_capacity(lists.len() * vlen_reference_size(&self.ctx));
12087 let mut pending = self.pending_heap_references.lock();
12088 for (axis, &(collection, index)) in placements.iter().enumerate() {
12089 for (k, path) in lists[axis].iter().enumerate() {
12090 pending.push(PendingHeapReference {
12091 collection,
12092 index,
12093 token_offset: k * sa,
12094 target: PendingHeapTarget::Object(path.clone()),
12095 });
12096 }
12097 data.extend_from_slice(&encode_vlen_reference(
12098 vlen_seq_len(lists[axis].len())?,
12099 collection,
12100 u32::from(index),
12101 &self.ctx,
12102 ));
12103 }
12104 drop(pending);
12105
12106 let attr = AttributeMessage {
12107 name: DIMENSION_LIST.to_string(),
12108 datatype: DatatypeMessage::VarLenSequence {
12109 base: Box::new(DatatypeMessage::object_reference(&self.ctx)),
12110 },
12111 dataspace: DataspaceMessage::simple(&[lists.len() as u64]),
12112 data,
12113 };
12114 self.insert_attribute(target, attr, origin)
12115 }
12116
12117 /// Scale `dsid`'s `REFERENCE_LIST` as (dataset path, axis) pairs; empty
12118 /// when it has no such attribute.
12119 fn reference_list(&self, dsid: usize) -> IoResult<Vec<(String, u32)>> {
12120 let Some(entry) = self.dataset_attribute(dsid, REFERENCE_LIST)? else {
12121 return Ok(Vec::new());
12122 };
12123 let name = || self.ds(dsid).lock().name.clone();
12124 let msg = entry.decoded().map_err(|reason| {
12125 crate::io::IoError::InvalidState(format!(
12126 "REFERENCE_LIST of '{}' cannot be decoded: {reason}",
12127 name()
12128 ))
12129 })?;
12130 let unsupported = |why: String| {
12131 crate::io::IoError::InvalidState(format!(
12132 "REFERENCE_LIST of '{}' is {}; {why}",
12133 name(),
12134 msg.datatype
12135 ))
12136 };
12137 let DatatypeMessage::Compound { size, members } = &msg.datatype else {
12138 return Err(unsupported("a compound is required".into()));
12139 };
12140 let member = |m: &str| {
12141 members
12142 .iter()
12143 .find(|c| c.name == m)
12144 .ok_or_else(|| unsupported(format!("member '{m}' is missing")))
12145 };
12146 let dataset = member("dataset")?;
12147 let dimension = member("dimension")?;
12148 let sa = self.ctx.sizeof_addr as usize;
12149 if !matches!(
12150 dataset.datatype,
12151 DatatypeMessage::Reference {
12152 kind: ReferenceKind::Object1,
12153 ..
12154 }
12155 ) {
12156 return Err(unsupported(
12157 "only an H5T_STD_REF_OBJ 'dataset' member is supported".into(),
12158 ));
12159 }
12160 let DatatypeMessage::FixedPoint {
12161 size: 4,
12162 byte_order,
12163 ..
12164 } = dimension.datatype
12165 else {
12166 return Err(unsupported(
12167 "a 4-byte integer 'dimension' member is required".into(),
12168 ));
12169 };
12170 let stride = *size as usize;
12171 let registered: Option<Vec<String>> = self
12172 .attribute_references
12173 .lock()
12174 .iter()
12175 .find(|r| r.scope == AttrScope::Dataset(dsid) && r.name == REFERENCE_LIST)
12176 .map(|r| r.targets.clone());
12177 let mut entries = Vec::with_capacity(msg.data.len() / stride);
12178 for (i, elem) in msg.data.chunks_exact(stride).enumerate() {
12179 let at = |offset: u32, len: usize| {
12180 elem.get(offset as usize..offset as usize + len)
12181 .ok_or_else(|| unsupported(format!("element {i} is too short for its members")))
12182 };
12183 let path = match ®istered {
12184 Some(targets) => targets.get(i).cloned().ok_or_else(|| {
12185 crate::io::IoError::InvalidState(format!(
12186 "REFERENCE_LIST of '{}' entry {i} has no registered target",
12187 name()
12188 ))
12189 })?,
12190 None => {
12191 let mut raw = [0u8; 8];
12192 raw[..sa].copy_from_slice(at(dataset.offset, sa)?);
12193 self.path_of_header_address(u64::from_le_bytes(raw))?
12194 }
12195 };
12196 let dim: [u8; 4] = at(dimension.offset, 4)?.try_into().expect("4 bytes");
12197 let dim = match byte_order {
12198 ByteOrder::LittleEndian => u32::from_le_bytes(dim),
12199 ByteOrder::BigEndian => u32::from_be_bytes(dim),
12200 };
12201 entries.push((path, dim));
12202 }
12203 Ok(entries)
12204 }
12205
12206 /// Store `entries` as scale `dsid`'s `REFERENCE_LIST`, replacing the one
12207 /// it has — deleted and created anew, as upstream does, so it takes a
12208 /// fresh creation index.
12209 ///
12210 /// The element is libhdf5's `ds_list_t` as it lands on disk: the
12211 /// reference at offset 0, `dimension` right after it, and the struct's
12212 /// trailing padding — 16 bytes over 8-byte addresses. The addresses are
12213 /// stamped at finalize through [`object_attributes`](Self::object_attributes)
12214 /// like any reference attribute's; the `dimension` fields are final here.
12215 fn write_reference_list(&self, dsid: usize, entries: &[(String, u32)]) -> IoResult<()> {
12216 use crate::format::messages::datatype::CompoundMember;
12217
12218 let target = AttrTarget::Dataset(dsid);
12219 self.evict_attr(target, REFERENCE_LIST)?;
12220 let sa = self.ctx.sizeof_addr as usize;
12221 let stride = sa + 8;
12222 let datatype = DatatypeMessage::compound(
12223 stride as u32,
12224 vec![
12225 CompoundMember {
12226 name: "dataset".to_string(),
12227 offset: 0,
12228 datatype: DatatypeMessage::object_reference(&self.ctx),
12229 },
12230 CompoundMember {
12231 name: "dimension".to_string(),
12232 offset: sa as u32,
12233 datatype: DatatypeMessage::u32_type(),
12234 },
12235 ],
12236 );
12237 let mut data = vec![0u8; entries.len() * stride];
12238 for (i, (_, dim)) in entries.iter().enumerate() {
12239 data[i * stride + sa..i * stride + sa + 4].copy_from_slice(&dim.to_le_bytes());
12240 }
12241 let attr = AttributeMessage::array_numeric(
12242 REFERENCE_LIST,
12243 datatype,
12244 &[entries.len() as u64],
12245 data,
12246 );
12247 self.insert_attribute(target, attr, Created)?;
12248 self.attribute_references
12249 .lock()
12250 .push(AttributeReferenceValue {
12251 scope: AttrScope::Dataset(dsid),
12252 name: REFERENCE_LIST.to_string(),
12253 targets: entries.iter().map(|(p, _)| p.clone()).collect(),
12254 stride,
12255 });
12256 Ok(())
12257 }
12258
12259 /// Take the attribute `name` off `target`'s list, releasing its heap
12260 /// objects. No-op when absent. Refused under SWMR — see
12261 /// [`set_attribute`](Self::set_attribute).
12262 ///
12263 /// What it answers is what the insert that follows it must be told: an
12264 /// attribute that was there is being rewritten and keeps its creation
12265 /// index, and one that was not is created.
12266 fn evict_attr(&self, target: AttrTarget<'_>, name: &str) -> IoResult<AttrOrigin> {
12267 if self.swmr_active {
12268 return Err(swmr_attr_error(name));
12269 }
12270 self.forget_attribute_reference(self.attr_scope(target)?, name);
12271 let old = self.with_attr_list(target, |attrs| {
12272 attrs
12273 .iter()
12274 .position(|a| a.name() == name)
12275 .map(|pos| attrs.remove(pos))
12276 })?;
12277 match old {
12278 Some(old) => {
12279 let origin = Rewritten(old.creation_index());
12280 self.release_attr_vlen(&old)?;
12281 Ok(origin)
12282 }
12283 None => Ok(Created),
12284 }
12285 }
12286
12287 /// Release the global-heap objects a superseded attribute owned.
12288 /// Recognizes top-level vlen datatypes only: a *compound* attribute
12289 /// with vlen members — which this crate cannot write, only a foreign
12290 /// file can carry — keeps its members' heap objects when replaced or
12291 /// deleted, the storage cost the foreign writer accepted. Every other
12292 /// class stores its value inline in the message. Per-object removal
12293 /// keeps collections shared with other refs (libhdf5-written files)
12294 /// intact.
12295 fn release_attr_vlen(&self, old: &AttributeEntry) -> IoResult<()> {
12296 use crate::format::messages::datatype::DatatypeMessage;
12297 // An attribute whose message this crate could not decode keeps
12298 // whatever heap space it references: releasing objects named by bytes
12299 // we cannot interpret would free storage that is still live.
12300 let Some(old) = old.readable() else {
12301 return Ok(());
12302 };
12303 if matches!(
12304 old.datatype,
12305 DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
12306 ) {
12307 self.release_vlen_references(&old.data)?;
12308 }
12309 Ok(())
12310 }
12311
12312 /// Run `f` on the attribute list `target` names — the accessor every
12313 /// attribute mutation shares.
12314 fn with_attr_list<R>(
12315 &self,
12316 target: AttrTarget<'_>,
12317 f: impl FnOnce(&mut Vec<AttributeEntry>) -> R,
12318 ) -> IoResult<R> {
12319 match target {
12320 AttrTarget::Root => Ok(f(&mut self.root_attributes.lock())),
12321 AttrTarget::Group(path) => {
12322 let path = self.canonical_group_path(path);
12323 for grp in self.group_refs() {
12324 let mut g = grp.lock();
12325 if g.name == path && !g.deleted {
12326 return Ok(f(&mut g.attributes));
12327 }
12328 }
12329 Err(crate::io::IoError::NotFound(format!(
12330 "group '{path}' not found"
12331 )))
12332 }
12333 AttrTarget::Dataset(index) => {
12334 let count = self.dataset_count();
12335 if index >= count {
12336 return Err(crate::io::IoError::InvalidState(format!(
12337 "dataset index {index} out of range (have {count})"
12338 )));
12339 }
12340 let ds = self.ds(index);
12341 let mut m = ds.lock();
12342 // Every caller of this mutates the list, and a reopened
12343 // dataset's header is rewritten only when it is marked stale.
12344 m.header_dirty = true;
12345 Ok(f(&mut m.attributes))
12346 }
12347 }
12348 }
12349
12350 /// Store each of `items` as a global heap object and return its
12351 /// placement `(collection address, object index)`, in input order —
12352 /// the writer side of libhdf5's `H5HG_insert`.
12353 ///
12354 /// Placement follows libhdf5: a collection from the CWFS list takes an
12355 /// item when its free space holds the object *and* a residual
12356 /// free-space marker header (`encode_at_size` always emits the
12357 /// marker); what no listed collection can take goes into a fresh
12358 /// collection, spilling into another at the 65535-object index cap.
12359 /// One batch may therefore span several collections — invisible to
12360 /// readers, which resolve each reference's own collection address. An
12361 /// empty batch allocates nothing: an empty collection still encodes
12362 /// to the 4096-byte `H5HG_MINALLOC` minimum, a block nothing would
12363 /// reference. libhdf5 additionally tries to extend a nearly-full
12364 /// collection's block in place (`H5MF_try_extend`); this writer does
12365 /// not — an oversized item always starts a fresh collection.
12366 ///
12367 /// The `cwfs` lock is held across every read-modify-rewrite of a
12368 /// listed collection block: it serializes concurrent inserts (two
12369 /// datasets' writers can pack the same block) and inserts against
12370 /// [`release_vlen_references`](Self::release_vlen_references), which
12371 /// rewrites the same blocks when objects are freed.
12372 ///
12373 /// Under SWMR the CWFS list is neither consulted nor updated and every
12374 /// batch gets fresh collections: packing rewrites a block a streaming
12375 /// reader may be mid-walk on — the same reason `place_chunk` keeps a
12376 /// relocated chunk's old block.
12377 fn insert_vlen_objects(&self, items: &[&[u8]]) -> IoResult<Vec<(u64, u16)>> {
12378 use crate::format::global_heap::{GlobalHeapCollection, GlobalHeapObject};
12379
12380 if items.is_empty() {
12381 return Ok(Vec::new());
12382 }
12383 let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
12384 let mut placements = Vec::with_capacity(items.len());
12385 let mut i = 0;
12386
12387 // Pack into listed collections while one can take the next item.
12388 if !self.swmr_active {
12389 let mut cwfs = self.cwfs.lock();
12390 while i < items.len() {
12391 let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
12392 let Some(pos) = cwfs.iter().position(|e| e.free >= need + objhdr) else {
12393 // Second pass of libhdf5's H5F_cwfs_find_free_heap: no
12394 // listed collection has room, so try to grow one in
12395 // place before falling back to a fresh collection.
12396 if self.extend_listed_collection(&mut cwfs, need + objhdr)? {
12397 continue;
12398 }
12399 break;
12400 };
12401 let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
12402 let image = self.handle.read_at(addr, size)?;
12403 let (mut gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
12404 // The disk is the truth for free space; the entry is a hint.
12405 let Some(mut free) = gcol.free_space_at(&self.ctx, size) else {
12406 cwfs.remove(pos);
12407 continue;
12408 };
12409 let mut next_idx = gcol.max_index();
12410 let mut took = false;
12411 while i < items.len() && next_idx < u16::MAX {
12412 let need = GlobalHeapCollection::object_disk_size(&self.ctx, items[i].len());
12413 if free < need + objhdr {
12414 break;
12415 }
12416 next_idx += 1;
12417 gcol.objects.push(GlobalHeapObject {
12418 index: next_idx,
12419 ref_count: 0,
12420 data: items[i].to_vec(),
12421 });
12422 placements.push((addr, next_idx));
12423 free -= need;
12424 took = true;
12425 i += 1;
12426 }
12427 if took {
12428 let rewritten = gcol.encode_at_size(&self.ctx, size)?;
12429 self.handle.write_at(addr, &rewritten)?;
12430 // Correct the entry to the measured free space and move
12431 // it to the front — libhdf5 keeps `cwfs` in
12432 // most-recently-used order.
12433 let mut e = cwfs.remove(pos);
12434 e.free = free;
12435 cwfs.insert(0, e);
12436 } else if next_idx == u16::MAX {
12437 // At the index cap nothing can be inserted no matter the
12438 // free space; drop the entry or the scan re-picks it
12439 // forever. (A removal can lower the top index again, and
12440 // the release side re-lists the collection then.)
12441 cwfs.remove(pos);
12442 } else {
12443 // The hint overstated the block's free space — shrink it
12444 // to the measured value so the scan moves on.
12445 cwfs[pos].free = free;
12446 }
12447 }
12448 }
12449
12450 // What remains goes into fresh collections.
12451 while i < items.len() {
12452 let mut gcol = GlobalHeapCollection::new();
12453 // Objects are pushed with a running index: `add_object` rescans
12454 // for the max index per call, O(n²) across a spill-sized batch.
12455 let mut next_idx: u16 = 0;
12456 while i < items.len() && next_idx < u16::MAX {
12457 next_idx += 1;
12458 gcol.objects.push(GlobalHeapObject {
12459 index: next_idx,
12460 ref_count: 0,
12461 data: items[i].to_vec(),
12462 });
12463 i += 1;
12464 }
12465 let encoded = gcol.encode(&self.ctx);
12466 let addr = self
12467 .allocator
12468 .allocate(encoded.len() as u64, FreeSpaceClass::RawData);
12469 self.handle.write_at(addr, &encoded)?;
12470 for idx in 1..=next_idx {
12471 placements.push((addr, idx));
12472 }
12473 // List the block's leftover free space for later inserts — the
12474 // minimum-size padding of a small batch is most of 4096 bytes.
12475 // Below two object headers not even an empty object fits.
12476 if !self.swmr_active {
12477 if let Some(free) = gcol.free_space_at(&self.ctx, encoded.len()) {
12478 if free >= 2 * objhdr {
12479 cwfs_note(&mut self.cwfs.lock(), addr, encoded.len(), free);
12480 }
12481 }
12482 }
12483 }
12484 Ok(placements)
12485 }
12486
12487 /// Try to extend one listed collection in place so it can take an
12488 /// object needing `want` bytes of free space — the second pass of
12489 /// libhdf5's `H5F_cwfs_find_free_heap`: grow the file allocation
12490 /// ([`FileAllocator::try_extend`], mirroring `H5MF_try_extend`) and then
12491 /// the collection itself (`H5HG_extend`: a larger declared size and a
12492 /// free-space marker covering the new tail — here by re-encoding at the
12493 /// grown size, which writes exactly those two things).
12494 ///
12495 /// Extension size is `max(collection_size, shortfall)` — at least a
12496 /// doubling — capped so the result stays within [`GCOL_MAX_SIZE`], both
12497 /// as upstream computes them. On success the grown entry moves to the
12498 /// front of the list and the caller's scan re-picks it; the free-space
12499 /// measurement is taken from the block on disk, not the list's hint, so
12500 /// the rewrite and the entry agree.
12501 ///
12502 /// Caller holds the `cwfs` lock (it passes the guarded list), which is
12503 /// what serializes this read-modify-rewrite against concurrent inserts
12504 /// and releases.
12505 fn extend_listed_collection(&self, cwfs: &mut Vec<CwfsEntry>, want: usize) -> IoResult<bool> {
12506 use crate::format::global_heap::{GlobalHeapCollection, GCOL_MAX_SIZE};
12507
12508 let mut pos = 0;
12509 while pos < cwfs.len() {
12510 let (addr, size) = (cwfs[pos].addr, cwfs[pos].size);
12511 let image = self.handle.read_at(addr, size)?;
12512 let (gcol, _) = GlobalHeapCollection::decode(&image[..size], &self.ctx)?;
12513 // The disk is the truth for free space; the entry is a hint.
12514 let Some(free) = gcol.free_space_at(&self.ctx, size) else {
12515 cwfs.remove(pos);
12516 continue;
12517 };
12518 // A hint can understate the block (upstream's FREE_SIZE is its
12519 // in-memory truth and cannot): if the block already has room,
12520 // correct the hint instead of doubling the collection.
12521 if free >= want {
12522 cwfs[pos].free = free;
12523 return Ok(true);
12524 }
12525 let new_need = size.max(want.saturating_sub(free));
12526 if size + new_need > GCOL_MAX_SIZE
12527 || !self.allocator.try_extend(
12528 addr,
12529 size as u64,
12530 new_need as u64,
12531 FreeSpaceClass::RawData,
12532 )
12533 {
12534 pos += 1;
12535 continue;
12536 }
12537 let new_size = size + new_need;
12538 let rewritten = gcol.encode_at_size(&self.ctx, new_size)?;
12539 self.handle.write_at(addr, &rewritten)?;
12540 let mut e = cwfs.remove(pos);
12541 e.size = new_size;
12542 e.free = free + new_need;
12543 cwfs.insert(0, e);
12544 return Ok(true);
12545 }
12546 Ok(false)
12547 }
12548
12549 /// Create a variable-length string dataset and write string data.
12550 ///
12551 /// Stores strings in the global heap. The dataset raw data consists of
12552 /// vlen references (collection_addr + object_index pairs).
12553 ///
12554 /// `charset` is the datatype's declared character set (0 = ASCII,
12555 /// 1 = UTF-8); the strings are checked against it before anything is
12556 /// written, so the type never misdescribes the bytes under it.
12557 pub fn create_vlen_string_dataset(
12558 &self,
12559 name: &str,
12560 strings: &[&str],
12561 charset: u8,
12562 ) -> IoResult<usize> {
12563 use crate::format::global_heap::encode_vlen_reference;
12564 use crate::format::messages::datatype::DatatypeMessage;
12565
12566 ensure_vlen_charset(charset, strings)?;
12567
12568 let create = self.begin_create(name)?;
12569 let name = create.name.as_str();
12570 let num_strings = strings.len() as u64;
12571
12572 // Store the strings as heap objects; a batch that fits an earlier
12573 // collection's free space shares its block.
12574 let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12575 let placements = self.insert_vlen_objects(&items)?;
12576
12577 // Build raw data: vlen references
12578 let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12579 let data_size = (num_strings as usize) * ref_size;
12580 let mut raw_data = Vec::with_capacity(data_size);
12581 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12582 let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12583 raw_data.extend_from_slice(&encode_vlen_reference(
12584 seq_len,
12585 gcol_addr,
12586 obj_idx as u32,
12587 &self.ctx,
12588 ));
12589 }
12590
12591 // Allocate and write raw data
12592 let data_addr = self
12593 .allocator
12594 .allocate(data_size as u64, FreeSpaceClass::RawData);
12595 self.handle.write_at(data_addr, &raw_data)?;
12596
12597 // Create the dataset with vlen string datatype
12598 let datatype = DatatypeMessage::VarLenString {
12599 padding: 0,
12600 charset,
12601 };
12602 let dataspace =
12603 crate::format::messages::dataspace::DataspaceMessage::simple(&[num_strings]);
12604
12605 let idx = self.push_dataset(
12606 &create,
12607 DatasetInfo {
12608 name: name.to_string(),
12609 datatype,
12610 committed_type: None,
12611 external: None,
12612 virtual_storage: None,
12613 dataspace,
12614 read_format: None,
12615 obj_header_addr: 0,
12616 data_addr,
12617 data_size: data_size as u64,
12618 compact: None,
12619 attributes: Vec::new(),
12620 obj_header_written_addr: None,
12621 obj_header_blocks: Vec::new(),
12622 filter_pipeline: None,
12623 deleted: false,
12624 extent_dirty: false,
12625 header_dirty: false,
12626 nlink_written: 1,
12627 creation_seq: self.take_creation_seq(),
12628 track_attr_order: self.track_order.attrs,
12629 fill_value: None,
12630 fill_time: FILL_TIME_IFSET,
12631 layout_version: 4,
12632 times: self.created_object_times(),
12633 chunked: None,
12634 fixed_array: None,
12635 implicit: None,
12636 single_chunk: None,
12637 btree_v1: None,
12638 btree_v2: None,
12639 append: None,
12640 },
12641 );
12642
12643 Ok(idx)
12644 }
12645
12646 /// Create a 1-D variable-length byte-array dataset.
12647 ///
12648 /// The `u8` case of [`create_vlen_sequence_dataset`], where an item's
12649 /// byte image and its element count are the same number.
12650 ///
12651 /// [`create_vlen_sequence_dataset`]: Self::create_vlen_sequence_dataset
12652 ///
12653 /// Superseded in production by [`write_vlen_numeric`](crate::H5File::write_vlen_numeric)
12654 /// (`H5Group::write_vlen_bytes` routes through it, not through here);
12655 /// kept as a direct entry point for this crate's own white-box tests.
12656 #[cfg(test)]
12657 pub fn create_vlen_bytes_dataset(&self, name: &str, items: &[&[u8]]) -> IoResult<usize> {
12658 use crate::format::messages::datatype::DatatypeMessage;
12659
12660 self.create_vlen_sequence_dataset(name, DatatypeMessage::u8_type(), items)
12661 }
12662
12663 /// Create a 1-D variable-length sequence dataset over `base`.
12664 ///
12665 /// Each item is the encoded image of one sequence — `n * base.element_size()`
12666 /// bytes in the base type's own byte order — and is stored as a global-heap
12667 /// object; the dataset holds one vlen reference per item, the same on-disk
12668 /// shape a vlen string dataset has. The `H5T_VLEN` length field counts base
12669 /// elements rather than bytes, so an image whose length is not a whole
12670 /// number of elements is refused here rather than stored under a length
12671 /// that misreads it.
12672 pub fn create_vlen_sequence_dataset(
12673 &self,
12674 name: &str,
12675 base: DatatypeMessage,
12676 items: &[&[u8]],
12677 ) -> IoResult<usize> {
12678 use crate::format::global_heap::encode_vlen_reference;
12679 use crate::format::messages::datatype::DatatypeMessage;
12680
12681 let elem_size = base.element_size() as usize;
12682 if elem_size == 0 {
12683 return Err(crate::io::IoError::InvalidState(format!(
12684 "vlen base datatype {base} has no element size"
12685 )));
12686 }
12687 for (i, item) in items.iter().enumerate() {
12688 if !item.len().is_multiple_of(elem_size) {
12689 return Err(crate::io::IoError::InvalidState(format!(
12690 "sequence {i} is {} bytes, not a whole number of {elem_size}-byte elements",
12691 item.len()
12692 )));
12693 }
12694 }
12695
12696 let create = self.begin_create(name)?;
12697 let name = create.name.as_str();
12698 let num_items = items.len() as u64;
12699
12700 // Store the sequence images as heap objects, sharing collection
12701 // blocks as `create_vlen_string_dataset` does.
12702 let placements = self.insert_vlen_objects(items)?;
12703
12704 // Build raw data: one vlen reference per item.
12705 let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12706 let data_size = (num_items as usize) * ref_size;
12707 let mut raw_data = Vec::with_capacity(data_size);
12708 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12709 let seq_len = crate::format::global_heap::vlen_seq_len(items[i].len() / elem_size)?;
12710 raw_data.extend_from_slice(&encode_vlen_reference(
12711 seq_len,
12712 gcol_addr,
12713 obj_idx as u32,
12714 &self.ctx,
12715 ));
12716 }
12717
12718 // Allocate and write raw data.
12719 let data_addr = self
12720 .allocator
12721 .allocate(data_size as u64, FreeSpaceClass::RawData);
12722 self.handle.write_at(data_addr, &raw_data)?;
12723
12724 let datatype = DatatypeMessage::VarLenSequence {
12725 base: Box::new(base),
12726 };
12727 let dataspace = crate::format::messages::dataspace::DataspaceMessage::simple(&[num_items]);
12728
12729 let idx = self.push_dataset(
12730 &create,
12731 DatasetInfo {
12732 name: name.to_string(),
12733 datatype,
12734 committed_type: None,
12735 external: None,
12736 virtual_storage: None,
12737 dataspace,
12738 read_format: None,
12739 obj_header_addr: 0,
12740 data_addr,
12741 data_size: data_size as u64,
12742 compact: None,
12743 attributes: Vec::new(),
12744 obj_header_written_addr: None,
12745 obj_header_blocks: Vec::new(),
12746 filter_pipeline: None,
12747 deleted: false,
12748 extent_dirty: false,
12749 header_dirty: false,
12750 nlink_written: 1,
12751 creation_seq: self.take_creation_seq(),
12752 track_attr_order: self.track_order.attrs,
12753 fill_value: None,
12754 fill_time: FILL_TIME_IFSET,
12755 layout_version: 4,
12756 times: self.created_object_times(),
12757 chunked: None,
12758 fixed_array: None,
12759 implicit: None,
12760 single_chunk: None,
12761 btree_v1: None,
12762 btree_v2: None,
12763 append: None,
12764 },
12765 );
12766
12767 Ok(idx)
12768 }
12769
12770 /// Create a chunked, compressed variable-length string dataset.
12771 ///
12772 /// Strings are stored in the global heap (same as `create_vlen_string_dataset`),
12773 /// but the vlen references are stored in chunked layout with the given filter
12774 /// pipeline (e.g., deflate, zstd). `chunk_size` is the number of strings per chunk.
12775 pub fn create_vlen_string_dataset_compressed(
12776 &self,
12777 name: &str,
12778 strings: &[&str],
12779 chunk_size: usize,
12780 pipeline: FilterPipeline,
12781 ) -> IoResult<usize> {
12782 use crate::format::global_heap::encode_vlen_reference;
12783 use crate::format::messages::datatype::DatatypeMessage;
12784
12785 let create = self.begin_create(name)?;
12786 let name = create.name.as_str();
12787 let num_strings = strings.len() as u64;
12788 validate_chunk_geometry(&[num_strings], &[num_strings], &[chunk_size as u64])?;
12789
12790 // Store the strings as heap objects; the geometry validation above
12791 // must precede this so a refused call allocates nothing.
12792 let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
12793 let placements = self.insert_vlen_objects(&items)?;
12794
12795 // Build raw data: vlen references
12796 let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
12797 let data_size = (num_strings as usize) * ref_size;
12798 let mut raw_data = Vec::with_capacity(data_size);
12799 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
12800 let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
12801 raw_data.extend_from_slice(&encode_vlen_reference(
12802 seq_len,
12803 gcol_addr,
12804 obj_idx as u32,
12805 &self.ctx,
12806 ));
12807 }
12808
12809 // Set up chunked compressed layout
12810 let datatype = DatatypeMessage::vlen_string_utf8();
12811 let element_size = datatype.element_size_ctx(&self.ctx) as u64;
12812 let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12813 let dims: Vec<u64> = vec![num_strings];
12814 let max_dims: Vec<u64> = vec![num_strings];
12815 let chunk_bytes = chunk_size as u64 * element_size;
12816 let layout_version = self.chunk_layout_version(true, chunk_bytes);
12817 let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
12818
12819 let earray_params = EarrayParams::default_params();
12820 let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
12821 let nsblk_addrs = compute_nsblk_addrs(
12822 earray_params.idx_blk_elmts,
12823 earray_params.data_blk_min_elmts,
12824 earray_params.sup_blk_min_data_ptrs,
12825 earray_params.max_nelmts_bits,
12826 )?;
12827
12828 // Create filtered EA header
12829 let mut ea_header =
12830 ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
12831 ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
12832 ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
12833 ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
12834 ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
12835 ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
12836
12837 let hdr_encoded = ea_header.encode(&self.ctx);
12838 let ea_header_addr = self
12839 .allocator
12840 .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
12841
12842 // Create filtered index block
12843 let filt_iblk = FilteredIndexBlock::new(
12844 ea_header_addr,
12845 earray_params.idx_blk_elmts,
12846 ndblk_addrs,
12847 nsblk_addrs,
12848 );
12849 let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
12850 let ea_iblk_addr = self
12851 .allocator
12852 .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
12853
12854 ea_header.idx_blk_addr = ea_iblk_addr;
12855
12856 let hdr_encoded = ea_header.encode(&self.ctx);
12857 self.handle.write_at(ea_header_addr, &hdr_encoded)?;
12858 self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
12859
12860 let dataspace = DataspaceMessage {
12861 // Chunked storage always requires at least one dimension, so
12862 // this is never Scalar or Null.
12863 class: DataspaceClass::Simple,
12864 dims: dims.to_vec(),
12865 max_dims: Some(max_dims.to_vec()),
12866 };
12867
12868 let ea_iblk = ExtensibleArrayIndexBlock::new(
12869 ea_header_addr,
12870 earray_params.idx_blk_elmts,
12871 ndblk_addrs,
12872 nsblk_addrs,
12873 );
12874
12875 let idx = self.push_dataset(
12876 &create,
12877 DatasetInfo {
12878 name: name.to_string(),
12879 datatype,
12880 committed_type: None,
12881 external: None,
12882 virtual_storage: None,
12883 dataspace,
12884 read_format: None,
12885 obj_header_addr: 0,
12886 data_addr: UNDEF_ADDR,
12887 data_size: 0,
12888 compact: None,
12889 attributes: Vec::new(),
12890 obj_header_written_addr: None,
12891 obj_header_blocks: Vec::new(),
12892 filter_pipeline: Some(pipeline),
12893 deleted: false,
12894 extent_dirty: false,
12895 header_dirty: false,
12896 nlink_written: 1,
12897 creation_seq: self.take_creation_seq(),
12898 track_attr_order: self.track_order.attrs,
12899 fill_value: None,
12900 fill_time: FILL_TIME_IFSET,
12901 layout_version,
12902 times: self.created_object_times(),
12903 fixed_array: None,
12904 implicit: None,
12905 single_chunk: None,
12906 btree_v1: None,
12907 btree_v2: None,
12908 chunked: Some(ChunkedDatasetInfo {
12909 chunk_dims: chunk_dims.clone(),
12910 earray_params,
12911 ea_header_addr,
12912 ea_iblk_addr,
12913 ea_header,
12914 ea_iblk,
12915 chunks_written: 0,
12916 filt_iblk: Some(filt_iblk),
12917 chunk_size_len,
12918 }),
12919 append: None,
12920 },
12921 );
12922
12923 // Write chunks of vlen references with compression
12924 let chunk_byte_size = chunk_bytes as usize;
12925 let num_chunks = raw_data.len().div_ceil(chunk_byte_size);
12926 for chunk_i in 0..num_chunks {
12927 let start = chunk_i * chunk_byte_size;
12928 let end = (start + chunk_byte_size).min(raw_data.len());
12929 let chunk_data = if end - start < chunk_byte_size {
12930 // Pad last chunk to full size (vlen datasets carry no user
12931 // fill value, so this resolves to zero = null vlen reference).
12932 let mut padded = self.new_chunk_buffer(idx, chunk_byte_size);
12933 padded[..end - start].copy_from_slice(&raw_data[start..end]);
12934 padded
12935 } else {
12936 raw_data[start..end].to_vec()
12937 };
12938 self.write_chunk(idx, chunk_i as u64, &chunk_data)?;
12939 }
12940
12941 Ok(idx)
12942 }
12943
12944 /// Create an empty chunked vlen string dataset ready for incremental appends.
12945 ///
12946 /// The dataset starts with `dims = [0]` and `max_dims = [unlimited]`.
12947 /// Use `append_vlen_strings` to add data.
12948 pub fn create_appendable_vlen_string_dataset(
12949 &self,
12950 name: &str,
12951 chunk_size: usize,
12952 pipeline: Option<FilterPipeline>,
12953 ) -> IoResult<usize> {
12954 let datatype = DatatypeMessage::vlen_string_utf8();
12955 let chunk_dims: Vec<u64> = vec![chunk_size as u64];
12956 let dims: Vec<u64> = vec![0];
12957 let max_dims: Vec<u64> = vec![u64::MAX];
12958
12959 if let Some(ref pl) = pipeline {
12960 self.create_chunked_dataset_with_pipeline(
12961 name,
12962 datatype,
12963 &dims,
12964 &max_dims,
12965 &chunk_dims,
12966 pl.clone(),
12967 )
12968 } else {
12969 self.create_chunked_dataset(name, datatype, &dims, &max_dims, &chunk_dims)
12970 }
12971 }
12972
12973 /// Append variable-length strings to an existing chunked vlen string dataset.
12974 ///
12975 /// Creates a new global heap collection for the strings, builds vlen
12976 /// references, and appends them as new chunks to the dataset.
12977 pub fn append_vlen_strings(&self, ds_index: usize, strings: &[&str]) -> IoResult<()> {
12978 use crate::format::global_heap::encode_vlen_reference;
12979 use crate::format::messages::datatype::DatatypeMessage;
12980
12981 if strings.is_empty() {
12982 return Ok(());
12983 }
12984
12985 // Whole-operation guard: buffer take, frame writes, re-buffer and
12986 // extend below are separate slot acquisitions that a concurrent
12987 // same-dataset append must not interleave with.
12988 let cell = self.ds(ds_index);
12989 let _op = cell.op.lock();
12990
12991 // The elements about to be written are vlen references; any other
12992 // element type would be overwritten with them as raw bytes.
12993 let charset = {
12994 let ds = self.ds(ds_index);
12995 let m = ds.lock();
12996 match m.datatype {
12997 DatatypeMessage::VarLenString { charset, .. } => charset,
12998 _ => {
12999 return Err(crate::io::IoError::InvalidState(
13000 "append_vlen_strings is only for variable-length string datasets".into(),
13001 ))
13002 }
13003 }
13004 };
13005 ensure_vlen_charset(charset, strings)?;
13006
13007 // Every deterministic rejection must precede the heap write below:
13008 // a collection written for a batch the append then refuses (a
13009 // contiguous dataset, or a reopened dataset whose chunk index was
13010 // not reconstructed) is a 4096-byte orphan nothing references.
13011 let chunk_dims = self
13012 .dataset_chunk_dims(ds_index)
13013 .ok_or_else(|| crate::io::IoError::InvalidState("not a chunked dataset".into()))?
13014 .to_vec();
13015 let dims = self.dataset_dims(ds_index).to_vec();
13016
13017 // Store the batch's strings as heap objects; a batch that fits an
13018 // earlier collection's free space shares its block.
13019 let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
13020 let placements = self.insert_vlen_objects(&items)?;
13021
13022 // Build raw vlen reference bytes
13023 let ref_size = crate::format::global_heap::vlen_reference_size(&self.ctx);
13024 let mut raw = Vec::with_capacity(strings.len() * ref_size);
13025 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13026 let seq_len = crate::format::global_heap::vlen_seq_len(strings[i].len())?;
13027 raw.extend_from_slice(&encode_vlen_reference(
13028 seq_len,
13029 gcol_addr,
13030 obj_idx as u32,
13031 &self.ctx,
13032 ));
13033 }
13034
13035 let n_new_frames = strings.len();
13036 let current_dim0 = dims[0] as usize;
13037 let chunk_dim0 = chunk_dims[0] as usize;
13038 let frame_bytes = ref_size;
13039
13040 // Merge the buffer with the new frames when it is the dataset's tail;
13041 // a buffer left mid-extent (the extent moved past it) keeps its
13042 // recorded place — flush it and start fresh at the current end.
13043 let taken = { self.ds(ds_index).lock().append.take() };
13044 let (base_dim0, buffered_frames, mut combined) = match taken {
13045 Some(b) if b.base + b.frames == current_dim0 as u64 => {
13046 (b.base as usize, b.frames as usize, b.bytes)
13047 }
13048 Some(b) => {
13049 self.write_append_frames(ds_index, b.base, b.frames, &b.bytes)?;
13050 (current_dim0, 0, Vec::new())
13051 }
13052 None => (current_dim0, 0, Vec::new()),
13053 };
13054 combined.extend_from_slice(&raw);
13055
13056 let total_frames = buffered_frames + n_new_frames;
13057
13058 // Rows up to the last chunk boundary are written now; the tail that
13059 // does not complete a chunk goes back in the buffer for the next
13060 // append (or the flush at close). The boundary can precede
13061 // `base_dim0` — a reopened file's flushed partial chunk leaves the
13062 // base mid-chunk — in which case everything is tail.
13063 let last_boundary = ((base_dim0 + total_frames) / chunk_dim0) * chunk_dim0;
13064 let write_frames = last_boundary.saturating_sub(base_dim0);
13065 let tail_frames = total_frames - write_frames;
13066 if write_frames > 0 {
13067 self.write_append_frames(
13068 ds_index,
13069 base_dim0 as u64,
13070 write_frames as u64,
13071 &combined[..write_frames * frame_bytes],
13072 )?;
13073 }
13074 if tail_frames > 0 {
13075 let ds = self.ds(ds_index);
13076 let mut m = ds.lock();
13077 m.append = Some(AppendBuffer {
13078 base: (base_dim0 + write_frames) as u64,
13079 frames: tail_frames as u64,
13080 bytes: combined[write_frames * frame_bytes..].to_vec(),
13081 });
13082 }
13083
13084 // Extend dims
13085 let logical_dim0 = base_dim0 + total_frames;
13086 let mut new_dims = dims;
13087 new_dims[0] = logical_dim0 as u64;
13088 self.extend_dataset_inner(ds_index, &new_dims)?;
13089
13090 Ok(())
13091 }
13092
13093 /// Replace elements `start .. start + strings.len()` of a 1-D
13094 /// variable-length string dataset, leaving its extent and every other
13095 /// element alone.
13096 ///
13097 /// The replacements go into the global heap and only the vlen
13098 /// references of the named elements are rewritten, so the cost is the
13099 /// new strings plus the chunks those references live in — not the column.
13100 /// The objects the old references pointed at are freed *before* the
13101 /// replacement is allocated, so repeated updates reuse space instead of
13102 /// growing the file — including across close/reopen cycles, where the
13103 /// in-memory free list starts empty and only this free-first order lets
13104 /// the session reuse the block it just released. This is what libhdf5
13105 /// does: `H5T__vlen_disk_write` deletes the reference it read into the
13106 /// conversion background buffer before storing the new one.
13107 ///
13108 /// Elements the append buffer still holds are flushed to their chunks
13109 /// first, so the whole range is on disk and one write path covers it.
13110 pub fn write_vlen_strings_slice(
13111 &self,
13112 ds_index: usize,
13113 start: u64,
13114 strings: &[&str],
13115 ) -> IoResult<()> {
13116 use crate::format::global_heap::{encode_vlen_reference, vlen_reference_size};
13117 use crate::format::messages::datatype::DatatypeMessage;
13118
13119 // An empty batch is a no-op: nothing to replace, nothing to free.
13120 if strings.is_empty() {
13121 return Ok(());
13122 }
13123
13124 // Whole-operation guard: the flush, the old-reference reads and the
13125 // slice write below must not interleave with a concurrent
13126 // same-dataset operation.
13127 let cell = self.ds(ds_index);
13128 let _op = cell.op.lock();
13129
13130 // Snapshot what the write needs, then drop the guard: `write_slice`
13131 // below re-locks the same slot.
13132 let (charset, dims, writable) = {
13133 let ds = self.ds(ds_index);
13134 let m = ds.lock();
13135 let charset = match m.datatype {
13136 DatatypeMessage::VarLenString { charset, .. } => charset,
13137 _ => {
13138 return Err(crate::io::IoError::InvalidState(
13139 "write_vlen_strings_slice is only for variable-length string datasets"
13140 .into(),
13141 ))
13142 }
13143 };
13144 let writable = if m.is_chunked() {
13145 Ok(())
13146 } else {
13147 match m.contiguous_target() {
13148 Some(ContiguousTarget::Virtual) => Err(virtual_write_refused()),
13149 Some(_) => Ok(()),
13150 None => Err(crate::io::IoError::InvalidState(
13151 "dataset has no data allocated".into(),
13152 )),
13153 }
13154 };
13155 (charset, m.dataspace.dims.clone(), writable)
13156 };
13157
13158 // `write_slice_inner` rejects a dataset with neither chunk machinery
13159 // nor allocated data (a reopened dataset whose index was not
13160 // reconstructed), and refuses a virtual one outright — those
13161 // rejections must come before the heap write below, or every failed
13162 // call orphans a 4096-byte collection.
13163 writable?;
13164
13165 if dims.len() != 1 {
13166 return Err(crate::io::IoError::InvalidState(format!(
13167 "write_vlen_strings_slice is only for 1-dimension datasets, this one has {}",
13168 dims.len()
13169 )));
13170 }
13171 let end = start + strings.len() as u64;
13172 if end > dims[0] {
13173 return Err(crate::io::IoError::InvalidState(format!(
13174 "elements {start}..{end} are outside the dataset's {} elements",
13175 dims[0]
13176 )));
13177 }
13178 ensure_vlen_charset(charset, strings)?;
13179
13180 let ref_size = vlen_reference_size(&self.ctx);
13181
13182 // Elements the append buffer holds are not in the chunks yet: hand
13183 // them to the chunks first so the whole range is on disk and the one
13184 // write path below covers it.
13185 self.flush_append_buffer_if_intersecting(ds_index, start, end)?;
13186
13187 // The on-disk references about to be overwritten, read before anything
13188 // moves. libhdf5 reads the same bytes into the conversion background
13189 // buffer (`H5D__scatgath_write` gathers the file's current elements
13190 // when `need_bkg` is set) and hands them to `H5T__vlen_disk_write`,
13191 // which deletes them before storing the new reference.
13192 let superseded = self.current_element_bytes(ds_index, start, end - start, ref_size)?;
13193
13194 // Free the superseded objects *before* allocating the replacement,
13195 // the order `H5T__vlen_disk_write` uses. The freed block satisfies
13196 // the allocation below within this same session, so a reopen-and-
13197 // replace loop keeps the file flat — no persisted free-space
13198 // information exists to carry it across sessions (issue #10). The
13199 // cost, shared with libhdf5: a failure between here and the ref
13200 // write below leaves the dataset's old references dangling.
13201 self.release_vlen_references(&superseded)?;
13202
13203 // The insert comes after the release above so the space the release
13204 // recovered — a freed block, or in-collection bytes the release just
13205 // listed in `cwfs` — can satisfy this batch.
13206 let items: Vec<&[u8]> = strings.iter().map(|s| s.as_bytes()).collect();
13207 let placements = self.insert_vlen_objects(&items)?;
13208
13209 let mut refs = Vec::with_capacity(strings.len() * ref_size);
13210 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13211 refs.extend_from_slice(&encode_vlen_reference(
13212 crate::format::global_heap::vlen_seq_len(strings[i].len())?,
13213 gcol_addr,
13214 obj_idx as u32,
13215 &self.ctx,
13216 ));
13217 }
13218
13219 self.write_slice_inner(ds_index, &[start], &[strings.len() as u64], &refs)?;
13220
13221 Ok(())
13222 }
13223
13224 /// The bytes elements `start .. start + count` of a 1-D dataset currently
13225 /// hold, whichever layout stores them.
13226 ///
13227 /// Elements no write has reached yet read as zeros — for a vlen dataset
13228 /// that is the nil reference, which names no heap object.
13229 fn current_element_bytes(
13230 &self,
13231 ds_index: usize,
13232 start: u64,
13233 count: u64,
13234 element_size: usize,
13235 ) -> IoResult<Vec<u8>> {
13236 let mut out = vec![0u8; count as usize * element_size];
13237 if count == 0 {
13238 return Ok(out);
13239 }
13240
13241 let (is_chunked, data_addr) = {
13242 let ds = self.ds(ds_index);
13243 let m = ds.lock();
13244 (m.is_chunked(), m.data_addr)
13245 };
13246
13247 if !is_chunked {
13248 if data_addr != UNDEF_ADDR {
13249 // `read_at_most`, not `read_at`: a contiguous dataset's block is
13250 // reserved when it is created, so the file can still be shorter
13251 // than the block until something writes it. What is missing has
13252 // never been written, which is the zeros above.
13253 let at = data_addr + start * element_size as u64;
13254 let got = self.handle.read_at_most(at, out.len())?;
13255 out[..got.len()].copy_from_slice(&got);
13256 }
13257 return Ok(out);
13258 }
13259
13260 let geo = self.chunk_geometry(ds_index)?;
13261 let per_chunk = geo.chunk_dims[0];
13262 // Only a corrupt or crafted file declares a zero-length chunk
13263 // dimension; the divisions below must reject it the way
13264 // `write_slice` does, not panic.
13265 if per_chunk == 0 {
13266 return Err(crate::io::IoError::InvalidState(
13267 "chunk shape has a zero-length dimension".into(),
13268 ));
13269 }
13270 let end = start + count;
13271 for c in (start / per_chunk)..=((end - 1) / per_chunk) {
13272 let origin = c * per_chunk;
13273 let lo = start.max(origin);
13274 let hi = end.min(origin + per_chunk);
13275 // A chunk with no block yet leaves this span as the zeros above.
13276 let Some(chunk) = self.read_chunk_at_coords(ds_index, &[c])? else {
13277 continue;
13278 };
13279 let src = ((lo - origin) as usize) * element_size;
13280 let dst = ((lo - start) as usize) * element_size;
13281 let len = ((hi - lo) as usize) * element_size;
13282 if src + len > chunk.len() {
13283 return Err(crate::io::IoError::InvalidState(format!(
13284 "chunk {c} is {} bytes, too short for elements {lo}..{hi}",
13285 chunk.len()
13286 )));
13287 }
13288 out[dst..dst + len].copy_from_slice(&chunk[src..src + len]);
13289 }
13290 Ok(out)
13291 }
13292
13293 /// Free the global heap objects `refs` names, so replacing a vlen element
13294 /// does not strand what it used to point at.
13295 ///
13296 /// Callers pass refs only for *top-level* vlen datatypes (the
13297 /// `collect_refs` / `is_vlen` decisions at the prune, delete and
13298 /// attribute-release sites all match `VarLenString`/`VarLenSequence`).
13299 /// A compound datatype with vlen members — writable only by a foreign
13300 /// library, never by this crate — keeps its members' heap objects when
13301 /// its storage is pruned, deleted or replaced.
13302 ///
13303 /// This is libhdf5's `H5HG_remove` reached through `H5T__vlen_disk_delete`:
13304 /// the object leaves its collection, the collection is rewritten at its
13305 /// existing size with the recovered bytes given to the free-space marker,
13306 /// and a collection that ends up empty returns its block to the allocator.
13307 /// A rewritten collection's recovered space is listed in `cwfs` for
13308 /// [`insert_vlen_objects`](Self::insert_vlen_objects) to pack into; a
13309 /// freed block leaves the list.
13310 /// A nil reference (address 0 or `UNDEF_ADDR`) names no object. The
13311 /// address decides, not the sequence length: this crate's writers store
13312 /// even the empty string as a real heap object, so a zero-length reference
13313 /// with a defined address still holds one that must be released. libhdf5
13314 /// diverges here against itself — `H5T__vlen_disk_delete` returns before
13315 /// `H5HG_remove` when the sequence length is zero, yet its write path
13316 /// (`H5VL__native_blob_put`) inserts a heap object even for an empty
13317 /// sequence, stranding it forever. The address rule frees those objects.
13318 ///
13319 /// Heap objects carry no reference count on this path, matching libhdf5:
13320 /// its vlen code never calls `H5HG_link` (only the virtual-dataset layer
13321 /// does). Releasing the same reference twice is absorbed by the
13322 /// missing-index check below, but a crafted file in which two elements
13323 /// share one heap object would lose it for the survivor when either is
13324 /// replaced — the same exposure the file has under libhdf5. This crate's
13325 /// writers never share: each element write inserts its own object.
13326 ///
13327 /// Under SWMR nothing is freed and no collection is rewritten: a reader may
13328 /// be following those references, the same reason `place_chunk` keeps a
13329 /// relocated chunk's old block.
13330 fn release_vlen_references(&self, refs: &[u8]) -> IoResult<()> {
13331 use crate::format::global_heap::{decode_vlen_reference, vlen_reference_size};
13332
13333 let ref_size = vlen_reference_size(&self.ctx);
13334 if ref_size == 0 || refs.len() < ref_size {
13335 return Ok(());
13336 }
13337
13338 // Group by collection so one holding several replaced objects is read,
13339 // rewritten and judged empty exactly once.
13340 let mut per_collection: std::collections::BTreeMap<u64, Vec<u16>> = Default::default();
13341 for r in refs.chunks_exact(ref_size) {
13342 let (_seq_len, addr, obj_idx) = decode_vlen_reference(r, &self.ctx)?;
13343 if addr == 0 || addr == UNDEF_ADDR {
13344 continue;
13345 }
13346 let Ok(idx) = u16::try_from(obj_idx) else {
13347 return Err(crate::io::IoError::InvalidState(format!(
13348 "global heap object index {obj_idx} does not fit the 16-bit on-disk field"
13349 )));
13350 };
13351 per_collection.entry(addr).or_default().push(idx);
13352 }
13353 self.remove_heap_objects(per_collection)
13354 }
13355
13356 /// Remove global heap objects — `H5HG_remove` — given the object indices
13357 /// grouped by the collection they live in.
13358 ///
13359 /// The single owner of heap-object removal: the vlen release path above
13360 /// reaches it with the objects a replaced element used to name, and
13361 /// [`release_dataset_storage`](Self::release_dataset_storage) with the
13362 /// one mapping-list object a deleted virtual dataset owned, which is what
13363 /// `H5D__virtual_delete` frees the same way.
13364 fn remove_heap_objects(
13365 &self,
13366 per_collection: std::collections::BTreeMap<u64, Vec<u16>>,
13367 ) -> IoResult<()> {
13368 use crate::format::global_heap::GlobalHeapCollection;
13369
13370 if self.swmr_active {
13371 return Ok(());
13372 }
13373
13374 // An object on its way out can hold no stamp: a reference this
13375 // session wrote into it would otherwise be stamped into whatever a
13376 // later insert puts at the same index. Pruned here, by the one owner
13377 // of removal, so no release path — attribute replacement, element
13378 // rewrite, dataset deletion — can leave one behind.
13379 self.pending_heap_references.lock().retain(|p| {
13380 !per_collection
13381 .get(&p.collection)
13382 .is_some_and(|indices| indices.contains(&p.index))
13383 });
13384
13385 // The `cwfs` lock is held across the sweep: it serializes these
13386 // collection-block rewrites (and frees) against
13387 // `insert_vlen_objects`, which may be packing new objects into the
13388 // same blocks.
13389 let objhdr = GlobalHeapCollection::object_disk_size(&self.ctx, 0);
13390 let mut cwfs = self.cwfs.lock();
13391 for (addr, indices) in per_collection {
13392 // A collection is at least 4096 bytes (H5HG_MINALLOC) and most are
13393 // exactly that, so one read usually covers the whole image; only
13394 // an oversized collection needs a second read at its declared size.
13395 let mut image = self.handle.read_at_most(addr, 4096)?;
13396 let declared = GlobalHeapCollection::decode_size(&image, &self.ctx)?;
13397 if declared > image.len() {
13398 image = self.handle.read_at(addr, declared)?;
13399 }
13400 let (mut gcol, _) = GlobalHeapCollection::decode(&image[..declared], &self.ctx)?;
13401 let mut removed_any = false;
13402 for idx in indices {
13403 removed_any |= gcol.remove_object(idx);
13404 }
13405 // Every index already gone (a stale or duplicate reference):
13406 // leave the image alone. Rewriting is not just wasted I/O — a
13407 // 100%-full collection written by libhdf5 has no free-space
13408 // marker, so re-encoding it at its declared size cannot fit one
13409 // and the whole element update would fail.
13410 if !removed_any {
13411 continue;
13412 }
13413 if gcol.is_empty() {
13414 self.allocator
13415 .free(addr, declared as u64, FreeSpaceClass::RawData);
13416 // The block is gone; a lingering entry would let an insert
13417 // pack into space the allocator can hand to anything.
13418 cwfs.retain(|e| e.addr != addr);
13419 } else {
13420 let rewritten = gcol.encode_at_size(&self.ctx, declared)?;
13421 self.handle.write_at(addr, &rewritten)?;
13422 // The recovered bytes are packable now — list them, the way
13423 // libhdf5's `H5HG_remove` adds the heap to `cwfs`.
13424 if let Some(free) = gcol.free_space_at(&self.ctx, declared) {
13425 if free >= 2 * objhdr {
13426 cwfs_note(&mut cwfs, addr, declared, free);
13427 }
13428 }
13429 }
13430 }
13431 Ok(())
13432 }
13433
13434 /// Add an attribute to a dataset.
13435 ///
13436 /// The attribute will be written as a message in the dataset's object
13437 /// header when the file is finalized.
13438 pub fn add_dataset_attribute(&self, ds_index: usize, attr: AttributeMessage) -> IoResult<()> {
13439 self.set_attribute(AttrTarget::Dataset(ds_index), attr)
13440 }
13441
13442 /// Build a variable-length UTF-8 string attribute message.
13443 ///
13444 /// The string is stored as one object in a global heap collection and the
13445 /// returned [`AttributeMessage`] carries the vlen reference as its data,
13446 /// with a vlen-string datatype and scalar dataspace. h5py reads the value
13447 /// back as a Python `str` (not `bytes`).
13448 ///
13449 /// This is the single owner of vlen-string-attribute construction: every
13450 /// public string-attribute setter (dataset, group, root, and the SWMR
13451 /// equivalents) routes through it, so a `VarLenUnicode` /
13452 /// `set_attr_string` value is always stored as a true variable-length
13453 /// string rather than the fixed-length string it used to be.
13454 ///
13455 /// The string's heap object is placed by
13456 /// [`insert_vlen_objects`](Self::insert_vlen_objects), so consecutive
13457 /// attributes pack into a shared collection instead of each paying the
13458 /// 4096-byte `H5HG_MINALLOC` minimum for a block that holds one string.
13459 fn vlen_string_attribute(&self, name: &str, value: &str) -> IoResult<AttributeMessage> {
13460 use crate::format::global_heap::encode_vlen_reference;
13461 use crate::format::messages::dataspace::DataspaceMessage;
13462 use crate::format::messages::datatype::DatatypeMessage;
13463
13464 let (gcol_addr, obj_idx) = self.insert_vlen_objects(&[value.as_bytes()])?[0];
13465 let seq_len = crate::format::global_heap::vlen_seq_len(value.len())?;
13466 let data = encode_vlen_reference(seq_len, gcol_addr, obj_idx as u32, &self.ctx);
13467 Ok(AttributeMessage {
13468 name: name.to_string(),
13469 datatype: DatatypeMessage::vlen_string_utf8(),
13470 dataspace: DataspaceMessage::scalar(),
13471 data,
13472 })
13473 }
13474
13475 /// Build a variable-length UTF-8 string **array** attribute message.
13476 ///
13477 /// The N-dimensional counterpart of
13478 /// [`vlen_string_attribute`](Self::vlen_string_attribute): every element
13479 /// string is stored as one object in a single global heap collection, and
13480 /// the attribute data is the row-major concatenation of one vlen reference
13481 /// per element. The datatype is the same vlen-string datatype; the dataspace
13482 /// is the simple dataspace described by `shape` (an empty `shape` is a
13483 /// scalar). h5py reads the value back as a numpy array of Python `str` with
13484 /// that shape.
13485 ///
13486 /// The caller owns the invariant that `values.len()` equals the product of
13487 /// `shape` (the public setters validate it before calling). The element
13488 /// objects are placed by
13489 /// [`insert_vlen_objects`](Self::insert_vlen_objects) — a zero-element
13490 /// array allocates nothing, and each reference carries its element's
13491 /// own collection address.
13492 fn vlen_string_array_attribute(
13493 &self,
13494 name: &str,
13495 values: &[&str],
13496 shape: &[u64],
13497 ) -> IoResult<AttributeMessage> {
13498 use crate::format::global_heap::encode_vlen_reference;
13499 use crate::format::messages::dataspace::DataspaceMessage;
13500 use crate::format::messages::datatype::DatatypeMessage;
13501
13502 debug_assert_eq!(
13503 values.len() as u64,
13504 shape.iter().product::<u64>(),
13505 "vlen_string_array_attribute values.len() must equal product(shape)"
13506 );
13507
13508 let items: Vec<&[u8]> = values.iter().map(|v| v.as_bytes()).collect();
13509 let placements = self.insert_vlen_objects(&items)?;
13510
13511 let mut data = Vec::with_capacity(values.len() * 16);
13512 for (i, &(gcol_addr, obj_idx)) in placements.iter().enumerate() {
13513 data.extend_from_slice(&encode_vlen_reference(
13514 crate::format::global_heap::vlen_seq_len(values[i].len())?,
13515 gcol_addr,
13516 obj_idx as u32,
13517 &self.ctx,
13518 ));
13519 }
13520 Ok(AttributeMessage {
13521 name: name.to_string(),
13522 datatype: DatatypeMessage::vlen_string_utf8(),
13523 dataspace: DataspaceMessage::simple(shape),
13524 data,
13525 })
13526 }
13527
13528 /// Set a user-defined fill value for a dataset.
13529 ///
13530 /// `bytes` must be exactly one element wide (matching the dataset's
13531 /// datatype). The value is emitted as a `fill_defined = 2` fill-value
13532 /// message in the dataset object header when the file is finalized.
13533 ///
13534 /// IMPORTANT: for a *contiguous* dataset this also immediately writes
13535 /// the tiled fill value across the whole data block, so it must be
13536 /// called BEFORE any `write_dataset_raw` / `write_slice` — otherwise the
13537 /// fill write clobbers data already written. (The high-level builder
13538 /// always calls this right after creating the dataset.)
13539 pub fn set_dataset_fill_value(&self, ds_index: usize, bytes: Vec<u8>) -> IoResult<()> {
13540 let count = self.dataset_count();
13541 if ds_index >= count {
13542 return Err(crate::io::IoError::InvalidState(format!(
13543 "dataset index {} out of range",
13544 ds_index
13545 )));
13546 }
13547 let ds_ref = self.ds(ds_index);
13548 let mut ds = ds_ref.lock();
13549 let es = ds.datatype.element_size() as usize;
13550 if bytes.len() != es {
13551 return Err(crate::io::IoError::InvalidState(format!(
13552 "fill value is {} bytes but dataset element size is {}",
13553 bytes.len(),
13554 es
13555 )));
13556 }
13557 // For a dataset with no per-chunk fill path the fill-value message
13558 // only declares fill-on-allocation — tile the fill value across the
13559 // storage itself now, so unwritten elements read back as the fill
13560 // value. Which storage that is depends on the layout: a compact
13561 // dataset's is the image inside its layout message, a contiguous
13562 // one's is its data block. (The high-level builder calls this
13563 // immediately after create, before any data is written; a subsequent
13564 // write_raw/write_slice overwrites its region.)
13565 // An implicitly indexed dataset is filled here too, and for the same
13566 // reason: that index has no per-chunk fill path because it has no
13567 // per-chunk anything — its whole chunk grid is one run of space,
13568 // allocated and filled at create like a contiguous block. So the test
13569 // is not "is it chunked" but "does something else fill its chunks".
13570 let fills_per_chunk = ds
13571 .chunk_index_kind()
13572 .is_some_and(|k| k != ChunkIndexKind::Implicit);
13573 // `H5D_FILL_TIME_NEVER` means exactly this: the library never writes
13574 // the fill value into allocated storage. Call `set_dataset_fill_time`
13575 // before this method to have it observed here — the storage this
13576 // would otherwise tile keeps whatever zero bytes its allocation
13577 // already gave it.
13578 if !fills_per_chunk && ds.fill_time != FILL_TIME_NEVER {
13579 if let Some(len) = ds.compact.as_ref().map(Vec::len) {
13580 ds.compact = Some(crate::format::messages::fill_value::tiled_fill(
13581 len,
13582 Some(&bytes),
13583 ));
13584 } else {
13585 // An implicit index's chunk grid is filled as one run, the
13586 // same way a contiguous block is, and storage this file did
13587 // not allocate is not filled at all; `allocated_storage_run`
13588 // is where both of those are decided.
13589 let run = ds.allocated_storage_run();
13590 if let Some((target, data_size)) = run.filter(|&(_, size)| size > 0) {
13591 let filled = crate::format::messages::fill_value::tiled_fill(
13592 data_size as usize,
13593 Some(&bytes),
13594 );
13595 self.write_contiguous_bytes(&target, 0, &filled)?;
13596 }
13597 }
13598 }
13599
13600 ds.fill_value = Some(bytes);
13601 ds.header_dirty = true;
13602 Ok(())
13603 }
13604
13605 /// Set when the fill value is written into allocated storage —
13606 /// `H5Pset_fill_time`. `time` is one of [`FILL_TIME_ALLOC`],
13607 /// [`FILL_TIME_NEVER`], [`FILL_TIME_IFSET`]; anything else is rejected
13608 /// the way `H5Pset_fill_time` rejects an out-of-range `H5D_fill_time_t`.
13609 ///
13610 /// Call this before [`set_dataset_fill_value`](Self::set_dataset_fill_value)
13611 /// so that a `FILL_TIME_NEVER` policy is in place before that call
13612 /// decides whether to eager-tile the value into storage. (The
13613 /// high-level builder always calls it first.)
13614 pub fn set_dataset_fill_time(&self, ds_index: usize, time: u8) -> IoResult<()> {
13615 if !matches!(time, FILL_TIME_ALLOC | FILL_TIME_NEVER | FILL_TIME_IFSET) {
13616 return Err(crate::io::IoError::InvalidState(format!(
13617 "invalid fill time {time}; must be {FILL_TIME_ALLOC} (alloc), \
13618 {FILL_TIME_NEVER} (never) or {FILL_TIME_IFSET} (if-set)"
13619 )));
13620 }
13621 let count = self.dataset_count();
13622 if ds_index >= count {
13623 return Err(crate::io::IoError::InvalidState(format!(
13624 "dataset index {} out of range",
13625 ds_index
13626 )));
13627 }
13628 let ds_ref = self.ds(ds_index);
13629 let mut ds = ds_ref.lock();
13630 ds.fill_time = time;
13631 ds.header_dirty = true;
13632 Ok(())
13633 }
13634
13635 /// Allocate a `chunk_bytes`-sized buffer pre-filled with dataset
13636 /// `ds_index`'s fill value (tiled one element wide), or zeros when no
13637 /// user-defined fill value exists.
13638 ///
13639 /// Every partial chunk the writer emits must be built on top of a
13640 /// buffer from this method, so that the unwritten element region of an
13641 /// allocated chunk reads back as the fill value rather than zero.
13642 ///
13643 /// Unconditional: a shrink's straddler refill
13644 /// (`refill_chunk_beyond_extent`) calls this to repair data about to
13645 /// become reachable again, which libhdf5's `H5D__chunk_prune_fill` does
13646 /// regardless of the fill-time policy. [`new_write_chunk_buffer`](Self::new_write_chunk_buffer)
13647 /// is the gated counterpart for a chunk touched for the first time
13648 /// during a write, where the policy does apply.
13649 pub(crate) fn new_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
13650 let ds = self.ds(ds_index);
13651 let m = ds.lock();
13652 let fv = m.fill_value.as_deref();
13653 crate::format::messages::fill_value::tiled_fill(chunk_bytes, fv)
13654 }
13655
13656 /// The buffer a chunk gets the first time a write touches it — this
13657 /// dataset's allocation-time fill gate. `H5D__chunk_lock`'s cache-miss
13658 /// path (H5Dchunk.c:4894) fills such a buffer only for `ALLOC`, or for
13659 /// `IFSET` with a fill value defined; `NEVER` leaves it as the zeros a
13660 /// fresh buffer already has. Everything else about the buffer is
13661 /// [`new_chunk_buffer`](Self::new_chunk_buffer)'s.
13662 fn new_write_chunk_buffer(&self, ds_index: usize, chunk_bytes: usize) -> Vec<u8> {
13663 let never = {
13664 let ds = self.ds(ds_index);
13665 let m = ds.lock();
13666 m.fill_time == FILL_TIME_NEVER
13667 };
13668 if never {
13669 vec![0u8; chunk_bytes]
13670 } else {
13671 self.new_chunk_buffer(ds_index, chunk_bytes)
13672 }
13673 }
13674
13675 /// Write `n_frames` whole frames whose first row is `base_frame`, for
13676 /// whichever chunk index the dataset uses and whatever its chunk shape.
13677 ///
13678 /// The single owner of an append's chunk writes. The frames are one
13679 /// hyperslab — rows `base_frame .. base_frame + n_frames` over the full
13680 /// row shape — so the write goes through
13681 /// [`write_slice_chunked`](Self::write_slice_chunked), the same engine
13682 /// `write_slice` uses: a chunk the span covers completely is written
13683 /// straight through, a partial one is read-modify-write on top of what
13684 /// is stored (or the fill value), and a chunk row narrower or wider
13685 /// than the frame row is scattered at the chunk stride. The previous
13686 /// owner required the extensible-array index and packed rows at the
13687 /// frame stride, so appends to a fixed-array or v2 B-tree dataset
13688 /// failed at close and lost the buffered rows.
13689 ///
13690 /// The caller holds the dataset's op lock or the writer exclusively.
13691 pub(crate) fn write_append_frames(
13692 &self,
13693 ds_index: usize,
13694 base_frame: u64,
13695 n_frames: u64,
13696 frames: &[u8],
13697 ) -> IoResult<()> {
13698 if n_frames == 0 {
13699 return Ok(());
13700 }
13701 let geo = self.chunk_geometry(ds_index)?;
13702 let mut starts = vec![0u64; geo.dims.len()];
13703 starts[0] = base_frame;
13704 let mut counts = geo.dims.clone();
13705 counts[0] = n_frames;
13706 let expected = counts.iter().product::<u64>() * geo.element_size;
13707 if frames.len() as u64 != expected {
13708 return Err(crate::io::IoError::InvalidState(format!(
13709 "{n_frames} frames at rows {base_frame}.. need {expected} bytes, got {}",
13710 frames.len()
13711 )));
13712 }
13713 self.write_slice_chunked(ds_index, &starts, &counts, frames)
13714 }
13715
13716 /// Write the dataset's append buffer (if any) into its chunks and clear
13717 /// it. The single owner of the buffer-to-chunks transition: the flush at
13718 /// close, an append meeting a non-contiguous buffer, and any operation
13719 /// about to write rows the buffer holds all come through here.
13720 ///
13721 /// The caller holds the dataset's op lock or the writer exclusively —
13722 /// the take and the frame writes are separate acquisitions.
13723 pub(crate) fn flush_append_buffer(&self, ds_index: usize) -> IoResult<()> {
13724 let taken = { self.ds(ds_index).lock().append.take() };
13725 match taken {
13726 Some(b) => self.write_append_frames(ds_index, b.base, b.frames, &b.bytes),
13727 None => Ok(()),
13728 }
13729 }
13730
13731 /// Flush the append buffer when rows `start_row .. end_row` intersect
13732 /// the buffered range — those rows' current content is the buffer, and
13733 /// writing them on disk while the buffer still holds them would be
13734 /// undone by the flush at close.
13735 ///
13736 /// The caller holds the dataset's op lock or the writer exclusively.
13737 pub(crate) fn flush_append_buffer_if_intersecting(
13738 &self,
13739 ds_index: usize,
13740 start_row: u64,
13741 end_row: u64,
13742 ) -> IoResult<()> {
13743 let intersects = {
13744 let ds = self.ds(ds_index);
13745 let m = ds.lock();
13746 m.append
13747 .as_ref()
13748 .is_some_and(|b| start_row < b.base + b.frames && end_row > b.base)
13749 };
13750 if intersects {
13751 self.flush_append_buffer(ds_index)
13752 } else {
13753 Ok(())
13754 }
13755 }
13756
13757 /// Read an already-written chunk's *decompressed* bytes when the chunk
13758 /// is allocated and resolvable from the in-memory extensible-array
13759 /// index. Handles index-block and data-block chunks, filtered and
13760 /// unfiltered.
13761 ///
13762 /// Returns `Ok(None)` only when the chunk has never been written
13763 /// (address `UNDEF`) or the index genuinely does not reach it, which for
13764 /// a read-modify-write means the chunk's content is the fill value.
13765 pub(crate) fn read_chunk_if_present(
13766 &self,
13767 ds_index: usize,
13768 chunk_idx: u64,
13769 ) -> IoResult<Option<Vec<u8>>> {
13770 // Phase 1: resolve the chunk's location from the in-memory index.
13771 // Hold the slot guard through Phase 1: `chunked` borrows it, while the
13772 // `self.handle`/`self.ctx` reads below touch disjoint fields.
13773 let ds = self.ds(ds_index);
13774 let m = ds.lock();
13775 let element_size = m.datatype.element_size() as u64;
13776 let pipeline = m.filter_pipeline.clone();
13777 let Some(chunked) = m.chunked.as_ref() else {
13778 return Ok(None);
13779 };
13780 let chunk_bytes = chunked.chunk_dims.iter().product::<u64>() * element_size;
13781 let max_nelmts_bits = chunked.earray_params.max_nelmts_bits;
13782 let chunk_size_len = chunked.chunk_size_len;
13783 let is_filtered = chunked.filt_iblk.is_some();
13784
13785 // The chunk entry is either read straight from an index block, or
13786 // located via a data block that must itself be read from disk.
13787 enum Loc {
13788 Direct(u64, u64, u32),
13789 DataBlock {
13790 dblk_addr: u64,
13791 offset: usize,
13792 nelmts: usize,
13793 },
13794 }
13795
13796 // Resolve the chunk's location with the libhdf5-compatible EA
13797 // geometry (super-block-grouped data blocks), matching `record_ea_chunk`.
13798 let ea_loc = {
13799 let p = &chunked.earray_params;
13800 EaGeometry::new(
13801 p.idx_blk_elmts,
13802 p.data_blk_min_elmts,
13803 p.sup_blk_min_data_ptrs,
13804 p.max_nelmts_bits,
13805 p.max_dblk_page_nelmts_bits,
13806 )?
13807 .locate(chunk_idx)?
13808 };
13809 let loc = match ea_loc {
13810 EaLoc::Index { elem } => {
13811 if is_filtered {
13812 let e = &chunked.filt_iblk.as_ref().unwrap().elements[elem];
13813 Loc::Direct(e.addr, e.nbytes, e.filter_mask)
13814 } else {
13815 Loc::Direct(chunked.ea_iblk.elements[elem], chunk_bytes, 0)
13816 }
13817 }
13818 EaLoc::Dblk(l) => {
13819 if l.paged {
13820 return Err(crate::io::IoError::InvalidState(format!(
13821 "chunk index {} lives in a paged extensible-array data \
13822 block, which is not yet supported for read-modify-write",
13823 chunk_idx
13824 )));
13825 }
13826 let dblk_addr = match l.path {
13827 EaDblkPath::Direct { idx } => {
13828 if is_filtered {
13829 chunked.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
13830 } else {
13831 chunked.ea_iblk.dblk_addrs[idx]
13832 }
13833 }
13834 EaDblkPath::ViaSblk {
13835 sblk_off,
13836 local_dblk,
13837 ndblks_in_sblk,
13838 ..
13839 } => {
13840 let sblk_addr = if is_filtered {
13841 chunked.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
13842 } else {
13843 chunked.ea_iblk.sblk_addrs[sblk_off]
13844 };
13845 if sblk_addr == UNDEF_ADDR {
13846 return Ok(None);
13847 }
13848 let sb_buf = self.handle.read_at_most(sblk_addr, 65536)?;
13849 let sb = ExtensibleArraySuperBlock::decode(
13850 &sb_buf,
13851 &self.ctx,
13852 max_nelmts_bits,
13853 ndblks_in_sblk,
13854 0,
13855 )?;
13856 sb.dblk_addrs[local_dblk]
13857 }
13858 };
13859 if dblk_addr == UNDEF_ADDR {
13860 return Ok(None);
13861 }
13862 Loc::DataBlock {
13863 dblk_addr,
13864 offset: l.offset_in_dblk as usize,
13865 nelmts: l.dblk_nelmts as usize,
13866 }
13867 }
13868 };
13869
13870 // Phase 2: resolve through the data block (if needed) and read. The
13871 // mask is the chunk's filter mask (0 for unfiltered), so a chunk
13872 // written via a direct chunk write with a skipped filter is reversed
13873 // correctly during read-modify-write.
13874 let (addr, nbytes, mask) = match loc {
13875 Loc::Direct(a, n, m) => (a, n, m),
13876 Loc::DataBlock {
13877 dblk_addr,
13878 offset,
13879 nelmts,
13880 } => {
13881 let buf = self.handle.read_at_most(dblk_addr, 65536)?;
13882 if is_filtered {
13883 let dblk = FilteredDataBlock::decode(
13884 &buf,
13885 &self.ctx,
13886 max_nelmts_bits,
13887 nelmts,
13888 chunk_size_len,
13889 )?;
13890 let e = &dblk.elements[offset];
13891 (e.addr, e.nbytes, e.filter_mask)
13892 } else {
13893 let dblk =
13894 ExtensibleArrayDataBlock::decode(&buf, &self.ctx, max_nelmts_bits, nelmts)?;
13895 (dblk.elements[offset], chunk_bytes, 0)
13896 }
13897 }
13898 };
13899 self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13900 }
13901
13902 /// Read one stored chunk block and undo its filters.
13903 ///
13904 /// `nbytes` is the *stored* length and `mask` the chunk's filter mask, so
13905 /// a chunk written by a direct chunk write with a skipped filter is
13906 /// reversed correctly. `Ok(None)` means the chunk has no block yet — the
13907 /// single place that judgement is made, shared by every chunk index.
13908 fn read_chunk_block(
13909 &self,
13910 pipeline: Option<&FilterPipeline>,
13911 addr: u64,
13912 nbytes: u64,
13913 mask: u32,
13914 ) -> IoResult<Option<Vec<u8>>> {
13915 if addr == UNDEF_ADDR || nbytes == 0 {
13916 return Ok(None);
13917 }
13918 let raw = self.handle.read_at(addr, nbytes as usize)?;
13919 match pipeline {
13920 Some(pl) => Ok(Some(filter::reverse_filters_masked(pl, &raw, mask)?)),
13921 None => Ok(Some(raw)),
13922 }
13923 }
13924
13925 /// Read the *decompressed* bytes of the chunk at `chunk_coords`, whichever
13926 /// chunk index the dataset uses, or `Ok(None)` when that chunk has never
13927 /// been written.
13928 ///
13929 /// This is the read half of a partial-chunk read-modify-write: a hyperslab
13930 /// write that covers only part of a chunk must start from what is already
13931 /// there. Keeping one entry point for all three index types is what lets
13932 /// [`write_slice`](Self::write_slice) stay index-agnostic.
13933 pub(crate) fn read_chunk_at_coords(
13934 &self,
13935 ds_index: usize,
13936 chunk_coords: &[u64],
13937 ) -> IoResult<Option<Vec<u8>>> {
13938 let geo = self.chunk_geometry(ds_index)?;
13939 // Only the linearly-addressed indexes compute a slot; a v2 B-tree is
13940 // keyed by the coordinates themselves (and may hold unlimited inner
13941 // dimensions, which have no linear slot).
13942 match geo.kind {
13943 ChunkIndexKind::ExtensibleArray => {
13944 let linear = geo.linear_index(chunk_coords)?;
13945 self.read_chunk_if_present(ds_index, linear)
13946 }
13947 ChunkIndexKind::FixedArray => {
13948 let linear = geo.linear_index(chunk_coords)?;
13949 let ds = self.ds(ds_index);
13950 let m = ds.lock();
13951 let pipeline = m.filter_pipeline.clone();
13952 let fa = m.fixed_array.as_ref().unwrap();
13953 let lidx = linear as usize;
13954 let (addr, nbytes, mask) = if pipeline.is_some() {
13955 match fa.fa_dblk.filtered_elements.get(lidx) {
13956 Some(e) => (e.address, e.chunk_size, e.filter_mask),
13957 None => return Ok(None),
13958 }
13959 } else {
13960 match fa.fa_dblk.elements.get(lidx) {
13961 Some(&a) => (a, geo.chunk_bytes(), 0),
13962 None => return Ok(None),
13963 }
13964 };
13965 drop(m);
13966 self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13967 }
13968 ChunkIndexKind::BtreeV2 => {
13969 let ds = self.ds(ds_index);
13970 let m = ds.lock();
13971 let pipeline = m.filter_pipeline.clone();
13972 let bt2 = m.btree_v2.as_ref().unwrap();
13973 // A filtered index records the stored size and mask per chunk;
13974 // an unfiltered one stores whole chunks, so their size is the
13975 // chunk shape and no filter ran.
13976 let found = if bt2.index.filtered {
13977 bt2.index
13978 .lookup_filtered(chunk_coords)
13979 .map(|r| (r.chunk_address, r.chunk_size, r.filter_mask))
13980 } else {
13981 bt2.index
13982 .lookup(chunk_coords)
13983 .map(|r| (r.chunk_address, geo.chunk_bytes(), 0))
13984 };
13985 drop(m);
13986 match found {
13987 Some((addr, nbytes, mask)) => {
13988 self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
13989 }
13990 None => Ok(None),
13991 }
13992 }
13993 // Every chunk of an implicitly indexed dataset exists from the
13994 // moment the dataset does, so there is no "never written" answer
13995 // to give: an untouched chunk reads back as the fill value the
13996 // create wrote there.
13997 ChunkIndexKind::Implicit => {
13998 let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
13999 self.read_chunk_block(None, grid + offset, geo.chunk_bytes(), 0)
14000 }
14001 // A single-chunk dataset's one chunk is never written until its
14002 // first write (unless the dataset was early-allocated and
14003 // unfiltered, in which case create already gave it an address) —
14004 // unlike Implicit, `UNDEF_ADDR` here is a real "never written".
14005 ChunkIndexKind::SingleChunk => {
14006 let ds = self.ds(ds_index);
14007 let m = ds.lock();
14008 let pipeline = m.filter_pipeline.clone();
14009 let sc = m.single_chunk.as_ref().unwrap();
14010 if sc.data_addr == UNDEF_ADDR {
14011 return Ok(None);
14012 }
14013 let (addr, nbytes, mask) = if pipeline.is_some() {
14014 (sc.data_addr, sc.nbytes, sc.filter_mask)
14015 } else {
14016 (sc.data_addr, geo.chunk_bytes(), 0)
14017 };
14018 drop(m);
14019 self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14020 }
14021 ChunkIndexKind::BtreeV1 => {
14022 let ds = self.ds(ds_index);
14023 let m = ds.lock();
14024 let pipeline = m.filter_pipeline.clone();
14025 let bt1 = m.btree_v1.as_ref().unwrap();
14026 let found = bt1
14027 .position(chunk_coords)
14028 .ok()
14029 .map(|i| &bt1.records[i])
14030 .map(|r| (r.address, r.nbytes as u64, r.filter_mask));
14031 drop(m);
14032 match found {
14033 Some((addr, nbytes, mask)) => {
14034 self.read_chunk_block(pipeline.as_ref(), addr, nbytes, mask)
14035 }
14036 None => Ok(None),
14037 }
14038 }
14039 }
14040 }
14041
14042 /// The slot one chunk of an implicitly indexed dataset occupies: the
14043 /// address its whole chunk grid starts at, and the chunk's offset within
14044 /// that grid. `data_addr + linear_index * chunk_bytes` is the whole of
14045 /// that index (`H5D__none_idx_get_addr`, H5Dnone.c).
14046 ///
14047 /// The one place a chunk of such a dataset is placed — read and write both
14048 /// come through here, so the bounds check below covers both. The grid it
14049 /// names is [`DatasetInfo::implicit_grid`], which is why the write side
14050 /// can hand [`ContiguousTarget::Local`] to
14051 /// [`write_contiguous_bytes`](Self::write_contiguous_bytes) without asking
14052 /// anything: the external and virtual destinations that owner also knows
14053 /// about are unreachable from a chunked dataset.
14054 fn implicit_chunk_slot(
14055 &self,
14056 ds_index: usize,
14057 geo: &ChunkGeometry,
14058 chunk_coords: &[u64],
14059 ) -> IoResult<(u64, u64)> {
14060 let linear = geo.linear_index(chunk_coords)?;
14061 let ds = self.ds(ds_index);
14062 let m = ds.lock();
14063 let (grid, grid_size) = m.implicit_grid().ok_or_else(|| {
14064 crate::io::IoError::InvalidState("no implicitly indexed chunk grid".into())
14065 })?;
14066 let offset = linear.checked_mul(geo.chunk_bytes()).ok_or_else(|| {
14067 crate::io::IoError::InvalidState("implicit chunk offset overflows u64".into())
14068 })?;
14069 if offset + geo.chunk_bytes() > grid_size {
14070 return Err(crate::io::IoError::InvalidState(format!(
14071 "chunk {chunk_coords:?} lies outside the {grid_size} bytes of chunk space \
14072 this implicitly indexed dataset was created with"
14073 )));
14074 }
14075 Ok((grid, offset))
14076 }
14077
14078 /// Write one whole chunk addressed by its grid coordinates, whichever
14079 /// chunk index the dataset uses. `data` is the chunk's unfiltered bytes;
14080 /// the dataset's filter pipeline (if any) runs here.
14081 ///
14082 /// The write half of the pair with
14083 /// [`read_chunk_at_coords`](Self::read_chunk_at_coords). Unlike the
14084 /// dataset-level `write_chunk_at`, this never grows the dataspace — a
14085 /// hyperslab write is bounded by the current extent by definition.
14086 ///
14087 /// The caller holds the dataset's op lock or the writer exclusively.
14088 pub(crate) fn write_chunk_at_coords(
14089 &self,
14090 ds_index: usize,
14091 chunk_coords: &[u64],
14092 data: &[u8],
14093 ) -> IoResult<()> {
14094 let geo = self.chunk_geometry(ds_index)?;
14095 match geo.kind {
14096 ChunkIndexKind::ExtensibleArray => {
14097 let linear = geo.linear_index(chunk_coords)?;
14098 self.write_chunk_inner(ds_index, linear, data)
14099 }
14100 ChunkIndexKind::FixedArray => {
14101 self.write_chunk_fixed_array_inner(ds_index, chunk_coords, data)
14102 }
14103 ChunkIndexKind::BtreeV2 => {
14104 self.write_chunk_btree_v2_inner(ds_index, chunk_coords, data)
14105 }
14106 ChunkIndexKind::Implicit => {
14107 self.write_chunk_implicit_inner(ds_index, chunk_coords, data)
14108 }
14109 ChunkIndexKind::SingleChunk => {
14110 self.write_chunk_single_chunk_inner(ds_index, chunk_coords, data)
14111 }
14112 ChunkIndexKind::BtreeV1 => {
14113 self.write_chunk_btree_v1_inner(ds_index, chunk_coords, data)
14114 }
14115 }
14116 }
14117
14118 /// Write one whole chunk to a dataset indexed by a version-1 B-tree.
14119 ///
14120 /// `chunk_coords` is the chunk's grid position. `data` is the chunk's
14121 /// unfiltered bytes; the dataset's filter pipeline runs here if it has
14122 /// one, and the key records the stored size and mask the way libhdf5's
14123 /// does (`H5D__btree_new_node`).
14124 ///
14125 /// The caller holds the dataset's op lock or the writer exclusively.
14126 pub(crate) fn write_chunk_btree_v1_inner(
14127 &self,
14128 ds_index: usize,
14129 chunk_coords: &[u64],
14130 data: &[u8],
14131 ) -> IoResult<()> {
14132 // Read what the write needs under a brief guard, then filter OUTSIDE
14133 // the lock, as every other index's write path does.
14134 let ds = self.ds(ds_index);
14135 let (chunk_bytes, pipeline) = {
14136 let m = ds.lock();
14137 let element_size = m.datatype.element_size() as u64;
14138 let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
14139 crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
14140 })?;
14141 (
14142 bt1.chunk_dims.iter().product::<u64>() * element_size,
14143 m.filter_pipeline.clone(),
14144 )
14145 };
14146 if data.len() as u64 != chunk_bytes {
14147 return Err(crate::io::IoError::InvalidState(format!(
14148 "chunk data size mismatch: expected {} bytes, got {}",
14149 chunk_bytes,
14150 data.len()
14151 )));
14152 }
14153
14154 let filtered;
14155 let stored = match pipeline {
14156 Some(ref pl) => {
14157 filtered = filter::apply_filters(pl, data)?;
14158 &filtered[..]
14159 }
14160 None => data,
14161 };
14162 self.record_btree_v1_chunk(ds_index, chunk_coords, stored, 0)
14163 }
14164
14165 /// Write a pre-filtered chunk verbatim to a version-1 B-tree dataset,
14166 /// recording the caller-supplied `filter_mask` — the classic-index half
14167 /// of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
14168 ///
14169 /// The caller holds the dataset's op lock or the writer exclusively.
14170 pub(crate) fn write_compressed_chunk_btree_v1_inner(
14171 &self,
14172 ds_index: usize,
14173 chunk_coords: &[u64],
14174 data: &[u8],
14175 filter_mask: u32,
14176 ) -> IoResult<()> {
14177 if self.ds(ds_index).lock().filter_pipeline.is_none() {
14178 return Err(crate::io::IoError::InvalidState(
14179 "write_chunk_raw requires a filtered dataset (an unfiltered chunk \
14180 is stored at its full size, so there is nothing for a stored size \
14181 or a filter mask to say)"
14182 .into(),
14183 ));
14184 }
14185 self.record_btree_v1_chunk(ds_index, chunk_coords, data, filter_mask)
14186 }
14187
14188 /// Place a chunk's already-final bytes in the file and record them in the
14189 /// version-1 B-tree under the caller-supplied `filter_mask`.
14190 ///
14191 /// Shared by the two writes above, so both reach the index through one
14192 /// placement rule. The records are kept in key order here — the bulk load
14193 /// at flush walks them in that order and a lookup bisects them.
14194 fn record_btree_v1_chunk(
14195 &self,
14196 ds_index: usize,
14197 chunk_coords: &[u64],
14198 final_bytes: &[u8],
14199 filter_mask: u32,
14200 ) -> IoResult<()> {
14201 let stored_len = final_bytes.len() as u64;
14202 // The key's size field is 32 bits wide (`H5D_btree_key_t::nbytes`),
14203 // which is also libhdf5's limit on a chunk in this index.
14204 let Ok(nbytes) = u32::try_from(stored_len) else {
14205 return Err(crate::io::IoError::InvalidState(format!(
14206 "stored chunk size {stored_len} does not fit in the 32-bit size \
14207 field of a version-1 B-tree chunk key"
14208 )));
14209 };
14210 let ds = self.ds(ds_index);
14211 let mut m = ds.lock();
14212 let bt1 = m.btree_v1.as_ref().ok_or_else(|| {
14213 crate::io::IoError::InvalidState("not a version-1 B-tree dataset".into())
14214 })?;
14215 if chunk_coords.len() != bt1.chunk_dims.len() {
14216 return Err(crate::io::IoError::InvalidState(format!(
14217 "chunk_coords has {} entries but the dataset has {} dimensions",
14218 chunk_coords.len(),
14219 bt1.chunk_dims.len()
14220 )));
14221 }
14222 // A coordinate past the maximum extent has no chunk to be: unlike the
14223 // array indexes there is no slot to run out of, so the bound is
14224 // checked here or not at all. An unlimited dimension has none.
14225 for (d, ((&c, &cd), &max)) in chunk_coords
14226 .iter()
14227 .zip(&bt1.chunk_dims)
14228 .zip(&bt1.max_dims)
14229 .enumerate()
14230 {
14231 if max != u64::MAX && c.saturating_mul(cd) >= max {
14232 return Err(crate::io::IoError::InvalidState(format!(
14233 "chunk coordinate {c} in dimension {d} is outside the maximum \
14234 extent {max}"
14235 )));
14236 }
14237 }
14238 let slot = bt1.position(chunk_coords);
14239 let old = slot.ok().map(|i| {
14240 let r = &bt1.records[i];
14241 (r.address, r.nbytes as u64)
14242 });
14243 // A rewrite whose stored size is unchanged stays where it is (always
14244 // so when unfiltered), one that no longer fits moves. See `place_chunk`.
14245 let address = self.place_chunk(old, stored_len);
14246 self.handle.write_at(address, final_bytes)?;
14247
14248 let bt1 = m.btree_v1.as_mut().unwrap();
14249 let record = BtreeV1ChunkRecord {
14250 scaled: chunk_coords.to_vec(),
14251 address,
14252 nbytes,
14253 filter_mask,
14254 };
14255 match slot {
14256 Ok(i) => bt1.records[i] = record,
14257 Err(i) => bt1.records.insert(i, record),
14258 }
14259 bt1.chunks_written += 1;
14260 Ok(())
14261 }
14262
14263 /// Write one whole chunk of an implicitly indexed dataset into the slot
14264 /// its coordinates name. There is no index to record anything in — the
14265 /// slot is where it always was — so this is the write in full.
14266 ///
14267 /// The bytes go through [`write_contiguous_bytes`](Self::write_contiguous_bytes),
14268 /// the one owner of a raw-byte write, against the grid
14269 /// [`implicit_chunk_slot`](Self::implicit_chunk_slot) names.
14270 ///
14271 /// The caller holds the dataset's op lock or the writer exclusively.
14272 pub(crate) fn write_chunk_implicit_inner(
14273 &self,
14274 ds_index: usize,
14275 chunk_coords: &[u64],
14276 data: &[u8],
14277 ) -> IoResult<()> {
14278 let geo = self.chunk_geometry(ds_index)?;
14279 let chunk_bytes = geo.chunk_bytes();
14280 if data.len() as u64 != chunk_bytes {
14281 return Err(crate::io::IoError::InvalidState(format!(
14282 "chunk data size mismatch: expected {} bytes, got {}",
14283 chunk_bytes,
14284 data.len()
14285 )));
14286 }
14287 let (grid, offset) = self.implicit_chunk_slot(ds_index, &geo, chunk_coords)?;
14288 self.write_contiguous_bytes(&ContiguousTarget::Local(grid), offset, data)
14289 }
14290
14291 /// Snapshot the geometry needed to address a chunked dataset's grid.
14292 ///
14293 /// Taken under one brief slot guard so the callers below — which re-lock
14294 /// the slot through `write_chunk`/`read_chunk_*` — never hold it across
14295 /// compression or I/O.
14296 fn chunk_geometry(&self, ds_index: usize) -> IoResult<ChunkGeometry> {
14297 let ds = self.ds(ds_index);
14298 let m = ds.lock();
14299 let Some(kind) = m.chunk_index_kind() else {
14300 return Err(crate::io::IoError::InvalidState(
14301 "not a chunked dataset".into(),
14302 ));
14303 };
14304 let chunk_dims = match kind {
14305 ChunkIndexKind::ExtensibleArray => m.chunked.as_ref().unwrap().chunk_dims.clone(),
14306 ChunkIndexKind::FixedArray => m.fixed_array.as_ref().unwrap().chunk_dims.clone(),
14307 ChunkIndexKind::BtreeV2 => m.btree_v2.as_ref().unwrap().chunk_dims.clone(),
14308 ChunkIndexKind::Implicit => m.implicit.as_ref().unwrap().chunk_dims.clone(),
14309 ChunkIndexKind::SingleChunk => m.single_chunk.as_ref().unwrap().chunk_dims.clone(),
14310 ChunkIndexKind::BtreeV1 => m.btree_v1.as_ref().unwrap().chunk_dims.clone(),
14311 };
14312 Ok(ChunkGeometry {
14313 kind,
14314 dims: m.dataspace.dims.clone(),
14315 max_dims: m.dataspace.max_dims.clone(),
14316 chunk_dims,
14317 element_size: m.datatype.element_size() as u64,
14318 })
14319 }
14320
14321 /// Index-grid slot of the chunk at grid `coords` (see
14322 /// [`crate::io::chunk_grid`]).
14323 pub(crate) fn chunk_slot(&self, ds_index: usize, coords: &[u64]) -> IoResult<u64> {
14324 self.chunk_geometry(ds_index)?.linear_index(coords)
14325 }
14326
14327 /// Grid coordinates of the chunk recorded under index-grid slot `linear`
14328 /// — the inverse of [`Self::chunk_slot`].
14329 pub(crate) fn chunk_coords_from_slot(
14330 &self,
14331 ds_index: usize,
14332 linear: u64,
14333 ) -> IoResult<Vec<u64>> {
14334 let geo = self.chunk_geometry(ds_index)?;
14335 crate::io::chunk_grid::coords_of(
14336 &geo.dims,
14337 geo.max_dims.as_deref(),
14338 &geo.chunk_dims,
14339 linear,
14340 )
14341 }
14342
14343 /// Define a chunked dataset indexed by a fixed array, fixed at its
14344 /// current shape (`max_dims == dims`). `chunk_dims` defines the chunk
14345 /// shape. Returns the dataset index.
14346 pub fn create_fixed_array_dataset(
14347 &self,
14348 name: &str,
14349 datatype: DatatypeMessage,
14350 dims: &[u64],
14351 chunk_dims: &[u64],
14352 ) -> IoResult<usize> {
14353 self.create_fixed_array_dataset_with_max(name, datatype, dims, dims, chunk_dims, None)
14354 }
14355
14356 /// Define a fixed-shape compressed chunked dataset indexed by a
14357 /// *filtered* Fixed Array (`max_dims == dims`).
14358 ///
14359 /// Like `create_fixed_array_dataset`, but the FA header carries the filtered
14360 /// client id and a `chunk_size_len`-wide compressed-size field per chunk
14361 /// (`FixedArrayFilteredChunkElement`), and the dataset gets a filter
14362 /// pipeline. Chunks written via `write_chunk_fixed_array` are compressed and
14363 /// their compressed size + filter mask are recorded in the data block.
14364 ///
14365 /// A convenience over [`create_fixed_array_dataset_with_max`]'s own
14366 /// pipeline argument; production dataset creation calls that directly,
14367 /// so this is kept as a direct entry point for this crate's own
14368 /// white-box tests.
14369 ///
14370 /// [`create_fixed_array_dataset_with_max`]: Self::create_fixed_array_dataset_with_max
14371 #[cfg(all(test, feature = "deflate"))]
14372 pub fn create_fixed_array_dataset_with_pipeline(
14373 &self,
14374 name: &str,
14375 datatype: DatatypeMessage,
14376 dims: &[u64],
14377 chunk_dims: &[u64],
14378 pipeline: FilterPipeline,
14379 ) -> IoResult<usize> {
14380 self.create_fixed_array_dataset_with_max(
14381 name,
14382 datatype,
14383 dims,
14384 dims,
14385 chunk_dims,
14386 Some(pipeline),
14387 )
14388 }
14389
14390 /// Define a chunked dataset indexed by a fixed array, growable up to
14391 /// `max_dims` (every maximum finite — libhdf5 picks this index exactly
14392 /// when no dimension is unlimited).
14393 ///
14394 /// The array is sized for the chunk grid of the *maximum* extent, the
14395 /// libhdf5 rule (`H5D__farray_idx_create` uses `max_nchunks`), so the
14396 /// dataset can be extended to `max_dims` without re-indexing chunks.
14397 pub fn create_fixed_array_dataset_with_max(
14398 &self,
14399 name: &str,
14400 datatype: DatatypeMessage,
14401 dims: &[u64],
14402 max_dims: &[u64],
14403 chunk_dims: &[u64],
14404 pipeline: Option<FilterPipeline>,
14405 ) -> IoResult<usize> {
14406 let create = self.begin_create(name)?;
14407 let name = create.name.as_str();
14408 validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14409 if max_dims.contains(&u64::MAX) {
14410 return Err(crate::io::IoError::InvalidState(
14411 "a fixed-array index requires a fixed maximum shape (no unlimited dimension)"
14412 .into(),
14413 ));
14414 }
14415 let mut num_chunks: u64 = 1;
14416 for g in crate::io::chunk_grid::index_grid(dims, Some(max_dims), chunk_dims)? {
14417 num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
14418 crate::io::IoError::InvalidState("chunk count overflows u64".into())
14419 })?;
14420 }
14421
14422 let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14423 let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
14424
14425 // Create the FA header. For a filtered FA, chunk_size_len is sized
14426 // the same way the filtered Extensible Array path computes it:
14427 // derived from the uncompressed chunk byte count under layout v4,
14428 // the fixed `sizeof_size` under layout v5.
14429 let mut fa_header = if pipeline.is_some() {
14430 let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
14431 FixedArrayHeader::new_for_filtered_chunks(&self.ctx, num_chunks, chunk_size_len)
14432 } else {
14433 FixedArrayHeader::new_for_chunks(&self.ctx, num_chunks)
14434 };
14435 let hdr_encoded = fa_header.encode(&self.ctx);
14436 let fa_header_addr = self
14437 .allocator
14438 .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14439
14440 // Create the FA data block. libhdf5 switches to a paged layout once
14441 // num_elmts exceeds dblk_page_nelmts; both layouts allocate space
14442 // for `num_chunks` entries up front, but the paged layout also
14443 // reserves the page-init bitmap and a per-page checksum.
14444 let fa_dblk = if pipeline.is_some() {
14445 FixedArrayDataBlock::new_filtered(fa_header_addr, num_chunks as usize)
14446 } else {
14447 FixedArrayDataBlock::new_unfiltered(fa_header_addr, num_chunks as usize)
14448 };
14449 let dblk_size = fixed_array_dblk_disk_size(&self.ctx, &fa_header);
14450 let fa_dblk_addr = self.allocator.allocate(dblk_size, FreeSpaceClass::Metadata);
14451
14452 // Update header with data block address
14453 fa_header.data_blk_addr = fa_dblk_addr;
14454
14455 // Write both. The data block content is finalized in `flush_dataset`
14456 // once all chunk addresses are known; here we just reserve space and
14457 // write the header so the file is structurally consistent.
14458 let hdr_encoded = fa_header.encode(&self.ctx);
14459 self.handle.write_at(fa_header_addr, &hdr_encoded)?;
14460 let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa_header, &fa_dblk);
14461 debug_assert_eq!(dblk_encoded.len() as u64, dblk_size);
14462 self.handle.write_at(fa_dblk_addr, &dblk_encoded)?;
14463
14464 // The maximum is stored even when it equals the dims: it is what
14465 // `extend_dataset` checks growth against, and the FA capacity above
14466 // is exactly its chunk grid.
14467 let dataspace = DataspaceMessage {
14468 // Chunked storage always requires at least one dimension, so
14469 // this is never Scalar or Null.
14470 class: DataspaceClass::Simple,
14471 dims: dims.to_vec(),
14472 max_dims: Some(max_dims.to_vec()),
14473 };
14474
14475 let idx = self.push_dataset(
14476 &create,
14477 DatasetInfo {
14478 name: name.to_string(),
14479 datatype,
14480 committed_type: None,
14481 external: None,
14482 virtual_storage: None,
14483 dataspace,
14484 read_format: None,
14485 obj_header_addr: 0,
14486 data_addr: UNDEF_ADDR,
14487 data_size: 0,
14488 compact: None,
14489 attributes: Vec::new(),
14490 obj_header_written_addr: None,
14491 obj_header_blocks: Vec::new(),
14492 filter_pipeline: pipeline,
14493 deleted: false,
14494 extent_dirty: false,
14495 header_dirty: false,
14496 nlink_written: 1,
14497 creation_seq: self.take_creation_seq(),
14498 track_attr_order: self.track_order.attrs,
14499 fill_value: None,
14500 fill_time: FILL_TIME_IFSET,
14501 layout_version,
14502 times: self.created_object_times(),
14503 chunked: None,
14504 btree_v2: None,
14505 implicit: None,
14506 single_chunk: None,
14507 btree_v1: None,
14508 fixed_array: Some(FixedArrayDatasetInfo {
14509 chunk_dims: chunk_dims.to_vec(),
14510 fa_header_addr,
14511 fa_dblk_addr,
14512 fa_header,
14513 fa_dblk,
14514 chunks_written: 0,
14515 }),
14516 append: None,
14517 },
14518 );
14519
14520 Ok(idx)
14521 }
14522
14523 /// Define a chunked dataset with the *implicit* index: no index structure
14524 /// at all, every chunk of the grid allocated at create in one contiguous
14525 /// run, addressed by arithmetic (`H5Dnone.c`).
14526 ///
14527 /// libhdf5 picks this index only where that arithmetic is total, and this
14528 /// enforces the same three conditions
14529 /// (`H5D__layout_set_latest_indexing`, H5Dlayout.c): no filter — a
14530 /// filtered chunk is not `chunk_bytes` long, so the run would not be a
14531 /// grid; no unlimited dimension — the run has to have a length; and early
14532 /// allocation, which is what this creator *does* rather than something it
14533 /// checks. The dataset's fill-value message says so
14534 /// (`build_dataset_header`), because a file claiming incremental
14535 /// allocation is one libhdf5 would never have chosen this index for.
14536 pub fn create_implicit_dataset(
14537 &self,
14538 name: &str,
14539 datatype: DatatypeMessage,
14540 dims: &[u64],
14541 chunk_dims: &[u64],
14542 ) -> IoResult<usize> {
14543 let create = self.begin_create(name)?;
14544 let name = create.name.as_str();
14545 validate_chunk_geometry(dims, dims, chunk_dims)?;
14546 let mut num_chunks: u64 = 1;
14547 for g in crate::io::chunk_grid::index_grid(dims, None, chunk_dims)? {
14548 num_chunks = num_chunks.checked_mul(g).ok_or_else(|| {
14549 crate::io::IoError::InvalidState("chunk count overflows u64".into())
14550 })?;
14551 }
14552 let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14553 let data_size = num_chunks.checked_mul(chunk_bytes).ok_or_else(|| {
14554 crate::io::IoError::InvalidState("implicit chunk storage overflows u64".into())
14555 })?;
14556 let layout_version = self.chunk_layout_version(false, chunk_bytes);
14557
14558 // Early allocation is the whole of this index: the run exists, and
14559 // holds the fill value, before any chunk is written. It is written
14560 // out rather than merely reserved because the file's end-of-file
14561 // address is what libhdf5 checks a file's completeness against — a
14562 // reserved-but-absent tail is a truncated file to it.
14563 let data_addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
14564 self.handle.write_at(
14565 data_addr,
14566 &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
14567 )?;
14568
14569 let dataspace = DataspaceMessage {
14570 // Chunked storage always requires at least one dimension, so
14571 // this is never Scalar or Null.
14572 class: DataspaceClass::Simple,
14573 dims: dims.to_vec(),
14574 max_dims: Some(dims.to_vec()),
14575 };
14576
14577 let idx = self.push_dataset(
14578 &create,
14579 DatasetInfo {
14580 name: name.to_string(),
14581 datatype,
14582 committed_type: None,
14583 external: None,
14584 virtual_storage: None,
14585 dataspace,
14586 read_format: None,
14587 obj_header_addr: 0,
14588 data_addr: UNDEF_ADDR,
14589 data_size: 0,
14590 compact: None,
14591 attributes: Vec::new(),
14592 obj_header_written_addr: None,
14593 obj_header_blocks: Vec::new(),
14594 filter_pipeline: None,
14595 deleted: false,
14596 extent_dirty: false,
14597 header_dirty: false,
14598 nlink_written: 1,
14599 creation_seq: self.take_creation_seq(),
14600 track_attr_order: self.track_order.attrs,
14601 fill_value: None,
14602 fill_time: FILL_TIME_IFSET,
14603 layout_version,
14604 times: self.created_object_times(),
14605 chunked: None,
14606 btree_v2: None,
14607 fixed_array: None,
14608 implicit: Some(ImplicitDatasetInfo {
14609 chunk_dims: chunk_dims.to_vec(),
14610 data_addr,
14611 data_size,
14612 }),
14613 single_chunk: None,
14614 btree_v1: None,
14615 append: None,
14616 },
14617 );
14618
14619 Ok(idx)
14620 }
14621
14622 /// Define a chunked dataset indexed by the single-chunk index: a fixed
14623 /// shape covered by exactly one whole chunk (`chunk_dims == dims`), its
14624 /// address — and, once written, size and filter mask if filtered — held
14625 /// directly in the layout message instead of any index structure
14626 /// (`H5Dsingle.c`). libhdf5 selects this index ahead of both Implicit and
14627 /// Fixed Array whenever the shape qualifies, filtered or not, early
14628 /// allocation or not (`H5D__layout_set_latest_indexing`).
14629 ///
14630 /// `early_alloc` mirrors [`create_implicit_dataset`](Self::create_implicit_dataset):
14631 /// when true, the chunk's storage is allocated and filled with the fill
14632 /// value immediately, matching an early-allocated unfiltered dataset
14633 /// whose one chunk covers the whole shape. When false, the chunk has no
14634 /// address until its first write, the same as an unfiltered Fixed Array
14635 /// element.
14636 pub fn create_single_chunk_dataset(
14637 &self,
14638 name: &str,
14639 datatype: DatatypeMessage,
14640 dims: &[u64],
14641 chunk_dims: &[u64],
14642 early_alloc: bool,
14643 ) -> IoResult<usize> {
14644 let create = self.begin_create(name)?;
14645 let name = create.name.as_str();
14646 validate_chunk_geometry(dims, dims, chunk_dims)?;
14647 let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14648 let layout_version = self.chunk_layout_version(false, data_size);
14649
14650 let data_addr = if early_alloc {
14651 // Same reasoning as `create_implicit_dataset`: the fill-value
14652 // bytes are written now, not merely reserved, because the
14653 // file's end-of-file address is what libhdf5 checks a file's
14654 // completeness against.
14655 let addr = self.allocator.allocate(data_size, FreeSpaceClass::RawData);
14656 self.handle.write_at(
14657 addr,
14658 &crate::format::messages::fill_value::tiled_fill(data_size as usize, None),
14659 )?;
14660 addr
14661 } else {
14662 UNDEF_ADDR
14663 };
14664
14665 let dataspace = DataspaceMessage {
14666 // Chunked storage always requires at least one dimension, so
14667 // this is never Scalar or Null.
14668 class: DataspaceClass::Simple,
14669 dims: dims.to_vec(),
14670 max_dims: Some(dims.to_vec()),
14671 };
14672
14673 let idx = self.push_dataset(
14674 &create,
14675 DatasetInfo {
14676 name: name.to_string(),
14677 datatype,
14678 committed_type: None,
14679 external: None,
14680 virtual_storage: None,
14681 dataspace,
14682 read_format: None,
14683 obj_header_addr: 0,
14684 data_addr: UNDEF_ADDR,
14685 data_size: 0,
14686 compact: None,
14687 attributes: Vec::new(),
14688 obj_header_written_addr: None,
14689 obj_header_blocks: Vec::new(),
14690 filter_pipeline: None,
14691 deleted: false,
14692 extent_dirty: false,
14693 header_dirty: false,
14694 nlink_written: 1,
14695 creation_seq: self.take_creation_seq(),
14696 track_attr_order: self.track_order.attrs,
14697 fill_value: None,
14698 fill_time: FILL_TIME_IFSET,
14699 layout_version,
14700 times: self.created_object_times(),
14701 chunked: None,
14702 btree_v2: None,
14703 fixed_array: None,
14704 implicit: None,
14705 single_chunk: Some(SingleChunkDatasetInfo {
14706 chunk_dims: chunk_dims.to_vec(),
14707 data_addr,
14708 data_size,
14709 nbytes: if early_alloc { data_size } else { 0 },
14710 filter_mask: 0,
14711 chunks_written: 0,
14712 early_alloc,
14713 }),
14714 btree_v1: None,
14715 append: None,
14716 },
14717 );
14718
14719 Ok(idx)
14720 }
14721
14722 /// Define a fixed-shape compressed chunked dataset — of exactly one
14723 /// whole chunk — indexed by a *filtered* single-chunk index
14724 /// (`H5O_LAYOUT_CHUNK_SINGLE_INDEX_WITH_FILTER`, H5Dsingle.c). The
14725 /// chunk's stored size and filter mask are recorded inline in the
14726 /// layout message once the chunk is written.
14727 ///
14728 /// Like [`create_fixed_array_dataset_with_pipeline`](Self::create_fixed_array_dataset_with_pipeline),
14729 /// there is nothing to allocate ahead of that first write — a filtered
14730 /// chunk's stored length isn't known until it is compressed — so this
14731 /// dataset is always incrementally allocated regardless of the caller's
14732 /// requested allocation time.
14733 pub fn create_single_chunk_dataset_with_pipeline(
14734 &self,
14735 name: &str,
14736 datatype: DatatypeMessage,
14737 dims: &[u64],
14738 chunk_dims: &[u64],
14739 pipeline: FilterPipeline,
14740 ) -> IoResult<usize> {
14741 let create = self.begin_create(name)?;
14742 let name = create.name.as_str();
14743 validate_chunk_geometry(dims, dims, chunk_dims)?;
14744 let data_size = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14745 let layout_version = self.chunk_layout_version(true, data_size);
14746
14747 let dataspace = DataspaceMessage {
14748 // Chunked storage always requires at least one dimension, so
14749 // this is never Scalar or Null.
14750 class: DataspaceClass::Simple,
14751 dims: dims.to_vec(),
14752 max_dims: Some(dims.to_vec()),
14753 };
14754
14755 let idx = self.push_dataset(
14756 &create,
14757 DatasetInfo {
14758 name: name.to_string(),
14759 datatype,
14760 committed_type: None,
14761 external: None,
14762 virtual_storage: None,
14763 dataspace,
14764 read_format: None,
14765 obj_header_addr: 0,
14766 data_addr: UNDEF_ADDR,
14767 data_size: 0,
14768 compact: None,
14769 attributes: Vec::new(),
14770 obj_header_written_addr: None,
14771 obj_header_blocks: Vec::new(),
14772 filter_pipeline: Some(pipeline),
14773 deleted: false,
14774 extent_dirty: false,
14775 header_dirty: false,
14776 nlink_written: 1,
14777 creation_seq: self.take_creation_seq(),
14778 track_attr_order: self.track_order.attrs,
14779 fill_value: None,
14780 fill_time: FILL_TIME_IFSET,
14781 layout_version,
14782 times: self.created_object_times(),
14783 chunked: None,
14784 btree_v2: None,
14785 fixed_array: None,
14786 implicit: None,
14787 single_chunk: Some(SingleChunkDatasetInfo {
14788 chunk_dims: chunk_dims.to_vec(),
14789 data_addr: UNDEF_ADDR,
14790 data_size,
14791 nbytes: 0,
14792 filter_mask: 0,
14793 chunks_written: 0,
14794 early_alloc: false,
14795 }),
14796 btree_v1: None,
14797 append: None,
14798 },
14799 );
14800
14801 Ok(idx)
14802 }
14803
14804 /// Define a chunked dataset indexed by a version-1 B-tree — the classic
14805 /// chunk index, and the only one a version-0/1 superblock file can carry.
14806 ///
14807 /// The tree itself is not created here: libhdf5 leaves the layout
14808 /// message's address undefined until the first chunk is inserted
14809 /// (`H5D__btree_idx_create` runs on that insert), and so does this — the
14810 /// flush that bulk-loads the records is what puts a node in the file.
14811 ///
14812 /// Unlike the array indexes this one has no grid to size, so it takes any
14813 /// number of unlimited dimensions: a key *is* the chunk's position, and
14814 /// the tree is ordered by it.
14815 pub fn create_btree_v1_dataset(
14816 &self,
14817 name: &str,
14818 datatype: DatatypeMessage,
14819 dims: &[u64],
14820 max_dims: &[u64],
14821 chunk_dims: &[u64],
14822 pipeline: Option<FilterPipeline>,
14823 ) -> IoResult<usize> {
14824 let create = self.begin_create(name)?;
14825 let name = create.name.as_str();
14826 validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14827 let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14828 if chunk_bytes > u32::MAX as u64 {
14829 return Err(crate::io::IoError::InvalidState(format!(
14830 "a {chunk_bytes}-byte chunk does not fit the 32-bit size field of a \
14831 version-1 B-tree chunk key"
14832 )));
14833 }
14834
14835 let dataspace = DataspaceMessage {
14836 // Chunked storage always requires at least one dimension, so
14837 // this is never Scalar or Null.
14838 class: DataspaceClass::Simple,
14839 dims: dims.to_vec(),
14840 max_dims: Some(max_dims.to_vec()),
14841 };
14842
14843 let idx = self.push_dataset(
14844 &create,
14845 DatasetInfo {
14846 name: name.to_string(),
14847 datatype,
14848 committed_type: None,
14849 external: None,
14850 virtual_storage: None,
14851 dataspace,
14852 read_format: None,
14853 obj_header_addr: 0,
14854 data_addr: UNDEF_ADDR,
14855 data_size: 0,
14856 compact: None,
14857 attributes: Vec::new(),
14858 obj_header_written_addr: None,
14859 obj_header_blocks: Vec::new(),
14860 filter_pipeline: pipeline,
14861 deleted: false,
14862 extent_dirty: false,
14863 header_dirty: false,
14864 nlink_written: 1,
14865 creation_seq: self.take_creation_seq(),
14866 track_attr_order: self.track_order.attrs,
14867 fill_value: None,
14868 fill_time: FILL_TIME_IFSET,
14869 // The version-3 data layout message this index encodes as:
14870 // `H5O_LAYOUT_VERSION_DEFAULT`, which is the floor of
14871 // `H5D__chunk_set_info`'s final MAX and the whole of it below
14872 // the version-4 gate — a bound whose row is lower does not
14873 // push the message down, it only keeps the v1.10 indexes out.
14874 layout_version: LAYOUT_VERSION_DEFAULT,
14875 times: self.created_object_times(),
14876 chunked: None,
14877 fixed_array: None,
14878 btree_v2: None,
14879 implicit: None,
14880 single_chunk: None,
14881 btree_v1: Some(BtreeV1DatasetInfo {
14882 chunk_dims: chunk_dims.to_vec(),
14883 max_dims: max_dims.to_vec(),
14884 config: self.btree_v1_config(),
14885 records: Vec::new(),
14886 node_addrs: Vec::new(),
14887 root_addr: UNDEF_ADDR,
14888 chunks_written: 0,
14889 }),
14890 append: None,
14891 },
14892 );
14893
14894 Ok(idx)
14895 }
14896
14897 /// Define a chunked dataset indexed by a B-tree v2 (multiple unlimited dimensions).
14898 ///
14899 /// Returns the dataset index.
14900 pub fn create_btree_v2_dataset(
14901 &self,
14902 name: &str,
14903 datatype: DatatypeMessage,
14904 dims: &[u64],
14905 max_dims: &[u64],
14906 chunk_dims: &[u64],
14907 ) -> IoResult<usize> {
14908 self.create_btree_v2_dataset_inner(name, datatype, dims, max_dims, chunk_dims, None)
14909 }
14910
14911 /// Define a *filtered* chunked dataset indexed by a B-tree v2.
14912 ///
14913 /// The v2 B-tree counterpart of
14914 /// [`create_chunked_dataset_with_pipeline`](Self::create_chunked_dataset_with_pipeline):
14915 /// chunks are compressed on write and the index records each chunk's
14916 /// stored size and filter mask (record type 11), the same shape libhdf5
14917 /// builds when a multi-unlimited-dimension dataset has a filter pipeline
14918 /// (`H5Dbtree2.c`, `H5D_BT2_FILT`).
14919 pub fn create_btree_v2_dataset_with_pipeline(
14920 &self,
14921 name: &str,
14922 datatype: DatatypeMessage,
14923 dims: &[u64],
14924 max_dims: &[u64],
14925 chunk_dims: &[u64],
14926 pipeline: FilterPipeline,
14927 ) -> IoResult<usize> {
14928 self.create_btree_v2_dataset_inner(
14929 name,
14930 datatype,
14931 dims,
14932 max_dims,
14933 chunk_dims,
14934 Some(pipeline),
14935 )
14936 }
14937
14938 fn create_btree_v2_dataset_inner(
14939 &self,
14940 name: &str,
14941 datatype: DatatypeMessage,
14942 dims: &[u64],
14943 max_dims: &[u64],
14944 chunk_dims: &[u64],
14945 pipeline: Option<FilterPipeline>,
14946 ) -> IoResult<usize> {
14947 use crate::format::chunk_index::btree_v2::Bt2Header;
14948
14949 let create = self.begin_create(name)?;
14950 let name = create.name.as_str();
14951 validate_chunk_geometry(dims, max_dims, chunk_dims)?;
14952 let ndims = dims.len();
14953 let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * datatype.element_size() as u64;
14954 let layout_version = self.chunk_layout_version(pipeline.is_some(), chunk_bytes);
14955
14956 // The filtered record's size field is as wide as libhdf5 will
14957 // recompute it — from the uncompressed chunk size under layout v4,
14958 // the fixed `sizeof_size` under layout v5 — exactly as the
14959 // extensible- and fixed-array filtered paths size theirs.
14960 let bt2_index = match pipeline {
14961 Some(_) => {
14962 let len = self.chunk_size_len_for(layout_version, chunk_bytes);
14963 Bt2ChunkIndex::new_filtered(ndims, len)
14964 }
14965 None => Bt2ChunkIndex::new_unfiltered(ndims),
14966 };
14967
14968 // The bulk loader spreads a level's records evenly over its nodes, one
14969 // separator between adjacent siblings, which needs room for a few
14970 // records per node. HDF5's rank limit of 32 leaves room for seven; a
14971 // wider rank than that has no valid geometry, so reject it here rather
14972 // than emit a tree no reader can walk.
14973 let record_size = bt2_index.record_size(&self.ctx) as usize;
14974 let node_size = bt2_index.node_size as usize;
14975 if node_size < 10 + 3 * record_size {
14976 return Err(crate::io::IoError::InvalidState(format!(
14977 "a {ndims}-dimension v2 B-tree record is {record_size} bytes, too wide \
14978 for a {node_size}-byte node"
14979 )));
14980 }
14981
14982 // Only the header gets a home now: it names an empty tree, whose root
14983 // is undefined until the first flush bulk-loads the index into nodes.
14984 let hdr = if bt2_index.filtered {
14985 Bt2Header::new_for_filtered_chunks(&self.ctx, ndims, bt2_index.chunk_size_len)
14986 } else {
14987 Bt2Header::new_for_chunks(&self.ctx, ndims)
14988 };
14989 let hdr_encoded = hdr.encode(&self.ctx);
14990 let bt2_header_addr = self
14991 .allocator
14992 .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
14993 self.handle.write_at(bt2_header_addr, &hdr_encoded)?;
14994
14995 let dataspace = DataspaceMessage {
14996 // Chunked storage always requires at least one dimension, so
14997 // this is never Scalar or Null.
14998 class: DataspaceClass::Simple,
14999 dims: dims.to_vec(),
15000 max_dims: Some(max_dims.to_vec()),
15001 };
15002
15003 let idx = self.push_dataset(
15004 &create,
15005 DatasetInfo {
15006 name: name.to_string(),
15007 datatype,
15008 committed_type: None,
15009 external: None,
15010 virtual_storage: None,
15011 dataspace,
15012 read_format: None,
15013 obj_header_addr: 0,
15014 data_addr: UNDEF_ADDR,
15015 data_size: 0,
15016 compact: None,
15017 attributes: Vec::new(),
15018 obj_header_written_addr: None,
15019 obj_header_blocks: Vec::new(),
15020 filter_pipeline: pipeline,
15021 deleted: false,
15022 extent_dirty: false,
15023 header_dirty: false,
15024 nlink_written: 1,
15025 creation_seq: self.take_creation_seq(),
15026 track_attr_order: self.track_order.attrs,
15027 fill_value: None,
15028 fill_time: FILL_TIME_IFSET,
15029 layout_version,
15030 times: self.created_object_times(),
15031 chunked: None,
15032 fixed_array: None,
15033 implicit: None,
15034 single_chunk: None,
15035 btree_v1: None,
15036 btree_v2: Some(Bt2DatasetInfo {
15037 chunk_dims: chunk_dims.to_vec(),
15038 bt2_header_addr,
15039 node_addrs: Vec::new(),
15040 index: bt2_index,
15041 chunks_written: 0,
15042 }),
15043 append: None,
15044 },
15045 );
15046
15047 Ok(idx)
15048 }
15049
15050 /// Create a chunked dataset with a custom filter pipeline.
15051 pub fn create_chunked_dataset_with_pipeline(
15052 &self,
15053 name: &str,
15054 datatype: DatatypeMessage,
15055 dims: &[u64],
15056 max_dims: &[u64],
15057 chunk_dims: &[u64],
15058 pipeline: FilterPipeline,
15059 ) -> IoResult<usize> {
15060 let create = self.begin_create(name)?;
15061 let name = create.name.as_str();
15062 validate_chunk_geometry(dims, max_dims, chunk_dims)?;
15063 ensure_at_most_one_unlimited(max_dims)?;
15064 let element_size = datatype.element_size() as u64;
15065 let chunk_bytes: u64 = chunk_dims.iter().product::<u64>() * element_size;
15066 let layout_version = self.chunk_layout_version(true, chunk_bytes);
15067 let chunk_size_len = self.chunk_size_len_for(layout_version, chunk_bytes);
15068
15069 let earray_params = EarrayParams::default_params();
15070 let ndblk_addrs = compute_ndblk_addrs(earray_params.sup_blk_min_data_ptrs)?;
15071 let nsblk_addrs = compute_nsblk_addrs(
15072 earray_params.idx_blk_elmts,
15073 earray_params.data_blk_min_elmts,
15074 earray_params.sup_blk_min_data_ptrs,
15075 earray_params.max_nelmts_bits,
15076 )?;
15077
15078 let mut ea_header =
15079 ExtensibleArrayHeader::new_for_filtered_chunks(&self.ctx, chunk_size_len);
15080 ea_header.max_nelmts_bits = earray_params.max_nelmts_bits;
15081 ea_header.idx_blk_elmts = earray_params.idx_blk_elmts;
15082 ea_header.data_blk_min_elmts = earray_params.data_blk_min_elmts;
15083 ea_header.sup_blk_min_data_ptrs = earray_params.sup_blk_min_data_ptrs;
15084 ea_header.max_dblk_page_nelmts_bits = earray_params.max_dblk_page_nelmts_bits;
15085
15086 let hdr_encoded = ea_header.encode(&self.ctx);
15087 let ea_header_addr = self
15088 .allocator
15089 .allocate(hdr_encoded.len() as u64, FreeSpaceClass::Metadata);
15090
15091 let filt_iblk = FilteredIndexBlock::new(
15092 ea_header_addr,
15093 earray_params.idx_blk_elmts,
15094 ndblk_addrs,
15095 nsblk_addrs,
15096 );
15097 let iblk_encoded = filt_iblk.encode(&self.ctx, chunk_size_len);
15098 let ea_iblk_addr = self
15099 .allocator
15100 .allocate(iblk_encoded.len() as u64, FreeSpaceClass::Metadata);
15101
15102 ea_header.idx_blk_addr = ea_iblk_addr;
15103 let hdr_encoded = ea_header.encode(&self.ctx);
15104 self.handle.write_at(ea_header_addr, &hdr_encoded)?;
15105 self.handle.write_at(ea_iblk_addr, &iblk_encoded)?;
15106
15107 let dataspace = DataspaceMessage {
15108 // Chunked storage always requires at least one dimension, so
15109 // this is never Scalar or Null.
15110 class: DataspaceClass::Simple,
15111 dims: dims.to_vec(),
15112 max_dims: Some(max_dims.to_vec()),
15113 };
15114 let ea_iblk = ExtensibleArrayIndexBlock::new(
15115 ea_header_addr,
15116 earray_params.idx_blk_elmts,
15117 ndblk_addrs,
15118 nsblk_addrs,
15119 );
15120
15121 let idx = self.push_dataset(
15122 &create,
15123 DatasetInfo {
15124 name: name.to_string(),
15125 datatype,
15126 committed_type: None,
15127 external: None,
15128 virtual_storage: None,
15129 dataspace,
15130 read_format: None,
15131 obj_header_addr: 0,
15132 data_addr: UNDEF_ADDR,
15133 data_size: 0,
15134 compact: None,
15135 attributes: Vec::new(),
15136 obj_header_written_addr: None,
15137 obj_header_blocks: Vec::new(),
15138 filter_pipeline: Some(pipeline),
15139 deleted: false,
15140 extent_dirty: false,
15141 header_dirty: false,
15142 nlink_written: 1,
15143 creation_seq: self.take_creation_seq(),
15144 track_attr_order: self.track_order.attrs,
15145 fill_value: None,
15146 fill_time: FILL_TIME_IFSET,
15147 layout_version,
15148 times: self.created_object_times(),
15149 fixed_array: None,
15150 implicit: None,
15151 single_chunk: None,
15152 btree_v1: None,
15153 btree_v2: None,
15154 chunked: Some(ChunkedDatasetInfo {
15155 chunk_dims: chunk_dims.to_vec(),
15156 earray_params,
15157 ea_header_addr,
15158 ea_iblk_addr,
15159 ea_header,
15160 ea_iblk,
15161 chunks_written: 0,
15162 filt_iblk: Some(filt_iblk),
15163 chunk_size_len,
15164 }),
15165 append: None,
15166 },
15167 );
15168 Ok(idx)
15169 }
15170
15171 /// Write a chunk to a fixed-array-indexed dataset.
15172 ///
15173 /// `chunk_coords` is the multidimensional chunk index (e.g., [row_chunk, col_chunk]).
15174 /// The uncompressed `data` must be exactly one chunk wide; the filter
15175 /// pipeline (if any) runs here before the bytes reach the index.
15176 pub fn write_chunk_fixed_array(
15177 &self,
15178 index: usize,
15179 chunk_coords: &[u64],
15180 data: &[u8],
15181 ) -> IoResult<()> {
15182 let ds = self.ds(index);
15183 let _op = ds.op.lock();
15184 self.write_chunk_fixed_array_inner(index, chunk_coords, data)
15185 }
15186
15187 /// [`Self::write_chunk_fixed_array`] body; the caller holds the dataset's
15188 /// op lock or the writer exclusively.
15189 pub(crate) fn write_chunk_fixed_array_inner(
15190 &self,
15191 index: usize,
15192 chunk_coords: &[u64],
15193 data: &[u8],
15194 ) -> IoResult<()> {
15195 // Read what we need under one brief slot guard, then compress
15196 // OUTSIDE the lock: `record_fixed_array_chunk` re-locks the same slot,
15197 // so the guard must be dropped before it (and before apply_filters).
15198 let ds = self.ds(index);
15199 let (chunk_bytes, pipeline) = {
15200 let m = ds.lock();
15201 let element_size = m.datatype.element_size() as u64;
15202 let fa = m.fixed_array.as_ref().ok_or_else(|| {
15203 crate::io::IoError::InvalidState("not a fixed-array dataset".into())
15204 })?;
15205 (
15206 fa.chunk_dims.iter().product::<u64>() * element_size,
15207 m.filter_pipeline.clone(),
15208 )
15209 };
15210
15211 if data.len() as u64 != chunk_bytes {
15212 return Err(crate::io::IoError::InvalidState(format!(
15213 "chunk data size mismatch: expected {} bytes, got {}",
15214 chunk_bytes,
15215 data.len()
15216 )));
15217 }
15218 let write_data;
15219 let data_to_write = if let Some(ref pipeline) = pipeline {
15220 write_data = filter::apply_filters(pipeline, data)?;
15221 &write_data[..]
15222 } else {
15223 data
15224 };
15225 // filter_mask = 0: the whole pipeline ran (or the dataset is
15226 // unfiltered), so no filter is skipped for this chunk.
15227 self.record_fixed_array_chunk(index, chunk_coords, data_to_write, 0)
15228 }
15229
15230 /// Write a pre-filtered chunk verbatim to a fixed-array dataset, recording
15231 /// the caller-supplied `filter_mask`.
15232 ///
15233 /// The bytes are stored exactly as given (no filter pipeline is run); this
15234 /// is the fixed-array half of the HDF5 "direct chunk write"
15235 /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15236 /// means filter *i* of the pipeline was **not** applied to this chunk and
15237 /// must be skipped on read; pass 0 when the full pipeline was applied
15238 /// upstream.
15239 ///
15240 /// Requires a filtered dataset — only the filtered FA element carries the
15241 /// size+mask slot.
15242 ///
15243 /// The caller holds the dataset's op lock or the writer exclusively.
15244 pub(crate) fn write_compressed_chunk_fixed_array_inner(
15245 &self,
15246 index: usize,
15247 chunk_coords: &[u64],
15248 data: &[u8],
15249 filter_mask: u32,
15250 ) -> IoResult<()> {
15251 if self.ds(index).lock().filter_pipeline.is_none() {
15252 return Err(crate::io::IoError::InvalidState(
15253 "write_compressed_chunk_fixed_array requires a filtered dataset \
15254 (no slot for a compressed size or filter mask on an unfiltered \
15255 chunk index)"
15256 .into(),
15257 ));
15258 }
15259 self.record_fixed_array_chunk(index, chunk_coords, data, filter_mask)
15260 }
15261
15262 /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
15263 /// filtered if the dataset is filtered, raw otherwise) into a fixed-array
15264 /// dataset's data block, recording the caller-supplied `filter_mask`.
15265 /// Shared by [`write_chunk_fixed_array`](Self::write_chunk_fixed_array)
15266 /// and [`write_compressed_chunk_fixed_array`](Self::write_compressed_chunk_fixed_array).
15267 fn record_fixed_array_chunk(
15268 &self,
15269 index: usize,
15270 chunk_coords: &[u64],
15271 final_bytes: &[u8],
15272 filter_mask: u32,
15273 ) -> IoResult<()> {
15274 // Hold one slot guard for the whole method; `self.allocator`/`self.handle`/
15275 // `self.ctx` below touch disjoint fields safe to use with the guard held.
15276 let ds = self.ds(index);
15277 let mut m = ds.lock();
15278 let is_filtered = m.filter_pipeline.is_some();
15279 let fa = m
15280 .fixed_array
15281 .as_ref()
15282 .ok_or_else(|| crate::io::IoError::InvalidState("not a fixed-array dataset".into()))?;
15283
15284 // Linear chunk index in the maximum-extent grid — the slot the fixed
15285 // array (sized from that grid at create) records the chunk under.
15286 let linear_idx = crate::io::chunk_grid::linear_index(
15287 &m.dataspace.dims,
15288 m.dataspace.max_dims.as_deref(),
15289 &fa.chunk_dims,
15290 chunk_coords,
15291 )?;
15292
15293 // Update the fixed array data block. The slot is read before the bytes
15294 // are placed so a rewrite can stay where it is (see `place_chunk`).
15295 let fa = m.fixed_array.as_mut().unwrap();
15296 let lidx = linear_idx as usize;
15297 if is_filtered {
15298 // Filtered FA: store address + stored size + filter mask. A
15299 // non-zero mask bit means "filter i was skipped for this chunk".
15300 let stored_size = final_bytes.len();
15301 // The stored size is encoded in the FA header's `chunk_size_len`-byte
15302 // field; libhdf5 errors if it does not fit (H5D_CHUNK_ENCODE_SIZE_CHECK)
15303 // rather than truncating silently. element_size = sizeof_addr +
15304 // chunk_size_len + 4 by construction.
15305 let chunk_size_len = (fa.fa_header.element_size as usize)
15306 .checked_sub(self.ctx.sizeof_addr as usize + 4)
15307 .ok_or_else(|| {
15308 crate::io::IoError::InvalidState(
15309 "filtered fixed-array element size is too small".into(),
15310 )
15311 })?;
15312 if chunk_size_len < 8 && stored_size >= (1usize << (chunk_size_len * 8)) {
15313 return Err(crate::io::IoError::InvalidState(format!(
15314 "compressed chunk size {stored_size} does not fit in the \
15315 {chunk_size_len}-byte fixed-array chunk-size field"
15316 )));
15317 }
15318 if lidx < fa.fa_dblk.filtered_elements.len() {
15319 let old = &fa.fa_dblk.filtered_elements[lidx];
15320 let chunk_addr =
15321 self.place_chunk(Some((old.address, old.chunk_size)), stored_size as u64);
15322 self.handle.write_at(chunk_addr, final_bytes)?;
15323 fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
15324 address: chunk_addr,
15325 chunk_size: stored_size as u64,
15326 filter_mask,
15327 };
15328 fa.chunks_written += 1;
15329 } else {
15330 return Err(crate::io::IoError::InvalidState(format!(
15331 "chunk index {} out of range (max {})",
15332 linear_idx,
15333 fa.fa_dblk.filtered_elements.len()
15334 )));
15335 }
15336 } else {
15337 // An unfiltered fixed array stores only addresses — there is no
15338 // slot for a filter mask, so a non-zero mask cannot be honored.
15339 if filter_mask != 0 {
15340 return Err(crate::io::IoError::InvalidState(
15341 "filter_mask is non-zero but the dataset is unfiltered".into(),
15342 ));
15343 }
15344 if lidx < fa.fa_dblk.elements.len() {
15345 // Unfiltered: the stored size is fixed by the chunk shape, so
15346 // a rewrite always fits its old block.
15347 let old = fa.fa_dblk.elements[lidx];
15348 let len = final_bytes.len() as u64;
15349 let chunk_addr = self.place_chunk(Some((old, len)), len);
15350 self.handle.write_at(chunk_addr, final_bytes)?;
15351 fa.fa_dblk.elements[lidx] = chunk_addr;
15352 fa.chunks_written += 1;
15353 } else {
15354 return Err(crate::io::IoError::InvalidState(format!(
15355 "chunk index {} out of range (max {})",
15356 linear_idx,
15357 fa.fa_dblk.elements.len()
15358 )));
15359 }
15360 }
15361
15362 Ok(())
15363 }
15364
15365 /// Write the one chunk of a single-chunk indexed dataset.
15366 ///
15367 /// `chunk_coords` is validated against the grid the same way every other
15368 /// coordinate-addressed index does (`ChunkGeometry::linear_index`), even
15369 /// though the grid holds exactly one slot — this is what rejects an
15370 /// out-of-range coordinate instead of silently writing to that slot.
15371 /// `data` is the chunk's unfiltered bytes; the dataset's filter pipeline
15372 /// runs here if it has one.
15373 ///
15374 /// The caller holds the dataset's op lock or the writer exclusively.
15375 pub(crate) fn write_chunk_single_chunk_inner(
15376 &self,
15377 index: usize,
15378 chunk_coords: &[u64],
15379 data: &[u8],
15380 ) -> IoResult<()> {
15381 let geo = self.chunk_geometry(index)?;
15382 geo.linear_index(chunk_coords)?;
15383 let chunk_bytes = geo.chunk_bytes();
15384 if data.len() as u64 != chunk_bytes {
15385 return Err(crate::io::IoError::InvalidState(format!(
15386 "chunk data size mismatch: expected {} bytes, got {}",
15387 chunk_bytes,
15388 data.len()
15389 )));
15390 }
15391 let pipeline = self.ds(index).lock().filter_pipeline.clone();
15392 let write_data;
15393 let data_to_write = if let Some(ref pipeline) = pipeline {
15394 write_data = filter::apply_filters(pipeline, data)?;
15395 &write_data[..]
15396 } else {
15397 data
15398 };
15399 // filter_mask = 0: the whole pipeline ran (or the dataset is
15400 // unfiltered), so no filter is skipped for this chunk.
15401 self.record_single_chunk(index, data_to_write, 0)
15402 }
15403
15404 /// Write a pre-filtered chunk verbatim to a single-chunk dataset,
15405 /// recording the caller-supplied `filter_mask`.
15406 ///
15407 /// The bytes are stored exactly as given (no filter pipeline is run); this
15408 /// is the single-chunk half of the HDF5 "direct chunk write"
15409 /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15410 /// means filter *i* of the pipeline was **not** applied to this chunk and
15411 /// must be skipped on read; pass 0 when the full pipeline was applied
15412 /// upstream.
15413 ///
15414 /// Requires a filtered dataset — only the filtered single-chunk layout
15415 /// carries a size+mask slot.
15416 ///
15417 /// The caller holds the dataset's op lock or the writer exclusively.
15418 pub(crate) fn write_compressed_chunk_single_chunk_inner(
15419 &self,
15420 index: usize,
15421 chunk_coords: &[u64],
15422 data: &[u8],
15423 filter_mask: u32,
15424 ) -> IoResult<()> {
15425 if self.ds(index).lock().filter_pipeline.is_none() {
15426 return Err(crate::io::IoError::InvalidState(
15427 "write_compressed_chunk_single_chunk requires a filtered dataset \
15428 (no slot for a compressed size or filter mask on an unfiltered \
15429 chunk index)"
15430 .into(),
15431 ));
15432 }
15433 let geo = self.chunk_geometry(index)?;
15434 geo.linear_index(chunk_coords)?;
15435 self.record_single_chunk(index, data, filter_mask)
15436 }
15437
15438 /// Place an already-final chunk (`final_bytes` is whatever goes to disk —
15439 /// filtered if the dataset is filtered, raw otherwise) into a single-chunk
15440 /// dataset's layout message fields, recording the caller-supplied
15441 /// `filter_mask`. Shared by
15442 /// [`write_chunk_single_chunk_inner`](Self::write_chunk_single_chunk_inner)
15443 /// and
15444 /// [`write_compressed_chunk_single_chunk_inner`](Self::write_compressed_chunk_single_chunk_inner).
15445 ///
15446 /// Unlike the array indexes there is no per-chunk slot to look up — the
15447 /// dataset has exactly one chunk, and its address/size/mask live directly
15448 /// in the layout message (`H5Dsingle.c`) — so this only ever rewrites the
15449 /// one chunk in place, via [`place_chunk`](Self::place_chunk) the same as
15450 /// every other index's rewrite path.
15451 fn record_single_chunk(
15452 &self,
15453 index: usize,
15454 final_bytes: &[u8],
15455 filter_mask: u32,
15456 ) -> IoResult<()> {
15457 let ds = self.ds(index);
15458 let mut m = ds.lock();
15459 let is_filtered = m.filter_pipeline.is_some();
15460 if !is_filtered && filter_mask != 0 {
15461 return Err(crate::io::IoError::InvalidState(
15462 "filter_mask is non-zero but the dataset is unfiltered".into(),
15463 ));
15464 }
15465 let sc = m
15466 .single_chunk
15467 .as_ref()
15468 .ok_or_else(|| crate::io::IoError::InvalidState("not a single-chunk dataset".into()))?;
15469
15470 // A rewrite whose stored size is unchanged stays where it is (always
15471 // so when unfiltered), one that no longer fits moves. See `place_chunk`.
15472 let old = if sc.data_addr == UNDEF_ADDR {
15473 None
15474 } else {
15475 Some((
15476 sc.data_addr,
15477 if is_filtered { sc.nbytes } else { sc.data_size },
15478 ))
15479 };
15480 let stored_size = final_bytes.len() as u64;
15481 let addr = self.place_chunk(old, stored_size);
15482 self.handle.write_at(addr, final_bytes)?;
15483
15484 let sc = m.single_chunk.as_mut().unwrap();
15485 sc.data_addr = addr;
15486 sc.nbytes = stored_size;
15487 sc.filter_mask = filter_mask;
15488 sc.chunks_written = 1;
15489 Ok(())
15490 }
15491
15492 /// Write a chunk to a B-tree v2 indexed dataset.
15493 ///
15494 /// `chunk_coords` is the scaled chunk coordinates (one per dimension).
15495 /// `data` is the chunk's unfiltered bytes; if the dataset has a filter
15496 /// pipeline it runs here and the index records the stored size and mask.
15497 ///
15498 /// Production writes call [`write_chunk_btree_v2_inner`](Self::write_chunk_btree_v2_inner)
15499 /// directly (they already hold the dataset's op lock); this self-locking
15500 /// form is kept as a direct entry point for this crate's own white-box
15501 /// tests.
15502 #[cfg(test)]
15503 pub fn write_chunk_btree_v2(
15504 &self,
15505 index: usize,
15506 chunk_coords: &[u64],
15507 data: &[u8],
15508 ) -> IoResult<()> {
15509 let ds = self.ds(index);
15510 let _op = ds.op.lock();
15511 self.write_chunk_btree_v2_inner(index, chunk_coords, data)
15512 }
15513
15514 /// [`Self::write_chunk_btree_v2`] body; the caller holds the dataset's op
15515 /// lock or the writer exclusively.
15516 pub(crate) fn write_chunk_btree_v2_inner(
15517 &self,
15518 index: usize,
15519 chunk_coords: &[u64],
15520 data: &[u8],
15521 ) -> IoResult<()> {
15522 // Read what the write needs under a brief guard, then compress OUTSIDE
15523 // the lock — filtering a chunk must not hold the dataset slot.
15524 let ds = self.ds(index);
15525 let (chunk_bytes, pipeline) = {
15526 let m = ds.lock();
15527 let element_size = m.datatype.element_size() as u64;
15528 let bt2 = m.btree_v2.as_ref().ok_or_else(|| {
15529 crate::io::IoError::InvalidState("not a B-tree v2 dataset".into())
15530 })?;
15531 (
15532 bt2.chunk_dims.iter().product::<u64>() * element_size,
15533 m.filter_pipeline.clone(),
15534 )
15535 };
15536
15537 if data.len() as u64 != chunk_bytes {
15538 return Err(crate::io::IoError::InvalidState(format!(
15539 "chunk data size mismatch: expected {} bytes, got {}",
15540 chunk_bytes,
15541 data.len()
15542 )));
15543 }
15544
15545 let filtered;
15546 let stored = match pipeline {
15547 Some(ref pl) => {
15548 filtered = filter::apply_filters(pl, data)?;
15549 &filtered[..]
15550 }
15551 None => data,
15552 };
15553
15554 // filter_mask = 0: the whole pipeline ran (or the dataset is
15555 // unfiltered), so no filter is skipped.
15556 self.record_btree_v2_chunk(index, chunk_coords, stored, 0)
15557 }
15558
15559 /// Write a pre-filtered chunk verbatim to a BT2-indexed dataset, recording
15560 /// the caller-supplied `filter_mask`.
15561 ///
15562 /// The v2-B-tree half of the HDF5 "direct chunk write" (`H5Dwrite_chunk`).
15563 /// The bytes are stored exactly as given; `filter_mask` bit *i* set means
15564 /// filter *i* of the pipeline was **not** applied and must be skipped on
15565 /// read. Requires a filtered dataset — only a type-11 record has a slot for
15566 /// a stored size and mask.
15567 ///
15568 /// The caller holds the dataset's op lock or the writer exclusively.
15569 pub(crate) fn write_compressed_chunk_btree_v2_inner(
15570 &self,
15571 index: usize,
15572 chunk_coords: &[u64],
15573 data: &[u8],
15574 filter_mask: u32,
15575 ) -> IoResult<()> {
15576 if self.ds(index).lock().filter_pipeline.is_none() {
15577 return Err(crate::io::IoError::InvalidState(
15578 "write_compressed_chunk_btree_v2 requires a filtered dataset (no \
15579 slot for a compressed size or filter mask on an unfiltered chunk \
15580 index)"
15581 .into(),
15582 ));
15583 }
15584 self.record_btree_v2_chunk(index, chunk_coords, data, filter_mask)
15585 }
15586
15587 /// Place a chunk's already-final bytes (filtered if the dataset is
15588 /// filtered, raw otherwise) in the file and record them in the v2 B-tree,
15589 /// under the caller-supplied `filter_mask`.
15590 ///
15591 /// Shared by [`write_chunk_btree_v2`](Self::write_chunk_btree_v2) and
15592 /// [`write_compressed_chunk_btree_v2`](Self::write_compressed_chunk_btree_v2),
15593 /// so both reach the index through one placement rule.
15594 fn record_btree_v2_chunk(
15595 &self,
15596 index: usize,
15597 chunk_coords: &[u64],
15598 final_bytes: &[u8],
15599 filter_mask: u32,
15600 ) -> IoResult<()> {
15601 let stored_len = final_bytes.len() as u64;
15602 let ds = self.ds(index);
15603 let mut m = ds.lock();
15604 let element_size = m.datatype.element_size() as u64;
15605 let bt2 = m
15606 .btree_v2
15607 .as_ref()
15608 .ok_or_else(|| crate::io::IoError::InvalidState("not a B-tree v2 dataset".into()))?;
15609 let chunk_bytes = bt2.chunk_dims.iter().product::<u64>() * element_size;
15610 // A filtered record encodes the stored size in a `chunk_size_len`-byte
15611 // field that truncates silently. Reject a size that would not fit, as
15612 // the extensible-array path does — the compress path never exceeds it,
15613 // but a direct write with caller-supplied bytes can.
15614 if bt2.index.filtered {
15615 let chunk_size_len = bt2.index.chunk_size_len as usize;
15616 if chunk_size_len < 8 && stored_len >= (1u64 << (chunk_size_len * 8)) {
15617 return Err(crate::io::IoError::InvalidState(format!(
15618 "filtered chunk size {stored_len} does not fit in the \
15619 {chunk_size_len}-byte v2 B-tree chunk-size field"
15620 )));
15621 }
15622 }
15623 // Place the bytes: a rewrite whose stored size is unchanged stays
15624 // where it is (always so when unfiltered — the size is fixed by the
15625 // chunk shape), and one that no longer fits moves, releasing its old
15626 // block. See `place_chunk`.
15627 let old = if bt2.index.filtered {
15628 bt2.index
15629 .lookup_filtered(chunk_coords)
15630 .map(|r| (r.chunk_address, r.chunk_size))
15631 } else {
15632 bt2.index
15633 .lookup(chunk_coords)
15634 .map(|r| (r.chunk_address, chunk_bytes))
15635 };
15636 let chunk_addr = self.place_chunk(old, stored_len);
15637 self.handle.write_at(chunk_addr, final_bytes)?;
15638
15639 let bt2 = m.btree_v2.as_mut().unwrap();
15640 if bt2.index.filtered {
15641 bt2.index
15642 .insert_filtered(chunk_coords.to_vec(), chunk_addr, stored_len, filter_mask);
15643 } else {
15644 bt2.index.insert(chunk_coords.to_vec(), chunk_addr);
15645 }
15646 bt2.chunks_written += 1;
15647
15648 Ok(())
15649 }
15650
15651 /// Write multiple chunks in a batch, optionally compressing in parallel.
15652 ///
15653 /// `chunks` is a list of (chunk_idx, data) pairs for an EA-indexed dataset.
15654 pub fn write_chunks_batch(&self, ds_index: usize, chunks: &[(u64, &[u8])]) -> IoResult<()> {
15655 let ds = self.ds(ds_index);
15656 let _op = ds.op.lock();
15657 self.write_chunks_batch_inner(ds_index, chunks)
15658 }
15659
15660 /// [`Self::write_chunks_batch`] body; the caller holds the dataset's op
15661 /// lock or the writer exclusively.
15662 pub(crate) fn write_chunks_batch_inner(
15663 &self,
15664 ds_index: usize,
15665 chunks: &[(u64, &[u8])],
15666 ) -> IoResult<()> {
15667 #[cfg(feature = "parallel")]
15668 {
15669 // If filter pipeline is set, compress all chunks in parallel.
15670 // Clone the pipeline out under a brief slot guard so the parallel
15671 // compression below runs off the lock.
15672 let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
15673 if let Some(ref pipeline) = pipeline {
15674 let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
15675 // Propagate a filter error rather than storing raw bytes under a
15676 // filter_mask that claims the pipeline ran (see
15677 // apply_filters_parallel). Ok reaching here means every chunk
15678 // compressed fully, so filter_mask = 0 is truthful.
15679 let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15680 for ((idx, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15681 self.write_compressed_chunk_inner(ds_index, *idx, compressed_data, 0)?;
15682 }
15683 return Ok(());
15684 }
15685 }
15686 // Fallback: sequential
15687 for (idx, data) in chunks {
15688 self.write_chunk_inner(ds_index, *idx, data)?;
15689 }
15690 Ok(())
15691 }
15692
15693 /// Write multiple fixed-array chunks in a batch, compressing them in
15694 /// parallel when a filter pipeline is set and the `parallel` feature is on.
15695 ///
15696 /// The fixed-array analogue of [`write_chunks_batch`](Self::write_chunks_batch):
15697 /// chunks are addressed by grid coordinates rather than a linear index.
15698 /// `record_fixed_array_chunk` writes already-compressed bytes verbatim, so
15699 /// the parallel compressor is the only place a filter runs. Falls back to
15700 /// per-chunk [`write_chunk_fixed_array`](Self::write_chunk_fixed_array) when
15701 /// unfiltered or when `parallel` is off.
15702 ///
15703 /// The caller holds the dataset's op lock or the writer exclusively.
15704 pub(crate) fn write_chunks_fixed_array_batch_inner(
15705 &self,
15706 ds_index: usize,
15707 chunks: &[(&[u64], &[u8])],
15708 ) -> IoResult<()> {
15709 #[cfg(feature = "parallel")]
15710 {
15711 // Clone the pipeline out under a brief slot guard so the parallel
15712 // compression below runs off the lock.
15713 let pipeline = self.ds(ds_index).lock().filter_pipeline.clone();
15714 if let Some(ref pipeline) = pipeline {
15715 let chunk_data: Vec<&[u8]> = chunks.iter().map(|&(_, d)| d).collect();
15716 // Same single owner as the EA batch: apply_filters_parallel
15717 // propagates a filter error instead of storing raw bytes under a
15718 // filter_mask that claims the pipeline ran. Ok here means every
15719 // chunk compressed fully, so filter_mask = 0 is truthful.
15720 let compressed = filter::apply_filters_parallel(pipeline, &chunk_data)?;
15721 for ((coords, _), compressed_data) in chunks.iter().zip(compressed.iter()) {
15722 self.record_fixed_array_chunk(ds_index, coords, compressed_data, 0)?;
15723 }
15724 return Ok(());
15725 }
15726 }
15727 // Fallback: sequential (write_chunk_fixed_array_inner compresses per
15728 // chunk).
15729 for (coords, data) in chunks {
15730 self.write_chunk_fixed_array_inner(ds_index, coords, data)?;
15731 }
15732 Ok(())
15733 }
15734
15735 /// Write a pre-filtered chunk verbatim to an EA-indexed dataset, recording
15736 /// the caller-supplied `filter_mask`.
15737 ///
15738 /// The bytes are stored exactly as given (no filter pipeline is run); this
15739 /// is the extensible-array half of the HDF5 "direct chunk write"
15740 /// (`H5Dwrite_chunk`) operation. `filter_mask` is a bitfield: bit *i* set
15741 /// means filter *i* of the pipeline was **not** applied to this chunk and
15742 /// must be skipped on read; pass 0 when the full pipeline was applied
15743 /// upstream.
15744 ///
15745 /// Requires a filtered dataset — only the filtered EA entry carries the
15746 /// size+mask slot. An unfiltered dataset has nowhere to record either.
15747 ///
15748 /// The caller holds the dataset's op lock or the writer exclusively.
15749 pub(crate) fn write_compressed_chunk_inner(
15750 &self,
15751 index: usize,
15752 chunk_idx: u64,
15753 compressed_data: &[u8],
15754 filter_mask: u32,
15755 ) -> IoResult<()> {
15756 if self.ds(index).lock().filter_pipeline.is_none() {
15757 return Err(crate::io::IoError::InvalidState(
15758 "write_compressed_chunk requires a filtered dataset (no slot for \
15759 a compressed size or filter mask on an unfiltered chunk index)"
15760 .into(),
15761 ));
15762 }
15763 self.record_ea_chunk(index, chunk_idx, compressed_data, filter_mask)
15764 }
15765
15766 /// Extend the dimensions of a chunked dataset.
15767 pub fn extend_dataset(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15768 let ds = self.ds(index);
15769 let _op = ds.op.lock();
15770 self.extend_dataset_inner(index, new_dims)
15771 }
15772
15773 /// [`Self::extend_dataset`] body; the caller holds the dataset's op lock
15774 /// or the writer exclusively.
15775 pub(crate) fn extend_dataset_inner(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15776 let ds = self.ds(index);
15777 let mut m = ds.lock();
15778 if !m.is_chunked() {
15779 return Err(crate::io::IoError::InvalidState(
15780 "can only extend chunked datasets".into(),
15781 ));
15782 }
15783 if new_dims.len() != m.dataspace.dims.len() {
15784 return Err(crate::io::IoError::InvalidState(format!(
15785 "extend_dataset rank mismatch: dataset has {} dimensions, got {}",
15786 m.dataspace.dims.len(),
15787 new_dims.len()
15788 )));
15789 }
15790 // The chunk index and append buffers assume the logical size only
15791 // grows; shrinking below already-written data desynchronizes them.
15792 for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15793 if new < cur {
15794 return Err(crate::io::IoError::InvalidState(format!(
15795 "extend_dataset cannot shrink dimension {d} from {cur} to {new}"
15796 )));
15797 }
15798 // An absent maximum shape means the shape is fixed (libhdf5
15799 // defaults maxdims to dims at creation), so any growth exceeds it.
15800 match m.dataspace.max_dims {
15801 Some(ref max) if new > max[d] => {
15802 return Err(crate::io::IoError::InvalidState(format!(
15803 "extend_dataset dimension {d} ({new}) exceeds the maximum {}",
15804 max[d]
15805 )));
15806 }
15807 None if new > cur => {
15808 return Err(crate::io::IoError::InvalidState(format!(
15809 "extend_dataset dimension {d} ({new}) exceeds the maximum {cur}: \
15810 a dataset without a stored maximum shape is fixed at its extent"
15811 )));
15812 }
15813 _ => {}
15814 }
15815 }
15816 if m.dataspace.dims != new_dims {
15817 m.dataspace.dims = new_dims.to_vec();
15818 m.extent_dirty = true;
15819 }
15820 Ok(())
15821 }
15822
15823 /// Set the logical extent of a chunked dataset, growing **or shrinking**
15824 /// any dimension (unlike [`extend_dataset`](Self::extend_dataset), which
15825 /// only grows).
15826 ///
15827 /// A shrink prunes the stored chunks the way libhdf5's
15828 /// `H5D__chunk_prune_by_extent` (H5Dchunk.c) does: a chunk entirely
15829 /// beyond the new extent leaves the chunk index and its block is freed
15830 /// for reuse (kept under SWMR, where a live reader may still hold its
15831 /// address — the rule `H5Dearray.c` applies in `idx_remove`), and a
15832 /// chunk the new extent cuts through has its out-of-extent region
15833 /// overwritten with the fill value, so growing the extent back exposes
15834 /// fill values rather than the stale data.
15835 pub fn set_dataset_extent(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15836 let ds = self.ds(index);
15837 let _op = ds.op.lock();
15838 let old_dims = {
15839 let m = ds.lock();
15840 if !m.is_chunked() {
15841 return Err(crate::io::IoError::InvalidState(
15842 "can only set the extent of chunked datasets".into(),
15843 ));
15844 }
15845 if new_dims.len() != m.dataspace.dims.len() {
15846 return Err(crate::io::IoError::InvalidState(format!(
15847 "set_extent rank mismatch: dataset has {} dimensions, got {}",
15848 m.dataspace.dims.len(),
15849 new_dims.len()
15850 )));
15851 }
15852 // A shrink can cut into buffered rows, whose recorded base would
15853 // then point past the extent; refuse rather than reconcile.
15854 if m.append.is_some() {
15855 return Err(crate::io::IoError::InvalidState(
15856 "set_extent cannot run while the dataset has buffered appends; \
15857 flush them first"
15858 .into(),
15859 ));
15860 }
15861 // An absent maximum shape means the shape is fixed (libhdf5
15862 // defaults maxdims to dims at creation), so growth is bounded by
15863 // the extent.
15864 match m.dataspace.max_dims {
15865 Some(ref max) => {
15866 for (d, (&new, &mx)) in new_dims.iter().zip(max).enumerate() {
15867 if new > mx {
15868 return Err(crate::io::IoError::InvalidState(format!(
15869 "set_extent dimension {d} ({new}) exceeds the maximum {mx}"
15870 )));
15871 }
15872 }
15873 }
15874 None => {
15875 for (d, (&new, &cur)) in new_dims.iter().zip(&m.dataspace.dims).enumerate() {
15876 if new > cur {
15877 return Err(crate::io::IoError::InvalidState(format!(
15878 "set_extent dimension {d} ({new}) exceeds the maximum {cur}: \
15879 a dataset without a stored maximum shape is fixed at its extent"
15880 )));
15881 }
15882 }
15883 }
15884 }
15885 m.dataspace.dims.clone()
15886 };
15887 // A shrink strands chunks; prune them (and refill the straddlers)
15888 // *before* the dims update — chunk addressing uses the
15889 // maximum-extent grid, which the update does not change, and the
15890 // helpers re-lock the slot themselves.
15891 if new_dims.iter().zip(&old_dims).any(|(&n, &o)| n < o) {
15892 self.prune_chunks_beyond(index, new_dims)?;
15893 }
15894 let mut m = ds.lock();
15895 if m.dataspace.dims != new_dims {
15896 m.dataspace.dims = new_dims.to_vec();
15897 m.extent_dirty = true;
15898 }
15899 Ok(())
15900 }
15901
15902 /// Remove and refill the chunks a shrink to `new_dims` strands — the
15903 /// libhdf5 `H5D__chunk_prune_by_extent` behavior. A chunk entirely
15904 /// beyond the new extent leaves the index and its block is freed (kept
15905 /// under SWMR, where a live reader may still hold its address); a chunk
15906 /// the extent cuts through gets its out-of-extent region refilled with
15907 /// the fill value, so a later regrow reads fill, not stale elements.
15908 ///
15909 /// Runs *before* the dims update: the index grid chunks are addressed in
15910 /// comes from the maximum extent, which a shrink never changes, so every
15911 /// stored entry still resolves. The caller holds the dataset's op lock.
15912 fn prune_chunks_beyond(&self, index: usize, new_dims: &[u64]) -> IoResult<()> {
15913 let geo = self.chunk_geometry(index)?;
15914 // A vlen dataset's elements are global-heap IDs: the pruned chunks
15915 // still reference live heap objects, so the walkers read each dead
15916 // chunk's bytes before freeing its block and the heap objects are
15917 // released here — otherwise every shrink strands its strings in the
15918 // file. `release_vlen_references` is a SWMR no-op, so the reads are
15919 // skipped under SWMR too.
15920 let collect_refs = !self.swmr_active && {
15921 let ds = self.ds(index);
15922 let m = ds.lock();
15923 matches!(
15924 m.datatype,
15925 DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
15926 )
15927 };
15928 let (straddlers, dead_refs) = match geo.kind {
15929 ChunkIndexKind::ExtensibleArray => {
15930 self.prune_ea_chunks(index, &geo, new_dims, collect_refs)?
15931 }
15932 ChunkIndexKind::FixedArray => {
15933 self.prune_fa_chunks(index, &geo, new_dims, collect_refs)?
15934 }
15935 ChunkIndexKind::BtreeV2 => {
15936 self.prune_bt2_chunks(index, &geo, new_dims, collect_refs)?
15937 }
15938 // Removing a chunk from the implicit index is
15939 // `H5D__none_idx_remove`: a no-op, because the chunk's space is
15940 // the dataset's space and stays allocated either way. Only the
15941 // straddlers matter, and they are refilled by the caller.
15942 ChunkIndexKind::Implicit => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15943 // A single-chunk index has no per-chunk remove either — its one
15944 // chunk's address lives in the layout message, not an index
15945 // structure, and stays exactly where it is; a shrink only ever
15946 // straddles that one chunk (`H5D__single_idx_remove` is likewise
15947 // a no-op).
15948 ChunkIndexKind::SingleChunk => (self.implicit_straddlers(&geo, new_dims)?, Vec::new()),
15949 ChunkIndexKind::BtreeV1 => {
15950 self.prune_btree_v1_chunks(index, &geo, new_dims, collect_refs)?
15951 }
15952 };
15953 if !dead_refs.is_empty() {
15954 self.release_vlen_references(&dead_refs)?;
15955 }
15956 // Whole-chunk read-modify-write per straddler: an unfiltered chunk
15957 // rewrites in place, a filtered one re-places through `place_chunk`.
15958 let chunk_bytes = geo.chunk_bytes() as usize;
15959 for coords in straddlers {
15960 let Some(mut data) = self.read_chunk_at_coords(index, &coords)? else {
15961 continue;
15962 };
15963 let fill = self.new_chunk_buffer(index, chunk_bytes);
15964 let replaced = refill_chunk_beyond_extent(
15965 &mut data,
15966 &fill,
15967 &coords,
15968 &geo.chunk_dims,
15969 new_dims,
15970 geo.element_size as usize,
15971 );
15972 // Release before the write-back: a filtered straddler re-places
15973 // its block, and freed heap space must be visible to that
15974 // allocation (free-before-alloc, as everywhere else).
15975 if collect_refs && !replaced.is_empty() {
15976 self.release_vlen_references(&replaced)?;
15977 }
15978 self.write_chunk_at_coords(index, &coords, &data)?;
15979 }
15980 Ok(())
15981 }
15982
15983 /// Extensible-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
15984 /// walk every slot the array has ever set, free and clear the entries of
15985 /// chunks entirely beyond `new_dims`, and return the grid coordinates of
15986 /// the chunks that straddle it, plus — when `collect_refs` — the dead
15987 /// chunks' element bytes so the caller can release their heap objects.
15988 fn prune_ea_chunks(
15989 &self,
15990 index: usize,
15991 geo: &ChunkGeometry,
15992 new_dims: &[u64],
15993 collect_refs: bool,
15994 ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
15995 let ds = self.ds(index);
15996 // One slot guard for the whole walk, the `record_ea_chunk` pattern:
15997 // `self.handle`/`self.allocator`/`self.ctx` are disjoint fields.
15998 let mut m = ds.lock();
15999 let is_filtered = m.filter_pipeline.is_some();
16000 let pipeline = m.filter_pipeline.clone();
16001 let chunk_bytes = geo.chunk_bytes();
16002 let (ea_geo, max_nelmts_bits, chunk_size_len, max_idx) = {
16003 let c = m.chunked.as_ref().unwrap();
16004 let p = &c.earray_params;
16005 (
16006 EaGeometry::new(
16007 p.idx_blk_elmts,
16008 p.data_blk_min_elmts,
16009 p.sup_blk_min_data_ptrs,
16010 p.max_nelmts_bits,
16011 p.max_dblk_page_nelmts_bits,
16012 )?,
16013 p.max_nelmts_bits,
16014 c.chunk_size_len,
16015 c.ea_header.max_idx_set,
16016 )
16017 };
16018
16019 let mut straddlers = Vec::new();
16020 let mut dead_refs = Vec::new();
16021
16022 // The decoded data block the walk is currently inside, written back
16023 // when the walk leaves it (or ends) having cleared an entry.
16024 enum Dblk {
16025 Unfiltered(ExtensibleArrayDataBlock),
16026 Filtered(FilteredDataBlock),
16027 }
16028 let mut cache: Option<(u64, Dblk, bool)> = None;
16029 let flush = |cache: &mut Option<(u64, Dblk, bool)>| -> IoResult<()> {
16030 if let Some((addr, blk, dirty)) = cache.take() {
16031 if dirty {
16032 let enc = match &blk {
16033 Dblk::Unfiltered(d) => d.encode(&self.ctx, max_nelmts_bits),
16034 Dblk::Filtered(d) => d.encode(&self.ctx, max_nelmts_bits, chunk_size_len),
16035 };
16036 self.handle.write_at(addr, &enc)?;
16037 }
16038 }
16039 Ok(())
16040 };
16041 // Consecutive slots resolve through the same super block, so keep
16042 // the last decode. Super blocks are only read here — clearing a
16043 // data-block element never moves the block — so it never dirties.
16044 let mut sblk_cache: Option<(usize, ExtensibleArraySuperBlock)> = None;
16045
16046 let mut slot = 0u64;
16047 while slot < max_idx {
16048 let coords = crate::io::chunk_grid::coords_of(
16049 &geo.dims,
16050 geo.max_dims.as_deref(),
16051 &geo.chunk_dims,
16052 slot,
16053 )?;
16054 if !chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
16055 if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16056 straddlers.push(coords);
16057 }
16058 slot += 1;
16059 continue;
16060 }
16061 match ea_geo.locate(slot)? {
16062 EaLoc::Index { elem } => {
16063 let c = m.chunked.as_mut().unwrap();
16064 if is_filtered {
16065 let fiblk = c.filt_iblk.as_mut().unwrap();
16066 let e = fiblk.elements[elem];
16067 if e.addr != UNDEF_ADDR {
16068 if collect_refs {
16069 if let Some(bytes) = self.read_chunk_block(
16070 pipeline.as_ref(),
16071 e.addr,
16072 e.nbytes,
16073 e.filter_mask,
16074 )? {
16075 dead_refs.extend_from_slice(&bytes);
16076 }
16077 }
16078 if !self.swmr_active {
16079 self.allocator
16080 .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
16081 }
16082 fiblk.elements[elem] = FilteredChunkEntry {
16083 addr: UNDEF_ADDR,
16084 nbytes: 0,
16085 filter_mask: 0,
16086 };
16087 }
16088 } else {
16089 let a = c.ea_iblk.elements[elem];
16090 if a != UNDEF_ADDR {
16091 if collect_refs {
16092 if let Some(bytes) =
16093 self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
16094 {
16095 dead_refs.extend_from_slice(&bytes);
16096 }
16097 }
16098 if !self.swmr_active {
16099 self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
16100 }
16101 c.ea_iblk.elements[elem] = UNDEF_ADDR;
16102 }
16103 }
16104 slot += 1;
16105 }
16106 EaLoc::Dblk(l) => {
16107 if l.paged {
16108 return Err(crate::io::IoError::InvalidState(format!(
16109 "chunk index {slot} lives in a paged extensible-array \
16110 data block, which is not yet supported"
16111 )));
16112 }
16113 let dblk_start = slot - l.offset_in_dblk;
16114 let dblk_end = dblk_start + l.dblk_nelmts;
16115 // Resolve the data block's address; an undefined super or
16116 // data block means nothing in its whole element range was
16117 // ever written, so the walk skips the range.
16118 let dblk_addr = {
16119 let c = m.chunked.as_ref().unwrap();
16120 match l.path {
16121 EaDblkPath::Direct { idx } => {
16122 if is_filtered {
16123 c.filt_iblk.as_ref().unwrap().dblk_addrs[idx]
16124 } else {
16125 c.ea_iblk.dblk_addrs[idx]
16126 }
16127 }
16128 EaDblkPath::ViaSblk {
16129 sblk_off,
16130 local_dblk,
16131 ndblks_in_sblk,
16132 ..
16133 } => {
16134 let sblk_addr = if is_filtered {
16135 c.filt_iblk.as_ref().unwrap().sblk_addrs[sblk_off]
16136 } else {
16137 c.ea_iblk.sblk_addrs[sblk_off]
16138 };
16139 if sblk_addr == UNDEF_ADDR {
16140 UNDEF_ADDR
16141 } else {
16142 if sblk_cache.as_ref().map(|&(o, _)| o) != Some(sblk_off) {
16143 let buf = self.handle.read_at_most(sblk_addr, 65536)?;
16144 let sb = ExtensibleArraySuperBlock::decode(
16145 &buf,
16146 &self.ctx,
16147 max_nelmts_bits,
16148 ndblks_in_sblk,
16149 0,
16150 )?;
16151 sblk_cache = Some((sblk_off, sb));
16152 }
16153 sblk_cache.as_ref().unwrap().1.dblk_addrs[local_dblk]
16154 }
16155 }
16156 }
16157 };
16158 if dblk_addr == UNDEF_ADDR {
16159 slot = dblk_end;
16160 continue;
16161 }
16162 if cache.as_ref().map(|&(a, _, _)| a) != Some(dblk_addr) {
16163 flush(&mut cache)?;
16164 let buf = self.handle.read_at_most(dblk_addr, 65536)?;
16165 let blk = if is_filtered {
16166 Dblk::Filtered(FilteredDataBlock::decode(
16167 &buf,
16168 &self.ctx,
16169 max_nelmts_bits,
16170 l.dblk_nelmts as usize,
16171 chunk_size_len,
16172 )?)
16173 } else {
16174 Dblk::Unfiltered(ExtensibleArrayDataBlock::decode(
16175 &buf,
16176 &self.ctx,
16177 max_nelmts_bits,
16178 l.dblk_nelmts as usize,
16179 )?)
16180 };
16181 cache = Some((dblk_addr, blk, false));
16182 }
16183 let (_, blk, dirty) = cache.as_mut().unwrap();
16184 match blk {
16185 Dblk::Filtered(d) => {
16186 let e = d.elements[l.offset_in_dblk as usize];
16187 if e.addr != UNDEF_ADDR {
16188 if collect_refs {
16189 if let Some(bytes) = self.read_chunk_block(
16190 pipeline.as_ref(),
16191 e.addr,
16192 e.nbytes,
16193 e.filter_mask,
16194 )? {
16195 dead_refs.extend_from_slice(&bytes);
16196 }
16197 }
16198 if !self.swmr_active {
16199 self.allocator
16200 .free(e.addr, e.nbytes, FreeSpaceClass::RawData);
16201 }
16202 d.elements[l.offset_in_dblk as usize] = FilteredChunkEntry {
16203 addr: UNDEF_ADDR,
16204 nbytes: 0,
16205 filter_mask: 0,
16206 };
16207 *dirty = true;
16208 }
16209 }
16210 Dblk::Unfiltered(d) => {
16211 let a = d.elements[l.offset_in_dblk as usize];
16212 if a != UNDEF_ADDR {
16213 if collect_refs {
16214 if let Some(bytes) =
16215 self.read_chunk_block(pipeline.as_ref(), a, chunk_bytes, 0)?
16216 {
16217 dead_refs.extend_from_slice(&bytes);
16218 }
16219 }
16220 if !self.swmr_active {
16221 self.allocator.free(a, chunk_bytes, FreeSpaceClass::RawData);
16222 }
16223 d.elements[l.offset_in_dblk as usize] = UNDEF_ADDR;
16224 *dirty = true;
16225 }
16226 }
16227 }
16228 slot += 1;
16229 }
16230 }
16231 }
16232 flush(&mut cache)?;
16233 Ok((straddlers, dead_refs))
16234 }
16235
16236 /// Fixed-array half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16237 /// the whole element array is in memory and flushed at close, so
16238 /// clearing an entry is pure bookkeeping.
16239 fn prune_fa_chunks(
16240 &self,
16241 index: usize,
16242 geo: &ChunkGeometry,
16243 new_dims: &[u64],
16244 collect_refs: bool,
16245 ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16246 let ds = self.ds(index);
16247 let mut m = ds.lock();
16248 let is_filtered = m.filter_pipeline.is_some();
16249 let pipeline = m.filter_pipeline.clone();
16250 let chunk_bytes = geo.chunk_bytes();
16251 let mut straddlers = Vec::new();
16252 let mut dead_refs = Vec::new();
16253 let fa = m.fixed_array.as_mut().unwrap();
16254 let nslots = if is_filtered {
16255 fa.fa_dblk.filtered_elements.len()
16256 } else {
16257 fa.fa_dblk.elements.len()
16258 };
16259 for lidx in 0..nslots {
16260 let (addr, stored, mask) = if is_filtered {
16261 let e = &fa.fa_dblk.filtered_elements[lidx];
16262 (e.address, e.chunk_size, e.filter_mask)
16263 } else {
16264 (fa.fa_dblk.elements[lidx], chunk_bytes, 0)
16265 };
16266 if addr == UNDEF_ADDR {
16267 continue;
16268 }
16269 let coords = crate::io::chunk_grid::coords_of(
16270 &geo.dims,
16271 geo.max_dims.as_deref(),
16272 &geo.chunk_dims,
16273 lidx as u64,
16274 )?;
16275 if chunk_outside_extent(&coords, &geo.chunk_dims, new_dims) {
16276 if collect_refs {
16277 if let Some(bytes) =
16278 self.read_chunk_block(pipeline.as_ref(), addr, stored, mask)?
16279 {
16280 dead_refs.extend_from_slice(&bytes);
16281 }
16282 }
16283 if !self.swmr_active {
16284 self.allocator.free(addr, stored, FreeSpaceClass::RawData);
16285 }
16286 if is_filtered {
16287 fa.fa_dblk.filtered_elements[lidx] = FixedArrayFilteredChunkElement {
16288 address: UNDEF_ADDR,
16289 chunk_size: 0,
16290 filter_mask: 0,
16291 };
16292 } else {
16293 fa.fa_dblk.elements[lidx] = UNDEF_ADDR;
16294 }
16295 } else if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16296 straddlers.push(coords);
16297 }
16298 }
16299 Ok((straddlers, dead_refs))
16300 }
16301
16302 /// Implicit half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16303 /// the grid coordinates of the chunks a shrink to `new_dims` cuts
16304 /// through. Nothing is freed or cleared — this index has no per-chunk
16305 /// state to clear and no per-chunk block to free — so the chunks wholly
16306 /// beyond the extent keep their bytes, exactly as `H5D__none_idx_remove`
16307 /// leaves them. That also means their elements stay reachable, so a
16308 /// variable-length dataset's heap objects must *not* be released here.
16309 fn implicit_straddlers(
16310 &self,
16311 geo: &ChunkGeometry,
16312 new_dims: &[u64],
16313 ) -> IoResult<Vec<Vec<u64>>> {
16314 let mut nchunks: u64 = 1;
16315 for g in
16316 crate::io::chunk_grid::index_grid(&geo.dims, geo.max_dims.as_deref(), &geo.chunk_dims)?
16317 {
16318 nchunks = nchunks.checked_mul(g).ok_or_else(|| {
16319 crate::io::IoError::InvalidState("chunk count overflows u64".into())
16320 })?;
16321 }
16322 let mut straddlers = Vec::new();
16323 for lidx in 0..nchunks {
16324 let coords = crate::io::chunk_grid::coords_of(
16325 &geo.dims,
16326 geo.max_dims.as_deref(),
16327 &geo.chunk_dims,
16328 lidx,
16329 )?;
16330 if chunk_straddles_extent(&coords, &geo.chunk_dims, new_dims) {
16331 straddlers.push(coords);
16332 }
16333 }
16334 Ok(straddlers)
16335 }
16336
16337 /// V2-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16338 /// drop the records of chunks beyond the extent — the next flush
16339 /// re-serializes the smaller tree over the node pool and releases the
16340 /// surplus node blocks.
16341 fn prune_bt2_chunks(
16342 &self,
16343 index: usize,
16344 geo: &ChunkGeometry,
16345 new_dims: &[u64],
16346 collect_refs: bool,
16347 ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16348 let ds = self.ds(index);
16349 let mut m = ds.lock();
16350 let pipeline = m.filter_pipeline.clone();
16351 let chunk_bytes = geo.chunk_bytes();
16352 let swmr = self.swmr_active;
16353 let mut straddlers = Vec::new();
16354 let mut dead_refs = Vec::new();
16355 let bt2 = m.btree_v2.as_mut().unwrap();
16356 if bt2.index.filtered {
16357 let records = std::mem::take(&mut bt2.index.filtered_records);
16358 let mut kept = Vec::with_capacity(records.len());
16359 for r in records {
16360 if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16361 if collect_refs {
16362 if let Some(bytes) = self.read_chunk_block(
16363 pipeline.as_ref(),
16364 r.chunk_address,
16365 r.chunk_size,
16366 r.filter_mask,
16367 )? {
16368 dead_refs.extend_from_slice(&bytes);
16369 }
16370 }
16371 if !swmr {
16372 self.allocator
16373 .free(r.chunk_address, r.chunk_size, FreeSpaceClass::RawData);
16374 }
16375 } else {
16376 if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16377 straddlers.push(r.scaled_offsets.clone());
16378 }
16379 kept.push(r);
16380 }
16381 }
16382 bt2.index.filtered_records = kept;
16383 } else {
16384 let records = std::mem::take(&mut bt2.index.records);
16385 let mut kept = Vec::with_capacity(records.len());
16386 for r in records {
16387 if chunk_outside_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16388 if collect_refs {
16389 if let Some(bytes) = self.read_chunk_block(
16390 pipeline.as_ref(),
16391 r.chunk_address,
16392 chunk_bytes,
16393 0,
16394 )? {
16395 dead_refs.extend_from_slice(&bytes);
16396 }
16397 }
16398 if !swmr {
16399 self.allocator
16400 .free(r.chunk_address, chunk_bytes, FreeSpaceClass::RawData);
16401 }
16402 } else {
16403 if chunk_straddles_extent(&r.scaled_offsets, &geo.chunk_dims, new_dims) {
16404 straddlers.push(r.scaled_offsets.clone());
16405 }
16406 kept.push(r);
16407 }
16408 }
16409 bt2.index.records = kept;
16410 }
16411 Ok((straddlers, dead_refs))
16412 }
16413
16414 /// Version-1-B-tree half of [`prune_chunks_beyond`](Self::prune_chunks_beyond):
16415 /// drop the records of chunks beyond the extent — the next flush
16416 /// re-serializes the smaller tree over the node pool and releases the
16417 /// surplus node blocks.
16418 fn prune_btree_v1_chunks(
16419 &self,
16420 index: usize,
16421 geo: &ChunkGeometry,
16422 new_dims: &[u64],
16423 collect_refs: bool,
16424 ) -> IoResult<(Vec<Vec<u64>>, Vec<u8>)> {
16425 let ds = self.ds(index);
16426 let mut m = ds.lock();
16427 let pipeline = m.filter_pipeline.clone();
16428 let swmr = self.swmr_active;
16429 let mut straddlers = Vec::new();
16430 let mut dead_refs = Vec::new();
16431 let bt1 = m.btree_v1.as_mut().unwrap();
16432 let records = std::mem::take(&mut bt1.records);
16433 let mut kept = Vec::with_capacity(records.len());
16434 for r in records {
16435 if chunk_outside_extent(&r.scaled, &geo.chunk_dims, new_dims) {
16436 if collect_refs {
16437 if let Some(bytes) = self.read_chunk_block(
16438 pipeline.as_ref(),
16439 r.address,
16440 r.nbytes as u64,
16441 r.filter_mask,
16442 )? {
16443 dead_refs.extend_from_slice(&bytes);
16444 }
16445 }
16446 if !swmr {
16447 self.allocator
16448 .free(r.address, r.nbytes as u64, FreeSpaceClass::RawData);
16449 }
16450 } else {
16451 if chunk_straddles_extent(&r.scaled, &geo.chunk_dims, new_dims) {
16452 straddlers.push(r.scaled.clone());
16453 }
16454 kept.push(r);
16455 }
16456 }
16457 m.btree_v1.as_mut().unwrap().records = kept;
16458 Ok((straddlers, dead_refs))
16459 }
16460
16461 /// Flush a chunked dataset's index structures to disk (durable).
16462 ///
16463 /// Writes the index blocks and issues an `fdatasync` so the data is
16464 /// durable — the guarantee SWMR readers and standalone callers rely on.
16465 pub fn flush_dataset(&self, index: usize) -> IoResult<()> {
16466 let ds = self.ds(index);
16467 let _op = ds.op.lock();
16468 self.flush_dataset_synced(index, true)
16469 }
16470
16471 /// Flush a chunked dataset's index structures, syncing only if `sync`.
16472 ///
16473 /// `finalize` threads its own durability choice here so that a
16474 /// [`close_no_sync`](Self::close_no_sync) skips this per-dataset
16475 /// `sync_data` too — otherwise gating only the final `sync_all` would
16476 /// leave one `fdatasync` per indexed dataset and defeat the fast close.
16477 fn flush_dataset_synced(&self, index: usize, sync: bool) -> IoResult<()> {
16478 // Hold one slot guard for the whole method; `self.handle`/`self.ctx`/
16479 // `self.allocator` below touch disjoint fields.
16480 let ds = self.ds(index);
16481 let mut m = ds.lock();
16482
16483 // EA-indexed dataset
16484 if let Some(ref chunked) = m.chunked {
16485 if let Some(ref fiblk) = chunked.filt_iblk {
16486 // Filtered EA
16487 let iblk_encoded = fiblk.encode(&self.ctx, chunked.chunk_size_len);
16488 self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
16489 } else {
16490 // Unfiltered EA
16491 let iblk_encoded = chunked.ea_iblk.encode(&self.ctx);
16492 self.handle.write_at(chunked.ea_iblk_addr, &iblk_encoded)?;
16493 }
16494 let hdr_encoded = chunked.ea_header.encode(&self.ctx);
16495 self.handle.write_at(chunked.ea_header_addr, &hdr_encoded)?;
16496 if sync {
16497 self.handle.sync_data()?;
16498 }
16499 return Ok(());
16500 }
16501
16502 // Fixed-array-indexed dataset
16503 if let Some(ref fa) = m.fixed_array {
16504 let dblk_encoded = encode_fixed_array_dblk(&self.ctx, &fa.fa_header, &fa.fa_dblk);
16505 self.handle.write_at(fa.fa_dblk_addr, &dblk_encoded)?;
16506 let hdr_encoded = fa.fa_header.encode(&self.ctx);
16507 self.handle.write_at(fa.fa_header_addr, &hdr_encoded)?;
16508 if sync {
16509 self.handle.sync_data()?;
16510 }
16511 return Ok(());
16512 }
16513
16514 // BT2-indexed dataset
16515 if let Some(ref bt2) = m.btree_v2 {
16516 // Bulk-load the index into fixed-size nodes and lay them over the
16517 // dataset's block pool. Because every node is the same size, the
16518 // blocks already on disk are reused in place and only the shortfall
16519 // is allocated — the pool is the single owner of these addresses,
16520 // so no flush leaves a block behind. The addresses a reader already
16521 // holds stay valid, which is also what SWMR needs.
16522 let tree = bt2.index.build_tree(&self.ctx);
16523 let mut node_addrs = bt2.node_addrs.clone();
16524 while node_addrs.len() < tree.nodes.len() {
16525 node_addrs.push(
16526 self.allocator
16527 .allocate(tree.node_size as u64, FreeSpaceClass::Metadata),
16528 );
16529 }
16530 // A tree with fewer nodes than last flush releases the surplus
16531 // rather than leaving it recorded and unreachable, so the pool is
16532 // exactly one block per node whichever way the count moved. Under
16533 // SWMR a reader may still hold a header naming those blocks, so
16534 // keep them out of the free list — the same rule `place_chunk`
16535 // applies to a relocated chunk.
16536 for addr in node_addrs.split_off(tree.nodes.len()) {
16537 if !self.swmr_active {
16538 self.allocator
16539 .free(addr, tree.node_size as u64, FreeSpaceClass::Metadata);
16540 }
16541 }
16542
16543 for (image, &addr) in tree.encode(&self.ctx, &node_addrs).iter().zip(&node_addrs) {
16544 self.handle.write_at(addr, image)?;
16545 }
16546
16547 // The root is the last node the bulk load emits.
16548 let root_addr = match tree.nodes.len() {
16549 0 => UNDEF_ADDR,
16550 n => node_addrs[n - 1],
16551 };
16552 let hdr_encoded = tree.header(root_addr).encode(&self.ctx);
16553 self.handle.write_at(bt2.bt2_header_addr, &hdr_encoded)?;
16554
16555 m.btree_v2.as_mut().unwrap().node_addrs = node_addrs;
16556
16557 if sync {
16558 self.handle.sync_data()?;
16559 }
16560 return Ok(());
16561 }
16562
16563 // Version-1-B-tree-indexed dataset
16564 if let Some(ref bt1) = m.btree_v1 {
16565 // Bulk-loaded over the same block pool the v2 B-tree above uses,
16566 // and for the same reason: every node of a v1 tree is the width
16567 // its "K" value gives, so a block stays usable however the tree
16568 // reshapes, and only the shortfall is ever allocated.
16569 let element_size = m.datatype.element_size() as u64;
16570 let tree = bt1.build_tree(element_size, self.ctx.sizeof_addr as usize);
16571 let node_size = tree.node_size() as u64;
16572 let mut node_addrs = bt1.node_addrs.clone();
16573 while node_addrs.len() < tree.node_count() {
16574 node_addrs.push(self.allocator.allocate(node_size, FreeSpaceClass::Metadata));
16575 }
16576 // A tree with fewer nodes than last flush releases the surplus
16577 // straight away, where the v2 B-tree has to keep it out of the
16578 // free list for a live SWMR reader: this index lives only in a
16579 // classic file, which `start_swmr` refuses outright (and upstream
16580 // says the same in `H5D_COPS_BTREE`).
16581 for addr in node_addrs.split_off(tree.node_count()) {
16582 self.allocator
16583 .free(addr, node_size, FreeSpaceClass::Metadata);
16584 }
16585 for (image, &addr) in tree.encode(&node_addrs)?.iter().zip(&node_addrs) {
16586 self.handle.write_at(addr, image)?;
16587 }
16588 // The root is the last node the bulk load emits, and is undefined
16589 // while the dataset has no chunks — what the version-3 data
16590 // layout message then carries, exactly as libhdf5 leaves it.
16591 let root_addr = tree.root_address(&node_addrs);
16592 let bt1 = m.btree_v1.as_mut().unwrap();
16593 bt1.node_addrs = node_addrs;
16594 bt1.root_addr = root_addr;
16595
16596 if sync {
16597 self.handle.sync_data()?;
16598 }
16599 return Ok(());
16600 }
16601
16602 Ok(())
16603 }
16604
16605 /// Finalize and close the file.
16606 ///
16607 /// Writes the dataset object headers, root group object header, and
16608 /// superblock. After this call the file is a valid HDF5 file.
16609 pub fn close(mut self) -> IoResult<()> {
16610 self.close_in_place()
16611 }
16612
16613 /// [`close`](Self::close) for a holder that cannot give the writer up by
16614 /// value because it has a `Drop` of its own ([`SwmrWriter`]): the same
16615 /// one-shot commit, after which this writer's `Drop` is a no-op.
16616 ///
16617 /// [`SwmrWriter`]: crate::io::swmr::SwmrWriter
16618 pub(crate) fn close_in_place(&mut self) -> IoResult<()> {
16619 // Mark closed BEFORE finalizing: finalize writes external truth
16620 // (object headers + superblock) and must run exactly once. If we
16621 // finalized first and it failed, the `?` would return with `closed`
16622 // still false, and dropping `self` would re-run `finalize` a second
16623 // time over a half-written file (and print the "call close()" notice
16624 // the caller already heeded). Committing to the close path first makes
16625 // `Drop` (the only other finalize site) a no-op regardless of outcome,
16626 // so the error is reported exactly once via this `Result`.
16627 self.closed = true;
16628 self.finalize(true)
16629 }
16630
16631 /// Finalize and close the file without a final `fsync`.
16632 ///
16633 /// Identical to [`close`](Self::close) — the same object headers and
16634 /// superblock are written, so on return the file is a complete, valid HDF5
16635 /// file readable by any process — except that the trailing `sync_all`
16636 /// (fsync) is skipped. The bytes are handed to the OS but are not
16637 /// guaranteed durable against power loss or an OS crash until the OS
16638 /// flushes its page cache; a normal process exit or a same-machine reader
16639 /// sees the full file regardless.
16640 ///
16641 /// This trades durability for speed: `sync_all` typically dominates close
16642 /// latency, so bulk writers that do not need crash durability (the file can
16643 /// be regenerated) can use this to avoid that cost. Use [`close`](Self::close)
16644 /// when durability matters. `Drop` always finalizes durably, so a writer
16645 /// finalized this way must reach `close_no_sync` explicitly.
16646 pub fn close_no_sync(mut self) -> IoResult<()> {
16647 // Same close-once discipline as `close`: commit to the close path
16648 // before finalizing so `Drop` cannot re-run `finalize` on failure.
16649 self.closed = true;
16650 self.finalize(false)
16651 }
16652
16653 /// Provide mutable access to the underlying file handle.
16654 pub fn handle(&mut self) -> &mut FileHandle {
16655 &mut self.handle
16656 }
16657
16658 /// The superblock version this file will be written with.
16659 ///
16660 /// `H5F__super_init` takes the oldest version that can describe the file
16661 /// and raises it to the one the file's library-version low bound implies:
16662 /// `super_vers = MAX(super_vers, HDF5_superblock_ver_bounds[low_bound])`,
16663 /// with the bounds table reading 0, 2, 3, 3, 3, 3, 3 for EARLIEST, V18,
16664 /// V110, V112, V114, V200, LATEST (H5Fsuper.c:68, :1128-1154). A file
16665 /// created at `H5F_LIBVER_EARLIEST` takes that bound's entry directly
16666 /// ([`SuperblockVersion::Chosen`], and the classic branch below) — version
16667 /// 0, or version 2 when the file carries shared messages, whose master
16668 /// table needs the superblock extension only a version-2 superblock has
16669 /// (H5Fsuper.c:1135). For every other file the bound is read back from
16670 /// what this crate writes:
16671 ///
16672 /// * The floor is `H5F_LIBVER_V18`, hence version 2. Every group such a
16673 /// file holds is a link-message group, which libhdf5 only writes at a
16674 /// low bound of V18 or newer (`use_at_least_v18`, H5Gobj.c:179), and
16675 /// every object header in it is version 2, which `H5O_obj_ver_bounds`
16676 /// likewise puts at V18 (H5Oint.c:125). A version-0 superblock over
16677 /// this content would claim a file libhdf5 1.6 can read, and no libhdf5
16678 /// writes that combination.
16679 /// * A chunked dataset — extensible array, fixed array or version-2
16680 /// B-tree, all reached through a version-4 or -5 data layout message —
16681 /// reads back as V110 (`H5O_layout_ver_bounds`, H5Dlayout.c:44), hence
16682 /// version 3.
16683 /// * SWMR writes version 3 outright (H5Fsuper.c:1129).
16684 ///
16685 /// A file whose caller *named* a bound skips the read-back and takes that
16686 /// bound's row directly, so `V18` stays at version 2 however its chunked
16687 /// datasets are indexed — which is what libhdf5 does, the layout version
16688 /// being no input to `H5F__super_init` at all.
16689 ///
16690 /// None of that applies to a reopened file. `H5F__super_read` validates
16691 /// the version it finds and never recomputes one, so the version written
16692 /// back is the version read, whatever this session appends — see
16693 /// [`SuperblockVersion`].
16694 fn superblock_version_for(&self, flags: u8) -> u8 {
16695 let chosen = match self.superblock_version {
16696 SuperblockVersion::Existing(version) => return version,
16697 SuperblockVersion::Chosen(version) => version,
16698 };
16699 if self.is_legacy() {
16700 // A classic file keeps the version it was created at — 0, or 2
16701 // when its shared messages needed the extension. Nothing a session
16702 // can add reaches past that: its objects get symbol-table links,
16703 // its chunked datasets the version-1 B-tree behind a version-3
16704 // layout message, and the two features that would raise the bound
16705 // — SWMR and the 2.0 format — are refused where the caller asks
16706 // for them.
16707 return chosen;
16708 }
16709 let mut version = chosen
16710 .max(SUPERBLOCK_V2)
16711 .max(self.effective_libver().superblock_version());
16712 if self.swmr_active || flags & FLAG_SWMR_WRITE != 0 {
16713 version = version.max(SUPERBLOCK_V3);
16714 }
16715 version
16716 }
16717
16718 /// The low bound a modern file this writer *created* is effectively
16719 /// written at: the one the caller named, or — with none named — the one
16720 /// its content reads back as. A reopened file never reaches here; its
16721 /// superblock version is not derived from its content at all.
16722 ///
16723 /// The read-back is what `superblock_version_for` needs and the field
16724 /// alone cannot give: this crate's default file names no bound, and the
16725 /// generation it writes is not one bound but two rows (see the `libver`
16726 /// field). The floor is `V18`, the oldest bound under which libhdf5 writes
16727 /// link-message groups (`use_at_least_v18`, H5Gobj.c:179) and version-2
16728 /// object headers (`H5O_obj_ver_bounds`, H5Oint.c:125), which is all such
16729 /// a file holds; a v1.10 chunk index in it raises that to `V110`, the
16730 /// oldest bound whose `H5O_layout_ver_bounds` row reaches the version-4
16731 /// layout message that index is written behind.
16732 fn effective_libver(&self) -> LibverBound {
16733 self.libver.unwrap_or_else(|| {
16734 if self.has_v110_chunk_index() {
16735 LibverBound::V110
16736 } else {
16737 LibverBound::V18
16738 }
16739 })
16740 }
16741
16742 /// Whether any dataset still in the file is indexed by a v1.10 chunk
16743 /// index — the markers `build_dataset_header` turns into a version-4/5
16744 /// data layout message, and nothing else it can emit reaches that
16745 /// version.
16746 ///
16747 /// Not "is any dataset chunked": the version-1 B-tree is a chunk index
16748 /// that encodes as a *version-3* layout message, the version
16749 /// `H5O_layout_ver_bounds` gives the earliest bound, so a dataset using
16750 /// it asks nothing of the superblock.
16751 fn has_v110_chunk_index(&self) -> bool {
16752 self.dataset_refs().iter().any(|d| {
16753 let m = d.lock();
16754 !m.deleted
16755 && m.chunk_index_kind()
16756 .is_some_and(|k| k != ChunkIndexKind::BtreeV1)
16757 })
16758 }
16759
16760 /// Write the superblock at offset 0 with the given flags.
16761 ///
16762 /// Requires that the root group has already been written (via `finalize`
16763 /// or `finalize_for_swmr`).
16764 pub fn write_superblock(&mut self, flags: u8) -> IoResult<()> {
16765 let root_addr = self
16766 .root_group_addr
16767 .ok_or_else(|| crate::io::IoError::InvalidState("root group not yet written".into()))?;
16768 // The userblock this file was opened with. `H5F__super_read` prefers
16769 // the located address over this field, but `H5Pget_userblock` reports
16770 // it, so a rewrite that zeroed it would hide the block from every
16771 // reader that asks for its size.
16772 let base = self.handle.base();
16773 // The end of file is the one address in the superblock measured from
16774 // the start of the *file* rather than from the base: `H5F__super_read`
16775 // sets the EOA to `stored_eof - base_addr` (H5Fsuper.c:635) and calls
16776 // the file truncated when `eof + base_addr < stored_eof` (:573). The
16777 // allocator counts in the based space, so the userblock is added back.
16778 let eof = self.allocator.eof() + base;
16779 let version = self.superblock_version_for(flags);
16780 // Which of the two images is written follows the version, not the
16781 // generation: a classic file carrying shared messages is a version-2
16782 // superblock over version-1 messages and symbol-table groups
16783 // (H5Fsuper.c:1135), and only the version-2/3 image has the extension
16784 // address that table is reached through. Below version 2 the file is
16785 // always a classic one — the other branch floors at 2.
16786 if let Some(legacy) = self.legacy.as_deref().filter(|_| version < SUPERBLOCK_V2) {
16787 // Re-emitted, not rebuilt: the "K" ranks, the userblock size and
16788 // the driver info address are recorded nowhere else in the file,
16789 // and every node width in it is derived from the ranks. Only the
16790 // three things this session can have changed are recomputed.
16791 let root_stab = self
16792 .symbol_tables
16793 .written
16794 .lock()
16795 .get(&LinkScope::Root)
16796 .copied();
16797 let mut sb = legacy.superblock.clone();
16798 sb.version = version;
16799 sb.file_consistency_flags = flags as u32;
16800 sb.end_of_file_address = eof;
16801 sb.root_symbol_table_entry.obj_header_addr = root_addr;
16802 // `H5G__stab_valid` (H5Groot.c) reads this pair back and compares
16803 // it against the root header's Symbol Table message, repairing the
16804 // superblock when they disagree. Writing the pair that message now
16805 // names is what keeps the file from needing that repair. A root
16806 // that keeps its links in messages has no such pair and no entry
16807 // in `written`, and gets `H5G_NOTHING_CACHED` — what libhdf5
16808 // writes for the same root.
16809 sb.root_symbol_table_entry.cache = match root_stab {
16810 Some(s) => SymbolTableCache::SymbolTable {
16811 btree_addr: s.btree_addr,
16812 heap_addr: s.heap_addr,
16813 },
16814 None => SymbolTableCache::Nothing,
16815 };
16816 self.handle.write_at(0, &sb.encode())?;
16817 return Ok(());
16818 }
16819 let sb = SuperblockV2V3 {
16820 version,
16821 sizeof_offsets: self.ctx.sizeof_addr,
16822 sizeof_lengths: self.ctx.sizeof_size,
16823 file_consistency_flags: flags,
16824 base_address: base,
16825 // Whatever `write_superblock_extension` put there, which is the
16826 // only place an extension is written.
16827 superblock_extension_address: self.extension.addr.lock().unwrap_or(UNDEF_ADDR),
16828 end_of_file_address: eof,
16829 root_group_object_header_address: root_addr,
16830 };
16831 self.handle.write_at(0, &sb.encode())?;
16832 Ok(())
16833 }
16834
16835 /// Re-write a dataset's object header in place (SWMR update).
16836 ///
16837 /// The header must have been written by `finalize_for_swmr`, and goes
16838 /// back over the same blocks: chunk 0 held to its block and the
16839 /// continuation chunk, when it has one, to its own. Only the dataspace
16840 /// dimensions are meant to change; a header that no longer fits is
16841 /// refused rather than moved, since a reader holds its address.
16842 pub fn write_dataset_header_inplace(&mut self, index: usize) -> IoResult<()> {
16843 // Scope the slot guard: `build_dataset_header` re-locks the same slot.
16844 let placement = {
16845 let ds = self.ds(index);
16846 let m = ds.lock();
16847 HeaderPlacement::over(&m.obj_header_blocks).ok_or_else(|| {
16848 crate::io::IoError::InvalidState("dataset header not yet written".into())
16849 })?
16850 };
16851
16852 let header = self.build_dataset_header(index)?;
16853 let nlink = self.object_link_count(HardLinkTarget::Dataset(index));
16854 let format = self.dataset_header_format(index);
16855 let images = self.encode_header_in(&header, nlink, format, &placement)?;
16856 let reserved = placement.blocks();
16857 let fits = images.len() == reserved.len()
16858 && images
16859 .iter()
16860 .zip(&reserved)
16861 .all(|((_, image), &(_, size))| image.len() as u64 == size);
16862 if !fits {
16863 return Err(crate::io::IoError::InvalidState(format!(
16864 "dataset header grew from {} to {} bytes; cannot rewrite in place",
16865 reserved.iter().map(|&(_, size)| size).sum::<u64>(),
16866 images.iter().map(|(_, image)| image.len()).sum::<usize>()
16867 )));
16868 }
16869 for (addr, image) in &images {
16870 self.handle.write_at(*addr, image)?;
16871 }
16872 // Only after the bytes are down: a failed write leaves the registry
16873 // describing the header the file still holds.
16874 self.ds(index).lock().header_written(nlink);
16875 Ok(())
16876 }
16877
16878 /// Perform a full finalize for SWMR mode.
16879 ///
16880 /// This writes all dataset object headers, the root group header, and the
16881 /// superblock with SWMR flags. After this call, the file is valid for
16882 /// SWMR readers. Subsequent writes use in-place updates.
16883 pub fn finalize_for_swmr(&mut self) -> IoResult<()> {
16884 self.reject_swmr()?;
16885 // 0. Flush all chunked dataset index structures.
16886 for i in 0..self.dataset_count() {
16887 let is_indexed = {
16888 let ds = self.ds(i);
16889 let m = ds.lock();
16890 !m.deleted && m.is_chunked()
16891 };
16892 if is_indexed {
16893 self.flush_dataset(i)?;
16894 }
16895 }
16896
16897 // 1. Allocate every object header (none for a dataset deleted before
16898 // start_swmr — its storage was freed at delete time). Same three
16899 // phases as the full finalize, and for the same reason: nothing a
16900 // header names can be laid out until every object has an address.
16901 let live: Vec<usize> = (0..self.dataset_count())
16902 .filter(|&i| !self.ds(i).lock().deleted)
16903 .collect();
16904 let kept = self.supersede_headers(&live);
16905 // Before any dataset header: a sharing dataset's header names the
16906 // committed type's address.
16907 self.write_committed_datatype_headers()?;
16908 let layout = self.allocate_object_headers(&live, &kept)?;
16909
16910 // 2. Build content against those addresses.
16911 self.prepare_dense_attributes(&live)?;
16912 self.prepare_link_storage()?;
16913 self.write_reference_values()?;
16914
16915 // 3. Write every object header.
16916 self.write_object_headers(&layout)?;
16917 // Where each header is published and how much room it has: what
16918 // `write_dataset_header_inplace` rewrites within, and what the
16919 // closing finalize writes over, since a reader may by then hold any
16920 // of these addresses.
16921 for &(i, placement) in &layout.datasets {
16922 let ds = self.ds(i);
16923 let mut m = ds.lock();
16924 m.obj_header_written_addr = Some(placement.addr);
16925 m.obj_header_blocks = placement.blocks();
16926 }
16927 for &(gi, placement) in &layout.groups {
16928 let grp = self.grp(gi);
16929 let mut g = grp.lock();
16930 g.obj_header_written_addr = Some(placement.addr);
16931 g.obj_header_blocks = placement.blocks();
16932 }
16933 self.superseded_root_header = layout.root.blocks();
16934
16935 // 4. Write superblock with SWMR flags.
16936 self.write_superblock(FLAG_WRITE_ACCESS | FLAG_SWMR_WRITE)?;
16937 self.handle.set_eof(self.allocator.eof())?;
16938
16939 self.handle.sync_all()?;
16940 // Readers can now be following this file, so a chunk that moves must
16941 // leave its old block intact for whoever is still holding the previous
16942 // index (see `swmr_active`).
16943 self.swmr_active = true;
16944 Ok(())
16945 }
16946
16947 // ------------------------------------------------------------------
16948 // Internal helpers
16949 // ------------------------------------------------------------------
16950
16951 /// Flush every dataset's append buffer into the chunks it belongs to,
16952 /// through [`flush_append_buffer`](Self::flush_append_buffer): frames
16953 /// already in the chunk survive, and the rest of it reads back as the
16954 /// dataset's fill value (zeros when none is defined).
16955 fn flush_append_buffers(&mut self) -> IoResult<()> {
16956 for i in 0..self.dataset_count() {
16957 if self.ds(i).lock().deleted {
16958 continue;
16959 }
16960 self.flush_append_buffer(i)?;
16961 }
16962 Ok(())
16963 }
16964
16965 /// Write all object headers and the superblock, producing a complete,
16966 /// valid HDF5 file.
16967 ///
16968 /// `sync == true` issues a final `sync_all` (fsync) so the bytes are
16969 /// durable against power loss / OS crash before returning. `sync == false`
16970 /// skips that fsync: the file is still fully written to the OS and readable
16971 /// by any process, but durability is left to the OS page-cache flush. This
16972 /// is the only difference between [`close`](Self::close) (durable) and
16973 /// [`close_no_sync`](Self::close_no_sync) (fast).
16974 fn finalize(&mut self, sync: bool) -> IoResult<()> {
16975 // Flush any partial append buffers before finalizing
16976 self.flush_append_buffers()?;
16977
16978 // A SWMR session (`finalize_for_swmr` already ran, so
16979 // `root_group_addr` is `Some`) is closed by the same full finalize as
16980 // a fresh write: every object header is rebuilt over its chunk 0 and
16981 // the superblock is written with clean-close flags. A full rebuild —
16982 // rather than the in-place header rewrite used by the live
16983 // `SwmrWriter::flush` path — is required so any structural change made
16984 // after `start_swmr` is committed to the final file. A hard link, in
16985 // particular, both grows its target's header with an object
16986 // reference-count message and adds a `MSG_LINK` record to a group
16987 // header; an in-place rewrite cannot accommodate the grown header and
16988 // never re-emits group/root headers. The fall-through below already
16989 // handles datasets whose header was written by `finalize_for_swmr`
16990 // (`obj_header_written_addr.is_some()`).
16991
16992 // 0. Flush chunked dataset index structures (only modified datasets).
16993 for i in 0..self.dataset_count() {
16994 let ds = self.ds(i);
16995 {
16996 let m = ds.lock();
16997 if m.deleted {
16998 continue;
16999 }
17000 if m.obj_header_written_addr.is_some() && !m.storage_dirty() {
17001 continue;
17002 }
17003 let is_indexed = m.is_chunked();
17004 if !is_indexed {
17005 continue;
17006 }
17007 }
17008 self.flush_dataset_synced(i, sync)?;
17009 }
17010
17011 // 1. Plan. Which datasets get a header (deleted datasets get none —
17012 // their storage was already freed at delete time) is settled first,
17013 // because everything the next phases lay out is laid out only for the
17014 // headers this finalize actually rewrites; and every header those
17015 // phases supersede is taken here, before the first allocation, so
17016 // the rewrite lands over it.
17017 let mut rewritten: Vec<usize> = Vec::new();
17018 // A finalize that lays the shared-message table out afresh reassigns
17019 // every heap ID in the file, so no existing header can keep its bytes:
17020 // the pointers in them name heap objects the new table does not have.
17021 let table_replaced = self.rebuilds_shared_messages();
17022 for i in 0..self.dataset_count() {
17023 // Before the slot guard: `object_link_count` re-locks every
17024 // dataset and group slot, this one included.
17025 let nlink = self.object_link_count(HardLinkTarget::Dataset(i));
17026 let ds = self.ds(i);
17027 let mut m = ds.lock();
17028 if m.deleted {
17029 continue;
17030 }
17031 // An existing dataset from append mode keeps its header — and
17032 // everything that header names — unless this session changed
17033 // what the header says.
17034 if let Some(written) = m.obj_header_written_addr {
17035 if !table_replaced && !m.header_stale_with(nlink) {
17036 // Keep the original object header address for the root group link.
17037 m.obj_header_addr = written;
17038 continue;
17039 }
17040 }
17041 rewritten.push(i);
17042 }
17043 let kept = self.supersede_headers(&rewritten);
17044
17045 // 2. Allocate. Committed datatype headers go down whole: a header of
17046 // theirs holds a datatype and a reference count, so it waits on
17047 // nothing, while a dataset sharing the type and the group naming it
17048 // both store its address. They are written before the shared-message
17049 // phase opens, so a committed type reaches the file as itself.
17050 self.write_committed_datatype_headers()?;
17051 self.begin_shared_message_layout();
17052 let layout = self.allocate_object_headers(&rewritten, &kept)?;
17053
17054 // 3. Build content, with every object header's address known. Dense
17055 // attribute storage holds the attribute messages themselves — an
17056 // object reference among them is a header address; dense links and
17057 // symbol tables name header addresses; a reference dataset's elements
17058 // are header addresses. Nothing here is a fixup: each is written once,
17059 // with the value the file keeps. The shared-message table comes last:
17060 // it counts the bodies the headers will hold, and the three above are
17061 // what settle them.
17062 self.prepare_dense_attributes(&rewritten)?;
17063 self.prepare_link_storage()?;
17064 self.write_reference_values()?;
17065 self.prepare_shared_messages(&rewritten)?;
17066 self.write_superblock_extension()?;
17067
17068 // 4. Write every object header over the block phase 2 reserved for it.
17069 self.write_object_headers(&layout)?;
17070
17071 // 5. Write superblock at offset 0.
17072 self.write_superblock(0)?;
17073
17074 // 6. End the file where its address space ends (`H5FD_truncate`, which
17075 // `H5F__dest` calls on every close). Allocated-but-unwritten space at
17076 // the end would otherwise leave the file shorter than the end-of-file
17077 // address the superblock just recorded, which libhdf5 reads as a
17078 // truncated file.
17079 self.handle.set_eof(self.allocator.eof())?;
17080
17081 // Durability is opt-in per call: `close` passes `true`, `close_no_sync`
17082 // passes `false`, and `Drop` passes `true` so an un-`close`d writer is
17083 // still finalized durably by default.
17084 if sync {
17085 self.handle.sync_all()?;
17086 }
17087 Ok(())
17088 }
17089
17090 /// Take the on-disk header of every object this finalize rewrites — the
17091 /// datasets in `datasets`, every group, and the root — and hand each
17092 /// chunk-0 block to [`allocate_object_headers`](Self::allocate_object_headers)
17093 /// to be written over.
17094 ///
17095 /// Chunk 0 stays where it is: its address is what every reference in the
17096 /// file holds. The continuation blocks behind it go back to the free
17097 /// list, so the rewrite reuses them instead of growing the file on every
17098 /// open/close cycle — nothing names one but its own header. Hard links
17099 /// can alias one header under several names; the set keeps an aliased
17100 /// chain from being taken twice. The registry forgets each header here,
17101 /// so a finalize that fails later describes none the file no longer holds.
17102 fn supersede_headers(&mut self, datasets: &[usize]) -> KeptChunks {
17103 let mut kept = KeptChunks::default();
17104 let mut taken = std::collections::HashSet::new();
17105 for &i in datasets {
17106 let ds = self.ds(i);
17107 let mut m = ds.lock();
17108 let Some(old) = m.obj_header_written_addr.take() else {
17109 continue;
17110 };
17111 let blocks = std::mem::take(&mut m.obj_header_blocks);
17112 if !blocks.is_empty() && taken.insert(old) {
17113 kept.datasets.insert(i, self.keep_chunk0(blocks));
17114 }
17115 }
17116 for gi in 0..self.group_count() {
17117 let grp = self.grp(gi);
17118 let mut g = grp.lock();
17119 let Some(old) = g.obj_header_written_addr.take() else {
17120 continue;
17121 };
17122 let blocks = std::mem::take(&mut g.obj_header_blocks);
17123 if !blocks.is_empty() && taken.insert(old) {
17124 kept.groups.insert(gi, self.keep_chunk0(blocks));
17125 }
17126 }
17127 let root_blocks = std::mem::take(&mut self.superseded_root_header);
17128 if root_blocks
17129 .first()
17130 .is_some_and(|&(addr, _)| taken.insert(addr))
17131 {
17132 kept.root = Some(self.keep_chunk0(root_blocks));
17133 }
17134 kept
17135 }
17136
17137 /// Keep `blocks`' chunk 0 for a rewrite and free the continuation blocks
17138 /// behind it — never under SWMR, where a live reader may be walking them,
17139 /// the same rule `release_vlen_references` and `place_chunk` follow.
17140 fn keep_chunk0(&self, blocks: crate::io::object_header_io::HeaderBlocks) -> (u64, u64) {
17141 let mut blocks = blocks.into_iter();
17142 let chunk0 = blocks.next().expect("a written header has a chunk 0");
17143 if !self.swmr_active {
17144 for (addr, len) in blocks {
17145 self.allocator.free(addr, len, FreeSpaceClass::Metadata);
17146 }
17147 }
17148 chunk0
17149 }
17150
17151 /// Give every object header this finalize writes an address, before
17152 /// anything that names one is built.
17153 ///
17154 /// INVARIANT: from the moment this returns until the file is closed, every
17155 /// object in it has the object header address it will be found at. That is
17156 /// what lets the phase after this one say an address wherever the format
17157 /// wants one — in a link message, in a symbol table entry, in a reference
17158 /// dataset's elements, and in an attribute's value, which is the one of the
17159 /// four that cannot be revisited after its header is written.
17160 ///
17161 /// An object in `kept` is placed over the chunk-0 block it already has, so
17162 /// its address is the one every reference in the file already holds. A
17163 /// header is measured before its content is final, which is sound because
17164 /// no address changes its length: every address is a fixed-width field,
17165 /// and an object that has none yet reads as zero, which is the same width.
17166 /// The storage a header names is laid out between the two passes for the
17167 /// same reason and answers the same way — `emit_attributes` and
17168 /// `emit_links` each fall back to a size-equal placeholder message. It is
17169 /// [`write_object_headers`](Self::write_object_headers) that checks this
17170 /// held, rather than either pass assuming it.
17171 fn allocate_object_headers(
17172 &mut self,
17173 datasets: &[usize],
17174 kept: &KeptChunks,
17175 ) -> IoResult<HeaderLayout> {
17176 let mut layout = HeaderLayout {
17177 datasets: Vec::with_capacity(datasets.len()),
17178 groups: Vec::new(),
17179 root: HeaderPlacement::fresh(0, 0),
17180 };
17181 for &i in datasets {
17182 let header = self.build_dataset_header(i)?;
17183 let format = self.dataset_header_format(i);
17184 let placement = self.place_header(&header, format, kept.datasets.get(&i).copied())?;
17185 self.ds(i).lock().obj_header_addr = placement.addr;
17186 layout.datasets.push((i, placement));
17187 }
17188 for gi in 0..self.group_count() {
17189 if self.grp(gi).lock().deleted {
17190 continue;
17191 }
17192 let header = self.build_group_header(gi)?;
17193 let format = self.group_header_format(gi);
17194 let placement = self.place_header(&header, format, kept.groups.get(&gi).copied())?;
17195 self.grp(gi).lock().obj_header_addr = placement.addr;
17196 layout.groups.push((gi, placement));
17197 }
17198 let header = self.build_root_group_header()?;
17199 let format = self.header_format(self.root_track_order);
17200 let placement = self.place_header(&header, format, kept.root)?;
17201 self.root_group_addr = Some(placement.addr);
17202 layout.root = placement;
17203 Ok(layout)
17204 }
17205
17206 /// Write every object header over the blocks
17207 /// [`allocate_object_headers`](Self::allocate_object_headers) reserved for
17208 /// it.
17209 ///
17210 /// The single owner of object header writing in both finalize paths, and
17211 /// the only place a header's body meets its block: a body that does not
17212 /// fill its measurement exactly fails the finalize here rather than
17213 /// overrunning the next object or leaving a tail of the previous one, which
17214 /// is how a message whose length turns out to depend on an address would
17215 /// show up.
17216 fn write_object_headers(&mut self, layout: &HeaderLayout) -> IoResult<()> {
17217 for &(i, placement) in &layout.datasets {
17218 let rc = self.object_link_count(HardLinkTarget::Dataset(i));
17219 let header = self.build_dataset_header(i)?;
17220 let format = self.dataset_header_format(i);
17221 let what = format!("dataset '{}'", self.ds(i).lock().name);
17222 self.write_header_in(&header, rc, format, &placement, &what)?;
17223 // Only after the bytes are down: a failed write leaves the registry
17224 // describing the header the file still holds.
17225 self.ds(i).lock().header_written(rc);
17226 }
17227 for &(gi, placement) in &layout.groups {
17228 let rc = self.object_link_count(HardLinkTarget::Group(gi));
17229 let header = self.build_group_header(gi)?;
17230 let format = self.group_header_format(gi);
17231 let what = format!("group '{}'", self.grp(gi).lock().name);
17232 self.write_header_in(&header, rc, format, &placement, &what)?;
17233 }
17234 let header = self.build_root_group_header()?;
17235 let format = self.header_format(self.root_track_order);
17236 self.write_header_in(&header, 1, format, &layout.root, "the root group")
17237 }
17238
17239 /// Encode `header` into `placement` and write it, after checking each
17240 /// image against the block reserved for it.
17241 fn write_header_in(
17242 &mut self,
17243 header: &ObjectHeader,
17244 rc: u32,
17245 format: ObjectFormat,
17246 placement: &HeaderPlacement,
17247 what: &str,
17248 ) -> IoResult<()> {
17249 let images = self.encode_header_in(header, rc, format, placement)?;
17250 let reserved =
17251 std::iter::once(placement.size).chain(placement.continuation.map(|(_, s)| s));
17252 for ((addr, image), size) in images.iter().zip(reserved) {
17253 check_header_size(image, size, || what.to_string())?;
17254 self.handle.write_at(*addr, image)?;
17255 }
17256 Ok(())
17257 }
17258
17259 fn build_dataset_header(&self, index: usize) -> IoResult<ObjectHeader> {
17260 // Compute the link count first: object_link_count re-locks dataset and
17261 // group slots (including this one), so it must run before we take this
17262 // dataset's slot guard — otherwise it would deadlock on the same slot.
17263 let rc = self.object_link_count(HardLinkTarget::Dataset(index));
17264 // Same reason: reading the committed type's address locks the
17265 // committed-datatype registry, which the slot guard below must not be
17266 // held across.
17267 let committed = self.ds(index).lock().committed_type;
17268 let committed_addr = committed.map(|r| match r {
17269 CommittedTypeRef::Session(ci) => self.committed_datatypes.lock()[ci].obj_header_addr,
17270 CommittedTypeRef::Preserved(addr) => addr,
17271 });
17272 // And again: an attribute holding an object reference is said in the
17273 // target's header address, which is read off that object's slot.
17274 let attributes = self.object_attributes(AttrScope::Dataset(index))?;
17275
17276 // Hold one slot guard for the whole header build.
17277 let ds = self.ds(index);
17278 let m = ds.lock();
17279 let mut header = ObjectHeader::new();
17280
17281 // Every message below is written in the format this dataset already
17282 // has, not the one this session would pick. libhdf5 grows a header in
17283 // place and never re-encodes a message it did not touch, so a reopen
17284 // at a newer bound leaves the version-1 dataspaces an EARLIEST-bound
17285 // creating session wrote exactly as they are. This writer has to lay
17286 // the whole header out again whenever the shared-message heap moves,
17287 // so preserving the encoding is the only way to land on the same
17288 // bytes.
17289 let format = m.read_format.unwrap_or_else(|| self.message_format());
17290 let libver = match format {
17291 ObjectFormat::Legacy => LibverBound::Earliest,
17292 ObjectFormat::Modern => self.encoding_libver(),
17293 };
17294
17295 // Dataspace message (type 0x01)
17296 let ds_msg = m.dataspace.encode_for(&self.ctx, format);
17297 let owner = ShareOwner::Header(m.obj_header_addr);
17298 let (flags, ds_msg) = self.share_message(owner, MSG_DATASPACE, 0x00, ds_msg);
17299 header.add_message(MSG_DATASPACE, flags, ds_msg);
17300
17301 // Datatype message (type 0x03). A dataset built on a committed type
17302 // stores a pointer to that object header in place of the message, and
17303 // the shared flag is what says the body is a pointer — the two are one
17304 // statement, so they are written together.
17305 match committed_addr {
17306 Some(addr) => header.add_message(
17307 MSG_DATATYPE,
17308 MSG_FLAG_CONSTANT | MSG_FLAG_SHARED,
17309 SharedMessagePointer::encode_committed(addr, &self.ctx),
17310 ),
17311 None => {
17312 let body = m.datatype.encode_at(&self.ctx, libver);
17313 let (flags, body) = if self.dataset_datatype_shareable(&m.datatype, libver) {
17314 self.share_message(owner, MSG_DATATYPE, MSG_FLAG_CONSTANT, body)
17315 } else {
17316 (MSG_FLAG_CONSTANT, body)
17317 };
17318 header.add_message(MSG_DATATYPE, flags, body)
17319 }
17320 }
17321
17322 // Fill Value message (type 0x05)
17323 let is_chunked = m.is_chunked();
17324 // `H5P__init_def_layout` gives each storage class its own default
17325 // allocation time: incremental for chunked and for virtual (whose
17326 // source datasets are allocated as they are written), early for
17327 // compact (the space is the header, so it exists as soon as the
17328 // dataset does), late for contiguous. An implicitly indexed dataset is
17329 // the one chunked exception, and not by default but by definition:
17330 // early allocation is a *condition* of that index
17331 // (`H5D__layout_set_latest_indexing`), so a header claiming
17332 // incremental would describe a file libhdf5 would never have chosen
17333 // this index for. A single-chunk dataset can go either way — unlike
17334 // Implicit, early allocation is not one of its selection conditions
17335 // — so its `early_alloc` flag (set only for an unfiltered dataset
17336 // created that way) is what this checks instead.
17337 let alloc_time = if m.compact.is_some()
17338 || m.implicit.is_some()
17339 || m.single_chunk.as_ref().is_some_and(|s| s.early_alloc)
17340 {
17341 1 // early
17342 } else if is_chunked || m.virtual_storage.is_some() {
17343 3 // incremental
17344 } else {
17345 2 // late
17346 };
17347 // `H5D__update_oh_info` (H5Dint.c:927-943): a variable-length
17348 // datatype with no explicit fill value forces ALLOC regardless of
17349 // the declared policy — its heap-reference encoding has no safe
17350 // all-zero "no fill" representation, so libhdf5 always writes the
17351 // (empty) fill value at allocation for such a dataset. `IFSET` is
17352 // the only declared policy this touches: an explicit `ALLOC` is
17353 // already what it forces, and upstream rejects `NEVER` for a
17354 // VL-typed dataset at `H5Dcreate` outright — this crate's
17355 // VL-typed datasets have no builder path to declare `NEVER` in the
17356 // first place, so that branch cannot be reached here.
17357 let is_vlen = matches!(
17358 m.datatype,
17359 DatatypeMessage::VarLenString { .. } | DatatypeMessage::VarLenSequence { .. }
17360 );
17361 let fill_write_time = if is_vlen && m.fill_value.is_none() && m.fill_time == FILL_TIME_IFSET
17362 {
17363 FILL_TIME_ALLOC
17364 } else {
17365 m.fill_time
17366 };
17367 let fv = if let Some(ref bytes) = m.fill_value {
17368 // User-defined fill value (fill_defined = 2).
17369 FillValueMessage {
17370 alloc_time,
17371 fill_write_time,
17372 fill_defined: 2,
17373 fill_value: Some(bytes.clone()),
17374 }
17375 } else {
17376 // No fill value of the dataset's own (fill_defined = 1, the
17377 // implicit default zero fill) — `alloc_time` above already
17378 // carries the per-layout-class default (`H5P__set_layout`,
17379 // H5Pdcpl.c:1864-1877), so this branch must use it too instead
17380 // of `FillValueMessage::default()`'s hardcoded LATE: that was
17381 // wrong for a compact (EARLY) or virtual (INCR) dataset with no
17382 // fill value, only coincidentally right for contiguous.
17383 FillValueMessage {
17384 alloc_time,
17385 fill_write_time,
17386 fill_defined: 1, // default value (zeros)
17387 fill_value: None,
17388 }
17389 };
17390 // `H5O_MSG_FLAG_CONSTANT`, as `H5D__update_oh_info` appends it
17391 // (H5Dint.c:965) — the same flag the datatype message beside it
17392 // carries (H5Dint.c:961) and the old fill value below (H5Dint.c:981).
17393 // A dataset's fill value is fixed at creation: `H5Pset_fill_value` is
17394 // a creation property, so nothing can rewrite the message in place and
17395 // libhdf5 tells the header so.
17396 let fv_msg = fv.encode_for(format);
17397 let (flags, fv_msg) = self.share_message(owner, MSG_FILL_VALUE, MSG_FLAG_CONSTANT, fv_msg);
17398 header.add_message(MSG_FILL_VALUE, flags, fv_msg);
17399
17400 // The "fill value (old)" message (type 0x04) beside the new one, for a
17401 // user-defined fill value below the v1.8 bound. `H5D__update_oh_info`
17402 // (H5Dint.c:1024-1035) appends `H5O_FILL_ID` whenever `fill_prop->buf`
17403 // is set and `use_at_least_v18` — `H5F_LOW_BOUND(file) >= V18`, which
17404 // here is exactly a non-`Legacy` message format — is false, so that a
17405 // reader that predates the new message still finds the value. The body
17406 // is the size and the bytes and nothing else: no allocation time, no
17407 // write time, no defined flag (`H5O__fill_old_encode`, H5Ofill.c:512).
17408 if matches!(format, ObjectFormat::Legacy) {
17409 if let Some(ref bytes) = m.fill_value {
17410 let mut old = Vec::with_capacity(4 + bytes.len());
17411 old.extend_from_slice(&(bytes.len() as u32).to_le_bytes());
17412 old.extend_from_slice(bytes);
17413 let (flags, old) =
17414 self.share_message(owner, MSG_FILL_VALUE_OLD, MSG_FLAG_CONSTANT, old);
17415 header.add_message(MSG_FILL_VALUE_OLD, flags, old);
17416 }
17417 }
17418
17419 // External Data Files message (type 0x07), before the layout message
17420 // and marked constant, exactly where `H5D__layout_oh_create` puts it.
17421 // It is what makes a reader route the dataset's I/O through the files
17422 // it names rather than through the undefined address the layout
17423 // message below still declares.
17424 if let Some(ref ext) = m.external {
17425 header.add_message(
17426 MSG_EXTERNAL_FILE_LIST,
17427 MSG_FLAG_CONSTANT,
17428 ext.message().encode(&self.ctx),
17429 );
17430 }
17431
17432 // Data Layout message (type 0x08)
17433 let layout = if let Some(ref chunked) = m.chunked {
17434 let mut layout_dims = chunked.chunk_dims.clone();
17435 layout_dims.push(m.datatype.element_size() as u64);
17436 DataLayoutMessage::chunked_v4_earray(
17437 m.layout_version,
17438 layout_dims,
17439 chunked.earray_params.clone(),
17440 chunked.ea_header_addr,
17441 )
17442 } else if let Some(ref fa) = m.fixed_array {
17443 let mut layout_dims = fa.chunk_dims.clone();
17444 layout_dims.push(m.datatype.element_size() as u64);
17445 DataLayoutMessage::chunked_v4_farray(
17446 m.layout_version,
17447 layout_dims,
17448 FixedArrayParams::default_params(),
17449 fa.fa_header_addr,
17450 )
17451 } else if let Some(ref bt2) = m.btree_v2 {
17452 let mut layout_dims = bt2.chunk_dims.clone();
17453 layout_dims.push(m.datatype.element_size() as u64);
17454 DataLayoutMessage::chunked_v4_btree_v2(
17455 m.layout_version,
17456 layout_dims,
17457 crate::format::messages::data_layout::Bt2Params {
17458 node_size: bt2.index.node_size,
17459 split_percent: bt2.index.split_percent,
17460 merge_percent: bt2.index.merge_percent,
17461 },
17462 bt2.bt2_header_addr,
17463 )
17464 } else if let Some(ref imp) = m.implicit {
17465 let mut layout_dims = imp.chunk_dims.clone();
17466 layout_dims.push(m.datatype.element_size() as u64);
17467 DataLayoutMessage::chunked_v4_implicit(m.layout_version, layout_dims, imp.data_addr)
17468 } else if let Some(ref sc) = m.single_chunk {
17469 let mut layout_dims = sc.chunk_dims.clone();
17470 layout_dims.push(m.datatype.element_size() as u64);
17471 if m.filter_pipeline.is_some() {
17472 DataLayoutMessage::chunked_v4_single_filtered(
17473 layout_dims,
17474 sc.data_addr,
17475 sc.nbytes,
17476 sc.filter_mask,
17477 )
17478 } else {
17479 DataLayoutMessage::chunked_v4_single(layout_dims, sc.data_addr)
17480 }
17481 } else if let Some(ref bt1) = m.btree_v1 {
17482 // The classic index: a version-3 layout message carrying the
17483 // address of the tree's root node, which is undefined until a
17484 // chunk is written.
17485 let mut layout_dims = bt1.chunk_dims.clone();
17486 layout_dims.push(m.datatype.element_size() as u64);
17487 DataLayoutMessage::chunked_v3_btree_v1(layout_dims, bt1.root_addr)
17488 } else if let Some(ref image) = m.compact {
17489 DataLayoutMessage::compact(image.clone())
17490 } else if let Some(ref virt) = m.virtual_storage {
17491 // Version 4 always: the virtual layout class did not exist before
17492 // it, so the default virtual layout is created at version 4 and
17493 // `H5Pset_virtual` raises any lower one to it (H5Pdcpl.c),
17494 // whatever the file's library-version bounds say — which is why a
17495 // v0-superblock file can still hold one.
17496 DataLayoutMessage::virtual_layout(4, virt.heap_addr, virt.heap_index)
17497 } else {
17498 DataLayoutMessage::contiguous(m.data_addr, m.data_size)
17499 };
17500 // `H5D__layout_oh_create` (H5Dlayout.c:530-536) marks the layout
17501 // message constant only where the storage it names is certain to be
17502 // there already: allocation time is early, the class is not compact,
17503 // no filter can change a chunk's size, and the dataspace holds at
17504 // least one element. Anything else leaves the address undefined at
17505 // creation and rewrites the message when the space is allocated, so
17506 // the flag would be a lie. `H5S_GET_EXTENT_NPOINTS` is zero for a
17507 // NULL dataspace and for any extent with a zero-length dimension.
17508 let npoints: u64 = if m.dataspace.is_null() {
17509 0
17510 } else {
17511 m.dataspace.dims.iter().product()
17512 };
17513 let filtered = m
17514 .filter_pipeline
17515 .as_ref()
17516 .is_some_and(|p| !p.filters.is_empty());
17517 let layout_flags = if alloc_time == 1 && m.compact.is_none() && !filtered && npoints != 0 {
17518 MSG_FLAG_CONSTANT
17519 } else {
17520 0x00
17521 };
17522 let layout_msg = layout.encode(&self.ctx);
17523 header.add_message(MSG_DATA_LAYOUT, layout_flags, layout_msg);
17524
17525 // Filter Pipeline message (type 0x0B) -- only if filters are
17526 // configured. `H5D__layout_oh_create` appends it with
17527 // `H5O_MSG_FLAG_CONSTANT` (H5Dlayout.c:462), as does the group
17528 // pipeline for dense links (H5Gobj.c:264): the pipeline is a creation
17529 // property, and every chunk already written was filtered through it,
17530 // so it can never be rewritten in place.
17531 if let Some(ref pipeline) = m.filter_pipeline {
17532 if !pipeline.filters.is_empty() {
17533 let (flags, filter_msg) = self.share_message(
17534 owner,
17535 MSG_FILTER_PIPELINE,
17536 MSG_FLAG_CONSTANT,
17537 pipeline.encode_for(format),
17538 );
17539 header.add_message(MSG_FILTER_PIPELINE, flags, filter_msg);
17540 }
17541 }
17542
17543 // A dataset has no links, so only attribute creation order can raise
17544 // its header past version 1 (`H5O__set_version`).
17545 let format = self.header_format(TrackOrder {
17546 links: CreationOrder::default(),
17547 attrs: m.track_attr_order,
17548 });
17549
17550 // Modification time, here and not earlier: `H5D__update_oh_info` makes
17551 // this the last message it writes (H5Dint.c:1022-1026), and the
17552 // attributes below it are added by `H5A` calls that come after the
17553 // dataset exists.
17554 touch_oh(&mut header, format, m.times, true);
17555
17556 // Attribute Info (type 0x15) + attribute messages (type 0x0C).
17557 self.emit_attributes(
17558 &mut header,
17559 AttrScope::Dataset(index),
17560 &attributes,
17561 m.track_attr_order,
17562 format,
17563 owner,
17564 );
17565
17566 self.emit_refcount(&mut header, rc, format);
17567
17568 Ok(header)
17569 }
17570
17571 /// Write the object header of every committed datatype something still
17572 /// reaches, recording the address each one landed at.
17573 ///
17574 /// Runs before the dataset and group headers because both name these
17575 /// addresses — a sharing dataset in its datatype message, the parent
17576 /// group in the link. One pass is enough: the header holds a datatype
17577 /// message and at most a reference count, neither of which depends on an
17578 /// address.
17579 fn write_committed_datatype_headers(&mut self) -> IoResult<()> {
17580 // The count is bound first: a lock guard in the `for` iterator
17581 // expression would live for the whole loop body, which locks the same
17582 // registry again.
17583 let count = self.committed_datatypes.lock().len();
17584 for i in 0..count {
17585 let rc = self.committed_datatype_refcount(i);
17586 if rc == 0 {
17587 // Its name's group was deleted and no dataset shares it, so
17588 // nothing in the file could reach the header.
17589 continue;
17590 }
17591 let format = self.committed_datatype_header_format();
17592 let encoded = self
17593 .build_committed_datatype_header(i, rc, format)
17594 .encode_for(format, rc)?;
17595 let addr = self
17596 .allocator
17597 .allocate(encoded.len() as u64, FreeSpaceClass::Metadata);
17598 self.handle.write_at(addr, &encoded)?;
17599 self.committed_datatypes.lock()[i].obj_header_addr = addr;
17600 }
17601 Ok(())
17602 }
17603
17604 /// The header format a committed datatype gets.
17605 ///
17606 /// `H5T__commit` creates the header from the datatype creation property
17607 /// list (H5Tcommit.c:468), which carries no link order and, by default, no
17608 /// attribute order — so the version is the file's floor exactly as
17609 /// `H5O__set_version` computes it, and a committed datatype in a classic
17610 /// file is a version-1 header like every other object in it.
17611 fn committed_datatype_header_format(&self) -> ObjectFormat {
17612 self.header_format(TrackOrder::default())
17613 }
17614
17615 /// Build the object header for a committed datatype: the type, and the
17616 /// reference count when more than one name reaches it.
17617 fn build_committed_datatype_header(
17618 &self,
17619 index: usize,
17620 rc: u32,
17621 format: ObjectFormat,
17622 ) -> ObjectHeader {
17623 let (datatype, times) = {
17624 let reg = self.committed_datatypes.lock();
17625 (reg[index].datatype.clone(), reg[index].times)
17626 };
17627 let mut header = ObjectHeader::new();
17628 // No attributes to emit, so nothing else would apply the file-wide
17629 // floor to this header. `store_msg_crt_idx` is a property of the file,
17630 // not of the object: every header created under it records creation
17631 // indices, a committed datatype's included.
17632 header.set_attribute_creation_order(self.header_attr_order(CreationOrder::default()));
17633 // `H5T__commit` marks the message constant and unshareable: this
17634 // header is where shared datatype bodies are read *from*, so its own
17635 // message must never become a pointer into the shared-message heap.
17636 header.add_message(
17637 MSG_DATATYPE,
17638 MSG_FLAG_CONSTANT | MSG_FLAG_DONTSHARE,
17639 datatype.encode_at(&self.ctx, self.encoding_libver()),
17640 );
17641 touch_oh(&mut header, format, times, false);
17642 // Through the same owner as every other object's count: a dataset
17643 // sharing this type raises it (`H5O__shared_link_adj`, H5Oshared.c:249)
17644 // just as a second name does, and where that count is recorded is the
17645 // header version's business, not the caller's.
17646 self.emit_refcount(&mut header, rc, format);
17647 header
17648 }
17649
17650 /// Build the object header for a subgroup.
17651 fn build_group_header(&self, group_idx: usize) -> IoResult<ObjectHeader> {
17652 let mut header = ObjectHeader::new();
17653
17654 // Link Info (type 0x02) + Group Info (type 0x0A) + the links
17655 // themselves, compact or dense.
17656 // Snapshot what the header needs, then drop the slot guard: the calls
17657 // below re-lock group slots (including this one).
17658 let (track_order, times, owner) = {
17659 let grp = self.grp(group_idx);
17660 let g = grp.lock();
17661 (
17662 g.track_order,
17663 g.times,
17664 ShareOwner::Header(g.obj_header_addr),
17665 )
17666 };
17667 let attributes = self.object_attributes(AttrScope::Group(group_idx))?;
17668 touch_oh(&mut header, self.header_format(track_order), times, false);
17669
17670 let links = self.group_links(LinkScope::Group(group_idx), track_order.links);
17671 self.emit_links(
17672 &mut header,
17673 LinkScope::Group(group_idx),
17674 &links,
17675 track_order.links,
17676 );
17677
17678 // Attribute Info (type 0x15) + attributes (type 0x0C) -- e.g. NeXus
17679 // `NX_class`.
17680 let format = self.header_format(track_order);
17681 self.emit_attributes(
17682 &mut header,
17683 AttrScope::Group(group_idx),
17684 &attributes,
17685 track_order.attrs,
17686 format,
17687 owner,
17688 );
17689
17690 self.emit_refcount(
17691 &mut header,
17692 self.object_link_count(HardLinkTarget::Group(group_idx)),
17693 format,
17694 );
17695
17696 Ok(header)
17697 }
17698
17699 fn build_root_group_header(&self) -> IoResult<ObjectHeader> {
17700 let mut header = ObjectHeader::new();
17701 touch_oh(
17702 &mut header,
17703 self.header_format(self.root_track_order),
17704 self.root_times,
17705 false,
17706 );
17707
17708 // Link Info (type 0x02) + Group Info (type 0x0A) + the links
17709 // themselves, compact or dense.
17710 let links = self.group_links(LinkScope::Root, self.root_track_order.links);
17711 self.emit_links(
17712 &mut header,
17713 LinkScope::Root,
17714 &links,
17715 self.root_track_order.links,
17716 );
17717
17718 // Root-level attributes
17719 let root_attributes = self.object_attributes(AttrScope::Root)?;
17720 self.emit_attributes(
17721 &mut header,
17722 AttrScope::Root,
17723 &root_attributes,
17724 self.root_track_order.attrs,
17725 self.header_format(self.root_track_order),
17726 ShareOwner::Header(self.root_group_addr.unwrap_or(0)),
17727 );
17728
17729 Ok(header)
17730 }
17731}
17732
17733impl Drop for Hdf5Writer {
17734 fn drop(&mut self) {
17735 if !self.closed {
17736 // Best-effort finalize on drop. Drop cannot return a Result, so a
17737 // failure here is otherwise invisible: it would leave a truncated
17738 // or unflushed file on disk while the caller believes the write
17739 // succeeded. Surface it on stderr instead of swallowing it.
17740 // Callers that need to handle the error must call
17741 // `H5File::close()` explicitly, which returns the Result.
17742 if let Err(e) = self.finalize(true) {
17743 eprintln!(
17744 "rust-hdf5: failed to finalize HDF5 file on drop: {e}. \
17745 The file may be incomplete or corrupt; call \
17746 H5File::close() to handle this error explicitly."
17747 );
17748 }
17749 }
17750 }
17751}
17752
17753#[cfg(test)]
17754mod tests {
17755 use super::*;
17756 use crate::format::messages::datatype::DatatypeMessage;
17757 use crate::io::reader::Hdf5Reader;
17758
17759 fn fixture(name: &str) -> std::path::PathBuf {
17760 std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
17761 .join("tests/fixtures")
17762 .join(name)
17763 }
17764
17765 /// Copy a fixture so a test that appends does not edit the checked-in file.
17766 fn fixture_copy(name: &str, tag: &str) -> std::path::PathBuf {
17767 let path = temp_path(tag);
17768 std::fs::copy(fixture(name), &path).unwrap();
17769 path
17770 }
17771
17772 fn temp_path(tag: &str) -> std::path::PathBuf {
17773 use std::sync::atomic::{AtomicU64, Ordering};
17774 static COUNTER: AtomicU64 = AtomicU64::new(0);
17775 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
17776 std::env::temp_dir().join(format!(
17777 "rust_hdf5_w_{}_{}_{}.h5",
17778 std::process::id(),
17779 tag,
17780 n
17781 ))
17782 }
17783
17784 /// A group past the link phase change keeps its links in a fractal heap
17785 /// with a v2 B-tree name index. The reopen that rewrites that group's
17786 /// header lays a fresh pair out, so both blocks the old header named must
17787 /// come back to the allocator — every block of the heap, and the index
17788 /// header with its nodes.
17789 ///
17790 /// Asserted on the free list rather than on the file size: a reopen does
17791 /// not yet carry dense links forward, so the rewritten group's links (and
17792 /// the datasets they name) are dropped, and the file size that follows
17793 /// says more about that than about this.
17794 #[test]
17795 fn a_reopen_frees_the_dense_link_storage_its_rewrite_supersedes() {
17796 let path = temp_path("dense_link_reclaim");
17797
17798 let writer = Hdf5Writer::create(&path).unwrap();
17799 writer.create_group("/", "run").unwrap();
17800 for i in 0..12 {
17801 writer
17802 .create_dataset(&format!("run/d{i:02}"), DatatypeMessage::i32_type(), &[2])
17803 .unwrap();
17804 }
17805 writer.close().unwrap();
17806
17807 let writer = Hdf5Writer::open_append(&path).unwrap();
17808 let gidx = (0..writer.group_count())
17809 .find(|&g| writer.grp(g).lock().name == "/run")
17810 .expect("the reopen registered the group");
17811 let linfo = writer
17812 .superseded_dense
17813 .lock()
17814 .as_ref()
17815 .and_then(|s| s.links.get(&LinkScope::Group(gidx)).cloned())
17816 .expect("the reopen recorded the group's dense link storage");
17817 assert_ne!(linfo.fractal_heap_address, UNDEF_ADDR);
17818 assert_ne!(linfo.name_btree_address, UNDEF_ADDR);
17819
17820 writer
17821 .release_superseded_dense_links(LinkScope::Group(gidx))
17822 .unwrap();
17823 let freed = writer.allocator.free_blocks();
17824 let covers = |addr: u64| {
17825 freed
17826 .iter()
17827 .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17828 };
17829 assert!(covers(linfo.fractal_heap_address), "heap header: {freed:?}");
17830 assert!(covers(linfo.name_btree_address), "name index: {freed:?}");
17831
17832 // And exactly once: the entry is gone, so the finalize that follows
17833 // cannot hand the same blocks back a second time.
17834 assert!(writer
17835 .superseded_dense
17836 .lock()
17837 .as_ref()
17838 .is_none_or(|s| s.links.is_empty()));
17839 writer
17840 .release_superseded_dense_links(LinkScope::Group(gidx))
17841 .unwrap();
17842 assert_eq!(writer.allocator.free_blocks(), freed);
17843
17844 writer.close().unwrap();
17845 std::fs::remove_file(&path).ok();
17846 }
17847
17848 /// The rewrite frees what it supersedes even when the replacement is not
17849 /// dense at all. An attribute set that drops back under `max_compact`
17850 /// goes into the object header, so nothing names the old heap any more —
17851 /// and a free driven by "the new set needs dense storage" would never
17852 /// reach this one.
17853 #[test]
17854 fn a_rewrite_that_drops_out_of_dense_storage_still_frees_it() {
17855 let path = temp_path("dense_attr_to_compact");
17856 let numeric = |name: &str| {
17857 AttributeMessage::scalar_numeric(
17858 name,
17859 DatatypeMessage::i32_type(),
17860 7i32.to_le_bytes().to_vec(),
17861 )
17862 };
17863
17864 let writer = Hdf5Writer::create(&path).unwrap();
17865 for i in 0..12 {
17866 writer
17867 .add_root_attribute(numeric(&format!("a{i:02}")))
17868 .unwrap();
17869 }
17870 writer.close().unwrap();
17871
17872 let writer = Hdf5Writer::open_append(&path).unwrap();
17873 let ainfo = writer
17874 .superseded_dense
17875 .lock()
17876 .as_ref()
17877 .and_then(|s| s.attrs.get(&AttrScope::Root).cloned())
17878 .expect("the reopen recorded the root's dense attribute storage");
17879 for i in 0..10 {
17880 writer
17881 .evict_attr(AttrTarget::Root, &format!("a{i:02}"))
17882 .unwrap();
17883 }
17884 assert!(!writer.attributes_need_dense(&writer.root_attributes.lock(), ObjectFormat::Modern));
17885
17886 writer.prepare_dense_attributes(&[]).unwrap();
17887 let freed = writer.allocator.free_blocks();
17888 let covers = |addr: u64| {
17889 freed
17890 .iter()
17891 .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17892 };
17893 assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17894 assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17895 assert!(writer
17896 .superseded_dense
17897 .lock()
17898 .as_ref()
17899 .is_none_or(|s| s.attrs.is_empty()));
17900
17901 writer.close().unwrap();
17902 std::fs::remove_file(&path).ok();
17903 }
17904
17905 /// Deleting a reopened object supersedes its dense storage as surely as
17906 /// rewriting one does: nothing in the finalized file names the heap, so
17907 /// the delete owner frees it through the same entry.
17908 #[test]
17909 fn deleting_a_reopened_group_frees_its_dense_attribute_storage() {
17910 let path = temp_path("dense_attr_delete");
17911 let numeric = |name: &str| {
17912 AttributeMessage::scalar_numeric(
17913 name,
17914 DatatypeMessage::i32_type(),
17915 7i32.to_le_bytes().to_vec(),
17916 )
17917 };
17918
17919 let writer = Hdf5Writer::create(&path).unwrap();
17920 writer.create_group("/", "run").unwrap();
17921 for i in 0..12 {
17922 writer
17923 .set_attribute(AttrTarget::Group("/run"), numeric(&format!("a{i:02}")))
17924 .unwrap();
17925 }
17926 writer.close().unwrap();
17927
17928 let writer = Hdf5Writer::open_append(&path).unwrap();
17929 let gidx = (0..writer.group_count())
17930 .find(|&g| writer.grp(g).lock().name == "/run")
17931 .expect("the reopen registered the group");
17932 let ainfo = writer
17933 .superseded_dense
17934 .lock()
17935 .as_ref()
17936 .and_then(|s| s.attrs.get(&AttrScope::Group(gidx)).cloned())
17937 .expect("the reopen recorded the group's dense attribute storage");
17938
17939 writer.delete_group("/run").unwrap();
17940 let freed = writer.allocator.free_blocks();
17941 let covers = |addr: u64| {
17942 freed
17943 .iter()
17944 .any(|&(a, len)| addr >= a && addr < a.saturating_add(len))
17945 };
17946 assert!(covers(ainfo.fractal_heap_address), "heap header: {freed:?}");
17947 assert!(covers(ainfo.name_btree_address), "name index: {freed:?}");
17948 assert!(writer
17949 .superseded_dense
17950 .lock()
17951 .as_ref()
17952 .is_none_or(|s| s.attrs.is_empty()));
17953
17954 writer.close().unwrap();
17955 std::fs::remove_file(&path).ok();
17956 }
17957
17958 /// The charset rule is one owner shared by every vlen string writer:
17959 /// appends into an ASCII-declared dataset reject non-ASCII strings the
17960 /// same way the slice writer does, and a dataset whose elements are not
17961 /// vlen references at all is refused instead of overwritten with them.
17962 #[test]
17963 fn append_vlen_strings_checks_the_datatype_and_charset() {
17964 let path = temp_path("append_vlen_charset");
17965
17966 let writer = Hdf5Writer::create(&path).unwrap();
17967 let idx = writer
17968 .create_appendable_vlen_string_dataset("d", 4, None)
17969 .unwrap();
17970 writer.ds(idx).lock().datatype = DatatypeMessage::vlen_string_ascii();
17971 let err = writer
17972 .append_vlen_strings(idx, &["ok", "안녕"])
17973 .unwrap_err();
17974 assert!(
17975 err.to_string().contains("is not ASCII"),
17976 "unexpected error: {err}"
17977 );
17978 writer.append_vlen_strings(idx, &["ok", "fine"]).unwrap();
17979
17980 let nums = writer
17981 .create_chunked_dataset("n", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
17982 .unwrap();
17983 let err = writer.append_vlen_strings(nums, &["x"]).unwrap_err();
17984 assert!(
17985 err.to_string()
17986 .contains("only for variable-length string datasets"),
17987 "unexpected error: {err}"
17988 );
17989
17990 writer.close().unwrap();
17991 std::fs::remove_file(&path).ok();
17992 }
17993
17994 /// `create_chunked_dataset` builds an extensible-array index unconditionally
17995 /// (the caller — the high-level dataset API — is the one that decides when
17996 /// two-or-more unlimited dimensions should go to a v2 B-tree instead), so
17997 /// its own guard is the last line of defense against a shape that index
17998 /// can't represent at all.
17999 #[test]
18000 fn create_chunked_dataset_rejects_two_unlimited_dimensions() {
18001 let path = temp_path("earray_two_unlimited");
18002 let writer = Hdf5Writer::create(&path).unwrap();
18003 let err = writer
18004 .create_chunked_dataset(
18005 "d",
18006 DatatypeMessage::i32_type(),
18007 &[4, 4],
18008 &[u64::MAX, u64::MAX],
18009 &[2, 2],
18010 )
18011 .unwrap_err();
18012 assert!(err.to_string().contains("at most one unlimited"), "{err}");
18013 writer.close().unwrap();
18014 std::fs::remove_file(&path).ok();
18015 }
18016
18017 /// Every creator must enter through `begin_create`; the four that used
18018 /// to bypass it could push a second dataset under an existing name and
18019 /// emit an invalid file with two same-named links.
18020 #[test]
18021 fn every_creator_rejects_an_existing_dataset_name() {
18022 let path = temp_path("create_gate");
18023
18024 let writer = Hdf5Writer::create(&path).unwrap();
18025 writer
18026 .create_dataset("d", DatatypeMessage::i32_type(), &[2])
18027 .unwrap();
18028
18029 let attempts: [(&str, IoResult<usize>); 4] = [
18030 (
18031 "vlen_string",
18032 writer.create_vlen_string_dataset("d", &["x"], 1),
18033 ),
18034 ("vlen_bytes", writer.create_vlen_bytes_dataset("d", &[b"x"])),
18035 (
18036 "vlen_string_compressed",
18037 writer.create_vlen_string_dataset_compressed(
18038 "d",
18039 &["x"],
18040 1,
18041 FilterPipeline::deflate(6),
18042 ),
18043 ),
18044 (
18045 "chunked_with_pipeline",
18046 writer.create_chunked_dataset_with_pipeline(
18047 "d",
18048 DatatypeMessage::i32_type(),
18049 &[0],
18050 &[u64::MAX],
18051 &[4],
18052 FilterPipeline::deflate(6),
18053 ),
18054 ),
18055 ];
18056 for (which, res) in attempts {
18057 match res {
18058 Ok(_) => panic!("{which} accepted a duplicate name"),
18059 Err(e) => assert!(
18060 e.to_string().contains("already exists"),
18061 "{which}: unexpected error: {e}"
18062 ),
18063 }
18064 }
18065
18066 writer.close().unwrap();
18067 std::fs::remove_file(&path).ok();
18068 }
18069
18070 /// Every creator and every kind of name meet at `ensure_name_free`.
18071 ///
18072 /// The gate's whole value is that it is one list: a creator must be
18073 /// blind neither to a name kind it does not itself make nor to one added
18074 /// after it. This crosses the two — six names, one of each kind the
18075 /// writer can put in a group, against every creator — so a creator that
18076 /// grows its own check, or a name kind that stops being on the list,
18077 /// fails here rather than in a file holding two links of one name.
18078 #[test]
18079 fn every_creator_refuses_every_kind_of_taken_name() {
18080 let path = temp_path("create_gate_matrix");
18081 let writer = Hdf5Writer::create(&path).unwrap();
18082
18083 let i32t = || DatatypeMessage::i32_type();
18084 writer.create_dataset("d", i32t(), &[2]).unwrap();
18085 writer.create_compact_dataset("c", i32t(), &[2]).unwrap();
18086 writer.create_group("/", "g").unwrap();
18087 writer.commit_datatype("t", i32t()).unwrap();
18088 writer.create_hard_link("/", "h", "d").unwrap();
18089 writer
18090 .create_symbolic_link(
18091 "/",
18092 "s",
18093 LinkTarget::Soft {
18094 target: "/d".into(),
18095 },
18096 )
18097 .unwrap();
18098 writer
18099 .create_symbolic_link(
18100 "/",
18101 "e",
18102 LinkTarget::External {
18103 file: "other.h5".into(),
18104 path: "/x".into(),
18105 },
18106 )
18107 .unwrap();
18108
18109 for taken in ["d", "c", "g", "t", "h", "s", "e"] {
18110 let attempts: [(&str, IoResult<()>); 8] = [
18111 (
18112 "dataset",
18113 writer.create_dataset(taken, i32t(), &[2]).map(|_| ()),
18114 ),
18115 (
18116 "compact",
18117 writer
18118 .create_compact_dataset(taken, i32t(), &[2])
18119 .map(|_| ()),
18120 ),
18121 (
18122 "chunked",
18123 writer
18124 .create_chunked_dataset(taken, i32t(), &[0], &[u64::MAX], &[4])
18125 .map(|_| ()),
18126 ),
18127 (
18128 "vlen_string",
18129 writer
18130 .create_vlen_string_dataset(taken, &["x"], 1)
18131 .map(|_| ()),
18132 ),
18133 (
18134 "committed datatype",
18135 writer.commit_datatype(taken, i32t()).map(|_| ()),
18136 ),
18137 ("group", writer.create_group("/", taken).map(|_| ())),
18138 ("hard link", writer.create_hard_link("/", taken, "d")),
18139 (
18140 "soft link",
18141 writer.create_symbolic_link(
18142 "/",
18143 taken,
18144 LinkTarget::Soft {
18145 target: "/d".into(),
18146 },
18147 ),
18148 ),
18149 ];
18150 for (which, res) in attempts {
18151 match res {
18152 Ok(()) => panic!("{which} accepted the taken name '{taken}'"),
18153 Err(e) => assert!(
18154 e.to_string().contains("already exists"),
18155 "{which} on '{taken}': unexpected error: {e}"
18156 ),
18157 }
18158 }
18159 }
18160
18161 writer.close().unwrap();
18162 std::fs::remove_file(&path).ok();
18163 }
18164
18165 /// The `H5T_VLEN` length field counts base elements, so an image that is
18166 /// not a whole number of them has no length that reads back as what was
18167 /// handed over; it is refused at the call rather than stored truncated.
18168 #[test]
18169 fn vlen_sequence_refuses_a_partial_element() {
18170 let path = temp_path("vlen_partial_element");
18171
18172 let writer = Hdf5Writer::create(&path).unwrap();
18173 let err = writer
18174 .create_vlen_sequence_dataset("d", DatatypeMessage::i32_type(), &[&[1u8, 2, 3, 4, 5]])
18175 .unwrap_err()
18176 .to_string();
18177 assert!(err.contains("5 bytes"), "unexpected error: {err}");
18178 assert!(err.contains("4-byte elements"), "unexpected error: {err}");
18179
18180 // The refusal is the length rule alone: the same base takes a whole
18181 // number of elements, and an empty sequence is a legal one.
18182 writer
18183 .create_vlen_sequence_dataset(
18184 "d",
18185 DatatypeMessage::i32_type(),
18186 &[&[1u8, 2, 3, 4], &[][..]],
18187 )
18188 .unwrap();
18189
18190 writer.close().unwrap();
18191 std::fs::remove_file(&path).ok();
18192 }
18193
18194 /// A corrupt file can declare a zero-length chunk dimension; the
18195 /// superseded-reference read must reject it the way `write_slice` does,
18196 /// not divide by it.
18197 #[test]
18198 fn vlen_slice_rejects_a_zero_chunk_dimension() {
18199 let path = temp_path("vlen_slice_zero_chunk");
18200
18201 let writer = Hdf5Writer::create(&path).unwrap();
18202 let idx = writer
18203 .create_appendable_vlen_string_dataset("d", 2, None)
18204 .unwrap();
18205 writer.append_vlen_strings(idx, &["a", "b"]).unwrap();
18206 writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 0;
18207 let err = writer.write_vlen_strings_slice(idx, 0, &["x"]).unwrap_err();
18208 assert!(
18209 err.to_string().contains("zero-length dimension"),
18210 "unexpected error: {err}"
18211 );
18212
18213 writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims[0] = 2;
18214 writer.close().unwrap();
18215 std::fs::remove_file(&path).ok();
18216 }
18217
18218 /// A libhdf5-written collection can be 100% full — no free-space marker,
18219 /// content exactly the declared size. When a stale reference names an
18220 /// index that is not there, nothing is removed, and the collection must
18221 /// be left alone: re-encoding it at its declared size cannot fit the
18222 /// free-space marker and would fail the whole update.
18223 #[test]
18224 fn release_leaves_a_full_collection_it_removed_nothing_from() {
18225 use crate::format::global_heap::encode_vlen_reference;
18226
18227 let path = temp_path("release_full_gcol");
18228 let writer = Hdf5Writer::create(&path).unwrap();
18229
18230 // Hand-built full collection: 16-byte header + one 16+8-byte object,
18231 // declared size exactly 40, no free-space marker.
18232 let mut img = Vec::new();
18233 img.extend_from_slice(b"GCOL");
18234 img.push(1);
18235 img.extend_from_slice(&[0u8; 3]);
18236 img.extend_from_slice(&40u64.to_le_bytes());
18237 img.extend_from_slice(&1u16.to_le_bytes()); // object index 1
18238 img.extend_from_slice(&1u16.to_le_bytes()); // ref_count
18239 img.extend_from_slice(&0u32.to_le_bytes()); // reserved
18240 img.extend_from_slice(&8u64.to_le_bytes()); // data size
18241 img.extend_from_slice(b"deadbeef");
18242 assert_eq!(img.len(), 40);
18243 let addr = writer
18244 .allocator
18245 .allocate(img.len() as u64, FreeSpaceClass::RawData);
18246 writer.handle.write_at(addr, &img).unwrap();
18247
18248 // The superseded reference names index 2, which the collection does
18249 // not hold — a no-op removal.
18250 let refs = encode_vlen_reference(3, addr, 2, &writer.ctx);
18251 writer.release_vlen_references(&refs).unwrap();
18252 assert_eq!(writer.handle.read_at(addr, 40).unwrap(), img);
18253
18254 writer.close().unwrap();
18255 std::fs::remove_file(&path).ok();
18256 }
18257
18258 /// The CWFS second pass (`H5F_cwfs_find_free_heap`): an object too big
18259 /// for the listed collection's remaining free space extends the
18260 /// collection in place — the file allocation grows off the end of the
18261 /// file (`H5MF_try_extend`) and the collection's declared size and
18262 /// free-space marker grow with it (`H5HG_extend`) — instead of opening
18263 /// a second collection.
18264 #[test]
18265 fn an_oversized_vlen_insert_extends_the_listed_collection() {
18266 use crate::format::global_heap::GlobalHeapCollection;
18267
18268 let path = temp_path("cwfs_extend_tail");
18269 let writer = Hdf5Writer::create(&path).unwrap();
18270 // A small object opens a minimum-size (4096) listed collection —
18271 // the file's last allocation, so the extension grows the file end.
18272 let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
18273 let big = vec![0x41u8; 5000]; // more than the ~4 KiB remaining
18274 let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
18275 assert_eq!(
18276 p2[0].0, p1[0].0,
18277 "the big object opened a second collection"
18278 );
18279
18280 // The block on disk is one grown collection holding both objects.
18281 let img = writer.handle.read_at_most(p1[0].0, 65536).unwrap();
18282 let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
18283 assert!(csize > 4096, "declared size did not grow: {csize}");
18284 assert_eq!(gcol.objects.len(), 2);
18285 assert_eq!(gcol.objects[1].data, big);
18286
18287 writer.close().unwrap();
18288 let bytes = std::fs::read(&path).unwrap();
18289 assert_eq!(
18290 bytes.windows(4).filter(|w| *w == b"GCOL").count(),
18291 1,
18292 "a second collection signature is in the file"
18293 );
18294 std::fs::remove_file(&path).ok();
18295 }
18296
18297 /// The non-tail counterpart: the collection is pinned away from the end
18298 /// of the file, but a released block starts right after it, so the
18299 /// extension consumes the front of that block (`H5MF_try_extend`'s
18300 /// free-section path) and the remainder stays reusable.
18301 #[test]
18302 fn extension_consumes_a_freed_block_after_the_collection() {
18303 use crate::format::global_heap::GlobalHeapCollection;
18304
18305 let path = temp_path("cwfs_extend_freed");
18306 let writer = Hdf5Writer::create(&path).unwrap();
18307 let p1 = writer.insert_vlen_objects(&[b"hello".as_slice()]).unwrap();
18308 let addr = p1[0].0;
18309 // Land a block right after the collection, pin the file end past
18310 // it, then release it: extension must use the released space.
18311 let spacer = writer.allocator.allocate(8192, FreeSpaceClass::RawData);
18312 assert_eq!(spacer, addr + 4096, "spacer not adjacent; layout changed");
18313 writer.allocator.allocate(8, FreeSpaceClass::RawData);
18314 writer.allocator.free(spacer, 8192, FreeSpaceClass::RawData);
18315
18316 let big = vec![0x42u8; 5000];
18317 let p2 = writer.insert_vlen_objects(&[big.as_slice()]).unwrap();
18318 assert_eq!(p2[0].0, addr, "the big object opened a second collection");
18319
18320 let img = writer.handle.read_at_most(addr, 65536).unwrap();
18321 let (gcol, csize) = GlobalHeapCollection::decode(&img, &writer.ctx).unwrap();
18322 assert_eq!(csize, 8192, "grew by max(size, shortfall) = 4096");
18323 assert_eq!(gcol.objects.len(), 2);
18324
18325 // The remainder of the released block is still allocatable.
18326 assert_eq!(
18327 writer.allocator.allocate(4096, FreeSpaceClass::RawData),
18328 addr + 8192,
18329 "the freed block's tail was lost"
18330 );
18331 writer.close().unwrap();
18332 std::fs::remove_file(&path).ok();
18333 }
18334
18335 /// Issue #10: a reopen-and-replace loop on a vlen string must not grow
18336 /// the file. The superseded heap objects are freed *before* the
18337 /// replacement is allocated, so each session reuses the block it just
18338 /// released even though the free list starts empty on reopen. The old
18339 /// free-after-alloc order failed this by one collection per session.
18340 #[test]
18341 fn vlen_replace_across_reopen_keeps_the_file_flat() {
18342 let path = temp_path("vlen_reopen_flat");
18343 let payload_a = "a".repeat(64 * 1024);
18344 let payload_b = "b".repeat(64 * 1024);
18345
18346 let writer = Hdf5Writer::create(&path).unwrap();
18347 writer
18348 .create_vlen_string_dataset("notes", &["initial"], 1)
18349 .unwrap();
18350 writer.close().unwrap();
18351
18352 let mut sizes = Vec::new();
18353 for i in 0..8 {
18354 let writer = Hdf5Writer::open_append(&path).unwrap();
18355 let payload = if i % 2 == 0 { &payload_a } else { &payload_b };
18356 writer
18357 .write_vlen_strings_slice(0, 0, &[payload.as_str()])
18358 .unwrap();
18359 writer.close().unwrap();
18360 sizes.push(std::fs::metadata(&path).unwrap().len());
18361 }
18362 // The first replacement grows the file once (the initial collection
18363 // cannot hold 64 KiB); every later equal-size replacement must land
18364 // in the block its own session just freed.
18365 assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
18366
18367 // The reused blocks still form a valid file holding the last value.
18368 let mut reader = Hdf5Reader::open(&path).unwrap();
18369 assert_eq!(
18370 reader.read_vlen_strings("notes").unwrap(),
18371 vec![payload_b.clone()]
18372 );
18373
18374 std::fs::remove_file(&path).ok();
18375 }
18376
18377 /// Replacing a vlen string attribute must release the superseded
18378 /// global-heap collection *before* the replacement's collection is
18379 /// allocated, so a reopen-replace loop lands each new value in the block
18380 /// it just freed instead of growing the file by one collection per
18381 /// session — the attribute counterpart of
18382 /// [`vlen_replace_across_reopen_keeps_the_file_flat`].
18383 #[test]
18384 fn vlen_attr_replace_across_reopen_keeps_the_file_flat() {
18385 let path = temp_path("vlen_attr_reopen_flat");
18386 let payload_a = "a".repeat(8 * 1024);
18387 let payload_b = "b".repeat(8 * 1024);
18388
18389 let writer = Hdf5Writer::create(&path).unwrap();
18390 writer
18391 .set_vlen_string_attribute(AttrTarget::Root, "note", &payload_a)
18392 .unwrap();
18393 writer.close().unwrap();
18394
18395 let mut sizes = Vec::new();
18396 for i in 0..8 {
18397 let writer = Hdf5Writer::open_append(&path).unwrap();
18398 let payload = if i % 2 == 0 { &payload_b } else { &payload_a };
18399 writer
18400 .set_vlen_string_attribute(AttrTarget::Root, "note", payload)
18401 .unwrap();
18402 writer.close().unwrap();
18403 sizes.push(std::fs::metadata(&path).unwrap().len());
18404 }
18405 assert_eq!(&sizes[1..], &vec![sizes[0]; 7][..], "sizes: {sizes:?}");
18406
18407 // The reused blocks still hold the last value.
18408 let reader = Hdf5Reader::open(&path).unwrap();
18409 let attr = reader.root_attr("note").unwrap().clone();
18410 let mut reader = reader;
18411 assert_eq!(reader.attr_string_value(&attr).unwrap(), payload_a);
18412
18413 std::fs::remove_file(&path).ok();
18414 }
18415
18416 /// A numeric attribute replacing a vlen one goes through the same list
18417 /// owner, so the superseded collection is released even though the new
18418 /// value holds no heap reference: a later same-size vlen attribute must
18419 /// land in the freed block, making the file exactly as large as one that
18420 /// never stored the replaced value.
18421 #[test]
18422 fn numeric_replacing_a_vlen_attr_releases_its_collection() {
18423 let payload = "x".repeat(8 * 1024);
18424 let numeric = || {
18425 AttributeMessage::scalar_numeric(
18426 "x",
18427 DatatypeMessage::i32_type(),
18428 7i32.to_le_bytes().to_vec(),
18429 )
18430 };
18431
18432 let path_a = temp_path("vlen_attr_cross_a");
18433 let writer = Hdf5Writer::create(&path_a).unwrap();
18434 writer
18435 .set_vlen_string_attribute(AttrTarget::Root, "x", &payload)
18436 .unwrap();
18437 writer.add_root_attribute(numeric()).unwrap();
18438 writer
18439 .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
18440 .unwrap();
18441 writer.close().unwrap();
18442
18443 // The same end state written without the replaced vlen value.
18444 let path_b = temp_path("vlen_attr_cross_b");
18445 let writer = Hdf5Writer::create(&path_b).unwrap();
18446 writer.add_root_attribute(numeric()).unwrap();
18447 writer
18448 .set_vlen_string_attribute(AttrTarget::Root, "y", &payload)
18449 .unwrap();
18450 writer.close().unwrap();
18451
18452 assert_eq!(
18453 std::fs::metadata(&path_a).unwrap().len(),
18454 std::fs::metadata(&path_b).unwrap().len()
18455 );
18456
18457 let reader = Hdf5Reader::open(&path_a).unwrap();
18458 let y = reader.root_attr("y").unwrap().clone();
18459 let mut reader = reader;
18460 assert_eq!(reader.attr_string_value(&y).unwrap(), payload);
18461
18462 std::fs::remove_file(&path_a).ok();
18463 std::fs::remove_file(&path_b).ok();
18464 }
18465
18466 /// Reopen/write/close cycles must not leak the object-header blocks
18467 /// finalize rewrites: the reopened root header, the reopened group
18468 /// header, and the modified chunked dataset's header are each freed
18469 /// before their replacements are allocated. The chunk rewrite itself is
18470 /// in place (unfiltered chunks never move), so a leak of any header
18471 /// block shows up as monotonic growth here.
18472 #[test]
18473 fn reopen_cycles_reuse_superseded_header_blocks() {
18474 let path = temp_path("header_reuse");
18475 {
18476 let writer = Hdf5Writer::create(&path).unwrap();
18477 writer.create_group("/", "g").unwrap();
18478 let idx = writer
18479 .create_chunked_dataset(
18480 "g/data",
18481 DatatypeMessage::i32_type(),
18482 &[4],
18483 &[u64::MAX],
18484 &[4],
18485 )
18486 .unwrap();
18487 let seed: Vec<u8> = [1i32, 2, 3, 4]
18488 .iter()
18489 .flat_map(|v| v.to_le_bytes())
18490 .collect();
18491 writer.write_chunk(idx, 0, &seed).unwrap();
18492 writer.close().unwrap();
18493 }
18494
18495 let mut sizes = Vec::new();
18496 for i in 0..6i32 {
18497 let writer = Hdf5Writer::open_append(&path).unwrap();
18498 let data: Vec<u8> = [i; 4].iter().flat_map(|v| v.to_le_bytes()).collect();
18499 writer.write_chunk(0, 0, &data).unwrap();
18500 writer.close().unwrap();
18501 sizes.push(std::fs::metadata(&path).unwrap().len());
18502 }
18503 assert_eq!(&sizes[1..], &vec![sizes[0]; 5][..], "sizes: {sizes:?}");
18504
18505 // The reused header blocks still form a valid file.
18506 let mut reader = Hdf5Reader::open(&path).unwrap();
18507 let raw = reader.read_dataset_raw("g/data").unwrap();
18508 let values: Vec<i32> = raw
18509 .chunks(4)
18510 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18511 .collect();
18512 assert_eq!(values, vec![5, 5, 5, 5]);
18513
18514 std::fs::remove_file(&path).ok();
18515 }
18516
18517 #[test]
18518 fn create_empty_file() {
18519 let path = temp_path("empty");
18520
18521 let writer = Hdf5Writer::create(&path).unwrap();
18522 writer.close().unwrap();
18523
18524 // Verify we can read it back
18525 let reader = Hdf5Reader::open(&path).unwrap();
18526 assert!(reader.dataset_names().is_empty());
18527
18528 std::fs::remove_file(&path).ok();
18529 }
18530
18531 #[test]
18532 fn create_single_dataset() {
18533 let path = temp_path("single");
18534
18535 let writer = Hdf5Writer::create(&path).unwrap();
18536 let idx = writer
18537 .create_dataset("data", DatatypeMessage::f64_type(), &[4])
18538 .unwrap();
18539 let values: Vec<f64> = vec![1.0, 2.0, 3.0, 4.0];
18540 let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18541 writer.write_dataset_raw(idx, &raw).unwrap();
18542 writer.close().unwrap();
18543
18544 // Read back
18545 let mut reader = Hdf5Reader::open(&path).unwrap();
18546 assert_eq!(reader.dataset_names(), vec!["data"]);
18547 assert_eq!(reader.dataset_shape("data").unwrap(), vec![4]);
18548 let readback = reader.read_dataset_raw("data").unwrap();
18549 assert_eq!(readback, raw);
18550
18551 std::fs::remove_file(&path).ok();
18552 }
18553
18554 #[test]
18555 fn create_multiple_datasets() {
18556 let path = temp_path("multi");
18557
18558 let writer = Hdf5Writer::create(&path).unwrap();
18559
18560 let idx0 = writer
18561 .create_dataset("ints", DatatypeMessage::i32_type(), &[3])
18562 .unwrap();
18563 let i_data: Vec<u8> = [10i32, 20, 30]
18564 .iter()
18565 .flat_map(|v| v.to_le_bytes())
18566 .collect();
18567 writer.write_dataset_raw(idx0, &i_data).unwrap();
18568
18569 let idx1 = writer
18570 .create_dataset("floats", DatatypeMessage::f32_type(), &[2, 2])
18571 .unwrap();
18572 let f_data: Vec<u8> = [1.0f32, 2.0, 3.0, 4.0]
18573 .iter()
18574 .flat_map(|v| v.to_le_bytes())
18575 .collect();
18576 writer.write_dataset_raw(idx1, &f_data).unwrap();
18577
18578 writer.close().unwrap();
18579
18580 let mut reader = Hdf5Reader::open(&path).unwrap();
18581 let names = reader.dataset_names();
18582 assert!(names.contains(&"ints"));
18583 assert!(names.contains(&"floats"));
18584 assert_eq!(reader.dataset_shape("ints").unwrap(), vec![3]);
18585 assert_eq!(reader.dataset_shape("floats").unwrap(), vec![2, 2]);
18586 assert_eq!(reader.read_dataset_raw("ints").unwrap(), i_data);
18587 assert_eq!(reader.read_dataset_raw("floats").unwrap(), f_data);
18588
18589 std::fs::remove_file(&path).ok();
18590 }
18591
18592 #[test]
18593 fn data_size_mismatch() {
18594 let path = temp_path("mismatch");
18595
18596 let writer = Hdf5Writer::create(&path).unwrap();
18597 let idx = writer
18598 .create_dataset("x", DatatypeMessage::u8_type(), &[4])
18599 .unwrap();
18600 let err = writer.write_dataset_raw(idx, &[1, 2, 3]); // 3 bytes instead of 4
18601 assert!(err.is_err());
18602
18603 std::fs::remove_file(&path).ok();
18604 }
18605
18606 #[test]
18607 fn create_chunked_dataset_simple() {
18608 let path = temp_path("chunked_simple");
18609
18610 let writer = Hdf5Writer::create(&path).unwrap();
18611 let idx = writer
18612 .create_chunked_dataset(
18613 "data",
18614 DatatypeMessage::f64_type(),
18615 &[0, 4], // start empty
18616 &[u64::MAX, 4], // unlimited first dim
18617 &[1, 4], // chunk = [1, 4]
18618 )
18619 .unwrap();
18620
18621 // Write 3 frames (chunks)
18622 for frame in 0..3u64 {
18623 let values: Vec<f64> = (0..4).map(|i| (frame * 4 + i) as f64).collect();
18624 let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18625 writer.write_chunk(idx, frame, &raw).unwrap();
18626 }
18627
18628 // Extend dimensions
18629 writer.extend_dataset(idx, &[3, 4]).unwrap();
18630
18631 writer.close().unwrap();
18632
18633 // Read back
18634 let mut reader = Hdf5Reader::open(&path).unwrap();
18635 assert_eq!(reader.dataset_names(), vec!["data"]);
18636 assert_eq!(reader.dataset_shape("data").unwrap(), vec![3, 4]);
18637
18638 let raw = reader.read_dataset_raw("data").unwrap();
18639 let values: Vec<f64> = raw
18640 .chunks(8)
18641 .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
18642 .collect();
18643 assert_eq!(values.len(), 12);
18644 for (i, val) in values.iter().enumerate() {
18645 assert_eq!(*val, i as f64);
18646 }
18647
18648 std::fs::remove_file(&path).ok();
18649 }
18650
18651 #[test]
18652 fn chunked_dataset_many_frames() {
18653 let path = temp_path("chunked_many");
18654
18655 let writer = Hdf5Writer::create(&path).unwrap();
18656 let idx = writer
18657 .create_chunked_dataset(
18658 "frames",
18659 DatatypeMessage::i32_type(),
18660 &[0, 2],
18661 &[u64::MAX, 2],
18662 &[1, 2],
18663 )
18664 .unwrap();
18665
18666 let n_frames = 10u64;
18667 for frame in 0..n_frames {
18668 let values = [(frame * 2) as i32, (frame * 2 + 1) as i32];
18669 let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
18670 writer.write_chunk(idx, frame, &raw).unwrap();
18671 }
18672
18673 writer.extend_dataset(idx, &[n_frames, 2]).unwrap();
18674 writer.close().unwrap();
18675
18676 // Read back
18677 let mut reader = Hdf5Reader::open(&path).unwrap();
18678 assert_eq!(reader.dataset_shape("frames").unwrap(), vec![10, 2]);
18679
18680 let raw = reader.read_dataset_raw("frames").unwrap();
18681 let values: Vec<i32> = raw
18682 .chunks(4)
18683 .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18684 .collect();
18685 assert_eq!(values.len(), 20);
18686 for (i, val) in values.iter().enumerate() {
18687 assert_eq!(*val, i as i32);
18688 }
18689
18690 std::fs::remove_file(&path).ok();
18691 }
18692
18693 #[test]
18694 fn create_fixed_array_dataset_roundtrip() {
18695 let path = temp_path("fixed_array");
18696
18697 let writer = Hdf5Writer::create(&path).unwrap();
18698 let idx = writer
18699 .create_fixed_array_dataset(
18700 "grid",
18701 DatatypeMessage::i32_type(),
18702 &[4, 6], // 4x6 grid
18703 &[2, 3], // chunk = 2x3
18704 )
18705 .unwrap();
18706
18707 // Write all chunks: 2x2 = 4 chunks
18708 // chunk (0,0): rows 0-1, cols 0-2
18709 let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18710 .iter()
18711 .flat_map(|v| v.to_le_bytes())
18712 .collect();
18713 writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18714
18715 // chunk (0,1): rows 0-1, cols 3-5
18716 let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18717 .iter()
18718 .flat_map(|v| v.to_le_bytes())
18719 .collect();
18720 writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
18721
18722 // chunk (1,0): rows 2-3, cols 0-2
18723 let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
18724 .iter()
18725 .flat_map(|v| v.to_le_bytes())
18726 .collect();
18727 writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
18728
18729 // chunk (1,1): rows 2-3, cols 3-5
18730 let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
18731 .iter()
18732 .flat_map(|v| v.to_le_bytes())
18733 .collect();
18734 writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
18735
18736 writer.close().unwrap();
18737
18738 // Read back
18739 let mut reader = Hdf5Reader::open(&path).unwrap();
18740 assert_eq!(reader.dataset_names(), vec!["grid"]);
18741 assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18742
18743 let raw = reader.read_dataset_raw("grid").unwrap();
18744 let values: Vec<i32> = raw
18745 .chunks(4)
18746 .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
18747 .collect();
18748 assert_eq!(values.len(), 24);
18749 for (i, val) in values.iter().enumerate() {
18750 assert_eq!(*val, i as i32);
18751 }
18752
18753 std::fs::remove_file(&path).ok();
18754 }
18755
18756 #[test]
18757 fn fixed_array_paged_dblk_disk_size() {
18758 let ctx = FormatContext {
18759 sizeof_addr: 8,
18760 sizeof_size: 8,
18761 };
18762 // 1024 elements per page (bits=10). 3000 chunks => 3 pages.
18763 let hdr = FixedArrayHeader::new_for_chunks(&ctx, 3000);
18764 assert!(hdr.is_paged());
18765 assert_eq!(hdr.npages(), 3);
18766 // prefix: 4+1+1+8 + bitmap(1) + cksum(4) = 19
18767 // elements: 3000 * 8 = 24000 ; per-page cksum: 3 * 4 = 12
18768 assert_eq!(fixed_array_dblk_disk_size(&ctx, &hdr), 19 + 24000 + 12);
18769
18770 // Non-paged: 1000 elements. prefix(14) + 1000*8 + cksum(4).
18771 let small = FixedArrayHeader::new_for_chunks(&ctx, 1000);
18772 assert!(!small.is_paged());
18773 assert_eq!(fixed_array_dblk_disk_size(&ctx, &small), 14 + 8000 + 4);
18774 }
18775
18776 #[test]
18777 fn fixed_array_paged_encode_matches_reader_layout() {
18778 let ctx = FormatContext {
18779 sizeof_addr: 8,
18780 sizeof_size: 8,
18781 };
18782 let mut hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18783 hdr.data_blk_addr = 0x9000;
18784 let npages = hdr.npages() as usize; // ceil(2500/1024) = 3
18785
18786 let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18787 for (i, e) in dblk.elements.iter_mut().enumerate() {
18788 *e = 0x10000 + (i as u64) * 0x100;
18789 }
18790
18791 let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18792 assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18793
18794 // Decode the prefix and pages exactly as the reader does.
18795 let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18796 assert_eq!(prefix.header_addr, 0x1000);
18797 for p in 0..npages {
18798 assert!(prefix.page_initialized(p), "page {p} should be initialized");
18799 }
18800
18801 let dblk_page_nelmts = hdr.dblk_page_nelmts() as usize;
18802 let page_stride = dblk_page_nelmts * 8 + 4;
18803 let mut recovered = Vec::new();
18804 for p in 0..npages {
18805 let page_nelmts = if p + 1 == npages {
18806 2500 - p * dblk_page_nelmts
18807 } else {
18808 dblk_page_nelmts
18809 };
18810 let off = prefix.prefix_size + p * page_stride;
18811 let page_buf = &encoded[off..];
18812 let addrs = crate::format::chunk_index::fixed_array::decode_unfiltered_page(
18813 page_buf,
18814 &ctx,
18815 page_nelmts,
18816 )
18817 .unwrap();
18818 recovered.extend(addrs);
18819 }
18820 assert_eq!(recovered, dblk.elements);
18821 }
18822
18823 #[test]
18824 fn fixed_array_paged_decode_roundtrip_with_uninitialized_page() {
18825 let ctx = FormatContext {
18826 sizeof_addr: 8,
18827 sizeof_size: 8,
18828 };
18829 let hdr = FixedArrayHeader::new_for_chunks(&ctx, 2500);
18830 let npages = hdr.npages() as usize; // 3
18831 let page = hdr.dblk_page_nelmts() as usize; // 1024
18832
18833 // Populate pages 0 and 2; leave page 1 entirely undefined so its
18834 // bitmap bit stays clear on encode.
18835 let mut dblk = FixedArrayDataBlock::new_unfiltered(0x1000, 2500);
18836 for i in (0..page).chain(2 * page..2500) {
18837 dblk.elements[i] = 0x10000 + (i as u64) * 0x100;
18838 }
18839
18840 let mut encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18841 let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
18842 assert!(prefix.page_initialized(0));
18843 assert!(!prefix.page_initialized(1));
18844 assert!(prefix.page_initialized(2));
18845
18846 // Corrupt the uninitialized page's bytes the way libhdf5 leaves
18847 // them: arbitrary, no valid checksum. Decode must not look at it.
18848 let page_stride = page * 8 + 4;
18849 let p1 = prefix.prefix_size + page_stride;
18850 for b in &mut encoded[p1..p1 + page_stride] {
18851 *b = 0x5A;
18852 }
18853
18854 let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, 0).unwrap();
18855 assert_eq!(decoded.elements, dblk.elements);
18856 assert_eq!(decoded.header_addr, 0x1000);
18857 }
18858
18859 #[test]
18860 fn fixed_array_paged_decode_filtered_roundtrip() {
18861 let ctx = FormatContext {
18862 sizeof_addr: 8,
18863 sizeof_size: 8,
18864 };
18865 let chunk_size_len = 4usize;
18866 let hdr = FixedArrayHeader::new_for_filtered_chunks(&ctx, 1500, chunk_size_len as u8);
18867 assert!(hdr.is_paged());
18868
18869 let mut dblk = FixedArrayDataBlock::new_filtered(0x2000, 1500);
18870 for (i, e) in dblk.filtered_elements.iter_mut().enumerate() {
18871 e.address = 0x8000 + (i as u64) * 0x40;
18872 e.chunk_size = 100 + i as u64;
18873 e.filter_mask = (i % 3) as u32;
18874 }
18875
18876 let encoded = encode_fixed_array_dblk(&ctx, &hdr, &dblk);
18877 assert_eq!(encoded.len() as u64, fixed_array_dblk_disk_size(&ctx, &hdr));
18878 let decoded = decode_fixed_array_dblk(&ctx, &hdr, &encoded, chunk_size_len).unwrap();
18879 assert_eq!(decoded.filtered_elements, dblk.filtered_elements);
18880 assert_eq!(decoded.client_id, FA_CLIENT_FILT_CHUNK);
18881 }
18882
18883 #[test]
18884 fn create_fixed_array_paged_dataset_roundtrip() {
18885 let path = temp_path("fixed_array_paged");
18886
18887 // 1D dataset of 3000 elements, chunk size 1 => 3000 chunks.
18888 // 3000 > 1024 (one page) => the FA data block must be paged.
18889 let n: usize = 3000;
18890 let writer = Hdf5Writer::create(&path).unwrap();
18891 let idx = writer
18892 .create_fixed_array_dataset("paged", DatatypeMessage::i32_type(), &[n as u64], &[1])
18893 .unwrap();
18894
18895 for i in 0..n {
18896 let v = (i as i32).to_le_bytes();
18897 writer
18898 .write_chunk_fixed_array(idx, &[i as u64], &v)
18899 .unwrap();
18900 }
18901 writer.close().unwrap();
18902
18903 let mut reader = Hdf5Reader::open(&path).unwrap();
18904 assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
18905 let raw = reader.read_dataset_raw("paged").unwrap();
18906 let values: Vec<i32> = raw
18907 .chunks(4)
18908 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18909 .collect();
18910 assert_eq!(values.len(), n);
18911 for (i, v) in values.iter().enumerate() {
18912 assert_eq!(*v, i as i32, "element {i}");
18913 }
18914
18915 std::fs::remove_file(&path).ok();
18916 }
18917
18918 #[cfg(feature = "deflate")]
18919 #[test]
18920 fn create_filtered_fixed_array_dataset_roundtrip() {
18921 // Small compressed fixed-shape chunked dataset: flat filtered FA.
18922 let path = temp_path("fixed_array_filt");
18923
18924 let writer = Hdf5Writer::create(&path).unwrap();
18925 let idx = writer
18926 .create_fixed_array_dataset_with_pipeline(
18927 "grid",
18928 DatatypeMessage::i32_type(),
18929 &[4, 6], // 4x6 grid
18930 &[2, 3], // chunk = 2x3 => 2x2 = 4 chunks
18931 FilterPipeline::deflate(6),
18932 )
18933 .unwrap();
18934
18935 let c00: Vec<u8> = [0i32, 1, 2, 6, 7, 8]
18936 .iter()
18937 .flat_map(|v| v.to_le_bytes())
18938 .collect();
18939 writer.write_chunk_fixed_array(idx, &[0, 0], &c00).unwrap();
18940 let c01: Vec<u8> = [3i32, 4, 5, 9, 10, 11]
18941 .iter()
18942 .flat_map(|v| v.to_le_bytes())
18943 .collect();
18944 writer.write_chunk_fixed_array(idx, &[0, 1], &c01).unwrap();
18945 let c10: Vec<u8> = [12i32, 13, 14, 18, 19, 20]
18946 .iter()
18947 .flat_map(|v| v.to_le_bytes())
18948 .collect();
18949 writer.write_chunk_fixed_array(idx, &[1, 0], &c10).unwrap();
18950 let c11: Vec<u8> = [15i32, 16, 17, 21, 22, 23]
18951 .iter()
18952 .flat_map(|v| v.to_le_bytes())
18953 .collect();
18954 writer.write_chunk_fixed_array(idx, &[1, 1], &c11).unwrap();
18955
18956 writer.close().unwrap();
18957
18958 let mut reader = Hdf5Reader::open(&path).unwrap();
18959 assert_eq!(reader.dataset_shape("grid").unwrap(), vec![4, 6]);
18960 let raw = reader.read_dataset_raw("grid").unwrap();
18961 let values: Vec<i32> = raw
18962 .chunks(4)
18963 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
18964 .collect();
18965 assert_eq!(values.len(), 24);
18966 for (i, v) in values.iter().enumerate() {
18967 assert_eq!(*v, i as i32, "element {i}");
18968 }
18969
18970 std::fs::remove_file(&path).ok();
18971 }
18972
18973 #[cfg(feature = "deflate")]
18974 #[test]
18975 fn create_filtered_fixed_array_paged_dataset_roundtrip() {
18976 // Large compressed fixed-shape chunked dataset (>1024 chunks): the
18977 // filtered FA data block must be paged.
18978 let path = temp_path("fixed_array_filt_paged");
18979
18980 let n: usize = 3000;
18981 let writer = Hdf5Writer::create(&path).unwrap();
18982 let idx = writer
18983 .create_fixed_array_dataset_with_pipeline(
18984 "paged",
18985 DatatypeMessage::i32_type(),
18986 &[n as u64],
18987 &[1],
18988 FilterPipeline::deflate(6),
18989 )
18990 .unwrap();
18991
18992 for i in 0..n {
18993 let v = (i as i32).to_le_bytes();
18994 writer
18995 .write_chunk_fixed_array(idx, &[i as u64], &v)
18996 .unwrap();
18997 }
18998 writer.close().unwrap();
18999
19000 let mut reader = Hdf5Reader::open(&path).unwrap();
19001 assert_eq!(reader.dataset_shape("paged").unwrap(), vec![n as u64]);
19002 let raw = reader.read_dataset_raw("paged").unwrap();
19003 let values: Vec<i32> = raw
19004 .chunks(4)
19005 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19006 .collect();
19007 assert_eq!(values.len(), n);
19008 for (i, v) in values.iter().enumerate() {
19009 assert_eq!(*v, i as i32, "element {i}");
19010 }
19011
19012 std::fs::remove_file(&path).ok();
19013 }
19014
19015 #[test]
19016 fn filtered_fixed_array_dblk_disk_size_and_encode() {
19017 // Cross-check filtered FA data-block sizing against the encoded length,
19018 // for both flat and paged layouts.
19019 let ctx = FormatContext {
19020 sizeof_addr: 8,
19021 sizeof_size: 8,
19022 };
19023 let csl = 3u8; // chunk_size_len
19024 let elem_size = 8 + csl as usize + 4; // addr + size + filter_mask
19025
19026 // Flat: 100 chunks. prefix(14) + 100*elem_size + cksum(4).
19027 let mut flat = FixedArrayHeader::new_for_filtered_chunks(&ctx, 100, csl);
19028 flat.data_blk_addr = 0x4000;
19029 assert!(!flat.is_paged());
19030 assert_eq!(
19031 fixed_array_dblk_disk_size(&ctx, &flat),
19032 (14 + 100 * elem_size + 4) as u64
19033 );
19034 let flat_dblk = FixedArrayDataBlock::new_filtered(0x1000, 100);
19035 assert_eq!(
19036 encode_fixed_array_dblk(&ctx, &flat, &flat_dblk).len() as u64,
19037 fixed_array_dblk_disk_size(&ctx, &flat)
19038 );
19039
19040 // Paged: 2500 chunks => 3 pages. prefix(4+1+1+8+1+4=19)
19041 // + 2500*elem_size + 3*cksum(4).
19042 let mut paged = FixedArrayHeader::new_for_filtered_chunks(&ctx, 2500, csl);
19043 paged.data_blk_addr = 0x9000;
19044 assert!(paged.is_paged());
19045 assert_eq!(paged.npages(), 3);
19046 assert_eq!(
19047 fixed_array_dblk_disk_size(&ctx, &paged),
19048 (19 + 2500 * elem_size + 12) as u64
19049 );
19050 let mut paged_dblk = FixedArrayDataBlock::new_filtered(0x1000, 2500);
19051 for (i, e) in paged_dblk.filtered_elements.iter_mut().enumerate() {
19052 e.address = 0x10000 + (i as u64) * 0x100;
19053 e.chunk_size = (i % 200) as u64;
19054 }
19055 let encoded = encode_fixed_array_dblk(&ctx, &paged, &paged_dblk);
19056 assert_eq!(
19057 encoded.len() as u64,
19058 fixed_array_dblk_disk_size(&ctx, &paged)
19059 );
19060
19061 // Decode the paged prefix + pages as the reader does.
19062 let npages = paged.npages() as usize;
19063 let prefix = FixedArrayPagedPrefix::decode(&encoded, &ctx, npages as u64).unwrap();
19064 for p in 0..npages {
19065 assert!(prefix.page_initialized(p), "page {p}");
19066 }
19067 let dblk_page_nelmts = paged.dblk_page_nelmts() as usize;
19068 let page_stride = dblk_page_nelmts * elem_size + 4;
19069 let mut recovered = Vec::new();
19070 for p in 0..npages {
19071 let page_nelmts = if p + 1 == npages {
19072 2500 - p * dblk_page_nelmts
19073 } else {
19074 dblk_page_nelmts
19075 };
19076 let off = prefix.prefix_size + p * page_stride;
19077 let elems = crate::format::chunk_index::fixed_array::decode_filtered_page(
19078 &encoded[off..],
19079 &ctx,
19080 page_nelmts,
19081 csl as usize,
19082 )
19083 .unwrap();
19084 recovered.extend(elems);
19085 }
19086 assert_eq!(recovered, paged_dblk.filtered_elements);
19087 }
19088
19089 #[test]
19090 fn create_btree_v2_dataset_roundtrip() {
19091 let path = temp_path("btree_v2");
19092
19093 let writer = Hdf5Writer::create(&path).unwrap();
19094 let idx = writer
19095 .create_btree_v2_dataset(
19096 "data",
19097 DatatypeMessage::f64_type(),
19098 &[0, 0], // start empty
19099 &[u64::MAX, u64::MAX], // both dims unlimited
19100 &[2, 3], // chunk = 2x3
19101 )
19102 .unwrap();
19103
19104 // Write chunks for a 4x6 dataset
19105 // chunk (0,0)
19106 let c00: Vec<u8> = [0.0f64, 1.0, 2.0, 6.0, 7.0, 8.0]
19107 .iter()
19108 .flat_map(|v| v.to_le_bytes())
19109 .collect();
19110 writer.write_chunk_btree_v2(idx, &[0, 0], &c00).unwrap();
19111
19112 // chunk (0,1)
19113 let c01: Vec<u8> = [3.0f64, 4.0, 5.0, 9.0, 10.0, 11.0]
19114 .iter()
19115 .flat_map(|v| v.to_le_bytes())
19116 .collect();
19117 writer.write_chunk_btree_v2(idx, &[0, 1], &c01).unwrap();
19118
19119 // chunk (1,0)
19120 let c10: Vec<u8> = [12.0f64, 13.0, 14.0, 18.0, 19.0, 20.0]
19121 .iter()
19122 .flat_map(|v| v.to_le_bytes())
19123 .collect();
19124 writer.write_chunk_btree_v2(idx, &[1, 0], &c10).unwrap();
19125
19126 // chunk (1,1)
19127 let c11: Vec<u8> = [15.0f64, 16.0, 17.0, 21.0, 22.0, 23.0]
19128 .iter()
19129 .flat_map(|v| v.to_le_bytes())
19130 .collect();
19131 writer.write_chunk_btree_v2(idx, &[1, 1], &c11).unwrap();
19132
19133 writer.extend_dataset(idx, &[4, 6]).unwrap();
19134 writer.close().unwrap();
19135
19136 // Read back
19137 let mut reader = Hdf5Reader::open(&path).unwrap();
19138 assert_eq!(reader.dataset_names(), vec!["data"]);
19139 assert_eq!(reader.dataset_shape("data").unwrap(), vec![4, 6]);
19140
19141 let raw = reader.read_dataset_raw("data").unwrap();
19142 let values: Vec<f64> = raw
19143 .chunks(8)
19144 .map(|chunk| f64::from_le_bytes(chunk.try_into().unwrap()))
19145 .collect();
19146 assert_eq!(values.len(), 24);
19147 for (i, val) in values.iter().enumerate() {
19148 assert_eq!(*val, i as f64);
19149 }
19150
19151 std::fs::remove_file(&path).ok();
19152 }
19153
19154 /// Bytes one chunk of [`btree_v2_flush_probe`]'s dataset occupies — an
19155 /// f64 element, so the allocator's alignment neither pads nor merges it and
19156 /// the file's growth is exactly the bytes asked for.
19157 const BT2_PROBE_CHUNK: u64 = 8;
19158
19159 /// Write chunks of a 1x1-chunked 2-D BT2 dataset, flushing at each batch
19160 /// boundary, and report `(node addresses, file length)` after every flush.
19161 /// Chunks are addressed down column 0 so the record count — and hence the
19162 /// tree's shape — grows one record at a time.
19163 fn btree_v2_flush_probe(path: &std::path::Path, batches: &[u64]) -> Vec<(Vec<u64>, u64)> {
19164 let writer = Hdf5Writer::create(path).unwrap();
19165 let idx = writer
19166 .create_btree_v2_dataset(
19167 "data",
19168 DatatypeMessage::f64_type(),
19169 &[0, 0],
19170 &[u64::MAX, u64::MAX],
19171 &[1, 1],
19172 )
19173 .unwrap();
19174 let mut written = 0u64;
19175 let mut out = Vec::new();
19176 for &upto in batches {
19177 while written < upto {
19178 writer
19179 .write_chunk_btree_v2(idx, &[written, 0], &(written as f64).to_le_bytes())
19180 .unwrap();
19181 written += 1;
19182 }
19183 writer.flush_dataset(idx).unwrap();
19184 let addrs = writer
19185 .ds(idx)
19186 .lock()
19187 .btree_v2
19188 .as_ref()
19189 .unwrap()
19190 .node_addrs
19191 .clone();
19192 out.push((addrs, std::fs::metadata(path).unwrap().len()));
19193 }
19194 writer.extend_dataset(idx, &[written.max(1), 1]).unwrap();
19195 writer.close().unwrap();
19196 out
19197 }
19198
19199 /// The node pool tracks the tree in both directions. Dropping records is
19200 /// what a removal path would do — [`Bt2ChunkIndex`] has none today, so the
19201 /// test drops them itself — and the flush that follows must hand the blocks
19202 /// its smaller tree no longer needs back to the allocator instead of
19203 /// leaving them recorded and unreachable.
19204 #[test]
19205 fn a_btree_v2_flush_frees_the_node_blocks_its_tree_gave_up() {
19206 use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
19207
19208 let path = temp_path("bt2_node_shrink");
19209 let writer = Hdf5Writer::create(&path).unwrap();
19210 let idx = writer
19211 .create_btree_v2_dataset(
19212 "data",
19213 DatatypeMessage::f64_type(),
19214 &[0, 0],
19215 &[u64::MAX, u64::MAX],
19216 &[1, 1],
19217 )
19218 .unwrap();
19219 // 85 records is one past a leaf, so the tree is two leaves and a root.
19220 for i in 0..85u64 {
19221 writer
19222 .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19223 .unwrap();
19224 }
19225 writer.flush_dataset(idx).unwrap();
19226 let grown = writer
19227 .ds(idx)
19228 .lock()
19229 .btree_v2
19230 .as_ref()
19231 .unwrap()
19232 .node_addrs
19233 .clone();
19234 assert_eq!(grown.len(), 3, "expected two leaves and a root");
19235
19236 // Back to 84 records: one leaf, so two of the three blocks are surplus.
19237 writer
19238 .ds(idx)
19239 .lock()
19240 .btree_v2
19241 .as_mut()
19242 .unwrap()
19243 .index
19244 .records
19245 .truncate(84);
19246 writer.flush_dataset(idx).unwrap();
19247 let shrunk = writer
19248 .ds(idx)
19249 .lock()
19250 .btree_v2
19251 .as_ref()
19252 .unwrap()
19253 .node_addrs
19254 .clone();
19255 assert_eq!(
19256 shrunk,
19257 grown[..1],
19258 "the pool still records the surplus blocks"
19259 );
19260
19261 // The surplus went back to the allocator, not on the floor: the next
19262 // node-sized allocation lands inside the region the two blocks covered.
19263 let reused = writer
19264 .allocator
19265 .allocate(BT2_NODE_SIZE as u64, FreeSpaceClass::Metadata);
19266 assert!(
19267 (grown[1]..grown[1] + 2 * BT2_NODE_SIZE as u64).contains(&reused),
19268 "a node block allocated at {reused:#x}, outside the freed \
19269 [{:#x}, {:#x}) the flush gave up",
19270 grown[1],
19271 grown[1] + 2 * BT2_NODE_SIZE as u64
19272 );
19273
19274 writer.extend_dataset(idx, &[85, 1]).unwrap();
19275 writer.close().unwrap();
19276 std::fs::remove_file(&path).ok();
19277 }
19278
19279 /// A v2 B-tree whose header declares a non-default node size — libhdf5
19280 /// built with a different `H5D_BT2_NODE_SIZE`, or any other writer —
19281 /// reopens for append: the reconstruction adopts the header's node_size,
19282 /// split and merge instead of refusing everything but 2048, and the next
19283 /// flush re-serializes at that size (upstream allocates every node at
19284 /// `hdr->node_size`, H5B2leaf.c / H5B2internal.c).
19285 #[test]
19286 fn a_btree_v2_with_a_foreign_node_size_reopens_and_grows() {
19287 let path = temp_path("bt2_foreign_node_size");
19288 {
19289 let writer = Hdf5Writer::create(&path).unwrap();
19290 let idx = writer
19291 .create_btree_v2_dataset(
19292 "data",
19293 DatatypeMessage::f64_type(),
19294 &[0, 0],
19295 &[u64::MAX, u64::MAX],
19296 &[1, 1],
19297 )
19298 .unwrap();
19299 // Act as a foreign writer: 512-byte nodes, non-default tuning.
19300 // record_size 24 => a 512-byte leaf holds 20 records, so 85
19301 // records make a depth-1 tree of 512-byte blocks.
19302 {
19303 let ds = writer.ds(idx);
19304 let mut m = ds.lock();
19305 let index = &mut m.btree_v2.as_mut().unwrap().index;
19306 index.node_size = 512;
19307 index.split_percent = 90;
19308 index.merge_percent = 30;
19309 }
19310 for i in 0..85u64 {
19311 writer
19312 .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19313 .unwrap();
19314 }
19315 writer.extend_dataset(idx, &[85, 1]).unwrap();
19316 writer.close().unwrap();
19317 }
19318 {
19319 let writer = Hdf5Writer::open_append(&path).unwrap();
19320 let idx = writer.dataset_index("data").unwrap();
19321 {
19322 let ds = writer.ds(idx);
19323 let m = ds.lock();
19324 let index = &m.btree_v2.as_ref().unwrap().index;
19325 assert_eq!(index.node_size, 512, "header node_size not adopted");
19326 assert_eq!(index.split_percent, 90);
19327 assert_eq!(index.merge_percent, 30);
19328 assert_eq!(index.records.len(), 85, "records not walked back");
19329 }
19330 for i in 85..115u64 {
19331 writer
19332 .write_chunk_btree_v2(idx, &[i, 0], &(i as f64).to_le_bytes())
19333 .unwrap();
19334 }
19335 writer.extend_dataset(idx, &[115, 1]).unwrap();
19336 writer.close().unwrap();
19337 }
19338
19339 let mut reader = Hdf5Reader::open(&path).unwrap();
19340 let raw = reader.read_dataset_raw("data").unwrap();
19341 let values: Vec<f64> = raw
19342 .chunks(8)
19343 .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19344 .collect();
19345 assert_eq!(values.len(), 115);
19346 for (i, v) in values.iter().enumerate() {
19347 assert_eq!(*v, i as f64, "element {i}");
19348 }
19349 std::fs::remove_file(&path).ok();
19350 }
19351
19352 /// A node's record count falls as well as rises: the tree's first leaf goes
19353 /// from a full 84 records to 42 when 85 records force it to split. The node
19354 /// image is padded to the whole block so re-serializing overwrites the
19355 /// block, not a prefix of it — otherwise that leaf keeps the tail of its
19356 /// 84-record self, stale records sitting in a live node block.
19357 #[test]
19358 fn a_shrinking_btree_v2_node_leaves_no_stale_records_behind() {
19359 use crate::format::chunk_index::btree_v2::{Bt2ChunkIndex, BT2_NODE_SIZE};
19360
19361 let path = temp_path("bt2_node_blocks");
19362 let probe = btree_v2_flush_probe(&path, &[84, 85]);
19363 let node0 = probe.last().unwrap().0[0];
19364
19365 // What the first leaf holds once the tree has split.
19366 let ctx = FormatContext {
19367 sizeof_addr: 8,
19368 sizeof_size: 8,
19369 };
19370 let mut index = Bt2ChunkIndex::new_unfiltered(2);
19371 for i in 0..85u64 {
19372 index.insert(vec![i, 0], 0);
19373 }
19374 let tree = index.build_tree(&ctx);
19375 assert!(
19376 tree.nodes[0].num_records < 84,
19377 "this test needs the first leaf to shrink, got {}",
19378 tree.nodes[0].num_records
19379 );
19380 // signature(4) + version(1) + type(1) + records + checksum(4)
19381 let used = 10 + tree.nodes[0].num_records as usize * tree.record_size as usize;
19382
19383 let bytes = std::fs::read(&path).unwrap();
19384 let block = &bytes[node0 as usize..node0 as usize + BT2_NODE_SIZE as usize];
19385 assert!(
19386 block[used..].iter().all(|&b| b == 0),
19387 "leaf block at {node0:#x} still holds {} bytes of its previous, larger image",
19388 block[used..].iter().rposition(|&b| b != 0).unwrap_or(0) + 1
19389 );
19390 std::fs::remove_file(&path).ok();
19391 }
19392
19393 /// The node pool is the single owner of the tree's block addresses: a flush
19394 /// reuses every block already in it and allocates only the shortfall. So
19395 /// re-flushing an unchanged index must cost nothing, and a flush that grows
19396 /// the tree must cost exactly the blocks it added — anything more means a
19397 /// block was stranded.
19398 #[test]
19399 fn a_btree_v2_flush_allocates_only_the_node_blocks_it_adds() {
19400 use crate::format::chunk_index::btree_v2::BT2_NODE_SIZE;
19401
19402 let path = temp_path("bt2_pool_growth");
19403 // Re-flush at 84 (still one leaf), then cross into a three-node depth-1
19404 // tree, then keep growing.
19405 let batches = [84u64, 84, 85, 200, 200];
19406 let probe = btree_v2_flush_probe(&path, &batches);
19407 for i in 1..probe.len() {
19408 let (prev_addrs, prev_len) = &probe[i - 1];
19409 let (addrs, len) = &probe[i];
19410 assert!(
19411 addrs.starts_with(prev_addrs),
19412 "flush {i} moved a node block instead of reusing it"
19413 );
19414 let new_blocks = (addrs.len() - prev_addrs.len()) as u64 * BT2_NODE_SIZE as u64;
19415 let new_chunks = (batches[i] - batches[i - 1]) * BT2_PROBE_CHUNK;
19416 assert_eq!(
19417 len - prev_len,
19418 new_blocks + new_chunks,
19419 "flush {i} grew the file by more than the blocks it added"
19420 );
19421 }
19422 // The unchanged re-flushes must be free.
19423 assert_eq!(probe[1].1, probe[0].1);
19424 assert_eq!(probe[4].1, probe[3].1);
19425 std::fs::remove_file(&path).ok();
19426 }
19427
19428 #[cfg(feature = "parallel")]
19429 #[test]
19430 fn parallel_batch_write_roundtrip() {
19431 let path = temp_path("parallel_batch");
19432
19433 let writer = Hdf5Writer::create(&path).unwrap();
19434 let idx = writer
19435 .create_chunked_dataset(
19436 "data",
19437 DatatypeMessage::i32_type(),
19438 &[0, 4],
19439 &[u64::MAX, 4],
19440 &[1, 4],
19441 )
19442 .unwrap();
19443
19444 // Prepare chunks
19445 let chunks_data: Vec<(u64, Vec<u8>)> = (0..8u64)
19446 .map(|frame| {
19447 let values: Vec<i32> = (0..4).map(|i| (frame * 4 + i) as i32).collect();
19448 let raw: Vec<u8> = values.iter().flat_map(|v| v.to_le_bytes()).collect();
19449 (frame, raw)
19450 })
19451 .collect();
19452
19453 let batch: Vec<(u64, &[u8])> = chunks_data
19454 .iter()
19455 .map(|(idx, data)| (*idx, data.as_slice()))
19456 .collect();
19457
19458 writer.write_chunks_batch(idx, &batch).unwrap();
19459 writer.extend_dataset(idx, &[8, 4]).unwrap();
19460 writer.close().unwrap();
19461
19462 // Read back
19463 let mut reader = Hdf5Reader::open(&path).unwrap();
19464 assert_eq!(reader.dataset_shape("data").unwrap(), vec![8, 4]);
19465 let raw = reader.read_dataset_raw("data").unwrap();
19466 let values: Vec<i32> = raw
19467 .chunks(4)
19468 .map(|chunk| i32::from_le_bytes(chunk.try_into().unwrap()))
19469 .collect();
19470 assert_eq!(values.len(), 32);
19471 for (i, val) in values.iter().enumerate() {
19472 assert_eq!(*val, i as i32);
19473 }
19474
19475 std::fs::remove_file(&path).ok();
19476 }
19477
19478 #[test]
19479 fn swmr_writer_append_frames() {
19480 use crate::io::swmr::SwmrWriter;
19481
19482 // Per-call unique path so concurrent cargo invocations and
19483 // kernel-side flock release races cannot collide.
19484 use std::sync::atomic::{AtomicU64, Ordering};
19485 static COUNTER: AtomicU64 = AtomicU64::new(0);
19486 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19487 let path = std::env::temp_dir().join(format!(
19488 "rust_hdf5_swmr_append_{}_{}.h5",
19489 std::process::id(),
19490 n
19491 ));
19492
19493 let mut swmr = SwmrWriter::create(&path).unwrap();
19494 let idx = swmr
19495 .create_streaming_dataset("detector", DatatypeMessage::u16_type(), &[4, 4])
19496 .unwrap();
19497
19498 swmr.start_swmr().unwrap();
19499
19500 // Append 5 frames
19501 for frame in 0..5u16 {
19502 let data: Vec<u16> = (0..16).map(|i| frame * 16 + i).collect();
19503 let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19504 swmr.append_frame(idx, &raw).unwrap();
19505 }
19506
19507 swmr.flush().unwrap();
19508 swmr.close().unwrap();
19509
19510 // Read back
19511 let mut reader = Hdf5Reader::open(&path).unwrap();
19512 assert_eq!(reader.dataset_shape("detector").unwrap(), vec![5, 4, 4]);
19513
19514 let raw = reader.read_dataset_raw("detector").unwrap();
19515 let values: Vec<u16> = raw
19516 .chunks(2)
19517 .map(|chunk| u16::from_le_bytes(chunk.try_into().unwrap()))
19518 .collect();
19519 assert_eq!(values.len(), 80); // 5 * 4 * 4
19520 // Verify first frame
19521 for (i, val) in values.iter().enumerate().take(16) {
19522 assert_eq!(*val, i as u16);
19523 }
19524 // Verify last frame
19525 for (i, val) in values[64..80].iter().enumerate() {
19526 assert_eq!(*val, 4 * 16 + i as u16);
19527 }
19528
19529 std::fs::remove_file(&path).ok();
19530 }
19531
19532 #[test]
19533 fn swmr_writer_tiled_frames() {
19534 use crate::io::swmr::SwmrWriter;
19535 use std::sync::atomic::{AtomicU64, Ordering};
19536 static COUNTER: AtomicU64 = AtomicU64::new(0);
19537 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19538 let path = std::env::temp_dir().join(format!(
19539 "rust_hdf5_swmr_tiled_{}_{}.h5",
19540 std::process::id(),
19541 n
19542 ));
19543
19544 let mut swmr = SwmrWriter::create(&path).unwrap();
19545 // 4x4 frames, tiled into 2x2 chunks -> 4 chunks per frame.
19546 let idx = swmr
19547 .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[4, 4], &[2, 2])
19548 .unwrap();
19549 swmr.start_swmr().unwrap();
19550
19551 for frame in 0..3u16 {
19552 let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
19553 let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19554 swmr.append_frame(idx, &raw).unwrap();
19555 }
19556 swmr.flush().unwrap();
19557 swmr.close().unwrap();
19558
19559 let mut reader = Hdf5Reader::open(&path).unwrap();
19560 assert_eq!(reader.dataset_shape("det").unwrap(), vec![3, 4, 4]);
19561 let raw = reader.read_dataset_raw("det").unwrap();
19562 let values: Vec<u16> = raw
19563 .chunks(2)
19564 .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19565 .collect();
19566 assert_eq!(values.len(), 48);
19567 // Every element must survive the frame -> tile split and the
19568 // tile -> frame reassembly on read.
19569 for frame in 0..3u16 {
19570 for i in 0..16usize {
19571 assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
19572 }
19573 }
19574 std::fs::remove_file(&path).ok();
19575 }
19576
19577 /// A chunk tile larger than the frame is geometry libhdf5 refuses to
19578 /// create (`H5D__chunk_construct`: chunk must not exceed a fixed maximum
19579 /// dimension), so no libhdf5-based writer — including the NDFileHDF5
19580 /// tiling controls this API mirrors — can produce such a file. Until
19581 /// 0.4.1 we accepted it and zero-padded the frame up to the tile; now
19582 /// the create is rejected like every other creator's.
19583 #[test]
19584 fn swmr_writer_tiled_chunk_larger_than_frame_is_rejected() {
19585 use crate::io::swmr::SwmrWriter;
19586 use std::sync::atomic::{AtomicU64, Ordering};
19587 static COUNTER: AtomicU64 = AtomicU64::new(0);
19588 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19589 let path = std::env::temp_dir().join(format!(
19590 "rust_hdf5_swmr_bigchunk_{}_{}.h5",
19591 std::process::id(),
19592 n
19593 ));
19594
19595 let mut swmr = SwmrWriter::create(&path).unwrap();
19596 let err = swmr
19597 .create_streaming_dataset_tiled("det", DatatypeMessage::u16_type(), &[3, 3], &[8, 8])
19598 .unwrap_err();
19599 assert!(
19600 err.to_string().contains("maximum dimension size"),
19601 "unexpected error: {err}"
19602 );
19603 swmr.close().unwrap();
19604 std::fs::remove_file(&path).ok();
19605 }
19606
19607 #[test]
19608 fn swmr_writer_multi_frame_chunks() {
19609 use crate::io::swmr::SwmrWriter;
19610 use std::sync::atomic::{AtomicU64, Ordering};
19611 static COUNTER: AtomicU64 = AtomicU64::new(0);
19612 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19613 let path = std::env::temp_dir().join(format!(
19614 "rust_hdf5_swmr_mfc_{}_{}.h5",
19615 std::process::id(),
19616 n
19617 ));
19618
19619 // 3x3 frames, chunk = 4 frames x full frame. 10 frames -> 3 bands
19620 // of 4, 4, 2 (the last band partial).
19621 let mut swmr = SwmrWriter::create(&path).unwrap();
19622 let idx = swmr
19623 .create_streaming_dataset_chunked(
19624 "det",
19625 DatatypeMessage::u16_type(),
19626 &[3, 3],
19627 &[4, 3, 3],
19628 )
19629 .unwrap();
19630 swmr.start_swmr().unwrap();
19631 for frame in 0..10u16 {
19632 let data: Vec<u16> = (0..9).map(|i| frame * 100 + i).collect();
19633 let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19634 swmr.append_frame(idx, &raw).unwrap();
19635 }
19636 swmr.flush().unwrap();
19637 swmr.close().unwrap();
19638
19639 let mut reader = Hdf5Reader::open(&path).unwrap();
19640 // The partial last band must not over-extend the frame count.
19641 assert_eq!(reader.dataset_shape("det").unwrap(), vec![10, 3, 3]);
19642 let raw = reader.read_dataset_raw("det").unwrap();
19643 let values: Vec<u16> = raw
19644 .chunks(2)
19645 .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19646 .collect();
19647 assert_eq!(values.len(), 90);
19648 for frame in 0..10u16 {
19649 for i in 0..9usize {
19650 assert_eq!(values[frame as usize * 9 + i], frame * 100 + i as u16);
19651 }
19652 }
19653 std::fs::remove_file(&path).ok();
19654 }
19655
19656 #[test]
19657 fn swmr_writer_multi_frame_tiled_chunks() {
19658 use crate::io::swmr::SwmrWriter;
19659 use std::sync::atomic::{AtomicU64, Ordering};
19660 static COUNTER: AtomicU64 = AtomicU64::new(0);
19661 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19662 let path = std::env::temp_dir().join(format!(
19663 "rust_hdf5_swmr_mftc_{}_{}.h5",
19664 std::process::id(),
19665 n
19666 ));
19667
19668 // 4x4 frames, chunk = 2 frames x 2x2 tiles. 5 frames -> bands of
19669 // 2, 2, 1; every frame is also split into a 2x2 tile grid.
19670 let mut swmr = SwmrWriter::create(&path).unwrap();
19671 let idx = swmr
19672 .create_streaming_dataset_chunked(
19673 "det",
19674 DatatypeMessage::u16_type(),
19675 &[4, 4],
19676 &[2, 2, 2],
19677 )
19678 .unwrap();
19679 swmr.start_swmr().unwrap();
19680 for frame in 0..5u16 {
19681 let data: Vec<u16> = (0..16).map(|i| frame * 100 + i).collect();
19682 let raw: Vec<u8> = data.iter().flat_map(|v| v.to_le_bytes()).collect();
19683 swmr.append_frame(idx, &raw).unwrap();
19684 }
19685 swmr.flush().unwrap();
19686 swmr.close().unwrap();
19687
19688 let mut reader = Hdf5Reader::open(&path).unwrap();
19689 assert_eq!(reader.dataset_shape("det").unwrap(), vec![5, 4, 4]);
19690 let raw = reader.read_dataset_raw("det").unwrap();
19691 let values: Vec<u16> = raw
19692 .chunks(2)
19693 .map(|c| u16::from_le_bytes(c.try_into().unwrap()))
19694 .collect();
19695 assert_eq!(values.len(), 80);
19696 for frame in 0..5u16 {
19697 for i in 0..16usize {
19698 assert_eq!(values[frame as usize * 16 + i], frame * 100 + i as u16);
19699 }
19700 }
19701 std::fs::remove_file(&path).ok();
19702 }
19703
19704 #[cfg(feature = "deflate")]
19705 #[test]
19706 fn swmr_writer_compressed_frames() {
19707 use crate::io::swmr::SwmrWriter;
19708 use std::sync::atomic::{AtomicU64, Ordering};
19709 static COUNTER: AtomicU64 = AtomicU64::new(0);
19710 let n = COUNTER.fetch_add(1, Ordering::Relaxed);
19711 let path = std::env::temp_dir().join(format!(
19712 "rust_hdf5_swmr_comp_{}_{}.h5",
19713 std::process::id(),
19714 n
19715 ));
19716
19717 let mut swmr = SwmrWriter::create(&path).unwrap();
19718 let pipeline = crate::format::messages::filter::FilterPipeline::deflate(4);
19719 let idx = swmr
19720 .create_streaming_dataset_compressed(
19721 "detector",
19722 DatatypeMessage::i32_type(),
19723 &[8],
19724 pipeline,
19725 )
19726 .unwrap();
19727 swmr.start_swmr().unwrap();
19728
19729 for frame in 0..40i32 {
19730 let raw: Vec<u8> = (0..8).flat_map(|i| (frame * 8 + i).to_le_bytes()).collect();
19731 swmr.append_frame(idx, &raw).unwrap();
19732 if frame % 7 == 0 {
19733 swmr.flush().unwrap();
19734 }
19735 }
19736 swmr.flush().unwrap();
19737 swmr.close().unwrap();
19738
19739 let mut reader = Hdf5Reader::open(&path).unwrap();
19740 assert_eq!(reader.dataset_shape("detector").unwrap(), vec![40, 8]);
19741 let raw = reader.read_dataset_raw("detector").unwrap();
19742 let values: Vec<i32> = raw
19743 .chunks(4)
19744 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19745 .collect();
19746 assert_eq!(values, (0..320).collect::<Vec<i32>>());
19747
19748 std::fs::remove_file(&path).ok();
19749 }
19750
19751 #[test]
19752 fn group_hierarchy_writer_reader() {
19753 let path = temp_path("group_hierarchy");
19754
19755 let writer = Hdf5Writer::create(&path).unwrap();
19756
19757 // Create groups
19758 let g0 = writer.create_group("/", "group1").unwrap();
19759 let g1 = writer.create_group("/group1", "sub").unwrap();
19760 assert_eq!(g0, 0);
19761 assert_eq!(g1, 1);
19762
19763 // Create datasets
19764 let ds_root = writer
19765 .create_dataset("root_data", DatatypeMessage::f64_type(), &[2])
19766 .unwrap();
19767 let raw_root: Vec<u8> = [1.0f64, 2.0].iter().flat_map(|v| v.to_le_bytes()).collect();
19768 writer.write_dataset_raw(ds_root, &raw_root).unwrap();
19769
19770 let ds_g0 = writer
19771 .create_dataset("group1/data", DatatypeMessage::i32_type(), &[3])
19772 .unwrap();
19773 let raw_g0: Vec<u8> = [10i32, 20, 30]
19774 .iter()
19775 .flat_map(|v| v.to_le_bytes())
19776 .collect();
19777 writer.write_dataset_raw(ds_g0, &raw_g0).unwrap();
19778
19779 let ds_g1 = writer
19780 .create_dataset("group1/sub/values", DatatypeMessage::u8_type(), &[4])
19781 .unwrap();
19782 writer.write_dataset_raw(ds_g1, &[1u8, 2, 3, 4]).unwrap();
19783
19784 writer.close().unwrap();
19785
19786 // Read back
19787 let mut reader = Hdf5Reader::open(&path).unwrap();
19788 let names = reader.dataset_names();
19789 assert!(names.contains(&"root_data"), "names: {:?}", names);
19790 assert!(names.contains(&"group1/data"), "names: {:?}", names);
19791 assert!(names.contains(&"group1/sub/values"), "names: {:?}", names);
19792
19793 let raw = reader.read_dataset_raw("root_data").unwrap();
19794 let vals: Vec<f64> = raw
19795 .chunks(8)
19796 .map(|c| f64::from_le_bytes(c.try_into().unwrap()))
19797 .collect();
19798 assert_eq!(vals, vec![1.0, 2.0]);
19799
19800 let raw = reader.read_dataset_raw("group1/data").unwrap();
19801 let vals: Vec<i32> = raw
19802 .chunks(4)
19803 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19804 .collect();
19805 assert_eq!(vals, vec![10, 20, 30]);
19806
19807 let raw = reader.read_dataset_raw("group1/sub/values").unwrap();
19808 assert_eq!(raw, vec![1, 2, 3, 4]);
19809
19810 std::fs::remove_file(&path).ok();
19811 }
19812
19813 /// libhdf5 (`H5D__chunk_construct`) rejects a chunk dimension that
19814 /// exceeds a fixed maximum dimension. Before this check, such a dataset
19815 /// was created and appends landed rows at the chunk stride instead of
19816 /// the row stride, reading back [1, 2, 0, 0] for [1, 2, 3, 4].
19817 #[test]
19818 fn create_rejects_a_chunk_wider_than_a_fixed_max_dimension() {
19819 let path = temp_path("chunk_wider_than_max");
19820
19821 let writer = Hdf5Writer::create(&path).unwrap();
19822 let err = writer
19823 .create_chunked_dataset(
19824 "data",
19825 DatatypeMessage::f64_type(),
19826 &[0, 2],
19827 &[u64::MAX, 2],
19828 &[2, 4],
19829 )
19830 .unwrap_err();
19831 assert!(
19832 err.to_string().contains("maximum dimension size"),
19833 "unexpected error: {err}"
19834 );
19835
19836 // The fixed-array creators derive the maximum from the fixed dims.
19837 let err = writer
19838 .create_fixed_array_dataset("fa", DatatypeMessage::f64_type(), &[3], &[5])
19839 .unwrap_err();
19840 assert!(
19841 err.to_string().contains("maximum dimension size"),
19842 "unexpected error: {err}"
19843 );
19844
19845 writer.close().unwrap();
19846 std::fs::remove_file(&path).ok();
19847 }
19848
19849 /// libhdf5 exempts a dimension whose *current* size is zero from the
19850 /// chunk-vs-maximum check (`curr_dims[u] &&` in `H5D__chunk_construct`),
19851 /// and rejects a zero chunk dimension on every path.
19852 #[test]
19853 fn create_mirrors_the_libhdf5_chunk_geometry_exemptions() {
19854 let path = temp_path("chunk_geometry_exemptions");
19855
19856 let writer = Hdf5Writer::create(&path).unwrap();
19857 // dims[1] == 0: chunk 4 > max 2 is allowed, as libhdf5 allows it.
19858 writer
19859 .create_chunked_dataset(
19860 "exempt",
19861 DatatypeMessage::f64_type(),
19862 &[0, 0],
19863 &[u64::MAX, 2],
19864 &[2, 4],
19865 )
19866 .unwrap();
19867
19868 let err = writer
19869 .create_chunked_dataset("zero", DatatypeMessage::f64_type(), &[0], &[u64::MAX], &[0])
19870 .unwrap_err();
19871 assert!(
19872 err.to_string().contains("chunk dimension 0 is zero"),
19873 "unexpected error: {err}"
19874 );
19875
19876 writer.close().unwrap();
19877 std::fs::remove_file(&path).ok();
19878 }
19879
19880 /// A file written by 0.4.0 can carry a chunk row wider than the frame
19881 /// row — create now rejects that geometry, but reopened files keep it.
19882 /// Appends must scatter frames at the chunk stride, not pack them at
19883 /// the frame stride (which read back `[1, 2, 0, 0]` for `[1, 2, 3, 4]`).
19884 /// The wide shape is simulated by widening the registered chunk dims
19885 /// after create, which also lands in the layout message at close.
19886 #[test]
19887 fn append_scatters_into_a_legacy_wider_than_row_chunk() {
19888 let path = temp_path("legacy_wide_chunk_append");
19889
19890 let writer = Hdf5Writer::create(&path).unwrap();
19891 let idx = writer
19892 .create_chunked_dataset(
19893 "data",
19894 DatatypeMessage::i32_type(),
19895 &[0, 2],
19896 &[u64::MAX, 2],
19897 &[2, 2],
19898 )
19899 .unwrap();
19900 writer.ds(idx).lock().chunked.as_mut().unwrap().chunk_dims = vec![2, 4];
19901
19902 let frames: Vec<u8> = [1i32, 2, 3, 4]
19903 .iter()
19904 .flat_map(|v| v.to_le_bytes())
19905 .collect();
19906 writer.write_append_frames(idx, 0, 2, &frames).unwrap();
19907 writer.extend_dataset(idx, &[2, 2]).unwrap();
19908 writer.close().unwrap();
19909
19910 let mut reader = Hdf5Reader::open(&path).unwrap();
19911 assert_eq!(reader.dataset_shape("data").unwrap(), vec![2, 2]);
19912 let raw = reader.read_dataset_raw("data").unwrap();
19913 let values: Vec<i32> = raw
19914 .chunks(4)
19915 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
19916 .collect();
19917 assert_eq!(values, vec![1, 2, 3, 4]);
19918 std::fs::remove_file(&path).ok();
19919 }
19920
19921 /// The compressed vlen creator sizes its chunked layout from a
19922 /// caller-supplied chunk size; it goes through the same geometry
19923 /// validation as every other creator (empty inputs are exempt because
19924 /// their current size is zero).
19925 #[test]
19926 #[cfg(feature = "deflate")]
19927 fn compressed_vlen_create_validates_its_chunk_size() {
19928 use crate::format::messages::filter::FilterPipeline;
19929 let path = temp_path("vlen_compressed_chunk");
19930
19931 let writer = Hdf5Writer::create(&path).unwrap();
19932 let err = writer
19933 .create_vlen_string_dataset_compressed(
19934 "texts",
19935 &["a", "b", "c"],
19936 100,
19937 FilterPipeline::deflate(6),
19938 )
19939 .unwrap_err();
19940 assert!(
19941 err.to_string().contains("maximum dimension size"),
19942 "unexpected error: {err}"
19943 );
19944
19945 writer
19946 .create_vlen_string_dataset_compressed("empty", &[], 16, FilterPipeline::deflate(6))
19947 .unwrap();
19948
19949 writer.close().unwrap();
19950 std::fs::remove_file(&path).ok();
19951 }
19952
19953 /// `set_libver_latest` moves *filtered* chunked datasets to layout v5 with
19954 /// fixed 8-byte chunk-size fields; unfiltered chunked and pre-opt-in
19955 /// datasets keep v4 with the derived width, matching libhdf5's
19956 /// `version_perf` rule (only the filtered index arms bump to 5).
19957 #[cfg(feature = "deflate")]
19958 #[test]
19959 fn libver_latest_selects_v5_for_filtered_chunks_only() {
19960 let path = temp_path("libver_v5_select");
19961
19962 let mut writer = Hdf5Writer::create(&path).unwrap();
19963 let before = writer
19964 .create_chunked_dataset_with_pipeline(
19965 "d4",
19966 DatatypeMessage::i32_type(),
19967 &[0],
19968 &[u64::MAX],
19969 &[16],
19970 FilterPipeline::deflate(4),
19971 )
19972 .unwrap();
19973 writer.set_libver_latest(true).unwrap();
19974 let ea5 = writer
19975 .create_chunked_dataset_with_pipeline(
19976 "ea5",
19977 DatatypeMessage::i32_type(),
19978 &[0],
19979 &[u64::MAX],
19980 &[16],
19981 FilterPipeline::deflate(4),
19982 )
19983 .unwrap();
19984 let plain = writer
19985 .create_chunked_dataset(
19986 "plain",
19987 DatatypeMessage::i32_type(),
19988 &[0],
19989 &[u64::MAX],
19990 &[16],
19991 )
19992 .unwrap();
19993 let fa5 = writer
19994 .create_fixed_array_dataset_with_pipeline(
19995 "fa5",
19996 DatatypeMessage::i32_type(),
19997 &[4, 6],
19998 &[2, 3],
19999 FilterPipeline::deflate(6),
20000 )
20001 .unwrap();
20002 let bt5 = writer
20003 .create_btree_v2_dataset_with_pipeline(
20004 "bt5",
20005 DatatypeMessage::i32_type(),
20006 &[0, 0],
20007 &[u64::MAX, u64::MAX],
20008 &[2, 3],
20009 FilterPipeline::deflate(6),
20010 )
20011 .unwrap();
20012
20013 {
20014 let d4 = writer.ds(before);
20015 let d4 = d4.lock();
20016 assert_eq!(d4.layout_version, 4);
20017 assert_eq!(
20018 d4.chunked.as_ref().unwrap().chunk_size_len,
20019 compute_chunk_size_len(16 * 4)
20020 );
20021 let e5 = writer.ds(ea5);
20022 let e5 = e5.lock();
20023 assert_eq!(e5.layout_version, 5);
20024 assert_eq!(e5.chunked.as_ref().unwrap().chunk_size_len, 8);
20025 assert_eq!(writer.ds(plain).lock().layout_version, 4);
20026 assert_eq!(writer.ds(fa5).lock().layout_version, 5);
20027 assert_eq!(writer.ds(bt5).lock().layout_version, 5);
20028 }
20029
20030 // Write through the FA and BT2 v5 indexes so their 8-byte chunk-size
20031 // fields are exercised end to end, not just selected.
20032 for (coords, vals) in [
20033 ([0u64, 0], [0i32, 1, 2, 6, 7, 8]),
20034 ([0, 1], [3, 4, 5, 9, 10, 11]),
20035 ([1, 0], [12, 13, 14, 18, 19, 20]),
20036 ([1, 1], [15, 16, 17, 21, 22, 23]),
20037 ] {
20038 let bytes: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
20039 writer
20040 .write_chunk_fixed_array(fa5, &coords, &bytes)
20041 .unwrap();
20042 writer.write_chunk_btree_v2(bt5, &coords, &bytes).unwrap();
20043 }
20044 writer.extend_dataset(bt5, &[4, 6]).unwrap();
20045 writer.close().unwrap();
20046
20047 let mut reader = Hdf5Reader::open(&path).unwrap();
20048 for name in ["fa5", "bt5"] {
20049 let raw = reader.read_dataset_raw(name).unwrap();
20050 let values: Vec<i32> = raw
20051 .chunks(4)
20052 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
20053 .collect();
20054 assert_eq!(values, (0..24).collect::<Vec<i32>>(), "dataset {name}");
20055 }
20056
20057 std::fs::remove_file(&path).ok();
20058 }
20059
20060 /// A v5 file reopened for append must stay v5: the decode → `DatasetInfo`
20061 /// → finalize path carries the version through, so the re-encoded layout
20062 /// message matches the 8-byte size fields the filtered index was built
20063 /// with. A silent v4 downgrade here would make libhdf5 derive a narrower
20064 /// field width than the index uses.
20065 #[cfg(feature = "deflate")]
20066 #[test]
20067 fn v5_layout_survives_reopen_and_append() {
20068 let path = temp_path("libver_v5_reopen");
20069 let chunk: usize = 8;
20070
20071 let mut writer = Hdf5Writer::create(&path).unwrap();
20072 writer.set_libver_latest(true).unwrap();
20073 let idx = writer
20074 .create_chunked_dataset_with_pipeline(
20075 "d",
20076 DatatypeMessage::i32_type(),
20077 &[0],
20078 &[u64::MAX],
20079 &[chunk as u64],
20080 FilterPipeline::deflate(4),
20081 )
20082 .unwrap();
20083 for c in 0..2u64 {
20084 let data: Vec<u8> = (0..chunk as i32)
20085 .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
20086 .collect();
20087 writer.write_chunk(idx, c, &data).unwrap();
20088 }
20089 writer.extend_dataset(idx, &[2 * chunk as u64]).unwrap();
20090 writer.close().unwrap();
20091
20092 // Reopen: the decoded layout version must be preserved, and appends
20093 // must keep working against the 8-byte-size-field index.
20094 let writer = Hdf5Writer::open_append(&path).unwrap();
20095 assert_eq!(writer.ds(0).lock().layout_version, 5);
20096 for c in 2..4u64 {
20097 let data: Vec<u8> = (0..chunk as i32)
20098 .flat_map(|i| (c as i32 * chunk as i32 + i).to_le_bytes())
20099 .collect();
20100 writer.write_chunk(0, c, &data).unwrap();
20101 }
20102 writer.extend_dataset(0, &[4 * chunk as u64]).unwrap();
20103 writer.close().unwrap();
20104
20105 // Still v5 after the second finalize, and fully readable.
20106 let writer = Hdf5Writer::open_append(&path).unwrap();
20107 assert_eq!(writer.ds(0).lock().layout_version, 5);
20108 writer.close().unwrap();
20109
20110 let mut reader = Hdf5Reader::open(&path).unwrap();
20111 let raw = reader.read_dataset_raw("d").unwrap();
20112 let values: Vec<i32> = raw
20113 .chunks(4)
20114 .map(|c| i32::from_le_bytes(c.try_into().unwrap()))
20115 .collect();
20116 assert_eq!(values, (0..4 * chunk as i32).collect::<Vec<i32>>());
20117
20118 std::fs::remove_file(&path).ok();
20119 }
20120
20121 /// A chunk strictly larger than `u32::MAX` bytes forces layout v5 with no
20122 /// opt-in — v4's size field cannot represent it — while a chunk of exactly
20123 /// `u32::MAX` bytes stays v4, matching libhdf5's `version_req` boundary
20124 /// (`> 0xffffffff`, filtered or not).
20125 #[test]
20126 fn oversized_chunk_forces_v5_without_opt_in() {
20127 let path = temp_path("libver_4gib_force");
20128
20129 let writer = Hdf5Writer::create(&path).unwrap();
20130 let at_limit = writer
20131 .create_chunked_dataset_with_pipeline(
20132 "at_limit",
20133 DatatypeMessage::u8_type(),
20134 &[0],
20135 &[u64::MAX],
20136 &[u32::MAX as u64],
20137 FilterPipeline::deflate(4),
20138 )
20139 .unwrap();
20140 let over = writer
20141 .create_chunked_dataset_with_pipeline(
20142 "over",
20143 DatatypeMessage::u8_type(),
20144 &[0],
20145 &[u64::MAX],
20146 &[u32::MAX as u64 + 1],
20147 FilterPipeline::deflate(4),
20148 )
20149 .unwrap();
20150 let over_unfiltered = writer
20151 .create_chunked_dataset(
20152 "over_plain",
20153 DatatypeMessage::u8_type(),
20154 &[0],
20155 &[u64::MAX],
20156 &[u32::MAX as u64 + 1],
20157 )
20158 .unwrap();
20159
20160 assert_eq!(writer.ds(at_limit).lock().layout_version, 4);
20161 {
20162 let ds = writer.ds(over);
20163 let ds = ds.lock();
20164 assert_eq!(ds.layout_version, 5);
20165 assert_eq!(ds.chunked.as_ref().unwrap().chunk_size_len, 8);
20166 }
20167 assert_eq!(writer.ds(over_unfiltered).lock().layout_version, 5);
20168 writer.close().unwrap();
20169 std::fs::remove_file(&path).ok();
20170 }
20171
20172 /// SWMR reaches version 3 on its own, without a chunked dataset to raise
20173 /// the bound — through the flags `finalize_for_swmr` passes, and then
20174 /// through `swmr_active` for every superblock written after it. Only a
20175 /// file with nothing else newer in it can tell the two arms apart, and
20176 /// the public SWMR API always creates a chunked streaming dataset.
20177 #[test]
20178 fn swmr_reaches_version_3_with_no_chunked_dataset_in_the_file() {
20179 let path = temp_path("swmr_superblock");
20180
20181 let mut writer = Hdf5Writer::create(&path).unwrap();
20182 writer
20183 .create_dataset("d", DatatypeMessage::i32_type(), &[2])
20184 .unwrap();
20185 assert_eq!(writer.superblock_version_for(0), SUPERBLOCK_V2);
20186
20187 writer.finalize_for_swmr().unwrap();
20188 // What `start_swmr` does after finalizing, and what lets a second
20189 // handle read the file while this writer lives — the writer's
20190 // exclusive lock is mandatory on Windows.
20191 writer.handle().release_lock().unwrap();
20192 assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
20193
20194 // The close-time finalize carries no SWMR flag; the file is still an
20195 // SWMR file and must not be handed back a version older than the one
20196 // its readers attached to.
20197 writer.close().unwrap();
20198 assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
20199 std::fs::remove_file(&path).ok();
20200 }
20201
20202 /// A named bound below `H5F_LIBVER_V110` refuses the session instead —
20203 /// the two checks `H5F__start_swmr_write` opens with, a version-3
20204 /// superblock (H5Fint.c:3814) and a low bound of at least V110
20205 /// (H5Fint.c:3818). Naming no bound at all is what the test above does,
20206 /// and that file is free to become version 3.
20207 #[test]
20208 fn a_named_bound_below_v110_refuses_an_swmr_session() {
20209 for bound in [LibverBound::Earliest, LibverBound::V18] {
20210 let path = temp_path(&format!("swmr_refused_{bound:?}"));
20211 let mut writer = Hdf5Writer::create_with_options(
20212 &path,
20213 FileCreateOptions {
20214 libver: Some(bound),
20215 ..Default::default()
20216 },
20217 )
20218 .unwrap();
20219 writer
20220 .create_dataset("d", DatatypeMessage::i32_type(), &[2])
20221 .unwrap();
20222
20223 let err = writer.finalize_for_swmr().unwrap_err().to_string();
20224 assert!(err.contains("SWMR"), "{bound:?}: {err}");
20225 assert!(err.contains("H5F_LIBVER_V110"), "{bound:?}: {err}");
20226
20227 // Refused, not half-done: nothing was published, and the close
20228 // writes the file the bound asked for.
20229 writer.close().unwrap();
20230 let version = std::fs::read(&path).unwrap()[8];
20231 assert_eq!(version, bound.superblock_version(), "{bound:?}");
20232 std::fs::remove_file(&path).ok();
20233 }
20234 }
20235
20236 /// A dataset header the SWMR publish could not fit into the chunk 0 it
20237 /// already had chains into a continuation block, and the in-place rewrite
20238 /// goes back over both: chunk 0 stays at the address the file's readers
20239 /// hold, and the continuation chunk at the one chunk 0 names.
20240 #[test]
20241 fn inplace_rewrite_goes_over_a_chained_header() {
20242 let path = temp_path("inplace_rewrite_chained");
20243 let writer = Hdf5Writer::create_with_options(
20244 &path,
20245 FileCreateOptions {
20246 libver: Some(LibverBound::V110),
20247 ..Default::default()
20248 },
20249 )
20250 .unwrap();
20251 writer
20252 .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
20253 .unwrap();
20254 writer.close().unwrap();
20255
20256 let mut writer = Hdf5Writer::open_append(&path).unwrap();
20257 let idx = 0;
20258 let published = writer.ds(idx).lock().obj_header_written_addr.unwrap();
20259 for i in 0..4 {
20260 writer
20261 .add_dataset_attribute(
20262 idx,
20263 AttributeMessage::array_numeric(
20264 &format!("wide{i}"),
20265 DatatypeMessage::f64_type(),
20266 &[32],
20267 vec![0u8; 256],
20268 ),
20269 )
20270 .unwrap();
20271 }
20272 writer.finalize_for_swmr().unwrap();
20273 let blocks = writer.ds(idx).lock().obj_header_blocks.clone();
20274 assert_eq!(blocks.len(), 2, "chunk 0 and a continuation: {blocks:?}");
20275 assert_eq!(blocks[0].0, published, "chunk 0 stayed where it was");
20276
20277 writer.write_dataset_header_inplace(idx).unwrap();
20278 assert_eq!(writer.ds(idx).lock().obj_header_blocks, blocks);
20279 writer.close().unwrap();
20280
20281 // The closing finalize wrote over the same chunk 0, and the chained
20282 // header reads back whole.
20283 let writer = Hdf5Writer::open_append(&path).unwrap();
20284 assert_eq!(writer.ds(0).lock().obj_header_written_addr, Some(published));
20285 assert_eq!(writer.ds(0).lock().attributes.len(), 4);
20286 std::fs::remove_file(&path).ok();
20287 }
20288
20289 /// `H5F__start_swmr_write` refuses a low bound below `H5F_LIBVER_V110`
20290 /// (H5Fint.c:3818) on a reopened file as on a created one, now that the
20291 /// superblock no longer raises it: an SWMR reader follows the v1.10 chunk
20292 /// indexes, which a lower bound's layout version cannot name. No bound
20293 /// named passes, the default's layout row being `V110`'s.
20294 #[test]
20295 fn swmr_on_a_reopened_file_refuses_a_named_bound_below_v110() {
20296 let path = temp_path("swmr_reopen_bound");
20297 let writer = Hdf5Writer::create_with_options(
20298 &path,
20299 FileCreateOptions {
20300 libver: Some(LibverBound::V110),
20301 ..Default::default()
20302 },
20303 )
20304 .unwrap();
20305 writer
20306 .create_dataset("d", DatatypeMessage::i32_type(), &[2])
20307 .unwrap();
20308 writer.close().unwrap();
20309 assert_eq!(std::fs::read(&path).unwrap()[8], SUPERBLOCK_V3);
20310
20311 for bound in [LibverBound::Earliest, LibverBound::V18] {
20312 let mut writer = Hdf5Writer::open_append(&path).unwrap();
20313 writer.set_libver_bound(bound).unwrap();
20314 let err = writer.finalize_for_swmr().unwrap_err().to_string();
20315 assert!(err.contains("H5F_LIBVER_V110"), "{bound:?}: {err}");
20316 writer.close().unwrap();
20317 }
20318 let mut writer = Hdf5Writer::open_append(&path).unwrap();
20319 writer.finalize_for_swmr().unwrap();
20320 writer.close().unwrap();
20321 std::fs::remove_file(&path).ok();
20322 }
20323
20324 /// After every writer of a dataset object header, `nlink_written` is the
20325 /// count that writer encoded.
20326 ///
20327 /// `header_stale_with` is the one authority for "does the on-disk header
20328 /// still describe this dataset?", and it reads `nlink_written`; the three
20329 /// writers — `finalize`, `finalize_for_swmr` and
20330 /// `write_dataset_header_inplace` — therefore all record through
20331 /// `DatasetInfo::header_written`. This walks the SWMR sequence, where the
20332 /// in-place writer is the one that could drift, and pins why it does not:
20333 /// a name added after the publish grows the header past the block it was
20334 /// published into, so the rewrite is refused rather than half-applied and
20335 /// the count on disk stays the one the registry names.
20336 #[test]
20337 fn every_dataset_header_write_records_its_link_count() {
20338 let path = temp_path("header_write_records_nlink");
20339 let writer = Hdf5Writer::create(&path).unwrap();
20340 let idx = writer
20341 .create_chunked_dataset("d", DatatypeMessage::i32_type(), &[0], &[u64::MAX], &[4])
20342 .unwrap();
20343 let mut writer = writer;
20344 writer.finalize_for_swmr().unwrap();
20345 assert_eq!(
20346 writer.ds(idx).lock().nlink_written,
20347 1,
20348 "the SWMR publish put one name in the header"
20349 );
20350 writer.write_dataset_header_inplace(idx).unwrap();
20351 assert_eq!(writer.ds(idx).lock().nlink_written, 1);
20352
20353 // A second name after the publish: the reference-count message it
20354 // adds does not fit the published block.
20355 writer.create_hard_link("/", "alias", "d").unwrap();
20356 assert_eq!(writer.object_link_count(HardLinkTarget::Dataset(idx)), 2);
20357 let grew = writer
20358 .write_dataset_header_inplace(idx)
20359 .unwrap_err()
20360 .to_string();
20361 assert!(
20362 grew.contains("cannot rewrite in place"),
20363 "a header that outgrew its block must be refused: {grew}"
20364 );
20365 assert_eq!(
20366 writer.ds(idx).lock().nlink_written,
20367 1,
20368 "a refused rewrite leaves the registry describing the header the file holds"
20369 );
20370
20371 // The close-time finalize is the writer that commits the second name,
20372 // and a reopen reads the same count back off the link graph.
20373 writer.close().unwrap();
20374 let writer = Hdf5Writer::open_append(&path).unwrap();
20375 assert_eq!(
20376 writer.ds(0).lock().nlink_written,
20377 2,
20378 "finalize wrote two names and the reopen reads two"
20379 );
20380 writer.close().unwrap();
20381 std::fs::remove_file(&path).ok();
20382 }
20383
20384 /// `H5D__chunk_set_info`'s `version_req` (H5Dchunk.c:909, :936): version 5
20385 /// is required for a chunk over 4 GiB — the version-4 layout message's
20386 /// stored-size field is 32 bits and cannot record one — and
20387 /// `LAYOUT_VERSION_DEFAULT` (3, `H5O_LAYOUT_VERSION_DEFAULT`) is the floor
20388 /// for everything at or under that limit. Pure arithmetic on the byte
20389 /// count: no chunk is ever allocated.
20390 #[test]
20391 fn required_chunk_layout_version_pins_5_past_4_gib() {
20392 assert_eq!(
20393 Hdf5Writer::required_chunk_layout_version(u32::MAX as u64),
20394 LAYOUT_VERSION_DEFAULT
20395 );
20396 assert_eq!(
20397 Hdf5Writer::required_chunk_layout_version(u32::MAX as u64 + 1),
20398 5
20399 );
20400 }
20401
20402 /// `H5D__chunk_set_info`'s index-selection gate (H5Dchunk.c:936): a chunk
20403 /// over 4 GiB reaches the v1.10 chunk indexes even under a bound whose
20404 /// `H5O_layout_ver_bounds` row (`LibverBound::layout_version`) is below
20405 /// 4 — `V18` (row 3) and `Earliest` (row 1) both normally keep an
20406 /// ordinary chunk on the version-1 B-tree, but
20407 /// `required_chunk_layout_version`'s own escape to 5 overrides that row
20408 /// for this one chunk. The default bound (`V110`, row 4) already crosses
20409 /// the threshold on its own, so it is asserted only as the baseline, not
20410 /// as a distinguishing case for the escape.
20411 #[test]
20412 fn uses_v110_chunk_indexing_escapes_past_4_gib_at_every_bound() {
20413 let over_4gib = u32::MAX as u64 + 1;
20414 let small = 1024u64;
20415
20416 let path = temp_path("uses_v110_default");
20417 let writer = Hdf5Writer::create(&path).unwrap();
20418 assert!(writer.uses_v110_chunk_indexing(small));
20419 assert!(writer.uses_v110_chunk_indexing(over_4gib));
20420 writer.close().unwrap();
20421 std::fs::remove_file(&path).ok();
20422
20423 let path = temp_path("uses_v110_v18");
20424 let mut writer = Hdf5Writer::create(&path).unwrap();
20425 writer.set_libver_bound(LibverBound::V18).unwrap();
20426 assert!(
20427 !writer.uses_v110_chunk_indexing(small),
20428 "V18's layout row (3) stays below the v1.10 gate for an ordinary chunk"
20429 );
20430 assert!(
20431 writer.uses_v110_chunk_indexing(over_4gib),
20432 "the >4 GiB escape reaches v1.10 indexing despite V18's row"
20433 );
20434 writer.close().unwrap();
20435 std::fs::remove_file(&path).ok();
20436
20437 let path = temp_path("uses_v110_earliest");
20438 let mut writer = Hdf5Writer::create(&path).unwrap();
20439 writer.set_libver_bound(LibverBound::Earliest).unwrap();
20440 assert!(
20441 !writer.uses_v110_chunk_indexing(small),
20442 "Earliest's layout row (1) stays below the v1.10 gate for an ordinary chunk"
20443 );
20444 assert!(
20445 writer.uses_v110_chunk_indexing(over_4gib),
20446 "the >4 GiB escape reaches v1.10 indexing despite Earliest's row"
20447 );
20448 writer.close().unwrap();
20449 std::fs::remove_file(&path).ok();
20450 }
20451
20452 /// `H5D__chunk_set_info`'s closing `MAX3` (H5Dchunk.c:1046): the same
20453 /// escape pins the layout message itself at version 5 for a chunk over
20454 /// 4 GiB regardless of bound — `required_chunk_layout_version` dominates
20455 /// the max chain ahead of both the bound-derived preference and
20456 /// `LAYOUT_VERSION_DEFAULT`.
20457 #[test]
20458 fn chunk_layout_version_pins_5_past_4_gib_at_every_bound() {
20459 let over_4gib = u32::MAX as u64 + 1;
20460 let small = 1024u64;
20461
20462 let path = temp_path("chunk_ver_default");
20463 let writer = Hdf5Writer::create(&path).unwrap();
20464 assert_eq!(writer.chunk_layout_version(false, small), 4);
20465 assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20466 writer.close().unwrap();
20467 std::fs::remove_file(&path).ok();
20468
20469 let path = temp_path("chunk_ver_v18");
20470 let mut writer = Hdf5Writer::create(&path).unwrap();
20471 writer.set_libver_bound(LibverBound::V18).unwrap();
20472 assert_eq!(writer.chunk_layout_version(false, small), 3);
20473 assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20474 writer.close().unwrap();
20475 std::fs::remove_file(&path).ok();
20476
20477 let path = temp_path("chunk_ver_earliest");
20478 let mut writer = Hdf5Writer::create(&path).unwrap();
20479 writer.set_libver_bound(LibverBound::Earliest).unwrap();
20480 assert_eq!(
20481 writer.chunk_layout_version(false, small),
20482 LAYOUT_VERSION_DEFAULT
20483 );
20484 assert_eq!(writer.chunk_layout_version(false, over_4gib), 5);
20485 writer.close().unwrap();
20486 std::fs::remove_file(&path).ok();
20487 }
20488 /// `fsm_persist.h5` persists two managers — metadata and raw data. The
20489 /// reopen reads both, hands their merged sections to the allocator, and
20490 /// claims the four blocks the managers themselves occupy.
20491 #[test]
20492 fn a_persisting_file_reopens_with_its_free_sections() {
20493 let path = fixture_copy("fsm_persist.h5", "fsm_read");
20494 let writer = Hdf5Writer::open_append(&path).unwrap();
20495 let fs = writer.free_space.as_deref().expect("managers were read");
20496
20497 assert!(fs.info.persist);
20498 assert_eq!(fs.info.strategy, FileSpaceStrategy::FsmAggr);
20499 assert_eq!(fs.info.threshold, 1);
20500
20501 let sections = writer.allocator.free_blocks();
20502 // h5stat -S reports 1910 bytes of tracked free space for this file.
20503 assert_eq!(sections.iter().map(|s| s.1).sum::<u64>(), 1910);
20504 // Address-ordered, and no two sections touch: what the two managers
20505 // held separately came out coalesced.
20506 for w in sections.windows(2) {
20507 assert!(w[0].0 + w[0].1 < w[1].0, "{sections:?}");
20508 }
20509 // Two headers plus the two sections blocks they name.
20510 assert_eq!(fs.superseded.len(), 4);
20511 for &(addr, len) in &fs.superseded {
20512 assert!(len > 0);
20513 assert!(
20514 !sections
20515 .iter()
20516 .any(|&(a, l)| addr < a + l && a < addr + len),
20517 "manager block {addr:#x}+{len} sits in a free section"
20518 );
20519 }
20520 drop(writer);
20521 let _ = std::fs::remove_file(&path);
20522 }
20523
20524 /// A file created with non-default file-space properties carries the
20525 /// message that declares them, and one created to persist gets real
20526 /// managers as soon as anything is freed.
20527 #[test]
20528 fn a_created_file_declares_the_strategy_it_was_made_with() {
20529 let path = temp_path("fsm_create");
20530 {
20531 let w = Hdf5Writer::create_with_options(
20532 &path,
20533 FileCreateOptions {
20534 file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
20535 ..Default::default()
20536 },
20537 )
20538 .unwrap();
20539 let i = w
20540 .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20541 .unwrap();
20542 w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20543 w.close().unwrap();
20544 }
20545
20546 let info = read_only_append(&path)
20547 .free_space
20548 .as_deref()
20549 .expect("the created file declares a strategy")
20550 .info
20551 .clone();
20552 assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
20553 assert!(info.persist);
20554 assert_eq!(info.threshold, 1);
20555 assert_eq!(info.page_size, 4096);
20556 // The alignment fragments the creation left behind are the file's
20557 // first free space, so the metadata manager already has an address
20558 // and the raw-data one, which nothing freed into, does not.
20559 assert_ne!(info.fs_addr[0], UNDEF_ADDR);
20560 assert!(info.fs_addr.iter().skip(1).all(|&a| a == UNDEF_ADDR));
20561
20562 // An append supersedes the root header and the extension, and that
20563 // freed space is what the managers now record.
20564 append_one(&path, "added", false);
20565 assert!(
20566 tracked_free_space(&path) > 0,
20567 "the append recorded no free space"
20568 );
20569 let _ = std::fs::remove_file(&path);
20570 }
20571
20572 /// The two strategies without managers, and the default. All three are
20573 /// `H5Pset_file_space_strategy` settings; only the default leaves the file
20574 /// without the message.
20575 #[test]
20576 fn a_strategy_without_managers_still_declares_itself() {
20577 for (strategy, persist) in [
20578 (FileSpaceStrategy::Aggr, true),
20579 (FileSpaceStrategy::None, false),
20580 ] {
20581 let path = temp_path("fsm_nomgr");
20582 {
20583 let w = Hdf5Writer::create_with_options(
20584 &path,
20585 FileCreateOptions {
20586 file_space: FileSpaceConfig::new(strategy, persist, 7),
20587 ..Default::default()
20588 },
20589 )
20590 .unwrap();
20591 w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20592 .unwrap();
20593 w.close().unwrap();
20594 }
20595 // Read through the reader, not the writer: a reopen only builds
20596 // free-space state for a file it will rewrite managers for, and
20597 // these two have none.
20598 let info = declared_file_space(&path).expect("the strategy is declared");
20599 assert_eq!(info.strategy, strategy);
20600 // `H5P__set_file_space_strategy` stores neither for a strategy
20601 // that has no managers, so both keep the library defaults.
20602 assert!(!info.persist);
20603 assert_eq!(info.threshold, 1);
20604 let _ = std::fs::remove_file(&path);
20605 }
20606 }
20607
20608 /// The library defaults are what a file says by saying nothing.
20609 #[test]
20610 fn the_default_strategy_writes_no_message() {
20611 let path = temp_path("fsm_default");
20612 {
20613 let w = Hdf5Writer::create_with_options(
20614 &path,
20615 FileCreateOptions {
20616 file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, false, 1),
20617 ..Default::default()
20618 },
20619 )
20620 .unwrap();
20621 w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20622 .unwrap();
20623 w.close().unwrap();
20624 }
20625 assert!(declared_file_space(&path).is_none());
20626 let _ = std::fs::remove_file(&path);
20627 }
20628
20629 /// The file-space info message a file carries, read back the way any
20630 /// reader sees it.
20631 fn declared_file_space(path: &std::path::Path) -> Option<FileSpaceInfoMessage> {
20632 crate::io::reader::Hdf5Reader::open(path)
20633 .unwrap()
20634 .superblock_extension()
20635 .file_space_info
20636 .clone()
20637 }
20638
20639 /// A created paged file is laid out on its page grid: the superblock takes
20640 /// the whole of page zero and the rest of that page is the metadata
20641 /// manager's first section, which is what `H5MF__alloc_pagefs` gives
20642 /// `H5F__super_init`'s `H5MF_alloc(f, H5FD_MEM_SUPER, ...)`.
20643 #[test]
20644 fn a_created_paged_file_lays_its_pages_out() {
20645 let path = temp_path("fsm_paged_created");
20646 {
20647 let w = Hdf5Writer::create_with_options(
20648 &path,
20649 FileCreateOptions {
20650 file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
20651 ..Default::default()
20652 },
20653 )
20654 .unwrap();
20655 let i = w
20656 .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20657 .unwrap();
20658 w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20659 w.close().unwrap();
20660 }
20661 let info = read_only_append(&path)
20662 .free_space
20663 .as_deref()
20664 .expect("the created file declares a strategy")
20665 .info
20666 .clone();
20667 assert_eq!(info.strategy, FileSpaceStrategy::Page);
20668 assert!(info.persist);
20669 assert_eq!(info.page_size, 4096);
20670 assert_eq!(
20671 std::fs::metadata(&path).unwrap().len() % info.page_size,
20672 0,
20673 "a paged file ends on a page boundary"
20674 );
20675 let _ = std::fs::remove_file(&path);
20676 }
20677
20678 /// A userblock has to be a whole number of pages, or every page boundary
20679 /// after it is off the file's own grid — `H5F__super_init` refuses one
20680 /// that is not (H5Fsuper.c:1182-1192).
20681 #[test]
20682 fn a_paged_file_refuses_a_userblock_smaller_than_its_page() {
20683 let path = temp_path("fsm_paged_userblock");
20684 let Err(err) = Hdf5Writer::create_with_options(
20685 &path,
20686 FileCreateOptions {
20687 file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1),
20688 userblock: 512,
20689 ..Default::default()
20690 },
20691 ) else {
20692 panic!("a 512-byte userblock was accepted on a 4096-byte page");
20693 };
20694 assert!(
20695 format!("{err}").contains("multiple of its 4096-byte"),
20696 "{err}"
20697 );
20698 let _ = std::fs::remove_file(&path);
20699 }
20700
20701 /// A page size the builder names is the page the file is actually laid
20702 /// out in, not just a number the message repeats: every allocation is
20703 /// shaped by it and the file ends on one of its boundaries.
20704 #[test]
20705 fn a_file_created_at_a_non_default_page_size_allocates_by_it() {
20706 let path = temp_path("fsm_page_size_8k");
20707 {
20708 let w = Hdf5Writer::create_with_options(
20709 &path,
20710 FileCreateOptions {
20711 file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20712 .with_page_size(8192),
20713 ..Default::default()
20714 },
20715 )
20716 .unwrap();
20717 let i = w
20718 .create_dataset("keep", DatatypeMessage::i32_type(), &[8])
20719 .unwrap();
20720 w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20721 w.close().unwrap();
20722 }
20723 let info = read_only_append(&path)
20724 .free_space
20725 .as_deref()
20726 .expect("the created file declares a strategy")
20727 .info
20728 .clone();
20729 assert_eq!(info.page_size, 8192);
20730 assert_eq!(
20731 std::fs::metadata(&path).unwrap().len() % 8192,
20732 0,
20733 "the file ends on one of the pages it was created with"
20734 );
20735 let _ = std::fs::remove_file(&path);
20736 }
20737
20738 /// The page size is the fourth of the four properties `H5F__super_init`
20739 /// compares against the library defaults (H5Fsuper.c:1092-1097), so
20740 /// naming it is on its own enough to give a file the message — under the
20741 /// default strategy, which allocates without it.
20742 #[test]
20743 fn a_non_default_page_size_alone_gives_the_file_a_message() {
20744 let path = temp_path("fsm_page_size_only");
20745 {
20746 let w = Hdf5Writer::create_with_options(
20747 &path,
20748 FileCreateOptions {
20749 file_space: FileSpaceConfig::default().with_page_size(1024),
20750 ..Default::default()
20751 },
20752 )
20753 .unwrap();
20754 w.close().unwrap();
20755 }
20756 let info = declared_file_space(&path)
20757 .expect("a file naming only a page size still carries the message");
20758 assert_eq!(info.strategy, FileSpaceStrategy::FsmAggr);
20759 assert!(!info.persist);
20760 assert_eq!(info.page_size, 1024);
20761 let _ = std::fs::remove_file(&path);
20762 }
20763
20764 /// `H5Pset_file_space_page_size` refuses anything below 512 or above
20765 /// 1 GiB (H5Pfcpl.c:1389-1393), and nothing between: no power of two is
20766 /// required, so a size the bounds admit is one the file may carry.
20767 #[test]
20768 fn a_page_size_outside_the_library_bounds_is_refused() {
20769 for size in [0, 1, 511, PAGE_SIZE_MAX + 1] {
20770 let path = temp_path(&format!("fsm_page_size_bad_{size}"));
20771 let Err(err) = Hdf5Writer::create_with_options(
20772 &path,
20773 FileCreateOptions {
20774 file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20775 .with_page_size(size),
20776 ..Default::default()
20777 },
20778 ) else {
20779 panic!("a {size}-byte file-space page was accepted");
20780 };
20781 assert!(
20782 format!("{err}").contains("between 512 bytes and 1073741824"),
20783 "{err}"
20784 );
20785 let _ = std::fs::remove_file(&path);
20786 }
20787 let path = temp_path("fsm_page_size_odd");
20788 let w = Hdf5Writer::create_with_options(
20789 &path,
20790 FileCreateOptions {
20791 file_space: FileSpaceConfig::new(FileSpaceStrategy::Page, true, 1)
20792 .with_page_size(513),
20793 ..Default::default()
20794 },
20795 )
20796 .expect("513 is inside the bounds, and no power of two is required");
20797 w.close().unwrap();
20798 let _ = std::fs::remove_file(&path);
20799 }
20800
20801 /// A paged file's managers are read on reopen, the same as any other
20802 /// file's: paged aggregation changes which manager a request maps to, not
20803 /// whether the file has managers to rewrite.
20804 #[test]
20805 fn a_paged_file_reports_the_managers_it_persists() {
20806 let path = fixture_copy("fsm_persist_page.h5", "fsm_read_paged");
20807 let writer = Hdf5Writer::open_append(&path).unwrap();
20808 let fs = writer.free_space.as_deref().expect("no managers read");
20809 assert_eq!(fs.info.strategy, FileSpaceStrategy::Page);
20810 assert!(
20811 !writer.allocator.free_extents().is_empty(),
20812 "the sections the file records were not put back in circulation"
20813 );
20814 drop(writer);
20815 let _ = std::fs::remove_file(&path);
20816 }
20817
20818 /// A file with no file-space info message at all — every file this crate
20819 /// creates — has nothing to read and nothing to write back.
20820 #[test]
20821 fn a_file_without_a_strategy_has_no_managers() {
20822 let path = temp_path("fsm_none");
20823 {
20824 let w = Hdf5Writer::create(&path).unwrap();
20825 w.create_dataset("d", DatatypeMessage::f64_type(), &[4])
20826 .unwrap();
20827 w.close().unwrap();
20828 }
20829 let writer = Hdf5Writer::open_append(&path).unwrap();
20830 assert!(writer.free_space.is_none());
20831 drop(writer);
20832 let _ = std::fs::remove_file(&path);
20833 }
20834 /// Sum of the sections the managers a file names actually hold — what
20835 /// `h5stat -S` prints as "Amount of tracked free space", read back through
20836 /// this crate's own decoder so a test can assert on it. A reopen seeds the
20837 /// allocator with exactly those sections, so its free list is the number.
20838 fn tracked_free_space(path: &std::path::Path) -> u64 {
20839 read_only_append(path)
20840 .allocator
20841 .free_blocks()
20842 .iter()
20843 .map(|b| b.1)
20844 .sum()
20845 }
20846
20847 /// Open for append and mark the writer closed, so dropping it releases the
20848 /// file lock instead of finalizing and rewriting what is being inspected.
20849 fn read_only_append(path: &std::path::Path) -> Hdf5Writer {
20850 let mut w = Hdf5Writer::open_append(path).unwrap();
20851 w.closed = true;
20852 w
20853 }
20854
20855 /// Add one small dataset, the smallest append that still rewrites the root
20856 /// header, the superblock extension and — on a persisting file — the
20857 /// free-space manager.
20858 fn append_one(path: &std::path::Path, name: &str, disable_managers: bool) {
20859 let mut w = Hdf5Writer::open_append(path).unwrap();
20860 if disable_managers {
20861 // Both halves of the change, so the control is the file as this
20862 // crate wrote it before: the session neither allocates from the
20863 // recorded sections nor writes any back.
20864 w.free_space = None;
20865 w.allocator.reset_free_list(&[]);
20866 }
20867 let i = w
20868 .create_dataset(name, DatatypeMessage::i32_type(), &[8])
20869 .unwrap();
20870 w.write_dataset_raw(
20871 i,
20872 &(0..8i32).flat_map(|v| v.to_le_bytes()).collect::<Vec<u8>>(),
20873 )
20874 .unwrap();
20875 w.close().unwrap();
20876 }
20877
20878 /// The block list a reopen carries for the superblock extension covers
20879 /// every chunk of the header, not just the first. The fixture's extension
20880 /// is a two-chunk header — libhdf5 put the file-space info message in a
20881 /// continuation — and freeing chunk zero alone left the continuation
20882 /// allocated with nothing naming it.
20883 #[test]
20884 fn a_reopen_carries_every_chunk_of_the_superblock_extension() {
20885 let path = fixture_copy("fsm_persist.h5", "fsm_ext_chunks");
20886 let blocks = read_only_append(&path).extension.superseded.clone();
20887 assert!(
20888 blocks.len() > 1,
20889 "the fixture's extension is one chunk, so this proves nothing: {blocks:?}"
20890 );
20891 let _ = std::fs::remove_file(&path);
20892 }
20893
20894 /// An append on a persisting file both spends and records the space its
20895 /// managers track: the new dataset comes out of the sections the file
20896 /// already had, and what the rewrite frees goes back into them.
20897 #[test]
20898 fn an_append_reuses_and_records_the_space_the_managers_track() {
20899 let path = fixture_copy("fsm_persist.h5", "fsm_write");
20900 let original = std::fs::metadata(&path).unwrap().len();
20901 let before = tracked_free_space(&path);
20902 assert_eq!(before, 1910, "the fixture's own managers");
20903
20904 append_one(&path, "added", false);
20905 let size = std::fs::metadata(&path).unwrap().len();
20906 let tracked = tracked_free_space(&path);
20907
20908 // Negative control: the same append with both halves of this off — no
20909 // allocating out of the recorded sections and no writing any back —
20910 // which is what this crate did before it read free space at all.
20911 let control = fixture_copy("fsm_persist.h5", "fsm_write_control");
20912 append_one(&control, "added", true);
20913 let control_size = std::fs::metadata(&control).unwrap().len();
20914 assert_eq!(
20915 tracked_free_space(&control),
20916 before,
20917 "with the manager rewrite disabled the number must not move"
20918 );
20919
20920 // The new dataset's raw data comes out of the raw-data sections the
20921 // file already recorded, so the append grows the file by less than the
20922 // same append with the reuse off. It does not stop the growth:
20923 // `H5MF_alloc` asks one manager and no other, and of this fixture's
20924 // 1910 free bytes 1848 are raw-data ones, so the metadata the append
20925 // writes still comes from the end of the file.
20926 assert!(
20927 size < control_size,
20928 "the append took nothing from the {before} bytes free: \
20929 {original} grew to {size}, the control to {control_size}"
20930 );
20931 assert!(
20932 control_size > original,
20933 "the control has to grow or it proves nothing"
20934 );
20935 // Space no manager and no object claims — `h5stat -S`'s "unaccounted
20936 // space" — is what the leak was, and it is smaller now.
20937 assert!(
20938 size - tracked < control_size - before,
20939 "unaccounted space went from {} to {}",
20940 control_size - before,
20941 size - tracked
20942 );
20943
20944 for p in [&path, &control] {
20945 let _ = std::fs::remove_file(p);
20946 }
20947 }
20948
20949 /// The set the writer holds free when it finishes is exactly the set the
20950 /// manager it just wrote records — the invariant that makes the on-disk
20951 /// managers a faithful account of the file's free space.
20952 #[test]
20953 fn the_manager_records_the_free_list_the_close_ends_with() {
20954 let path = fixture_copy("fsm_persist.h5", "fsm_roundtrip");
20955 let internal = {
20956 let mut w = Hdf5Writer::open_append(&path).unwrap();
20957 let i = w
20958 .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20959 .unwrap();
20960 w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20961 w.finalize(true).unwrap();
20962 let blocks = w.allocator.free_extents();
20963 w.closed = true;
20964 blocks
20965 };
20966 assert!(!internal.is_empty(), "the append freed nothing");
20967
20968 // Classes included: a section read back out of the wrong manager is a
20969 // section libhdf5 would offer to the wrong kind of allocation.
20970 let reread = {
20971 let w = read_only_append(&path);
20972 assert!(w.free_space.is_some(), "managers were written");
20973 w.allocator.free_extents()
20974 };
20975 assert_eq!(internal, reread);
20976 let _ = std::fs::remove_file(&path);
20977 }
20978
20979 /// The paged half of
20980 /// [`the_manager_records_the_free_list_the_close_ends_with`]: a paged
20981 /// file's sections carry a page and a class as well as an address, and a
20982 /// section written into the wrong manager or split across a page boundary
20983 /// would come back different.
20984 #[test]
20985 fn the_manager_records_the_free_list_a_paged_close_ends_with() {
20986 let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_roundtrip");
20987 let internal = {
20988 let mut w = Hdf5Writer::open_append(&path).unwrap();
20989 let i = w
20990 .create_dataset("added", DatatypeMessage::i32_type(), &[8])
20991 .unwrap();
20992 w.write_dataset_raw(i, &[0u8; 32]).unwrap();
20993 w.finalize(true).unwrap();
20994 let blocks = w.allocator.free_extents();
20995 w.closed = true;
20996 blocks
20997 };
20998 assert!(!internal.is_empty(), "the append freed nothing");
20999
21000 let reread = {
21001 let w = read_only_append(&path);
21002 assert!(w.free_space.is_some(), "managers were written");
21003 w.allocator.free_extents()
21004 };
21005 assert_eq!(internal, reread);
21006 let _ = std::fs::remove_file(&path);
21007 }
21008
21009 /// Negative control for the paged managers: with the read and the rewrite
21010 /// both off — the file as this crate handled a paged file before — the
21011 /// space the append frees is recorded nowhere, and the number this crate
21012 /// reads back is the fixture's own.
21013 #[test]
21014 fn a_paged_append_records_nothing_without_the_manager_rewrite() {
21015 let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_measured");
21016 let control = fixture_copy("fsm_persist_page.h5", "fsm_paged_control");
21017 let before = tracked_free_space(&path);
21018 let original = std::fs::metadata(&path).unwrap().len();
21019
21020 append_one(&path, "added", false);
21021 append_one(&control, "added", true);
21022
21023 assert_eq!(
21024 tracked_free_space(&control),
21025 before,
21026 "the control moved the number it is there to hold still"
21027 );
21028 assert_eq!(
21029 std::fs::metadata(&path).unwrap().len(),
21030 original,
21031 "the append grew a paged file with {before} bytes recorded free"
21032 );
21033 assert!(
21034 std::fs::metadata(&control).unwrap().len() > original,
21035 "the control has to grow or it proves nothing"
21036 );
21037 assert_ne!(
21038 tracked_free_space(&path),
21039 before,
21040 "the managers came back holding what the fixture wrote"
21041 );
21042 for p in [&path, &control] {
21043 let _ = std::fs::remove_file(p);
21044 }
21045 }
21046
21047 /// A block released from a dataset's raw data is recorded by the manager
21048 /// `H5MF_ALLOC_TO_FS_AGGR_TYPE` maps `H5FD_MEM_DRAW` to, and nothing else
21049 /// is: the dichotomy the sec2 driver installs is what decides, and the two
21050 /// managers it collapses to are the file-space info message's slots 0 and
21051 /// 2.
21052 #[test]
21053 fn a_released_raw_block_lands_in_the_raw_data_manager() {
21054 let path = temp_path("fsm_dichotomy");
21055 {
21056 let w = Hdf5Writer::create_with_options(
21057 &path,
21058 FileCreateOptions {
21059 file_space: FileSpaceConfig::new(FileSpaceStrategy::FsmAggr, true, 1),
21060 ..Default::default()
21061 },
21062 )
21063 .unwrap();
21064 let i = w
21065 .create_dataset("bulk", DatatypeMessage::i32_type(), &[256])
21066 .unwrap();
21067 w.write_dataset_raw(i, &vec![0u8; 1024]).unwrap();
21068 w.create_dataset("keep", DatatypeMessage::i32_type(), &[8])
21069 .unwrap();
21070 w.close().unwrap();
21071 }
21072 let (raw_addr, raw_len) = {
21073 let w = read_only_append(&path);
21074 let i = w.dataset_index("bulk").unwrap();
21075 let ds = w.ds(i);
21076 let m = ds.lock();
21077 (m.data_addr, m.data_size)
21078 };
21079 assert!(raw_len >= 1024, "the raw block is {raw_len} bytes");
21080 {
21081 let w = Hdf5Writer::open_append(&path).unwrap();
21082 w.delete_dataset("bulk").unwrap();
21083 w.close().unwrap();
21084 }
21085
21086 let mut w = read_only_append(&path);
21087 let info = w
21088 .free_space
21089 .as_deref()
21090 .expect("the file persists managers")
21091 .info
21092 .clone();
21093 assert_ne!(info.fs_addr[0], UNDEF_ADDR, "no metadata manager");
21094 assert_ne!(info.fs_addr[2], UNDEF_ADDR, "no raw-data manager");
21095 for (slot, &addr) in info.fs_addr.iter().enumerate() {
21096 if slot != 0 && slot != 2 {
21097 assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
21098 }
21099 }
21100
21101 let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21102 let inside = |b: &FreeBlock| b.addr >= raw_addr && b.addr + b.len <= raw_addr + raw_len;
21103 let raw: Vec<&FreeBlock> = found
21104 .sections
21105 .iter()
21106 .filter(|b| b.manager == FreeSpaceManager::RawData)
21107 .collect();
21108 assert!(
21109 !raw.is_empty(),
21110 "the deleted dataset's bytes were not recorded"
21111 );
21112 assert!(
21113 raw.iter().all(|b| inside(b)),
21114 "a raw-data section is outside the deleted dataset's block: {raw:?}"
21115 );
21116 assert!(
21117 found
21118 .sections
21119 .iter()
21120 .filter(|b| b.manager == FreeSpaceManager::Metadata)
21121 .all(|b| !inside(b)),
21122 "raw-data bytes were recorded by the metadata manager"
21123 );
21124 drop(w);
21125 let _ = std::fs::remove_file(&path);
21126 }
21127
21128 /// A reopened paged file's managers are this writer's to rewrite, and the
21129 /// three the sec2 driver can reach are the only ones it names.
21130 ///
21131 /// `H5MF__alloc_to_fs_type` (H5MF.c:265) sends a request of at least one
21132 /// page to `H5F_MEM_PAGE_GENERIC` unless the driver declares
21133 /// `H5FD_FEAT_PAGED_AGGR`, which only the multi and split drivers do, so a
21134 /// sec2 file has the dichotomy's two small managers and that one large
21135 /// one: message slots 0, 2 and 6.
21136 #[test]
21137 fn a_paged_file_names_only_the_managers_sec2_can_reach() {
21138 let path = fixture_copy("fsm_persist_page.h5", "fsm_write_paged");
21139 assert!(
21140 read_only_append(&path).free_space.is_some(),
21141 "the paged fixture's managers were not read"
21142 );
21143 append_one(&path, "added", false);
21144
21145 let mut w = read_only_append(&path);
21146 let info = w
21147 .free_space
21148 .as_deref()
21149 .expect("the file persists managers")
21150 .info
21151 .clone();
21152 assert_eq!(info.strategy, FileSpaceStrategy::Page);
21153 for (slot, &addr) in info.fs_addr.iter().enumerate() {
21154 if !matches!(slot, 0 | 2 | 6) {
21155 assert_eq!(addr, UNDEF_ADDR, "slot {slot} names a manager");
21156 }
21157 }
21158 assert!(
21159 info.fs_addr.iter().any(|&a| a != UNDEF_ADDR),
21160 "the rewritten file records nothing free"
21161 );
21162 crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21163 drop(w);
21164 let _ = std::fs::remove_file(&path);
21165 }
21166
21167 /// Every section a paged file records sits inside one page, and the pages
21168 /// its small managers use are pages of their own kind — the invariant
21169 /// `H5MF__alloc_pagefs` maintains by giving each small request a whole
21170 /// page of its class and recording the rest of it in that class's manager.
21171 #[test]
21172 fn a_paged_files_small_sections_stay_inside_one_page_of_one_kind() {
21173 let path = fixture_copy("fsm_persist_page.h5", "fsm_paged_pages");
21174 append_one(&path, "added", false);
21175
21176 let mut w = read_only_append(&path);
21177 let info = w
21178 .free_space
21179 .as_deref()
21180 .expect("the file persists managers")
21181 .info
21182 .clone();
21183 let page = info.page_size;
21184 let found = crate::io::free_space_io::read_managers(&mut w.handle, &w.ctx, &info).unwrap();
21185 let mut kind_of_page: std::collections::HashMap<u64, FreeSpaceManager> =
21186 std::collections::HashMap::new();
21187 for section in &found.sections {
21188 if section.manager == FreeSpaceManager::Large {
21189 continue;
21190 }
21191 assert_eq!(
21192 section.addr / page,
21193 (section.addr + section.len - 1) / page,
21194 "the section at {:#x} crosses a page boundary",
21195 section.addr
21196 );
21197 let owner = kind_of_page
21198 .entry(section.addr / page)
21199 .or_insert(section.manager);
21200 assert_eq!(
21201 *owner,
21202 section.manager,
21203 "page {} holds sections of two kinds",
21204 section.addr / page
21205 );
21206 }
21207 drop(w);
21208 let _ = std::fs::remove_file(&path);
21209 }
21210}