Skip to main content

rust_hdf5/format/messages/
attribute.rs

1//! Attribute message (type 0x0C) -- describes an attribute attached to an object.
2//!
3//! Binary layout (version 3, no shared datatypes):
4//!   Byte 0:    version = 3
5//!   Byte 1:    flags (0 for non-shared)
6//!   Bytes 2-3: name_size (u16 LE, including null terminator)
7//!   Bytes 4-5: datatype_size (u16 LE)
8//!   Bytes 6-7: dataspace_size (u16 LE)
9//!   Byte 8:    name character set encoding (0=ASCII, 1=UTF-8)
10//!   <name: name_size bytes, null-terminated>
11//!   <encoded datatype message: datatype_size bytes>
12//!   <encoded dataspace message: dataspace_size bytes>
13//!   <raw attribute data>
14
15use crate::format::messages::dataspace::DataspaceMessage;
16use crate::format::messages::datatype::DatatypeMessage;
17use crate::format::{FormatContext, FormatError, FormatResult, LibverBound};
18
19const ATTR_VERSION: u8 = 3;
20
21/// `H5O_ATTR_FLAG_TYPE_SHARED` (H5Oattr.c:88): the datatype field holds a
22/// shared-message pointer rather than the datatype message.
23pub const ATTR_FLAG_TYPE_SHARED: u8 = 0x01;
24/// `H5O_ATTR_FLAG_SPACE_SHARED` (H5Oattr.c:89), the same for the dataspace.
25pub const ATTR_FLAG_SPACE_SHARED: u8 = 0x02;
26
27/// One attribute message body, and where its datatype and dataspace fields
28/// sit inside it.
29///
30/// The offsets are what a caller that put a shared-message pointer in either
31/// field needs in order to fill the pointer's heap ID in later, once the heap
32/// it points into has been laid out.
33#[derive(Debug, Clone, PartialEq, Eq)]
34pub struct EncodedAttribute {
35    /// The message payload.
36    pub body: Vec<u8>,
37    /// Offset of the datatype field in `body`.
38    pub datatype_at: usize,
39    /// Offset of the dataspace field in `body`.
40    pub dataspace_at: usize,
41}
42
43/// An HDF5 attribute message.
44#[derive(Debug, Clone, PartialEq)]
45pub struct AttributeMessage {
46    /// Attribute name.
47    pub name: String,
48    /// Datatype of the attribute value.
49    pub datatype: DatatypeMessage,
50    /// Dataspace (scalar or simple).
51    pub dataspace: DataspaceMessage,
52    /// Raw attribute value data.
53    pub data: Vec<u8>,
54}
55
56impl AttributeMessage {
57    /// Create a scalar string attribute with the given name and value.
58    ///
59    /// Uses a null-terminated UTF-8 fixed-length string datatype with
60    /// size = value.len() + 1 (for the null terminator), and a scalar
61    /// dataspace.
62    pub fn scalar_string(name: &str, value: &str) -> Self {
63        let str_size = (value.len() + 1) as u32; // +1 for null terminator
64        let datatype = DatatypeMessage::fixed_string_utf8(str_size);
65        let dataspace = DataspaceMessage::scalar();
66
67        // Data: string bytes + null terminator
68        let mut data = Vec::with_capacity(str_size as usize);
69        data.extend_from_slice(value.as_bytes());
70        data.push(0); // null terminator
71
72        Self {
73            name: name.to_string(),
74            datatype,
75            dataspace,
76            data,
77        }
78    }
79
80    /// Create a scalar numeric attribute with raw bytes as value.
81    pub fn scalar_numeric(name: &str, datatype: DatatypeMessage, data: Vec<u8>) -> Self {
82        Self {
83            name: name.to_string(),
84            datatype,
85            dataspace: DataspaceMessage::scalar(),
86            data,
87        }
88    }
89
90    /// Create a numeric array attribute with a simple dataspace.
91    ///
92    /// `dims` are the dimension sizes (e.g. `&[3]` for the 1-D array
93    /// attributes AreaDetector writes). `data` is the row-major raw bytes and
94    /// must hold `product(dims) * datatype.element_size()` bytes — the caller
95    /// owns that invariant. An empty `dims` yields a scalar dataspace; prefer
96    /// [`Self::scalar_numeric`] for that case.
97    pub fn array_numeric(
98        name: &str,
99        datatype: DatatypeMessage,
100        dims: &[u64],
101        data: Vec<u8>,
102    ) -> Self {
103        debug_assert_eq!(
104            data.len() as u64,
105            dims.iter().product::<u64>() * datatype.element_size() as u64,
106            "array_numeric data length must equal product(dims) * element_size"
107        );
108        Self {
109            name: name.to_string(),
110            datatype,
111            dataspace: DataspaceMessage::simple(dims),
112            data,
113        }
114    }
115
116    /// Encode the attribute message into a byte vector.
117    ///
118    /// The result is the raw payload for an object header message of type
119    /// 0x0C (MSG_ATTRIBUTE). It does NOT include the object header message
120    /// envelope (type, size, flags bytes); that is handled by the caller.
121    pub fn encode(&self, ctx: &FormatContext) -> Vec<u8> {
122        self.encode_at(ctx, LibverBound::Earliest)
123    }
124
125    /// Encode the attribute message for a file whose low libver bound is
126    /// `libver`, which the datatype message inside it follows.
127    pub fn encode_at(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
128        self.encode_for(ctx, libver, crate::format::ObjectFormat::Modern)
129    }
130
131    /// Encode a version-1 attribute message (`H5O__attr_encode`, H5Oattr.c).
132    ///
133    /// Version 1 has no flags byte and no name character set: byte 1 is
134    /// reserved, and the three size fields are followed by the name, the
135    /// datatype and the dataspace each padded out to a multiple of eight
136    /// bytes. The size fields record the *unpadded* lengths, so a decoder that
137    /// forgets the padding walks into the middle of the next field — which is
138    /// why the version is not something a writer may pick freely.
139    fn encode_v1(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
140        /// `H5O_ALIGN_OLD`, which version 1 of this message applies to each of
141        /// its three variable-length fields.
142        fn pad_to_8(buf: &mut Vec<u8>) {
143            let padded = (buf.len() + 7) & !7;
144            buf.resize(padded, 0);
145        }
146
147        let encoded_dt = self.datatype.encode_at(ctx, libver);
148        let encoded_ds = self
149            .dataspace
150            .encode_for(ctx, crate::format::ObjectFormat::Legacy);
151        let name_bytes = self.name.as_bytes();
152        let name_size = name_bytes.len() + 1;
153
154        let mut buf = Vec::with_capacity(8 + name_size + encoded_dt.len() + encoded_ds.len() + 24);
155        buf.push(1); // version
156        buf.push(0); // reserved
157        buf.extend_from_slice(&(name_size as u16).to_le_bytes());
158        buf.extend_from_slice(&(encoded_dt.len() as u16).to_le_bytes());
159        buf.extend_from_slice(&(encoded_ds.len() as u16).to_le_bytes());
160        buf.extend_from_slice(name_bytes);
161        buf.push(0);
162        pad_to_8(&mut buf);
163        buf.extend_from_slice(&encoded_dt);
164        pad_to_8(&mut buf);
165        buf.extend_from_slice(&encoded_ds);
166        pad_to_8(&mut buf);
167        buf.extend_from_slice(&self.data);
168        buf
169    }
170
171    /// Encode the attribute message at the version a file of this `format`
172    /// calls for, with the datatype inside it at `libver`.
173    pub fn encode_for(
174        &self,
175        ctx: &FormatContext,
176        libver: LibverBound,
177        format: crate::format::ObjectFormat,
178    ) -> Vec<u8> {
179        if format.attribute_version() == 1 {
180            return self.encode_v1(ctx, libver);
181        }
182        let encoded_dt = self.datatype.encode_at(ctx, libver);
183        let encoded_ds = self.dataspace.encode_for(ctx, format);
184        self.encode_with_fields(0x00, &encoded_dt, &encoded_ds).body
185    }
186
187    /// The version-3 body with its datatype and dataspace fields supplied.
188    ///
189    /// `H5O__attr_encode` writes each of the two through its message class's
190    /// encoder, which is the *shared* encoder when that piece is a shared
191    /// message — the field then holds a `H5O_shared_t` and the attribute's own
192    /// flags byte says so (`H5O_ATTR_FLAG_TYPE_SHARED` /
193    /// `H5O_ATTR_FLAG_SPACE_SHARED`, H5Oattr.c:358-359). Whichever it is, the
194    /// size fields record what is actually stored, so the caller supplies the
195    /// bytes and the matching flag bits and this lays the message out around
196    /// them.
197    pub fn encode_with_fields(
198        &self,
199        flags: u8,
200        datatype: &[u8],
201        dataspace: &[u8],
202    ) -> EncodedAttribute {
203        // Name with null terminator
204        let name_bytes = self.name.as_bytes();
205        let name_size = name_bytes.len() + 1; // +1 for null terminator
206
207        // Total: 9 (header) + name_size + datatype_size + dataspace_size + data_size
208        let total = 9 + name_size + datatype.len() + dataspace.len() + self.data.len();
209        let mut buf = Vec::with_capacity(total);
210
211        // Byte 0: version
212        buf.push(ATTR_VERSION);
213
214        // Byte 1: flags — which of the two fields below is a shared pointer.
215        buf.push(flags);
216
217        // Bytes 2-3: name size (u16 LE)
218        buf.extend_from_slice(&(name_size as u16).to_le_bytes());
219
220        // Bytes 4-5: datatype size (u16 LE)
221        buf.extend_from_slice(&(datatype.len() as u16).to_le_bytes());
222
223        // Bytes 6-7: dataspace size (u16 LE)
224        buf.extend_from_slice(&(dataspace.len() as u16).to_le_bytes());
225
226        // Byte 8: name character set encoding (1 = UTF-8)
227        buf.push(0x01);
228
229        // Name (null-terminated)
230        buf.extend_from_slice(name_bytes);
231        buf.push(0x00);
232
233        let datatype_at = buf.len();
234        buf.extend_from_slice(datatype);
235
236        let dataspace_at = buf.len();
237        buf.extend_from_slice(dataspace);
238
239        // Raw data
240        buf.extend_from_slice(&self.data);
241
242        debug_assert_eq!(buf.len(), total);
243        EncodedAttribute {
244            body: buf,
245            datatype_at,
246            dataspace_at,
247        }
248    }
249
250    /// Decode an attribute message from a byte buffer.
251    ///
252    /// Supports versions 1, 2, and 3:
253    /// - v1: 8-byte header, each field padded to 8-byte alignment
254    /// - v2: 8-byte header, no alignment padding
255    /// - v3: 9-byte header (adds charset byte), no alignment padding
256    pub fn decode(buf: &[u8], ctx: &FormatContext) -> FormatResult<(Self, usize)> {
257        let AttributeHeader {
258            name,
259            datatype_size,
260            dataspace_size,
261            align,
262            mut pos,
263        } = AttributeHeader::decode(buf)?;
264
265        // Datatype
266        let needed = pos + datatype_size;
267        if buf.len() < needed {
268            return Err(FormatError::BufferTooShort {
269                needed,
270                available: buf.len(),
271            });
272        }
273        let (datatype, _) = DatatypeMessage::decode(&buf[pos..pos + datatype_size], ctx)?;
274        pos += datatype_size;
275        if align > 1 {
276            pos = (pos + align - 1) & !(align - 1);
277        }
278
279        // Dataspace
280        let needed = pos + dataspace_size;
281        if buf.len() < needed {
282            return Err(FormatError::BufferTooShort {
283                needed,
284                available: buf.len(),
285            });
286        }
287        let (dataspace, _) = DataspaceMessage::decode(&buf[pos..pos + dataspace_size], ctx)?;
288        pos += dataspace_size;
289        if align > 1 {
290            pos = (pos + align - 1) & !(align - 1);
291        }
292
293        // Data: remaining bytes = datatype.element_size() * number_of_elements
294        let num_elements: u64 = if dataspace.dims.is_empty() {
295            1 // scalar
296        } else {
297            // dims are file-derived; saturate so a crafted attribute with
298            // absurd dimensions is rejected by the buffer check below
299            // instead of overflowing.
300            dataspace
301                .dims
302                .iter()
303                .fold(1u64, |acc, &d| acc.saturating_mul(d))
304        };
305        let data_size = num_elements
306            .saturating_mul(datatype.element_size() as u64)
307            .min(usize::MAX as u64) as usize;
308        let needed = pos.saturating_add(data_size);
309        if buf.len() < needed {
310            return Err(FormatError::BufferTooShort {
311                needed,
312                available: buf.len(),
313            });
314        }
315        let data = buf[pos..pos + data_size].to_vec();
316        pos += data_size;
317
318        Ok((
319            Self {
320                name,
321                datatype,
322                dataspace,
323                data,
324            },
325            pos,
326        ))
327    }
328}
329
330/// The part of an attribute message that identifies it: the envelope and the
331/// name, both of which sit ahead of the datatype.
332///
333/// Split out because that ordering is what makes an undecodable attribute
334/// nameable — see [`AttributeEntry::parse`].
335struct AttributeHeader {
336    name: String,
337    datatype_size: usize,
338    dataspace_size: usize,
339    /// Field alignment: 8 for version 1, 1 for versions 2 and 3.
340    align: usize,
341    /// Offset just past the (aligned) name, where the datatype begins.
342    pos: usize,
343}
344
345impl AttributeHeader {
346    fn decode(buf: &[u8]) -> FormatResult<Self> {
347        if buf.len() < 8 {
348            return Err(FormatError::BufferTooShort {
349                needed: 8,
350                available: buf.len(),
351            });
352        }
353
354        let version = buf[0];
355        if !(1..=ATTR_VERSION).contains(&version) {
356            return Err(FormatError::InvalidVersion(version));
357        }
358
359        // Byte 1 says whether the datatype and dataspace that follow are
360        // bodies or references (`H5O_ATTR_FLAG_TYPE_SHARED` /
361        // `H5O_ATTR_FLAG_SPACE_SHARED`). A reference decoded as a body reads
362        // its version byte as the body's, which invents a type rather than
363        // failing, so an attribute that carries one is named here instead.
364        // Resolving it needs the file the reference points into, which a
365        // message decoder does not have — an attribute read out of an object
366        // header has been resolved before it gets here, one read out of dense
367        // storage has not.
368        let flags = buf[1];
369        if flags & (ATTR_FLAG_TYPE_SHARED | ATTR_FLAG_SPACE_SHARED) != 0 {
370            let what = if flags & ATTR_FLAG_TYPE_SHARED != 0 {
371                "datatype"
372            } else {
373                "dataspace"
374            };
375            return Err(FormatError::UnsupportedFeature(format!(
376                "attribute whose {what} is a shared-message reference"
377            )));
378        }
379        let name_size = u16::from_le_bytes([buf[2], buf[3]]) as usize;
380        let datatype_size = u16::from_le_bytes([buf[4], buf[5]]) as usize;
381        let dataspace_size = u16::from_le_bytes([buf[6], buf[7]]) as usize;
382
383        let mut pos = if version >= 3 {
384            // v3 has charset byte at offset 8
385            9
386        } else {
387            // v1, v2: no charset byte
388            8
389        };
390
391        // v1 pads each field to 8-byte alignment
392        let align = if version == 1 { 8 } else { 1 };
393
394        // Name
395        let needed = pos + name_size;
396        if buf.len() < needed {
397            return Err(FormatError::BufferTooShort {
398                needed,
399                available: buf.len(),
400            });
401        }
402        // Strip trailing null
403        let name_end = if name_size > 0 && buf[pos + name_size - 1] == 0 {
404            pos + name_size - 1
405        } else {
406            pos + name_size
407        };
408        let name = String::from_utf8_lossy(&buf[pos..name_end]).to_string();
409        pos += name_size;
410        // v1 alignment
411        if align > 1 {
412            pos = (pos + align - 1) & !(align - 1);
413        }
414
415        Ok(Self {
416            name,
417            datatype_size,
418            dataspace_size,
419            align,
420            pos,
421        })
422    }
423}
424
425/// One attribute as an object header holds it: the message, plus the creation
426/// index the file records for it.
427///
428/// [`AttributeMessage::decode`] fails on a payload this crate cannot model —
429/// an object-reference datatype, say — but the name sits ahead of the datatype
430/// in the message, so such an attribute is still identifiable. Carrying the
431/// unreadable case in the same list is what lets a listing answer "this object
432/// has an attribute named X that I cannot read" instead of answering as though
433/// X were not there.
434///
435/// The creation index is a property of the attribute, exactly as its name is —
436/// `H5A_shared_t::crt_idx`, stored in the object header message envelope when
437/// the set is compact and in the index records when it is dense. Keeping it
438/// here is what stops a rewrite from re-deriving it from the position an
439/// attribute happens to occupy in a list: a dense set is read back in name-hash
440/// order, so a position-derived index re-stamps the whole set with the order
441/// the hash walk took.
442#[derive(Debug, Clone, PartialEq)]
443pub struct AttributeEntry {
444    body: AttributeBody,
445    /// The index this attribute was created with, when its object tracks
446    /// creation order. `None` when the object does not, which is what
447    /// `H5O_MAX_CRT_ORDER_IDX` says on disk.
448    creation_index: Option<u16>,
449}
450
451/// The message an [`AttributeEntry`] carries, decoded or not.
452#[derive(Debug, Clone, PartialEq)]
453enum AttributeBody {
454    /// Decoded, and usable through the typed accessors.
455    Readable(AttributeMessage),
456    /// Named, with the reason it could not be decoded and the message payload
457    /// verbatim — so a header rewrite puts back exactly what it read rather
458    /// than dropping what it could not model.
459    Unreadable {
460        name: String,
461        raw: Vec<u8>,
462        reason: String,
463    },
464}
465
466impl AttributeEntry {
467    /// Parse one attribute message. The entry carries no creation index —
468    /// only the envelope or index record it came out of knows one, so the
469    /// caller that has it attaches it with
470    /// [`with_creation_index`](Self::with_creation_index).
471    ///
472    /// Total over every message whose envelope and name parse: a payload this
473    /// crate cannot decode is named, never an absence. Only a message too
474    /// damaged to yield a name at all is an error, because there is then no
475    /// name to report.
476    pub fn parse(buf: &[u8], ctx: &FormatContext) -> FormatResult<Self> {
477        let body = match AttributeMessage::decode(buf, ctx) {
478            Ok((attr, _)) => AttributeBody::Readable(attr),
479            Err(payload_err) => {
480                let header = AttributeHeader::decode(buf)?;
481                AttributeBody::Unreadable {
482                    name: header.name,
483                    raw: buf.to_vec(),
484                    reason: payload_err.to_string(),
485                }
486            }
487        };
488        Ok(Self {
489            body,
490            creation_index: None,
491        })
492    }
493
494    /// This entry with `creation_index` attached.
495    pub fn with_creation_index(mut self, creation_index: Option<u16>) -> Self {
496        self.creation_index = creation_index;
497        self
498    }
499
500    /// Attach `creation_index` in place.
501    pub fn set_creation_index(&mut self, creation_index: Option<u16>) {
502        self.creation_index = creation_index;
503    }
504
505    /// The index this attribute was created with, or `None` when its object
506    /// does not track creation order.
507    pub fn creation_index(&self) -> Option<u16> {
508        self.creation_index
509    }
510
511    /// The attribute's name, whether or not its payload decoded.
512    pub fn name(&self) -> &str {
513        match &self.body {
514            AttributeBody::Readable(attr) => &attr.name,
515            AttributeBody::Unreadable { name, .. } => name,
516        }
517    }
518
519    /// The decoded message, or the reason there is none — exactly one of the
520    /// two, so a caller reporting the failure never needs a branch for an
521    /// attribute that is neither.
522    pub fn decoded(&self) -> Result<&AttributeMessage, &str> {
523        match &self.body {
524            AttributeBody::Readable(attr) => Ok(attr),
525            AttributeBody::Unreadable { reason, .. } => Err(reason),
526        }
527    }
528
529    /// The decoded message, or `None` when only the name is known.
530    pub fn readable(&self) -> Option<&AttributeMessage> {
531        self.decoded().ok()
532    }
533
534    /// Why this attribute cannot be read, or `None` when it can be.
535    pub fn unreadable_reason(&self) -> Option<&str> {
536        self.decoded().err()
537    }
538
539    /// The message payload to write back into an object header.
540    ///
541    /// An unreadable attribute returns the bytes it was read from: re-encoding
542    /// is impossible without a decoded form, and dropping it would delete an
543    /// attribute the caller never asked to change.
544    pub fn encode(&self, ctx: &FormatContext) -> Vec<u8> {
545        self.encode_at(ctx, LibverBound::Earliest)
546    }
547
548    /// The same, for a file whose low libver bound is `libver`: the datatype
549    /// message inside a readable attribute follows it. An unreadable one is
550    /// bytes, and bytes have no version to choose.
551    pub fn encode_at(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
552        self.encode_for(ctx, libver, crate::format::ObjectFormat::Modern)
553    }
554
555    /// The same, at the message version `format` calls for.
556    pub fn encode_for(
557        &self,
558        ctx: &FormatContext,
559        libver: LibverBound,
560        format: crate::format::ObjectFormat,
561    ) -> Vec<u8> {
562        match &self.body {
563            AttributeBody::Readable(attr) => attr.encode_for(ctx, libver, format),
564            AttributeBody::Unreadable { raw, .. } => raw.clone(),
565        }
566    }
567}
568
569impl From<AttributeMessage> for AttributeEntry {
570    fn from(attr: AttributeMessage) -> Self {
571        Self {
572            body: AttributeBody::Readable(attr),
573            creation_index: None,
574        }
575    }
576}
577
578#[cfg(test)]
579mod tests {
580    use super::*;
581
582    fn ctx() -> FormatContext {
583        FormatContext {
584            sizeof_addr: 8,
585            sizeof_size: 8,
586        }
587    }
588
589    #[test]
590    fn scalar_string_roundtrip() {
591        let msg = AttributeMessage::scalar_string("my_attr", "hello");
592        let encoded = msg.encode(&ctx());
593        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
594        assert_eq!(consumed, encoded.len());
595        assert_eq!(decoded.name, "my_attr");
596        assert_eq!(decoded.data, b"hello\0");
597        assert_eq!(decoded, msg);
598    }
599
600    #[test]
601    fn scalar_string_empty() {
602        let msg = AttributeMessage::scalar_string("empty", "");
603        let encoded = msg.encode(&ctx());
604        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
605        assert_eq!(consumed, encoded.len());
606        assert_eq!(decoded.name, "empty");
607        assert_eq!(decoded.data, b"\0");
608        assert_eq!(decoded, msg);
609    }
610
611    #[test]
612    fn version_is_three() {
613        let msg = AttributeMessage::scalar_string("test", "val");
614        let encoded = msg.encode(&ctx());
615        assert_eq!(encoded[0], 3);
616    }
617
618    #[test]
619    fn decode_buffer_too_short() {
620        let buf = [0u8; 4];
621        let err = AttributeMessage::decode(&buf, &ctx()).unwrap_err();
622        match err {
623            FormatError::BufferTooShort { .. } => {}
624            other => panic!("unexpected error: {:?}", other),
625        }
626    }
627
628    #[test]
629    fn decode_bad_version() {
630        let msg = AttributeMessage::scalar_string("x", "y");
631        let mut encoded = msg.encode(&ctx());
632        encoded[0] = 0; // invalid version
633        let err = AttributeMessage::decode(&encoded, &ctx()).unwrap_err();
634        match err {
635            FormatError::InvalidVersion(0) => {}
636            other => panic!("unexpected error: {:?}", other),
637        }
638    }
639
640    #[test]
641    fn array_numeric_1d_roundtrip() {
642        use crate::format::messages::datatype::DatatypeMessage;
643        // Three int32 values, 1-D array attribute (NDArrayDimOffset-style).
644        let vals: [i32; 3] = [10, -20, 30];
645        let mut data = Vec::new();
646        for v in vals {
647            data.extend_from_slice(&v.to_le_bytes());
648        }
649        let msg = AttributeMessage::array_numeric(
650            "dim_offset",
651            DatatypeMessage::i32_type(),
652            &[3],
653            data.clone(),
654        );
655        assert_eq!(msg.dataspace.dims, vec![3]);
656        let encoded = msg.encode(&ctx());
657        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
658        assert_eq!(consumed, encoded.len());
659        assert_eq!(decoded.name, "dim_offset");
660        assert_eq!(decoded.dataspace.dims, vec![3]);
661        assert_eq!(decoded.data, data);
662        // The attribute's dataspace comes back naming the maximum dimensions
663        // a simple extent is always written with, which `array_numeric` left
664        // to the encoder to fill in.
665        assert_eq!(decoded.dataspace.max_dims, Some(vec![3]));
666        assert_eq!(decoded.datatype, msg.datatype);
667        assert_eq!(decoded.dataspace.class, msg.dataspace.class);
668    }
669
670    #[test]
671    fn scalar_string_utf8_content() {
672        let msg = AttributeMessage::scalar_string("desc", "caf\u{00e9}");
673        let encoded = msg.encode(&ctx());
674        let (decoded, _) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
675        assert_eq!(decoded.name, "desc");
676        // "caf\u{e9}" is 5 bytes in UTF-8 + null = 6
677        assert_eq!(decoded.data.len(), 6);
678        assert_eq!(&decoded.data[..5], "caf\u{00e9}".as_bytes());
679        assert_eq!(decoded.data[5], 0);
680    }
681
682    /// The 48 bytes libhdf5 1.14.6 wrote for `f.attrs["ra"] = 42` on the root
683    /// group of a default (superblock-0) file. Version 1 pads the name, the
684    /// datatype and the dataspace each out to eight bytes while the size
685    /// fields keep the unpadded lengths, so the byte comparison is the only
686    /// thing that catches a padding rule applied in the wrong place.
687    #[test]
688    fn a_legacy_attribute_matches_the_bytes_libhdf5_wrote() {
689        let ctx = FormatContext::default_v3();
690        let attr = AttributeMessage::scalar_numeric(
691            "ra",
692            DatatypeMessage::i64_type(),
693            42i64.to_le_bytes().to_vec(),
694        );
695        let buf = attr.encode_for(
696            &ctx,
697            LibverBound::Earliest,
698            crate::format::ObjectFormat::Legacy,
699        );
700        assert_eq!(
701            buf,
702            vec![
703                0x01, 0x00, 0x03, 0x00, 0x0c, 0x00, 0x08, 0x00, 0x72, 0x61, 0x00, 0x00, 0x00, 0x00,
704                0x00, 0x00, 0x10, 0x08, 0x00, 0x00, 0x08, 0x00, 0x00, 0x00, 0x00, 0x00, 0x40, 0x00,
705                0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x2a, 0x00,
706                0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
707            ]
708        );
709        let (back, consumed) = AttributeMessage::decode(&buf, &ctx).unwrap();
710        assert_eq!(consumed, buf.len());
711        assert_eq!(back.name, "ra");
712        assert_eq!(back.data, 42i64.to_le_bytes().to_vec());
713    }
714
715    /// A name whose padded length differs from the datatype's, so a decoder
716    /// that pads one field and not the other lands mid-value.
717    #[test]
718    fn a_legacy_attribute_round_trips_at_every_field_padding() {
719        let ctx = FormatContext::default_v3();
720        for name in ["a", "ab", "abcdefg", "abcdefgh", "abcdefghi"] {
721            let attr = AttributeMessage::array_numeric(
722                name,
723                DatatypeMessage::i32_type(),
724                &[3],
725                vec![1u8, 0, 0, 0, 2, 0, 0, 0, 3, 0, 0, 0],
726            );
727            let buf = attr.encode_for(
728                &ctx,
729                LibverBound::Earliest,
730                crate::format::ObjectFormat::Legacy,
731            );
732            assert_eq!(buf[0], 1, "{name}");
733            let (back, consumed) = AttributeMessage::decode(&buf, &ctx).unwrap();
734            assert_eq!(consumed, buf.len(), "{name}");
735            assert_eq!(back.name, name);
736            assert_eq!(back.data, attr.data, "{name}");
737            assert_eq!(back.dataspace.dims, vec![3], "{name}");
738        }
739    }
740}