Skip to main content

rust_hdf5/format/messages/
attribute.rs

1//! Attribute message (type 0x0C) -- describes an attribute attached to an object.
2//!
3//! Binary layout (version 3, no shared datatypes):
4//!
5//! ```text
6//!   Byte 0:    version = 3
7//!   Byte 1:    flags (0 for non-shared)
8//!   Bytes 2-3: name_size (u16 LE, including null terminator)
9//!   Bytes 4-5: datatype_size (u16 LE)
10//!   Bytes 6-7: dataspace_size (u16 LE)
11//!   Byte 8:    name character set encoding (0=ASCII, 1=UTF-8)
12//!   <name: name_size bytes, null-terminated>
13//!   <encoded datatype message: datatype_size bytes>
14//!   <encoded dataspace message: dataspace_size bytes>
15//!   <raw attribute data>
16//! ```
17
18use crate::format::messages::dataspace::DataspaceMessage;
19use crate::format::messages::datatype::DatatypeMessage;
20use crate::format::{FormatContext, FormatError, FormatResult, LibverBound};
21
22const ATTR_VERSION: u8 = 3;
23
24/// `H5O_ATTR_FLAG_TYPE_SHARED` (H5Oattr.c:88): the datatype field holds a
25/// shared-message pointer rather than the datatype message.
26pub const ATTR_FLAG_TYPE_SHARED: u8 = 0x01;
27/// `H5O_ATTR_FLAG_SPACE_SHARED` (H5Oattr.c:89), the same for the dataspace.
28pub const ATTR_FLAG_SPACE_SHARED: u8 = 0x02;
29
30/// One attribute message body, and where its datatype and dataspace fields
31/// sit inside it.
32///
33/// The offsets are what a caller that put a shared-message pointer in either
34/// field needs in order to fill the pointer's heap ID in later, once the heap
35/// it points into has been laid out.
36#[derive(Debug, Clone, PartialEq, Eq)]
37pub struct EncodedAttribute {
38    /// The message payload.
39    pub body: Vec<u8>,
40    /// Offset of the datatype field in `body`.
41    pub datatype_at: usize,
42    /// Offset of the dataspace field in `body`.
43    pub dataspace_at: usize,
44}
45
46/// An HDF5 attribute message.
47#[derive(Debug, Clone, PartialEq)]
48pub struct AttributeMessage {
49    /// Attribute name.
50    pub name: String,
51    /// Datatype of the attribute value.
52    pub datatype: DatatypeMessage,
53    /// Dataspace (scalar or simple).
54    pub dataspace: DataspaceMessage,
55    /// Raw attribute value data.
56    pub data: Vec<u8>,
57}
58
59impl AttributeMessage {
60    /// Create a scalar string attribute with the given name and value.
61    ///
62    /// Uses a null-terminated UTF-8 fixed-length string datatype with
63    /// size = value.len() + 1 (for the null terminator), and a scalar
64    /// dataspace.
65    pub fn scalar_string(name: &str, value: &str) -> Self {
66        let str_size = (value.len() + 1) as u32; // +1 for null terminator
67        let datatype = DatatypeMessage::fixed_string_utf8(str_size);
68        let dataspace = DataspaceMessage::scalar();
69
70        // Data: string bytes + null terminator
71        let mut data = Vec::with_capacity(str_size as usize);
72        data.extend_from_slice(value.as_bytes());
73        data.push(0); // null terminator
74
75        Self {
76            name: name.to_string(),
77            datatype,
78            dataspace,
79            data,
80        }
81    }
82
83    /// Create a scalar numeric attribute with raw bytes as value.
84    pub fn scalar_numeric(name: &str, datatype: DatatypeMessage, data: Vec<u8>) -> Self {
85        Self {
86            name: name.to_string(),
87            datatype,
88            dataspace: DataspaceMessage::scalar(),
89            data,
90        }
91    }
92
93    /// Create a numeric array attribute with a simple dataspace.
94    ///
95    /// `dims` are the dimension sizes (e.g. `&[3]` for the 1-D array
96    /// attributes AreaDetector writes). `data` is the row-major raw bytes and
97    /// must hold `product(dims) * datatype.element_size()` bytes — the caller
98    /// owns that invariant. An empty `dims` yields a scalar dataspace; prefer
99    /// [`Self::scalar_numeric`] for that case.
100    pub fn array_numeric(
101        name: &str,
102        datatype: DatatypeMessage,
103        dims: &[u64],
104        data: Vec<u8>,
105    ) -> Self {
106        debug_assert_eq!(
107            data.len() as u64,
108            dims.iter().product::<u64>() * datatype.element_size() as u64,
109            "array_numeric data length must equal product(dims) * element_size"
110        );
111        Self {
112            name: name.to_string(),
113            datatype,
114            dataspace: DataspaceMessage::simple(dims),
115            data,
116        }
117    }
118
119    /// Encode the attribute message into a byte vector.
120    ///
121    /// The result is the raw payload for an object header message of type
122    /// 0x0C (MSG_ATTRIBUTE). It does NOT include the object header message
123    /// envelope (type, size, flags bytes); that is handled by the caller.
124    pub fn encode(&self, ctx: &FormatContext) -> Vec<u8> {
125        self.encode_at(ctx, LibverBound::Earliest)
126    }
127
128    /// Encode the attribute message for a file whose low libver bound is
129    /// `libver`, which the datatype message inside it follows.
130    pub fn encode_at(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
131        self.encode_for(ctx, libver, crate::format::ObjectFormat::Modern)
132    }
133
134    /// Encode a version-1 attribute message (`H5O__attr_encode`, H5Oattr.c).
135    ///
136    /// Version 1 has no flags byte and no name character set: byte 1 is
137    /// reserved, and the three size fields are followed by the name, the
138    /// datatype and the dataspace each padded out to a multiple of eight
139    /// bytes. The size fields record the *unpadded* lengths, so a decoder that
140    /// forgets the padding walks into the middle of the next field — which is
141    /// why the version is not something a writer may pick freely.
142    fn encode_v1(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
143        /// `H5O_ALIGN_OLD`, which version 1 of this message applies to each of
144        /// its three variable-length fields.
145        fn pad_to_8(buf: &mut Vec<u8>) {
146            let padded = (buf.len() + 7) & !7;
147            buf.resize(padded, 0);
148        }
149
150        let encoded_dt = self.datatype.encode_at(ctx, libver);
151        let encoded_ds = self
152            .dataspace
153            .encode_for(ctx, crate::format::ObjectFormat::Legacy);
154        let name_bytes = self.name.as_bytes();
155        let name_size = name_bytes.len() + 1;
156
157        let mut buf = Vec::with_capacity(8 + name_size + encoded_dt.len() + encoded_ds.len() + 24);
158        buf.push(1); // version
159        buf.push(0); // reserved
160        buf.extend_from_slice(&(name_size as u16).to_le_bytes());
161        buf.extend_from_slice(&(encoded_dt.len() as u16).to_le_bytes());
162        buf.extend_from_slice(&(encoded_ds.len() as u16).to_le_bytes());
163        buf.extend_from_slice(name_bytes);
164        buf.push(0);
165        pad_to_8(&mut buf);
166        buf.extend_from_slice(&encoded_dt);
167        pad_to_8(&mut buf);
168        buf.extend_from_slice(&encoded_ds);
169        pad_to_8(&mut buf);
170        buf.extend_from_slice(&self.data);
171        buf
172    }
173
174    /// Encode the attribute message at the version a file of this `format`
175    /// calls for, with the datatype inside it at `libver`.
176    pub fn encode_for(
177        &self,
178        ctx: &FormatContext,
179        libver: LibverBound,
180        format: crate::format::ObjectFormat,
181    ) -> Vec<u8> {
182        if format.attribute_version() == 1 {
183            return self.encode_v1(ctx, libver);
184        }
185        let encoded_dt = self.datatype.encode_at(ctx, libver);
186        let encoded_ds = self.dataspace.encode_for(ctx, format);
187        self.encode_with_fields(0x00, &encoded_dt, &encoded_ds).body
188    }
189
190    /// The version-3 body with its datatype and dataspace fields supplied.
191    ///
192    /// `H5O__attr_encode` writes each of the two through its message class's
193    /// encoder, which is the *shared* encoder when that piece is a shared
194    /// message — the field then holds a `H5O_shared_t` and the attribute's own
195    /// flags byte says so (`H5O_ATTR_FLAG_TYPE_SHARED` /
196    /// `H5O_ATTR_FLAG_SPACE_SHARED`, H5Oattr.c:358-359). Whichever it is, the
197    /// size fields record what is actually stored, so the caller supplies the
198    /// bytes and the matching flag bits and this lays the message out around
199    /// them.
200    pub fn encode_with_fields(
201        &self,
202        flags: u8,
203        datatype: &[u8],
204        dataspace: &[u8],
205    ) -> EncodedAttribute {
206        // Name with null terminator
207        let name_bytes = self.name.as_bytes();
208        let name_size = name_bytes.len() + 1; // +1 for null terminator
209
210        // Total: 9 (header) + name_size + datatype_size + dataspace_size + data_size
211        let total = 9 + name_size + datatype.len() + dataspace.len() + self.data.len();
212        let mut buf = Vec::with_capacity(total);
213
214        // Byte 0: version
215        buf.push(ATTR_VERSION);
216
217        // Byte 1: flags — which of the two fields below is a shared pointer.
218        buf.push(flags);
219
220        // Bytes 2-3: name size (u16 LE)
221        buf.extend_from_slice(&(name_size as u16).to_le_bytes());
222
223        // Bytes 4-5: datatype size (u16 LE)
224        buf.extend_from_slice(&(datatype.len() as u16).to_le_bytes());
225
226        // Bytes 6-7: dataspace size (u16 LE)
227        buf.extend_from_slice(&(dataspace.len() as u16).to_le_bytes());
228
229        // Byte 8: name character set encoding (1 = UTF-8)
230        buf.push(0x01);
231
232        // Name (null-terminated)
233        buf.extend_from_slice(name_bytes);
234        buf.push(0x00);
235
236        let datatype_at = buf.len();
237        buf.extend_from_slice(datatype);
238
239        let dataspace_at = buf.len();
240        buf.extend_from_slice(dataspace);
241
242        // Raw data
243        buf.extend_from_slice(&self.data);
244
245        debug_assert_eq!(buf.len(), total);
246        EncodedAttribute {
247            body: buf,
248            datatype_at,
249            dataspace_at,
250        }
251    }
252
253    /// Decode an attribute message from a byte buffer.
254    ///
255    /// Supports versions 1, 2, and 3:
256    /// - v1: 8-byte header, each field padded to 8-byte alignment
257    /// - v2: 8-byte header, no alignment padding
258    /// - v3: 9-byte header (adds charset byte), no alignment padding
259    pub fn decode(buf: &[u8], ctx: &FormatContext) -> FormatResult<(Self, usize)> {
260        let AttributeHeader {
261            name,
262            datatype_size,
263            dataspace_size,
264            align,
265            mut pos,
266        } = AttributeHeader::decode(buf)?;
267
268        // Datatype
269        let needed = pos + datatype_size;
270        if buf.len() < needed {
271            return Err(FormatError::BufferTooShort {
272                needed,
273                available: buf.len(),
274            });
275        }
276        let (datatype, _) = DatatypeMessage::decode(&buf[pos..pos + datatype_size], ctx)?;
277        pos += datatype_size;
278        if align > 1 {
279            pos = (pos + align - 1) & !(align - 1);
280        }
281
282        // Dataspace
283        let needed = pos + dataspace_size;
284        if buf.len() < needed {
285            return Err(FormatError::BufferTooShort {
286                needed,
287                available: buf.len(),
288            });
289        }
290        let (dataspace, _) = DataspaceMessage::decode(&buf[pos..pos + dataspace_size], ctx)?;
291        pos += dataspace_size;
292        if align > 1 {
293            pos = (pos + align - 1) & !(align - 1);
294        }
295
296        // Data: remaining bytes = datatype.element_size() * number_of_elements
297        let num_elements: u64 = if dataspace.dims.is_empty() {
298            1 // scalar
299        } else {
300            // dims are file-derived; saturate so a crafted attribute with
301            // absurd dimensions is rejected by the buffer check below
302            // instead of overflowing.
303            dataspace
304                .dims
305                .iter()
306                .fold(1u64, |acc, &d| acc.saturating_mul(d))
307        };
308        let data_size = num_elements
309            .saturating_mul(datatype.element_size() as u64)
310            .min(usize::MAX as u64) as usize;
311        let needed = pos.saturating_add(data_size);
312        if buf.len() < needed {
313            return Err(FormatError::BufferTooShort {
314                needed,
315                available: buf.len(),
316            });
317        }
318        let data = buf[pos..pos + data_size].to_vec();
319        pos += data_size;
320
321        Ok((
322            Self {
323                name,
324                datatype,
325                dataspace,
326                data,
327            },
328            pos,
329        ))
330    }
331}
332
333/// The part of an attribute message that identifies it: the envelope and the
334/// name, both of which sit ahead of the datatype.
335///
336/// Split out because that ordering is what makes an undecodable attribute
337/// nameable — see [`AttributeEntry::parse`].
338struct AttributeHeader {
339    name: String,
340    datatype_size: usize,
341    dataspace_size: usize,
342    /// Field alignment: 8 for version 1, 1 for versions 2 and 3.
343    align: usize,
344    /// Offset just past the (aligned) name, where the datatype begins.
345    pos: usize,
346}
347
348impl AttributeHeader {
349    fn decode(buf: &[u8]) -> FormatResult<Self> {
350        if buf.len() < 8 {
351            return Err(FormatError::BufferTooShort {
352                needed: 8,
353                available: buf.len(),
354            });
355        }
356
357        let version = buf[0];
358        if !(1..=ATTR_VERSION).contains(&version) {
359            return Err(FormatError::InvalidVersion(version));
360        }
361
362        // Byte 1 says whether the datatype and dataspace that follow are
363        // bodies or references (`H5O_ATTR_FLAG_TYPE_SHARED` /
364        // `H5O_ATTR_FLAG_SPACE_SHARED`). A reference decoded as a body reads
365        // its version byte as the body's, which invents a type rather than
366        // failing, so an attribute that carries one is named here instead.
367        // Resolving it needs the file the reference points into, which a
368        // message decoder does not have — an attribute read out of an object
369        // header has been resolved before it gets here, one read out of dense
370        // storage has not.
371        let flags = buf[1];
372        if flags & (ATTR_FLAG_TYPE_SHARED | ATTR_FLAG_SPACE_SHARED) != 0 {
373            let what = if flags & ATTR_FLAG_TYPE_SHARED != 0 {
374                "datatype"
375            } else {
376                "dataspace"
377            };
378            return Err(FormatError::UnsupportedFeature(format!(
379                "attribute whose {what} is a shared-message reference"
380            )));
381        }
382        let name_size = u16::from_le_bytes([buf[2], buf[3]]) as usize;
383        let datatype_size = u16::from_le_bytes([buf[4], buf[5]]) as usize;
384        let dataspace_size = u16::from_le_bytes([buf[6], buf[7]]) as usize;
385
386        let mut pos = if version >= 3 {
387            // v3 has charset byte at offset 8
388            9
389        } else {
390            // v1, v2: no charset byte
391            8
392        };
393
394        // v1 pads each field to 8-byte alignment
395        let align = if version == 1 { 8 } else { 1 };
396
397        // Name
398        let needed = pos + name_size;
399        if buf.len() < needed {
400            return Err(FormatError::BufferTooShort {
401                needed,
402                available: buf.len(),
403            });
404        }
405        // Strip trailing null
406        let name_end = if name_size > 0 && buf[pos + name_size - 1] == 0 {
407            pos + name_size - 1
408        } else {
409            pos + name_size
410        };
411        let name = String::from_utf8_lossy(&buf[pos..name_end]).to_string();
412        pos += name_size;
413        // v1 alignment
414        if align > 1 {
415            pos = (pos + align - 1) & !(align - 1);
416        }
417
418        Ok(Self {
419            name,
420            datatype_size,
421            dataspace_size,
422            align,
423            pos,
424        })
425    }
426}
427
428/// One attribute as an object header holds it: the message, plus the creation
429/// index the file records for it.
430///
431/// [`AttributeMessage::decode`] fails on a payload this crate cannot model —
432/// an object-reference datatype, say — but the name sits ahead of the datatype
433/// in the message, so such an attribute is still identifiable. Carrying the
434/// unreadable case in the same list is what lets a listing answer "this object
435/// has an attribute named X that I cannot read" instead of answering as though
436/// X were not there.
437///
438/// The creation index is a property of the attribute, exactly as its name is —
439/// `H5A_shared_t::crt_idx`, stored in the object header message envelope when
440/// the set is compact and in the index records when it is dense. Keeping it
441/// here is what stops a rewrite from re-deriving it from the position an
442/// attribute happens to occupy in a list: a dense set is read back in name-hash
443/// order, so a position-derived index re-stamps the whole set with the order
444/// the hash walk took.
445#[derive(Debug, Clone, PartialEq)]
446pub struct AttributeEntry {
447    body: AttributeBody,
448    /// The index this attribute was created with, when its object tracks
449    /// creation order. `None` when the object does not, which is what
450    /// `H5O_MAX_CRT_ORDER_IDX` says on disk.
451    creation_index: Option<u16>,
452}
453
454/// The message an [`AttributeEntry`] carries, decoded or not.
455#[derive(Debug, Clone, PartialEq)]
456enum AttributeBody {
457    /// Decoded, and usable through the typed accessors.
458    Readable(AttributeMessage),
459    /// Named, with the reason it could not be decoded and the message payload
460    /// verbatim — so a header rewrite puts back exactly what it read rather
461    /// than dropping what it could not model.
462    Unreadable {
463        name: String,
464        raw: Vec<u8>,
465        reason: String,
466    },
467}
468
469impl AttributeEntry {
470    /// Parse one attribute message. The entry carries no creation index —
471    /// only the envelope or index record it came out of knows one, so the
472    /// caller that has it attaches it with
473    /// [`with_creation_index`](Self::with_creation_index).
474    ///
475    /// Total over every message whose envelope and name parse: a payload this
476    /// crate cannot decode is named, never an absence. Only a message too
477    /// damaged to yield a name at all is an error, because there is then no
478    /// name to report.
479    pub fn parse(buf: &[u8], ctx: &FormatContext) -> FormatResult<Self> {
480        let body = match AttributeMessage::decode(buf, ctx) {
481            Ok((attr, _)) => AttributeBody::Readable(attr),
482            Err(payload_err) => {
483                let header = AttributeHeader::decode(buf)?;
484                AttributeBody::Unreadable {
485                    name: header.name,
486                    raw: buf.to_vec(),
487                    reason: payload_err.to_string(),
488                }
489            }
490        };
491        Ok(Self {
492            body,
493            creation_index: None,
494        })
495    }
496
497    /// This entry with `creation_index` attached.
498    pub fn with_creation_index(mut self, creation_index: Option<u16>) -> Self {
499        self.creation_index = creation_index;
500        self
501    }
502
503    /// Attach `creation_index` in place.
504    pub fn set_creation_index(&mut self, creation_index: Option<u16>) {
505        self.creation_index = creation_index;
506    }
507
508    /// The index this attribute was created with, or `None` when its object
509    /// does not track creation order.
510    pub fn creation_index(&self) -> Option<u16> {
511        self.creation_index
512    }
513
514    /// The attribute's name, whether or not its payload decoded.
515    pub fn name(&self) -> &str {
516        match &self.body {
517            AttributeBody::Readable(attr) => &attr.name,
518            AttributeBody::Unreadable { name, .. } => name,
519        }
520    }
521
522    /// The decoded message, or the reason there is none — exactly one of the
523    /// two, so a caller reporting the failure never needs a branch for an
524    /// attribute that is neither.
525    pub fn decoded(&self) -> Result<&AttributeMessage, &str> {
526        match &self.body {
527            AttributeBody::Readable(attr) => Ok(attr),
528            AttributeBody::Unreadable { reason, .. } => Err(reason),
529        }
530    }
531
532    /// The decoded message, or `None` when only the name is known.
533    pub fn readable(&self) -> Option<&AttributeMessage> {
534        self.decoded().ok()
535    }
536
537    /// Why this attribute cannot be read, or `None` when it can be.
538    pub fn unreadable_reason(&self) -> Option<&str> {
539        self.decoded().err()
540    }
541
542    /// The message payload to write back into an object header.
543    ///
544    /// An unreadable attribute returns the bytes it was read from: re-encoding
545    /// is impossible without a decoded form, and dropping it would delete an
546    /// attribute the caller never asked to change.
547    pub fn encode(&self, ctx: &FormatContext) -> Vec<u8> {
548        self.encode_at(ctx, LibverBound::Earliest)
549    }
550
551    /// The same, for a file whose low libver bound is `libver`: the datatype
552    /// message inside a readable attribute follows it. An unreadable one is
553    /// bytes, and bytes have no version to choose.
554    pub fn encode_at(&self, ctx: &FormatContext, libver: LibverBound) -> Vec<u8> {
555        self.encode_for(ctx, libver, crate::format::ObjectFormat::Modern)
556    }
557
558    /// The same, at the message version `format` calls for.
559    pub fn encode_for(
560        &self,
561        ctx: &FormatContext,
562        libver: LibverBound,
563        format: crate::format::ObjectFormat,
564    ) -> Vec<u8> {
565        match &self.body {
566            AttributeBody::Readable(attr) => attr.encode_for(ctx, libver, format),
567            AttributeBody::Unreadable { raw, .. } => raw.clone(),
568        }
569    }
570}
571
572impl From<AttributeMessage> for AttributeEntry {
573    fn from(attr: AttributeMessage) -> Self {
574        Self {
575            body: AttributeBody::Readable(attr),
576            creation_index: None,
577        }
578    }
579}
580
581#[cfg(test)]
582mod tests {
583    use super::*;
584
585    fn ctx() -> FormatContext {
586        FormatContext {
587            sizeof_addr: 8,
588            sizeof_size: 8,
589        }
590    }
591
592    #[test]
593    fn scalar_string_roundtrip() {
594        let msg = AttributeMessage::scalar_string("my_attr", "hello");
595        let encoded = msg.encode(&ctx());
596        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
597        assert_eq!(consumed, encoded.len());
598        assert_eq!(decoded.name, "my_attr");
599        assert_eq!(decoded.data, b"hello\0");
600        assert_eq!(decoded, msg);
601    }
602
603    #[test]
604    fn scalar_string_empty() {
605        let msg = AttributeMessage::scalar_string("empty", "");
606        let encoded = msg.encode(&ctx());
607        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
608        assert_eq!(consumed, encoded.len());
609        assert_eq!(decoded.name, "empty");
610        assert_eq!(decoded.data, b"\0");
611        assert_eq!(decoded, msg);
612    }
613
614    #[test]
615    fn version_is_three() {
616        let msg = AttributeMessage::scalar_string("test", "val");
617        let encoded = msg.encode(&ctx());
618        assert_eq!(encoded[0], 3);
619    }
620
621    #[test]
622    fn decode_buffer_too_short() {
623        let buf = [0u8; 4];
624        let err = AttributeMessage::decode(&buf, &ctx()).unwrap_err();
625        match err {
626            FormatError::BufferTooShort { .. } => {}
627            other => panic!("unexpected error: {:?}", other),
628        }
629    }
630
631    #[test]
632    fn decode_bad_version() {
633        let msg = AttributeMessage::scalar_string("x", "y");
634        let mut encoded = msg.encode(&ctx());
635        encoded[0] = 0; // invalid version
636        let err = AttributeMessage::decode(&encoded, &ctx()).unwrap_err();
637        match err {
638            FormatError::InvalidVersion(0) => {}
639            other => panic!("unexpected error: {:?}", other),
640        }
641    }
642
643    #[test]
644    fn array_numeric_1d_roundtrip() {
645        use crate::format::messages::datatype::DatatypeMessage;
646        // Three int32 values, 1-D array attribute (NDArrayDimOffset-style).
647        let vals: [i32; 3] = [10, -20, 30];
648        let mut data = Vec::new();
649        for v in vals {
650            data.extend_from_slice(&v.to_le_bytes());
651        }
652        let msg = AttributeMessage::array_numeric(
653            "dim_offset",
654            DatatypeMessage::i32_type(),
655            &[3],
656            data.clone(),
657        );
658        assert_eq!(msg.dataspace.dims, vec![3]);
659        let encoded = msg.encode(&ctx());
660        let (decoded, consumed) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
661        assert_eq!(consumed, encoded.len());
662        assert_eq!(decoded.name, "dim_offset");
663        assert_eq!(decoded.dataspace.dims, vec![3]);
664        assert_eq!(decoded.data, data);
665        // The attribute's dataspace comes back naming the maximum dimensions
666        // a simple extent is always written with, which `array_numeric` left
667        // to the encoder to fill in.
668        assert_eq!(decoded.dataspace.max_dims, Some(vec![3]));
669        assert_eq!(decoded.datatype, msg.datatype);
670        assert_eq!(decoded.dataspace.class, msg.dataspace.class);
671    }
672
673    #[test]
674    fn scalar_string_utf8_content() {
675        let msg = AttributeMessage::scalar_string("desc", "caf\u{00e9}");
676        let encoded = msg.encode(&ctx());
677        let (decoded, _) = AttributeMessage::decode(&encoded, &ctx()).unwrap();
678        assert_eq!(decoded.name, "desc");
679        // "caf\u{e9}" is 5 bytes in UTF-8 + null = 6
680        assert_eq!(decoded.data.len(), 6);
681        assert_eq!(&decoded.data[..5], "caf\u{00e9}".as_bytes());
682        assert_eq!(decoded.data[5], 0);
683    }
684
685    /// The 48 bytes libhdf5 1.14.6 wrote for `f.attrs["ra"] = 42` on the root
686    /// group of a default (superblock-0) file. Version 1 pads the name, the
687    /// datatype and the dataspace each out to eight bytes while the size
688    /// fields keep the unpadded lengths, so the byte comparison is the only
689    /// thing that catches a padding rule applied in the wrong place.
690    #[test]
691    fn a_legacy_attribute_matches_the_bytes_libhdf5_wrote() {
692        let ctx = FormatContext::default_v3();
693        let attr = AttributeMessage::scalar_numeric(
694            "ra",
695            DatatypeMessage::i64_type(),
696            42i64.to_le_bytes().to_vec(),
697        );
698        let buf = attr.encode_for(
699            &ctx,
700            LibverBound::Earliest,
701            crate::format::ObjectFormat::Legacy,
702        );
703        assert_eq!(
704            buf,
705            vec![
706                0x01, 0x00, 0x03, 0x00, 0x0c, 0x00, 0x08, 0x00, 0x72, 0x61, 0x00, 0x00, 0x00, 0x00,
707                0x00, 0x00, 0x10, 0x08, 0x00, 0x00, 0x08, 0x00, 0x00, 0x00, 0x00, 0x00, 0x40, 0x00,
708                0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x2a, 0x00,
709                0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
710            ]
711        );
712        let (back, consumed) = AttributeMessage::decode(&buf, &ctx).unwrap();
713        assert_eq!(consumed, buf.len());
714        assert_eq!(back.name, "ra");
715        assert_eq!(back.data, 42i64.to_le_bytes().to_vec());
716    }
717
718    /// A name whose padded length differs from the datatype's, so a decoder
719    /// that pads one field and not the other lands mid-value.
720    #[test]
721    fn a_legacy_attribute_round_trips_at_every_field_padding() {
722        let ctx = FormatContext::default_v3();
723        for name in ["a", "ab", "abcdefg", "abcdefgh", "abcdefghi"] {
724            let attr = AttributeMessage::array_numeric(
725                name,
726                DatatypeMessage::i32_type(),
727                &[3],
728                vec![1u8, 0, 0, 0, 2, 0, 0, 0, 3, 0, 0, 0],
729            );
730            let buf = attr.encode_for(
731                &ctx,
732                LibverBound::Earliest,
733                crate::format::ObjectFormat::Legacy,
734            );
735            assert_eq!(buf[0], 1, "{name}");
736            let (back, consumed) = AttributeMessage::decode(&buf, &ctx).unwrap();
737            assert_eq!(consumed, buf.len(), "{name}");
738            assert_eq!(back.name, name);
739            assert_eq!(back.data, attr.data, "{name}");
740            assert_eq!(back.dataspace.dims, vec![3], "{name}");
741        }
742    }
743}