Skip to main content

rete_core/
header.rs

1//! The fixed-size 1024-byte file header (SPEC.md §4.1).
2//!
3//! A client's very first request is `bytes=0-1023`; this struct is what those
4//! bytes decode to, and it points at every other section of the file via a
5//! **typed section directory**, so new top-level sections (e.g. a future text
6//! index) are added as a new directory entry without reshaping the header.
7//!
8//! Layout: a fixed 64-byte **core** (magic, version, flags, content hash, counts,
9//! codecs, `section_count`, `schema_meta_len`) followed by up to
10//! [`MAX_SECTIONS`] **directory entries** of 24 bytes each `(kind, flags, offset,
11//! length)`, zero-padded to 1024. The known section kinds populate the named
12//! convenience fields below; unknown kinds (written by a newer build) are
13//! preserved verbatim in [`Header::extra_sections`].
14
15use std::convert::TryInto;
16
17/// Magic bytes at offset 0: ASCII `RETE`.
18pub const MAGIC: [u8; 4] = *b"RETE";
19
20/// Current format generation written by this crate.
21///
22/// `0x05` is stable format generation 1, frozen on 2026-07-14 and first
23/// released in Rete 0.3.0. It retains the six-wide index-permutation addressing
24/// and the 1 KiB section-directory layout finalized in the last experimental
25/// generation; *which* of those six a given file stores is [`Header::perms`]
26/// (byte 50), not the generation.
27///
28/// The generation number is not a release version — there is no Rete 1.0.0, and
29/// it is the Rust/CLI/WASM APIs, not the format, that are waiting on one. Nor
30/// does it pin reader capability: writer semantics changed inside `0x05` nine
31/// days after the freeze (#68 split oversized index groups across tiles), so a
32/// `0x05` file can need a reader newer than the one that froze `0x05`.
33pub const CURRENT_FORMAT_VERSION: u8 = 0x05;
34
35/// Oldest stable format generation accepted by this reader.
36///
37/// Files written before the generation-1 freeze used experimental generations
38/// `0x01` through `0x04` and must be rebuilt from their RDF source. `0x05` has
39/// not moved since it froze on 2026-07-14, but that is a track record, not a
40/// promise: no backwards compatibility is guaranteed before 1.0.0, so this
41/// constant may yet be raised and a `0x05` file may yet need a rebuild. See
42/// `docs/compatibility.md`.
43pub const MIN_STABLE_READ_VERSION: u8 = 0x05;
44
45/// Fixed header size in bytes.
46pub const HEADER_LEN: usize = 1024;
47
48/// Byte offset of the first section-directory entry (i.e. the core size).
49const SECTION_DIR_OFFSET: usize = 64;
50/// Size of one section-directory entry.
51const SECTION_ENTRY_LEN: usize = 24;
52/// How many directory entries fit in the 1 KB frame.
53pub const MAX_SECTIONS: usize = (HEADER_LEN - SECTION_DIR_OFFSET) / SECTION_ENTRY_LEN;
54
55/// Flag bit: the file contains named graphs (quads) rather than triples only.
56pub const FLAG_HAS_QUADS: u8 = 0b0000_0001;
57
58/// Flag bit: each tiled permutation section carries a **tile-synopsis trailer**
59/// — per-tile min/max of the two non-leading columns, appended after the tile
60/// payloads. It lets a range reader prune a routed tile by a bound secondary
61/// component *before* fetching it. See `file.rs::encode_tiled_section`.
62pub const FLAG_TILE_SYNOPSIS: u8 = 0b0000_0010;
63
64/// Flag bit: the file contains **RDF-star quoted triples** (`<< s p o >>`) as
65/// dictionary terms. Purely informational for compatibility — a plain-RDF
66/// consumer can detect from the header alone that some terms are quoted triples
67/// (which it may not understand) without scanning the dictionary. The file is
68/// otherwise a normal `.rete`: quoted triples are stored like any other term, so
69/// this needs no format-version bump and old readers stay forward-compatible.
70pub const FLAG_HAS_QUOTED_TRIPLES: u8 = 0b0000_0100;
71
72/// A top-level file section, addressed by [`SectionKind`] in the header directory.
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub enum SectionKind {
75    /// Dataset-card metadata.
76    Metadata,
77    /// Dictionary container (front-coded term sections).
78    Dictionary,
79    /// Permutation index container (SPO/POS/OSP).
80    Index,
81    /// Community + schema pyramid metadata.
82    PyramidMeta,
83    /// Named-graphs section.
84    NamedGraphs,
85    /// Full-text (word) index over literals — `token → subjects`.
86    TextIndex,
87    /// Build-conditions record (timestamp, builder version, build parameters,
88    /// measured starter-query costs), stored as an opaque JSON blob owned by the
89    /// application layer (the CLI). **Deliberately excluded from the content
90    /// hash**: two builds of identical data must hash identically, and this
91    /// section is exactly the per-build facts (when, by which binary, how fast)
92    /// that differ between them. `verify` therefore does not cover it.
93    BuildInfo,
94    /// A section kind this build doesn't know — preserved verbatim on round-trip
95    /// so a newer writer's sections survive an older reader.
96    Unknown(u16),
97}
98
99impl SectionKind {
100    fn to_u16(self) -> u16 {
101        match self {
102            SectionKind::Metadata => 1,
103            SectionKind::Dictionary => 2,
104            SectionKind::Index => 3,
105            SectionKind::PyramidMeta => 4,
106            SectionKind::NamedGraphs => 5,
107            SectionKind::TextIndex => 6,
108            SectionKind::BuildInfo => 7,
109            SectionKind::Unknown(k) => k,
110        }
111    }
112
113    fn from_u16(k: u16) -> Self {
114        match k {
115            1 => SectionKind::Metadata,
116            2 => SectionKind::Dictionary,
117            3 => SectionKind::Index,
118            4 => SectionKind::PyramidMeta,
119            5 => SectionKind::NamedGraphs,
120            6 => SectionKind::TextIndex,
121            7 => SectionKind::BuildInfo,
122            other => SectionKind::Unknown(other),
123        }
124    }
125}
126
127/// One parsed/encoded section-directory entry: a typed `(offset, length)` into
128/// the file, plus 16 bits of per-section flags (reserved).
129#[derive(Debug, Clone, Copy, PartialEq, Eq)]
130pub struct Section {
131    pub kind: SectionKind,
132    pub flags: u16,
133    pub offset: u64,
134    pub length: u64,
135}
136
137#[derive(Debug, thiserror::Error)]
138#[non_exhaustive]
139pub enum HeaderError {
140    #[error("buffer too small: need {HEADER_LEN} bytes, got {0}")]
141    TooSmall(usize),
142    #[error("bad magic: expected RETE")]
143    BadMagic,
144    #[error(
145        "unsupported .rete format {found:#04x}; this Rete build reads {min:#04x}..={max:#04x}. Pre-1.0 files must be rebuilt from RDF source with `rete build`"
146    )]
147    UnsupportedVersion { found: u8, min: u8, max: u8 },
148    #[error("section count {0} overruns the header frame")]
149    BadSectionCount(usize),
150    #[error("unusable index permutation mask {mask:#04x}: {why}")]
151    BadPermMask { mask: u8, why: &'static str },
152}
153
154/// Decoded file header. All multi-byte fields are little-endian on disk. The
155/// `*_offset` / `*_len` fields are a convenience view over the section directory
156/// (populated from it on parse, emitted back to it on serialize).
157#[derive(Debug, Clone, PartialEq, Eq)]
158pub struct Header {
159    /// Format version of the parsed file (currently
160    /// [`MIN_STABLE_READ_VERSION`] through [`CURRENT_FORMAT_VERSION`]).
161    pub version: u8,
162    pub flags: u8,
163    pub metadata_offset: u64,
164    pub metadata_len: u64,
165    pub dictionary_offset: u64,
166    pub dictionary_len: u64,
167    pub root_dir_offset: u64,
168    pub root_dir_len: u64,
169    pub pyramid_meta_offset: u64,
170    pub pyramid_meta_len: u64,
171    pub dict_codec: u8,
172    pub block_codec: u8,
173    pub pyramid_levels: u16,
174    pub quad_count: u64,
175    pub term_count: u64,
176    /// First 16 bytes of the blake3 content hash — an immutable validator.
177    pub content_hash: [u8; 16],
178    /// Named-graphs section (0 if the file has only the default graph).
179    pub named_graphs_offset: u64,
180    pub named_graphs_len: u64,
181    /// Byte length of the trailing **schema-pyramid block** within the pyramid-meta
182    /// section (0 if none). A reader fetches *only* that block — at
183    /// `pyramid_meta_offset + pyramid_meta_len - schema_meta_len` — for an
184    /// index/dictionary/summary-free Tier-0 coherence check.
185    pub schema_meta_len: u32,
186    /// Full-text index section (0 if the file has none; built with
187    /// `rete build --text-index`). See [`SectionKind::TextIndex`].
188    pub text_index_offset: u64,
189    pub text_index_len: u64,
190    /// Build-conditions section (0 if absent). See [`SectionKind::BuildInfo`]:
191    /// an opaque per-build record that is **not** covered by the content hash.
192    pub build_info_offset: u64,
193    pub build_info_len: u64,
194    /// Which index permutations the file's index containers carry
195    /// ([`crate::index::PermSet`]).
196    ///
197    /// On disk this is one byte at `[50]`, inside the reserved core span, and
198    /// **`0` means all six** — so every file written before the mask existed
199    /// decodes as [`PermSet::ALL`] and a full six-permutation build stays
200    /// byte-identical to what it always was. A lean file writes its mask
201    /// (`0b000_0111` for SPO+POS+OSP), and its index container then holds three
202    /// sections rather than six, which is what an older reader trips over.
203    ///
204    /// [`PermSet::ALL`]: crate::index::PermSet::ALL
205    pub perms: crate::index::PermSet,
206    /// Directory entries whose [`SectionKind`] this build doesn't recognize,
207    /// preserved verbatim. Empty for a file this crate wrote.
208    pub extra_sections: Vec<Section>,
209}
210
211impl Header {
212    /// Serialize into a fixed 1024-byte array.
213    pub fn to_bytes(&self) -> [u8; HEADER_LEN] {
214        let mut b = [0u8; HEADER_LEN];
215        // --- core (64 bytes) ---
216        b[0..4].copy_from_slice(&MAGIC);
217        b[4] = self.version;
218        b[5] = self.flags;
219        b[6..8].copy_from_slice(&(HEADER_LEN as u16).to_le_bytes());
220        b[8..24].copy_from_slice(&self.content_hash);
221        b[24..32].copy_from_slice(&self.quad_count.to_le_bytes());
222        b[32..40].copy_from_slice(&self.term_count.to_le_bytes());
223        b[40..42].copy_from_slice(&self.pyramid_levels.to_le_bytes());
224        b[42] = self.dict_codec;
225        b[43] = self.block_codec;
226        // [44..46) section_count written below.
227        b[46..50].copy_from_slice(&self.schema_meta_len.to_le_bytes());
228        // [50] permutation mask; 0 = all six, so a full build is byte-identical
229        // to every file written before the field existed.
230        b[50] = if self.perms == crate::index::PermSet::ALL {
231            0
232        } else {
233            self.perms.bits()
234        };
235        // [51..64) reserved.
236
237        // --- section directory ---
238        // The five always-present sections (verbatim, so the named offsets
239        // round-trip exactly), then the optional text index (only when present, so
240        // a file without one stays byte-identical), then preserved unknown kinds.
241        let entry = |kind, offset, length| Section {
242            kind,
243            flags: 0,
244            offset,
245            length,
246        };
247        let mut entries: Vec<Section> = vec![
248            entry(
249                SectionKind::Metadata,
250                self.metadata_offset,
251                self.metadata_len,
252            ),
253            entry(
254                SectionKind::Dictionary,
255                self.dictionary_offset,
256                self.dictionary_len,
257            ),
258            entry(SectionKind::Index, self.root_dir_offset, self.root_dir_len),
259            entry(
260                SectionKind::PyramidMeta,
261                self.pyramid_meta_offset,
262                self.pyramid_meta_len,
263            ),
264            entry(
265                SectionKind::NamedGraphs,
266                self.named_graphs_offset,
267                self.named_graphs_len,
268            ),
269        ];
270        if self.text_index_len > 0 {
271            entries.push(entry(
272                SectionKind::TextIndex,
273                self.text_index_offset,
274                self.text_index_len,
275            ));
276        }
277        if self.build_info_len > 0 {
278            entries.push(entry(
279                SectionKind::BuildInfo,
280                self.build_info_offset,
281                self.build_info_len,
282            ));
283        }
284        entries.extend(self.extra_sections.iter().copied());
285        debug_assert!(
286            entries.len() <= MAX_SECTIONS,
287            "too many sections for a 1 KB header"
288        );
289        let n = entries.len().min(MAX_SECTIONS);
290        b[44..46].copy_from_slice(&(n as u16).to_le_bytes());
291        for (i, s) in entries.iter().take(n).enumerate() {
292            let p = SECTION_DIR_OFFSET + i * SECTION_ENTRY_LEN;
293            b[p..p + 2].copy_from_slice(&s.kind.to_u16().to_le_bytes());
294            b[p + 2..p + 4].copy_from_slice(&s.flags.to_le_bytes());
295            // [p+4..p+8) reserved.
296            b[p + 8..p + 16].copy_from_slice(&s.offset.to_le_bytes());
297            b[p + 16..p + 24].copy_from_slice(&s.length.to_le_bytes());
298        }
299        b
300    }
301
302    /// Parse a header from the first 1024 bytes of a file.
303    pub fn from_bytes(b: &[u8]) -> Result<Self, HeaderError> {
304        if b.len() < HEADER_LEN {
305            return Err(HeaderError::TooSmall(b.len()));
306        }
307        if b[0..4] != MAGIC {
308            return Err(HeaderError::BadMagic);
309        }
310        if !(MIN_STABLE_READ_VERSION..=CURRENT_FORMAT_VERSION).contains(&b[4]) {
311            return Err(HeaderError::UnsupportedVersion {
312                found: b[4],
313                min: MIN_STABLE_READ_VERSION,
314                max: CURRENT_FORMAT_VERSION,
315            });
316        }
317        let u16_at = |o: usize| u16::from_le_bytes(b[o..o + 2].try_into().unwrap());
318        let u32_at = |o: usize| u32::from_le_bytes(b[o..o + 4].try_into().unwrap());
319        let u64_at = |o: usize| u64::from_le_bytes(b[o..o + 8].try_into().unwrap());
320
321        let section_count = u16_at(44) as usize;
322        if SECTION_DIR_OFFSET + section_count * SECTION_ENTRY_LEN > HEADER_LEN {
323            return Err(HeaderError::BadSectionCount(section_count));
324        }
325
326        let perms = if b[50] == 0 {
327            crate::index::PermSet::ALL
328        } else {
329            crate::index::PermSet::from_bits(b[50])
330                .map_err(|why| HeaderError::BadPermMask { mask: b[50], why })?
331        };
332
333        let mut h = Header {
334            version: b[4],
335            flags: b[5],
336            metadata_offset: 0,
337            metadata_len: 0,
338            dictionary_offset: 0,
339            dictionary_len: 0,
340            root_dir_offset: 0,
341            root_dir_len: 0,
342            pyramid_meta_offset: 0,
343            pyramid_meta_len: 0,
344            dict_codec: b[42],
345            block_codec: b[43],
346            pyramid_levels: u16_at(40),
347            quad_count: u64_at(24),
348            term_count: u64_at(32),
349            content_hash: b[8..24].try_into().unwrap(),
350            named_graphs_offset: 0,
351            named_graphs_len: 0,
352            schema_meta_len: u32_at(46),
353            perms,
354            text_index_offset: 0,
355            text_index_len: 0,
356            build_info_offset: 0,
357            build_info_len: 0,
358            extra_sections: Vec::new(),
359        };
360        for i in 0..section_count {
361            let p = SECTION_DIR_OFFSET + i * SECTION_ENTRY_LEN;
362            let kind = SectionKind::from_u16(u16_at(p));
363            let offset = u64_at(p + 8);
364            let length = u64_at(p + 16);
365            match kind {
366                SectionKind::Metadata => {
367                    h.metadata_offset = offset;
368                    h.metadata_len = length;
369                }
370                SectionKind::Dictionary => {
371                    h.dictionary_offset = offset;
372                    h.dictionary_len = length;
373                }
374                SectionKind::Index => {
375                    h.root_dir_offset = offset;
376                    h.root_dir_len = length;
377                }
378                SectionKind::PyramidMeta => {
379                    h.pyramid_meta_offset = offset;
380                    h.pyramid_meta_len = length;
381                }
382                SectionKind::NamedGraphs => {
383                    h.named_graphs_offset = offset;
384                    h.named_graphs_len = length;
385                }
386                SectionKind::TextIndex => {
387                    h.text_index_offset = offset;
388                    h.text_index_len = length;
389                }
390                SectionKind::BuildInfo => {
391                    h.build_info_offset = offset;
392                    h.build_info_len = length;
393                }
394                SectionKind::Unknown(_) => h.extra_sections.push(Section {
395                    kind,
396                    flags: u16_at(p + 2),
397                    offset,
398                    length,
399                }),
400            }
401        }
402        Ok(h)
403    }
404
405    pub fn has_quads(&self) -> bool {
406        self.flags & FLAG_HAS_QUADS != 0
407    }
408
409    /// The total file length implied by this header alone: the writer lays the
410    /// sections out back-to-back after the 1 KiB header and ends the file with
411    /// the 4-byte `RETE` footer, so the furthest `offset + length` in the
412    /// section directory plus 4 **is** the file length.
413    ///
414    /// This is the one length signal a *remote* reader can always trust: an
415    /// HTTP host may advertise the size of a compressed representation in
416    /// `Content-Length` (GitHub Pages gzips — a 71 MB file HEADs as 58 MB)
417    /// while serving ranges over the identity bytes, and may hide
418    /// `Content-Range` from cross-origin JS by omitting
419    /// `Access-Control-Expose-Headers`. The header travels *in band* — in the
420    /// first 1 KiB range read — so it cannot be skewed by transport encoding.
421    ///
422    /// `None` if any directory entry overflows `u64` (a crafted or corrupt
423    /// header must yield "unknown", never a panic or a wrapped length — the
424    /// weekly fuzz caught `verify()` on exactly this class of input).
425    pub fn expected_file_len(&self) -> Option<u64> {
426        let mut end = HEADER_LEN as u64;
427        let named = [
428            (self.metadata_offset, self.metadata_len),
429            (self.dictionary_offset, self.dictionary_len),
430            (self.root_dir_offset, self.root_dir_len),
431            (self.pyramid_meta_offset, self.pyramid_meta_len),
432            (self.named_graphs_offset, self.named_graphs_len),
433            (self.text_index_offset, self.text_index_len),
434            (self.build_info_offset, self.build_info_len),
435        ];
436        let extras = self.extra_sections.iter().map(|s| (s.offset, s.length));
437        for (offset, length) in named.into_iter().chain(extras) {
438            if length == 0 {
439                continue; // absent section — its (0, 0) entry says nothing
440            }
441            end = end.max(offset.checked_add(length)?);
442        }
443        end.checked_add(MAGIC.len() as u64)
444    }
445
446    /// Does the file contain RDF-star quoted triples ([`FLAG_HAS_QUOTED_TRIPLES`])?
447    pub fn has_quoted_triples(&self) -> bool {
448        self.flags & FLAG_HAS_QUOTED_TRIPLES != 0
449    }
450
451    /// Do the tiled index sections carry a [`FLAG_TILE_SYNOPSIS`] trailer?
452    pub fn has_tile_synopsis(&self) -> bool {
453        self.flags & FLAG_TILE_SYNOPSIS != 0
454    }
455
456    /// The directory entry for a section kind, or `None` if absent. Known kinds
457    /// read from the named convenience fields; unknown kinds from
458    /// [`extra_sections`](Self::extra_sections).
459    pub fn section(&self, kind: SectionKind) -> Option<Section> {
460        let (offset, length) = match kind {
461            SectionKind::Metadata => (self.metadata_offset, self.metadata_len),
462            SectionKind::Dictionary => (self.dictionary_offset, self.dictionary_len),
463            SectionKind::Index => (self.root_dir_offset, self.root_dir_len),
464            SectionKind::PyramidMeta => (self.pyramid_meta_offset, self.pyramid_meta_len),
465            SectionKind::NamedGraphs => (self.named_graphs_offset, self.named_graphs_len),
466            SectionKind::TextIndex => (self.text_index_offset, self.text_index_len),
467            SectionKind::BuildInfo => (self.build_info_offset, self.build_info_len),
468            SectionKind::Unknown(_) => {
469                return self.extra_sections.iter().find(|s| s.kind == kind).copied()
470            }
471        };
472        Some(Section {
473            kind,
474            flags: 0,
475            offset,
476            length,
477        })
478    }
479
480    /// Attach (or overwrite) a section's `(offset, length)` — the extension point
481    /// for new top-level sections. Known kinds set the named fields; an unknown
482    /// kind is appended to [`extra_sections`](Self::extra_sections).
483    pub fn with_section(mut self, kind: SectionKind, offset: u64, length: u64) -> Self {
484        match kind {
485            SectionKind::Metadata => {
486                self.metadata_offset = offset;
487                self.metadata_len = length;
488            }
489            SectionKind::Dictionary => {
490                self.dictionary_offset = offset;
491                self.dictionary_len = length;
492            }
493            SectionKind::Index => {
494                self.root_dir_offset = offset;
495                self.root_dir_len = length;
496            }
497            SectionKind::PyramidMeta => {
498                self.pyramid_meta_offset = offset;
499                self.pyramid_meta_len = length;
500            }
501            SectionKind::NamedGraphs => {
502                self.named_graphs_offset = offset;
503                self.named_graphs_len = length;
504            }
505            SectionKind::TextIndex => {
506                self.text_index_offset = offset;
507                self.text_index_len = length;
508            }
509            SectionKind::BuildInfo => {
510                self.build_info_offset = offset;
511                self.build_info_len = length;
512            }
513            SectionKind::Unknown(_) => self.extra_sections.push(Section {
514                kind,
515                flags: 0,
516                offset,
517                length,
518            }),
519        }
520        self
521    }
522}
523
524#[cfg(test)]
525mod tests {
526    use super::*;
527
528    fn sample() -> Header {
529        Header {
530            version: CURRENT_FORMAT_VERSION,
531            flags: FLAG_HAS_QUADS,
532            metadata_offset: 1024,
533            metadata_len: 42,
534            dictionary_offset: 1066,
535            dictionary_len: 2048,
536            root_dir_offset: 3114,
537            root_dir_len: 256,
538            pyramid_meta_offset: 3370,
539            pyramid_meta_len: 64,
540            dict_codec: 1,
541            block_codec: 2,
542            pyramid_levels: 3,
543            perms: crate::index::PermSet::ALL,
544            quad_count: 5,
545            term_count: 9,
546            content_hash: [7u8; 16],
547            named_graphs_offset: 3434,
548            named_graphs_len: 48,
549            schema_meta_len: 99,
550            text_index_offset: 0,
551            text_index_len: 0,
552            build_info_offset: 0,
553            build_info_len: 0,
554            extra_sections: Vec::new(),
555        }
556    }
557
558    #[test]
559    fn round_trip() {
560        let h = sample();
561        let bytes = h.to_bytes();
562        assert_eq!(bytes.len(), HEADER_LEN);
563        assert_eq!(&bytes[0..4], b"RETE");
564        let back = Header::from_bytes(&bytes).unwrap();
565        assert_eq!(h, back);
566        assert!(back.has_quads());
567    }
568
569    #[test]
570    fn byte_layout_matches_spec() {
571        // Pins the core fields and the first directory entry to exact offsets.
572        let h = Header {
573            content_hash: [0xCC; 16],
574            quad_count: 0x99,
575            term_count: 0xAA,
576            pyramid_levels: 0xABCD,
577            dict_codec: 0xA1,
578            block_codec: 0xA2,
579            schema_meta_len: 0xD00D,
580            metadata_offset: 0x11,
581            metadata_len: 0x22,
582            dictionary_offset: 0x33,
583            dictionary_len: 0x44,
584            ..sample()
585        };
586        let b = h.to_bytes();
587        let u16_at = |o: usize| u16::from_le_bytes(b[o..o + 2].try_into().unwrap());
588        let u64_at = |o: usize| u64::from_le_bytes(b[o..o + 8].try_into().unwrap());
589
590        // core
591        assert_eq!(&b[0..4], b"RETE");
592        assert_eq!(b[4], CURRENT_FORMAT_VERSION);
593        assert_eq!(b[5], FLAG_HAS_QUADS);
594        assert_eq!(u16_at(6), HEADER_LEN as u16);
595        assert_eq!(&b[8..24], &[0xCC; 16]); // content hash
596        assert_eq!(u64_at(24), 0x99); // quad count
597        assert_eq!(u64_at(32), 0xAA); // term count
598        assert_eq!(u16_at(40), 0xABCD); // pyramid levels
599        assert_eq!(b[42], 0xA1); // dict codec
600        assert_eq!(b[43], 0xA2); // block codec
601        assert_eq!(u16_at(44), 5); // section_count: the 5 known sections
602        assert_eq!(u32::from_le_bytes(b[46..50].try_into().unwrap()), 0xD00D); // schema-meta len
603                                                                               // first directory entry = Metadata, at offset 64
604        assert_eq!(u16_at(64), 1); // kind = Metadata
605        assert_eq!(u64_at(72), 0x11); // metadata offset
606        assert_eq!(u64_at(80), 0x22); // metadata length
607                                      // second entry = Dictionary, at 88
608        assert_eq!(u16_at(88), 2);
609        assert_eq!(u64_at(96), 0x33);
610        assert_eq!(u64_at(104), 0x44);
611        assert_eq!(b.len(), HEADER_LEN);
612    }
613
614    #[test]
615    fn expected_file_len_is_last_section_end_plus_footer() {
616        // sample(): named graphs end furthest, at 3434 + 48 = 3482.
617        assert_eq!(sample().expected_file_len(), Some(3482 + 4));
618
619        // A zero-length section's offset must not count (absent sections are
620        // written as (0, 0), and a stale offset with len 0 addresses nothing).
621        let mut h = sample();
622        h.text_index_offset = 1 << 40;
623        h.text_index_len = 0;
624        assert_eq!(h.expected_file_len(), Some(3482 + 4));
625
626        // An empty file (header + footer only) is 1028 bytes.
627        let mut e = sample();
628        e.metadata_len = 0;
629        e.dictionary_len = 0;
630        e.root_dir_len = 0;
631        e.pyramid_meta_len = 0;
632        e.named_graphs_len = 0;
633        assert_eq!(e.expected_file_len(), Some(1024 + 4));
634
635        // A crafted header whose entry overflows u64 yields None, not a panic
636        // or a wrapped (tiny) length.
637        let mut c = sample();
638        c.named_graphs_offset = u64::MAX - 8;
639        c.named_graphs_len = 64;
640        assert_eq!(c.expected_file_len(), None);
641
642        // Unknown (future) sections extend the file too.
643        let f = sample().with_section(SectionKind::Unknown(99), 1 << 20, 512);
644        assert_eq!(f.expected_file_len(), Some((1 << 20) + 512 + 4));
645    }
646
647    #[test]
648    fn rejects_bad_magic() {
649        let mut bytes = [0u8; HEADER_LEN];
650        bytes[4] = CURRENT_FORMAT_VERSION;
651        assert!(matches!(
652            Header::from_bytes(&bytes),
653            Err(HeaderError::BadMagic)
654        ));
655    }
656
657    #[test]
658    fn stable_reader_accepts_v1_baseline_and_rejects_pre_v1() {
659        let current = sample().to_bytes();
660        assert_eq!(current[4], 0x05);
661        assert_eq!(Header::from_bytes(&current).unwrap().version, 0x05);
662
663        for old in 0x01..=0x04 {
664            let mut bytes = current;
665            bytes[4] = old;
666            let error = Header::from_bytes(&bytes).unwrap_err();
667            assert!(matches!(
668                &error,
669                HeaderError::UnsupportedVersion {
670                    found,
671                    min: 0x05,
672                    max: 0x05
673                } if *found == old
674            ));
675            assert!(error
676                .to_string()
677                .contains("Pre-1.0 files must be rebuilt from RDF source with `rete build`"));
678        }
679
680        for unsupported in [0x00, 0x06, 0xff] {
681            let mut bytes = current;
682            bytes[4] = unsupported;
683            assert!(matches!(
684                Header::from_bytes(&bytes),
685                Err(HeaderError::UnsupportedVersion {
686                    found,
687                    min: 0x05,
688                    max: 0x05
689                }) if found == unsupported
690            ));
691        }
692    }
693
694    #[test]
695    fn rejects_overrunning_section_count() {
696        let mut bad = sample().to_bytes();
697        bad[44..46].copy_from_slice(&9999u16.to_le_bytes());
698        assert!(matches!(
699            Header::from_bytes(&bad),
700            Err(HeaderError::BadSectionCount(9999))
701        ));
702    }
703
704    #[test]
705    fn unknown_section_survives_round_trip() {
706        // A section a future build added (kind 99) must be preserved verbatim by a
707        // reader that doesn't know it, and readable via `section()`.
708        let h = sample().with_section(SectionKind::Unknown(99), 4096, 512);
709        let back = Header::from_bytes(&h.to_bytes()).unwrap();
710        assert_eq!(back.extra_sections.len(), 1);
711        let s = back.section(SectionKind::Unknown(99)).unwrap();
712        assert_eq!((s.offset, s.length), (4096, 512));
713        // Known sections still resolve.
714        let dict = back.section(SectionKind::Dictionary).unwrap();
715        assert_eq!(dict.offset, h.dictionary_offset);
716        assert_eq!(h, back);
717    }
718
719    #[test]
720    fn build_info_section_round_trips_and_is_optional() {
721        // Absent (len 0): only the 5 always-present sections — a file without a
722        // build-info section is byte-identical to one built before it existed.
723        let plain = sample().to_bytes();
724        assert_eq!(u16::from_le_bytes(plain[44..46].try_into().unwrap()), 5);
725        assert_eq!(sample().section(SectionKind::BuildInfo).unwrap().length, 0);
726
727        // Present: a directory entry that round-trips and resolves as kind 7.
728        let h = sample().with_section(SectionKind::BuildInfo, 1066, 300);
729        let bytes = h.to_bytes();
730        assert_eq!(u16::from_le_bytes(bytes[44..46].try_into().unwrap()), 6);
731        let back = Header::from_bytes(&bytes).unwrap();
732        assert_eq!(h, back);
733        let s = back.section(SectionKind::BuildInfo).unwrap();
734        assert_eq!((s.offset, s.length), (1066, 300));
735        assert!(back.extra_sections.is_empty(), "BuildInfo is a known kind");
736        // And it extends the expected file length like any other section.
737        let far = sample().with_section(SectionKind::BuildInfo, 1 << 21, 128);
738        assert_eq!(far.expected_file_len(), Some((1 << 21) + 128 + 4));
739    }
740
741    #[test]
742    fn text_index_section_round_trips_and_is_optional() {
743        // Absent (len 0): only the 5 always-present sections, so a file without a
744        // text index is byte-identical to one built before this section existed.
745        assert_eq!(
746            u16::from_le_bytes(sample().to_bytes()[44..46].try_into().unwrap()),
747            5
748        );
749        assert!(sample().section(SectionKind::TextIndex).unwrap().length == 0);
750
751        // Present: a 6th directory entry that round-trips and resolves.
752        let h = sample().with_section(SectionKind::TextIndex, 5000, 4096);
753        let bytes = h.to_bytes();
754        assert_eq!(u16::from_le_bytes(bytes[44..46].try_into().unwrap()), 6);
755        let back = Header::from_bytes(&bytes).unwrap();
756        assert_eq!(h, back);
757        let s = back.section(SectionKind::TextIndex).unwrap();
758        assert_eq!((s.offset, s.length), (5000, 4096));
759        assert!(back.extra_sections.is_empty(), "TextIndex is a known kind");
760    }
761}