Skip to main content

urna_format/reader/
decode.rs

1//! Decoded section access + identity hashes (file_hash, content_hash).
2
3use std::borrow::Cow;
4
5use super::UrnaView;
6use crate::bytes::le_u64;
7use crate::encoding::{decode_dedup_map, decode_payload, decode_payload_with_dict, expand_dedup};
8use crate::error::UrnaError;
9use crate::layout::{
10    CANONICAL_SECTIONS, SECTION_CHUNKS_CANONICAL, SECTION_DEDUP_MAP, SECTION_DICTIONARY,
11    SECTION_SEARCH_CONTRACT,
12};
13use crate::sections::{
14    SearchContract, decode_chunks_canonical, decode_search_contract, encode_chunks_canonical,
15};
16
17impl<'a> UrnaView<'a> {
18    /// Logical (decoded) bytes of a section's payload. Borrows for raw
19    /// encoding; copies for zstd. Float16/int8 embedding payloads are
20    /// returned as-is - the runtime dispatches on `manifest.dtype`.
21    ///
22    /// The chunks_canonical (0x02) section gets two extra, content_hash-
23    /// invariant rewrites here so its decoded bytes are byte-identical to a
24    /// plain build: a dict-framed (`zstd_dict`, id 5) payload is decoded
25    /// against the shared dictionary in section 0x0A, and a deduped pool is
26    /// re-expanded through the back-reference array in section 0x0B. Both
27    /// optional sections are excluded from content_hash, so they never move a
28    /// citation; the re-expansion happens BEFORE content_hash sees the bytes.
29    pub fn decoded_section(&self, section_id: u32) -> crate::Result<Cow<'a, [u8]>> {
30        if section_id == SECTION_CHUNKS_CANONICAL {
31            return self.decoded_chunks_canonical();
32        }
33        self.decoded_section_plain(section_id)
34    }
35
36    /// Decode a section that needs no dict/dedup context (everything but
37    /// chunks_canonical, which `decoded_section` special-cases).
38    fn decoded_section_plain(&self, section_id: u32) -> crate::Result<Cow<'a, [u8]>> {
39        let entry = self.entry(section_id)?;
40        let phys = self.get_section_data(section_id)?;
41        decode_payload(entry.encoding, phys).map_err(|e| Self::tag_err(section_id, e))
42    }
43
44    /// Decode chunks_canonical with dict (0x0A) + dedup (0x0B) awareness,
45    /// rebuilding the byte-identical canonical payload a plain build emits.
46    fn decoded_chunks_canonical(&self) -> crate::Result<Cow<'a, [u8]>> {
47        let entry = self.entry(SECTION_CHUNKS_CANONICAL)?;
48        let phys = self.get_section_data(SECTION_CHUNKS_CANONICAL)?;
49        let dict = self
50            .entry(SECTION_DICTIONARY)
51            .ok()
52            .and_then(|_| self.get_section_data(SECTION_DICTIONARY).ok());
53        let decoded = decode_payload_with_dict(entry.encoding, phys, dict)
54            .map_err(|e| Self::tag_err(SECTION_CHUNKS_CANONICAL, e))?;
55        // no dedup map => the decoded bytes already are the full canonical
56        // payload (dict/fsst/zstd all decode byte-identically to raw).
57        if self.entry(SECTION_DEDUP_MAP).is_err() {
58            return Ok(decoded);
59        }
60        // dedup map present: the section holds only the first-seen unique
61        // pool; re-expand it through the back-references to the exact original
62        // ordered byte stream BEFORE content_hash sees it.
63        let map_phys = self.get_section_data(SECTION_DEDUP_MAP)?;
64        let back_refs =
65            decode_dedup_map(map_phys).map_err(|e| Self::tag_err(SECTION_DEDUP_MAP, e))?;
66        let unique = decode_chunks_canonical(&decoded, count_prefix(&decoded)?)?;
67        let full = expand_dedup(&unique, &back_refs)?;
68        Ok(Cow::Owned(encode_chunks_canonical(&full)?))
69    }
70
71    fn tag_err(section_id: u32, e: UrnaError) -> UrnaError {
72        match e {
73            UrnaError::UnsupportedSectionEncoding { encoding, .. } => {
74                UrnaError::UnsupportedSectionEncoding {
75                    section_id,
76                    encoding,
77                }
78            }
79            UrnaError::MalformedSectionPayload { reason, .. } => {
80                UrnaError::MalformedSectionPayload { section_id, reason }
81            }
82            other => other,
83        }
84    }
85
86    /// Decode the `search_contract` section. Already validated to agree
87    /// with the manifest at construction time.
88    pub fn search_contract(&self) -> crate::Result<SearchContract> {
89        let bytes = self.decoded_section(SECTION_SEARCH_CONTRACT)?;
90        decode_search_contract(&bytes)
91    }
92
93    /// `sha256:<hex>` of the file as written, including the footer.
94    pub fn file_hash_hex(&self) -> String {
95        use sha2::{Digest, Sha256};
96        let h = Sha256::digest(self.data);
97        format!("sha256:{}", hex::encode(h))
98    }
99
100    /// `sha256:<hex>` of the canonical sections in the order fixed by spec
101    /// (see `CANONICAL_SECTIONS`). Hashes the **decoded** bytes so two
102    /// files that wire-compress the same logical content (zstd vs raw)
103    /// produce the same content_hash and therefore stable citations.
104    /// Quantized embeddings (float16 / int8) hash their on-disk bytes -
105    /// they're already the canonical representation for that precision.
106    /// Optional sections (HNSW, BM25, and every reserved 0x09+ section) are
107    /// NOT included, and neither is the manifest: the manifest is covered by
108    /// file_hash only. So additive manifest fields (a new Option field, a
109    /// `capabilities_ext` flag) move file_hash but NEVER content_hash, which
110    /// is why adding a capability cannot invalidate a urna:// citation.
111    pub fn content_hash_hex(&self) -> crate::Result<String> {
112        use sha2::{Digest, Sha256};
113        let mut h = Sha256::new();
114        for (id, name) in CANONICAL_SECTIONS {
115            let bytes = self.decoded_section(*id)?;
116            // Domain-separate by name length + name bytes so hashes for
117            // different sections cannot collide via concatenation.
118            h.update((name.len() as u32).to_le_bytes());
119            h.update(name.as_bytes());
120            h.update((bytes.len() as u64).to_le_bytes());
121            h.update(bytes.as_ref());
122        }
123        Ok(format!("sha256:{}", hex::encode(h.finalize())))
124    }
125}
126
127/// read the u64 entry count from a canonical section payload's 12-byte
128/// prefix (u32 version + u64 count) so the unique pool can be decoded back
129/// to strings. bounds-checked; never panics on a short buffer.
130fn count_prefix(payload: &[u8]) -> crate::Result<usize> {
131    if payload.len() < 12 {
132        return Err(UrnaError::MalformedSectionPayload {
133            section_id: SECTION_CHUNKS_CANONICAL,
134            reason: "chunks_canonical: truncated count prefix".into(),
135        });
136    }
137    Ok(le_u64(&payload[4..12])? as usize)
138}