Skip to main content

urna_format/sections/
chunk_ids.rs

1//! `chunk_ids` section (`SECTION_CHUNK_IDS = 0x01`). Length-prefixed
2//! UTF-8 strings of the form `sha256:<64 hex>` - one per chunk.
3//!
4//! the raw encoding stores each id as 71 ascii bytes. the `intpack`
5//! repack (encoding id 4, kind 0) stores only the 32 raw digest bytes
6//! per id (~2.3x smaller) and reconstructs the exact ascii payload on
7//! decode, so `content_hash` (computed over the decoded bytes) is
8//! byte-identical to the raw form and citations stay stable.
9
10use super::REPACK_KIND_CHUNK_IDS;
11use super::codec::{Cursor, read_prefix, write_lp_str, write_prefix};
12use crate::bytes::le_u32;
13use crate::error::UrnaError;
14use crate::layout::SECTION_CHUNK_IDS;
15
16/// canonical chunk-id prefix; the rest is 64 lowercase hex digits.
17const SHA256_PREFIX: &str = "sha256:";
18const DIGEST_LEN: usize = 32;
19
20pub fn encode_chunk_ids(ids: &[String]) -> crate::Result<Vec<u8>> {
21    let mut buf = Vec::new();
22    write_prefix(&mut buf, ids.len() as u64);
23    for id in ids {
24        write_lp_str(&mut buf, id)?;
25    }
26    Ok(buf)
27}
28
29pub fn decode_chunk_ids(data: &[u8], expected_count: usize) -> crate::Result<Vec<String>> {
30    let mut c = Cursor::new(data, SECTION_CHUNK_IDS);
31    let count = read_prefix(&mut c)? as usize;
32    if count != expected_count {
33        return Err(UrnaError::SectionCountMismatch {
34            section_id: SECTION_CHUNK_IDS,
35            expected: expected_count,
36            got: count,
37        });
38    }
39    let mut ids = Vec::with_capacity(count);
40    for _ in 0..count {
41        ids.push(c.read_lp_str()?);
42    }
43    c.finish()?;
44    Ok(ids)
45}
46
47/// encode the chunk-ids section as an `intpack` repack payload: a kind
48/// byte, the count, then the 32 raw digest bytes per id. returns `None`
49/// when any id is not a canonical `sha256:<64 lowercase hex>` string, so
50/// the writer falls back to the raw encoding and reconstruction stays
51/// guaranteed byte-exact.
52pub fn encode_chunk_ids_intpack(ids: &[String]) -> Option<Vec<u8>> {
53    let mut digests: Vec<u8> = Vec::with_capacity(ids.len() * DIGEST_LEN);
54    for id in ids {
55        let hex_part = id.strip_prefix(SHA256_PREFIX)?;
56        if hex_part.len() != DIGEST_LEN * 2 {
57            return None;
58        }
59        let d = hex::decode(hex_part).ok()?;
60        // guard against any non-canonical (e.g. uppercase) hex so the
61        // round-trip reproduces the original ascii byte-for-byte.
62        if hex::encode(&d) != hex_part {
63            return None;
64        }
65        digests.extend_from_slice(&d);
66    }
67    let mut out = Vec::with_capacity(1 + 4 + digests.len());
68    out.push(REPACK_KIND_CHUNK_IDS);
69    out.extend_from_slice(&(ids.len() as u32).to_le_bytes());
70    out.extend_from_slice(&digests);
71    Some(out)
72}
73
74/// reconstruct the canonical (raw-encoding) chunk-ids payload from the
75/// body of an `intpack` repack (the bytes after the kind byte). the
76/// output is byte-identical to [`encode_chunk_ids`] so `content_hash`
77/// is preserved.
78pub fn decode_chunk_ids_intpack(rest: &[u8]) -> crate::Result<Vec<u8>> {
79    let malformed = |reason: &str| UrnaError::MalformedSectionPayload {
80        section_id: SECTION_CHUNK_IDS,
81        reason: reason.into(),
82    };
83    if rest.len() < 4 {
84        return Err(malformed("chunk_ids intpack: truncated count"));
85    }
86    let count = le_u32(&rest[0..4])? as usize;
87    let body = &rest[4..];
88    if body.len() != count * DIGEST_LEN {
89        return Err(malformed("chunk_ids intpack: digest body size mismatch"));
90    }
91    let mut buf = Vec::with_capacity(12 + count * (4 + 71));
92    write_prefix(&mut buf, count as u64);
93    for d in body.chunks_exact(DIGEST_LEN) {
94        let s = format!("{}{}", SHA256_PREFIX, hex::encode(d));
95        write_lp_str(&mut buf, &s)?;
96    }
97    Ok(buf)
98}
99
100#[cfg(test)]
101mod tests {
102    use super::*;
103    use crate::layout::SECTION_CHUNK_IDS;
104
105    #[test]
106    fn roundtrip() {
107        let ids = vec!["sha256:aaa".to_string(), "sha256:bbb".to_string()];
108        let bytes = encode_chunk_ids(&ids).unwrap();
109        let back = decode_chunk_ids(&bytes, 2).unwrap();
110        assert_eq!(ids, back);
111    }
112
113    #[test]
114    fn count_mismatch() {
115        let ids = vec!["a".to_string()];
116        let bytes = encode_chunk_ids(&ids).unwrap();
117        let err = decode_chunk_ids(&bytes, 5).unwrap_err();
118        assert!(matches!(err, UrnaError::SectionCountMismatch { .. }));
119    }
120
121    fn sample_ids() -> Vec<String> {
122        (0u8..4)
123            .map(|i| format!("sha256:{}", hex::encode([i; 32])))
124            .collect()
125    }
126
127    #[test]
128    fn intpack_decodes_byte_identical_to_raw() {
129        // the whole point: the packed form must decode to the exact raw
130        // payload so content_hash (over decoded bytes) is unchanged.
131        let ids = sample_ids();
132        let packed = encode_chunk_ids_intpack(&ids).unwrap();
133        assert_eq!(packed[0], REPACK_KIND_CHUNK_IDS);
134        // packed is ~2.3x smaller than the raw ascii payload.
135        let raw = encode_chunk_ids(&ids).unwrap();
136        assert!(packed.len() < raw.len());
137        let reconstructed = decode_chunk_ids_intpack(&packed[1..]).unwrap();
138        assert_eq!(reconstructed, raw, "intpack repack must rebuild raw bytes");
139        // and the reconstructed payload decodes back to the same ids.
140        assert_eq!(decode_chunk_ids(&reconstructed, ids.len()).unwrap(), ids);
141    }
142
143    #[test]
144    fn intpack_rejects_non_canonical_ids() {
145        assert!(encode_chunk_ids_intpack(&["not-a-hash".to_string()]).is_none());
146        assert!(encode_chunk_ids_intpack(&["sha256:ABCD".to_string()]).is_none());
147        // uppercase hex is not the canonical lowercase form.
148        let up = format!("sha256:{}", "A".repeat(64));
149        assert!(encode_chunk_ids_intpack(&[up]).is_none());
150    }
151
152    #[test]
153    fn intpack_empty_roundtrips() {
154        let packed = encode_chunk_ids_intpack(&[]).unwrap();
155        let reconstructed = decode_chunk_ids_intpack(&packed[1..]).unwrap();
156        assert_eq!(reconstructed, encode_chunk_ids(&[]).unwrap());
157    }
158
159    #[test]
160    fn intpack_body_size_mismatch_errors() {
161        assert!(decode_chunk_ids_intpack(&[]).is_err());
162        let mut bad = 2u32.to_le_bytes().to_vec();
163        bad.extend_from_slice(&[0u8; 40]); // 40 != 2*32
164        assert!(decode_chunk_ids_intpack(&bad).is_err());
165    }
166
167    #[test]
168    fn rejects_unsupported_version() {
169        let mut buf = Vec::new();
170        buf.extend_from_slice(&99u32.to_le_bytes());
171        buf.extend_from_slice(&0u64.to_le_bytes());
172        let err = decode_chunk_ids(&buf, 0).unwrap_err();
173        assert!(matches!(
174            err,
175            UrnaError::UnsupportedSectionVersion {
176                section_id: SECTION_CHUNK_IDS,
177                version: 99,
178            }
179        ));
180    }
181}