Skip to main content

codec_cbor/
cid.rs

1// SPDX-FileCopyrightText: Copyright © 2026 ReallyMe LLC. All rights reserved
2//
3// SPDX-License-Identifier: MIT OR Apache-2.0
4
5use cid::multibase::{decode as multibase_decode, Base};
6use cid::Cid;
7use multihash::Multihash;
8use multihash_codetable::{Code, MultihashDigest};
9use sha2::{Digest, Sha256};
10
11/// dag-cbor multicodec code (IPLD)
12pub const DAG_CBOR_CODEC: u64 = 0x71;
13
14/// Maximum CID string size accepted before multibase decoding.
15///
16/// A CID with the supported 64-byte digest and four u64 varints occupies at
17/// most 104 binary bytes. This budget accommodates even base2 and the UTF-8
18/// base256emoji representation, while bounding quadratic base conversions.
19pub const MAX_CID_STRING_LEN: usize = 1024;
20
21const CID_V0_STRING_LEN: usize = 46;
22
23/// Hash output for sha2-256
24pub type ContentHash = [u8; 32];
25
26/// Multihash envelope size used by the CID stack for sha2-256 digests.
27pub type DagCborMultihash = Multihash<64>;
28
29/// Returns the raw sha2-256 digest of `bytes`.
30pub fn sha2_256_content_hash(bytes: &[u8]) -> ContentHash {
31    Sha256::digest(bytes).into()
32}
33
34/// Returns a sha2-256 multihash of `bytes` for use in a CID.
35pub fn dag_cbor_multihash(bytes: &[u8]) -> DagCborMultihash {
36    Code::Sha2_256.digest(bytes)
37}
38
39/// Computes the CIDv1 (dag-cbor, sha2-256) of `bytes` in canonical
40/// base32-lower string form.
41///
42/// Hashes the supplied bytes as-is, without parsing CBOR or applying the
43/// encoder/decoder size limit. Encode a value first to obtain a canonical block.
44pub fn compute_cid_dag_cbor(bytes: &[u8]) -> String {
45    let hash = dag_cbor_multihash(bytes);
46    let cid = Cid::new_v1(DAG_CBOR_CODEC, hash);
47    cid.to_string()
48}
49
50/// Recomputes the CID of `bytes` and compares it to `cid_str`.
51///
52/// Returns whether the parsed CID values match, plus the expected CID and the
53/// parsed actual CID in canonical string form. Invalid CID input never matches
54/// and returns an empty actual string so unvalidated caller input does not cross
55/// diagnostic or FFI boundaries.
56/// Like [`compute_cid_dag_cbor`], this hashes bytes without validating CBOR.
57/// Verification rejects uppercase base16, base32, and base36 variants, including
58/// uppercase payloads with lowercase prefixes; generic [`try_parse_cid`] also
59/// accepts those case variants.
60pub fn verify_dag_cbor_cid(cid_str: &str, bytes: &[u8]) -> (bool, String, String) {
61    let expected_hash = dag_cbor_multihash(bytes);
62    let expected_cid = Cid::new_v1(DAG_CBOR_CODEC, expected_hash);
63    let expected = expected_cid.to_string();
64    let Some(actual_cid) = parse_verification_cid(cid_str) else {
65        return (false, expected, String::new());
66    };
67    let actual = actual_cid.to_string();
68    (expected_cid == actual_cid, expected, actual)
69}
70
71fn parse_verification_cid(cid_str: &str) -> Option<Cid> {
72    let (actual, base) = parse_cid_string(cid_str)?;
73    let Some(base) = base else {
74        return Some(actual);
75    };
76    if rejects_case_variant_base(base) || has_uppercase_payload_for_lowercase_base(base, cid_str) {
77        return None;
78    }
79    Some(actual)
80}
81
82fn rejects_case_variant_base(base: Base) -> bool {
83    matches!(
84        base,
85        Base::Base16Upper
86            | Base::Base32Upper
87            | Base::Base32PadUpper
88            | Base::Base32HexUpper
89            | Base::Base32HexPadUpper
90            | Base::Base36Upper
91    )
92}
93
94fn has_uppercase_payload_for_lowercase_base(base: Base, cid_str: &str) -> bool {
95    if !matches!(
96        base,
97        Base::Base16Lower
98            | Base::Base32Lower
99            | Base::Base32PadLower
100            | Base::Base32HexLower
101            | Base::Base32HexPadLower
102            | Base::Base36Lower
103    ) {
104        return false;
105    }
106    cid_str
107        .get(1..)
108        .is_some_and(|payload| payload.bytes().any(|byte| byte.is_ascii_uppercase()))
109}
110
111/// Returns whether `s` parses as a valid CID string.
112pub fn is_valid_cid_string(s: &str) -> bool {
113    try_parse_cid(s).is_some()
114}
115
116/// Parses `s` as a CID, returning `None` if it is not a valid CID string.
117///
118/// Accepts CIDv0 and multibase CID strings up to [`MAX_CID_STRING_LEN`]. Paths,
119/// non-minimal binary encodings, and trailing decoded bytes are rejected.
120pub fn try_parse_cid(s: &str) -> Option<Cid> {
121    parse_cid_string(s).map(|(cid, _base)| cid)
122}
123
124fn parse_cid_string(s: &str) -> Option<(Cid, Option<Base>)> {
125    if s.len() > MAX_CID_STRING_LEN {
126        return None;
127    }
128    // The upstream string convenience parser also extracts CIDs from paths.
129    // Decode the entire identifier ourselves so no prefix can be discarded.
130    let (base, decoded) = if s.len() == CID_V0_STRING_LEN && s.starts_with("Qm") {
131        (None, Base::Base58Btc.decode(s).ok()?)
132    } else {
133        let (base, bytes) = multibase_decode(s).ok()?;
134        (Some(base), bytes)
135    };
136    let mut remaining = decoded.as_slice();
137    let cid = Cid::read_bytes(&mut remaining).ok()?;
138    // read_bytes is a stream parser. Exhaustion is essential for validating
139    // an identifier, and byte equality also enforces minimal varint forms.
140    if !remaining.is_empty() || cid.to_bytes() != decoded {
141        return None;
142    }
143    Some((cid, base))
144}