Skip to main content

khive_storage/
blob.rs

1//! Blob storage capability — content-addressed binary object CRUD.
2//!
3//! `BlobStore` is the trait family added by khive#292: bytes that do not
4//! belong inside the primary SQLite database (source PDFs, images, large
5//! opaque payloads) are stored by a dedicated backend and referenced from
6//! the graph by an opaque [`ContentRef`]. Per ADR-005's "zero
7//! implementations" constraint, this module defines the contract only — the
8//! first backend (filesystem, BLAKE3-addressed) lives in `khive-db`.
9
10use std::collections::HashSet;
11
12use async_trait::async_trait;
13use serde::{Deserialize, Serialize};
14
15use crate::capability::StorageCapability;
16use crate::error::StorageError;
17use crate::sql::SqlAccess;
18use crate::types::StorageResult;
19
20/// Number of hex characters in a BLAKE3-256 digest (32 bytes -> 64 hex chars).
21const CONTENT_REF_HEX_LEN: usize = 64;
22
23/// Portable v1 ceiling for a whole-buffer blob operation (64 MiB).
24///
25/// Callers needing larger objects require a future streaming contract. A
26/// [`BlobStore::get_bounded_verified`] request above this limit is invalid
27/// even when the selected backend could otherwise satisfy it.
28pub const MAX_BLOB_WHOLE_BYTES: u64 = 64 * 1024 * 1024;
29
30/// An opaque, content-addressed reference to a stored blob.
31///
32/// Backed by a lowercase-hex BLAKE3 digest of the blob's bytes: identical
33/// content always produces the same `ContentRef`, so storing the same bytes
34/// twice is a no-op after the first write. Callers must treat the value as
35/// opaque — the backend, not the caller, decides how a `ContentRef` maps to
36/// physical storage.
37///
38/// `Deserialize` is hand-written (below) to reject any string that is not 64
39/// lowercase hex characters — a naive derive would let an unvalidated value
40/// panic later in `shard_path`'s slicing.
41/// See `crates/khive-storage/docs/api/blob-store.md` for the full rationale.
42#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize)]
43#[serde(transparent)]
44pub struct ContentRef(String);
45
46/// Opaque capability for a backend's staged upload, encoded as 128-bit hex.
47///
48/// This is not a content reference. The caller owns hashing and upload-session
49/// state; retaining this identifier does not make a session restartable.
50#[derive(Clone, Debug, PartialEq, Eq, Hash, Serialize)]
51#[serde(transparent)]
52pub struct UploadId(String);
53
54impl UploadId {
55    /// Construct an identifier from freshly generated random bytes.
56    pub fn from_bytes(bytes: &[u8; 16]) -> Self {
57        Self(hex_encode(bytes))
58    }
59
60    /// Parse exactly 32 lowercase hex characters, never a backend pathname.
61    pub fn from_hex(value: impl Into<String>) -> Result<Self, String> {
62        let value = value.into();
63        if value.len() != 32
64            || !value
65                .bytes()
66                .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte))
67        {
68            return Err("upload_id must be 32 lowercase hex characters".into());
69        }
70        Ok(Self(value))
71    }
72
73    /// Canonical wire spelling.
74    pub fn as_str(&self) -> &str {
75        &self.0
76    }
77}
78
79impl std::fmt::Display for UploadId {
80    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
81        f.write_str(self.as_str())
82    }
83}
84
85impl<'de> Deserialize<'de> for UploadId {
86    fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
87        Self::from_hex(String::deserialize(deserializer)?).map_err(serde::de::Error::custom)
88    }
89}
90
91impl<'de> Deserialize<'de> for ContentRef {
92    fn deserialize<D>(deserializer: D) -> Result<Self, D::Error>
93    where
94        D: serde::Deserializer<'de>,
95    {
96        let raw = String::deserialize(deserializer)?;
97        ContentRef::from_hex(raw).map_err(serde::de::Error::custom)
98    }
99}
100
101impl ContentRef {
102    /// Parse a `ContentRef` from a caller-supplied hex string.
103    ///
104    /// Rejects anything that is not exactly 64 lowercase hex characters.
105    /// Uppercase is rejected (not normalized) to keep one canonical string
106    /// form per digest — see `docs/api/blob-store.md`.
107    pub fn from_hex(hex: impl Into<String>) -> Result<Self, String> {
108        let hex = hex.into();
109        if hex.len() != CONTENT_REF_HEX_LEN {
110            return Err(format!(
111                "content_ref must be {CONTENT_REF_HEX_LEN} hex characters, got length {} ({hex:?})",
112                hex.len()
113            ));
114        }
115        if !hex
116            .bytes()
117            .all(|b| b.is_ascii_digit() || (b.is_ascii_lowercase() && b.is_ascii_hexdigit()))
118        {
119            return Err(format!(
120                "content_ref must be lowercase hex (0-9, a-f), got {hex:?}"
121            ));
122        }
123        Ok(Self(hex))
124    }
125
126    /// Construct a `ContentRef` directly from a BLAKE3 digest's raw bytes.
127    pub fn from_digest_bytes(digest: &[u8; 32]) -> Self {
128        Self(hex_encode(digest))
129    }
130
131    /// Borrow the underlying hex string.
132    pub fn as_str(&self) -> &str {
133        &self.0
134    }
135}
136
137impl std::fmt::Display for ContentRef {
138    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
139        f.write_str(&self.0)
140    }
141}
142
143impl AsRef<str> for ContentRef {
144    fn as_ref(&self) -> &str {
145        &self.0
146    }
147}
148
149fn hex_encode(bytes: &[u8]) -> String {
150    const HEX: &[u8; 16] = b"0123456789abcdef";
151    let mut out = String::with_capacity(bytes.len() * 2);
152    for &b in bytes {
153        out.push(HEX[(b >> 4) as usize] as char);
154        out.push(HEX[(b & 0x0f) as usize] as char);
155    }
156    out
157}
158
159/// Configuration for [`BlobStore::orphan_sweep`].
160///
161/// `live_refs` is a point-in-time snapshot the caller assembles (this trait
162/// has no visibility into SQL substrates — ADR-005 constraint 4), not a live
163/// query. See [`BlobStore::orphan_sweep`] for the concurrency hazard this
164/// implies, and `crates/khive-storage/docs/api/blob-store.md` for the full
165/// rationale.
166#[derive(Clone, Debug, Default, Serialize, Deserialize)]
167pub struct BlobOrphanSweepConfig {
168    /// Content refs currently referenced by at least one committed record
169    /// attachment, as of when the caller assembled this set. Anything
170    /// this backend stores that is NOT in this set is treated as orphaned
171    /// and deleted (or reported, in `dry_run` mode) — including a
172    /// `content_ref` that becomes live after this snapshot was taken.
173    pub live_refs: HashSet<ContentRef>,
174    /// When `true`, report what would be deleted without deleting anything.
175    pub dry_run: bool,
176}
177
178/// Result of a [`BlobStore::orphan_sweep`] call.
179#[derive(Clone, Debug, Default, Serialize, Deserialize)]
180pub struct BlobOrphanSweepResult {
181    /// Total objects examined in this backend.
182    pub scanned: u64,
183    /// Objects actually deleted (always 0 when `dry_run = true`).
184    pub deleted: u64,
185    /// Objects that are orphaned (would be deleted whether or not `dry_run`
186    /// is set — populated in both modes so a dry run reports the same count
187    /// a real run would delete).
188    pub would_delete: u64,
189    /// Objects with zero live references that were left alone because they
190    /// are still inside their publish grace period — recently written and
191    /// not yet orphaned, just not yet referenced by a record attachment.
192    /// Reported in both modes; never counted in `would_delete` or `deleted`.
193    pub grace_period_skipped: u64,
194}
195
196/// Immutable owner policy conveyed by the host, never by upload wire arguments.
197#[derive(Clone, Copy, Debug)]
198pub struct UploadLeaseConfig {
199    owner: uuid::Uuid,
200    idle_secs: u64,
201}
202
203impl UploadLeaseConfig {
204    /// Construct a positive whole-second bound; the backend enforces its cap.
205    pub fn new(owner: uuid::Uuid, idle: std::time::Duration) -> StorageResult<Self> {
206        if owner.is_nil() {
207            return Err(StorageError::InvalidInput {
208                capability: StorageCapability::Blob,
209                operation: "upload_lease_config".into(),
210                message: "upload lease owner must be a non-nil durable UUID".into(),
211            });
212        }
213        if idle.is_zero() || idle.subsec_nanos() != 0 {
214            return Err(StorageError::InvalidInput {
215                capability: StorageCapability::Blob,
216                operation: "upload_lease_config".into(),
217                message: "upload lease bound must be positive whole seconds".into(),
218            });
219        }
220        Ok(Self {
221            owner,
222            idle_secs: idle.as_secs(),
223        })
224    }
225
226    /// Validated durable MAIN/store identity supplied by the host.
227    pub fn owner(&self) -> uuid::Uuid {
228        self.owner
229    }
230    /// The owning daemon's fixed bound, in seconds.
231    pub fn idle_secs(&self) -> u64 {
232        self.idle_secs
233    }
234}
235
236/// Content-addressed binary object CRUD.
237///
238/// Every method is backend-agnostic: the filesystem backend
239/// (`khive-db::stores::blob::FsBlobStore`) is the first implementation, and
240/// any future backend (object storage, a different CAS layout) implements
241/// the same operations. Per ADR-005 constraint 4, a `BlobStore` instance
242/// talks to exactly one backend.
243// `Debug` is a supertrait so boot-path tests can distinguish which concrete
244// backend was installed behind `Arc<dyn BlobStore>` via `format!("{:?}", ..)`
245// without adding a downcast/type-name method to the production surface.
246#[async_trait]
247pub trait BlobStore: Send + Sync + std::fmt::Debug + 'static {
248    /// Store `bytes`, returning the content-addressed reference under which
249    /// they are now retrievable. Storing byte-identical content more than
250    /// once returns the same `ContentRef` and does not re-write the object.
251    async fn put(&self, bytes: Vec<u8>) -> StorageResult<ContentRef>;
252
253    /// Restart the publish grace period for an existing object and return its
254    /// size, or `None` if absent. Backends with a live orphan sweep must make
255    /// the existence check and refresh atomic with that sweep's root lock.
256    /// The default is sufficient for backends without a live orphan sweep.
257    async fn refresh_publish_grace(&self, content_ref: &ContentRef) -> StorageResult<Option<u64>> {
258        self.size(content_ref).await
259    }
260
261    /// Lease policy capability, not an authorization or ownership decision.
262    /// None preserves non-filesystem policy and explicit Unsupported defaults.
263    fn upload_lease_idle_cap(&self) -> Option<std::time::Duration> {
264        None
265    }
266
267    /// Create and sync staging plus an atomic complete immutable-owner lease.
268    /// Hosts convey durable identity and the resolved bound before admission.
269    async fn begin_upload_with_lease(
270        &self,
271        declared_size: u64,
272        config: UploadLeaseConfig,
273    ) -> StorageResult<UploadId> {
274        let _ = (declared_size, config);
275        Err(unsupported_upload("begin_upload_with_lease"))
276    }
277
278    /// Renew a validated lease without appending bytes, including tail resends.
279    /// Success follows file sync, replacing rename and supported dir barriers.
280    async fn renew_upload(&self, id: &UploadId) -> StorageResult<()> {
281        let _ = id;
282        Err(unsupported_upload("renew_upload"))
283    }
284
285    /// Create an empty staging object. The pack keeps the declared size,
286    /// incremental hash, sequence and idle clock. Backends enforce their
287    /// capacity policy on each append. Unsupported backends refuse explicitly.
288    async fn begin_upload(&self, declared_size: u64) -> StorageResult<UploadId> {
289        let _ = declared_size;
290        Err(unsupported_upload("begin_upload"))
291    }
292
293    /// Append and synchronize bytes, returning the total staged length.
294    /// Leased backends synchronize bytes then renew under one root ownership.
295    /// The caller serializes parts and aborts after any uncertain append/renewal.
296    async fn append_part(&self, id: &UploadId, bytes: Vec<u8>) -> StorageResult<u64> {
297        let _ = (id, bytes);
298        Err(unsupported_upload("append_part"))
299    }
300
301    /// Publish through the same routine as put, without hashing a second time.
302    /// The caller proves the supplied digest and declared size before invoking
303    /// this method. Success consumes the staging object, including on dedup.
304    async fn commit_upload(&self, id: &UploadId, content_ref: &ContentRef) -> StorageResult<()> {
305        let _ = (id, content_ref);
306        Err(unsupported_upload("commit_upload"))
307    }
308
309    /// Discard staging; an already absent staging object is a successful no-op.
310    async fn abort_upload(&self, id: &UploadId) -> StorageResult<()> {
311        let _ = id;
312        Err(unsupported_upload("abort_upload"))
313    }
314
315    /// Remove staging objects the backend can show are abandoned. The filesystem
316    /// backend ignores `idle_for`: it removes a leased upload once it has seen that
317    /// lease unchanged for the lease's own bound plus a fixed margin, or once the
318    /// lease's last renewal plus its bound plus 24 hours has passed, and an upload
319    /// with no lease once its file is 24 hours old. Backends without staged uploads
320    /// refuse. This never visits committed objects. Open S3 multipart uploads
321    /// require the deployment's incomplete-multipart lifecycle rule instead.
322    async fn sweep_uploads(&self, idle_for: std::time::Duration) -> StorageResult<u64> {
323        let _ = idle_for;
324        Err(unsupported_upload("sweep_uploads"))
325    }
326
327    /// Fetch at most `max_bytes` from `content_ref` and verify its BLAKE3
328    /// digest before returning any bytes.
329    ///
330    /// `max_bytes` may be zero (only an empty object can then succeed) and
331    /// must not exceed [`MAX_BLOB_WHOLE_BYTES`]. Implementations must enforce
332    /// the limit while reading the authoritative object, not by composing a
333    /// metadata-only [`Self::size`] check with another read. A successful
334    /// result is complete, no larger than the declared maximum,
335    /// metadata-size-consistent, and digest-matched to `content_ref`.
336    async fn get_bounded_verified(
337        &self,
338        content_ref: &ContentRef,
339        max_bytes: u64,
340    ) -> StorageResult<Vec<u8>>;
341
342    /// Whether an object currently exists for `content_ref`.
343    async fn exists(&self, content_ref: &ContentRef) -> StorageResult<bool>;
344
345    /// The size in bytes of the object stored under `content_ref`, without
346    /// hydrating its bytes.
347    ///
348    /// Returns `Ok(None)` when no object exists for this reference — this is
349    /// the existence check and the size read in one call, so a caller never
350    /// pays for a full read just to answer "does this exist and how big is
351    /// it". On the filesystem backend this maps to a file metadata stat; on
352    /// an object-storage backend it maps to a `HEAD Object` request.
353    async fn size(&self, content_ref: &ContentRef) -> StorageResult<Option<u64>>;
354
355    /// Remove the object stored under `content_ref`.
356    ///
357    /// Returns `true` when an object was actually removed, `false` when
358    /// none existed — deleting an absent object is not an error.
359    ///
360    /// # Safety / concurrency hazard (ADR-111 §8, amended)
361    ///
362    /// Unconditional physical removal with **no coordination against any
363    /// record or attachment that might reference `content_ref`**. Safe to call only when
364    /// the caller has independently quiesced every writer that could commit a
365    /// new SQL liveness reference for the duration of the call — this
366    /// trait does not detect or prevent a race. Offline-maintenance-only.
367    /// See `crates/khive-storage/docs/api/blob-store.md`.
368    async fn delete(&self, content_ref: &ContentRef) -> StorageResult<bool>;
369
370    /// Enumerate every object this backend holds and delete (or, in
371    /// `dry_run` mode, report) those absent from `config.live_refs`.
372    /// Operator-side GC path (khive#292 deliverable 5) — admin-only, not an
373    /// MCP verb. Default returns `StorageError::Unsupported`; the filesystem
374    /// backend currently returns the same typed refusal for every call (see
375    /// below) rather than performing a real directory walk.
376    ///
377    /// # Safety / concurrency hazard (ADR-111 §8, amended)
378    ///
379    /// `config.live_refs` is a **snapshot**; a `content_ref` that becomes
380    /// newly live between the snapshot and the sweep is deleted anyway.
381    /// **Callers MUST quiesce attachment writes** for the duration of
382    /// snapshot-plus-sweep. See `crates/khive-storage/docs/api/blob-store.md`
383    /// for the hazard. This API also has no [`SqlAccess`] capability with
384    /// which to prove a completed V21 attachment epoch, so — unlike
385    /// [`Self::transactional_orphan_sweep`] — it cannot honor that gate. The
386    /// filesystem backend therefore disables this method entirely in this
387    /// compatibility release, in both `dry_run` modes: concurrent AND
388    /// offline callers alike must use [`Self::transactional_orphan_sweep`]
389    /// instead.
390    async fn orphan_sweep(
391        &self,
392        config: &BlobOrphanSweepConfig,
393    ) -> StorageResult<BlobOrphanSweepResult> {
394        let _ = config;
395        Err(StorageError::Unsupported {
396            capability: StorageCapability::Blob,
397            operation: "orphan_sweep".into(),
398            message: "this backend does not support orphan sweep".into(),
399        })
400    }
401
402    /// Select live attachment references and sweep orphaned blobs behind a
403    /// database-coordinated, bounded claim protocol.
404    ///
405    /// Unlike [`Self::orphan_sweep`], this operation obtains liveness itself
406    /// from `sql`; callers do not assemble a stale snapshot. `sql` must be the
407    /// canonical main database capability used for the attachment writes that own
408    /// references. Implementations must also ensure an object published after
409    /// the sweep's candidate set is captured cannot be mistaken for an orphan,
410    /// including when it is published between selecting live references and
411    /// physical deletion. Implementations must not perform filesystem or
412    /// other external I/O while holding the database writer transaction;
413    /// durable claims/triggers or an equivalently fail-closed fence must keep
414    /// attachment writes safe after each short transaction commits. Claim/result
415    /// materialization and cleanup must have an explicit per-transaction
416    /// cardinality bound rather than scale one writer hold with the complete
417    /// object population. A file-backed `sql` implementation must expose its
418    /// canonical [`SqlAccess::database_path`] so crash recovery can retain
419    /// cross-process database ownership independently of mutable blob-root
420    /// spelling or relocation.
421    /// Coordination may be advisory, so callers must publish through the
422    /// backend rather than mutate its physical storage directly.
423    /// Backends that cannot provide both guarantees return
424    /// `StorageError::Unsupported`.
425    ///
426    /// The filesystem implementation is schema-epoch gated and supports both
427    /// report-only and destructive modes only when `sql` proves the exact
428    /// completed V21 attachment cutover: durable complete marker and ledger
429    /// row, attachment table/indexes and INSERT/UPDATE claim fences, and
430    /// absence of every legacy entity reference column/index/fence. V20,
431    /// pending, incomplete, missing-required-object, retained-legacy, and
432    /// ahead-of-V21 epochs return typed `Unsupported` before root locking,
433    /// filesystem walking, or abandoned-claim cleanup. Malformed stored
434    /// evidence or a nonfunctional named fence fails closed with its validation,
435    /// storage, or typed `Unsupported` error before claim cleanup or deletion.
436    /// Once admitted, every attachment role is live; soft deletion alone does
437    /// not make its blob collectible.
438    ///
439    /// This is the Phase-4a GC compatibility gate. Phase 4a changes no schema or
440    /// data. Every older process sharing the database/blob root must be drained
441    /// before Phase 4b performs the attachment backfill and legacy-column drop.
442    /// Phase-4a application readers/writers must also be quiesced during cutover;
443    /// only a GC-only worker has narrow compatibility with exact completed V21.
444    /// Callers must not fall back to [`Self::orphan_sweep`] or [`Self::delete`]
445    /// when this gate refuses.
446    ///
447    /// Publishing a blob and committing the attachment write that references it
448    /// are two separate client steps; nothing serializes them against this
449    /// sweep. Implementations must therefore also give a just-published,
450    /// not-yet-referenced object a bounded grace period before treating it as
451    /// an orphan (the filesystem backend does this via file age). A client
452    /// whose own gap between the two steps exceeds that grace period is not
453    /// protected — this narrows, but does not eliminate, the hazard.
454    async fn transactional_orphan_sweep(
455        &self,
456        sql: &dyn SqlAccess,
457        dry_run: bool,
458    ) -> StorageResult<BlobOrphanSweepResult> {
459        let _ = (sql, dry_run);
460        Err(StorageError::Unsupported {
461            capability: StorageCapability::Blob,
462            operation: "transactional_orphan_sweep".into(),
463            message: "this backend does not support a database-coordinated orphan sweep".into(),
464        })
465    }
466}
467
468fn unsupported_upload(operation: &'static str) -> StorageError {
469    StorageError::Unsupported {
470        capability: StorageCapability::Blob,
471        operation: operation.into(),
472        message: "this backend does not support staged uploads".into(),
473    }
474}
475
476#[cfg(test)]
477mod tests {
478    use super::*;
479
480    #[test]
481    fn upload_id_roundtrips_and_rejects_path_or_noncanonical_input() {
482        let id = UploadId::from_bytes(&[0xab; 16]);
483        assert_eq!(id.as_str(), "ab".repeat(16));
484        assert_eq!(
485            serde_json::from_str::<UploadId>(&serde_json::to_string(&id).unwrap()).unwrap(),
486            id
487        );
488        for invalid in [
489            String::new(),
490            "a".repeat(31),
491            "a".repeat(33),
492            "A".repeat(32),
493            "g".repeat(32),
494            "../outside".into(),
495            "a/b".repeat(11),
496        ] {
497            assert!(UploadId::from_hex(&invalid).is_err());
498            assert!(serde_json::from_value::<UploadId>(serde_json::json!(invalid)).is_err());
499        }
500    }
501
502    #[test]
503    fn from_hex_accepts_valid_lowercase_digest() {
504        let hex = "a".repeat(64);
505        let cref = ContentRef::from_hex(hex.clone()).unwrap();
506        assert_eq!(cref.as_str(), hex);
507        assert_eq!(cref.to_string(), hex);
508    }
509
510    #[test]
511    fn from_hex_rejects_short_string() {
512        let err = ContentRef::from_hex("abc").unwrap_err();
513        assert!(
514            err.contains("64"),
515            "error must mention expected length: {err}"
516        );
517    }
518
519    #[test]
520    fn from_hex_rejects_long_string() {
521        let err = ContentRef::from_hex("a".repeat(65)).unwrap_err();
522        assert!(
523            err.contains("64"),
524            "error must mention expected length: {err}"
525        );
526    }
527
528    #[test]
529    fn from_hex_rejects_uppercase() {
530        let err = ContentRef::from_hex("A".repeat(64)).unwrap_err();
531        assert!(
532            err.contains("lowercase"),
533            "error must mention lowercase requirement: {err}"
534        );
535    }
536
537    #[test]
538    fn from_hex_rejects_non_hex_characters() {
539        let mut hex = "a".repeat(63);
540        hex.push('z');
541        let err = ContentRef::from_hex(hex).unwrap_err();
542        assert!(
543            err.contains("lowercase hex"),
544            "error must mention hex requirement: {err}"
545        );
546    }
547
548    #[test]
549    fn from_digest_bytes_matches_known_blake3_hash() {
550        // BLAKE3("") -> af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262
551        let hash = blake3_hash_of_empty();
552        let cref = ContentRef::from_digest_bytes(&hash);
553        assert_eq!(
554            cref.as_str(),
555            "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262"
556        );
557    }
558
559    // hand-rolled BLAKE3("") vector (see docs/api/blob-store.md)
560    fn blake3_hash_of_empty() -> [u8; 32] {
561        let hex = "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262";
562        let mut out = [0u8; 32];
563        for (i, chunk) in hex.as_bytes().chunks(2).enumerate() {
564            let s = std::str::from_utf8(chunk).unwrap();
565            out[i] = u8::from_str_radix(s, 16).unwrap();
566        }
567        out
568    }
569
570    #[test]
571    fn deserialize_accepts_a_valid_lowercase_digest() {
572        let hex = "d".repeat(64);
573        let json = serde_json::to_string(&hex).unwrap();
574        let cref: ContentRef = serde_json::from_str(&json).unwrap();
575        assert_eq!(cref.as_str(), hex);
576    }
577
578    #[test]
579    fn deserialize_rejects_short_string() {
580        let err = serde_json::from_str::<ContentRef>("\"x\"").unwrap_err();
581        assert!(
582            err.to_string().contains("64"),
583            "deserialize error must mention the expected length: {err}"
584        );
585    }
586
587    #[test]
588    fn deserialize_rejects_uppercase() {
589        let hex = "A".repeat(64);
590        let json = serde_json::to_string(&hex).unwrap();
591        let err = serde_json::from_str::<ContentRef>(&json).unwrap_err();
592        assert!(
593            err.to_string().contains("lowercase"),
594            "deserialize error must mention the lowercase requirement: {err}"
595        );
596    }
597
598    #[test]
599    fn deserialize_rejects_non_hex_characters() {
600        let mut hex = "a".repeat(63);
601        hex.push('z');
602        let json = serde_json::to_string(&hex).unwrap();
603        let err = serde_json::from_str::<ContentRef>(&json).unwrap_err();
604        assert!(
605            err.to_string().contains("lowercase hex"),
606            "deserialize error must mention the hex requirement: {err}"
607        );
608    }
609
610    #[test]
611    fn content_ref_equality_and_hash_are_string_based() {
612        let a = ContentRef::from_hex("b".repeat(64)).unwrap();
613        let b = ContentRef::from_hex("b".repeat(64)).unwrap();
614        let c = ContentRef::from_hex("c".repeat(64)).unwrap();
615        assert_eq!(a, b);
616        assert_ne!(a, c);
617
618        use std::collections::HashSet;
619        let mut set = HashSet::new();
620        set.insert(a.clone());
621        assert!(set.contains(&b));
622        assert!(!set.contains(&c));
623    }
624}