khive_storage/blob.rs
1//! Blob storage capability — content-addressed binary object CRUD.
2//!
3//! `BlobStore` is the trait family added by khive#292: bytes that do not
4//! belong inside the primary SQLite database (source PDFs, images, large
5//! opaque payloads) are stored by a dedicated backend and referenced from
6//! the graph by an opaque [`ContentRef`]. Per ADR-005's "zero
7//! implementations" constraint, this module defines the contract only — the
8//! first backend (filesystem, BLAKE3-addressed) lives in `khive-db`.
9
10use std::collections::HashSet;
11
12use async_trait::async_trait;
13use serde::{Deserialize, Serialize};
14
15use crate::capability::StorageCapability;
16use crate::error::StorageError;
17use crate::sql::SqlAccess;
18use crate::types::StorageResult;
19
20/// Number of hex characters in a BLAKE3-256 digest (32 bytes -> 64 hex chars).
21const CONTENT_REF_HEX_LEN: usize = 64;
22
23/// Portable v1 ceiling for a whole-buffer blob operation (64 MiB).
24///
25/// Callers needing larger objects require a future streaming contract. A
26/// [`BlobStore::get_bounded_verified`] request above this limit is invalid
27/// even when the selected backend could otherwise satisfy it.
28pub const MAX_BLOB_WHOLE_BYTES: u64 = 64 * 1024 * 1024;
29
30/// An opaque, content-addressed reference to a stored blob.
31///
32/// Backed by a lowercase-hex BLAKE3 digest of the blob's bytes: identical
33/// content always produces the same `ContentRef`, so storing the same bytes
34/// twice is a no-op after the first write. Callers must treat the value as
35/// opaque — the backend, not the caller, decides how a `ContentRef` maps to
36/// physical storage.
37///
38/// `Deserialize` is hand-written (below) to reject any string that is not 64
39/// lowercase hex characters — a naive derive would let an unvalidated value
40/// panic later in `shard_path`'s slicing.
41/// See `crates/khive-storage/docs/api/blob-store.md` for the full rationale.
42#[derive(Clone, Debug, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize)]
43#[serde(transparent)]
44pub struct ContentRef(String);
45
46/// Opaque capability for a backend's staged upload, encoded as 128-bit hex.
47///
48/// This is not a content reference. The caller owns hashing and upload-session
49/// state; retaining this identifier does not make a session restartable.
50#[derive(Clone, Debug, PartialEq, Eq, Hash, Serialize)]
51#[serde(transparent)]
52pub struct UploadId(String);
53
54impl UploadId {
55 /// Construct an identifier from freshly generated random bytes.
56 pub fn from_bytes(bytes: &[u8; 16]) -> Self {
57 Self(hex_encode(bytes))
58 }
59
60 /// Parse exactly 32 lowercase hex characters, never a backend pathname.
61 pub fn from_hex(value: impl Into<String>) -> Result<Self, String> {
62 let value = value.into();
63 if value.len() != 32 || !khive_types::is_lowercase_hex(&value) {
64 return Err("upload_id must be 32 lowercase hex characters".into());
65 }
66 Ok(Self(value))
67 }
68
69 /// Canonical wire spelling.
70 pub fn as_str(&self) -> &str {
71 &self.0
72 }
73}
74
75impl std::fmt::Display for UploadId {
76 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
77 f.write_str(self.as_str())
78 }
79}
80
81impl<'de> Deserialize<'de> for UploadId {
82 fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
83 Self::from_hex(String::deserialize(deserializer)?).map_err(serde::de::Error::custom)
84 }
85}
86
87impl<'de> Deserialize<'de> for ContentRef {
88 fn deserialize<D>(deserializer: D) -> Result<Self, D::Error>
89 where
90 D: serde::Deserializer<'de>,
91 {
92 let raw = String::deserialize(deserializer)?;
93 ContentRef::from_hex(raw).map_err(serde::de::Error::custom)
94 }
95}
96
97impl ContentRef {
98 /// Parse a `ContentRef` from a caller-supplied hex string.
99 ///
100 /// Rejects anything that is not exactly 64 lowercase hex characters.
101 /// Uppercase is rejected (not normalized) to keep one canonical string
102 /// form per digest — see `docs/api/blob-store.md`.
103 pub fn from_hex(hex: impl Into<String>) -> Result<Self, String> {
104 let hex = hex.into();
105 if hex.len() != CONTENT_REF_HEX_LEN {
106 return Err(format!(
107 "content_ref must be {CONTENT_REF_HEX_LEN} hex characters, got length {} ({hex:?})",
108 hex.len()
109 ));
110 }
111 if !khive_types::is_lowercase_hex(&hex) {
112 return Err(format!(
113 "content_ref must be lowercase hex (0-9, a-f), got {hex:?}"
114 ));
115 }
116 Ok(Self(hex))
117 }
118
119 /// Construct a `ContentRef` directly from a BLAKE3 digest's raw bytes.
120 pub fn from_digest_bytes(digest: &[u8; 32]) -> Self {
121 Self(hex_encode(digest))
122 }
123
124 /// Borrow the underlying hex string.
125 pub fn as_str(&self) -> &str {
126 &self.0
127 }
128}
129
130impl std::fmt::Display for ContentRef {
131 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
132 f.write_str(&self.0)
133 }
134}
135
136impl AsRef<str> for ContentRef {
137 fn as_ref(&self) -> &str {
138 &self.0
139 }
140}
141
142fn hex_encode(bytes: &[u8]) -> String {
143 const HEX: &[u8; 16] = b"0123456789abcdef";
144 let mut out = String::with_capacity(bytes.len() * 2);
145 for &b in bytes {
146 out.push(HEX[(b >> 4) as usize] as char);
147 out.push(HEX[(b & 0x0f) as usize] as char);
148 }
149 out
150}
151
152/// Configuration for [`BlobStore::orphan_sweep`].
153///
154/// `live_refs` is a point-in-time snapshot the caller assembles (this trait
155/// has no visibility into SQL substrates — ADR-005 constraint 4), not a live
156/// query. See [`BlobStore::orphan_sweep`] for the concurrency hazard this
157/// implies, and `crates/khive-storage/docs/api/blob-store.md` for the full
158/// rationale.
159#[derive(Clone, Debug, Default, Serialize, Deserialize)]
160pub struct BlobOrphanSweepConfig {
161 /// Content refs currently referenced by at least one committed record
162 /// attachment, as of when the caller assembled this set. Anything
163 /// this backend stores that is NOT in this set is treated as orphaned
164 /// and deleted (or reported, in `dry_run` mode) — including a
165 /// `content_ref` that becomes live after this snapshot was taken.
166 pub live_refs: HashSet<ContentRef>,
167 /// When `true`, report what would be deleted without deleting anything.
168 pub dry_run: bool,
169}
170
171/// Result of a [`BlobStore::orphan_sweep`] call.
172#[derive(Clone, Debug, Default, Serialize, Deserialize)]
173pub struct BlobOrphanSweepResult {
174 /// Total objects examined in this backend.
175 pub scanned: u64,
176 /// Objects actually deleted (always 0 when `dry_run = true`).
177 pub deleted: u64,
178 /// Objects that are orphaned (would be deleted whether or not `dry_run`
179 /// is set — populated in both modes so a dry run reports the same count
180 /// a real run would delete).
181 pub would_delete: u64,
182 /// Objects with zero live references that were left alone because they
183 /// are still inside their publish grace period — recently written and
184 /// not yet orphaned, just not yet referenced by a record attachment.
185 /// Reported in both modes; never counted in `would_delete` or `deleted`.
186 pub grace_period_skipped: u64,
187}
188
189/// Immutable owner policy conveyed by the host, never by upload wire arguments.
190#[derive(Clone, Copy, Debug)]
191pub struct UploadLeaseConfig {
192 owner: uuid::Uuid,
193 idle_secs: u64,
194}
195
196impl UploadLeaseConfig {
197 /// Construct a positive whole-second bound; the backend enforces its cap.
198 pub fn new(owner: uuid::Uuid, idle: std::time::Duration) -> StorageResult<Self> {
199 if owner.is_nil() {
200 return Err(StorageError::InvalidInput {
201 capability: StorageCapability::Blob,
202 operation: "upload_lease_config".into(),
203 message: "upload lease owner must be a non-nil durable UUID".into(),
204 });
205 }
206 if idle.is_zero() || idle.subsec_nanos() != 0 {
207 return Err(StorageError::InvalidInput {
208 capability: StorageCapability::Blob,
209 operation: "upload_lease_config".into(),
210 message: "upload lease bound must be positive whole seconds".into(),
211 });
212 }
213 Ok(Self {
214 owner,
215 idle_secs: idle.as_secs(),
216 })
217 }
218
219 /// Validated durable MAIN/store identity supplied by the host.
220 pub fn owner(&self) -> uuid::Uuid {
221 self.owner
222 }
223 /// The owning daemon's fixed bound, in seconds.
224 pub fn idle_secs(&self) -> u64 {
225 self.idle_secs
226 }
227}
228
229/// Content-addressed binary object CRUD.
230///
231/// Every method is backend-agnostic: the filesystem backend
232/// (`khive-db::stores::blob::FsBlobStore`) is the first implementation, and
233/// any future backend (object storage, a different CAS layout) implements
234/// the same operations. Per ADR-005 constraint 4, a `BlobStore` instance
235/// talks to exactly one backend.
236// `Debug` is a supertrait so boot-path tests can distinguish which concrete
237// backend was installed behind `Arc<dyn BlobStore>` via `format!("{:?}", ..)`
238// without adding a downcast/type-name method to the production surface.
239#[async_trait]
240pub trait BlobStore: Send + Sync + std::fmt::Debug + 'static {
241 /// Store `bytes`, returning the content-addressed reference under which
242 /// they are now retrievable. Storing byte-identical content more than
243 /// once returns the same `ContentRef` and does not re-write the object.
244 async fn put(&self, bytes: Vec<u8>) -> StorageResult<ContentRef>;
245
246 /// Restart the publish grace period for an existing object and return its
247 /// size, or `None` if absent. Backends with a live orphan sweep must make
248 /// the existence check and refresh atomic with that sweep's root lock.
249 /// The default is sufficient for backends without a live orphan sweep.
250 async fn refresh_publish_grace(&self, content_ref: &ContentRef) -> StorageResult<Option<u64>> {
251 self.size(content_ref).await
252 }
253
254 /// Lease policy capability, not an authorization or ownership decision.
255 /// None preserves non-filesystem policy and explicit Unsupported defaults.
256 fn upload_lease_idle_cap(&self) -> Option<std::time::Duration> {
257 None
258 }
259
260 /// Create and sync staging plus an atomic complete immutable-owner lease.
261 /// Hosts convey durable identity and the resolved bound before admission.
262 async fn begin_upload_with_lease(
263 &self,
264 declared_size: u64,
265 config: UploadLeaseConfig,
266 ) -> StorageResult<UploadId> {
267 let _ = (declared_size, config);
268 Err(unsupported_upload("begin_upload_with_lease"))
269 }
270
271 /// Renew a validated lease without appending bytes, including tail resends.
272 /// Success follows file sync, replacing rename and supported dir barriers.
273 async fn renew_upload(&self, id: &UploadId) -> StorageResult<()> {
274 let _ = id;
275 Err(unsupported_upload("renew_upload"))
276 }
277
278 /// Create an empty staging object. The pack keeps the declared size,
279 /// incremental hash, sequence and idle clock. Backends enforce their
280 /// capacity policy on each append. Unsupported backends refuse explicitly.
281 async fn begin_upload(&self, declared_size: u64) -> StorageResult<UploadId> {
282 let _ = declared_size;
283 Err(unsupported_upload("begin_upload"))
284 }
285
286 /// Append and synchronize bytes, returning the total staged length.
287 /// Leased backends synchronize bytes then renew under one root ownership.
288 /// The caller serializes parts and aborts after any uncertain append/renewal.
289 async fn append_part(&self, id: &UploadId, bytes: Vec<u8>) -> StorageResult<u64> {
290 let _ = (id, bytes);
291 Err(unsupported_upload("append_part"))
292 }
293
294 /// Publish through the same routine as put, without hashing a second time.
295 /// The caller proves the supplied digest and declared size before invoking
296 /// this method. Success consumes the staging object, including on dedup.
297 async fn commit_upload(&self, id: &UploadId, content_ref: &ContentRef) -> StorageResult<()> {
298 let _ = (id, content_ref);
299 Err(unsupported_upload("commit_upload"))
300 }
301
302 /// Discard staging; an already absent staging object is a successful no-op.
303 async fn abort_upload(&self, id: &UploadId) -> StorageResult<()> {
304 let _ = id;
305 Err(unsupported_upload("abort_upload"))
306 }
307
308 /// Remove staging objects the backend can show are abandoned. The filesystem
309 /// backend ignores `idle_for`: it removes a leased upload once it has seen that
310 /// lease unchanged for the lease's own bound plus a fixed margin, or once the
311 /// lease's last renewal plus its bound plus 24 hours has passed, and an upload
312 /// with no lease once its file is 24 hours old. Backends without staged uploads
313 /// refuse. This never visits committed objects. Open S3 multipart uploads
314 /// require the deployment's incomplete-multipart lifecycle rule instead.
315 async fn sweep_uploads(&self, idle_for: std::time::Duration) -> StorageResult<u64> {
316 let _ = idle_for;
317 Err(unsupported_upload("sweep_uploads"))
318 }
319
320 /// Fetch at most `max_bytes` from `content_ref` and verify its BLAKE3
321 /// digest before returning any bytes.
322 ///
323 /// `max_bytes` may be zero (only an empty object can then succeed) and
324 /// must not exceed [`MAX_BLOB_WHOLE_BYTES`]. Implementations must enforce
325 /// the limit while reading the authoritative object, not by composing a
326 /// metadata-only [`Self::size`] check with another read. A successful
327 /// result is complete, no larger than the declared maximum,
328 /// metadata-size-consistent, and digest-matched to `content_ref`.
329 async fn get_bounded_verified(
330 &self,
331 content_ref: &ContentRef,
332 max_bytes: u64,
333 ) -> StorageResult<Vec<u8>>;
334
335 /// Whether an object currently exists for `content_ref`.
336 async fn exists(&self, content_ref: &ContentRef) -> StorageResult<bool>;
337
338 /// The size in bytes of the object stored under `content_ref`, without
339 /// hydrating its bytes.
340 ///
341 /// Returns `Ok(None)` when no object exists for this reference — this is
342 /// the existence check and the size read in one call, so a caller never
343 /// pays for a full read just to answer "does this exist and how big is
344 /// it". On the filesystem backend this maps to a file metadata stat; on
345 /// an object-storage backend it maps to a `HEAD Object` request.
346 async fn size(&self, content_ref: &ContentRef) -> StorageResult<Option<u64>>;
347
348 /// Remove the object stored under `content_ref`.
349 ///
350 /// Returns `true` when an object was actually removed, `false` when
351 /// none existed — deleting an absent object is not an error.
352 ///
353 /// # Safety / concurrency hazard (ADR-111 §8, amended)
354 ///
355 /// Unconditional physical removal with **no coordination against any
356 /// record or attachment that might reference `content_ref`**. Safe to call only when
357 /// the caller has independently quiesced every writer that could commit a
358 /// new SQL liveness reference for the duration of the call — this
359 /// trait does not detect or prevent a race. Offline-maintenance-only.
360 /// See `crates/khive-storage/docs/api/blob-store.md`.
361 async fn delete(&self, content_ref: &ContentRef) -> StorageResult<bool>;
362
363 /// Enumerate every object this backend holds and delete (or, in
364 /// `dry_run` mode, report) those absent from `config.live_refs`.
365 /// Operator-side GC path (khive#292 deliverable 5) — admin-only, not an
366 /// MCP verb. Default returns `StorageError::Unsupported`; the filesystem
367 /// backend currently returns the same typed refusal for every call (see
368 /// below) rather than performing a real directory walk.
369 ///
370 /// # Safety / concurrency hazard (ADR-111 §8, amended)
371 ///
372 /// `config.live_refs` is a **snapshot**; a `content_ref` that becomes
373 /// newly live between the snapshot and the sweep is deleted anyway.
374 /// **Callers MUST quiesce attachment writes** for the duration of
375 /// snapshot-plus-sweep. See `crates/khive-storage/docs/api/blob-store.md`
376 /// for the hazard. This API also has no [`SqlAccess`] capability with
377 /// which to prove a completed V21 attachment epoch, so — unlike
378 /// [`Self::transactional_orphan_sweep`] — it cannot honor that gate. The
379 /// filesystem backend therefore disables this method entirely in this
380 /// compatibility release, in both `dry_run` modes: concurrent AND
381 /// offline callers alike must use [`Self::transactional_orphan_sweep`]
382 /// instead.
383 async fn orphan_sweep(
384 &self,
385 config: &BlobOrphanSweepConfig,
386 ) -> StorageResult<BlobOrphanSweepResult> {
387 let _ = config;
388 Err(StorageError::Unsupported {
389 capability: StorageCapability::Blob,
390 operation: "orphan_sweep".into(),
391 message: "this backend does not support orphan sweep".into(),
392 })
393 }
394
395 /// Select live attachment references and sweep orphaned blobs behind a
396 /// database-coordinated, bounded claim protocol.
397 ///
398 /// Unlike [`Self::orphan_sweep`], this operation obtains liveness itself
399 /// from `sql`; callers do not assemble a stale snapshot. `sql` must be the
400 /// canonical main database capability used for the attachment writes that own
401 /// references. Implementations must also ensure an object published after
402 /// the sweep's candidate set is captured cannot be mistaken for an orphan,
403 /// including when it is published between selecting live references and
404 /// physical deletion. Implementations must not perform filesystem or
405 /// other external I/O while holding the database writer transaction;
406 /// durable claims/triggers or an equivalently fail-closed fence must keep
407 /// attachment writes safe after each short transaction commits. Claim/result
408 /// materialization and cleanup must have an explicit per-transaction
409 /// cardinality bound rather than scale one writer hold with the complete
410 /// object population. A file-backed `sql` implementation must expose its
411 /// canonical [`SqlAccess::database_path`] so crash recovery can retain
412 /// cross-process database ownership independently of mutable blob-root
413 /// spelling or relocation.
414 /// Coordination may be advisory, so callers must publish through the
415 /// backend rather than mutate its physical storage directly.
416 /// Backends that cannot provide both guarantees return
417 /// `StorageError::Unsupported`.
418 ///
419 /// The filesystem implementation is schema-epoch gated and supports both
420 /// report-only and destructive modes only when `sql` proves the exact
421 /// completed V21 attachment cutover: durable complete marker and ledger
422 /// row, attachment table/indexes and INSERT/UPDATE claim fences, and
423 /// absence of every legacy entity reference column/index/fence. V20,
424 /// pending, incomplete, missing-required-object, retained-legacy, and
425 /// ahead-of-V21 epochs return typed `Unsupported` before root locking,
426 /// filesystem walking, or abandoned-claim cleanup. Malformed stored
427 /// evidence or a nonfunctional named fence fails closed with its validation,
428 /// storage, or typed `Unsupported` error before claim cleanup or deletion.
429 /// Once admitted, every attachment role is live; soft deletion alone does
430 /// not make its blob collectible.
431 ///
432 /// This is the Phase-4a GC compatibility gate. Phase 4a changes no schema or
433 /// data. Every older process sharing the database/blob root must be drained
434 /// before Phase 4b performs the attachment backfill and legacy-column drop.
435 /// Phase-4a application readers/writers must also be quiesced during cutover;
436 /// only a GC-only worker has narrow compatibility with exact completed V21.
437 /// Callers must not fall back to [`Self::orphan_sweep`] or [`Self::delete`]
438 /// when this gate refuses.
439 ///
440 /// Publishing a blob and committing the attachment write that references it
441 /// are two separate client steps; nothing serializes them against this
442 /// sweep. Implementations must therefore also give a just-published,
443 /// not-yet-referenced object a bounded grace period before treating it as
444 /// an orphan (the filesystem backend does this via file age). A client
445 /// whose own gap between the two steps exceeds that grace period is not
446 /// protected — this narrows, but does not eliminate, the hazard.
447 async fn transactional_orphan_sweep(
448 &self,
449 sql: &dyn SqlAccess,
450 dry_run: bool,
451 ) -> StorageResult<BlobOrphanSweepResult> {
452 let _ = (sql, dry_run);
453 Err(StorageError::Unsupported {
454 capability: StorageCapability::Blob,
455 operation: "transactional_orphan_sweep".into(),
456 message: "this backend does not support a database-coordinated orphan sweep".into(),
457 })
458 }
459}
460
461fn unsupported_upload(operation: &'static str) -> StorageError {
462 StorageError::Unsupported {
463 capability: StorageCapability::Blob,
464 operation: operation.into(),
465 message: "this backend does not support staged uploads".into(),
466 }
467}
468
469#[cfg(test)]
470mod tests {
471 use super::*;
472
473 #[test]
474 fn upload_id_roundtrips_and_rejects_path_or_noncanonical_input() {
475 let id = UploadId::from_bytes(&[0xab; 16]);
476 assert_eq!(id.as_str(), "ab".repeat(16));
477 assert_eq!(
478 serde_json::from_str::<UploadId>(&serde_json::to_string(&id).unwrap()).unwrap(),
479 id
480 );
481 for invalid in [
482 String::new(),
483 "a".repeat(31),
484 "a".repeat(33),
485 "A".repeat(32),
486 "g".repeat(32),
487 "../outside".into(),
488 "a/b".repeat(11),
489 ] {
490 assert!(UploadId::from_hex(&invalid).is_err());
491 assert!(serde_json::from_value::<UploadId>(serde_json::json!(invalid)).is_err());
492 }
493 }
494
495 #[test]
496 fn from_hex_accepts_valid_lowercase_digest() {
497 let hex = "a".repeat(64);
498 let cref = ContentRef::from_hex(hex.clone()).unwrap();
499 assert_eq!(cref.as_str(), hex);
500 assert_eq!(cref.to_string(), hex);
501 }
502
503 #[test]
504 fn from_hex_rejects_short_string() {
505 let err = ContentRef::from_hex("abc").unwrap_err();
506 assert!(
507 err.contains("64"),
508 "error must mention expected length: {err}"
509 );
510 }
511
512 #[test]
513 fn from_hex_rejects_long_string() {
514 let err = ContentRef::from_hex("a".repeat(65)).unwrap_err();
515 assert!(
516 err.contains("64"),
517 "error must mention expected length: {err}"
518 );
519 }
520
521 #[test]
522 fn from_hex_rejects_uppercase() {
523 let err = ContentRef::from_hex("A".repeat(64)).unwrap_err();
524 assert!(
525 err.contains("lowercase"),
526 "error must mention lowercase requirement: {err}"
527 );
528 }
529
530 #[test]
531 fn from_hex_rejects_non_hex_characters() {
532 let mut hex = "a".repeat(63);
533 hex.push('z');
534 let err = ContentRef::from_hex(hex).unwrap_err();
535 assert!(
536 err.contains("lowercase hex"),
537 "error must mention hex requirement: {err}"
538 );
539 }
540
541 #[test]
542 fn from_digest_bytes_matches_known_blake3_hash() {
543 // BLAKE3("") -> af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262
544 let hash = blake3_hash_of_empty();
545 let cref = ContentRef::from_digest_bytes(&hash);
546 assert_eq!(
547 cref.as_str(),
548 "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262"
549 );
550 }
551
552 // hand-rolled BLAKE3("") vector (see docs/api/blob-store.md)
553 fn blake3_hash_of_empty() -> [u8; 32] {
554 let hex = "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262";
555 let mut out = [0u8; 32];
556 for (i, chunk) in hex.as_bytes().chunks(2).enumerate() {
557 let s = std::str::from_utf8(chunk).unwrap();
558 out[i] = u8::from_str_radix(s, 16).unwrap();
559 }
560 out
561 }
562
563 #[test]
564 fn deserialize_accepts_a_valid_lowercase_digest() {
565 let hex = "d".repeat(64);
566 let json = serde_json::to_string(&hex).unwrap();
567 let cref: ContentRef = serde_json::from_str(&json).unwrap();
568 assert_eq!(cref.as_str(), hex);
569 }
570
571 #[test]
572 fn deserialize_rejects_short_string() {
573 let err = serde_json::from_str::<ContentRef>("\"x\"").unwrap_err();
574 assert!(
575 err.to_string().contains("64"),
576 "deserialize error must mention the expected length: {err}"
577 );
578 }
579
580 #[test]
581 fn deserialize_rejects_uppercase() {
582 let hex = "A".repeat(64);
583 let json = serde_json::to_string(&hex).unwrap();
584 let err = serde_json::from_str::<ContentRef>(&json).unwrap_err();
585 assert!(
586 err.to_string().contains("lowercase"),
587 "deserialize error must mention the lowercase requirement: {err}"
588 );
589 }
590
591 #[test]
592 fn deserialize_rejects_non_hex_characters() {
593 let mut hex = "a".repeat(63);
594 hex.push('z');
595 let json = serde_json::to_string(&hex).unwrap();
596 let err = serde_json::from_str::<ContentRef>(&json).unwrap_err();
597 assert!(
598 err.to_string().contains("lowercase hex"),
599 "deserialize error must mention the hex requirement: {err}"
600 );
601 }
602
603 #[test]
604 fn content_ref_equality_and_hash_are_string_based() {
605 let a = ContentRef::from_hex("b".repeat(64)).unwrap();
606 let b = ContentRef::from_hex("b".repeat(64)).unwrap();
607 let c = ContentRef::from_hex("c".repeat(64)).unwrap();
608 assert_eq!(a, b);
609 assert_ne!(a, c);
610
611 use std::collections::HashSet;
612 let mut set = HashSet::new();
613 set.insert(a.clone());
614 assert!(set.contains(&b));
615 assert!(!set.contains(&c));
616 }
617}