lgwks_std 1.1.0

Everyday Rust primitives with no async runtime required: codecs, timestamps, ids, hashing, regex, JSON/RON/wire, HTTP, and default structured-debugging install. One audited stack per feature — a stack is not always one crate (json is serde + serde_json, ron is serde + ron, http is ureq + iri-string).
//! `hash` owns content-addressable hashing and enforces INV-HASH-DETERMINISTIC:
//! the same input bytes always produce the same digest, and the digest is the
//! BLAKE3 algorithm, the sole content-identity hash in this crate.

/// A 32-byte BLAKE3 digest.
///
/// **Equality is constant-time; ordering is not.** `==`, `!=` and
/// [`Digest::ct_eq`] compare all 32 bytes through `blake3::Hash`'s equality,
/// which is the `constant_time_eq` routine behind an optimisation barrier, so
/// the time taken does not depend on where two digests first differ. The
/// `examples/digest_timing.rs` dudect harness measures that on the release
/// build, beside an early-exit negative control it must catch.
///
/// `Ord`/`PartialOrd` are a lexicographic byte compare that stops at the first
/// differing byte. They exist so a digest can key a `BTreeMap` or be sorted,
/// and are **variable-time**: never order, sort, `binary_search` or `max`
/// digests where one side is secret-derived. Compare such values with `==` or
/// [`Digest::ct_eq`] only.
///
/// Under `wire` the archive is derived too, so a digest can sit inside an
/// archived record and be read back in place. [`rkyv(compare(PartialEq))`]
/// rather than a derived comparison on the archived form, because the archived
/// type is a bare `[u8; 32]` and a byte-loop over it is exactly the timing
/// channel this type exists to close — the comparison is delegated back to the
/// constant-time one below rather than re-derived beside it.
///
/// [`rkyv(compare(PartialEq))`]: https://rkyv.org/derive-attributes.html
#[cfg_attr(
    feature = "wire",
    derive(rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)
)]
#[cfg_attr(feature = "wire", rkyv(compare(PartialEq), derive(Debug)))]
#[derive(Clone, Copy, PartialOrd, Ord)]
pub struct Digest([u8; 32]);

impl std::hash::Hash for Digest {
    fn hash<H: std::hash::Hasher>(&self, state: &mut H) {
        self.0.hash(state);
    }
}

impl PartialEq for Digest {
    /// Delegated to `blake3::Hash`'s equality rather than a byte loop written
    /// here: an XOR/OR fold has no optimisation barrier, so the compiler may
    /// legally turn it into an early-exit compare, while `blake3` routes
    /// through `constant_time_eq`, whose maintainers check the generated code.
    fn eq(&self, other: &Self) -> bool {
        blake3::Hash::from_bytes(self.0) == blake3::Hash::from_bytes(other.0)
    }
}

impl Eq for Digest {}

impl Digest {
    /// Rebuild a digest from its raw 32 bytes, the inverse of
    /// [`Digest::as_bytes`].
    ///
    /// Public because a durable record stores digest bytes — a journal's
    /// chain heads, a receipt's commitment — and re-deriving the value they
    /// committed to means reading them back. Without this, every such reader
    /// would need a digest type of its own beside the estate's.
    #[must_use]
    pub const fn from_bytes(bytes: [u8; 32]) -> Self {
        Self(bytes)
    }

    /// Constant-time equality, by name.
    ///
    /// The same comparison as `==`. It exists so a check against a
    /// secret-derived or adversary-supplied value (a stored chain head, a
    /// receipt's commitment, an artifact key) says at the call site that its
    /// timing matters, and so a later edit cannot quietly swap it for an
    /// ordering or a comparison of the raw bytes.
    #[must_use]
    pub fn ct_eq(&self, other: &Self) -> bool {
        self == other
    }

    /// The raw 32-byte digest.
    #[must_use]
    pub fn as_bytes(&self) -> &[u8; 32] {
        &self.0
    }

    /// Lowercase hex encoding of the digest (64 characters).
    #[must_use]
    pub fn to_hex(&self) -> String {
        crate::hex::encode(self.0)
    }

    /// Parse a 64-character hex string into a digest.
    ///
    /// Upper and lower case are both accepted. Length is validated first, then
    /// the fixed-size hex decoder fills the digest bytes without a temporary
    /// heap allocation.
    pub fn from_hex(text: &str) -> Result<Self, DigestParseError> {
        if text.len() != 64 {
            let refusal = Err(DigestParseError::WrongLength { len: text.len() });
            #[cfg(feature = "trace")]
            crate::trace::debug!(error = ?refusal.as_ref().err(), "from_hex: returning an error to the caller");
            return refusal;
        }
        let mut raw = [0; 32];
        crate::hex::decode_into(text, &mut raw).map_err(DigestParseError::Hex)?;
        Ok(Self(raw))
    }
}

/// Error from parsing a hex string into a [`Digest`].
///
/// Variants are stable and machine-readable; `#[non_exhaustive]` lets a later
/// revision add a rejection reason without breaking callers that match on the
/// current two.
#[derive(Debug, Clone)]
#[non_exhaustive]
pub enum DigestParseError {
    /// Input was not exactly 64 hex characters (32 bytes).
    WrongLength {
        /// Actual length of the input.
        len: usize,
    },
    /// Input contained invalid hex.
    Hex(crate::hex::DecodeError),
}

impl core::fmt::Display for DigestParseError {
    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
        match *self {
            Self::WrongLength { len } => {
                write!(f, "digest hex must be 64 characters, got {len}")
            }
            Self::Hex(ref err) => write!(f, "{err}"),
        }
    }
}

impl std::error::Error for DigestParseError {}

impl core::fmt::Debug for Digest {
    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
        f.write_str("Digest(")?;
        crate::hex::write_lowercase(&self.0, f)?;
        f.write_str(")")
    }
}

impl core::fmt::Display for Digest {
    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
        crate::hex::write_lowercase(&self.0, f)
    }
}

/// Hash `data` with BLAKE3 and return the 32-byte digest.
#[must_use]
pub fn blake3(data: &[u8]) -> Digest {
    Digest(*blake3::hash(data).as_bytes())
}

/// Hash `data` with keyed BLAKE3 under `key` and return the 32-byte digest.
///
/// This is the message-authentication primitive: whoever holds `key`
/// can recompute the tag, whoever does not cannot forge one. Use it where a
/// checksum is not enough because the writer is adversarial (audit chains,
/// sealed receipts). The key must come from outside the sealed artifact
/// (environment, keyring); a key stored beside the tags proves nothing.
#[must_use]
pub fn keyed(key: &[u8; 32], data: &[u8]) -> Digest {
    Digest(*blake3::keyed_hash(key, data).as_bytes())
}

/// Incremental hasher for streaming data.
///
/// Manual `Debug` rather than a derive: the wrapped engine hasher's own
/// formatting is an implementation detail of the `blake3` edge, and a consumer
/// asking to print this type wants to know where the stream stands, not which
/// crate is underneath. `bytes_written` is the one field that answers that.
pub struct Hasher(blake3::Hasher);

impl core::fmt::Debug for Hasher {
    fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
        f.debug_struct("Hasher")
            .field("bytes_written", &self.0.count())
            .finish_non_exhaustive()
    }
}

impl Hasher {
    /// Create a new incremental hasher.
    #[must_use]
    pub fn new() -> Self {
        Self(blake3::Hasher::new())
    }

    /// Feed bytes into the hasher.
    ///
    /// Returns `&mut Self` so calls chain; the digest is unchanged by how the
    /// input was split across calls, which the test below pins.
    pub fn update(&mut self, data: &[u8]) -> &mut Self {
        self.0.update(data);
        self
    }

    /// Feed one variable-length byte string into the hasher under a length
    /// prefix.
    ///
    /// A stream built from concatenated parts is ambiguous at the boundaries:
    /// `("ab", "c")` and `("a", "bc")` feed the same bytes. The `u64`
    /// little-endian length prefix makes each part self-delimiting, so the
    /// digest of a framed stream depends on the parts, not only on their
    /// concatenation. Anything hashed from more than one variable-length
    /// field — an identity, a canonical record, a digest over a structure —
    /// should feed its parts through this rather than through [`Self::update`].
    ///
    /// Returns `&mut Self` so calls chain, for the same reason `update` does.
    pub fn write_framed(&mut self, data: &[u8]) -> &mut Self {
        let len = u64::try_from(data.len()).unwrap_or(u64::MAX);
        self.0.update(&len.to_le_bytes());
        self.0.update(data);
        self
    }

    /// Finalize and return the digest.
    ///
    /// Borrows rather than consumes, so a caller can keep feeding the same
    /// hasher; the digest is a snapshot of the bytes written so far.
    #[must_use]
    pub fn finalize(&self) -> Digest {
        Digest(*self.0.finalize().as_bytes())
    }
}

impl Default for Hasher {
    fn default() -> Self {
        Self::new()
    }
}

// ── Tests ───────────────────────────────────────────────────────────────────

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn empty_input_matches_blake3_spec() {
        let digest = blake3(b"");
        assert_eq!(
            digest.to_hex(),
            "af1349b9f5f9a1a6a0404dea36dcc9499bcb25c9adc112b7cc9a93cae41f3262"
        );
    }

    #[test]
    fn deterministic_across_calls() {
        let first = blake3(b"hello world");
        let second = blake3(b"hello world");
        assert_eq!(first, second);
    }

    #[test]
    fn different_input_different_digest() {
        assert_ne!(blake3(b"a"), blake3(b"b"));
    }

    #[test]
    fn incremental_matches_oneshot() {
        let oneshot = blake3(b"hello world");
        let mut hasher = Hasher::new();
        hasher.update(b"hello ");
        hasher.update(b"world");
        assert_eq!(hasher.finalize(), oneshot);
    }

    #[test]
    fn framing_separates_splits_concatenation_conflates() {
        // The ambiguity framing exists to remove: unframed, both splits feed
        // the same bytes. Framed, each split is its own digest.
        let mut unframed_left = Hasher::new();
        unframed_left.update(b"ab").update(b"c");
        let mut unframed_right = Hasher::new();
        unframed_right.update(b"a").update(b"bc");
        assert_eq!(unframed_left.finalize(), unframed_right.finalize());

        let mut framed_left = Hasher::new();
        framed_left.write_framed(b"ab").write_framed(b"c");
        let mut framed_right = Hasher::new();
        framed_right.write_framed(b"a").write_framed(b"bc");
        assert_ne!(framed_left.finalize(), framed_right.finalize());

        // And framed streams remain deterministic.
        let mut repeat = Hasher::new();
        repeat.write_framed(b"ab").write_framed(b"c");
        assert_eq!(framed_left.finalize(), repeat.finalize());
    }

    /// Equality sees every byte: a digest differing from another at any one
    /// of the 32 positions, by any single bit, is unequal to it, under `==`,
    /// `!=` and `ct_eq` alike, and equal to an unchanged copy of itself.
    #[test]
    fn equality_sees_each_of_the_32_bytes() {
        let base = blake3(b"position");
        assert!(base.ct_eq(&Digest::from_bytes(*base.as_bytes())));
        for position in 0..32 {
            for mask in [0x01_u8, 0x02, 0x04, 0x08, 0x10, 0x20, 0x40, 0x80] {
                let mut bytes = *base.as_bytes();
                if let Some(byte) = bytes.get_mut(position) {
                    *byte ^= mask;
                }
                let flipped = Digest::from_bytes(bytes);
                assert!(
                    base != flipped && !base.ct_eq(&flipped) && !flipped.ct_eq(&base),
                    "a flip of mask {mask:#04x} at byte {position} compared equal"
                );
            }
        }
    }

    #[test]
    fn hex_roundtrip() -> Result<(), DigestParseError> {
        let digest = blake3(b"test");
        let hex = digest.to_hex();
        let parsed = Digest::from_hex(&hex)?;
        assert_eq!(digest, parsed);
        Ok(())
    }

    #[test]
    fn display_is_hex() {
        let digest = blake3(b"");
        assert_eq!(format!("{digest}"), digest.to_hex());
    }

    #[test]
    fn from_hex_rejects_wrong_length() {
        assert!(Digest::from_hex("abcd").is_err());
    }

    #[test]
    fn keyed_differs_from_unkeyed() {
        let key = [0x42u8; 32];
        assert_ne!(keyed(&key, b"hello world"), blake3(b"hello world"));
    }

    #[test]
    fn keyed_is_key_sensitive() {
        assert_ne!(keyed(&[0x01u8; 32], b"data"), keyed(&[0x02u8; 32], b"data"));
    }

    #[test]
    fn keyed_deterministic_across_calls() {
        let key = [0x07u8; 32];
        assert_eq!(keyed(&key, b"receipt"), keyed(&key, b"receipt"));
    }

    #[test]
    fn keyed_matches_blake3_keyed_hash() {
        let key = [0xABu8; 32];
        let expected = *blake3::keyed_hash(&key, b"vector").as_bytes();
        assert_eq!(keyed(&key, b"vector").as_bytes(), &expected);
    }

    /// A digest archives, and the archive is read back in place.
    ///
    /// Two things are checked, and the second is the one the derive attribute
    /// is there for: `access` hands back a value borrowed from the buffer with
    /// no allocation, and comparing two archived digests still resolves through
    /// [`Digest::eq`] rather than a derived byte loop, so the archive does not
    /// quietly reintroduce the timing channel the type exists to close.
    #[cfg(feature = "wire")]
    #[test]
    fn archives_and_reads_back_in_place() -> Result<(), crate::wire::WireError> {
        let digest = blake3(b"receipt");
        let bytes = crate::wire::to_bytes::<crate::wire::WireError>(&digest)?;
        // Each read is compared against the original rather than against the
        // other read: `compare(PartialEq)` generates the cross-type comparison,
        // and that is the one that resolves through the constant-time `eq`.
        // An archived-to-archived comparison is a different impl entirely.
        assert_eq!(
            &digest,
            crate::wire::access::<ArchivedDigest, crate::wire::WireError>(&bytes)?,
            "the archived form compares equal to the digest it was made from"
        );
        let other = crate::wire::to_bytes::<crate::wire::WireError>(&blake3(b"other"))?;
        assert_ne!(
            &digest,
            crate::wire::access::<ArchivedDigest, crate::wire::WireError>(&other)?,
            "and a different digest does not"
        );
        Ok(())
    }
}