Skip to main content

ic_hash/
sha2.rs

1//! FIPS 180-4 SHA-2 family.
2//!
3//! Two cores (32-bit and 64-bit) are shared by six published output variants,
4//! which differ only in their initial hash value and truncation length.
5
6//! Indexed loops over fixed-size limb and word arrays are used throughout; they
7//! mirror the index algebra in the specifications these routines implement, so
8//! `needless_range_loop` is allowed rather than obscuring the correspondence.
9#![allow(clippy::needless_range_loop)]
10
11use ic_core::traits::{Algorithm, Digest, SelfTest};
12use ic_core::{ensure, Result, Zeroize};
13
14// SHA-NI, where the CPU has it. Only under `std`, because the detection does:
15// a `no_std` build has no way to ask, and guessing wrong is an illegal
16// instruction rather than a wrong answer.
17#[cfg(all(target_arch = "x86_64", feature = "std"))]
18mod x86;
19
20/// Whether this CPU has the instructions [`x86::compress`] needs.
21///
22/// Asked once. `is_x86_feature_detected!` is not free, and SHA-256 is called
23/// often enough on small inputs that paying for the query per block would show
24/// up in exactly the workloads this is meant to help.
25#[cfg(all(target_arch = "x86_64", feature = "std"))]
26fn sha_ni() -> bool {
27    use core::sync::atomic::{AtomicU8, Ordering};
28    // 0 not yet asked, 1 yes, 2 no.
29    static CACHED: AtomicU8 = AtomicU8::new(0);
30    match CACHED.load(Ordering::Relaxed) {
31        1 => true,
32        2 => false,
33        _ => {
34            let have = std::is_x86_feature_detected!("sha")
35                && std::is_x86_feature_detected!("sse2")
36                && std::is_x86_feature_detected!("ssse3")
37                && std::is_x86_feature_detected!("sse4.1");
38            CACHED.store(u8::from(!have) + 1, Ordering::Relaxed);
39            have
40        }
41    }
42}
43
44const K256: [u32; 64] = [
45    0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5,
46    0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174,
47    0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
48    0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967,
49    0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85,
50    0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
51    0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
52    0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
53];
54
55const K512: [u64; 80] = [
56    0x428a2f98d728ae22,
57    0x7137449123ef65cd,
58    0xb5c0fbcfec4d3b2f,
59    0xe9b5dba58189dbbc,
60    0x3956c25bf348b538,
61    0x59f111f1b605d019,
62    0x923f82a4af194f9b,
63    0xab1c5ed5da6d8118,
64    0xd807aa98a3030242,
65    0x12835b0145706fbe,
66    0x243185be4ee4b28c,
67    0x550c7dc3d5ffb4e2,
68    0x72be5d74f27b896f,
69    0x80deb1fe3b1696b1,
70    0x9bdc06a725c71235,
71    0xc19bf174cf692694,
72    0xe49b69c19ef14ad2,
73    0xefbe4786384f25e3,
74    0x0fc19dc68b8cd5b5,
75    0x240ca1cc77ac9c65,
76    0x2de92c6f592b0275,
77    0x4a7484aa6ea6e483,
78    0x5cb0a9dcbd41fbd4,
79    0x76f988da831153b5,
80    0x983e5152ee66dfab,
81    0xa831c66d2db43210,
82    0xb00327c898fb213f,
83    0xbf597fc7beef0ee4,
84    0xc6e00bf33da88fc2,
85    0xd5a79147930aa725,
86    0x06ca6351e003826f,
87    0x142929670a0e6e70,
88    0x27b70a8546d22ffc,
89    0x2e1b21385c26c926,
90    0x4d2c6dfc5ac42aed,
91    0x53380d139d95b3df,
92    0x650a73548baf63de,
93    0x766a0abb3c77b2a8,
94    0x81c2c92e47edaee6,
95    0x92722c851482353b,
96    0xa2bfe8a14cf10364,
97    0xa81a664bbc423001,
98    0xc24b8b70d0f89791,
99    0xc76c51a30654be30,
100    0xd192e819d6ef5218,
101    0xd69906245565a910,
102    0xf40e35855771202a,
103    0x106aa07032bbd1b8,
104    0x19a4c116b8d2d0c8,
105    0x1e376c085141ab53,
106    0x2748774cdf8eeb99,
107    0x34b0bcb5e19b48a8,
108    0x391c0cb3c5c95a63,
109    0x4ed8aa4ae3418acb,
110    0x5b9cca4f7763e373,
111    0x682e6ff3d6b2b8a3,
112    0x748f82ee5defb2fc,
113    0x78a5636f43172f60,
114    0x84c87814a1f0ab72,
115    0x8cc702081a6439ec,
116    0x90befffa23631e28,
117    0xa4506cebde82bde9,
118    0xbef9a3f7b2c67915,
119    0xc67178f2e372532b,
120    0xca273eceea26619c,
121    0xd186b8c721c0c207,
122    0xeada7dd6cde0eb1e,
123    0xf57d4f7fee6ed178,
124    0x06f067aa72176fba,
125    0x0a637dc5a2c898a6,
126    0x113f9804bef90dae,
127    0x1b710b35131c471b,
128    0x28db77f523047d84,
129    0x32caab7b40c72493,
130    0x3c9ebe0a15c9bebc,
131    0x431d67c49c100d4c,
132    0x4cc5d4becb3e42b6,
133    0x597f299cfc657e2a,
134    0x5fcb6fab3ad6faec,
135    0x6c44198c4a475817,
136];
137
138/// The shared 32-bit SHA-2 compression core (SHA-224 / SHA-256).
139#[derive(Clone)]
140struct Core256 {
141    h: [u32; 8],
142    buf: [u8; 64],
143    buffered: usize,
144    len: u64,
145}
146
147impl Drop for Core256 {
148    /// Wipe the chaining state and the buffered block.
149    ///
150    /// A hash is not a secret, but this state is not only used for hashing:
151    /// `Hmac<D>` holds two of these with the key already absorbed into them,
152    /// so the ipad and opad states are key-derived material. Putting the wipe
153    /// here rather than on `Hmac` means every consumer inherits it through
154    /// ordinary field drop, with no `Zeroize` bound threaded through the
155    /// `Digest` trait and no chance of a new wrapper forgetting.
156    fn drop(&mut self) {
157        self.h.zeroize();
158        self.buf.zeroize();
159        self.buffered = 0;
160        self.len = 0;
161    }
162}
163
164impl Core256 {
165    const fn new(iv: [u32; 8]) -> Self {
166        Self {
167            h: iv,
168            buf: [0u8; 64],
169            buffered: 0,
170            len: 0,
171        }
172    }
173
174    fn compress(&mut self, block: &[u8]) {
175        let mut w = [0u32; 64];
176        for i in 0..16 {
177            w[i] = u32::from_be_bytes([
178                block[i * 4],
179                block[i * 4 + 1],
180                block[i * 4 + 2],
181                block[i * 4 + 3],
182            ]);
183        }
184        for i in 16..64 {
185            let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
186            let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
187            w[i] = w[i - 16]
188                .wrapping_add(s0)
189                .wrapping_add(w[i - 7])
190                .wrapping_add(s1);
191        }
192        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = self.h;
193        for i in 0..64 {
194            let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
195            let ch = (e & f) ^ ((!e) & g);
196            let t1 = hh
197                .wrapping_add(s1)
198                .wrapping_add(ch)
199                .wrapping_add(K256[i])
200                .wrapping_add(w[i]);
201            let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
202            let maj = (a & b) ^ (a & c) ^ (b & c);
203            let t2 = s0.wrapping_add(maj);
204            hh = g;
205            g = f;
206            f = e;
207            e = d.wrapping_add(t1);
208            d = c;
209            c = b;
210            b = a;
211            a = t1.wrapping_add(t2);
212        }
213        let upd = [a, b, c, d, e, f, g, hh];
214        for i in 0..8 {
215            self.h[i] = self.h[i].wrapping_add(upd[i]);
216        }
217        w.zeroize();
218    }
219
220    /// Compress a whole number of blocks, using the hardware path when there
221    /// is one.
222    ///
223    /// Taking a run rather than a block at a time is the point: the SHA-NI
224    /// backend shuffles the state into and out of its register layout once per
225    /// call, so feeding it one block at a time would pay that on every block.
226    fn compress_blocks(&mut self, data: &[u8]) {
227        debug_assert!(data.len() % 64 == 0);
228        if data.is_empty() {
229            return;
230        }
231        #[cfg(all(target_arch = "x86_64", feature = "std"))]
232        if sha_ni() {
233            // SAFETY: `sha_ni()` is exactly the feature test this requires, and
234            // the length is a multiple of the block size by the assertion above.
235            unsafe { x86::compress(&mut self.h, data) };
236            return;
237        }
238        for block in data.chunks_exact(64) {
239            self.compress(block);
240        }
241    }
242
243    fn update(&mut self, mut data: &[u8]) {
244        self.len = self.len.wrapping_add(data.len() as u64);
245        if self.buffered > 0 {
246            let need = 64 - self.buffered;
247            let take = core::cmp::min(need, data.len());
248            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
249            self.buffered += take;
250            data = &data[take..];
251            if self.buffered < 64 {
252                // The whole input fit in the partial block; nothing to compress.
253                return;
254            }
255            let block = self.buf;
256            self.compress(&block);
257            self.buffered = 0;
258        }
259        let whole = data.len() - data.len() % 64;
260        self.compress_blocks(&data[..whole]);
261        let rest = &data[whole..];
262        self.buf[..rest.len()].copy_from_slice(rest);
263        self.buffered = rest.len();
264    }
265
266    fn finalize(mut self) -> [u32; 8] {
267        let bit_len = self.len.wrapping_mul(8);
268        let mut pad = [0u8; 72];
269        pad[0] = 0x80;
270        // Pad so that (buffered + 1 + zeros) % 64 == 56.
271        let zeros = (55 + 64 - (self.buffered % 64)) % 64;
272        pad[1 + zeros..1 + zeros + 8].copy_from_slice(&bit_len.to_be_bytes());
273        self.update_no_count(&pad[..1 + zeros + 8]);
274        let out = self.h;
275        self.buf.zeroize();
276        self.h.zeroize();
277        out
278    }
279
280    /// Absorb padding without disturbing the message-length counter.
281    fn update_no_count(&mut self, data: &[u8]) {
282        let saved = self.len;
283        self.update(data);
284        self.len = saved;
285    }
286}
287
288/// The shared 64-bit SHA-2 compression core (SHA-384 / SHA-512 / SHA-512-t).
289#[derive(Clone)]
290struct Core512 {
291    h: [u64; 8],
292    buf: [u8; 128],
293    buffered: usize,
294    len: u128,
295    /// The message schedule, kept here rather than built on the stack.
296    ///
297    /// It lives in the struct so that wiping it costs once per hash instead of
298    /// once per block. `Zeroize` writes element by element through
299    /// `write_volatile`, which is what makes the wipe non-elidable and also
300    /// what makes it expensive: eighty volatile stores cannot be merged into a
301    /// `memset` or vectorised, and measured in isolation they were 159ns of a
302    /// 258ns block -- 62% of SHA-512's compression spent clearing the schedule
303    /// rather than computing it. `what_the_schedule_wipe_costs` is that
304    /// measurement.
305    ///
306    /// The schedule is still wiped, in `finalize`, beside `h` and `buf`. What
307    /// changes is how often. Wiping after every block bought nothing that
308    /// survived the block anyway: the next block immediately overwrites the
309    /// whole array, and the material it is derived from is in the caller's
310    /// input buffer, which this library neither owns nor clears.
311    w: [u64; 80],
312}
313
314impl Drop for Core512 {
315    /// Wipe the chaining state and the buffered block. See [`Core256`].
316    fn drop(&mut self) {
317        self.h.zeroize();
318        self.buf.zeroize();
319        self.buffered = 0;
320        self.len = 0;
321    }
322}
323
324/// SHA-512's compression, and what has already been tried on it.
325///
326/// It runs about 1.5 times behind RustCrypto's, which has an AVX2 backend for
327/// the message schedule. There is no SHA-512 instruction on x86 the way there
328/// is for SHA-256, so the portable path below is what runs.
329///
330/// What closed the gap from 1.76x was not an optimisation of the arithmetic at
331/// all: 62% of the compression was the per-block `zeroize` of the schedule,
332/// whose volatile writes cannot be merged into a `memset`. The schedule now
333/// lives in the struct and is wiped once per hash. See the note on `w`.
334///
335/// Two source-level optimisations were measured and reverted, and are recorded
336/// so they are not tried a third time:
337///
338/// - **A rolling sixteen-word schedule window** instead of the eighty-word
339///   array. The array is 640 bytes cleared per 128-byte block, five bytes wiped
340///   per byte hashed, which looks like the cost. A controlled A/B showed it
341///   *slower*: 640 bytes sits in L1, and the modulo indexing defeats whatever
342///   unrolling the flat array was getting.
343/// - **Unrolling the round loop by eight**, naming the working variables in
344///   rotation so the eight moves per round disappear. No measurable change in
345///   either direction; LLVM already renames and unrolls this shape.
346///
347/// Both failed the same way: the waste was visible in the source and absent
348/// from the object code. What did pay elsewhere in this workspace was work the
349/// compiler cannot do -- breaking a serial dependency chain, selecting a
350/// hardware instruction, changing the algorithm. The remaining gap here is the
351/// vectorised schedule, and even that addresses only the third or so of the
352/// work the schedule represents, since the rounds are inherently serial.
353impl Core512 {
354    const fn new(iv: [u64; 8]) -> Self {
355        Self {
356            h: iv,
357            buf: [0u8; 128],
358            buffered: 0,
359            len: 0,
360            w: [0u64; 80],
361        }
362    }
363
364    fn compress(&mut self, block: &[u8]) {
365        Self::compress_into(&mut self.h, &mut self.w, block);
366    }
367
368    /// The compression function proper, over borrowed state.
369    ///
370    /// Split out so the schedule is reached through a plain `&mut [u64; 80]`
371    /// rather than through `self`, which keeps the indexing the same as it was
372    /// when the array was a local.
373    fn compress_into(h: &mut [u64; 8], w: &mut [u64; 80], block: &[u8]) {
374        for i in 0..16 {
375            let mut b = [0u8; 8];
376            b.copy_from_slice(&block[i * 8..i * 8 + 8]);
377            w[i] = u64::from_be_bytes(b);
378        }
379        for i in 16..80 {
380            let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
381            let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
382            w[i] = w[i - 16]
383                .wrapping_add(s0)
384                .wrapping_add(w[i - 7])
385                .wrapping_add(s1);
386        }
387        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
388        for i in 0..80 {
389            let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
390            let ch = (e & f) ^ ((!e) & g);
391            let t1 = hh
392                .wrapping_add(s1)
393                .wrapping_add(ch)
394                .wrapping_add(K512[i])
395                .wrapping_add(w[i]);
396            let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
397            let maj = (a & b) ^ (a & c) ^ (b & c);
398            let t2 = s0.wrapping_add(maj);
399            hh = g;
400            g = f;
401            f = e;
402            e = d.wrapping_add(t1);
403            d = c;
404            c = b;
405            b = a;
406            a = t1.wrapping_add(t2);
407        }
408        let upd = [a, b, c, d, e, f, g, hh];
409        for i in 0..8 {
410            h[i] = h[i].wrapping_add(upd[i]);
411        }
412    }
413
414    fn update(&mut self, mut data: &[u8]) {
415        self.len = self.len.wrapping_add(data.len() as u128);
416        if self.buffered > 0 {
417            let need = 128 - self.buffered;
418            let take = core::cmp::min(need, data.len());
419            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
420            self.buffered += take;
421            data = &data[take..];
422            if self.buffered < 128 {
423                // The whole input fit in the partial block; nothing to compress.
424                return;
425            }
426            let block = self.buf;
427            self.compress(&block);
428            self.buffered = 0;
429        }
430        let mut chunks = data.chunks_exact(128);
431        for block in &mut chunks {
432            self.compress(block);
433        }
434        let rest = chunks.remainder();
435        self.buf[..rest.len()].copy_from_slice(rest);
436        self.buffered = rest.len();
437    }
438
439    fn finalize(mut self) -> [u64; 8] {
440        let bit_len = self.len.wrapping_mul(8);
441        let mut pad = [0u8; 145];
442        pad[0] = 0x80;
443        let zeros = (111 + 128 - (self.buffered % 128)) % 128;
444        pad[1 + zeros..1 + zeros + 16].copy_from_slice(&bit_len.to_be_bytes());
445        let saved = self.len;
446        self.update(&pad[..1 + zeros + 16]);
447        self.len = saved;
448        let out = self.h;
449        self.buf.zeroize();
450        self.h.zeroize();
451        self.w.zeroize();
452        out
453    }
454}
455
456macro_rules! sha2_32 {
457    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
458        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
459        #[derive(Clone)]
460        pub struct $name(Core256);
461
462        impl Default for $name {
463            fn default() -> Self {
464                Self(Core256::new($iv))
465            }
466        }
467
468        impl Algorithm for $name {
469            const ID: &'static str = $id;
470            const NAME: &'static str = $disp;
471        }
472
473        impl Digest for $name {
474            type Output = [u8; $out];
475            const OUTPUT_LEN: usize = $out;
476            const BLOCK_LEN: usize = 64;
477
478            fn update(&mut self, data: &[u8]) {
479                self.0.update(data);
480            }
481
482            fn finalize(self) -> Self::Output {
483                let h = self.0.finalize();
484                let mut full = [0u8; 32];
485                for i in 0..8 {
486                    full[i * 4..i * 4 + 4].copy_from_slice(&h[i].to_be_bytes());
487                }
488                let mut out = [0u8; $out];
489                out.copy_from_slice(&full[..$out]);
490                out
491            }
492        }
493
494        impl SelfTest for $name {
495            fn self_test() -> Result<()> {
496                let got = <Self as Digest>::digest(b"abc");
497                let mut want = [0u8; $out];
498                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
499                ensure!(
500                    ic_core::ct::verify(&want, got.as_ref()),
501                    SelfTestFailed,
502                    $id
503                );
504                Ok(())
505            }
506        }
507    };
508}
509
510macro_rules! sha2_64 {
511    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
512        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
513        #[derive(Clone)]
514        pub struct $name(Core512);
515
516        impl Default for $name {
517            fn default() -> Self {
518                Self(Core512::new($iv))
519            }
520        }
521
522        impl Algorithm for $name {
523            const ID: &'static str = $id;
524            const NAME: &'static str = $disp;
525        }
526
527        impl Digest for $name {
528            type Output = [u8; $out];
529            const OUTPUT_LEN: usize = $out;
530            const BLOCK_LEN: usize = 128;
531
532            fn update(&mut self, data: &[u8]) {
533                self.0.update(data);
534            }
535
536            fn finalize(self) -> Self::Output {
537                let h = self.0.finalize();
538                let mut full = [0u8; 64];
539                for i in 0..8 {
540                    full[i * 8..i * 8 + 8].copy_from_slice(&h[i].to_be_bytes());
541                }
542                let mut out = [0u8; $out];
543                out.copy_from_slice(&full[..$out]);
544                out
545            }
546        }
547
548        impl SelfTest for $name {
549            fn self_test() -> Result<()> {
550                let got = <Self as Digest>::digest(b"abc");
551                let mut want = [0u8; $out];
552                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
553                ensure!(
554                    ic_core::ct::verify(&want, got.as_ref()),
555                    SelfTestFailed,
556                    $id
557                );
558                Ok(())
559            }
560        }
561    };
562}
563
564sha2_32!(
565    Sha224,
566    "sha2-224",
567    "SHA-224",
568    28,
569    [
570        0xc1059ed8, 0x367cd507, 0x3070dd17, 0xf70e5939, 0xffc00b31, 0x68581511, 0x64f98fa7,
571        0xbefa4fa4
572    ],
573    "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
574);
575
576sha2_32!(
577    Sha256,
578    "sha2-256",
579    "SHA-256",
580    32,
581    [
582        0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
583        0x5be0cd19
584    ],
585    "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
586);
587
588sha2_64!(
589    Sha384,
590    "sha2-384",
591    "SHA-384",
592    48,
593    [
594        0xcbbb9d5dc1059ed8, 0x629a292a367cd507, 0x9159015a3070dd17, 0x152fecd8f70e5939,
595        0x67332667ffc00b31, 0x8eb44a8768581511, 0xdb0c2e0d64f98fa7, 0x47b5481dbefa4fa4
596    ],
597    "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
598);
599
600sha2_64!(
601    Sha512,
602    "sha2-512",
603    "SHA-512",
604    64,
605    [
606        0x6a09e667f3bcc908, 0xbb67ae8584caa73b, 0x3c6ef372fe94f82b, 0xa54ff53a5f1d36f1,
607        0x510e527fade682d1, 0x9b05688c2b3e6c1f, 0x1f83d9abfb41bd6b, 0x5be0cd19137e2179
608    ],
609    "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
610);
611
612sha2_64!(
613    Sha512_224,
614    "sha2-512-224",
615    "SHA-512/224",
616    28,
617    [
618        0x8c3d37c819544da2,
619        0x73e1996689dcd4d6,
620        0x1dfab7ae32ff9c82,
621        0x679dd514582f9fcf,
622        0x0f6d2b697bd44da8,
623        0x77e36f7304c48942,
624        0x3f9d85a86a1d36c8,
625        0x1112e6ad91d692a1
626    ],
627    "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
628);
629
630sha2_64!(
631    Sha512_256,
632    "sha2-512-256",
633    "SHA-512/256",
634    32,
635    [
636        0x22312194fc2bf72c,
637        0x9f555fa3c84c64c2,
638        0x2393b86b6f53b151,
639        0x963877195940eabd,
640        0x96283ee2a88effe3,
641        0xbe5e1e2553863992,
642        0x2b0199fc2c85b8aa,
643        0x0eb72ddc81c52ca2
644    ],
645    "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
646);
647
648#[cfg(test)]
649mod tests {
650    use super::*;
651
652    /// The hardware path must agree with the portable one, block for block.
653    ///
654    /// The published vectors above do not establish this. They pass whichever
655    /// path runs, so on a machine with SHA-NI they check the backend and on one
656    /// without they check the fallback -- and either way they cannot notice
657    /// that the two disagree, which is the failure a second implementation
658    /// introduces. This runs both over the same input and compares the states.
659    ///
660    /// It reports which path it took rather than asserting one, because a CPU
661    /// without the instructions is a legitimate machine to run the suite on.
662    /// What it does assert is that the comparison happened when it could.
663    #[cfg(all(target_arch = "x86_64", feature = "std"))]
664    #[test]
665    fn the_sha_ni_backend_agrees_with_the_portable_one() {
666        if !sha_ni() {
667            println!("no SHA-NI on this CPU; the backend was not exercised");
668            return;
669        }
670
671        // Lengths either side of the block boundary, and long enough to run the
672        // message schedule over several blocks.
673        let mut checked = 0;
674        for blocks in [1usize, 2, 3, 4, 7, 16] {
675            let mut data = vec![0u8; blocks * 64];
676            // Not random, but not uniform either: a counter through a couple of
677            // multiplications, so every byte position varies between cases.
678            for (i, b) in data.iter_mut().enumerate() {
679                *b = ((i as u64).wrapping_mul(0x9e37_79b9).rotate_left(7) & 0xff) as u8;
680            }
681
682            // FIPS 180-4 section 5.3.3, the same value the macro below passes.
683            const IV: [u32; 8] = [
684                0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
685                0x5be0cd19,
686            ];
687            let mut portable = Core256::new(IV);
688            for block in data.chunks_exact(64) {
689                portable.compress(block);
690            }
691
692            let mut hardware = Core256::new(IV);
693            // SAFETY: guarded by the `sha_ni()` check above.
694            unsafe { x86::compress(&mut hardware.h, &data) };
695
696            assert_eq!(
697                portable.h, hardware.h,
698                "SHA-NI and portable disagree after {blocks} blocks"
699            );
700            checked += 1;
701        }
702        assert_eq!(checked, 6, "the comparison did not run");
703    }
704
705    /// Say which path this build will take, so a benchmark or a vector run is
706    /// not silently measuring the fallback.
707    #[cfg(all(target_arch = "x86_64", feature = "std"))]
708    #[test]
709    fn the_active_sha256_path_is_reported() {
710        println!(
711            "sha-256 backend: {}",
712            if sha_ni() { "SHA-NI" } else { "portable" }
713        );
714    }
715
716    #[test]
717    fn nist_abc_vectors() {
718        assert_eq!(
719            ic_core::codec::hex(Sha256::digest(b"abc").as_ref()),
720            "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
721        );
722        assert_eq!(
723            ic_core::codec::hex(Sha224::digest(b"abc").as_ref()),
724            "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
725        );
726        assert_eq!(
727            ic_core::codec::hex(Sha512::digest(b"abc").as_ref()),
728            "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
729        );
730        assert_eq!(
731            ic_core::codec::hex(Sha384::digest(b"abc").as_ref()),
732            "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
733        );
734    }
735
736    #[test]
737    fn empty_input_vectors() {
738        assert_eq!(
739            ic_core::codec::hex(Sha256::digest(b"").as_ref()),
740            "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
741        );
742        assert_eq!(
743            ic_core::codec::hex(Sha512::digest(b"").as_ref()),
744            "cf83e1357eefb8bdf1542850d66d8007d620e4050b5715dc83f4a921d36ce9ce47d0d13c5d85f2b0ff8318d2877eec2f63b931bd47417a81a538327af927da3e"
745        );
746    }
747
748    /// The 448-bit boundary case: input length forces an extra padding block.
749    #[test]
750    fn two_block_vector() {
751        let msg = b"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq";
752        assert_eq!(
753            ic_core::codec::hex(Sha256::digest(msg).as_ref()),
754            "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"
755        );
756    }
757
758    #[test]
759    fn million_a_vector() {
760        let mut h = Sha256::new();
761        let chunk = [b'a'; 1000];
762        for _ in 0..1000 {
763            h.update(&chunk);
764        }
765        assert_eq!(
766            ic_core::codec::hex(h.finalize().as_ref()),
767            "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"
768        );
769    }
770
771    #[test]
772    fn streaming_matches_one_shot() {
773        let data: [u8; 300] = core::array::from_fn(|i| i as u8);
774        for split in [0usize, 1, 63, 64, 65, 127, 128, 200, 300] {
775            let mut h = Sha512::new();
776            h.update(&data[..split]);
777            h.update(&data[split..]);
778            assert_eq!(h.finalize(), Sha512::digest(&data), "split at {split}");
779        }
780    }
781
782    #[test]
783    fn truncated_variants() {
784        assert_eq!(
785            ic_core::codec::hex(Sha512_224::digest(b"abc").as_ref()),
786            "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
787        );
788        assert_eq!(
789            ic_core::codec::hex(Sha512_256::digest(b"abc").as_ref()),
790            "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
791        );
792    }
793
794    #[test]
795    fn all_self_tests_pass() {
796        Sha224::self_test().unwrap();
797        Sha256::self_test().unwrap();
798        Sha384::self_test().unwrap();
799        Sha512::self_test().unwrap();
800        Sha512_224::self_test().unwrap();
801        Sha512_256::self_test().unwrap();
802    }
803
804    /// What the schedule wipe costs SHA-512, measured in one process.
805    ///
806    /// Ignored: a measurement. Run it with
807    /// `cargo test -p ic-hash --release -- --ignored --nocapture what_the_schedule_wipe_costs`.
808    ///
809    /// Both variants are timed in the same binary, alternating, because this
810    /// machine has other work on it: an attempt to compare across two benchmark
811    /// runs had RustCrypto's own SHA-512 moving 700 -> 1087 MiB/s between them,
812    /// untouched, which is larger than the effect being looked for.
813    #[test]
814    #[ignore = "diagnostic, not a test"]
815    fn what_the_schedule_wipe_costs() {
816        use std::time::Instant;
817
818        // A copy of Core512::compress with the wipe left out, and nothing else
819        // changed. Only for this measurement.
820        fn compress_unwiped(h: &mut [u64; 8], block: &[u8]) {
821            let mut w = [0u64; 80];
822            for i in 0..16 {
823                let mut b = [0u8; 8];
824                b.copy_from_slice(&block[i * 8..i * 8 + 8]);
825                w[i] = u64::from_be_bytes(b);
826            }
827            for i in 16..80 {
828                let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
829                let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
830                w[i] = w[i - 16]
831                    .wrapping_add(s0)
832                    .wrapping_add(w[i - 7])
833                    .wrapping_add(s1);
834            }
835            let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
836            for i in 0..80 {
837                let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
838                let ch = (e & f) ^ ((!e) & g);
839                let t1 = hh
840                    .wrapping_add(s1)
841                    .wrapping_add(ch)
842                    .wrapping_add(K512[i])
843                    .wrapping_add(w[i]);
844                let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
845                let maj = (a & b) ^ (a & c) ^ (b & c);
846                let t2 = s0.wrapping_add(maj);
847                hh = g;
848                g = f;
849                f = e;
850                e = d.wrapping_add(t1);
851                d = c;
852                c = b;
853                b = a;
854                a = t1.wrapping_add(t2);
855            }
856            let upd = [a, b, c, d, e, f, g, hh];
857            for i in 0..8 {
858                h[i] = h[i].wrapping_add(upd[i]);
859            }
860        }
861
862        let block: Vec<u8> = (0..128u32).map(|i| (i * 7 + 1) as u8).collect();
863        let n = 50_000;
864        let (mut best_wiped, mut best_plain) = (f64::INFINITY, f64::INFINITY);
865
866        for _ in 0..30 {
867            let mut core = Core512::new([1, 2, 3, 4, 5, 6, 7, 8]);
868            let t = Instant::now();
869            for _ in 0..n {
870                core.compress(core::hint::black_box(&block));
871            }
872            best_wiped = best_wiped.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
873
874            let mut h = [1u64, 2, 3, 4, 5, 6, 7, 8];
875            let t = Instant::now();
876            for _ in 0..n {
877                compress_unwiped(&mut h, core::hint::black_box(&block));
878            }
879            best_plain = best_plain.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
880        }
881        println!(
882            "
883  sha-512 compress, with w.zeroize()   {best_wiped:>8.1} ns/block"
884        );
885        println!("  sha-512 compress, without            {best_plain:>8.1} ns/block");
886        println!(
887            "  the wipe costs                       {:>8.1} ns/block ({:.0}%)",
888            best_wiped - best_plain,
889            (best_wiped - best_plain) / best_wiped * 100.0
890        );
891    }
892}