Skip to main content

ic_hash/
sha2.rs

1//! FIPS 180-4 SHA-2 family.
2//!
3//! Two cores (32-bit and 64-bit) are shared by six published output variants,
4//! which differ only in their initial hash value and truncation length.
5
6//! Indexed loops over fixed-size limb and word arrays are used throughout; they
7//! mirror the index algebra in the specifications these routines implement, so
8//! `needless_range_loop` is allowed rather than obscuring the correspondence.
9#![allow(clippy::needless_range_loop)]
10
11use ic_core::traits::{Algorithm, Digest, SelfTest};
12use ic_core::{ensure, Result, Zeroize};
13
14// SHA-NI, where the CPU has it. Only under `std`, because the detection does:
15// a `no_std` build has no way to ask, and guessing wrong is an illegal
16// instruction rather than a wrong answer.
17#[cfg(all(target_arch = "x86_64", feature = "std"))]
18mod x86;
19
20// SHA-512's schedule, four words at a time. Same reasoning as `x86`: only
21// under `std`, because the detection needs it.
22#[cfg(all(target_arch = "x86_64", feature = "std"))]
23mod avx2_512;
24
25/// Whether this CPU has AVX2, asked once.
26#[cfg(all(target_arch = "x86_64", feature = "std"))]
27fn avx2() -> bool {
28    use core::sync::atomic::{AtomicU8, Ordering};
29    // 0 not yet asked, 1 yes, 2 no.
30    static CACHED: AtomicU8 = AtomicU8::new(0);
31    match CACHED.load(Ordering::Relaxed) {
32        1 => true,
33        2 => false,
34        _ => {
35            let have = std::is_x86_feature_detected!("avx2");
36            CACHED.store(u8::from(!have) + 1, Ordering::Relaxed);
37            have
38        }
39    }
40}
41
42/// Whether this CPU has the instructions [`x86::compress`] needs.
43///
44/// Asked once. `is_x86_feature_detected!` is not free, and SHA-256 is called
45/// often enough on small inputs that paying for the query per block would show
46/// up in exactly the workloads this is meant to help.
47#[cfg(all(target_arch = "x86_64", feature = "std"))]
48fn sha_ni() -> bool {
49    use core::sync::atomic::{AtomicU8, Ordering};
50    // 0 not yet asked, 1 yes, 2 no.
51    static CACHED: AtomicU8 = AtomicU8::new(0);
52    match CACHED.load(Ordering::Relaxed) {
53        1 => true,
54        2 => false,
55        _ => {
56            let have = std::is_x86_feature_detected!("sha")
57                && std::is_x86_feature_detected!("sse2")
58                && std::is_x86_feature_detected!("ssse3")
59                && std::is_x86_feature_detected!("sse4.1");
60            CACHED.store(u8::from(!have) + 1, Ordering::Relaxed);
61            have
62        }
63    }
64}
65
66const K256: [u32; 64] = [
67    0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5,
68    0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174,
69    0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
70    0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967,
71    0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85,
72    0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
73    0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
74    0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
75];
76
77const K512: [u64; 80] = [
78    0x428a2f98d728ae22,
79    0x7137449123ef65cd,
80    0xb5c0fbcfec4d3b2f,
81    0xe9b5dba58189dbbc,
82    0x3956c25bf348b538,
83    0x59f111f1b605d019,
84    0x923f82a4af194f9b,
85    0xab1c5ed5da6d8118,
86    0xd807aa98a3030242,
87    0x12835b0145706fbe,
88    0x243185be4ee4b28c,
89    0x550c7dc3d5ffb4e2,
90    0x72be5d74f27b896f,
91    0x80deb1fe3b1696b1,
92    0x9bdc06a725c71235,
93    0xc19bf174cf692694,
94    0xe49b69c19ef14ad2,
95    0xefbe4786384f25e3,
96    0x0fc19dc68b8cd5b5,
97    0x240ca1cc77ac9c65,
98    0x2de92c6f592b0275,
99    0x4a7484aa6ea6e483,
100    0x5cb0a9dcbd41fbd4,
101    0x76f988da831153b5,
102    0x983e5152ee66dfab,
103    0xa831c66d2db43210,
104    0xb00327c898fb213f,
105    0xbf597fc7beef0ee4,
106    0xc6e00bf33da88fc2,
107    0xd5a79147930aa725,
108    0x06ca6351e003826f,
109    0x142929670a0e6e70,
110    0x27b70a8546d22ffc,
111    0x2e1b21385c26c926,
112    0x4d2c6dfc5ac42aed,
113    0x53380d139d95b3df,
114    0x650a73548baf63de,
115    0x766a0abb3c77b2a8,
116    0x81c2c92e47edaee6,
117    0x92722c851482353b,
118    0xa2bfe8a14cf10364,
119    0xa81a664bbc423001,
120    0xc24b8b70d0f89791,
121    0xc76c51a30654be30,
122    0xd192e819d6ef5218,
123    0xd69906245565a910,
124    0xf40e35855771202a,
125    0x106aa07032bbd1b8,
126    0x19a4c116b8d2d0c8,
127    0x1e376c085141ab53,
128    0x2748774cdf8eeb99,
129    0x34b0bcb5e19b48a8,
130    0x391c0cb3c5c95a63,
131    0x4ed8aa4ae3418acb,
132    0x5b9cca4f7763e373,
133    0x682e6ff3d6b2b8a3,
134    0x748f82ee5defb2fc,
135    0x78a5636f43172f60,
136    0x84c87814a1f0ab72,
137    0x8cc702081a6439ec,
138    0x90befffa23631e28,
139    0xa4506cebde82bde9,
140    0xbef9a3f7b2c67915,
141    0xc67178f2e372532b,
142    0xca273eceea26619c,
143    0xd186b8c721c0c207,
144    0xeada7dd6cde0eb1e,
145    0xf57d4f7fee6ed178,
146    0x06f067aa72176fba,
147    0x0a637dc5a2c898a6,
148    0x113f9804bef90dae,
149    0x1b710b35131c471b,
150    0x28db77f523047d84,
151    0x32caab7b40c72493,
152    0x3c9ebe0a15c9bebc,
153    0x431d67c49c100d4c,
154    0x4cc5d4becb3e42b6,
155    0x597f299cfc657e2a,
156    0x5fcb6fab3ad6faec,
157    0x6c44198c4a475817,
158];
159
160/// The shared 32-bit SHA-2 compression core (SHA-224 / SHA-256).
161#[derive(Clone)]
162struct Core256 {
163    h: [u32; 8],
164    buf: [u8; 64],
165    buffered: usize,
166    len: u64,
167}
168
169impl Drop for Core256 {
170    /// Wipe the chaining state and the buffered block.
171    ///
172    /// A hash is not a secret, but this state is not only used for hashing:
173    /// `Hmac<D>` holds two of these with the key already absorbed into them,
174    /// so the ipad and opad states are key-derived material. Putting the wipe
175    /// here rather than on `Hmac` means every consumer inherits it through
176    /// ordinary field drop, with no `Zeroize` bound threaded through the
177    /// `Digest` trait and no chance of a new wrapper forgetting.
178    fn drop(&mut self) {
179        self.h.zeroize();
180        self.buf.zeroize();
181        self.buffered = 0;
182        self.len = 0;
183    }
184}
185
186impl Core256 {
187    const fn new(iv: [u32; 8]) -> Self {
188        Self {
189            h: iv,
190            buf: [0u8; 64],
191            buffered: 0,
192            len: 0,
193        }
194    }
195
196    fn compress(&mut self, block: &[u8]) {
197        let mut w = [0u32; 64];
198        for i in 0..16 {
199            w[i] = u32::from_be_bytes([
200                block[i * 4],
201                block[i * 4 + 1],
202                block[i * 4 + 2],
203                block[i * 4 + 3],
204            ]);
205        }
206        for i in 16..64 {
207            let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
208            let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
209            w[i] = w[i - 16]
210                .wrapping_add(s0)
211                .wrapping_add(w[i - 7])
212                .wrapping_add(s1);
213        }
214        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = self.h;
215        for i in 0..64 {
216            let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
217            let ch = (e & f) ^ ((!e) & g);
218            let t1 = hh
219                .wrapping_add(s1)
220                .wrapping_add(ch)
221                .wrapping_add(K256[i])
222                .wrapping_add(w[i]);
223            let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
224            let maj = (a & b) ^ (a & c) ^ (b & c);
225            let t2 = s0.wrapping_add(maj);
226            hh = g;
227            g = f;
228            f = e;
229            e = d.wrapping_add(t1);
230            d = c;
231            c = b;
232            b = a;
233            a = t1.wrapping_add(t2);
234        }
235        let upd = [a, b, c, d, e, f, g, hh];
236        for i in 0..8 {
237            self.h[i] = self.h[i].wrapping_add(upd[i]);
238        }
239        w.zeroize();
240    }
241
242    /// Compress a whole number of blocks, using the hardware path when there
243    /// is one.
244    ///
245    /// Taking a run rather than a block at a time is the point: the SHA-NI
246    /// backend shuffles the state into and out of its register layout once per
247    /// call, so feeding it one block at a time would pay that on every block.
248    fn compress_blocks(&mut self, data: &[u8]) {
249        debug_assert!(data.len().is_multiple_of(64));
250        if data.is_empty() {
251            return;
252        }
253        #[cfg(all(target_arch = "x86_64", feature = "std"))]
254        if sha_ni() {
255            // SAFETY: `sha_ni()` is exactly the feature test this requires, and
256            // the length is a multiple of the block size by the assertion above.
257            unsafe { x86::compress(&mut self.h, data) };
258            return;
259        }
260        for block in data.chunks_exact(64) {
261            self.compress(block);
262        }
263    }
264
265    fn update(&mut self, mut data: &[u8]) {
266        self.len = self.len.wrapping_add(data.len() as u64);
267        if self.buffered > 0 {
268            let need = 64 - self.buffered;
269            let take = core::cmp::min(need, data.len());
270            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
271            self.buffered += take;
272            data = &data[take..];
273            if self.buffered < 64 {
274                // The whole input fit in the partial block; nothing to compress.
275                return;
276            }
277            // Through the dispatcher, not `compress`: that is the portable
278            // round function, and calling it here sent every block assembled
279            // in the buffer -- the tail of any message not a multiple of 64
280            // bytes, and every final padding block -- past SHA-NI. Short
281            // messages are mostly such blocks, so a 200-byte hash ran at a
282            // third of the bulk rate and HMAC, which finishes two hashes per
283            // tag, at about a quarter of what SHA-NI allows.
284            let block = self.buf;
285            self.compress_blocks(&block);
286            self.buffered = 0;
287        }
288        let whole = data.len() - data.len() % 64;
289        self.compress_blocks(&data[..whole]);
290        let rest = &data[whole..];
291        self.buf[..rest.len()].copy_from_slice(rest);
292        self.buffered = rest.len();
293    }
294
295    fn finalize(mut self) -> [u32; 8] {
296        let bit_len = self.len.wrapping_mul(8);
297        let mut pad = [0u8; 72];
298        pad[0] = 0x80;
299        // Pad so that (buffered + 1 + zeros) % 64 == 56.
300        let zeros = (55 + 64 - (self.buffered % 64)) % 64;
301        pad[1 + zeros..1 + zeros + 8].copy_from_slice(&bit_len.to_be_bytes());
302        self.update_no_count(&pad[..1 + zeros + 8]);
303        // `self` is dropped on return, and `Drop` wipes `h` and `buf`. They
304        // used to be wiped here as well, so every hash wiped its state twice;
305        // HMAC finishes two hashes per tag, so that was four wipes of 96 bytes
306        // where two do the same job.
307        self.h
308    }
309
310    /// Absorb padding without disturbing the message-length counter.
311    fn update_no_count(&mut self, data: &[u8]) {
312        let saved = self.len;
313        self.update(data);
314        self.len = saved;
315    }
316}
317
318/// The shared 64-bit SHA-2 compression core (SHA-384 / SHA-512 / SHA-512-t).
319#[derive(Clone)]
320struct Core512 {
321    h: [u64; 8],
322    buf: [u8; 128],
323    buffered: usize,
324    len: u128,
325    /// The message schedule, kept here rather than built on the stack.
326    ///
327    /// It lives in the struct so that wiping it costs once per hash instead of
328    /// once per block. `Zeroize` writes element by element through
329    /// `write_volatile`, which is what makes the wipe non-elidable and also
330    /// what makes it expensive: eighty volatile stores cannot be merged into a
331    /// `memset` or vectorised, and measured in isolation they were 159ns of a
332    /// 258ns block -- 62% of SHA-512's compression spent clearing the schedule
333    /// rather than computing it. `what_the_schedule_wipe_costs` is that
334    /// measurement.
335    ///
336    /// The schedule is still wiped, in `finalize`, beside `h` and `buf`. What
337    /// changes is how often. Wiping after every block bought nothing that
338    /// survived the block anyway: the next block immediately overwrites the
339    /// whole array, and the material it is derived from is in the caller's
340    /// input buffer, which this library neither owns nor clears.
341    w: [u64; 80],
342}
343
344impl Drop for Core512 {
345    /// Wipe the chaining state and the buffered block. See [`Core256`].
346    fn drop(&mut self) {
347        self.h.zeroize();
348        self.buf.zeroize();
349        self.buffered = 0;
350        self.len = 0;
351    }
352}
353
354/// SHA-512's compression, and what has already been tried on it.
355///
356/// It runs about 1.5 times behind RustCrypto's, which has an AVX2 backend for
357/// the message schedule. There is no SHA-512 instruction on x86 the way there
358/// is for SHA-256, so the portable path below is what runs.
359///
360/// What closed the gap from 1.76x was not an optimisation of the arithmetic at
361/// all: 62% of the compression was the per-block `zeroize` of the schedule,
362/// whose volatile writes cannot be merged into a `memset`. The schedule now
363/// lives in the struct and is wiped once per hash. See the note on `w`.
364///
365/// The remaining gap is not scalar slack, and it is worth saying why before
366/// anyone looks for some. At 715 MiB/s this compresses a block in about 180ns,
367/// which on this machine is roughly 2900 instructions in 600 cycles: close to
368/// five per cycle, near what a four-wide core can retire. RustCrypto's 1070
369/// MiB/s would need better than seven per cycle for the same instruction count,
370/// which is not possible -- so they are executing fewer instructions, not
371/// scheduling the same ones better. `sha2 0.10.9` has an AVX2 SHA-512 backend
372/// in `sha512/x86.rs`, selected by runtime detection; that is the difference.
373/// Matching it means writing one, not tuning this.
374///
375/// Two source-level optimisations were measured and reverted, and are recorded
376/// so they are not tried a third time:
377///
378/// - **A rolling sixteen-word schedule window** instead of the eighty-word
379///   array. The array is 640 bytes cleared per 128-byte block, five bytes wiped
380///   per byte hashed, which looks like the cost. A controlled A/B showed it
381///   *slower*: 640 bytes sits in L1, and the modulo indexing defeats whatever
382///   unrolling the flat array was getting.
383/// - **Unrolling the round loop by eight**, naming the working variables in
384///   rotation so the eight moves per round disappear. No measurable change in
385///   either direction; LLVM already renames and unrolls this shape.
386///
387/// Both failed the same way: the waste was visible in the source and absent
388/// from the object code. What did pay elsewhere in this workspace was work the
389/// compiler cannot do -- breaking a serial dependency chain, selecting a
390/// hardware instruction, changing the algorithm. The remaining gap here is the
391/// vectorised schedule, and even that addresses only the third or so of the
392/// work the schedule represents, since the rounds are inherently serial.
393impl Core512 {
394    const fn new(iv: [u64; 8]) -> Self {
395        Self {
396            h: iv,
397            buf: [0u8; 128],
398            buffered: 0,
399            len: 0,
400            w: [0u64; 80],
401        }
402    }
403
404    fn compress(&mut self, block: &[u8]) {
405        // The schedule is about a third of the work and is the only part with
406        // anything to run in parallel; the rounds are a chain. So the backends
407        // differ in how `w` is filled and share everything after it.
408        #[cfg(all(target_arch = "x86_64", feature = "std"))]
409        if avx2() {
410            // SAFETY: `avx2()` is the feature test this requires, and `block`
411            // is one block by `update`'s chunking.
412            unsafe { avx2_512::compress(&mut self.h, &mut self.w, block) };
413            return;
414        }
415        Self::schedule(&mut self.w, block);
416        Self::rounds(&mut self.h, &self.w);
417    }
418
419    /// Build the message schedule.
420    ///
421    /// The AVX2 backend does not call this -- it interleaves the same steps
422    /// into its round loop, which is the whole reason it is faster -- but it is
423    /// checked against that backend word for word.
424    fn schedule(w: &mut [u64; 80], block: &[u8]) {
425        for i in 0..16 {
426            let mut b = [0u8; 8];
427            b.copy_from_slice(&block[i * 8..i * 8 + 8]);
428            w[i] = u64::from_be_bytes(b);
429        }
430        for i in 16..80 {
431            let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
432            let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
433            w[i] = w[i - 16]
434                .wrapping_add(s0)
435                .wrapping_add(w[i - 7])
436                .wrapping_add(s1);
437        }
438    }
439
440    /// The eighty rounds.
441    fn rounds(h: &mut [u64; 8], w: &[u64; 80]) {
442        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
443        for i in 0..80 {
444            let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
445            let ch = (e & f) ^ ((!e) & g);
446            let t1 = hh
447                .wrapping_add(s1)
448                .wrapping_add(ch)
449                .wrapping_add(K512[i])
450                .wrapping_add(w[i]);
451            let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
452            let maj = (a & b) ^ (a & c) ^ (b & c);
453            let t2 = s0.wrapping_add(maj);
454            hh = g;
455            g = f;
456            f = e;
457            e = d.wrapping_add(t1);
458            d = c;
459            c = b;
460            b = a;
461            a = t1.wrapping_add(t2);
462        }
463        let upd = [a, b, c, d, e, f, g, hh];
464        for i in 0..8 {
465            h[i] = h[i].wrapping_add(upd[i]);
466        }
467    }
468
469    fn update(&mut self, mut data: &[u8]) {
470        self.len = self.len.wrapping_add(data.len() as u128);
471        if self.buffered > 0 {
472            let need = 128 - self.buffered;
473            let take = core::cmp::min(need, data.len());
474            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
475            self.buffered += take;
476            data = &data[take..];
477            if self.buffered < 128 {
478                // The whole input fit in the partial block; nothing to compress.
479                return;
480            }
481            let block = self.buf;
482            self.compress(&block);
483            self.buffered = 0;
484        }
485        let mut chunks = data.chunks_exact(128);
486        for block in &mut chunks {
487            self.compress(block);
488        }
489        let rest = chunks.remainder();
490        self.buf[..rest.len()].copy_from_slice(rest);
491        self.buffered = rest.len();
492    }
493
494    fn finalize(mut self) -> [u64; 8] {
495        let bit_len = self.len.wrapping_mul(8);
496        let mut pad = [0u8; 145];
497        pad[0] = 0x80;
498        let zeros = (111 + 128 - (self.buffered % 128)) % 128;
499        pad[1 + zeros..1 + zeros + 16].copy_from_slice(&bit_len.to_be_bytes());
500        let saved = self.len;
501        self.update(&pad[..1 + zeros + 16]);
502        self.len = saved;
503        let out = self.h;
504        self.buf.zeroize();
505        self.h.zeroize();
506        self.w.zeroize();
507        out
508    }
509}
510
511macro_rules! sha2_32 {
512    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
513        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
514        #[derive(Clone)]
515        pub struct $name(Core256);
516
517        impl Default for $name {
518            fn default() -> Self {
519                Self(Core256::new($iv))
520            }
521        }
522
523        impl Algorithm for $name {
524            const ID: &'static str = $id;
525            const NAME: &'static str = $disp;
526        }
527
528        impl Digest for $name {
529            type Output = [u8; $out];
530            const OUTPUT_LEN: usize = $out;
531            const BLOCK_LEN: usize = 64;
532
533            fn update(&mut self, data: &[u8]) {
534                self.0.update(data);
535            }
536
537            fn finalize(self) -> Self::Output {
538                let h = self.0.finalize();
539                let mut full = [0u8; 32];
540                for i in 0..8 {
541                    full[i * 4..i * 4 + 4].copy_from_slice(&h[i].to_be_bytes());
542                }
543                let mut out = [0u8; $out];
544                out.copy_from_slice(&full[..$out]);
545                out
546            }
547        }
548
549        impl SelfTest for $name {
550            fn self_test() -> Result<()> {
551                let got = <Self as Digest>::digest(b"abc");
552                let mut want = [0u8; $out];
553                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
554                ensure!(
555                    ic_core::ct::verify(&want, got.as_ref()),
556                    SelfTestFailed,
557                    $id
558                );
559                Ok(())
560            }
561        }
562    };
563}
564
565macro_rules! sha2_64 {
566    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
567        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
568        #[derive(Clone)]
569        pub struct $name(Core512);
570
571        impl Default for $name {
572            fn default() -> Self {
573                Self(Core512::new($iv))
574            }
575        }
576
577        impl Algorithm for $name {
578            const ID: &'static str = $id;
579            const NAME: &'static str = $disp;
580        }
581
582        impl Digest for $name {
583            type Output = [u8; $out];
584            const OUTPUT_LEN: usize = $out;
585            const BLOCK_LEN: usize = 128;
586
587            fn update(&mut self, data: &[u8]) {
588                self.0.update(data);
589            }
590
591            fn finalize(self) -> Self::Output {
592                let h = self.0.finalize();
593                let mut full = [0u8; 64];
594                for i in 0..8 {
595                    full[i * 8..i * 8 + 8].copy_from_slice(&h[i].to_be_bytes());
596                }
597                let mut out = [0u8; $out];
598                out.copy_from_slice(&full[..$out]);
599                out
600            }
601        }
602
603        impl SelfTest for $name {
604            fn self_test() -> Result<()> {
605                let got = <Self as Digest>::digest(b"abc");
606                let mut want = [0u8; $out];
607                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
608                ensure!(
609                    ic_core::ct::verify(&want, got.as_ref()),
610                    SelfTestFailed,
611                    $id
612                );
613                Ok(())
614            }
615        }
616    };
617}
618
619sha2_32!(
620    Sha224,
621    "sha2-224",
622    "SHA-224",
623    28,
624    [
625        0xc1059ed8, 0x367cd507, 0x3070dd17, 0xf70e5939, 0xffc00b31, 0x68581511, 0x64f98fa7,
626        0xbefa4fa4
627    ],
628    "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
629);
630
631sha2_32!(
632    Sha256,
633    "sha2-256",
634    "SHA-256",
635    32,
636    [
637        0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
638        0x5be0cd19
639    ],
640    "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
641);
642
643sha2_64!(
644    Sha384,
645    "sha2-384",
646    "SHA-384",
647    48,
648    [
649        0xcbbb9d5dc1059ed8, 0x629a292a367cd507, 0x9159015a3070dd17, 0x152fecd8f70e5939,
650        0x67332667ffc00b31, 0x8eb44a8768581511, 0xdb0c2e0d64f98fa7, 0x47b5481dbefa4fa4
651    ],
652    "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
653);
654
655sha2_64!(
656    Sha512,
657    "sha2-512",
658    "SHA-512",
659    64,
660    [
661        0x6a09e667f3bcc908, 0xbb67ae8584caa73b, 0x3c6ef372fe94f82b, 0xa54ff53a5f1d36f1,
662        0x510e527fade682d1, 0x9b05688c2b3e6c1f, 0x1f83d9abfb41bd6b, 0x5be0cd19137e2179
663    ],
664    "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
665);
666
667sha2_64!(
668    Sha512_224,
669    "sha2-512-224",
670    "SHA-512/224",
671    28,
672    [
673        0x8c3d37c819544da2,
674        0x73e1996689dcd4d6,
675        0x1dfab7ae32ff9c82,
676        0x679dd514582f9fcf,
677        0x0f6d2b697bd44da8,
678        0x77e36f7304c48942,
679        0x3f9d85a86a1d36c8,
680        0x1112e6ad91d692a1
681    ],
682    "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
683);
684
685sha2_64!(
686    Sha512_256,
687    "sha2-512-256",
688    "SHA-512/256",
689    32,
690    [
691        0x22312194fc2bf72c,
692        0x9f555fa3c84c64c2,
693        0x2393b86b6f53b151,
694        0x963877195940eabd,
695        0x96283ee2a88effe3,
696        0xbe5e1e2553863992,
697        0x2b0199fc2c85b8aa,
698        0x0eb72ddc81c52ca2
699    ],
700    "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
701);
702
703#[cfg(test)]
704mod tests {
705    use super::*;
706
707    /// The hardware path must agree with the portable one, block for block.
708    ///
709    /// The published vectors above do not establish this. They pass whichever
710    /// path runs, so on a machine with SHA-NI they check the backend and on one
711    /// without they check the fallback -- and either way they cannot notice
712    /// that the two disagree, which is the failure a second implementation
713    /// introduces. This runs both over the same input and compares the states.
714    ///
715    /// It reports which path it took rather than asserting one, because a CPU
716    /// without the instructions is a legitimate machine to run the suite on.
717    /// What it does assert is that the comparison happened when it could.
718    #[cfg(all(target_arch = "x86_64", feature = "std"))]
719    #[test]
720    fn the_sha_ni_backend_agrees_with_the_portable_one() {
721        if !sha_ni() {
722            println!("no SHA-NI on this CPU; the backend was not exercised");
723            return;
724        }
725
726        // Lengths either side of the block boundary, and long enough to run the
727        // message schedule over several blocks.
728        let mut checked = 0;
729        for blocks in [1usize, 2, 3, 4, 7, 16] {
730            let mut data = vec![0u8; blocks * 64];
731            // Not random, but not uniform either: a counter through a couple of
732            // multiplications, so every byte position varies between cases.
733            for (i, b) in data.iter_mut().enumerate() {
734                *b = ((i as u64).wrapping_mul(0x9e37_79b9).rotate_left(7) & 0xff) as u8;
735            }
736
737            // FIPS 180-4 section 5.3.3, the same value the macro below passes.
738            const IV: [u32; 8] = [
739                0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
740                0x5be0cd19,
741            ];
742            let mut portable = Core256::new(IV);
743            for block in data.chunks_exact(64) {
744                portable.compress(block);
745            }
746
747            let mut hardware = Core256::new(IV);
748            // SAFETY: guarded by the `sha_ni()` check above.
749            unsafe { x86::compress(&mut hardware.h, &data) };
750
751            assert_eq!(
752                portable.h, hardware.h,
753                "SHA-NI and portable disagree after {blocks} blocks"
754            );
755            checked += 1;
756        }
757        assert_eq!(checked, 6, "the comparison did not run");
758    }
759
760    /// Say which path this build will take, so a benchmark or a vector run is
761    /// not silently measuring the fallback.
762    #[cfg(all(target_arch = "x86_64", feature = "std"))]
763    #[test]
764    fn the_active_sha256_path_is_reported() {
765        println!(
766            "sha-256 backend: {}",
767            if sha_ni() { "SHA-NI" } else { "portable" }
768        );
769    }
770
771    #[test]
772    fn nist_abc_vectors() {
773        assert_eq!(
774            ic_core::codec::hex(Sha256::digest(b"abc").as_ref()),
775            "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
776        );
777        assert_eq!(
778            ic_core::codec::hex(Sha224::digest(b"abc").as_ref()),
779            "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
780        );
781        assert_eq!(
782            ic_core::codec::hex(Sha512::digest(b"abc").as_ref()),
783            "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
784        );
785        assert_eq!(
786            ic_core::codec::hex(Sha384::digest(b"abc").as_ref()),
787            "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
788        );
789    }
790
791    #[test]
792    fn empty_input_vectors() {
793        assert_eq!(
794            ic_core::codec::hex(Sha256::digest(b"").as_ref()),
795            "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
796        );
797        assert_eq!(
798            ic_core::codec::hex(Sha512::digest(b"").as_ref()),
799            "cf83e1357eefb8bdf1542850d66d8007d620e4050b5715dc83f4a921d36ce9ce47d0d13c5d85f2b0ff8318d2877eec2f63b931bd47417a81a538327af927da3e"
800        );
801    }
802
803    /// The 448-bit boundary case: input length forces an extra padding block.
804    #[test]
805    fn two_block_vector() {
806        let msg = b"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq";
807        assert_eq!(
808            ic_core::codec::hex(Sha256::digest(msg).as_ref()),
809            "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"
810        );
811    }
812
813    #[test]
814    fn million_a_vector() {
815        let mut h = Sha256::new();
816        let chunk = [b'a'; 1000];
817        for _ in 0..1000 {
818            h.update(&chunk);
819        }
820        assert_eq!(
821            ic_core::codec::hex(h.finalize().as_ref()),
822            "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"
823        );
824    }
825
826    #[test]
827    fn streaming_matches_one_shot() {
828        let data: [u8; 300] = core::array::from_fn(|i| i as u8);
829        for split in [0usize, 1, 63, 64, 65, 127, 128, 200, 300] {
830            let mut h = Sha512::new();
831            h.update(&data[..split]);
832            h.update(&data[split..]);
833            assert_eq!(h.finalize(), Sha512::digest(&data), "split at {split}");
834        }
835    }
836
837    #[test]
838    fn truncated_variants() {
839        assert_eq!(
840            ic_core::codec::hex(Sha512_224::digest(b"abc").as_ref()),
841            "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
842        );
843        assert_eq!(
844            ic_core::codec::hex(Sha512_256::digest(b"abc").as_ref()),
845            "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
846        );
847    }
848
849    #[test]
850    fn all_self_tests_pass() {
851        Sha224::self_test().unwrap();
852        Sha256::self_test().unwrap();
853        Sha384::self_test().unwrap();
854        Sha512::self_test().unwrap();
855        Sha512_224::self_test().unwrap();
856        Sha512_256::self_test().unwrap();
857    }
858
859    /// What the schedule wipe costs SHA-512, measured in one process.
860    ///
861    /// Ignored: a measurement. Run it with
862    /// `cargo test -p ic-hash --release -- --ignored --nocapture what_the_schedule_wipe_costs`.
863    ///
864    /// Both variants are timed in the same binary, alternating, because this
865    /// machine has other work on it: an attempt to compare across two benchmark
866    /// runs had RustCrypto's own SHA-512 moving 700 -> 1087 MiB/s between them,
867    /// untouched, which is larger than the effect being looked for.
868    #[test]
869    #[ignore = "diagnostic, not a test"]
870    fn what_the_schedule_wipe_costs() {
871        use std::time::Instant;
872
873        // A copy of Core512::compress with the wipe left out, and nothing else
874        // changed. Only for this measurement.
875        fn compress_unwiped(h: &mut [u64; 8], block: &[u8]) {
876            let mut w = [0u64; 80];
877            for i in 0..16 {
878                let mut b = [0u8; 8];
879                b.copy_from_slice(&block[i * 8..i * 8 + 8]);
880                w[i] = u64::from_be_bytes(b);
881            }
882            for i in 16..80 {
883                let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
884                let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
885                w[i] = w[i - 16]
886                    .wrapping_add(s0)
887                    .wrapping_add(w[i - 7])
888                    .wrapping_add(s1);
889            }
890            let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
891            for i in 0..80 {
892                let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
893                let ch = (e & f) ^ ((!e) & g);
894                let t1 = hh
895                    .wrapping_add(s1)
896                    .wrapping_add(ch)
897                    .wrapping_add(K512[i])
898                    .wrapping_add(w[i]);
899                let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
900                let maj = (a & b) ^ (a & c) ^ (b & c);
901                let t2 = s0.wrapping_add(maj);
902                hh = g;
903                g = f;
904                f = e;
905                e = d.wrapping_add(t1);
906                d = c;
907                c = b;
908                b = a;
909                a = t1.wrapping_add(t2);
910            }
911            let upd = [a, b, c, d, e, f, g, hh];
912            for i in 0..8 {
913                h[i] = h[i].wrapping_add(upd[i]);
914            }
915        }
916
917        let block: Vec<u8> = (0..128u32).map(|i| (i * 7 + 1) as u8).collect();
918        let n = 50_000;
919        let (mut best_wiped, mut best_plain) = (f64::INFINITY, f64::INFINITY);
920
921        for _ in 0..30 {
922            let mut core = Core512::new([1, 2, 3, 4, 5, 6, 7, 8]);
923            let t = Instant::now();
924            for _ in 0..n {
925                core.compress(core::hint::black_box(&block));
926            }
927            best_wiped = best_wiped.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
928
929            let mut h = [1u64, 2, 3, 4, 5, 6, 7, 8];
930            let t = Instant::now();
931            for _ in 0..n {
932                compress_unwiped(&mut h, core::hint::black_box(&block));
933            }
934            best_plain = best_plain.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
935        }
936        println!(
937            "
938  sha-512 compress, with w.zeroize()   {best_wiped:>8.1} ns/block"
939        );
940        println!("  sha-512 compress, without            {best_plain:>8.1} ns/block");
941        println!(
942            "  the wipe costs                       {:>8.1} ns/block ({:.0}%)",
943            best_wiped - best_plain,
944            (best_wiped - best_plain) / best_wiped * 100.0
945        );
946    }
947}