Skip to main content

ic_hash/
sha2.rs

1//! FIPS 180-4 SHA-2 family.
2//!
3//! Two cores (32-bit and 64-bit) are shared by six published output variants,
4//! which differ only in their initial hash value and truncation length.
5
6//! Indexed loops over fixed-size limb and word arrays are used throughout; they
7//! mirror the index algebra in the specifications these routines implement, so
8//! `needless_range_loop` is allowed rather than obscuring the correspondence.
9#![allow(clippy::needless_range_loop)]
10
11use ic_core::traits::{Algorithm, Digest, SelfTest};
12use ic_core::{ensure, Result, Zeroize};
13
14// SHA-NI, where the CPU has it. Only under `std`, because the detection does:
15// a `no_std` build has no way to ask, and guessing wrong is an illegal
16// instruction rather than a wrong answer.
17#[cfg(all(target_arch = "x86_64", feature = "std"))]
18mod x86;
19
20// SHA-512's schedule, four words at a time. Same reasoning as `x86`: only
21// under `std`, because the detection needs it.
22#[cfg(all(target_arch = "x86_64", feature = "std"))]
23mod avx2_512;
24
25/// Whether this CPU has AVX2, asked once.
26#[cfg(all(target_arch = "x86_64", feature = "std"))]
27fn avx2() -> bool {
28    use core::sync::atomic::{AtomicU8, Ordering};
29    // 0 not yet asked, 1 yes, 2 no.
30    static CACHED: AtomicU8 = AtomicU8::new(0);
31    match CACHED.load(Ordering::Relaxed) {
32        1 => true,
33        2 => false,
34        _ => {
35            let have = std::is_x86_feature_detected!("avx2");
36            CACHED.store(u8::from(!have) + 1, Ordering::Relaxed);
37            have
38        }
39    }
40}
41
42/// Whether this CPU has the instructions [`x86::compress`] needs.
43///
44/// Asked once. `is_x86_feature_detected!` is not free, and SHA-256 is called
45/// often enough on small inputs that paying for the query per block would show
46/// up in exactly the workloads this is meant to help.
47#[cfg(all(target_arch = "x86_64", feature = "std"))]
48fn sha_ni() -> bool {
49    use core::sync::atomic::{AtomicU8, Ordering};
50    // 0 not yet asked, 1 yes, 2 no.
51    static CACHED: AtomicU8 = AtomicU8::new(0);
52    match CACHED.load(Ordering::Relaxed) {
53        1 => true,
54        2 => false,
55        _ => {
56            let have = std::is_x86_feature_detected!("sha")
57                && std::is_x86_feature_detected!("sse2")
58                && std::is_x86_feature_detected!("ssse3")
59                && std::is_x86_feature_detected!("sse4.1");
60            CACHED.store(u8::from(!have) + 1, Ordering::Relaxed);
61            have
62        }
63    }
64}
65
66const K256: [u32; 64] = [
67    0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4, 0xab1c5ed5,
68    0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174,
69    0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
70    0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967,
71    0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85,
72    0xa2bfe8a1, 0xa81a664b, 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
73    0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
74    0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
75];
76
77const K512: [u64; 80] = [
78    0x428a2f98d728ae22,
79    0x7137449123ef65cd,
80    0xb5c0fbcfec4d3b2f,
81    0xe9b5dba58189dbbc,
82    0x3956c25bf348b538,
83    0x59f111f1b605d019,
84    0x923f82a4af194f9b,
85    0xab1c5ed5da6d8118,
86    0xd807aa98a3030242,
87    0x12835b0145706fbe,
88    0x243185be4ee4b28c,
89    0x550c7dc3d5ffb4e2,
90    0x72be5d74f27b896f,
91    0x80deb1fe3b1696b1,
92    0x9bdc06a725c71235,
93    0xc19bf174cf692694,
94    0xe49b69c19ef14ad2,
95    0xefbe4786384f25e3,
96    0x0fc19dc68b8cd5b5,
97    0x240ca1cc77ac9c65,
98    0x2de92c6f592b0275,
99    0x4a7484aa6ea6e483,
100    0x5cb0a9dcbd41fbd4,
101    0x76f988da831153b5,
102    0x983e5152ee66dfab,
103    0xa831c66d2db43210,
104    0xb00327c898fb213f,
105    0xbf597fc7beef0ee4,
106    0xc6e00bf33da88fc2,
107    0xd5a79147930aa725,
108    0x06ca6351e003826f,
109    0x142929670a0e6e70,
110    0x27b70a8546d22ffc,
111    0x2e1b21385c26c926,
112    0x4d2c6dfc5ac42aed,
113    0x53380d139d95b3df,
114    0x650a73548baf63de,
115    0x766a0abb3c77b2a8,
116    0x81c2c92e47edaee6,
117    0x92722c851482353b,
118    0xa2bfe8a14cf10364,
119    0xa81a664bbc423001,
120    0xc24b8b70d0f89791,
121    0xc76c51a30654be30,
122    0xd192e819d6ef5218,
123    0xd69906245565a910,
124    0xf40e35855771202a,
125    0x106aa07032bbd1b8,
126    0x19a4c116b8d2d0c8,
127    0x1e376c085141ab53,
128    0x2748774cdf8eeb99,
129    0x34b0bcb5e19b48a8,
130    0x391c0cb3c5c95a63,
131    0x4ed8aa4ae3418acb,
132    0x5b9cca4f7763e373,
133    0x682e6ff3d6b2b8a3,
134    0x748f82ee5defb2fc,
135    0x78a5636f43172f60,
136    0x84c87814a1f0ab72,
137    0x8cc702081a6439ec,
138    0x90befffa23631e28,
139    0xa4506cebde82bde9,
140    0xbef9a3f7b2c67915,
141    0xc67178f2e372532b,
142    0xca273eceea26619c,
143    0xd186b8c721c0c207,
144    0xeada7dd6cde0eb1e,
145    0xf57d4f7fee6ed178,
146    0x06f067aa72176fba,
147    0x0a637dc5a2c898a6,
148    0x113f9804bef90dae,
149    0x1b710b35131c471b,
150    0x28db77f523047d84,
151    0x32caab7b40c72493,
152    0x3c9ebe0a15c9bebc,
153    0x431d67c49c100d4c,
154    0x4cc5d4becb3e42b6,
155    0x597f299cfc657e2a,
156    0x5fcb6fab3ad6faec,
157    0x6c44198c4a475817,
158];
159
160/// The shared 32-bit SHA-2 compression core (SHA-224 / SHA-256).
161#[derive(Clone)]
162struct Core256 {
163    h: [u32; 8],
164    buf: [u8; 64],
165    buffered: usize,
166    len: u64,
167}
168
169impl Drop for Core256 {
170    /// Wipe the chaining state and the buffered block.
171    ///
172    /// A hash is not a secret, but this state is not only used for hashing:
173    /// `Hmac<D>` holds two of these with the key already absorbed into them,
174    /// so the ipad and opad states are key-derived material. Putting the wipe
175    /// here rather than on `Hmac` means every consumer inherits it through
176    /// ordinary field drop, with no `Zeroize` bound threaded through the
177    /// `Digest` trait and no chance of a new wrapper forgetting.
178    fn drop(&mut self) {
179        self.h.zeroize();
180        self.buf.zeroize();
181        self.buffered = 0;
182        self.len = 0;
183    }
184}
185
186impl Core256 {
187    const fn new(iv: [u32; 8]) -> Self {
188        Self {
189            h: iv,
190            buf: [0u8; 64],
191            buffered: 0,
192            len: 0,
193        }
194    }
195
196    fn compress(&mut self, block: &[u8]) {
197        let mut w = [0u32; 64];
198        for i in 0..16 {
199            w[i] = u32::from_be_bytes([
200                block[i * 4],
201                block[i * 4 + 1],
202                block[i * 4 + 2],
203                block[i * 4 + 3],
204            ]);
205        }
206        for i in 16..64 {
207            let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
208            let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
209            w[i] = w[i - 16]
210                .wrapping_add(s0)
211                .wrapping_add(w[i - 7])
212                .wrapping_add(s1);
213        }
214        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = self.h;
215        for i in 0..64 {
216            let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
217            let ch = (e & f) ^ ((!e) & g);
218            let t1 = hh
219                .wrapping_add(s1)
220                .wrapping_add(ch)
221                .wrapping_add(K256[i])
222                .wrapping_add(w[i]);
223            let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
224            let maj = (a & b) ^ (a & c) ^ (b & c);
225            let t2 = s0.wrapping_add(maj);
226            hh = g;
227            g = f;
228            f = e;
229            e = d.wrapping_add(t1);
230            d = c;
231            c = b;
232            b = a;
233            a = t1.wrapping_add(t2);
234        }
235        let upd = [a, b, c, d, e, f, g, hh];
236        for i in 0..8 {
237            self.h[i] = self.h[i].wrapping_add(upd[i]);
238        }
239        w.zeroize();
240    }
241
242    /// Compress a whole number of blocks, using the hardware path when there
243    /// is one.
244    ///
245    /// Taking a run rather than a block at a time is the point: the SHA-NI
246    /// backend shuffles the state into and out of its register layout once per
247    /// call, so feeding it one block at a time would pay that on every block.
248    fn compress_blocks(&mut self, data: &[u8]) {
249        debug_assert!(data.len() % 64 == 0);
250        if data.is_empty() {
251            return;
252        }
253        #[cfg(all(target_arch = "x86_64", feature = "std"))]
254        if sha_ni() {
255            // SAFETY: `sha_ni()` is exactly the feature test this requires, and
256            // the length is a multiple of the block size by the assertion above.
257            unsafe { x86::compress(&mut self.h, data) };
258            return;
259        }
260        for block in data.chunks_exact(64) {
261            self.compress(block);
262        }
263    }
264
265    fn update(&mut self, mut data: &[u8]) {
266        self.len = self.len.wrapping_add(data.len() as u64);
267        if self.buffered > 0 {
268            let need = 64 - self.buffered;
269            let take = core::cmp::min(need, data.len());
270            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
271            self.buffered += take;
272            data = &data[take..];
273            if self.buffered < 64 {
274                // The whole input fit in the partial block; nothing to compress.
275                return;
276            }
277            let block = self.buf;
278            self.compress(&block);
279            self.buffered = 0;
280        }
281        let whole = data.len() - data.len() % 64;
282        self.compress_blocks(&data[..whole]);
283        let rest = &data[whole..];
284        self.buf[..rest.len()].copy_from_slice(rest);
285        self.buffered = rest.len();
286    }
287
288    fn finalize(mut self) -> [u32; 8] {
289        let bit_len = self.len.wrapping_mul(8);
290        let mut pad = [0u8; 72];
291        pad[0] = 0x80;
292        // Pad so that (buffered + 1 + zeros) % 64 == 56.
293        let zeros = (55 + 64 - (self.buffered % 64)) % 64;
294        pad[1 + zeros..1 + zeros + 8].copy_from_slice(&bit_len.to_be_bytes());
295        self.update_no_count(&pad[..1 + zeros + 8]);
296        let out = self.h;
297        self.buf.zeroize();
298        self.h.zeroize();
299        out
300    }
301
302    /// Absorb padding without disturbing the message-length counter.
303    fn update_no_count(&mut self, data: &[u8]) {
304        let saved = self.len;
305        self.update(data);
306        self.len = saved;
307    }
308}
309
310/// The shared 64-bit SHA-2 compression core (SHA-384 / SHA-512 / SHA-512-t).
311#[derive(Clone)]
312struct Core512 {
313    h: [u64; 8],
314    buf: [u8; 128],
315    buffered: usize,
316    len: u128,
317    /// The message schedule, kept here rather than built on the stack.
318    ///
319    /// It lives in the struct so that wiping it costs once per hash instead of
320    /// once per block. `Zeroize` writes element by element through
321    /// `write_volatile`, which is what makes the wipe non-elidable and also
322    /// what makes it expensive: eighty volatile stores cannot be merged into a
323    /// `memset` or vectorised, and measured in isolation they were 159ns of a
324    /// 258ns block -- 62% of SHA-512's compression spent clearing the schedule
325    /// rather than computing it. `what_the_schedule_wipe_costs` is that
326    /// measurement.
327    ///
328    /// The schedule is still wiped, in `finalize`, beside `h` and `buf`. What
329    /// changes is how often. Wiping after every block bought nothing that
330    /// survived the block anyway: the next block immediately overwrites the
331    /// whole array, and the material it is derived from is in the caller's
332    /// input buffer, which this library neither owns nor clears.
333    w: [u64; 80],
334}
335
336impl Drop for Core512 {
337    /// Wipe the chaining state and the buffered block. See [`Core256`].
338    fn drop(&mut self) {
339        self.h.zeroize();
340        self.buf.zeroize();
341        self.buffered = 0;
342        self.len = 0;
343    }
344}
345
346/// SHA-512's compression, and what has already been tried on it.
347///
348/// It runs about 1.5 times behind RustCrypto's, which has an AVX2 backend for
349/// the message schedule. There is no SHA-512 instruction on x86 the way there
350/// is for SHA-256, so the portable path below is what runs.
351///
352/// What closed the gap from 1.76x was not an optimisation of the arithmetic at
353/// all: 62% of the compression was the per-block `zeroize` of the schedule,
354/// whose volatile writes cannot be merged into a `memset`. The schedule now
355/// lives in the struct and is wiped once per hash. See the note on `w`.
356///
357/// The remaining gap is not scalar slack, and it is worth saying why before
358/// anyone looks for some. At 715 MiB/s this compresses a block in about 180ns,
359/// which on this machine is roughly 2900 instructions in 600 cycles: close to
360/// five per cycle, near what a four-wide core can retire. RustCrypto's 1070
361/// MiB/s would need better than seven per cycle for the same instruction count,
362/// which is not possible -- so they are executing fewer instructions, not
363/// scheduling the same ones better. `sha2 0.10.9` has an AVX2 SHA-512 backend
364/// in `sha512/x86.rs`, selected by runtime detection; that is the difference.
365/// Matching it means writing one, not tuning this.
366///
367/// Two source-level optimisations were measured and reverted, and are recorded
368/// so they are not tried a third time:
369///
370/// - **A rolling sixteen-word schedule window** instead of the eighty-word
371///   array. The array is 640 bytes cleared per 128-byte block, five bytes wiped
372///   per byte hashed, which looks like the cost. A controlled A/B showed it
373///   *slower*: 640 bytes sits in L1, and the modulo indexing defeats whatever
374///   unrolling the flat array was getting.
375/// - **Unrolling the round loop by eight**, naming the working variables in
376///   rotation so the eight moves per round disappear. No measurable change in
377///   either direction; LLVM already renames and unrolls this shape.
378///
379/// Both failed the same way: the waste was visible in the source and absent
380/// from the object code. What did pay elsewhere in this workspace was work the
381/// compiler cannot do -- breaking a serial dependency chain, selecting a
382/// hardware instruction, changing the algorithm. The remaining gap here is the
383/// vectorised schedule, and even that addresses only the third or so of the
384/// work the schedule represents, since the rounds are inherently serial.
385impl Core512 {
386    const fn new(iv: [u64; 8]) -> Self {
387        Self {
388            h: iv,
389            buf: [0u8; 128],
390            buffered: 0,
391            len: 0,
392            w: [0u64; 80],
393        }
394    }
395
396    fn compress(&mut self, block: &[u8]) {
397        // The schedule is about a third of the work and is the only part with
398        // anything to run in parallel; the rounds are a chain. So the backends
399        // differ in how `w` is filled and share everything after it.
400        #[cfg(all(target_arch = "x86_64", feature = "std"))]
401        if avx2() {
402            // SAFETY: `avx2()` is the feature test this requires, and `block`
403            // is one block by `update`'s chunking.
404            unsafe { avx2_512::compress(&mut self.h, &mut self.w, block) };
405            return;
406        }
407        Self::schedule(&mut self.w, block);
408        Self::rounds(&mut self.h, &self.w);
409    }
410
411    /// Build the message schedule.
412    ///
413    /// The AVX2 backend does not call this -- it interleaves the same steps
414    /// into its round loop, which is the whole reason it is faster -- but it is
415    /// checked against that backend word for word.
416    fn schedule(w: &mut [u64; 80], block: &[u8]) {
417        for i in 0..16 {
418            let mut b = [0u8; 8];
419            b.copy_from_slice(&block[i * 8..i * 8 + 8]);
420            w[i] = u64::from_be_bytes(b);
421        }
422        for i in 16..80 {
423            let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
424            let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
425            w[i] = w[i - 16]
426                .wrapping_add(s0)
427                .wrapping_add(w[i - 7])
428                .wrapping_add(s1);
429        }
430    }
431
432    /// The eighty rounds.
433    fn rounds(h: &mut [u64; 8], w: &[u64; 80]) {
434        let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
435        for i in 0..80 {
436            let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
437            let ch = (e & f) ^ ((!e) & g);
438            let t1 = hh
439                .wrapping_add(s1)
440                .wrapping_add(ch)
441                .wrapping_add(K512[i])
442                .wrapping_add(w[i]);
443            let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
444            let maj = (a & b) ^ (a & c) ^ (b & c);
445            let t2 = s0.wrapping_add(maj);
446            hh = g;
447            g = f;
448            f = e;
449            e = d.wrapping_add(t1);
450            d = c;
451            c = b;
452            b = a;
453            a = t1.wrapping_add(t2);
454        }
455        let upd = [a, b, c, d, e, f, g, hh];
456        for i in 0..8 {
457            h[i] = h[i].wrapping_add(upd[i]);
458        }
459    }
460
461    fn update(&mut self, mut data: &[u8]) {
462        self.len = self.len.wrapping_add(data.len() as u128);
463        if self.buffered > 0 {
464            let need = 128 - self.buffered;
465            let take = core::cmp::min(need, data.len());
466            self.buf[self.buffered..self.buffered + take].copy_from_slice(&data[..take]);
467            self.buffered += take;
468            data = &data[take..];
469            if self.buffered < 128 {
470                // The whole input fit in the partial block; nothing to compress.
471                return;
472            }
473            let block = self.buf;
474            self.compress(&block);
475            self.buffered = 0;
476        }
477        let mut chunks = data.chunks_exact(128);
478        for block in &mut chunks {
479            self.compress(block);
480        }
481        let rest = chunks.remainder();
482        self.buf[..rest.len()].copy_from_slice(rest);
483        self.buffered = rest.len();
484    }
485
486    fn finalize(mut self) -> [u64; 8] {
487        let bit_len = self.len.wrapping_mul(8);
488        let mut pad = [0u8; 145];
489        pad[0] = 0x80;
490        let zeros = (111 + 128 - (self.buffered % 128)) % 128;
491        pad[1 + zeros..1 + zeros + 16].copy_from_slice(&bit_len.to_be_bytes());
492        let saved = self.len;
493        self.update(&pad[..1 + zeros + 16]);
494        self.len = saved;
495        let out = self.h;
496        self.buf.zeroize();
497        self.h.zeroize();
498        self.w.zeroize();
499        out
500    }
501}
502
503macro_rules! sha2_32 {
504    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
505        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
506        #[derive(Clone)]
507        pub struct $name(Core256);
508
509        impl Default for $name {
510            fn default() -> Self {
511                Self(Core256::new($iv))
512            }
513        }
514
515        impl Algorithm for $name {
516            const ID: &'static str = $id;
517            const NAME: &'static str = $disp;
518        }
519
520        impl Digest for $name {
521            type Output = [u8; $out];
522            const OUTPUT_LEN: usize = $out;
523            const BLOCK_LEN: usize = 64;
524
525            fn update(&mut self, data: &[u8]) {
526                self.0.update(data);
527            }
528
529            fn finalize(self) -> Self::Output {
530                let h = self.0.finalize();
531                let mut full = [0u8; 32];
532                for i in 0..8 {
533                    full[i * 4..i * 4 + 4].copy_from_slice(&h[i].to_be_bytes());
534                }
535                let mut out = [0u8; $out];
536                out.copy_from_slice(&full[..$out]);
537                out
538            }
539        }
540
541        impl SelfTest for $name {
542            fn self_test() -> Result<()> {
543                let got = <Self as Digest>::digest(b"abc");
544                let mut want = [0u8; $out];
545                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
546                ensure!(
547                    ic_core::ct::verify(&want, got.as_ref()),
548                    SelfTestFailed,
549                    $id
550                );
551                Ok(())
552            }
553        }
554    };
555}
556
557macro_rules! sha2_64 {
558    ($name:ident, $id:literal, $disp:literal, $out:literal, $iv:expr, $kat:literal) => {
559        #[doc = concat!("FIPS 180-4 ", $disp, ".")]
560        #[derive(Clone)]
561        pub struct $name(Core512);
562
563        impl Default for $name {
564            fn default() -> Self {
565                Self(Core512::new($iv))
566            }
567        }
568
569        impl Algorithm for $name {
570            const ID: &'static str = $id;
571            const NAME: &'static str = $disp;
572        }
573
574        impl Digest for $name {
575            type Output = [u8; $out];
576            const OUTPUT_LEN: usize = $out;
577            const BLOCK_LEN: usize = 128;
578
579            fn update(&mut self, data: &[u8]) {
580                self.0.update(data);
581            }
582
583            fn finalize(self) -> Self::Output {
584                let h = self.0.finalize();
585                let mut full = [0u8; 64];
586                for i in 0..8 {
587                    full[i * 8..i * 8 + 8].copy_from_slice(&h[i].to_be_bytes());
588                }
589                let mut out = [0u8; $out];
590                out.copy_from_slice(&full[..$out]);
591                out
592            }
593        }
594
595        impl SelfTest for $name {
596            fn self_test() -> Result<()> {
597                let got = <Self as Digest>::digest(b"abc");
598                let mut want = [0u8; $out];
599                ic_core::codec::hex_decode($kat.as_bytes(), &mut want)?;
600                ensure!(
601                    ic_core::ct::verify(&want, got.as_ref()),
602                    SelfTestFailed,
603                    $id
604                );
605                Ok(())
606            }
607        }
608    };
609}
610
611sha2_32!(
612    Sha224,
613    "sha2-224",
614    "SHA-224",
615    28,
616    [
617        0xc1059ed8, 0x367cd507, 0x3070dd17, 0xf70e5939, 0xffc00b31, 0x68581511, 0x64f98fa7,
618        0xbefa4fa4
619    ],
620    "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
621);
622
623sha2_32!(
624    Sha256,
625    "sha2-256",
626    "SHA-256",
627    32,
628    [
629        0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
630        0x5be0cd19
631    ],
632    "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
633);
634
635sha2_64!(
636    Sha384,
637    "sha2-384",
638    "SHA-384",
639    48,
640    [
641        0xcbbb9d5dc1059ed8, 0x629a292a367cd507, 0x9159015a3070dd17, 0x152fecd8f70e5939,
642        0x67332667ffc00b31, 0x8eb44a8768581511, 0xdb0c2e0d64f98fa7, 0x47b5481dbefa4fa4
643    ],
644    "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
645);
646
647sha2_64!(
648    Sha512,
649    "sha2-512",
650    "SHA-512",
651    64,
652    [
653        0x6a09e667f3bcc908, 0xbb67ae8584caa73b, 0x3c6ef372fe94f82b, 0xa54ff53a5f1d36f1,
654        0x510e527fade682d1, 0x9b05688c2b3e6c1f, 0x1f83d9abfb41bd6b, 0x5be0cd19137e2179
655    ],
656    "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
657);
658
659sha2_64!(
660    Sha512_224,
661    "sha2-512-224",
662    "SHA-512/224",
663    28,
664    [
665        0x8c3d37c819544da2,
666        0x73e1996689dcd4d6,
667        0x1dfab7ae32ff9c82,
668        0x679dd514582f9fcf,
669        0x0f6d2b697bd44da8,
670        0x77e36f7304c48942,
671        0x3f9d85a86a1d36c8,
672        0x1112e6ad91d692a1
673    ],
674    "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
675);
676
677sha2_64!(
678    Sha512_256,
679    "sha2-512-256",
680    "SHA-512/256",
681    32,
682    [
683        0x22312194fc2bf72c,
684        0x9f555fa3c84c64c2,
685        0x2393b86b6f53b151,
686        0x963877195940eabd,
687        0x96283ee2a88effe3,
688        0xbe5e1e2553863992,
689        0x2b0199fc2c85b8aa,
690        0x0eb72ddc81c52ca2
691    ],
692    "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
693);
694
695#[cfg(test)]
696mod tests {
697    use super::*;
698
699    /// The hardware path must agree with the portable one, block for block.
700    ///
701    /// The published vectors above do not establish this. They pass whichever
702    /// path runs, so on a machine with SHA-NI they check the backend and on one
703    /// without they check the fallback -- and either way they cannot notice
704    /// that the two disagree, which is the failure a second implementation
705    /// introduces. This runs both over the same input and compares the states.
706    ///
707    /// It reports which path it took rather than asserting one, because a CPU
708    /// without the instructions is a legitimate machine to run the suite on.
709    /// What it does assert is that the comparison happened when it could.
710    #[cfg(all(target_arch = "x86_64", feature = "std"))]
711    #[test]
712    fn the_sha_ni_backend_agrees_with_the_portable_one() {
713        if !sha_ni() {
714            println!("no SHA-NI on this CPU; the backend was not exercised");
715            return;
716        }
717
718        // Lengths either side of the block boundary, and long enough to run the
719        // message schedule over several blocks.
720        let mut checked = 0;
721        for blocks in [1usize, 2, 3, 4, 7, 16] {
722            let mut data = vec![0u8; blocks * 64];
723            // Not random, but not uniform either: a counter through a couple of
724            // multiplications, so every byte position varies between cases.
725            for (i, b) in data.iter_mut().enumerate() {
726                *b = ((i as u64).wrapping_mul(0x9e37_79b9).rotate_left(7) & 0xff) as u8;
727            }
728
729            // FIPS 180-4 section 5.3.3, the same value the macro below passes.
730            const IV: [u32; 8] = [
731                0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
732                0x5be0cd19,
733            ];
734            let mut portable = Core256::new(IV);
735            for block in data.chunks_exact(64) {
736                portable.compress(block);
737            }
738
739            let mut hardware = Core256::new(IV);
740            // SAFETY: guarded by the `sha_ni()` check above.
741            unsafe { x86::compress(&mut hardware.h, &data) };
742
743            assert_eq!(
744                portable.h, hardware.h,
745                "SHA-NI and portable disagree after {blocks} blocks"
746            );
747            checked += 1;
748        }
749        assert_eq!(checked, 6, "the comparison did not run");
750    }
751
752    /// Say which path this build will take, so a benchmark or a vector run is
753    /// not silently measuring the fallback.
754    #[cfg(all(target_arch = "x86_64", feature = "std"))]
755    #[test]
756    fn the_active_sha256_path_is_reported() {
757        println!(
758            "sha-256 backend: {}",
759            if sha_ni() { "SHA-NI" } else { "portable" }
760        );
761    }
762
763    #[test]
764    fn nist_abc_vectors() {
765        assert_eq!(
766            ic_core::codec::hex(Sha256::digest(b"abc").as_ref()),
767            "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
768        );
769        assert_eq!(
770            ic_core::codec::hex(Sha224::digest(b"abc").as_ref()),
771            "23097d223405d8228642a477bda255b32aadbce4bda0b3f7e36c9da7"
772        );
773        assert_eq!(
774            ic_core::codec::hex(Sha512::digest(b"abc").as_ref()),
775            "ddaf35a193617abacc417349ae20413112e6fa4e89a97ea20a9eeee64b55d39a2192992a274fc1a836ba3c23a3feebbd454d4423643ce80e2a9ac94fa54ca49f"
776        );
777        assert_eq!(
778            ic_core::codec::hex(Sha384::digest(b"abc").as_ref()),
779            "cb00753f45a35e8bb5a03d699ac65007272c32ab0eded1631a8b605a43ff5bed8086072ba1e7cc2358baeca134c825a7"
780        );
781    }
782
783    #[test]
784    fn empty_input_vectors() {
785        assert_eq!(
786            ic_core::codec::hex(Sha256::digest(b"").as_ref()),
787            "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
788        );
789        assert_eq!(
790            ic_core::codec::hex(Sha512::digest(b"").as_ref()),
791            "cf83e1357eefb8bdf1542850d66d8007d620e4050b5715dc83f4a921d36ce9ce47d0d13c5d85f2b0ff8318d2877eec2f63b931bd47417a81a538327af927da3e"
792        );
793    }
794
795    /// The 448-bit boundary case: input length forces an extra padding block.
796    #[test]
797    fn two_block_vector() {
798        let msg = b"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq";
799        assert_eq!(
800            ic_core::codec::hex(Sha256::digest(msg).as_ref()),
801            "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"
802        );
803    }
804
805    #[test]
806    fn million_a_vector() {
807        let mut h = Sha256::new();
808        let chunk = [b'a'; 1000];
809        for _ in 0..1000 {
810            h.update(&chunk);
811        }
812        assert_eq!(
813            ic_core::codec::hex(h.finalize().as_ref()),
814            "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"
815        );
816    }
817
818    #[test]
819    fn streaming_matches_one_shot() {
820        let data: [u8; 300] = core::array::from_fn(|i| i as u8);
821        for split in [0usize, 1, 63, 64, 65, 127, 128, 200, 300] {
822            let mut h = Sha512::new();
823            h.update(&data[..split]);
824            h.update(&data[split..]);
825            assert_eq!(h.finalize(), Sha512::digest(&data), "split at {split}");
826        }
827    }
828
829    #[test]
830    fn truncated_variants() {
831        assert_eq!(
832            ic_core::codec::hex(Sha512_224::digest(b"abc").as_ref()),
833            "4634270f707b6a54daae7530460842e20e37ed265ceee9a43e8924aa"
834        );
835        assert_eq!(
836            ic_core::codec::hex(Sha512_256::digest(b"abc").as_ref()),
837            "53048e2681941ef99b2e29b76b4c7dabe4c2d0c634fc6d46e0e2f13107e7af23"
838        );
839    }
840
841    #[test]
842    fn all_self_tests_pass() {
843        Sha224::self_test().unwrap();
844        Sha256::self_test().unwrap();
845        Sha384::self_test().unwrap();
846        Sha512::self_test().unwrap();
847        Sha512_224::self_test().unwrap();
848        Sha512_256::self_test().unwrap();
849    }
850
851    /// What the schedule wipe costs SHA-512, measured in one process.
852    ///
853    /// Ignored: a measurement. Run it with
854    /// `cargo test -p ic-hash --release -- --ignored --nocapture what_the_schedule_wipe_costs`.
855    ///
856    /// Both variants are timed in the same binary, alternating, because this
857    /// machine has other work on it: an attempt to compare across two benchmark
858    /// runs had RustCrypto's own SHA-512 moving 700 -> 1087 MiB/s between them,
859    /// untouched, which is larger than the effect being looked for.
860    #[test]
861    #[ignore = "diagnostic, not a test"]
862    fn what_the_schedule_wipe_costs() {
863        use std::time::Instant;
864
865        // A copy of Core512::compress with the wipe left out, and nothing else
866        // changed. Only for this measurement.
867        fn compress_unwiped(h: &mut [u64; 8], block: &[u8]) {
868            let mut w = [0u64; 80];
869            for i in 0..16 {
870                let mut b = [0u8; 8];
871                b.copy_from_slice(&block[i * 8..i * 8 + 8]);
872                w[i] = u64::from_be_bytes(b);
873            }
874            for i in 16..80 {
875                let s0 = w[i - 15].rotate_right(1) ^ w[i - 15].rotate_right(8) ^ (w[i - 15] >> 7);
876                let s1 = w[i - 2].rotate_right(19) ^ w[i - 2].rotate_right(61) ^ (w[i - 2] >> 6);
877                w[i] = w[i - 16]
878                    .wrapping_add(s0)
879                    .wrapping_add(w[i - 7])
880                    .wrapping_add(s1);
881            }
882            let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = *h;
883            for i in 0..80 {
884                let s1 = e.rotate_right(14) ^ e.rotate_right(18) ^ e.rotate_right(41);
885                let ch = (e & f) ^ ((!e) & g);
886                let t1 = hh
887                    .wrapping_add(s1)
888                    .wrapping_add(ch)
889                    .wrapping_add(K512[i])
890                    .wrapping_add(w[i]);
891                let s0 = a.rotate_right(28) ^ a.rotate_right(34) ^ a.rotate_right(39);
892                let maj = (a & b) ^ (a & c) ^ (b & c);
893                let t2 = s0.wrapping_add(maj);
894                hh = g;
895                g = f;
896                f = e;
897                e = d.wrapping_add(t1);
898                d = c;
899                c = b;
900                b = a;
901                a = t1.wrapping_add(t2);
902            }
903            let upd = [a, b, c, d, e, f, g, hh];
904            for i in 0..8 {
905                h[i] = h[i].wrapping_add(upd[i]);
906            }
907        }
908
909        let block: Vec<u8> = (0..128u32).map(|i| (i * 7 + 1) as u8).collect();
910        let n = 50_000;
911        let (mut best_wiped, mut best_plain) = (f64::INFINITY, f64::INFINITY);
912
913        for _ in 0..30 {
914            let mut core = Core512::new([1, 2, 3, 4, 5, 6, 7, 8]);
915            let t = Instant::now();
916            for _ in 0..n {
917                core.compress(core::hint::black_box(&block));
918            }
919            best_wiped = best_wiped.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
920
921            let mut h = [1u64, 2, 3, 4, 5, 6, 7, 8];
922            let t = Instant::now();
923            for _ in 0..n {
924                compress_unwiped(&mut h, core::hint::black_box(&block));
925            }
926            best_plain = best_plain.min(t.elapsed().as_secs_f64() / n as f64 * 1e9);
927        }
928        println!(
929            "
930  sha-512 compress, with w.zeroize()   {best_wiped:>8.1} ns/block"
931        );
932        println!("  sha-512 compress, without            {best_plain:>8.1} ns/block");
933        println!(
934            "  the wipe costs                       {:>8.1} ns/block ({:.0}%)",
935            best_wiped - best_plain,
936            (best_wiped - best_plain) / best_wiped * 100.0
937        );
938    }
939}