Skip to main content

ic_cipher/
gcm.rs

1//! NIST SP 800-38D Galois/Counter Mode.
2//!
3//! GHASH is implemented with a branch-free bit-by-bit multiplication in
4//! GF(2^128). Table-driven GHASH is faster but indexes memory with key-derived
5//! values; the portable backend refuses that trade.
6//!
7//! # Nonce discipline
8//!
9//! Reusing a `(key, nonce)` pair under GCM is catastrophic: it leaks the
10//! authentication subkey and lets an attacker forge arbitrary messages. The
11//! ontology records this as a hard usage constraint
12//! (`nonce_reuse_consequence: "catastrophic"`) so an agent selecting GCM is
13//! told to pair it with a counter or a random 96-bit nonce under a message
14//! limit. See [`GcmLimits`].
15
16//! Indexed loops over fixed-size limb and word arrays are used throughout; they
17//! mirror the index algebra in the specifications these routines implement, so
18//! `needless_range_loop` is allowed rather than obscuring the correspondence.
19#![allow(clippy::needless_range_loop)]
20
21use crate::aes::{Aes128, Aes192, Aes256, BLOCK_LEN};
22use crate::modes::increment_be32;
23use ic_core::traits::{Aead, Algorithm, BlockCipher, SelfTest};
24use ic_core::{ensure, Result, Zeroize};
25
26/// The GF(2^128) reduction constant for GHASH, `x^128 + x^7 + x^2 + x + 1`.
27const R: u8 = 0xe1;
28
29/// Invocation limits an agent must respect for a single GCM key.
30///
31/// From SP 800-38D §8.3 and the AES-GCM analysis behind RFC 8446.
32pub struct GcmLimits;
33
34impl GcmLimits {
35    /// Maximum plaintext bytes in one invocation: `2^39 - 256` bits.
36    pub const MAX_PLAINTEXT_BYTES: u64 = (1 << 36) - 32;
37    /// Maximum invocations under one key with random 96-bit nonces.
38    pub const MAX_RANDOM_NONCE_INVOCATIONS: u64 = 1 << 32;
39    /// Recommended nonce length in bytes; other lengths are legal but slower
40    /// and lose the injectivity guarantee that makes counters safe.
41    pub const RECOMMENDED_NONCE_LEN: usize = 12;
42}
43
44/// Whether GHASH can use the carry-less multiply on this CPU.
45///
46/// Needs `ssse3` for the byte-reversal shuffle as well as `pclmulqdq` for the
47/// multiply itself. Detection lives in `ic-core`, like AES-NI's, so the
48/// ontology reports the same answer this dispatches on.
49#[inline]
50#[must_use]
51pub fn ghash_accelerated() -> bool {
52    ic_core::cpu::has_ghash_clmul()
53}
54
55/// Multiply `x` by `h` in GF(2^128) using the GCM bit ordering, in place.
56///
57/// Exposed within the crate so the accelerated backend can be differentially
58/// tested against it.
59pub(crate) fn portable_ghash_mul(x: &mut [u8; BLOCK_LEN], h: &[u8; BLOCK_LEN]) {
60    let mut z = [0u8; BLOCK_LEN];
61    let mut v = *h;
62    for i in 0..128 {
63        let bit = (x[i / 8] >> (7 - (i % 8))) & 1;
64        let m = bit.wrapping_neg();
65        for j in 0..BLOCK_LEN {
66            z[j] ^= v[j] & m;
67        }
68        // v >>= 1 over the whole 128-bit word, then conditionally reduce.
69        let lsb = v[BLOCK_LEN - 1] & 1;
70        let mut carry = 0u8;
71        for byte in v.iter_mut() {
72            let next = *byte & 1;
73            *byte = (*byte >> 1) | (carry << 7);
74            carry = next;
75        }
76        v[0] ^= R & lsb.wrapping_neg();
77    }
78    *x = z;
79    z.zeroize();
80    v.zeroize();
81}
82
83/// The GHASH universal hash over a sequence of 16-byte blocks.
84struct Ghash {
85    h: [u8; BLOCK_LEN],
86    /// `H`, `H^2` .. `H^8`: index `i` holds `H^(i+1)`. The first four serve
87    /// the four-block path; the rest the eight-block `PCLMULQDQ` kernel, and
88    /// are only computed once an input long enough to use them arrives.
89    ///
90    /// GHASH is a serial chain by definition -- each block's product feeds the
91    /// next -- and on a CPU where one multiply has several cycles of latency
92    /// and issues one per cycle, that chain, not the multiplier, is what bounds
93    /// AES-GCM. Expanding four steps of the recurrence removes it:
94    ///
95    /// ```text
96    /// Y' = (Y ^ X0)*H^4  ^  X1*H^3  ^  X2*H^2  ^  X3*H
97    /// ```
98    ///
99    /// Four independent multiplies where there were four dependent ones. The
100    /// identity is just distributivity over XOR in GF(2^128), and every product
101    /// here is separately reduced, so this reuses `mul` exactly as it is rather
102    /// than introducing a second reduction to get wrong.
103    powers: [[u8; BLOCK_LEN]; 8],
104    /// Whether `powers[4..]` has been computed.
105    #[cfg(all(target_arch = "x86_64", feature = "std"))]
106    high_powers: bool,
107    acc: [u8; BLOCK_LEN],
108    /// Whether the `PCLMULQDQ` multiply is available. Decided once per value,
109    /// from the CPU alone, so it is not a side channel.
110    ///
111    /// Only exists where an accelerated backend could be compiled in; on other
112    /// targets there is nothing to select between.
113    #[cfg(all(target_arch = "x86_64", feature = "std"))]
114    accelerated: bool,
115}
116
117impl Ghash {
118    fn new(h: [u8; BLOCK_LEN]) -> Self {
119        let mut me = Self {
120            h,
121            powers: [[0u8; BLOCK_LEN]; 8],
122            #[cfg(all(target_arch = "x86_64", feature = "std"))]
123            high_powers: false,
124            acc: [0u8; BLOCK_LEN],
125            #[cfg(all(target_arch = "x86_64", feature = "std"))]
126            accelerated: ghash_accelerated(),
127        };
128        // H^2, H^3, H^4, each built from the previous one by the same multiply
129        // the hot path uses. Once per key, off the hot path.
130        me.powers[0] = h;
131        for slot in 1..4 {
132            let mut p = me.powers[slot - 1];
133            me.mul_by(&mut p, &h);
134            me.powers[slot] = p;
135        }
136        me
137    }
138
139    /// `x *= y` in GCM's field, via whichever backend is live.
140    #[inline]
141    fn mul_by(&self, x: &mut [u8; BLOCK_LEN], y: &[u8; BLOCK_LEN]) {
142        #[cfg(all(target_arch = "x86_64", feature = "std"))]
143        if self.accelerated {
144            // SAFETY: `accelerated` is only true when `ghash_accelerated()`
145            // confirmed both `pclmulqdq` and `ssse3`.
146            unsafe { crate::clmul::mul(x, y) };
147            return;
148        }
149        portable_ghash_mul(x, y);
150    }
151
152    /// Absorb four whole blocks with four independent multiplies.
153    ///
154    /// Correct for the same reason the one-at-a-time path is: see `powers`.
155    #[inline]
156    fn absorb4(&mut self, blocks: &[u8]) {
157        debug_assert_eq!(blocks.len(), BLOCK_LEN * 4);
158        let mut terms = [[0u8; BLOCK_LEN]; 4];
159        for (i, t) in terms.iter_mut().enumerate() {
160            t.copy_from_slice(&blocks[i * BLOCK_LEN..(i + 1) * BLOCK_LEN]);
161        }
162        // The first term carries the accumulator in, and takes the highest
163        // power because it is the oldest.
164        for j in 0..BLOCK_LEN {
165            terms[0][j] ^= self.acc[j];
166        }
167        let multipliers = [
168            &self.powers[3],
169            &self.powers[2],
170            &self.powers[1],
171            &self.powers[0],
172        ];
173        for (t, m) in terms.iter_mut().zip(multipliers) {
174            self.mul_by(t, m);
175        }
176        self.acc = terms[0];
177        for t in &terms[1..] {
178            for j in 0..BLOCK_LEN {
179                self.acc[j] ^= t[j];
180            }
181        }
182    }
183
184    /// Multiply the accumulator by `H`, via whichever backend is live.
185    #[inline]
186    fn mul_acc(&mut self) {
187        #[cfg(all(target_arch = "x86_64", feature = "std"))]
188        if self.accelerated {
189            // SAFETY: `accelerated` is only true when `ghash_accelerated()`
190            // confirmed both `pclmulqdq` and `ssse3`.
191            unsafe { crate::clmul::mul(&mut self.acc, &self.h) };
192            return;
193        }
194        portable_ghash_mul(&mut self.acc, &self.h);
195    }
196
197    /// Absorb `data`, zero-padding the final partial block.
198    fn update_padded(&mut self, mut data: &[u8]) {
199        // Whole groups of eight through the `PCLMULQDQ` kernel, which reduces
200        // once per group; see `clmul::absorb8`.
201        #[cfg(all(target_arch = "x86_64", feature = "std"))]
202        if self.accelerated && data.len() >= BLOCK_LEN * 8 {
203            if !self.high_powers {
204                for slot in 4..8 {
205                    let mut p = self.powers[slot - 1];
206                    self.mul_by(&mut p, &self.h);
207                    self.powers[slot] = p;
208                }
209                self.high_powers = true;
210            }
211            let whole = data.len() - data.len() % (BLOCK_LEN * 8);
212            // SAFETY: `accelerated` is only true when `ghash_accelerated()`
213            // confirmed both `pclmulqdq` and `ssse3`, and `whole` is a
214            // multiple of 128 bytes.
215            unsafe { crate::clmul::absorb8(&mut self.acc, &self.powers, &data[..whole]) };
216            data = &data[whole..];
217        }
218        // Then whole groups of four; the tail falls through to the serial
219        // path, which also handles the final partial block.
220        while data.len() >= BLOCK_LEN * 4 {
221            self.absorb4(&data[..BLOCK_LEN * 4]);
222            data = &data[BLOCK_LEN * 4..];
223        }
224        for chunk in data.chunks(BLOCK_LEN) {
225            let mut block = [0u8; BLOCK_LEN];
226            block[..chunk.len()].copy_from_slice(chunk);
227            for j in 0..BLOCK_LEN {
228                self.acc[j] ^= block[j];
229            }
230            self.mul_acc();
231        }
232    }
233
234    fn finalize(self) -> [u8; BLOCK_LEN] {
235        self.acc
236    }
237}
238
239impl Drop for Ghash {
240    fn drop(&mut self) {
241        self.h.zeroize();
242        self.acc.zeroize();
243        // Powers of `H` are as key-derived as `H` itself. These were not wiped
244        // before, when there were three of them.
245        for p in self.powers.iter_mut() {
246            p.zeroize();
247        }
248    }
249}
250
251/// GCM's counter mode: XOR `data` with the keystream from `counter`,
252/// incrementing its last 32 bits big-endian (`inc32`).
253///
254/// A trait so the AES types can take their backend's own path; see
255/// `aes::x86::ctr32_xor`. The default is the portable construction, and it is
256/// what the accelerated path is tested against.
257pub(crate) trait Ctr32: BlockCipher {
258    fn ctr32_xor(&self, counter: &mut [u8; BLOCK_LEN], data: &mut [u8]) -> Result<()> {
259        ctr32_xor_generic(self, counter, data)
260    }
261}
262
263/// [`Ctr32`] through `encrypt_blocks`, eight blocks per call.
264pub(crate) fn ctr32_xor_generic<C: BlockCipher + ?Sized>(
265    cipher: &C,
266    counter: &mut [u8; BLOCK_LEN],
267    data: &mut [u8],
268) -> Result<()> {
269    const CTR_BATCH: usize = 8;
270    let mut keystream = [0u8; BLOCK_LEN * CTR_BATCH];
271    for chunk in data.chunks_mut(BLOCK_LEN * CTR_BATCH) {
272        let blocks = chunk.len().div_ceil(BLOCK_LEN);
273        for i in 0..blocks {
274            keystream[i * BLOCK_LEN..(i + 1) * BLOCK_LEN].copy_from_slice(counter);
275            increment_be32(counter);
276        }
277        cipher.encrypt_blocks(&mut keystream[..blocks * BLOCK_LEN])?;
278        for (d, k) in chunk.iter_mut().zip(keystream.iter()) {
279            *d ^= k;
280        }
281    }
282    keystream.zeroize();
283    Ok(())
284}
285
286/// Derive the initial counter block J0 from a nonce of any length.
287fn derive_j0(nonce: &[u8], h: &[u8; BLOCK_LEN]) -> [u8; BLOCK_LEN] {
288    if nonce.len() == 12 {
289        let mut j0 = [0u8; BLOCK_LEN];
290        j0[..12].copy_from_slice(nonce);
291        j0[15] = 1;
292        j0
293    } else {
294        let mut g = Ghash::new(*h);
295        g.update_padded(nonce);
296        let mut len_block = [0u8; BLOCK_LEN];
297        len_block[8..].copy_from_slice(&((nonce.len() as u64) * 8).to_be_bytes());
298        g.update_padded(&len_block);
299        g.finalize()
300    }
301}
302
303/// Shared GCM machinery over any 128-bit block cipher.
304fn gcm_core<C: Ctr32>(
305    cipher: &C,
306    nonce: &[u8],
307    aad: &[u8],
308    in_out: &mut [u8],
309    encrypting: bool,
310) -> Result<[u8; BLOCK_LEN]> {
311    ensure!(
312        !nonce.is_empty(),
313        InvalidParameter,
314        "gcm nonce must be non-empty"
315    );
316    ensure!(
317        in_out.len() as u64 <= GcmLimits::MAX_PLAINTEXT_BYTES,
318        CounterExhausted,
319        "gcm plaintext exceeds 2^39-256 bits"
320    );
321
322    // H = E_K(0^128)
323    let mut h = [0u8; BLOCK_LEN];
324    cipher.encrypt_block(&mut h)?;
325
326    let j0 = derive_j0(nonce, &h);
327
328    // When decrypting, GHASH must run over the ciphertext, which is what
329    // `in_out` holds *before* the CTR pass.
330    let mut g = Ghash::new(h);
331    g.update_padded(aad);
332    if !encrypting {
333        g.update_padded(in_out);
334    }
335
336    // CTR starting at inc32(J0).
337    let mut counter = j0;
338    increment_be32(&mut counter);
339    cipher.ctr32_xor(&mut counter, in_out)?;
340
341    if encrypting {
342        g.update_padded(in_out);
343    }
344
345    let mut len_block = [0u8; BLOCK_LEN];
346    len_block[..8].copy_from_slice(&((aad.len() as u64) * 8).to_be_bytes());
347    len_block[8..].copy_from_slice(&((in_out.len() as u64) * 8).to_be_bytes());
348    g.update_padded(&len_block);
349
350    let mut tag = g.finalize();
351    let mut ek_j0 = j0;
352    cipher.encrypt_block(&mut ek_j0)?;
353    for j in 0..BLOCK_LEN {
354        tag[j] ^= ek_j0[j];
355    }
356    ek_j0.zeroize();
357    h.zeroize();
358    Ok(tag)
359}
360
361macro_rules! aes_gcm {
362    ($name:ident, $inner:ty, $id:literal, $disp:literal, $keylen:literal) => {
363        #[doc = concat!("SP 800-38D ", $disp, ".")]
364        pub struct $name($inner);
365
366        impl Algorithm for $name {
367            const ID: &'static str = $id;
368            const NAME: &'static str = $disp;
369        }
370
371        impl Aead for $name {
372            const KEY_LEN: usize = $keylen;
373            const NONCE_LEN: usize = 12;
374            const TAG_LEN: usize = 16;
375
376            fn new(key: &[u8]) -> Result<Self> {
377                ic_core::module::operational()?;
378                Ok(Self(<$inner as BlockCipher>::new(key)?))
379            }
380
381            fn seal_detached(
382                &self,
383                nonce: &[u8],
384                aad: &[u8],
385                in_out: &mut [u8],
386                tag: &mut [u8],
387            ) -> Result<()> {
388                ic_core::module::operational()?;
389                ensure!(tag.len() == 16, InvalidLength, "gcm tag buffer");
390                let t = gcm_core(&self.0, nonce, aad, in_out, true)?;
391                tag.copy_from_slice(&t);
392                Ok(())
393            }
394
395            fn open_detached(
396                &self,
397                nonce: &[u8],
398                aad: &[u8],
399                in_out: &mut [u8],
400                tag: &[u8],
401            ) -> Result<()> {
402                ic_core::module::operational()?;
403                ensure!(tag.len() == 16, InvalidLength, "gcm tag");
404                let expected = gcm_core(&self.0, nonce, aad, in_out, false)?;
405                if ic_core::ct::verify(&expected, tag) {
406                    Ok(())
407                } else {
408                    // Never hand back unauthenticated plaintext.
409                    in_out.zeroize();
410                    Err(ic_core::err!(AuthenticationFailed, $id))
411                }
412            }
413        }
414
415        impl SelfTest for $name {
416            fn self_test() -> Result<()> {
417                let key = [0u8; $keylen];
418                let nonce = [0u8; 12];
419                let c = <Self as Aead>::new(&key)?;
420                let mut buf = [0u8; 16];
421                let mut tag = [0u8; 16];
422                c.seal_detached(&nonce, &[], &mut buf, &mut tag)?;
423                c.open_detached(&nonce, &[], &mut buf, &tag)?;
424                ensure!(buf == [0u8; 16], SelfTestFailed, $id);
425                // A flipped tag bit must be rejected.
426                tag[0] ^= 1;
427                ensure!(
428                    c.open_detached(&nonce, &[], &mut buf, &tag).is_err(),
429                    SelfTestFailed,
430                    $id
431                );
432                Ok(())
433            }
434        }
435    };
436}
437
438aes_gcm!(Aes128Gcm, Aes128, "aes-128-gcm", "AES-128-GCM", 16);
439aes_gcm!(Aes192Gcm, Aes192, "aes-192-gcm", "AES-192-GCM", 24);
440aes_gcm!(Aes256Gcm, Aes256, "aes-256-gcm", "AES-256-GCM", 32);
441
442#[cfg(test)]
443mod tests {
444    use super::*;
445
446    /// The backend's counter mode against the generic one, on the portable
447    /// cipher.
448    ///
449    /// The specification vectors are at most four blocks, so they never reach
450    /// the kernel's whole-group path more than once, never cross the 32-bit
451    /// counter wrap, and never end a message partway through a group. This
452    /// covers every length from 0 to 300 bytes at counters that wrap inside a
453    /// group, and the counter left behind.
454    #[test]
455    fn the_backend_counter_mode_agrees_with_the_generic_one() {
456        use crate::aes::{Aes128, Aes256, Backend};
457        type CtrFn = std::boxed::Box<dyn Fn(&mut [u8; 16], &mut [u8])>;
458        let mut checked = 0;
459        for key_len in [16usize, 32] {
460            let key: std::vec::Vec<u8> = (0..key_len).map(|i| (i * 7 + 3) as u8).collect();
461            let (fast, slow): (CtrFn, CtrFn) = if key_len == 16 {
462                let f = Aes128::new(&key).unwrap();
463                let s = Aes128::new_portable(&key).unwrap();
464                if crate::aes::aesni_available() {
465                    assert_eq!(f.backend(), Backend::Aesni);
466                }
467                (
468                    std::boxed::Box::new(move |c, d| f.ctr32_xor(c, d).unwrap()),
469                    std::boxed::Box::new(move |c, d| ctr32_xor_generic(&s, c, d).unwrap()),
470                )
471            } else {
472                let f = Aes256::new(&key).unwrap();
473                let s = Aes256::new_portable(&key).unwrap();
474                (
475                    std::boxed::Box::new(move |c, d| f.ctr32_xor(c, d).unwrap()),
476                    std::boxed::Box::new(move |c, d| ctr32_xor_generic(&s, c, d).unwrap()),
477                )
478            };
479            // A counter far from the edge, and ones that wrap inside the last
480            // 32 bits partway through a group of eight.
481            for tail in [0x0000_0001u32, 0xffff_fffd, 0xffff_fff9, 0xffff_ffff] {
482                for len in 0..=300usize {
483                    let mut counter = [0x5au8; 16];
484                    counter[12..].copy_from_slice(&tail.to_be_bytes());
485                    let data: std::vec::Vec<u8> = (0..len).map(|i| (i * 13) as u8).collect();
486                    let (mut a, mut b) = (data.clone(), data.clone());
487                    let (mut ca, mut cb) = (counter, counter);
488                    fast(&mut ca, &mut a);
489                    slow(&mut cb, &mut b);
490                    assert_eq!(a, b, "key {key_len}, counter tail {tail:#x}, {len} bytes");
491                    assert_eq!(
492                        ca, cb,
493                        "final counter, key {key_len}, tail {tail:#x}, {len} bytes"
494                    );
495                    checked += 1;
496                }
497            }
498        }
499        assert_eq!(checked, 2 * 4 * 301);
500    }
501
502    /// The four-at-a-time path must agree with the one-at-a-time path.
503    ///
504    /// The specification vectors do not establish this. The longest of them is
505    /// four blocks, so they barely reach `absorb4` and never exercise it
506    /// repeatedly or alongside a tail -- a version that was wrong from the
507    /// second group onward, or wrong about the remainder, would pass all of
508    /// them. This drives both paths over every length either side of the group
509    /// boundary and compares the accumulators.
510    #[test]
511    fn the_batched_ghash_agrees_with_the_serial_one() {
512        let h = [
513            0x66, 0xe9, 0x4b, 0xd4, 0xef, 0x8a, 0x2c, 0x3b, 0x88, 0x4c, 0xfa, 0x59, 0xca, 0x34,
514            0x2b, 0x2e,
515        ];
516
517        let mut checked = 0;
518        // Around one group, two groups, and the ragged lengths between.
519        for len in [
520            0usize, 1, 15, 16, 17, 31, 63, 64, 65, 79, 80, 127, 128, 129, 255, 256, 1024, 1025,
521        ] {
522            let data: std::vec::Vec<u8> = (0..len)
523                .map(|i| ((i as u64).wrapping_mul(0x9e37_79b9) >> 3) as u8)
524                .collect();
525
526            let mut batched = Ghash::new(h);
527            batched.update_padded(&data);
528
529            // The serial reference: one block at a time, no grouping.
530            let mut serial = Ghash::new(h);
531            for chunk in data.chunks(BLOCK_LEN) {
532                let mut block = [0u8; BLOCK_LEN];
533                block[..chunk.len()].copy_from_slice(chunk);
534                for j in 0..BLOCK_LEN {
535                    serial.acc[j] ^= block[j];
536                }
537                serial.mul_acc();
538            }
539
540            assert_eq!(
541                batched.acc, serial.acc,
542                "batched and serial GHASH disagree at {len} bytes"
543            );
544            checked += 1;
545        }
546        assert_eq!(checked, 18, "the comparison did not run");
547
548        // The eight-block kernel with an accumulator already carrying state,
549        // as it does after the AAD: a short prefix first, then long inputs
550        // either side of a group boundary.
551        for (prefix, len) in [(17usize, 128usize), (16, 2048), (5, 2048 + 48), (33, 1023)] {
552            let data: std::vec::Vec<u8> = (0..prefix + len)
553                .map(|i| ((i as u64).wrapping_mul(0x2545_f491_4f6c_dd1d) >> 11) as u8)
554                .collect();
555            let (head, tail) = data.split_at(prefix);
556            let mut batched = Ghash::new(h);
557            batched.update_padded(head);
558            batched.update_padded(tail);
559
560            let mut serial = Ghash::new(h);
561            for part in [head, tail] {
562                for chunk in part.chunks(BLOCK_LEN) {
563                    let mut block = [0u8; BLOCK_LEN];
564                    block[..chunk.len()].copy_from_slice(chunk);
565                    for j in 0..BLOCK_LEN {
566                        serial.acc[j] ^= block[j];
567                    }
568                    serial.mul_acc();
569                }
570            }
571            assert_eq!(batched.acc, serial.acc, "prefix {prefix}, then {len} bytes");
572
573            // Where the kernel exists it must have run, or this is the
574            // four-block path agreeing with the serial one again.
575            #[cfg(all(target_arch = "x86_64", feature = "std"))]
576            assert_eq!(batched.high_powers, batched.accelerated);
577        }
578
579        // And the grouping must actually have been used, or the agreement
580        // above is two serial paths agreeing with each other.
581        let long = std::vec![0xa5u8; BLOCK_LEN * 4];
582        let mut g = Ghash::new(h);
583        g.absorb4(&long);
584        let mut serial = Ghash::new(h);
585        for chunk in long.chunks(BLOCK_LEN) {
586            for j in 0..BLOCK_LEN {
587                serial.acc[j] ^= chunk[j];
588            }
589            serial.mul_acc();
590        }
591        assert_eq!(g.acc, serial.acc, "absorb4 alone disagrees with four steps");
592    }
593
594    /// `H^2`, `H^3` and `H^4` must be what they claim.
595    #[test]
596    fn the_precomputed_powers_are_powers_of_h() {
597        let h = [0x3cu8; BLOCK_LEN];
598        let mut g = Ghash::new(h);
599        // Long enough to make the eight-block path build H^5..H^8 where it
600        // runs; elsewhere only the first four exist.
601        g.update_padded(&[0u8; BLOCK_LEN * 8]);
602        #[cfg(all(target_arch = "x86_64", feature = "std"))]
603        let built = if g.high_powers { 8 } else { 4 };
604        #[cfg(not(all(target_arch = "x86_64", feature = "std")))]
605        let built = 4;
606
607        let mut expect = h;
608        for (i, stored) in g.powers[..built].iter().enumerate() {
609            if i > 0 {
610                g.mul_by(&mut expect, &h);
611            }
612            assert_eq!(*stored, expect, "powers[{i}] is not H^{}", i + 1);
613        }
614        // Distinct, so a table of copies would fail rather than pass.
615        for i in 1..built {
616            assert_ne!(g.powers[i - 1], g.powers[i]);
617        }
618    }
619    use ic_core::codec::{hex, unhex};
620
621    /// Runs one of the McGrew–Viega GCM test vectors end to end.
622    fn check(key: &str, nonce: &str, pt: &str, aad: &str, ct: &str, tag: &str) {
623        let k = unhex(key).unwrap();
624        let mut buf = unhex(pt).unwrap();
625        let mut got_tag = [0u8; 16];
626        let n = unhex(nonce).unwrap();
627        let a = unhex(aad).unwrap();
628
629        match k.len() {
630            16 => {
631                let c = Aes128Gcm::new(&k).unwrap();
632                c.seal_detached(&n, &a, &mut buf, &mut got_tag).unwrap();
633            }
634            24 => {
635                let c = Aes192Gcm::new(&k).unwrap();
636                c.seal_detached(&n, &a, &mut buf, &mut got_tag).unwrap();
637            }
638            _ => {
639                let c = Aes256Gcm::new(&k).unwrap();
640                c.seal_detached(&n, &a, &mut buf, &mut got_tag).unwrap();
641            }
642        }
643        assert_eq!(hex(&buf), ct, "ciphertext");
644        assert_eq!(hex(&got_tag), tag, "tag");
645    }
646
647    #[test]
648    fn gcm_spec_case_1_empty() {
649        check(
650            "00000000000000000000000000000000",
651            "000000000000000000000000",
652            "",
653            "",
654            "",
655            "58e2fccefa7e3061367f1d57a4e7455a",
656        );
657    }
658
659    #[test]
660    fn gcm_spec_case_2_single_block() {
661        check(
662            "00000000000000000000000000000000",
663            "000000000000000000000000",
664            "00000000000000000000000000000000",
665            "",
666            "0388dace60b6a392f328c2b971b2fe78",
667            "ab6e47d42cec13bdf53a67b21257bddf",
668        );
669    }
670
671    /// Case 3 authenticates the full 64-byte plaintext with no AAD; case 4
672    /// below truncates it to 60 bytes and adds AAD, exercising both the
673    /// partial-block and the AAD paths through GHASH.
674    #[test]
675    fn gcm_spec_case_3_multi_block() {
676        check(
677            "feffe9928665731c6d6a8f9467308308",
678            "cafebabefacedbaddecaf888",
679            "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b391aafd255",
680            "",
681            "42831ec2217774244b7221b784d0d49ce3aa212f2c02a4e035c17e2329aca12e21d514b25466931c7d8f6a5aac84aa051ba30b396a0aac973d58e091473f5985",
682            "4d5c2af327cd64a62cf35abd2ba6fab4",
683        );
684    }
685
686    #[test]
687    fn gcm_spec_case_4_with_aad() {
688        check(
689            "feffe9928665731c6d6a8f9467308308",
690            "cafebabefacedbaddecaf888",
691            "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b39",
692            "feedfacedeadbeeffeedfacedeadbeefabaddad2",
693            "42831ec2217774244b7221b784d0d49ce3aa212f2c02a4e035c17e2329aca12e21d514b25466931c7d8f6a5aac84aa051ba30b396a0aac973d58e091",
694            "5bc94fbc3221a5db94fae95ae7121a47",
695        );
696    }
697
698    /// Case 5: a 64-bit nonce, which exercises the GHASH-based J0 derivation.
699    #[test]
700    fn gcm_short_nonce_uses_ghash_j0() {
701        check(
702            "feffe9928665731c6d6a8f9467308308",
703            "cafebabefacedbad",
704            "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b39",
705            "feedfacedeadbeeffeedfacedeadbeefabaddad2",
706            "61353b4c2806934a777ff51fa22a4755699b2a714fcdc6f83766e5f97b6c742373806900e49f24b22b097544d4896b424989b5e1ebac0f07c23f4598",
707            "3612d2e79e3b0785561be14aaca2fccb",
708        );
709    }
710
711    #[test]
712    fn aes256_gcm_vector() {
713        check(
714            "feffe9928665731c6d6a8f9467308308feffe9928665731c6d6a8f9467308308",
715            "cafebabefacedbaddecaf888",
716            "d9313225f88406e5a55909c5aff5269a86a7a9531534f7da2e4c303d8a318a721c3c0c95956809532fcf0e2449a6b525b16aedf5aa0de657ba637b39",
717            "feedfacedeadbeeffeedfacedeadbeefabaddad2",
718            "522dc1f099567d07f47f37a32a84427d643a8cdcbfe5c0c97598a2bd2555d1aa8cb08e48590dbb3da7b08b1056828838c5f61e6393ba7a0abcc9f662",
719            "76fc6ece0f4e1768cddf8853bb2d551b",
720        );
721    }
722
723    #[test]
724    fn roundtrip_and_tamper_detection() {
725        let c = Aes256Gcm::new(&[7u8; 32]).unwrap();
726        let nonce = [9u8; 12];
727        let aad = b"header";
728        let plaintext = b"attack at dawn, bring the ontology";
729
730        let mut buf = plaintext.to_vec();
731        let mut tag = [0u8; 16];
732        c.seal_detached(&nonce, aad, &mut buf, &mut tag).unwrap();
733        assert_ne!(&buf[..], &plaintext[..]);
734
735        let mut ok = buf.clone();
736        c.open_detached(&nonce, aad, &mut ok, &tag).unwrap();
737        assert_eq!(&ok[..], &plaintext[..]);
738
739        // Tampered ciphertext must fail and must not leak plaintext.
740        let mut bad = buf.clone();
741        bad[0] ^= 1;
742        assert!(c.open_detached(&nonce, aad, &mut bad, &tag).is_err());
743        assert_eq!(
744            bad,
745            vec![0u8; bad.len()],
746            "plaintext must be wiped on failure"
747        );
748
749        // Wrong AAD must fail.
750        let mut wrong_aad = buf.clone();
751        assert!(c
752            .open_detached(&nonce, b"other", &mut wrong_aad, &tag)
753            .is_err());
754
755        // Wrong nonce must fail.
756        let mut wrong_nonce = buf.clone();
757        assert!(c
758            .open_detached(&[0u8; 12], aad, &mut wrong_nonce, &tag)
759            .is_err());
760    }
761
762    #[test]
763    fn self_tests_pass() {
764        Aes128Gcm::self_test().unwrap();
765        Aes192Gcm::self_test().unwrap();
766        Aes256Gcm::self_test().unwrap();
767    }
768}