philbin 1.0.1

A pure Rust AEGIS library with SIMD and runtime CPU detection
Documentation
use crate::arch::Simd;
use std::ops::{BitAnd, BitXor};
use zerocopy::transmute;

// An array of 128 bits (16 bytes). Same size as an AES block.
//
// NOTE: Do NOT confuse an "AES block" with an "AEGIS state block". While
// related and similar, they are not the same.
pub type Array128 = [u8; 16];

// The C0 constant from RFC 10032.
pub const RAW_C0: Array128 = [
  0x00, 0x01, 0x01, 0x02, 0x03, 0x05, 0x08, 0x0d, 0x15, 0x22, 0x37, 0x59, 0x90,
  0xe9, 0x79, 0x62,
];

// The C1 constant from RFC 10032.
pub const RAW_C1: Array128 = [
  0xdb, 0x3d, 0x18, 0x55, 0x6d, 0xc2, 0x2f, 0xf1, 0x20, 0x11, 0x31, 0x42, 0x73,
  0xb5, 0x28, 0xdd,
];

// Produces the `ctx` computed constant used in data-parallel AEGIS ciphers.
//
// From the `Init(key, nonce)` function in the AEGIS-128X (RFC 10032) spec:
//
// for i in 0..D:
//     ctx[i] = ZeroPad(Byte(i) || Byte(D - 1), 128)
//
// "D" is the degree of parallelism, so 1, 2 or 4.
const fn make_ctx<const BYTES: usize>(d: usize) -> [u8; BYTES] {
  let mut ctx = [0u8; BYTES];
  let mut i: usize = 0;
  // No FOR loops in const context!
  while i < d {
    ctx[i * 16] = i as u8;
    ctx[i * 16 + 1] = (d - 1) as u8;
    i += 1;
  }
  ctx
}

// D == 1 -> 128-bit SIMD vector
// D == 2 -> 256-bit SIMD vector, 2x128 lanes
// D == 4 -> 512-bit SIMD vector, 4x128 lanes
pub const RAW_CTX_D1: [u8; 16] = make_ctx(/* D = */ 1);
pub const RAW_CTX_D2: [u8; 32] = make_ctx(/* D = */ 2);
pub const RAW_CTX_D4: [u8; 64] = make_ctx(/* D = */ 4);

/// Represents a block of bytes of one "state unit" of AEGIS.
///
/// A Block can only be 128, 256 or 512 bits (16, 32 or 64 bytes) wide.
///
/// AEGIS-128L uses 8 blocks of 128 bits (8x128 as shorthand).
/// AEGIS-256 uses 6 blocks of 128 bits (6x128).
///
/// Here's a full table including the parallel "X" variants:
///
/// | Cipher      | Block Spec | Output Rate |
/// | ----------- | ---------- | ----------- |
/// | AEGIS-128L  | 8x128      | 256         |
/// | AEGIS-128X2 | 8x256      | 512         |
/// | AEGIS-128X4 | 8x512      | 1024        |
/// | AEGIS-256   | 6x128      | 128         |
/// | AEGIS-256X2 | 6x256      | 256         |
/// | AEGIS-256X4 | 6x512      | 512         |
///
/// The table also includes the Output Rate AKA Absorption Rate (in bits) for
/// each cipher.
///
/// IMPORTANT! Notice that for AEGIS-128[L|X] ciphers the Output Rate is ALWAYS
/// 2x the block size while for AEGIS-256[X] ciphers it equals the Output Rate.
/// This is the reason why we also have the [`BlockDoubleRate`] trait which is
/// necessary for implementing AEGIS-128[L|X] ciphers.
///
/// (Most) implementations of Block use SIMD vectors internally. Each SIMD
/// "lane" is 128 bits, so a 512-bit SIMD vector has 4 128-bit lanes.
///
/// Depending on the CPU, we may "emulate" a 256-bit SIMD vector with two
/// 128-bit SIMD vectors (and similar for 512-bit SIMD). Or Block might be
/// implemented entirely with plain byte arrays (the fallback path).
pub trait Block
where
  Self: Copy,
  Self: BitXor<Self, Output = Self>,
  Self: BitAnd<Self, Output = Self>,
{
  /// A Block represented as an array of bytes.
  type SelfArray: AsRef<[u8]>;
  type Simd: Simd;

  // Self instances initialized from `RAW_C[0|1]` constants, as appropriate for
  // the width of the Block.
  fn c0(simd: Self::Simd) -> Self;
  fn c1(simd: Self::Simd) -> Self;
  /// Impls should assign the correct `RAW_CTX_D#` constant (defined above).
  /// See those constants and `make_ctx()` for docs.
  fn ctx(simd: Self::Simd) -> Self;

  /// Create a Block from a same-size array.
  fn new(simd: Self::Simd, input: Self::SelfArray) -> Self;

  /// Create a Block from a 128-bit (16 byte) array.
  ///
  /// This must work even when Block is a 2x or 4x multiple of such an array! As
  /// per RFC 10032 ("The Init Function" section for AEGIS-[128|256]X), each
  /// 128-bit SIMD lane is initialized with the same bytes. So if Block is 2x/4x
  /// wider than `Array128`, we just concat `Array128` 2/4 times and init with
  /// that.
  fn from_128_bits(simd: Self::Simd, input: Array128) -> Self;
  fn into_bytes(self) -> Self::SelfArray;

  /// Creates a Block from two integers. Used in the spec's:
  ///
  /// ```ignore
  /// Finalize(ad_len_bits, msg_len_bits)
  /// ```
  ///
  /// function. For Blocks with multiple 128-bit SIMD lanes, each lane is
  /// initialized with the same bytes.
  #[inline(always)]
  fn from_ints(simd: Self::Simd, top: u64, bottom: u64) -> Self {
    Self::from_128_bits(
      simd,
      transmute!([top.to_le_bytes(), bottom.to_le_bytes()]),
    )
  }

  /// XOR's the block's lanes down to one lane and returns it as an array.
  /// Thus:
  ///  If Block is 1x128, this is `into_bytes()`.
  ///  If Block is 2x128, the top and bottom halves are XOR'd and returned.
  ///  If Block is 4x128, all four lanes are XOR'd and returned as one lane.
  ///
  /// (This is used in the `Finalize()` spec function for data-parallel AEGIS.)
  fn xor_down(self) -> Array128;

  /// Computes a single AES round using `state` and `round_key` and returns a
  /// new Block.
  fn aes_encrypt_round(state: Self, round_key: Self) -> Self;
}

/// A [`Block`] subtrait with additional methods needed when implementing
/// AEGIS-128[L|X] ciphers due to Output Rate being 2x the Block width.
///
/// See [`Block`] for more details.
pub trait BlockDoubleRate<const OUTPUT_RATE_BYTES: usize>
where
  Self: Block,
{
  // We'd LIKE to use:
  //  type SelfArray: [u8; OUTPUT_RATE_BYTES / 2];
  //  type OutputRateArray: [u8; OUTPUT_RATE_BYTES];
  // and then use those types in method params, but as of Rust 1.96 this can't
  // be done due to missing support in the compiler. Specifically:
  // generic const expressions and associated type defaults.
  // The whole OUTPUT_RATE_BYTES const generic shouldn't be necessary here, but
  // we live with the compiler we have, not the compiler we'd like to have.

  /// Split an array of 2x Self width into 2 Selves.
  /// Takes argument by-ref so we can pass a correctly aligned reference.
  fn split(simd: Self::Simd, chunk: &[u8; OUTPUT_RATE_BYTES]) -> [Self; 2];

  /// Concat two Selves into an array of 2x Self width.
  fn concat(first: Self, second: Self) -> [u8; OUTPUT_RATE_BYTES];
}

// Takes the width of a Block in bytes and generates a blanket
// BlockDoubleRate impl for that byte width.
//
// Call example:
//
//   gen_block_double_rate_blanket!(16);
macro_rules! gen_block_double_rate_blanket {
  ($width:expr) => {
    impl<T> BlockDoubleRate<{ $width * 2 }> for T
    where
      T: Block<SelfArray = [u8; $width]>,
    {
      #[inline(always)]
      fn split(simd: Self::Simd, chunk: &[u8; $width * 2]) -> [Self; 2] {
        let (p1, p2) = transmute!(*chunk);
        [Self::new(simd, p1), Self::new(simd, p2)]
      }

      #[inline(always)]
      fn concat(first: Self, second: Self) -> [u8; $width * 2] {
        transmute!([first.into_bytes(), second.into_bytes()])
      }
    }
  };
}

// Blanket impls of BlockDoubleRate for supported SIMD widths.
gen_block_double_rate_blanket!(16 /* bytes; 128 bits */);
gen_block_double_rate_blanket!(32 /* bytes; 256 bits */);
gen_block_double_rate_blanket!(64 /* bytes; 512 bits */);

// Takes the `Simd` generic parameter.
// Expands to common method impls for _all_ Block implementations.
// Called by `gen_shared_blockXXX` macros.
macro_rules! gen_shared_block {
  ($simd_ty:ty) => {
    type Simd = $simd_ty;

    #[inline(always)]
    fn c0(simd: $simd_ty) -> Self {
      Self::from_128_bits(simd, crate::base::block::RAW_C0)
    }

    #[inline(always)]
    fn c1(simd: $simd_ty) -> Self {
      Self::from_128_bits(simd, crate::base::block::RAW_C1)
    }

    #[inline(always)]
    fn into_bytes(self) -> Self::SelfArray {
      transmute!([self.val])
    }

    #[inline(always)]
    fn new(simd: $simd_ty, input: Self::SelfArray) -> Self {
      Self {
        val: transmute!([input]),
        simd,
      }
    }
  };
}
pub(crate) use gen_shared_block;

// Takes the `Simd` generic parameter.
// Expands to common method impls for all 128-bit Block implementations.
macro_rules! gen_shared_block128 {
  ($simd_ty:ty) => {
    crate::base::block::gen_shared_block!($simd_ty);
    type SelfArray = [u8; 16];

    #[inline(always)]
    fn ctx(simd: $simd_ty) -> Self {
      Self::new(simd, crate::base::block::RAW_CTX_D1)
    }

    #[inline(always)]
    fn from_128_bits(simd: $simd_ty, input: Array128) -> Self {
      Self {
        // Double-array to avoid silly "transmute to itself" error for Fallback
        // types.
        val: transmute!([input]),
        simd,
      }
    }

    #[inline(always)]
    fn xor_down(self) -> Array128 {
      // Already one 128-bit lane wide so just convert to array.
      Self::into_bytes(self)
    }
  };
}
pub(crate) use gen_shared_block128;

// Takes the `Simd` generic parameter.
// Expands to common method impls for all 256-bit Block implementations.
macro_rules! gen_shared_block256 {
  ($simd_ty:ty) => {
    crate::base::block::gen_shared_block!($simd_ty);
    type SelfArray = [u8; 32];

    #[inline(always)]
    fn ctx(simd: $simd_ty) -> Self {
      Self::new(simd, crate::base::block::RAW_CTX_D2)
    }

    #[inline(always)]
    fn from_128_bits(simd: $simd_ty, input: Array128) -> Self {
      Self {
        // Double-array to avoid silly "transmute to itself" error for Fallback
        // types.
        val: transmute!([[input, input]]),
        simd,
      }
    }
  };
}
pub(crate) use gen_shared_block256;

// Takes the `Simd` generic parameter.
// Expands to common method impls for all 512-bit Block implementations.
macro_rules! gen_shared_block512 {
  ($simd_ty:ty) => {
    crate::base::block::gen_shared_block!($simd_ty);
    type SelfArray = [u8; 64];

    #[inline(always)]
    fn ctx(simd: $simd_ty) -> Self {
      Self::new(simd, crate::base::block::RAW_CTX_D4)
    }

    #[inline(always)]
    fn from_128_bits(simd: $simd_ty, input: Array128) -> Self {
      Self {
        // Double-array to avoid silly "transmute to itself" error for Fallback
        // types.
        val: transmute!([[input, input, input, input]]),
        simd,
      }
    }
  };
}
pub(crate) use gen_shared_block512;