philbin 1.0.1

A pure Rust AEGIS library with SIMD and runtime CPU detection
Documentation
// # What is this "aligned buffer" business?
//
// We need a byte array with an alignemnt of OUTPUT_RATE_BYTES that the Aegis
// trait uses. OUTPUT_RATE_BYTES depends on AEGIS algorithm variant. The byte
// width of the SIMD vector we use with the variant is _always_ less than or
// equal to OUTPUT_RATE_BYTES, so reading from a byte array aligned to
// OUTPUT_RATE_BYTES ensures we make _aligned_ (instead of _unaligned_)
// reads[^1] from the byte array when copying into that SIMD vector.
//
// So we pass OUTPUT_RATE_BYTES as the BYTES parameter for AlignedBuf.
//
// To get a properly aligned byte array, we have to put it in a struct and then
// use `#[align(N)]` to align it to N bytes... but we don't want to end up with
// any padding bytes in our struct, so `#[align(N)]` must match the width of the
// internal byte array.
//
// Rust does not allow using a const generic parameter for that N in
// `#[align(N)]`. Thus we use a macro to generate AlignedBuffer[16|32|64|128]
// structs, all of which implement the AlignedBuf trait.
//
// The Aegis trait uses the AlignedBuf trait as an associated type. Our
// Aegis128L and Aegis256 structs then use AlignedBufHolder and AlignedBufRouter
// to choose the correct AlignedBufferXX struct to "fill in" the AlignedBuf
// associated type.
//
// Confused? Perfectly understandable; we're hitting some rough edge cases of
// the Rust type system that require significant boilerplate to work around.
//
// # Preserving Alignment
//
// _References_ to the array field inside AlignedBufferXX (or the struct itself)
// are guaranteed to be aligned correctly.
//
// ...BUT ONLY REFERENCES! If you pass the array _by value_ to some function,
// you lose the alignment guarantee. Passing the AlignedBufferXX struct by-value
// is fine and the alignment is preserved.
//
// TL;DR: Use the `as_ref()`/`as_mut()` methods on AlignedBufferXX (or through
// the AlignedBuf trait) which return a (mutable) reference to the internal
// array. Do as much work as possible _through the reference_ to get any perf
// benefits from aligned reads.
//
// NOTE: It is NOT possible to somehow transmute an unaligned array reference
// like `&[u8; 16]` to an `&AlignedBuffer16`. Data at an unaligned address
// _must be copied_ to an aligned address to get an aligned reference to it.
// Thus you'll need to call `AlignedBuffer16::from_array([u8; 16])`.
//
// [^1]: On modern CPUs the perf cost of reading unaligned data is effectively
// zero. Here's why:
//
// - Unaligned reads that are within a cache line (64 B) are free.
// - Unaligned reads which cross a cache line have a tiny cost, BUT:
//    - Big out-of-order execution windows on modern CPUs hide the latency.
// - Crossing an OS page boundary (4 KiB) has a small cost, but hitting that is
//   rare enough to not matter.
//
// So yes, there are many systems in place that will hide this cost, but it's
// still better to just make aligned reads in the first place. And on older or
// embedded CPUs unaligned reads are still an issue.

use crate::utils::const_assert;
use std::mem::size_of;

/// A trait representing an array of `BYTES` bytes that are also aligned
/// exactly to the provided `BYTES`.
pub trait AlignedBuf<const BYTES: usize>
where
  Self: AsRef<[u8; BYTES]> + AsMut<[u8; BYTES]>,
{
  /// Creates an `AlignedBuf` initialized with the bytes from the provided array.
  fn from_array(data: [u8; BYTES]) -> Self;

  /// Creates an all-zero `AlignedBuf`.
  fn new() -> Self;
}

/// Part of the necessary machinery that maps a number of bytes to the correct
/// `AlignedBufferXX` type.
///
/// Here's how to use the machinery:
///
/// ```ignore
/// <AlignedBufHolder as AlignedBufRouter<BYTES>>::AlignedBuf
/// ```
///
/// That expression should resolve to the correct `AlignedBufferXX` type.
///
/// -  `<AlignedBufHolder as AlignedBufRouter<32>>::AlignedBuf`
///    resolves to `AlignedBuffer32`
/// -  `<AlignedBufHolder as AlignedBufRouter<64>>::AlignedBuf`
///    resolves to `AlignedBuffer64`
///
/// etc.
pub trait AlignedBufRouter<const N: usize> {
  type AlignedBuf: AlignedBuf<N>;
}

/// See the comment on [`AlignedBufRouter`].
pub struct AlignedBufHolder;

// Takes a number of bytes and produces an AlignedBufferXX type where XX is the
// number of bytes provided.
//
// Call example:
//
//  gen_aligned_buffer!(16);
macro_rules! gen_aligned_buffer {
  ($bytes:expr) => {
    pastey::paste! {
      #[repr(align($bytes))]
      #[derive(Clone, Copy)]
      pub struct [<AlignedBuffer $bytes>]([u8; $bytes]);

      const_assert!(size_of::<[<AlignedBuffer $bytes>]>() == $bytes,
                    "Verify no padding in AlignedBufferXX");

      impl AsRef<[u8; $bytes]> for [<AlignedBuffer $bytes>] {
        fn as_ref(&self) -> &[u8; $bytes] {
          &self.0
        }
      }

      impl AsMut<[u8; $bytes]> for [<AlignedBuffer $bytes>] {
        fn as_mut(&mut self) -> &mut [u8; $bytes] {
          &mut self.0
        }
      }

      impl AlignedBuf<$bytes> for [<AlignedBuffer $bytes>] {
        fn from_array(data: [u8; $bytes]) -> Self {
          [<AlignedBuffer $bytes>](data)
        }

        fn new() -> Self {
          [<AlignedBuffer $bytes>]([0; $bytes])
        }
      }

      impl AlignedBufRouter<$bytes> for AlignedBufHolder {
        type AlignedBuf = [<AlignedBuffer $bytes>];
      }
    }
  };
}

// We need an AlignedBufferXX for each OUTPUT_RATE_BYTES that we end up using.
gen_aligned_buffer!(16 /* bytes; 128 bits */);
gen_aligned_buffer!(32 /* bytes; 256 bits */);
gen_aligned_buffer!(64 /* bytes; 512 bits */);
gen_aligned_buffer!(128 /* bytes; 1024 bits */);