use crate::error::Error;
const BC7_PARALLEL_MIN_BLOCKS: usize = 16_384;
fn alloc_and_decode(
data: &[u8],
width: u32,
height: u32,
block_bytes: usize,
f: impl FnOnce(&mut [u8]) -> Result<(), Error>,
) -> Result<Vec<u8>, Error> {
let (_, _, expected) = block_grid(width, height, block_bytes)?;
if data.len() < expected {
return Err(Error::TruncatedData);
}
let need = (width as usize)
.checked_mul(height as usize)
.and_then(|n| n.checked_mul(4))
.ok_or(Error::OutOfBounds)?;
let mut out = vec![0u8; need];
f(&mut out)?;
Ok(out)
}
pub fn decode_bc1(data: &[u8], width: u32, height: u32) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 8, |o| {
decode_bc1_into(data, width, height, o)
})
}
pub fn decode_bc1_into(data: &[u8], width: u32, height: u32, out: &mut [u8]) -> Result<(), Error> {
decode_rgba_blocks_into(data, width, height, 8, out, |block, dst, pitch| {
bcdec_rs::bc1(block, dst, pitch);
})
}
pub fn decode_bc2(data: &[u8], width: u32, height: u32) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 16, |o| {
decode_bc2_into(data, width, height, o)
})
}
pub fn decode_bc2_into(data: &[u8], width: u32, height: u32, out: &mut [u8]) -> Result<(), Error> {
decode_rgba_blocks_into(data, width, height, 16, out, |block, dst, pitch| {
bcdec_rs::bc2(block, dst, pitch);
})
}
pub fn decode_bc3(data: &[u8], width: u32, height: u32) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 16, |o| {
decode_bc3_into(data, width, height, o)
})
}
pub fn decode_bc3_into(data: &[u8], width: u32, height: u32, out: &mut [u8]) -> Result<(), Error> {
decode_rgba_blocks_into(data, width, height, 16, out, |block, dst, pitch| {
bcdec_rs::bc3(block, dst, pitch);
})
}
pub fn decode_bc4(
data: &[u8],
width: u32,
height: u32,
is_signed: bool,
) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 8, |o| {
decode_bc4_into(data, width, height, is_signed, o)
})
}
pub fn decode_bc4_into(
data: &[u8],
width: u32,
height: u32,
is_signed: bool,
out: &mut [u8],
) -> Result<(), Error> {
let (blocks_x, blocks_y, expected) = block_grid(width, height, 8)?;
if data.len() < expected {
return Err(Error::TruncatedData);
}
let out_w = width as usize;
let out_h = height as usize;
check_out_len(out, out_w, out_h)?;
let mut block_r = [0u8; 16];
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * 8;
bcdec_rs::bc4(&data[bi..bi + 8], &mut block_r, 4, is_signed);
blit_r_to_rgba(&block_r, out, out_w, out_h, bx * 4, by * 4);
}
}
Ok(())
}
pub fn decode_bc5(
data: &[u8],
width: u32,
height: u32,
is_signed: bool,
) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 16, |o| {
decode_bc5_into(data, width, height, is_signed, o)
})
}
pub fn decode_bc5_into(
data: &[u8],
width: u32,
height: u32,
is_signed: bool,
out: &mut [u8],
) -> Result<(), Error> {
let (blocks_x, blocks_y, expected) = block_grid(width, height, 16)?;
if data.len() < expected {
return Err(Error::TruncatedData);
}
let out_w = width as usize;
let out_h = height as usize;
check_out_len(out, out_w, out_h)?;
let mut block_rg = [0u8; 32];
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * 16;
bcdec_rs::bc5(&data[bi..bi + 16], &mut block_rg, 8, is_signed);
blit_rg_to_rgba(&block_rg, out, out_w, out_h, bx * 4, by * 4);
}
}
Ok(())
}
pub fn decode_bc7(data: &[u8], width: u32, height: u32) -> Result<Vec<u8>, Error> {
alloc_and_decode(data, width, height, 16, |o| {
decode_bc7_into(data, width, height, o)
})
}
pub fn decode_bc7_into(data: &[u8], width: u32, height: u32, out: &mut [u8]) -> Result<(), Error> {
let (blocks_x, blocks_y, expected) = block_grid(width, height, 16)?;
if data.len() < expected {
return Err(Error::TruncatedData);
}
let out_w = width as usize;
let out_h = height as usize;
check_out_len(out, out_w, out_h)?;
let aligned = width % 4 == 0 && height % 4 == 0;
let parallel = aligned
&& blocks_y >= 2
&& blocks_x.saturating_mul(blocks_y) >= BC7_PARALLEL_MIN_BLOCKS;
if parallel {
decode_bc7_parallel(data, out, out_w, blocks_x, blocks_y);
} else if aligned {
decode_bc7_direct(data, out, out_w, blocks_x, blocks_y);
} else {
decode_bc7_scratch(data, out, out_w, out_h, blocks_x, blocks_y);
}
Ok(())
}
fn check_out_len(out: &[u8], out_w: usize, out_h: usize) -> Result<(), Error> {
let need = out_w
.checked_mul(out_h)
.and_then(|n| n.checked_mul(4))
.ok_or(Error::OutOfBounds)?;
if out.len() != need {
return Err(Error::OutOfBounds);
}
Ok(())
}
fn decode_rgba_blocks_into(
data: &[u8],
width: u32,
height: u32,
block_bytes: usize,
out: &mut [u8],
decode_block: impl Fn(&[u8], &mut [u8], usize),
) -> Result<(), Error> {
let (blocks_x, blocks_y, expected) = block_grid(width, height, block_bytes)?;
if data.len() < expected {
return Err(Error::TruncatedData);
}
let out_w = width as usize;
let out_h = height as usize;
check_out_len(out, out_w, out_h)?;
let pitch = out_w * 4;
if width % 4 == 0 && height % 4 == 0 {
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * block_bytes;
let offset = (by * 4 * out_w + bx * 4) * 4;
decode_block(&data[bi..bi + block_bytes], &mut out[offset..], pitch);
}
}
} else {
let mut scratch = [0u8; 64];
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * block_bytes;
decode_block(&data[bi..bi + block_bytes], &mut scratch, 16);
blit_rgba4(&scratch, out, out_w, out_h, bx * 4, by * 4);
}
}
}
Ok(())
}
fn decode_bc7_direct(
data: &[u8],
out: &mut [u8],
out_w: usize,
blocks_x: usize,
blocks_y: usize,
) {
let pitch = out_w * 4;
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * 16;
let offset = (by * 4 * out_w + bx * 4) * 4;
let (blk, dst) = (&data[bi..bi + 16], &mut out[offset..]);
if !bc7_fast_block(blk, dst, pitch) {
bcdec_rs::bc7(blk, dst, pitch);
}
}
}
}
fn decode_bc7_scratch(
data: &[u8],
out: &mut [u8],
out_w: usize,
out_h: usize,
blocks_x: usize,
blocks_y: usize,
) {
let mut scratch = [0u8; 64];
for by in 0..blocks_y {
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * 16;
let blk = &data[bi..bi + 16];
if !bc7_fast_block(blk, &mut scratch, 16) {
bcdec_rs::bc7(blk, &mut scratch, 16);
}
blit_rgba4(&scratch, out, out_w, out_h, bx * 4, by * 4);
}
}
}
fn decode_bc7_parallel(
data: &[u8],
out: &mut [u8],
out_w: usize,
blocks_x: usize,
blocks_y: usize,
) {
let pitch = out_w * 4;
let strip_bytes = 4 * pitch;
static CORES: std::sync::OnceLock<usize> = std::sync::OnceLock::new();
let cores = *CORES.get_or_init(|| {
std::thread::available_parallelism()
.map(|n| n.get())
.unwrap_or(1)
});
let workers = cores.clamp(1, blocks_y);
let mut ranges: Vec<(usize, usize)> = Vec::with_capacity(workers);
let base = blocks_y / workers;
let extra = blocks_y % workers;
let mut start = 0;
for w in 0..workers {
let len = base + usize::from(w < extra);
ranges.push((start, start + len));
start += len;
}
std::thread::scope(|s| {
let mut rest = out;
let mut consumed_rows = 0usize;
for &(by0, by1) in &ranges {
let row0 = by0 * 4;
debug_assert_eq!(row0, consumed_rows);
let strip_len = (by1 - by0) * strip_bytes;
let (band, tail) = rest.split_at_mut(strip_len);
rest = tail;
consumed_rows = by1 * 4;
s.spawn(move || {
for by in by0..by1 {
let local_y = by - by0;
for bx in 0..blocks_x {
let bi = (by * blocks_x + bx) * 16;
let offset = (local_y * 4 * out_w + bx * 4) * 4;
let (blk, dst) = (&data[bi..bi + 16], &mut band[offset..]);
if !bc7_fast_block(blk, dst, pitch) {
bcdec_rs::bc7(blk, dst, pitch);
}
}
}
});
}
debug_assert!(rest.is_empty());
});
}
fn block_grid(width: u32, height: u32, block_bytes: usize) -> Result<(usize, usize, usize), Error> {
if width == 0 || height == 0 {
return Err(Error::InvalidField("zero image dimension".into()));
}
let blocks_x = (width as usize + 3) / 4;
let blocks_y = (height as usize + 3) / 4;
let expected = blocks_x
.checked_mul(blocks_y)
.and_then(|n| n.checked_mul(block_bytes))
.ok_or(Error::OutOfBounds)?;
Ok((blocks_x, blocks_y, expected))
}
#[inline]
fn blit_rgba4(scratch: &[u8; 64], out: &mut [u8], out_w: usize, out_h: usize, px0: usize, py0: usize) {
let copy_w = 4.min(out_w - px0);
let copy_h = 4.min(out_h - py0);
for row in 0..copy_h {
let src = row * 16;
let dst = ((py0 + row) * out_w + px0) * 4;
out[dst..dst + copy_w * 4].copy_from_slice(&scratch[src..src + copy_w * 4]);
}
}
#[inline]
fn blit_r_to_rgba(block_r: &[u8; 16], out: &mut [u8], out_w: usize, out_h: usize, px0: usize, py0: usize) {
let copy_w = 4.min(out_w - px0);
let copy_h = 4.min(out_h - py0);
for row in 0..copy_h {
for col in 0..copy_w {
let v = block_r[row * 4 + col];
let dst = ((py0 + row) * out_w + px0 + col) * 4;
out[dst] = v;
out[dst + 1] = 0;
out[dst + 2] = 0;
out[dst + 3] = 255;
}
}
}
#[inline]
fn blit_rg_to_rgba(
block_rg: &[u8; 32],
out: &mut [u8],
out_w: usize,
out_h: usize,
px0: usize,
py0: usize,
) {
let copy_w = 4.min(out_w - px0);
let copy_h = 4.min(out_h - py0);
for row in 0..copy_h {
for col in 0..copy_w {
let src = row * 8 + col * 2;
let dst = ((py0 + row) * out_w + px0 + col) * 4;
out[dst] = block_rg[src];
out[dst + 1] = block_rg[src + 1];
out[dst + 2] = 0;
out[dst + 3] = 255;
}
}
}
#[inline]
fn bc7_fast_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] == 0 {
return false;
}
match blk[0].trailing_zeros() {
0 => bc7_mode0_block(blk, out, pitch),
1 => bc7_mode1_block(blk, out, pitch),
2 => bc7_mode2_block(blk, out, pitch),
3 => bc7_mode3_block(blk, out, pitch),
4 => bc7_mode4_block(blk, out, pitch),
5 => bc7_mode5_block(blk, out, pitch),
6 => bc7_mode6_block(blk, out, pitch),
7 => bc7_mode7_block(blk, out, pitch),
_ => false,
}
}
const BC7_P2_SUBSET: [u16; 64] = [
0xcccc, 0x8888, 0xeeee, 0xecc8, 0xc880, 0xfeec, 0xfec8, 0xec80,
0xc800, 0xffec, 0xfe80, 0xe800, 0xffe8, 0xff00, 0xfff0, 0xf000,
0xf710, 0x008e, 0x7100, 0x08ce, 0x008c, 0x7310, 0x3100, 0x8cce,
0x088c, 0x3110, 0x6666, 0x366c, 0x17e8, 0x0ff0, 0x718e, 0x399c,
0xaaaa, 0xf0f0, 0x5a5a, 0x33cc, 0x3c3c, 0x55aa, 0x9696, 0xa55a,
0x73ce, 0x13c8, 0x324c, 0x3bdc, 0x6996, 0xc33c, 0x9966, 0x0660,
0x0272, 0x04e4, 0x4e40, 0x2720, 0xc936, 0x936c, 0x39c6, 0x639c,
0x9336, 0x9cc6, 0x817e, 0xe718, 0xccf0, 0x0fcc, 0x7744, 0xee22,
];
const BC7_P2_FIXUP: [u8; 64] = [
15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15, 15,
15, 2, 8, 2, 2, 8, 8, 15, 2, 8, 2, 2, 8, 8, 2, 2,
15, 15, 6, 8, 2, 8, 15, 15, 2, 8, 2, 2, 2, 15, 15, 6,
6, 2, 6, 8, 15, 15, 2, 2, 15, 15, 15, 15, 15, 2, 2, 15,
];
const BC7_WEIGHTS3: [u32; 8] = [0, 9, 18, 27, 37, 46, 55, 64];
const BC7_WEIGHTS2: [u32; 4] = [0, 21, 43, 64];
#[inline(always)]
fn bc7_bd3(e: &[[u32; 3]], pairs: usize) -> [([i32; 3], [i32; 3]); 3] {
let mut out = [([0i32; 3], [0i32; 3]); 3];
for (k, slot) in out.iter_mut().enumerate().take(pairs) {
let (a, c) = (&e[k * 2], &e[k * 2 + 1]);
for i in 0..3 {
slot.0[i] = a[i] as i32 * 64 + 32;
slot.1[i] = c[i] as i32 - a[i] as i32;
}
}
out
}
#[inline(always)]
fn bc7_bd4(e: &[[u32; 4]], pairs: usize) -> [([i32; 4], [i32; 4]); 2] {
let mut out = [([0i32; 4], [0i32; 4]); 2];
for (k, slot) in out.iter_mut().enumerate().take(pairs) {
let (a, c) = (&e[k * 2], &e[k * 2 + 1]);
for i in 0..4 {
slot.0[i] = a[i] as i32 * 64 + 32;
slot.1[i] = c[i] as i32 - a[i] as i32;
}
}
out
}
#[inline(always)]
fn bc7_p2_index_at(p: usize, fixup: usize, bits: u32) -> (u32, u32) {
let short = usize::from(p > 0) + usize::from(p > fixup);
let off = bits as usize * p - short;
let w = bits - u32::from(p == 0 || p == fixup);
(off as u32, w)
}
fn bc7_mode1_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0x3 != 0x2 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let partition = ((b >> 2) & 0x3f) as usize;
let p0 = ((b >> 80) & 1) as u32;
let p1 = ((b >> 81) & 1) as u32;
let ep = |shift: u32, pbit: u32| {
let v = ((((b >> shift) & 0x3f) as u32) << 1) | pbit;
let t = v << 1;
t | (t >> 7)
};
let e = [
[ep(8, p0), ep(32, p0), ep(56, p0)],
[ep(14, p0), ep(38, p0), ep(62, p0)],
[ep(20, p1), ep(44, p1), ep(68, p1)],
[ep(26, p1), ep(50, p1), ep(74, p1)],
];
let subsets = BC7_P2_SUBSET[partition];
let fixup = BC7_P2_FIXUP[partition] as usize;
let idx = (b >> 82) as u64;
let bd = bc7_bd3(&e, 2);
let weight_of = |p: usize| {
let (off, w) = bc7_p2_index_at(p, fixup, 3);
BC7_WEIGHTS3[((idx >> off) & ((1u64 << w) - 1)) as usize]
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let bdp = crate::decode::simd::pack_bd3(&bd, 2);
for p in (0..16usize).step_by(2) {
let (b0, d0) = bdp[((subsets >> p) & 1) as usize];
let q = p + 1;
let (b1, d1) = bdp[((subsets >> q) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2(
b0,
d0,
b1,
d1,
weight_of(p) as i16,
weight_of(q) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let weight = weight_of(p) as i32;
let (base, delta) = &bd[((subsets >> p) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
out[o] = ((base[0] + weight * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + weight * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + weight * delta[2]) >> 6) as u8;
out[o + 3] = 0xff;
}
}
true
}
fn bc7_mode3_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0xf != 0x8 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let partition = ((b >> 4) & 0x3f) as usize;
let ep = |shift: u32, pbit_shift: u32| {
((((b >> shift) & 0x7f) as u32) << 1) | (((b >> pbit_shift) & 1) as u32)
};
let e = [
[ep(10, 94), ep(38, 94), ep(66, 94)],
[ep(17, 95), ep(45, 95), ep(73, 95)],
[ep(24, 96), ep(52, 96), ep(80, 96)],
[ep(31, 97), ep(59, 97), ep(87, 97)],
];
let subsets = BC7_P2_SUBSET[partition];
let fixup = BC7_P2_FIXUP[partition] as usize;
let idx = (b >> 98) as u64;
let bd = bc7_bd3(&e, 2);
let weight_of = |p: usize| {
let (off, w) = bc7_p2_index_at(p, fixup, 2);
BC7_WEIGHTS2[((idx >> off) & ((1u64 << w) - 1)) as usize]
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let bdp = crate::decode::simd::pack_bd3(&bd, 2);
for p in (0..16usize).step_by(2) {
let (b0, d0) = bdp[((subsets >> p) & 1) as usize];
let q = p + 1;
let (b1, d1) = bdp[((subsets >> q) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2(
b0,
d0,
b1,
d1,
weight_of(p) as i16,
weight_of(q) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let weight = weight_of(p) as i32;
let (base, delta) = &bd[((subsets >> p) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
out[o] = ((base[0] + weight * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + weight * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + weight * delta[2]) >> 6) as u8;
out[o + 3] = 0xff;
}
}
true
}
fn bc7_mode7_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] != 0x80 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let partition = ((b >> 8) & 0x3f) as usize;
let ep = |shift: u32, pbit_shift: u32| {
let v = ((((b >> shift) & 0x1f) as u32) << 1) | (((b >> pbit_shift) & 1) as u32);
let t = v << 2;
t | (t >> 6)
};
let e = [
[ep(14, 94), ep(34, 94), ep(54, 94), ep(74, 94)],
[ep(19, 95), ep(39, 95), ep(59, 95), ep(79, 95)],
[ep(24, 96), ep(44, 96), ep(64, 96), ep(84, 96)],
[ep(29, 97), ep(49, 97), ep(69, 97), ep(89, 97)],
];
let subsets = BC7_P2_SUBSET[partition];
let fixup = BC7_P2_FIXUP[partition] as usize;
let idx = (b >> 98) as u64;
let bd = bc7_bd4(&e, 2);
let weight_of = |p: usize| {
let (off, w) = bc7_p2_index_at(p, fixup, 2);
BC7_WEIGHTS2[((idx >> off) & ((1u64 << w) - 1)) as usize]
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let bdp = crate::decode::simd::pack_bd4(&bd, 2);
for p in (0..16usize).step_by(2) {
let (b0, d0) = bdp[((subsets >> p) & 1) as usize];
let q = p + 1;
let (b1, d1) = bdp[((subsets >> q) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2(
b0,
d0,
b1,
d1,
weight_of(p) as i16,
weight_of(q) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let weight = weight_of(p) as i32;
let (base, delta) = &bd[((subsets >> p) & 1) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
out[o] = ((base[0] + weight * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + weight * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + weight * delta[2]) >> 6) as u8;
out[o + 3] = ((base[3] + weight * delta[3]) >> 6) as u8;
}
}
true
}
#[cfg(test)]
mod bc7_mode7_tests {
use super::bc7_mode7_block;
#[test]
fn mode7_matches_the_general_decoder() {
let mut state = 0x0ddc_0ffe_e0dd_f00du64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
for partition in 0..64u8 {
for case in 0..200 {
let mut raw = [0u8; 16];
match case {
0 => {}
1 => raw.iter_mut().for_each(|x| *x = 0xff),
_ => {
let (x, y) = (next(), next());
raw[..8].copy_from_slice(&x.to_le_bytes());
raw[8..].copy_from_slice(&y.to_le_bytes());
}
}
let v = u128::from_le_bytes(raw);
let v = (v & !(0x3fu128 << 8)) | ((partition as u128) << 8);
let mut blk = v.to_le_bytes();
blk[0] = 0x80;
let mut ours = [0u8; 64];
assert!(bc7_mode7_block(&blk, &mut ours, 16), "partition {partition}");
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&blk, &mut theirs, 16);
assert_eq!(
ours, theirs,
"mode 7, partition {partition}, case {case} diverged"
);
}
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
let mut blk = [0u8; 16];
blk[0] = 1u8 << mode;
let mut px = [0u8; 64];
assert_eq!(bc7_mode7_block(&blk, &mut px, 16), mode == 7, "vs mode {mode}");
}
let mut px = [0u8; 64];
assert!(!bc7_mode7_block(&[0u8; 16], &mut px, 16));
}
}
#[inline(always)]
fn bc7_p1_index_at(p: usize, bits: u32) -> (u32, u32) {
let off = bits as usize * p - usize::from(p > 0);
let w = bits - u32::from(p == 0);
(off as u32, w)
}
fn bc7_mode4_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0x1f != 0x10 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let rotation = ((b >> 5) & 0x3) as usize;
let isb = ((b >> 7) & 1) != 0;
let c = |shift: u32| {
let t = (((b >> shift) & 0x1f) as u32) << 3;
t | (t >> 5)
};
let a = |shift: u32| {
let t = (((b >> shift) & 0x3f) as u32) << 2;
t | (t >> 6)
};
let e0 = [c(8), c(18), c(28), a(38)];
let e1 = [c(13), c(23), c(33), a(44)];
let i2 = (b >> 50) as u64;
let i3 = (b >> 81) as u64;
let mut map = [0usize, 1, 2, 3];
if rotation != 0 {
map.swap(3, rotation - 1);
}
let base = [
e0[0] as i32 * 64 + 32,
e0[1] as i32 * 64 + 32,
e0[2] as i32 * 64 + 32,
e0[3] as i32 * 64 + 32,
];
let delta = [
e1[0] as i32 - e0[0] as i32,
e1[1] as i32 - e0[1] as i32,
e1[2] as i32 - e0[2] as i32,
e1[3] as i32 - e0[3] as i32,
];
let weights = |p: usize| {
let (o2, w2) = bc7_p1_index_at(p, 2);
let (o3, w3) = bc7_p1_index_at(p, 3);
let wa = BC7_WEIGHTS2[((i2 >> o2) & ((1u64 << w2) - 1)) as usize];
let wb = BC7_WEIGHTS3[((i3 >> o3) & ((1u64 << w3) - 1)) as usize];
if isb {
(wb, wa)
} else {
(wa, wb)
}
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let (mut bo, mut dobj) = ([0i32; 4], [0i32; 4]);
for k in 0..4 {
bo[map[k]] = base[k];
dobj[map[k]] = delta[k];
}
let (bp, dp) = (
crate::decode::simd::pack4(bo),
crate::decode::simd::pack4(dobj),
);
for p in (0..16usize).step_by(2) {
let (c0, a0) = weights(p);
let (c1, a1) = weights(p + 1);
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2_split(
bp,
dp,
bp,
dp,
(c0 as i16, c1 as i16),
(a0 as i16, a1 as i16),
map[3],
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let (wc, walpha) = weights(p);
let (wc, walpha) = (wc as i32, walpha as i32);
let o = (p / 4) * pitch + (p % 4) * 4;
out[o + map[0]] = ((base[0] + wc * delta[0]) >> 6) as u8;
out[o + map[1]] = ((base[1] + wc * delta[1]) >> 6) as u8;
out[o + map[2]] = ((base[2] + wc * delta[2]) >> 6) as u8;
out[o + map[3]] = ((base[3] + walpha * delta[3]) >> 6) as u8;
}
return true;
}
}
#[cfg(test)]
mod bc7_mode4_tests {
use super::bc7_mode4_block;
#[test]
fn mode4_matches_the_general_decoder() {
let mut state = 0xfeed_face_dead_10ccu64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
for rotation in 0..4u8 {
for isb in 0..2u8 {
for case in 0..3_000 {
let mut raw = [0u8; 16];
match case {
0 => {}
1 => raw.iter_mut().for_each(|x| *x = 0xff),
_ => {
let (x, y) = (next(), next());
raw[..8].copy_from_slice(&x.to_le_bytes());
raw[8..].copy_from_slice(&y.to_le_bytes());
}
}
raw[0] = 0x10 | (rotation << 5) | (isb << 7);
let mut ours = [0u8; 64];
assert!(
bc7_mode4_block(&raw, &mut ours, 16),
"rot {rotation} isb {isb} not recognised"
);
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&raw, &mut theirs, 16);
assert_eq!(
ours, theirs,
"mode 4, rotation {rotation}, isb {isb}, case {case} diverged"
);
}
}
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
for rot in 0..4u8 {
for isb in 0..2u8 {
let mut blk = [0u8; 16];
blk[0] = (1u8 << mode) | (rot << 5) | (isb << 7);
let mut px = [0u8; 64];
let expect = mode == 4;
assert_eq!(
bc7_mode4_block(&blk, &mut px, 16),
expect,
"mode {mode} rot {rot} isb {isb}"
);
}
}
}
}
}
const BC7_P3_SUBSET: [u32; 64] = [
0xaa685050, 0x6a5a5040, 0x5a5a4200, 0x5450a0a8,
0xa5a50000, 0xa0a05050, 0x5555a0a0, 0x5a5a5050,
0xaa550000, 0xaa555500, 0xaaaa5500, 0x90909090,
0x94949494, 0xa4a4a4a4, 0xa9a59450, 0x2a0a4250,
0xa5945040, 0x0a425054, 0xa5a5a500, 0x55a0a0a0,
0xa8a85454, 0x6a6a4040, 0xa4a45000, 0x1a1a0500,
0x0050a4a4, 0xaaa59090, 0x14696914, 0x69691400,
0xa08585a0, 0xaa821414, 0x50a4a450, 0x6a5a0200,
0xa9a58000, 0x5090a0a8, 0xa8a09050, 0x24242424,
0x00aa5500, 0x24924924, 0x24499224, 0x50a50a50,
0x500aa550, 0xaaaa4444, 0x66660000, 0xa5a0a5a0,
0x50a050a0, 0x69286928, 0x44aaaa44, 0x66666600,
0xaa444444, 0x54a854a8, 0x95809580, 0x96969600,
0xa85454a8, 0x80959580, 0xaa141414, 0x96960000,
0xaaaa1414, 0xa05050a0, 0xa0a5a5a0, 0x96000000,
0x40804080, 0xa9a8a9a8, 0xaaaaaa44, 0x2a4a5254,
];
const BC7_P3_ANCHOR: [[u8; 2]; 64] = [
[ 3,15], [ 3, 8], [15, 8], [15, 3], [ 8,15], [ 3,15], [15, 3], [15, 8],
[ 8,15], [ 8,15], [ 6,15], [ 6,15], [ 6,15], [ 5,15], [ 3,15], [ 3, 8],
[ 3,15], [ 3, 8], [ 8,15], [15, 3], [ 3,15], [ 3, 8], [ 6,15], [10, 8],
[ 5, 3], [ 8,15], [ 8, 6], [ 6,10], [ 8,15], [ 5,15], [15,10], [15, 8],
[ 8,15], [15, 3], [ 3,15], [ 5,10], [ 6,10], [10, 8], [ 8, 9], [15,10],
[15, 6], [ 3,15], [15, 8], [ 5,15], [15, 3], [15, 6], [15, 6], [15, 8],
[ 3,15], [15, 3], [ 5,15], [ 5,15], [ 5,15], [ 8,15], [ 5,15], [10,15],
[ 5,15], [10,15], [ 8,15], [13,15], [15, 3], [12,15], [ 3,15], [ 3, 8],
];
#[inline(always)]
fn bc7_p3_index_at(p: usize, anchors: [u8; 2], bits: u32) -> (u32, u32) {
let (a1, a2) = (anchors[0] as usize, anchors[1] as usize);
let short = usize::from(p > 0) + usize::from(p > a1) + usize::from(p > a2);
let off = bits as usize * p - short;
let w = bits - u32::from(p == 0 || p == a1 || p == a2);
(off as u32, w)
}
fn bc7_mode0_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0x1 != 0x1 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let partition = ((b >> 1) & 0xf) as usize;
let ep = |shift: u32, pbit_shift: u32| {
let v = ((((b >> shift) & 0xf) as u32) << 1) | (((b >> pbit_shift) & 1) as u32);
let t = v << 3;
t | (t >> 5)
};
let e = [
[ep(5, 77), ep(29, 77), ep(53, 77)],
[ep(9, 78), ep(33, 78), ep(57, 78)],
[ep(13, 79), ep(37, 79), ep(61, 79)],
[ep(17, 80), ep(41, 80), ep(65, 80)],
[ep(21, 81), ep(45, 81), ep(69, 81)],
[ep(25, 82), ep(49, 82), ep(73, 82)],
];
let subsets = BC7_P3_SUBSET[partition];
let anchors = BC7_P3_ANCHOR[partition];
let idx = (b >> 83) as u64;
let bd = bc7_bd3(&e, 3);
let weight_of = |p: usize| {
let (off, w) = bc7_p3_index_at(p, anchors, 3);
BC7_WEIGHTS3[((idx >> off) & ((1u64 << w) - 1)) as usize]
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let bdp = crate::decode::simd::pack_bd3(&bd, 3);
for p in (0..16usize).step_by(2) {
let (b0, d0) = bdp[((subsets >> (2 * p)) & 0x3) as usize];
let q = p + 1;
let (b1, d1) = bdp[((subsets >> (2 * q)) & 0x3) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2(
b0,
d0,
b1,
d1,
weight_of(p) as i16,
weight_of(q) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let weight = weight_of(p) as i32;
let (base, delta) = &bd[((subsets >> (2 * p)) & 0x3) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
out[o] = ((base[0] + weight * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + weight * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + weight * delta[2]) >> 6) as u8;
out[o + 3] = 0xff;
}
}
true
}
fn bc7_mode2_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0x7 != 0x4 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let partition = ((b >> 3) & 0x3f) as usize;
let ep = |shift: u32| {
let t = (((b >> shift) & 0x1f) as u32) << 3;
t | (t >> 5)
};
let e = [
[ep(9), ep(39), ep(69)],
[ep(14), ep(44), ep(74)],
[ep(19), ep(49), ep(79)],
[ep(24), ep(54), ep(84)],
[ep(29), ep(59), ep(89)],
[ep(34), ep(64), ep(94)],
];
let subsets = BC7_P3_SUBSET[partition];
let anchors = BC7_P3_ANCHOR[partition];
let idx = (b >> 99) as u64;
let bd = bc7_bd3(&e, 3);
let weight_of = |p: usize| {
let (off, w) = bc7_p3_index_at(p, anchors, 2);
BC7_WEIGHTS2[((idx >> off) & ((1u64 << w) - 1)) as usize]
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let bdp = crate::decode::simd::pack_bd3(&bd, 3);
for p in (0..16usize).step_by(2) {
let (b0, d0) = bdp[((subsets >> (2 * p)) & 0x3) as usize];
let q = p + 1;
let (b1, d1) = bdp[((subsets >> (2 * q)) & 0x3) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2(
b0,
d0,
b1,
d1,
weight_of(p) as i16,
weight_of(q) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let weight = weight_of(p) as i32;
let (base, delta) = &bd[((subsets >> (2 * p)) & 0x3) as usize];
let o = (p / 4) * pitch + (p % 4) * 4;
out[o] = ((base[0] + weight * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + weight * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + weight * delta[2]) >> 6) as u8;
out[o + 3] = 0xff;
}
}
true
}
#[cfg(test)]
mod bc7_p3_tests {
use super::{bc7_mode0_block, bc7_mode2_block, BC7_P3_ANCHOR, BC7_P3_SUBSET};
#[test]
fn three_subset_modes_match_the_general_decoder() {
for &(mode, mask, set, pshift, pbits) in
&[(0u32, 0x1u8, 0x1u8, 1u32, 4u32), (2, 0x7, 0x4, 3, 6)]
{
let mut state = 0x1234_9876_abcd_5678u64 ^ mode as u64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
let partitions = 1u32 << pbits;
for partition in 0..partitions {
for case in 0..200 {
let mut raw = [0u8; 16];
match case {
0 => {}
1 => raw.iter_mut().for_each(|x| *x = 0xff),
_ => {
let (x, y) = (next(), next());
raw[..8].copy_from_slice(&x.to_le_bytes());
raw[8..].copy_from_slice(&y.to_le_bytes());
}
}
let pmask = ((1u128 << pbits) - 1) << pshift;
let v = u128::from_le_bytes(raw);
let v = (v & !pmask) | ((partition as u128) << pshift);
let mut blk = v.to_le_bytes();
blk[0] = (blk[0] & !mask) | set;
let mut ours = [0u8; 64];
let claimed = if mode == 0 {
bc7_mode0_block(&blk, &mut ours, 16)
} else {
bc7_mode2_block(&blk, &mut ours, 16)
};
assert!(claimed, "mode {mode} partition {partition} not recognised");
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&blk, &mut theirs, 16);
assert_eq!(
ours, theirs,
"mode {mode}, partition {partition}, case {case} diverged"
);
}
}
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
let mut blk = [0u8; 16];
blk[0] = 1u8 << mode;
let mut px = [0u8; 64];
assert_eq!(bc7_mode0_block(&blk, &mut px, 16), mode == 0, "m0 vs {mode}");
assert_eq!(bc7_mode2_block(&blk, &mut px, 16), mode == 2, "m2 vs {mode}");
}
}
#[test]
fn partition_tables_are_the_spec_tables() {
assert_eq!(BC7_P3_SUBSET[0], 0xaa68_5050);
assert_eq!(BC7_P3_SUBSET[1], 0x6a5a_5040);
assert!(BC7_P3_SUBSET.iter().all(|m| m & 0x3 == 0));
for (i, m) in BC7_P3_SUBSET.iter().enumerate() {
let mut seen = [false; 4];
for p in 0..16 {
seen[((m >> (2 * p)) & 0x3) as usize] = true;
}
assert!(seen[0] && seen[1] && seen[2], "partition {i} misses a subset");
assert!(!seen[3], "partition {i} uses subset 3");
let [a1, a2] = BC7_P3_ANCHOR[i];
assert_eq!((m >> (2 * a1 as u32)) & 0x3, 1, "partition {i} anchor 1");
assert_eq!((m >> (2 * a2 as u32)) & 0x3, 2, "partition {i} anchor 2");
}
}
}
fn bc7_mode5_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] & 0x3f != 0x20 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let rotation = ((b >> 6) & 0x3) as usize;
let c = |shift: u32| {
let t = (((b >> shift) & 0x7f) as u32) << 1;
t | (t >> 7)
};
let a = |shift: u32| ((b >> shift) & 0xff) as u32;
let e0 = [c(8), c(22), c(36), a(50)];
let e1 = [c(15), c(29), c(43), a(58)];
let ci = (b >> 66) as u64;
let ai = (b >> 97) as u64;
let mut map = [0usize, 1, 2, 3];
if rotation != 0 {
map.swap(3, rotation - 1);
}
let base = [
e0[0] as i32 * 64 + 32,
e0[1] as i32 * 64 + 32,
e0[2] as i32 * 64 + 32,
e0[3] as i32 * 64 + 32,
];
let delta = [
e1[0] as i32 - e0[0] as i32,
e1[1] as i32 - e0[1] as i32,
e1[2] as i32 - e0[2] as i32,
e1[3] as i32 - e0[3] as i32,
];
let weights = |p: usize| {
let (off, w) = bc7_p1_index_at(p, 2);
let mask = (1u64 << w) - 1;
(
BC7_WEIGHTS2[((ci >> off) & mask) as usize],
BC7_WEIGHTS2[((ai >> off) & mask) as usize],
)
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let (mut bo, mut dobj) = ([0i32; 4], [0i32; 4]);
for k in 0..4 {
bo[map[k]] = base[k];
dobj[map[k]] = delta[k];
}
let (bp, dp) = (
crate::decode::simd::pack4(bo),
crate::decode::simd::pack4(dobj),
);
for p in (0..16usize).step_by(2) {
let (c0, a0) = weights(p);
let (c1, a1) = weights(p + 1);
let o = (p / 4) * pitch + (p % 4) * 4;
crate::decode::simd::write2_split(
bp,
dp,
bp,
dp,
(c0 as i16, c1 as i16),
(a0 as i16, a1 as i16),
map[3],
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for p in 0..16usize {
let (wc, wa) = weights(p);
let (wc, wa) = (wc as i32, wa as i32);
let o = (p / 4) * pitch + (p % 4) * 4;
out[o + map[0]] = ((base[0] + wc * delta[0]) >> 6) as u8;
out[o + map[1]] = ((base[1] + wc * delta[1]) >> 6) as u8;
out[o + map[2]] = ((base[2] + wc * delta[2]) >> 6) as u8;
out[o + map[3]] = ((base[3] + wa * delta[3]) >> 6) as u8;
}
return true;
}
}
#[cfg(test)]
mod bc7_mode5_tests {
use super::bc7_mode5_block;
#[test]
fn mode5_matches_the_general_decoder() {
let mut state = 0x9e37_79b9_7f4a_7c15u64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
for rotation in 0..4u8 {
for case in 0..10_000 {
let mut raw = [0u8; 16];
match case {
0 => {}
1 => raw.iter_mut().for_each(|x| *x = 0xff),
_ => {
let (x, y) = (next(), next());
raw[..8].copy_from_slice(&x.to_le_bytes());
raw[8..].copy_from_slice(&y.to_le_bytes());
}
}
raw[0] = 0x20 | (rotation << 6);
let mut ours = [0u8; 64];
assert!(bc7_mode5_block(&raw, &mut ours, 16), "rot {rotation}");
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&raw, &mut theirs, 16);
assert_eq!(
ours, theirs,
"mode 5, rotation {rotation}, case {case} diverged"
);
}
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
for rot in 0..4u8 {
let mut blk = [0u8; 16];
blk[0] = (1u8 << mode) | (rot << 6);
let mut px = [0u8; 64];
assert_eq!(
bc7_mode5_block(&blk, &mut px, 16),
mode == 5,
"mode {mode} rot {rot}"
);
}
}
}
}
#[cfg(test)]
mod bc7_p2_tests {
use super::{bc7_mode1_block, bc7_mode3_block, BC7_P2_FIXUP, BC7_P2_SUBSET};
#[test]
fn two_subset_modes_match_the_general_decoder() {
for &(mode, mask, set) in &[(1u32, 0x3u8, 0x2u8), (3, 0xf, 0x8)] {
let mut state = 0xdead_beef_cafe_f00du64 ^ mode as u64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
for partition in 0..64u8 {
for case in 0..200 {
let mut raw = [0u8; 16];
match case {
0 => {}
1 => raw.iter_mut().for_each(|x| *x = 0xff),
_ => {
let (x, y) = (next(), next());
raw[..8].copy_from_slice(&x.to_le_bytes());
raw[8..].copy_from_slice(&y.to_le_bytes());
}
}
let pshift = mode + 1;
let v = u128::from_le_bytes(raw);
let v = (v & !(0x3fu128 << pshift)) | ((partition as u128) << pshift);
let mut blk = v.to_le_bytes();
blk[0] = (blk[0] & !mask) | set;
let mut ours = [0u8; 64];
let claimed = if mode == 1 {
bc7_mode1_block(&blk, &mut ours, 16)
} else {
bc7_mode3_block(&blk, &mut ours, 16)
};
assert!(claimed, "mode {mode} partition {partition} not recognised");
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&blk, &mut theirs, 16);
assert_eq!(
ours, theirs,
"mode {mode}, partition {partition}, case {case} diverged"
);
}
}
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
let mut blk = [0u8; 16];
blk[0] = 1u8 << mode;
let mut px = [0u8; 64];
assert_eq!(bc7_mode1_block(&blk, &mut px, 16), mode == 1, "m1 vs {mode}");
assert_eq!(bc7_mode3_block(&blk, &mut px, 16), mode == 3, "m3 vs {mode}");
}
}
#[test]
fn partition_tables_are_the_spec_tables() {
assert_eq!(BC7_P2_SUBSET[0], 0xcccc);
assert_eq!(BC7_P2_SUBSET[1], 0x8888);
assert_eq!(BC7_P2_SUBSET[2], 0xeeee);
assert!(BC7_P2_SUBSET.iter().all(|m| m & 1 == 0));
assert!(BC7_P2_SUBSET.iter().all(|&m| m != 0));
for (i, &f) in BC7_P2_FIXUP.iter().enumerate() {
assert!((1..=15).contains(&f), "partition {i} anchor {f}");
assert_eq!(
(BC7_P2_SUBSET[i] >> f) & 1,
1,
"partition {i} anchor not in subset 1"
);
}
}
}
const BC7_WEIGHTS4: [u32; 16] = [0, 4, 9, 13, 17, 21, 26, 30, 34, 38, 43, 47, 51, 55, 60, 64];
fn bc7_mode6_block(blk: &[u8], out: &mut [u8], pitch: usize) -> bool {
if blk[0] != 0x40 {
return false;
}
let Ok(bytes) = <[u8; 16]>::try_from(&blk[..16]) else {
return false;
};
let b = u128::from_le_bytes(bytes);
let f = |shift: u32| ((b >> shift) & 0x7f) as u32;
let p0 = ((b >> 63) & 1) as u32;
let p1 = ((b >> 64) & 1) as u32;
let e0 = [
(f(7) << 1) | p0,
(f(21) << 1) | p0,
(f(35) << 1) | p0,
(f(49) << 1) | p0,
];
let e1 = [
(f(14) << 1) | p1,
(f(28) << 1) | p1,
(f(42) << 1) | p1,
(f(56) << 1) | p1,
];
let base = [
(e0[0] as i32) * 64 + 32,
(e0[1] as i32) * 64 + 32,
(e0[2] as i32) * 64 + 32,
(e0[3] as i32) * 64 + 32,
];
let delta = [
e1[0] as i32 - e0[0] as i32,
e1[1] as i32 - e0[1] as i32,
e1[2] as i32 - e0[2] as i32,
e1[3] as i32 - e0[3] as i32,
];
let idx = (b >> 65) as u64;
let weight = |i: usize| {
if i == 0 {
BC7_WEIGHTS4[(idx & 0x7) as usize]
} else {
BC7_WEIGHTS4[((idx >> (3 + (i - 1) * 4)) & 0xf) as usize]
}
};
#[cfg(all(feature = "simd", target_arch = "x86_64"))]
{
let (bp, dp) = (crate::decode::simd::pack4(base), crate::decode::simd::pack4(delta));
for i in (0..16usize).step_by(2) {
let o = (i / 4) * pitch + (i % 4) * 4;
crate::decode::simd::write2(
bp,
dp,
bp,
dp,
weight(i) as i16,
weight(i + 1) as i16,
&mut out[o..o + 8],
);
}
return true;
}
#[cfg(not(all(feature = "simd", target_arch = "x86_64")))]
{
for i in 0..16usize {
let w = weight(i) as i32;
let o = (i / 4) * pitch + (i % 4) * 4;
out[o] = ((base[0] + w * delta[0]) >> 6) as u8;
out[o + 1] = ((base[1] + w * delta[1]) >> 6) as u8;
out[o + 2] = ((base[2] + w * delta[2]) >> 6) as u8;
out[o + 3] = ((base[3] + w * delta[3]) >> 6) as u8;
}
return true;
}
}
#[cfg(test)]
mod bc7_mode6_tests {
use super::{bc7_mode6_block, BC7_WEIGHTS4};
#[test]
fn mode6_matches_the_general_decoder() {
let mut state = 0x243f_6a88_85a3_08d3u64;
let mut next = move || {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
state
};
for case in 0..20_000 {
let mut blk = [0u8; 16];
match case {
0 => {}
1 => blk.iter_mut().for_each(|b| *b = 0xff),
_ => {
let (a, b) = (next(), next());
blk[..8].copy_from_slice(&a.to_le_bytes());
blk[8..].copy_from_slice(&b.to_le_bytes());
}
}
blk[0] = 0x40;
let mut ours = [0u8; 64];
assert!(bc7_mode6_block(&blk, &mut ours, 16), "case {case}: not recognised");
let mut theirs = [0u8; 64];
bcdec_rs::bc7(&blk, &mut theirs, 16);
assert_eq!(
ours, theirs,
"case {case}: mode-6 fast path diverged
block {blk:02x?}"
);
}
}
#[test]
fn other_modes_are_declined() {
for mode in 0..8u32 {
let mut blk = [0u8; 16];
blk[0] = 1 << mode;
let mut px = [0u8; 64];
assert_eq!(
bc7_mode6_block(&blk, &mut px, 16),
mode == 6,
"mode {mode} handled incorrectly"
);
}
let mut px = [0u8; 64];
assert!(!bc7_mode6_block(&[0u8; 16], &mut px, 16));
}
#[test]
fn weights_are_the_spec_table() {
assert_eq!(BC7_WEIGHTS4[0], 0);
assert_eq!(BC7_WEIGHTS4[15], 64);
assert!(BC7_WEIGHTS4.windows(2).all(|w| w[0] < w[1]));
}
}