use bytemuck::{Pod, Zeroable};
use vello_common::filter::gaussian_blur::{DecimationSizer, GaussianBlur};
use vello_common::filter_effects::EdgeMode;
use vello_common::geometry::SizeU16;
pub use vello_common::filter::gaussian_blur::MAX_KERNEL_SIZE;
use crate::filters::{FilterPassKind, FilterStep};
pub const FILTER_ATLAS_PADDING: u16 = MAX_KERNEL_SIZE as u16 / 2;
pub const FILTER_SIZE_BYTES: usize = 48;
const FILTER_SIZE_U32: usize = FILTER_SIZE_BYTES / 4;
const BYTES_PER_TEXEL: usize = 16;
const COMPOSITE_ORIGINAL_SHIFT: u32 = 13;
const COMPOSITE_ORIGINAL_MASK: u32 = 1 << COMPOSITE_ORIGINAL_SHIFT;
const _: () = assert!(
size_of::<GpuFilterData>() == FILTER_SIZE_BYTES,
"every filter's parameter block is one uniform size, which is what makes the type-erased \
block addressable by a plain texel multiple"
);
const _: () = assert!(
size_of::<GpuGaussianBlur>() == FILTER_SIZE_BYTES,
"every filter's parameter block is one uniform size, which is what makes the type-erased \
block addressable by a plain texel multiple"
);
const _: () = assert!(
FILTER_SIZE_BYTES.is_multiple_of(BYTES_PER_TEXEL),
"a parameter block that did not fill whole texels could not be addressed by a texel offset"
);
const _: () = assert!(
pack_blur_header(0x1F, 3, 15, 3) & COMPOSITE_ORIGINAL_MASK == 0,
"the blur header's fields must not grow into the drop shadow's composite-original bit"
);
pub mod filter_type {
pub const OFFSET: u32 = 0;
pub const FLOOD: u32 = 1;
pub const GAUSSIAN_BLUR: u32 = 2;
pub const DROP_SHADOW: u32 = 3;
}
pub mod edge_mode {
pub const DUPLICATE: u32 = 0;
pub const WRAP: u32 = 1;
pub const MIRROR: u32 = 2;
pub const NONE: u32 = 3;
}
#[must_use]
pub const fn edge_mode_code(mode: EdgeMode) -> u32 {
match mode {
EdgeMode::Duplicate => edge_mode::DUPLICATE,
EdgeMode::Wrap => edge_mode::WRAP,
EdgeMode::Mirror => edge_mode::MIRROR,
EdgeMode::None => edge_mode::NONE,
}
}
pub const MAX_TAPS_PER_SIDE: usize = (MAX_KERNEL_SIZE / 2).div_ceil(2);
const _: () = assert!(
MAX_TAPS_PER_SIDE == 3,
"the shader packs the tap weights and offsets into one `vec3<f32>` each"
);
#[derive(Debug, Clone, Copy, PartialEq)]
pub struct LinearKernel {
pub center_weight: f32,
pub weights: [f32; MAX_TAPS_PER_SIDE],
pub offsets: [f32; MAX_TAPS_PER_SIDE],
pub n_taps: u8,
}
impl LinearKernel {
#[must_use]
pub fn new(kernel: &[f32; MAX_KERNEL_SIZE], kernel_size: u8) -> Self {
let kernel_size = usize::from(kernel_size).min(MAX_KERNEL_SIZE);
let radius = kernel_size / 2;
let center_weight = kernel.get(radius).copied().unwrap_or(1.0);
let mut weights = [0.0_f32; MAX_TAPS_PER_SIDE];
let mut offsets = [0.0_f32; MAX_TAPS_PER_SIDE];
let mut n_taps = 0_usize;
let positive_side = kernel
.get(radius.saturating_add(1)..kernel_size)
.unwrap_or(&[]);
let (pairs, remainder) = positive_side.as_chunks::<2>();
for (k, &[w1, w2]) in pairs.iter().enumerate() {
let merged_weight = w1 + w2;
let offset1 = (2 * k + 1) as f32;
let merged_offset = if merged_weight > 0.0 {
(w1 * offset1 + w2 * (offset1 + 1.0)) / merged_weight
} else {
offset1
};
if let (Some(weight), Some(offset)) = (weights.get_mut(k), offsets.get_mut(k)) {
*weight = merged_weight;
*offset = merged_offset;
n_taps = k.saturating_add(1);
}
}
if let [leftover] = remainder
&& let (Some(weight), Some(offset)) = (weights.get_mut(n_taps), offsets.get_mut(n_taps))
{
*weight = *leftover;
*offset = radius as f32;
n_taps = n_taps.saturating_add(1);
}
Self {
center_weight,
weights,
offsets,
n_taps: u8::try_from(n_taps).unwrap_or(0),
}
}
}
const fn pack_blur_header(
filter_type: u32,
edge_mode: u32,
n_decimations: u32,
n_linear_taps: u32,
) -> u32 {
(filter_type & 0x1F)
| ((edge_mode & 0x3) << 5)
| ((n_decimations & 0xF) << 7)
| ((n_linear_taps & 0x3) << 11)
}
#[repr(C, align(16))]
#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
pub struct GpuGaussianBlur {
pub header: u32,
pub center_weight: f32,
pub linear_weights: [f32; MAX_TAPS_PER_SIDE],
pub linear_offsets: [f32; MAX_TAPS_PER_SIDE],
pub _padding: [u32; 4],
}
impl From<&GaussianBlur> for GpuGaussianBlur {
fn from(blur: &GaussianBlur) -> Self {
let kernel = LinearKernel::new(&blur.kernel, blur.kernel_size);
Self {
header: pack_blur_header(
filter_type::GAUSSIAN_BLUR,
edge_mode_code(blur.edge_mode),
u32::try_from(blur.n_decimations).unwrap_or(u32::MAX),
u32::from(kernel.n_taps),
),
center_weight: kernel.center_weight,
linear_weights: kernel.weights,
linear_offsets: kernel.offsets,
_padding: [0; 4],
}
}
}
#[repr(C, align(16))]
#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
pub struct GpuFilterData {
data: [u32; FILTER_SIZE_U32],
}
impl GpuFilterData {
pub const SIZE_TEXELS: u32 = (FILTER_SIZE_BYTES / BYTES_PER_TEXEL) as u32;
#[must_use]
pub fn filter_type(&self) -> u32 {
self.data.first().copied().unwrap_or_default() & 0x1F
}
#[must_use]
pub fn n_decimations(&self) -> usize {
((self.data.first().copied().unwrap_or_default() >> 7) & 0xF) as usize
}
#[must_use]
pub fn words(&self) -> &[u32; FILTER_SIZE_U32] {
&self.data
}
}
impl From<GpuGaussianBlur> for GpuFilterData {
fn from(blur: GpuGaussianBlur) -> Self {
bytemuck::cast(blur)
}
}
#[repr(C)]
#[derive(Debug, Clone, Copy, PartialEq, Pod, Zeroable)]
pub struct FilterInstanceData {
pub source_origin: u32,
pub source_size: u32,
pub dest_origin: u32,
pub dest_size: u32,
pub dest_texture_size: u32,
pub filter_data_offset: u32,
pub original_size: u32,
pub filter_pass_kind: u32,
}
impl FilterInstanceData {
#[must_use]
pub fn new(
step: &FilterStep,
filter_data_offset: u32,
source_origin: (u16, u16),
dest_origin: (u16, u16),
dest_texture_size: SizeU16,
original_size: SizeU16,
) -> Self {
Self {
source_origin: pack_u16_pair(source_origin.0, source_origin.1),
source_size: pack_u16_pair(step.source.width(), step.source.height()),
dest_origin: pack_u16_pair(dest_origin.0, dest_origin.1),
dest_size: pack_u16_pair(step.dest.width(), step.dest.height()),
dest_texture_size: pack_u16_pair(dest_texture_size.width(), dest_texture_size.height()),
filter_data_offset,
original_size: pack_u16_pair(original_size.width(), original_size.height()),
filter_pass_kind: step.kind.code(),
}
}
}
#[must_use]
pub const fn pack_u16_pair(low: u16, high: u16) -> u32 {
(low as u32) | ((high as u32) << 16)
}
#[must_use]
pub fn blur_passes(blur: &GaussianBlur, size: SizeU16) -> Vec<FilterStep> {
let mut sizer = DecimationSizer::new(size.width(), size.height());
let mut steps: Vec<FilterStep> = Vec::new();
let mut pending_upscales = 0_usize;
for _ in 0..blur.n_decimations {
let source = current(&sizer);
let (width, height) = sizer.downscale();
steps.push(FilterStep {
kind: FilterPassKind::Downscale,
source,
dest: SizeU16::from_wh(width, height),
});
pending_upscales = pending_upscales.saturating_add(1);
}
let decimated = current(&sizer);
steps.push(FilterStep {
kind: FilterPassKind::BlurH,
source: decimated,
dest: decimated,
});
steps.push(FilterStep {
kind: FilterPassKind::BlurV,
source: decimated,
dest: decimated,
});
while pending_upscales > 0 {
let source = current(&sizer);
let (width, height) = sizer.upscale();
steps.push(FilterStep {
kind: FilterPassKind::Upscale,
source,
dest: SizeU16::from_wh(width, height),
});
pending_upscales = pending_upscales.saturating_sub(1);
}
if !steps.len().is_multiple_of(2) {
let size = current(&sizer);
steps.push(FilterStep {
kind: FilterPassKind::Copy,
source: size,
dest: size,
});
}
steps
}
fn current(sizer: &DecimationSizer) -> SizeU16 {
let (width, height) = sizer.current();
SizeU16::from_wh(width, height)
}