frust_engine/filters/blur.rs
1//! The Gaussian blur filter: its GPU parameter block, and the pass sequence a
2//! blurred layer renders as.
3//!
4//! Ported from the sparse-strip reference renderer's `vello_hybrid` 0.2.0
5//! `filter.rs`, narrowed to the single-filter path — the multi-primitive filter
6//! *graph* planner is unimplemented upstream too (`vello_common`'s
7//! `PreparedFilter::new` refuses a graph of more than one primitive), so there
8//! is nothing here to port it from. The kernel itself is not re-derived: the
9//! decimation plan, the discrete Gaussian weights and the dimension bookkeeping
10//! all come from [`vello_common::filter::gaussian_blur`], so this tier and the
11//! CPU rasterizer blur with byte-identical kernels.
12//!
13//! ## Two representations of one kernel
14//!
15//! `vello_common` hands over a discrete kernel of up to [`MAX_KERNEL_SIZE`]
16//! weights, which the CPU rasterizer convolves with directly, one multiply per
17//! tap. A GPU has bilinear sampling, so each *pair* of adjacent taps is merged
18//! here into one sample at a fractional offset between them, weighted by their
19//! sum ([`LinearKernel`]) — the same Gaussian at roughly half the samples. See
20//! <https://www.rastergrid.com/blog/2010/09/efficient-gaussian-blur-with-linear-sampling/>.
21//!
22//! ## Every filter's parameter block is the same size
23//!
24//! A filter's parameters reach the fragment stage through one texture the whole
25//! frame's filters are packed into, addressed by a texel offset. Every kind's
26//! block is [`FILTER_SIZE_BYTES`] wide, whatever it carries, which is what lets
27//! that offset be a plain multiple rather than a per-filter lookup — the
28//! reason [`GpuGaussianBlur`] pads to a footprint it does not need (a drop
29//! shadow's is larger) and the reason each block carries a `size_of` assert of
30//! its own. A kind added later carries the same assert or the erasure to
31//! [`GpuFilterData`] stops being sound.
32
33use bytemuck::{Pod, Zeroable};
34use vello_common::filter::gaussian_blur::{DecimationSizer, GaussianBlur};
35use vello_common::filter_effects::EdgeMode;
36use vello_common::geometry::SizeU16;
37
38pub use vello_common::filter::gaussian_blur::MAX_KERNEL_SIZE;
39
40use crate::filters::{FilterPassKind, FilterStep};
41
42/// Transparent padding a decimated pass overdraws around the region it writes.
43///
44/// The blur kernels sample bilinearly, so a tap can reach half a kernel past
45/// the region it is filtering. Reserving that much transparent border is what
46/// lets every kernel skip bounds checks entirely: a tap that lands outside
47/// reads transparent black instead of a stale texel, which is exactly what the
48/// convolution wants there.
49///
50/// Keep this in sync with `FILTER_ATLAS_PADDING` in `shaders/filter.wgsl`.
51pub const FILTER_ATLAS_PADDING: u16 = MAX_KERNEL_SIZE as u16 / 2;
52
53/// Bytes one filter's parameter block occupies in the filter-data texture,
54/// uniform across every filter kind.
55///
56/// Keep this in sync with `FILTER_SIZE_BYTES` in `shaders/filter.wgsl`.
57pub const FILTER_SIZE_BYTES: usize = 48;
58
59/// Words one filter's parameter block occupies.
60const FILTER_SIZE_U32: usize = FILTER_SIZE_BYTES / 4;
61
62/// Bytes per texel of the filter-data texture, which is `Rgba32Uint`.
63const BYTES_PER_TEXEL: usize = 16;
64
65/// Bit of the packed header a drop shadow sets to ask for the unfiltered layer
66/// to be composited back over the shadow.
67///
68/// Unused by the blur, and reserved here rather than left implicit: the
69/// const assertion below is what keeps the blur's own header fields from
70/// growing into it.
71const COMPOSITE_ORIGINAL_SHIFT: u32 = 13;
72
73/// Mask of [`COMPOSITE_ORIGINAL_SHIFT`].
74const COMPOSITE_ORIGINAL_MASK: u32 = 1 << COMPOSITE_ORIGINAL_SHIFT;
75
76const _: () = assert!(
77 size_of::<GpuFilterData>() == FILTER_SIZE_BYTES,
78 "every filter's parameter block is one uniform size, which is what makes the type-erased \
79 block addressable by a plain texel multiple"
80);
81const _: () = assert!(
82 size_of::<GpuGaussianBlur>() == FILTER_SIZE_BYTES,
83 "every filter's parameter block is one uniform size, which is what makes the type-erased \
84 block addressable by a plain texel multiple"
85);
86const _: () = assert!(
87 FILTER_SIZE_BYTES.is_multiple_of(BYTES_PER_TEXEL),
88 "a parameter block that did not fill whole texels could not be addressed by a texel offset"
89);
90const _: () = assert!(
91 pack_blur_header(0x1F, 3, 15, 3) & COMPOSITE_ORIGINAL_MASK == 0,
92 "the blur header's fields must not grow into the drop shadow's composite-original bit"
93);
94
95/// The kind each filter's packed header names, matching the reference's own
96/// numbering so a kind added later needs no renumbering.
97pub mod filter_type {
98 /// An offset filter. Reserved; not served.
99 pub const OFFSET: u32 = 0;
100 /// A flood filter. Reserved; not served.
101 pub const FLOOD: u32 = 1;
102 /// A Gaussian blur.
103 pub const GAUSSIAN_BLUR: u32 = 2;
104 /// A drop shadow. Reserved; not served.
105 pub const DROP_SHADOW: u32 = 3;
106}
107
108/// How a filter samples past the edge of its input, matching the reference's
109/// own numbering.
110pub mod edge_mode {
111 /// Clamp to the edge texel.
112 pub const DUPLICATE: u32 = 0;
113 /// Wrap around to the opposite edge.
114 pub const WRAP: u32 = 1;
115 /// Mirror across the edge.
116 pub const MIRROR: u32 = 2;
117 /// Read transparent black.
118 pub const NONE: u32 = 3;
119}
120
121/// `mode` as the fragment stage reads it.
122#[must_use]
123pub const fn edge_mode_code(mode: EdgeMode) -> u32 {
124 match mode {
125 EdgeMode::Duplicate => edge_mode::DUPLICATE,
126 EdgeMode::Wrap => edge_mode::WRAP,
127 EdgeMode::Mirror => edge_mode::MIRROR,
128 EdgeMode::None => edge_mode::NONE,
129 }
130}
131
132/// The most merged bilinear tap pairs per side a kernel can produce.
133///
134/// Keep this in sync with `MAX_TAPS_PER_SIDE` in `shaders/filters_blur.wgsl`:
135/// the shader's weight and offset accessors read a `vec3<f32>` each, which is
136/// only enough because the decimation plan bounds the kernel at
137/// [`MAX_KERNEL_SIZE`].
138pub const MAX_TAPS_PER_SIDE: usize = (MAX_KERNEL_SIZE / 2).div_ceil(2);
139
140const _: () = assert!(
141 MAX_TAPS_PER_SIDE == 3,
142 "the shader packs the tap weights and offsets into one `vec3<f32>` each"
143);
144
145/// A discrete Gaussian kernel re-expressed for bilinear sampling.
146///
147/// The centre tap is sampled on its own; every other pair of adjacent taps is
148/// merged into a single sample placed between them, so the fragment stage runs
149/// `1 + 2 * n_taps` samples rather than one per kernel entry.
150#[derive(Debug, Clone, Copy, PartialEq)]
151pub struct LinearKernel {
152 /// Weight of the centre tap.
153 pub center_weight: f32,
154 /// Merged weight of each tap pair. Only the first `n_taps` are meaningful.
155 ///
156 /// One side only: the kernel is symmetric, so each entry is applied on both
157 /// sides of the centre.
158 pub weights: [f32; MAX_TAPS_PER_SIDE],
159 /// Fractional offset, in texels from the centre, each tap pair is sampled
160 /// at. Only the first `n_taps` are meaningful.
161 pub offsets: [f32; MAX_TAPS_PER_SIDE],
162 /// How many tap pairs per side are meaningful.
163 pub n_taps: u8,
164}
165
166impl LinearKernel {
167 /// The bilinear form of the first `kernel_size` weights of `kernel`.
168 ///
169 /// A `kernel_size` past [`MAX_KERNEL_SIZE`] is clamped to it rather than
170 /// indexing out of bounds; the decimation plan never produces one, and
171 /// clamping keeps that a fact about the plan rather than a precondition on
172 /// this function (E17).
173 #[must_use]
174 pub fn new(kernel: &[f32; MAX_KERNEL_SIZE], kernel_size: u8) -> Self {
175 let kernel_size = usize::from(kernel_size).min(MAX_KERNEL_SIZE);
176 let radius = kernel_size / 2;
177 let center_weight = kernel.get(radius).copied().unwrap_or(1.0);
178
179 let mut weights = [0.0_f32; MAX_TAPS_PER_SIDE];
180 let mut offsets = [0.0_f32; MAX_TAPS_PER_SIDE];
181 let mut n_taps = 0_usize;
182
183 // Symmetric, so only the positive side is walked.
184 let positive_side = kernel
185 .get(radius.saturating_add(1)..kernel_size)
186 .unwrap_or(&[]);
187 let (pairs, remainder) = positive_side.as_chunks::<2>();
188
189 for (k, &[w1, w2]) in pairs.iter().enumerate() {
190 let merged_weight = w1 + w2;
191 let offset1 = (2 * k + 1) as f32;
192 // The merged sample sits at the weighted mean of the two taps it
193 // stands in for; with no weight at all there is nothing to average,
194 // so it stays on the first of them.
195 let merged_offset = if merged_weight > 0.0 {
196 (w1 * offset1 + w2 * (offset1 + 1.0)) / merged_weight
197 } else {
198 offset1
199 };
200 if let (Some(weight), Some(offset)) = (weights.get_mut(k), offsets.get_mut(k)) {
201 *weight = merged_weight;
202 *offset = merged_offset;
203 n_taps = k.saturating_add(1);
204 }
205 }
206
207 // An odd number of taps on the positive side leaves one over. It is
208 // sampled at whole-texel offset, so bilinear filtering reads it alone.
209 if let [leftover] = remainder
210 && let (Some(weight), Some(offset)) = (weights.get_mut(n_taps), offsets.get_mut(n_taps))
211 {
212 *weight = *leftover;
213 *offset = radius as f32;
214 n_taps = n_taps.saturating_add(1);
215 }
216
217 Self {
218 center_weight,
219 weights,
220 offsets,
221 n_taps: u8::try_from(n_taps).unwrap_or(0),
222 }
223 }
224}
225
226/// The packed header of a blur's parameter block.
227///
228/// See the bit layout documented in `shaders/filter.wgsl`.
229const fn pack_blur_header(
230 filter_type: u32,
231 edge_mode: u32,
232 n_decimations: u32,
233 n_linear_taps: u32,
234) -> u32 {
235 (filter_type & 0x1F)
236 | ((edge_mode & 0x3) << 5)
237 | ((n_decimations & 0xF) << 7)
238 | ((n_linear_taps & 0x3) << 11)
239}
240
241/// A Gaussian blur's parameter block, as the fragment stage reads it.
242///
243/// `_padding` carries the block to [`FILTER_SIZE_BYTES`]; the blur needs none
244/// of it, but a drop shadow's block is larger and every kind shares one stride.
245#[repr(C, align(16))]
246#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
247pub struct GpuGaussianBlur {
248 /// Packed filter kind, edge mode, decimation count and tap count.
249 pub header: u32,
250 /// Weight of the kernel's centre tap.
251 pub center_weight: f32,
252 /// Merged weight of each bilinear tap pair.
253 pub linear_weights: [f32; MAX_TAPS_PER_SIDE],
254 /// Fractional offset of each bilinear tap pair.
255 pub linear_offsets: [f32; MAX_TAPS_PER_SIDE],
256 /// Unused by the blur; present so every filter kind is one stride wide.
257 pub _padding: [u32; 4],
258}
259
260impl From<&GaussianBlur> for GpuGaussianBlur {
261 fn from(blur: &GaussianBlur) -> Self {
262 let kernel = LinearKernel::new(&blur.kernel, blur.kernel_size);
263
264 Self {
265 header: pack_blur_header(
266 filter_type::GAUSSIAN_BLUR,
267 edge_mode_code(blur.edge_mode),
268 // Four bits is room for fifteen halvings, which is more than a
269 // page-sized layer has axes to halve; a plan past that is
270 // truncated by the mask rather than corrupting the fields above
271 // it, and the pass sequence is planned from the plan itself
272 // rather than from the header.
273 u32::try_from(blur.n_decimations).unwrap_or(u32::MAX),
274 u32::from(kernel.n_taps),
275 ),
276 center_weight: kernel.center_weight,
277 linear_weights: kernel.weights,
278 linear_offsets: kernel.offsets,
279 _padding: [0; 4],
280 }
281 }
282}
283
284/// A filter's parameter block with its kind erased, as it is serialized into
285/// the filter-data texture.
286///
287/// Every kind casts to this because every kind is [`FILTER_SIZE_BYTES`] wide;
288/// the fragment stage reads the kind back out of the header.
289#[repr(C, align(16))]
290#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
291pub struct GpuFilterData {
292 data: [u32; FILTER_SIZE_U32],
293}
294
295impl GpuFilterData {
296 /// Texels one parameter block occupies in the filter-data texture — the
297 /// stride a filter's own texel offset is a multiple of.
298 pub const SIZE_TEXELS: u32 = (FILTER_SIZE_BYTES / BYTES_PER_TEXEL) as u32;
299
300 /// The filter kind the header names, one of [`filter_type`]'s constants.
301 #[must_use]
302 pub fn filter_type(&self) -> u32 {
303 self.data.first().copied().unwrap_or_default() & 0x1F
304 }
305
306 /// How many 2x decimation levels the header names, meaningful for a blur.
307 #[must_use]
308 pub fn n_decimations(&self) -> usize {
309 ((self.data.first().copied().unwrap_or_default() >> 7) & 0xF) as usize
310 }
311
312 /// The block's words, in the order they are uploaded.
313 #[must_use]
314 pub fn words(&self) -> &[u32; FILTER_SIZE_U32] {
315 &self.data
316 }
317}
318
319impl From<GpuGaussianBlur> for GpuFilterData {
320 fn from(blur: GpuGaussianBlur) -> Self {
321 bytemuck::cast(blur)
322 }
323}
324
325/// One filter pass's per-instance vertex data.
326///
327/// Must stay byte-compatible with `FilterInstanceData` in
328/// `shaders/filter.wgsl`. Every extent is a `u16` pair packed into one word,
329/// the same packing the strip and copy instances use.
330#[repr(C)]
331#[derive(Debug, Clone, Copy, PartialEq, Pod, Zeroable)]
332pub struct FilterInstanceData {
333 /// Origin of the source region inside the source page.
334 pub source_origin: u32,
335 /// Extent of the source region.
336 pub source_size: u32,
337 /// Origin of the destination region inside the destination page.
338 pub dest_origin: u32,
339 /// Extent of the destination region.
340 pub dest_size: u32,
341 /// Extent of the whole destination page.
342 pub dest_texture_size: u32,
343 /// Texel offset of this filter's parameter block in the filter-data
344 /// texture.
345 pub filter_data_offset: u32,
346 /// Extent of the filter layer before any decimation, which bounds the
347 /// transparent border a decimated pass overdraws.
348 pub original_size: u32,
349 /// Which pass of the filter's sequence this instance runs, one of
350 /// [`FilterPassKind::code`]'s values.
351 pub filter_pass_kind: u32,
352}
353
354impl FilterInstanceData {
355 /// The instance running `step` of a filter whose parameters start at
356 /// `filter_data_offset`, from `source_origin` of the source page into
357 /// `dest_origin` of a destination page of `dest_texture_size`.
358 ///
359 /// `original_size` is the filter layer's own extent, before decimation.
360 #[must_use]
361 pub fn new(
362 step: &FilterStep,
363 filter_data_offset: u32,
364 source_origin: (u16, u16),
365 dest_origin: (u16, u16),
366 dest_texture_size: SizeU16,
367 original_size: SizeU16,
368 ) -> Self {
369 Self {
370 source_origin: pack_u16_pair(source_origin.0, source_origin.1),
371 source_size: pack_u16_pair(step.source.width(), step.source.height()),
372 dest_origin: pack_u16_pair(dest_origin.0, dest_origin.1),
373 dest_size: pack_u16_pair(step.dest.width(), step.dest.height()),
374 dest_texture_size: pack_u16_pair(dest_texture_size.width(), dest_texture_size.height()),
375 filter_data_offset,
376 original_size: pack_u16_pair(original_size.width(), original_size.height()),
377 filter_pass_kind: step.kind.code(),
378 }
379 }
380}
381
382/// Two `u16`s in one word, low half first — the packing every engine instance
383/// layout uses and `unpack_u16_pair` in `shaders/helpers.wgsl` reads.
384#[must_use]
385pub const fn pack_u16_pair(low: u16, high: u16) -> u32 {
386 (low as u32) | ((high as u32) << 16)
387}
388
389/// The passes a blur of `blur` over a layer of `size` renders as, in execution
390/// order.
391///
392/// The shape is the reference's `emit_blur_sequence` followed by its
393/// `ensure_result_in_original`: `n_decimations` halvings, one horizontal and
394/// one vertical convolution at the decimated resolution, then the matching
395/// doublings back. Decimating is what keeps the convolution kernel bounded —
396/// the plan halves until the remaining variance is small enough for a kernel of
397/// at most [`MAX_KERNEL_SIZE`] taps, so a σ of 32 costs more passes rather than
398/// a wider kernel.
399///
400/// Each step names the extent it reads and the extent it writes; a step's
401/// destination is the next step's source, on the opposite page.
402#[must_use]
403pub fn blur_passes(blur: &GaussianBlur, size: SizeU16) -> Vec<FilterStep> {
404 let mut sizer = DecimationSizer::new(size.width(), size.height());
405 let mut steps: Vec<FilterStep> = Vec::new();
406
407 // Counted rather than inferred: `DecimationSizer::upscale` pops the stack
408 // its downscales pushed, so an upscale emitted without a matching downscale
409 // would panic inside it (E17). Pairing them through this counter is what
410 // makes that unreachable by construction rather than by argument.
411 let mut pending_upscales = 0_usize;
412 for _ in 0..blur.n_decimations {
413 let source = current(&sizer);
414 let (width, height) = sizer.downscale();
415 steps.push(FilterStep {
416 kind: FilterPassKind::Downscale,
417 source,
418 dest: SizeU16::from_wh(width, height),
419 });
420 pending_upscales = pending_upscales.saturating_add(1);
421 }
422
423 let decimated = current(&sizer);
424 steps.push(FilterStep {
425 kind: FilterPassKind::BlurH,
426 source: decimated,
427 dest: decimated,
428 });
429 steps.push(FilterStep {
430 kind: FilterPassKind::BlurV,
431 source: decimated,
432 dest: decimated,
433 });
434
435 while pending_upscales > 0 {
436 let source = current(&sizer);
437 let (width, height) = sizer.upscale();
438 steps.push(FilterStep {
439 kind: FilterPassKind::Upscale,
440 source,
441 dest: SizeU16::from_wh(width, height),
442 });
443 pending_upscales = pending_upscales.saturating_sub(1);
444 }
445
446 // The reference's `ensure_result_in_original`. A pass writes the page it
447 // did not read, so an odd-length sequence would leave the result in the
448 // scratch page rather than the one the parent's composite samples. A blur's
449 // sequence is `2 * n_decimations + 2` passes and so always even; this is
450 // what keeps that a checked property rather than an assumed one.
451 if !steps.len().is_multiple_of(2) {
452 let size = current(&sizer);
453 steps.push(FilterStep {
454 kind: FilterPassKind::Copy,
455 source: size,
456 dest: size,
457 });
458 }
459
460 steps
461}
462
463/// The sizer's current extent.
464fn current(sizer: &DecimationSizer) -> SizeU16 {
465 let (width, height) = sizer.current();
466 SizeU16::from_wh(width, height)
467}