Skip to main content

frust_engine/filters/
blur.rs

1//! The Gaussian blur filter: its GPU parameter block, and the pass sequence a
2//! blurred layer renders as.
3//!
4//! Ported from the sparse-strip reference renderer's `vello_hybrid` 0.2.0
5//! `filter.rs`, narrowed to the single-filter path — the multi-primitive filter
6//! *graph* planner is unimplemented upstream too (`vello_common`'s
7//! `PreparedFilter::new` refuses a graph of more than one primitive), so there
8//! is nothing here to port it from. The kernel itself is not re-derived: the
9//! decimation plan, the discrete Gaussian weights and the dimension bookkeeping
10//! all come from [`vello_common::filter::gaussian_blur`], so this tier and the
11//! CPU rasterizer blur with byte-identical kernels.
12//!
13//! ## Two representations of one kernel
14//!
15//! `vello_common` hands over a discrete kernel of up to [`MAX_KERNEL_SIZE`]
16//! weights, which the CPU rasterizer convolves with directly, one multiply per
17//! tap. A GPU has bilinear sampling, so each *pair* of adjacent taps is merged
18//! here into one sample at a fractional offset between them, weighted by their
19//! sum ([`LinearKernel`]) — the same Gaussian at roughly half the samples. See
20//! <https://www.rastergrid.com/blog/2010/09/efficient-gaussian-blur-with-linear-sampling/>.
21//!
22//! ## Every filter's parameter block is the same size
23//!
24//! A filter's parameters reach the fragment stage through one texture the whole
25//! frame's filters are packed into, addressed by a texel offset. Every kind's
26//! block is [`FILTER_SIZE_BYTES`] wide, whatever it carries, which is what lets
27//! that offset be a plain multiple rather than a per-filter lookup — the
28//! reason [`GpuGaussianBlur`] pads to a footprint it does not need (a drop
29//! shadow's is larger) and the reason each block carries a `size_of` assert of
30//! its own. A kind added later carries the same assert or the erasure to
31//! [`GpuFilterData`] stops being sound.
32
33use bytemuck::{Pod, Zeroable};
34use vello_common::filter::gaussian_blur::{DecimationSizer, GaussianBlur};
35use vello_common::filter_effects::EdgeMode;
36use vello_common::geometry::SizeU16;
37
38pub use vello_common::filter::gaussian_blur::MAX_KERNEL_SIZE;
39
40use crate::filters::{FilterPassKind, FilterStep};
41
42/// Transparent padding a decimated pass overdraws around the region it writes.
43///
44/// The blur kernels sample bilinearly, so a tap can reach half a kernel past
45/// the region it is filtering. Reserving that much transparent border is what
46/// lets every kernel skip bounds checks entirely: a tap that lands outside
47/// reads transparent black instead of a stale texel, which is exactly what the
48/// convolution wants there.
49///
50/// Keep this in sync with `FILTER_ATLAS_PADDING` in `shaders/filter.wgsl`.
51pub const FILTER_ATLAS_PADDING: u16 = MAX_KERNEL_SIZE as u16 / 2;
52
53/// Bytes one filter's parameter block occupies in the filter-data texture,
54/// uniform across every filter kind.
55///
56/// Keep this in sync with `FILTER_SIZE_BYTES` in `shaders/filter.wgsl`.
57pub const FILTER_SIZE_BYTES: usize = 48;
58
59/// Words one filter's parameter block occupies.
60const FILTER_SIZE_U32: usize = FILTER_SIZE_BYTES / 4;
61
62/// Bytes per texel of the filter-data texture, which is `Rgba32Uint`.
63const BYTES_PER_TEXEL: usize = 16;
64
65/// Bit of the packed header a drop shadow sets to ask for the unfiltered layer
66/// to be composited back over the shadow.
67///
68/// Unused by the blur, and reserved here rather than left implicit: the
69/// const assertion below is what keeps the blur's own header fields from
70/// growing into it.
71const COMPOSITE_ORIGINAL_SHIFT: u32 = 13;
72
73/// Mask of [`COMPOSITE_ORIGINAL_SHIFT`].
74const COMPOSITE_ORIGINAL_MASK: u32 = 1 << COMPOSITE_ORIGINAL_SHIFT;
75
76const _: () = assert!(
77    size_of::<GpuFilterData>() == FILTER_SIZE_BYTES,
78    "every filter's parameter block is one uniform size, which is what makes the type-erased \
79     block addressable by a plain texel multiple"
80);
81const _: () = assert!(
82    size_of::<GpuGaussianBlur>() == FILTER_SIZE_BYTES,
83    "every filter's parameter block is one uniform size, which is what makes the type-erased \
84     block addressable by a plain texel multiple"
85);
86const _: () = assert!(
87    FILTER_SIZE_BYTES.is_multiple_of(BYTES_PER_TEXEL),
88    "a parameter block that did not fill whole texels could not be addressed by a texel offset"
89);
90const _: () = assert!(
91    pack_blur_header(0x1F, 3, 15, 3) & COMPOSITE_ORIGINAL_MASK == 0,
92    "the blur header's fields must not grow into the drop shadow's composite-original bit"
93);
94
95/// The kind each filter's packed header names, matching the reference's own
96/// numbering so a kind added later needs no renumbering.
97pub mod filter_type {
98    /// An offset filter. Reserved; not served.
99    pub const OFFSET: u32 = 0;
100    /// A flood filter. Reserved; not served.
101    pub const FLOOD: u32 = 1;
102    /// A Gaussian blur.
103    pub const GAUSSIAN_BLUR: u32 = 2;
104    /// A drop shadow. Reserved; not served.
105    pub const DROP_SHADOW: u32 = 3;
106}
107
108/// How a filter samples past the edge of its input, matching the reference's
109/// own numbering.
110pub mod edge_mode {
111    /// Clamp to the edge texel.
112    pub const DUPLICATE: u32 = 0;
113    /// Wrap around to the opposite edge.
114    pub const WRAP: u32 = 1;
115    /// Mirror across the edge.
116    pub const MIRROR: u32 = 2;
117    /// Read transparent black.
118    pub const NONE: u32 = 3;
119}
120
121/// `mode` as the fragment stage reads it.
122#[must_use]
123pub const fn edge_mode_code(mode: EdgeMode) -> u32 {
124    match mode {
125        EdgeMode::Duplicate => edge_mode::DUPLICATE,
126        EdgeMode::Wrap => edge_mode::WRAP,
127        EdgeMode::Mirror => edge_mode::MIRROR,
128        EdgeMode::None => edge_mode::NONE,
129    }
130}
131
132/// The most merged bilinear tap pairs per side a kernel can produce.
133///
134/// Keep this in sync with `MAX_TAPS_PER_SIDE` in `shaders/filters_blur.wgsl`:
135/// the shader's weight and offset accessors read a `vec3<f32>` each, which is
136/// only enough because the decimation plan bounds the kernel at
137/// [`MAX_KERNEL_SIZE`].
138pub const MAX_TAPS_PER_SIDE: usize = (MAX_KERNEL_SIZE / 2).div_ceil(2);
139
140const _: () = assert!(
141    MAX_TAPS_PER_SIDE == 3,
142    "the shader packs the tap weights and offsets into one `vec3<f32>` each"
143);
144
145/// A discrete Gaussian kernel re-expressed for bilinear sampling.
146///
147/// The centre tap is sampled on its own; every other pair of adjacent taps is
148/// merged into a single sample placed between them, so the fragment stage runs
149/// `1 + 2 * n_taps` samples rather than one per kernel entry.
150#[derive(Debug, Clone, Copy, PartialEq)]
151pub struct LinearKernel {
152    /// Weight of the centre tap.
153    pub center_weight: f32,
154    /// Merged weight of each tap pair. Only the first `n_taps` are meaningful.
155    ///
156    /// One side only: the kernel is symmetric, so each entry is applied on both
157    /// sides of the centre.
158    pub weights: [f32; MAX_TAPS_PER_SIDE],
159    /// Fractional offset, in texels from the centre, each tap pair is sampled
160    /// at. Only the first `n_taps` are meaningful.
161    pub offsets: [f32; MAX_TAPS_PER_SIDE],
162    /// How many tap pairs per side are meaningful.
163    pub n_taps: u8,
164}
165
166impl LinearKernel {
167    /// The bilinear form of the first `kernel_size` weights of `kernel`.
168    ///
169    /// A `kernel_size` past [`MAX_KERNEL_SIZE`] is clamped to it rather than
170    /// indexing out of bounds; the decimation plan never produces one, and
171    /// clamping keeps that a fact about the plan rather than a precondition on
172    /// this function (E17).
173    #[must_use]
174    pub fn new(kernel: &[f32; MAX_KERNEL_SIZE], kernel_size: u8) -> Self {
175        let kernel_size = usize::from(kernel_size).min(MAX_KERNEL_SIZE);
176        let radius = kernel_size / 2;
177        let center_weight = kernel.get(radius).copied().unwrap_or(1.0);
178
179        let mut weights = [0.0_f32; MAX_TAPS_PER_SIDE];
180        let mut offsets = [0.0_f32; MAX_TAPS_PER_SIDE];
181        let mut n_taps = 0_usize;
182
183        // Symmetric, so only the positive side is walked.
184        let positive_side = kernel
185            .get(radius.saturating_add(1)..kernel_size)
186            .unwrap_or(&[]);
187        let (pairs, remainder) = positive_side.as_chunks::<2>();
188
189        for (k, &[w1, w2]) in pairs.iter().enumerate() {
190            let merged_weight = w1 + w2;
191            let offset1 = (2 * k + 1) as f32;
192            // The merged sample sits at the weighted mean of the two taps it
193            // stands in for; with no weight at all there is nothing to average,
194            // so it stays on the first of them.
195            let merged_offset = if merged_weight > 0.0 {
196                (w1 * offset1 + w2 * (offset1 + 1.0)) / merged_weight
197            } else {
198                offset1
199            };
200            if let (Some(weight), Some(offset)) = (weights.get_mut(k), offsets.get_mut(k)) {
201                *weight = merged_weight;
202                *offset = merged_offset;
203                n_taps = k.saturating_add(1);
204            }
205        }
206
207        // An odd number of taps on the positive side leaves one over. It is
208        // sampled at whole-texel offset, so bilinear filtering reads it alone.
209        if let [leftover] = remainder
210            && let (Some(weight), Some(offset)) = (weights.get_mut(n_taps), offsets.get_mut(n_taps))
211        {
212            *weight = *leftover;
213            *offset = radius as f32;
214            n_taps = n_taps.saturating_add(1);
215        }
216
217        Self {
218            center_weight,
219            weights,
220            offsets,
221            n_taps: u8::try_from(n_taps).unwrap_or(0),
222        }
223    }
224}
225
226/// The packed header of a blur's parameter block.
227///
228/// See the bit layout documented in `shaders/filter.wgsl`.
229const fn pack_blur_header(
230    filter_type: u32,
231    edge_mode: u32,
232    n_decimations: u32,
233    n_linear_taps: u32,
234) -> u32 {
235    (filter_type & 0x1F)
236        | ((edge_mode & 0x3) << 5)
237        | ((n_decimations & 0xF) << 7)
238        | ((n_linear_taps & 0x3) << 11)
239}
240
241/// A Gaussian blur's parameter block, as the fragment stage reads it.
242///
243/// `_padding` carries the block to [`FILTER_SIZE_BYTES`]; the blur needs none
244/// of it, but a drop shadow's block is larger and every kind shares one stride.
245#[repr(C, align(16))]
246#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
247pub struct GpuGaussianBlur {
248    /// Packed filter kind, edge mode, decimation count and tap count.
249    pub header: u32,
250    /// Weight of the kernel's centre tap.
251    pub center_weight: f32,
252    /// Merged weight of each bilinear tap pair.
253    pub linear_weights: [f32; MAX_TAPS_PER_SIDE],
254    /// Fractional offset of each bilinear tap pair.
255    pub linear_offsets: [f32; MAX_TAPS_PER_SIDE],
256    /// Unused by the blur; present so every filter kind is one stride wide.
257    pub _padding: [u32; 4],
258}
259
260impl From<&GaussianBlur> for GpuGaussianBlur {
261    fn from(blur: &GaussianBlur) -> Self {
262        let kernel = LinearKernel::new(&blur.kernel, blur.kernel_size);
263
264        Self {
265            header: pack_blur_header(
266                filter_type::GAUSSIAN_BLUR,
267                edge_mode_code(blur.edge_mode),
268                // Four bits is room for fifteen halvings, which is more than a
269                // page-sized layer has axes to halve; a plan past that is
270                // truncated by the mask rather than corrupting the fields above
271                // it, and the pass sequence is planned from the plan itself
272                // rather than from the header.
273                u32::try_from(blur.n_decimations).unwrap_or(u32::MAX),
274                u32::from(kernel.n_taps),
275            ),
276            center_weight: kernel.center_weight,
277            linear_weights: kernel.weights,
278            linear_offsets: kernel.offsets,
279            _padding: [0; 4],
280        }
281    }
282}
283
284/// A filter's parameter block with its kind erased, as it is serialized into
285/// the filter-data texture.
286///
287/// Every kind casts to this because every kind is [`FILTER_SIZE_BYTES`] wide;
288/// the fragment stage reads the kind back out of the header.
289#[repr(C, align(16))]
290#[derive(Debug, Clone, Copy, PartialEq, Zeroable, Pod)]
291pub struct GpuFilterData {
292    data: [u32; FILTER_SIZE_U32],
293}
294
295impl GpuFilterData {
296    /// Texels one parameter block occupies in the filter-data texture — the
297    /// stride a filter's own texel offset is a multiple of.
298    pub const SIZE_TEXELS: u32 = (FILTER_SIZE_BYTES / BYTES_PER_TEXEL) as u32;
299
300    /// The filter kind the header names, one of [`filter_type`]'s constants.
301    #[must_use]
302    pub fn filter_type(&self) -> u32 {
303        self.data.first().copied().unwrap_or_default() & 0x1F
304    }
305
306    /// How many 2x decimation levels the header names, meaningful for a blur.
307    #[must_use]
308    pub fn n_decimations(&self) -> usize {
309        ((self.data.first().copied().unwrap_or_default() >> 7) & 0xF) as usize
310    }
311
312    /// The block's words, in the order they are uploaded.
313    #[must_use]
314    pub fn words(&self) -> &[u32; FILTER_SIZE_U32] {
315        &self.data
316    }
317}
318
319impl From<GpuGaussianBlur> for GpuFilterData {
320    fn from(blur: GpuGaussianBlur) -> Self {
321        bytemuck::cast(blur)
322    }
323}
324
325/// One filter pass's per-instance vertex data.
326///
327/// Must stay byte-compatible with `FilterInstanceData` in
328/// `shaders/filter.wgsl`. Every extent is a `u16` pair packed into one word,
329/// the same packing the strip and copy instances use.
330#[repr(C)]
331#[derive(Debug, Clone, Copy, PartialEq, Pod, Zeroable)]
332pub struct FilterInstanceData {
333    /// Origin of the source region inside the source page.
334    pub source_origin: u32,
335    /// Extent of the source region.
336    pub source_size: u32,
337    /// Origin of the destination region inside the destination page.
338    pub dest_origin: u32,
339    /// Extent of the destination region.
340    pub dest_size: u32,
341    /// Extent of the whole destination page.
342    pub dest_texture_size: u32,
343    /// Texel offset of this filter's parameter block in the filter-data
344    /// texture.
345    pub filter_data_offset: u32,
346    /// Extent of the filter layer before any decimation, which bounds the
347    /// transparent border a decimated pass overdraws.
348    pub original_size: u32,
349    /// Which pass of the filter's sequence this instance runs, one of
350    /// [`FilterPassKind::code`]'s values.
351    pub filter_pass_kind: u32,
352}
353
354impl FilterInstanceData {
355    /// The instance running `step` of a filter whose parameters start at
356    /// `filter_data_offset`, from `source_origin` of the source page into
357    /// `dest_origin` of a destination page of `dest_texture_size`.
358    ///
359    /// `original_size` is the filter layer's own extent, before decimation.
360    #[must_use]
361    pub fn new(
362        step: &FilterStep,
363        filter_data_offset: u32,
364        source_origin: (u16, u16),
365        dest_origin: (u16, u16),
366        dest_texture_size: SizeU16,
367        original_size: SizeU16,
368    ) -> Self {
369        Self {
370            source_origin: pack_u16_pair(source_origin.0, source_origin.1),
371            source_size: pack_u16_pair(step.source.width(), step.source.height()),
372            dest_origin: pack_u16_pair(dest_origin.0, dest_origin.1),
373            dest_size: pack_u16_pair(step.dest.width(), step.dest.height()),
374            dest_texture_size: pack_u16_pair(dest_texture_size.width(), dest_texture_size.height()),
375            filter_data_offset,
376            original_size: pack_u16_pair(original_size.width(), original_size.height()),
377            filter_pass_kind: step.kind.code(),
378        }
379    }
380}
381
382/// Two `u16`s in one word, low half first — the packing every engine instance
383/// layout uses and `unpack_u16_pair` in `shaders/helpers.wgsl` reads.
384#[must_use]
385pub const fn pack_u16_pair(low: u16, high: u16) -> u32 {
386    (low as u32) | ((high as u32) << 16)
387}
388
389/// The passes a blur of `blur` over a layer of `size` renders as, in execution
390/// order.
391///
392/// The shape is the reference's `emit_blur_sequence` followed by its
393/// `ensure_result_in_original`: `n_decimations` halvings, one horizontal and
394/// one vertical convolution at the decimated resolution, then the matching
395/// doublings back. Decimating is what keeps the convolution kernel bounded —
396/// the plan halves until the remaining variance is small enough for a kernel of
397/// at most [`MAX_KERNEL_SIZE`] taps, so a σ of 32 costs more passes rather than
398/// a wider kernel.
399///
400/// Each step names the extent it reads and the extent it writes; a step's
401/// destination is the next step's source, on the opposite page.
402#[must_use]
403pub fn blur_passes(blur: &GaussianBlur, size: SizeU16) -> Vec<FilterStep> {
404    let mut sizer = DecimationSizer::new(size.width(), size.height());
405    let mut steps: Vec<FilterStep> = Vec::new();
406
407    // Counted rather than inferred: `DecimationSizer::upscale` pops the stack
408    // its downscales pushed, so an upscale emitted without a matching downscale
409    // would panic inside it (E17). Pairing them through this counter is what
410    // makes that unreachable by construction rather than by argument.
411    let mut pending_upscales = 0_usize;
412    for _ in 0..blur.n_decimations {
413        let source = current(&sizer);
414        let (width, height) = sizer.downscale();
415        steps.push(FilterStep {
416            kind: FilterPassKind::Downscale,
417            source,
418            dest: SizeU16::from_wh(width, height),
419        });
420        pending_upscales = pending_upscales.saturating_add(1);
421    }
422
423    let decimated = current(&sizer);
424    steps.push(FilterStep {
425        kind: FilterPassKind::BlurH,
426        source: decimated,
427        dest: decimated,
428    });
429    steps.push(FilterStep {
430        kind: FilterPassKind::BlurV,
431        source: decimated,
432        dest: decimated,
433    });
434
435    while pending_upscales > 0 {
436        let source = current(&sizer);
437        let (width, height) = sizer.upscale();
438        steps.push(FilterStep {
439            kind: FilterPassKind::Upscale,
440            source,
441            dest: SizeU16::from_wh(width, height),
442        });
443        pending_upscales = pending_upscales.saturating_sub(1);
444    }
445
446    // The reference's `ensure_result_in_original`. A pass writes the page it
447    // did not read, so an odd-length sequence would leave the result in the
448    // scratch page rather than the one the parent's composite samples. A blur's
449    // sequence is `2 * n_decimations + 2` passes and so always even; this is
450    // what keeps that a checked property rather than an assumed one.
451    if !steps.len().is_multiple_of(2) {
452        let size = current(&sizer);
453        steps.push(FilterStep {
454            kind: FilterPassKind::Copy,
455            source: size,
456            dest: size,
457        });
458    }
459
460    steps
461}
462
463/// The sizer's current extent.
464fn current(sizer: &DecimationSizer) -> SizeU16 {
465    let (width, height) = sizer.current();
466    SizeU16::from_wh(width, height)
467}