pub const FILTER: &str = "// Copyright 2024 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/helpers/*.wesl`).\n//\n// Shared, binding-free helper functions: packing, quad geometry, extend\n// modes, encoded-paint accessors, gradient/blur evaluation and atlas\n// sampling. Every entry-point module (`strip.wgsl`, `clear.wgsl`,\n// `copy.wgsl`) is compiled with this file prepended, so the WESL reference\'s\n// `import package::helpers::...` lines are resolved by concatenation rather\n// than by a resolver at build time.\n//\n// Two consequences of that flattening are visible below:\n//\n// 1. Nothing here declares a `@group`/`@binding` global. A texture a helper\n// reads is always a function parameter, so prepending this file never\n// changes an entry-point module\'s derived bind-group layout.\n// 2. Parameters that named a texture or an extend mode in the reference are\n// renamed (`tex`, `mode`) where the reference\'s own module boundary was\n// the only thing keeping them from shadowing an entry-point global or a\n// function declared here.\n\n// Mathematical constants.\nconst PI: f32 = 3.1415926535897932384626433832795028;\nconst TWO_PI: f32 = 2.0 * PI;\n// Tolerance for nearly-zero comparisons. Must match `SCALAR_NEARLY_ZERO` in\n// `vello_common::math`, which the CPU-side rasterizer compares against.\nconst NEARLY_ZERO_TOLERANCE: f32 = 1.0 / 4096.0;\n\n// Extend modes, shared by images and gradients.\nconst EXTEND_PAD: u32 = 0u;\nconst EXTEND_REPEAT: u32 = 1u;\nconst EXTEND_REFLECT: u32 = 2u;\n\n// Image rendering quality.\nconst IMAGE_QUALITY_LOW: u32 = 0u;\nconst IMAGE_QUALITY_MEDIUM: u32 = 1u;\nconst IMAGE_QUALITY_HIGH: u32 = 2u;\n\n// Image source kinds.\nconst IMAGE_SOURCE_ATLAS: u32 = 0u;\nconst IMAGE_SOURCE_EXTERNAL: u32 = 1u;\n\n// Tint modes.\nconst TINT_MODE_ALPHA_MASK: u32 = 0u;\nconst TINT_MODE_MULTIPLY: u32 = 1u;\n\n// Gradient types.\nconst GRADIENT_TYPE_LINEAR: u32 = 0u;\nconst GRADIENT_TYPE_RADIAL: u32 = 1u;\nconst GRADIENT_TYPE_SWEEP: u32 = 2u;\n\n// Radial gradient types.\nconst RADIAL_GRADIENT_TYPE_STANDARD: u32 = 0u;\nconst RADIAL_GRADIENT_TYPE_STRIP: u32 = 1u;\nconst RADIAL_GRADIENT_TYPE_FOCAL: u32 = 2u;\n\n// -----------------------------------------------------------------------------\n// Packing\n// -----------------------------------------------------------------------------\n\nfn unpack_u16_pair(value: u32) -> vec2<u32> {\n return vec2<u32>(value & 0xffffu, value >> 16u);\n}\n\n// -----------------------------------------------------------------------------\n// Texture addressing\n// -----------------------------------------------------------------------------\n\nfn flat_index_to_texture_coord(index: u32, width: u32) -> vec2<u32> {\n return vec2<u32>(index % width, index / width);\n}\n\n// -----------------------------------------------------------------------------\n// Quad geometry\n// -----------------------------------------------------------------------------\n\nfn quad_corner(vertex_index: u32) -> vec2<f32> {\n return vec2<f32>(\n f32(vertex_index & 1u),\n f32(vertex_index >> 1u),\n );\n}\n\nfn pixel_to_ndc(pixel: vec2<f32>, target_size: vec2<f32>) -> vec2<f32> {\n return vec2<f32>(\n pixel.x * 2.0 / target_size.x - 1.0,\n 1.0 - pixel.y * 2.0 / target_size.y,\n );\n}\n\n// -----------------------------------------------------------------------------\n// Strip alpha unpacking\n// -----------------------------------------------------------------------------\n\n// Alpha textures store 16 1-byte alpha values per texel, with each color\n// channel packing the 4 alpha values of a single strip column.\nfn unpack_alphas_from_channel(rgba: vec4<u32>, channel_index: u32) -> u32 {\n switch channel_index {\n case 0u: { return rgba.x; }\n case 1u: { return rgba.y; }\n case 2u: { return rgba.z; }\n case 3u: { return rgba.w; }\n // Fallback, should never happen.\n default: { return rgba.x; }\n }\n}\n\n// -----------------------------------------------------------------------------\n// Extend modes\n// -----------------------------------------------------------------------------\n\nfn extend_mode(t: f32, mode: u32, max: f32) -> f32 {\n switch mode {\n case EXTEND_PAD: {\n return clamp(t, 0.0, max - 1.0);\n }\n case EXTEND_REPEAT: {\n return extend_mode_normalized(t / max, mode) * max;\n }\n case EXTEND_REFLECT, default: {\n return extend_mode_normalized(t / max, mode) * max;\n }\n }\n}\n\nfn extend_mode_normalized(t: f32, mode: u32) -> f32 {\n switch mode {\n case EXTEND_PAD: {\n return clamp(t, 0.0, 1.0);\n }\n case EXTEND_REPEAT: {\n return fract(t);\n }\n case EXTEND_REFLECT, default: {\n return abs(t - 2.0 * round(0.5 * t));\n }\n }\n}\n\n// -----------------------------------------------------------------------------\n// Encoded gradient accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// Sample from the gradient LUT texture at the calculated position.\nfn sample_gradient_lut(\n tex: texture_2d<f32>,\n t_value: f32,\n mode: u32,\n gradient_start: u32,\n texture_width: u32,\n) -> vec4<f32> {\n // Apply the extend mode to t_value.\n let clamped_t = extend_mode_normalized(t_value, mode);\n // Convert t_value to a texture coordinate.\n let t_offset = u32(clamped_t * f32(texture_width - 1u));\n // Absolute position in the flat gradient texture.\n let flat_coord = gradient_start + t_offset;\n let gradient_tex_width = textureDimensions(tex).x;\n let texture_coord = flat_index_to_texture_coord(flat_coord, gradient_tex_width);\n return textureLoad(tex, texture_coord, 0);\n}\n\n// Width of the gradient\'s own ramp, in texels.\nfn get_gradient_texture_width(texel0: vec4<u32>) -> u32 { return texel0.x & 0x0FFFFFFFu; }\n\n// The extend mode for the gradient.\nfn get_gradient_extend_mode(texel0: vec4<u32>) -> u32 { return (texel0.x >> 30u) & 3u; }\n\n// Start coordinate in the flat gradient texture.\nfn get_gradient_start(texel0: vec4<u32>) -> u32 { return texel0.y; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_gradient_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w)),\n vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_gradient_translate(texel1: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.z), bitcast<f32>(texel1.w));\n}\n\nfn apply_gradient_transform(\n texel0: vec4<u32>,\n texel1: vec4<u32>,\n fragment_pos: vec2<f32>,\n) -> vec2<f32> {\n return get_gradient_transform(texel0, texel1) * fragment_pos + get_gradient_translate(texel1);\n}\n\n// Kind of radial gradient (0=Radial, 1=Strip, 2=Focal).\nfn get_radial_kind(texel2: vec4<u32>) -> u32 { return texel2.x & 0x3u; }\n\n// Whether the focal point is swapped for the radial gradient (0=false, 1=true).\nfn get_radial_f_is_swapped(texel2: vec4<u32>) -> u32 { return (texel2.x >> 2u) & 1u; }\n\n// Bias value for radial gradient calculation.\nfn get_radial_bias(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Scale factor for radial gradient calculation.\nfn get_radial_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\n// Focal point 0 parameter for radial gradient.\nfn get_radial_fp0(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\n// Focal point 1 parameter for radial gradient.\nfn get_radial_fp1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\n// Focal radius 1 parameter for radial gradient.\nfn get_radial_fr1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\n// Focal X coordinate for radial gradient.\nfn get_radial_f_focal_x(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\n// Scaled radius 0 squared parameter for the radial gradient strip kind.\nfn get_radial_scaled_r0_squared(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\n// Starting angle for sweep gradient (in radians).\nfn get_sweep_start_angle(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\n// Inverse of angle delta for sweep gradient.\nfn get_sweep_inv_angle_delta(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Fast polynomial approximation for xy_to_unit_angle from Skia.\n// Returns an angle in the [0, 1) range representing [0, 2*PI).\n// See: https://github.com/google/skia/blob/30bba741989865c157c7a997a0caebe94921276b/src/opts/SkRasterPipeline_opts.h#L5859\nfn xy_to_unit_angle(x: f32, y: f32) -> f32 {\n let xabs = abs(x);\n let yabs = abs(y);\n let slope = min(xabs, yabs) / max(xabs, yabs);\n let s = slope * slope;\n // A 7th degree polynomial approximating atan, generated with\n // sollya.gforge.inria.fr via\n // P1 = fpminimax((1/(2*Pi))*atan(x),[|1,3,5,7|],[|24...|],[2^(-40),1],relative);\n var phi = slope * (0.15912117063999176025390625 + s * (-5.185396969318389892578125e-2 + s * (2.476101927459239959716796875e-2 + s * (-7.0547382347285747528076171875e-3))));\n // Map from the first octant to the full circle using quadrant information.\n // Handle the [0, 90] degree range.\n phi = select(phi, 0.25 - phi, xabs < yabs);\n // Handle the [90, 180] degree range.\n phi = select(phi, 0.5 - phi, x < 0.0);\n // Handle the [180, 360] degree range.\n phi = select(phi, 1.0 - phi, y < 0.0);\n // Handle NaN cases (using the property that NaN != NaN).\n phi = select(phi, 0.0, phi != phi);\n return phi;\n}\n\n// Calculate a radial gradient; matches the CPU rasterizer\'s implementation.\n// Returns [t_value, validity], where validity is 0.0 for a sample that has no\n// gradient coverage at all.\nfn calculate_radial_gradient(\n grad_pos: vec2<f32>,\n texel2: vec4<u32>,\n texel3: vec4<u32>,\n) -> vec2<f32> {\n let x_pos = grad_pos.x;\n let y_pos = grad_pos.y;\n\n var t_value: f32;\n var is_valid: bool;\n let kind = get_radial_kind(texel2);\n\n switch kind {\n case RADIAL_GRADIENT_TYPE_STANDARD: {\n // Standard radial gradient: bias + scale * sqrt(x^2 + y^2).\n let radius = sqrt(x_pos * x_pos + y_pos * y_pos);\n t_value = get_radial_bias(texel2) + get_radial_scale(texel2) * radius;\n // Radial gradients are always valid.\n is_valid = true;\n }\n case RADIAL_GRADIENT_TYPE_STRIP: {\n // Strip gradient: x + sqrt(scaled_r0_squared - y^2).\n let p1 = get_radial_scaled_r0_squared(texel3) - y_pos * y_pos;\n // Invalid if negative under the square root.\n is_valid = p1 >= 0.0;\n if is_valid {\n t_value = x_pos + sqrt(p1);\n } else {\n // Value doesn\'t matter when invalid.\n t_value = 0.0;\n }\n }\n case RADIAL_GRADIENT_TYPE_FOCAL, default: {\n var t = 0.0;\n let fp0 = get_radial_fp0(texel2);\n let fp1 = get_radial_fp1(texel3);\n let fr1 = get_radial_fr1(texel3);\n let f_focal_x = get_radial_f_focal_x(texel3);\n let is_swapped = get_radial_f_is_swapped(texel2);\n\n // Focal flags, derived from the encoded field values.\n let is_focal_on_circle = abs(1.0 - fr1) <= NEARLY_ZERO_TOLERANCE;\n let is_well_behaved = !is_focal_on_circle && fr1 > 1.0;\n let is_natively_focal = abs(f_focal_x) <= NEARLY_ZERO_TOLERANCE;\n\n // Start with the valid assumption.\n is_valid = true;\n\n if is_focal_on_circle {\n t = x_pos + y_pos * y_pos / x_pos;\n // Check for division by zero and negative t.\n is_valid = t >= 0.0 && x_pos != 0.0;\n } else if is_well_behaved {\n t = sqrt(x_pos * x_pos + y_pos * y_pos) - x_pos * fp0;\n } else {\n // For non-well-behaved gradients, check whether the\n // calculation is valid.\n let xx = x_pos * x_pos;\n let yy = y_pos * y_pos;\n let discriminant = xx - yy;\n\n if is_swapped != 0u || (1.0 - f_focal_x < 0.0) {\n t = -sqrt(discriminant) - x_pos * fp0;\n } else {\n t = sqrt(discriminant) - x_pos * fp0;\n }\n\n // Invalid if the discriminant is negative or t is negative.\n is_valid = discriminant >= 0.0 && t >= 0.0;\n }\n\n // Apply the additional focal transforms only if still valid.\n if is_valid {\n if 1.0 - f_focal_x < 0.0 {\n t = -t;\n }\n\n if !is_natively_focal {\n t = t + fp1;\n }\n\n if is_swapped != 0u {\n t = 1.0 - t;\n }\n }\n\n t_value = t;\n }\n }\n\n return vec2<f32>(t_value, select(0.0, 1.0, is_valid));\n}\n\n// -----------------------------------------------------------------------------\n// Encoded blurred-rounded-rect accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_blurred_rounded_rect_transform(texel0: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.x), bitcast<f32>(texel0.y)),\n vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_blurred_rounded_rect_translate(texel1: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y));\n}\n\n// Premultiplied rectangle color.\nfn get_blurred_rounded_rect_color(texel1: vec4<u32>) -> vec4<f32> { return unpack4x8unorm(texel1.z); }\n\n// Whether to paint the inverse (`1 - alpha`) of the blur coverage.\nfn get_blurred_rounded_rect_invert(texel1: vec4<u32>) -> u32 { return texel1.w; }\n\nfn get_blurred_rounded_rect_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\nfn get_blurred_rounded_rect_recip_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\nfn get_blurred_rounded_rect_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\nfn get_blurred_rounded_rect_std_dev_inv(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\nfn get_blurred_rounded_rect_min_edge(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\nfn get_blurred_rounded_rect_w(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\nfn get_blurred_rounded_rect_h(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\nfn get_blurred_rounded_rect_r1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\nfn get_blurred_rounded_rect_width(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.x); }\n\nfn get_blurred_rounded_rect_height(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.y); }\n\n// Approximation to erf, matching the CPU rasterizer\'s blur painter.\nfn erf7(x: f32) -> f32 {\n let y = clamp(x * 1.1283791671, -100.0, 100.0);\n let yy = y * y;\n let z = y + (0.24295 + (0.03395 + 0.0104 * yy) * yy) * (y * yy);\n return z / sqrt(1.0 + z * z);\n}\n\n// Approximation for the convolution of a gaussian filter with a rounded\n// rectangle, modelled after the CPU rasterizer\'s blurred-rounded-rect painter.\nfn calculate_blurred_rounded_rect(\n fragment_pos: vec2<f32>,\n texel0: vec4<u32>,\n texel1: vec4<u32>,\n texel2: vec4<u32>,\n texel3: vec4<u32>,\n texel4: vec4<u32>,\n) -> vec4<f32> {\n let transform = get_blurred_rounded_rect_transform(texel0);\n let translate = get_blurred_rounded_rect_translate(texel1);\n let color = get_blurred_rounded_rect_color(texel1);\n let invert = get_blurred_rounded_rect_invert(texel1);\n let exponent = get_blurred_rounded_rect_exponent(texel2);\n let recip_exponent = get_blurred_rounded_rect_recip_exponent(texel2);\n let scale = get_blurred_rounded_rect_scale(texel2);\n let std_dev_inv = get_blurred_rounded_rect_std_dev_inv(texel2);\n let min_edge = get_blurred_rounded_rect_min_edge(texel3);\n let w = get_blurred_rounded_rect_w(texel3);\n let h = get_blurred_rounded_rect_h(texel3);\n let r1 = get_blurred_rounded_rect_r1(texel3);\n let width = get_blurred_rounded_rect_width(texel4);\n let height = get_blurred_rounded_rect_height(texel4);\n\n let local_xy = transform * fragment_pos + translate;\n // The 0.5 and 0.0 constants correspond to the CPU painter\'s v1 and v0.\n let y = local_xy.y - 0.5 * height;\n let y0 = r1 + abs(y) - 0.5 * h;\n let y1 = max(y0, 0.0);\n\n let x = local_xy.x - 0.5 * width;\n let x0 = r1 + abs(x) - 0.5 * w;\n let x1 = max(x0, 0.0);\n\n let d_pos = pow(\n pow(x1, exponent) + pow(y1, exponent),\n recip_exponent,\n );\n let d_neg = min(max(x0, y0), 0.0);\n let d = d_pos + d_neg - r1;\n let blur_coverage = scale * (\n erf7(std_dev_inv * (min_edge + d)) -\n erf7(std_dev_inv * d)\n );\n\n // Invert alpha when the `invert` flag is set.\n let blur_alpha = select(blur_coverage, 1.0 - blur_coverage, invert != 0u);\n\n return color * blur_alpha;\n}\n\n// -----------------------------------------------------------------------------\n// Encoded image accessors\n// -----------------------------------------------------------------------------\n\n// Encoded image layout. Must match `GpuEncodedImage` in `gpu::paint_texture`.\n//\n// texel0.x: image_params\n// bits 0-1: quality\n// bits 2-3: extend_x\n// bits 4-5: extend_y\n// bits 6-13: atlas_index\n// bit 14: source_kind (0=atlas, 1=external texture)\n// texel0.y: image_size, packed as [width:16, height:16]\n// texel0.z: image_offset, packed as [x:16, y:16]\n// texel0.w/texel1.x/texel1.y/texel1.z: transform matrix [a, b, c, d]\n// texel1.w/texel2.x: translation [tx, ty]\n// texel2.y: premultiplied tint color packed as RGBA8 unorm\n// texel2.z: tint mode\n// texel2.w: transparent padding pixels around the image in the atlas\n\n// The rendering quality of the image.\nfn get_image_quality(texel0: vec4<u32>) -> u32 { return texel0.x & 0x3u; }\n\n// The extend modes in the horizontal and vertical direction.\nfn get_image_extend_modes(texel0: vec4<u32>) -> vec2<u32> {\n return vec2<u32>((texel0.x >> 2u) & 0x3u, (texel0.x >> 4u) & 0x3u);\n}\n\n// The size of the image in pixels.\nfn get_image_size(texel0: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(f32(texel0.y >> 16u), f32(texel0.y & 0xFFFFu));\n}\n\n// The offset of the image in pixels.\nfn get_image_offset(texel0: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(f32(texel0.z >> 16u), f32(texel0.z & 0xFFFFu));\n}\n\n// The atlas index containing this image.\nfn get_image_atlas_index(texel0: vec4<u32>) -> u32 { return (texel0.x >> 6u) & 0xFFu; }\n\n// Whether the image is sourced from the atlas or the externally bound texture.\nfn get_image_source_kind(texel0: vec4<u32>) -> u32 { return (texel0.x >> 14u) & 0x1u; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_image_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.w), bitcast<f32>(texel1.x)),\n vec2<f32>(bitcast<f32>(texel1.y), bitcast<f32>(texel1.z))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_image_translate(texel1: vec4<u32>, texel2: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.w), bitcast<f32>(texel2.x));\n}\n\n// Number of transparent padding pixels around the image in the atlas.\nfn get_image_padding(texel2: vec4<u32>) -> f32 { return f32(texel2.w); }\n\n// -----------------------------------------------------------------------------\n// Atlas-array sampling\n// -----------------------------------------------------------------------------\n\n// Bilinear filtering: sample the 4 surrounding texels of the target point and\n// interpolate them with a bilinear filter.\nfn bilinear_sample(\n tex: texture_2d_array<f32>,\n coords: vec2<f32>,\n atlas_idx: i32,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n _extend_modes: vec2<u32>,\n _image_padding: f32,\n) -> vec4<f32> {\n let atlas_max = image_offset + image_size - vec2(1.0);\n let atlas_uv_clamped = clamp(coords, image_offset, atlas_max);\n let uv_quad = vec4(floor(atlas_uv_clamped), ceil(atlas_uv_clamped));\n let uv_frac = fract(coords);\n let a = textureLoad(tex, vec2<i32>(uv_quad.xy), atlas_idx, 0);\n let b = textureLoad(tex, vec2<i32>(uv_quad.xw), atlas_idx, 0);\n let c = textureLoad(tex, vec2<i32>(uv_quad.zy), atlas_idx, 0);\n let d = textureLoad(tex, vec2<i32>(uv_quad.zw), atlas_idx, 0);\n return mix(mix(a, b, uv_frac.y), mix(c, d, uv_frac.y), uv_frac.x);\n}\n\n// Bicubic filtering with a Mitchell filter (B=1/3, C=1/3): sample the 16\n// surrounding texels of the target point and interpolate them with a cubic\n// filter. The 4x4 matrix holds the coefficients of the cubic function used to\n// derive the weights from the fractional part of the sample location.\nfn bicubic_sample(\n tex: texture_2d_array<f32>,\n coords: vec2<f32>,\n atlas_idx: i32,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n _extend_modes: vec2<u32>,\n _image_padding: f32,\n) -> vec4<f32> {\n let atlas_max = image_offset + image_size - vec2(1.0);\n let frac_coords = fract(coords + 0.5);\n // Cubic weights for the x and y directions.\n let cx = cubic_weights(frac_coords.x);\n let cy = cubic_weights(frac_coords.y);\n\n // Sample the 4x4 grid around `coords`.\n let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n // Interpolate in the x direction for each row.\n let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n // Interpolate in the y direction.\n let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n // Clamp alpha first, then clamp the premultiplied color channels against it.\n let a = clamp(result.a, 0.0, 1.0);\n return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\n// Mitchell-Netravali cubic filter coefficients with B=1/3 and C=1/3, matching\n// the CPU rasterizer\'s cubic resampler.\nconst MF: array<vec4<f32>, 4> = array<vec4<f32>, 4>(\n vec4<f32>(\n (1.0 / 6.0) / 3.0,\n -(3.0 / 6.0) / 3.0 - 1.0 / 3.0,\n (3.0 / 6.0) / 3.0 + 2.0 * 1.0 / 3.0,\n -(1.0 / 6.0) / 3.0 - 1.0 / 3.0\n ),\n vec4<f32>(\n 1.0 - (2.0 / 6.0) / 3.0,\n 0.0,\n -3.0 + (12.0 / 6.0) / 3.0 + 1.0 / 3.0,\n 2.0 - (9.0 / 6.0) / 3.0 - 1.0 / 3.0\n ),\n vec4<f32>(\n (1.0 / 6.0) / 3.0,\n (3.0 / 6.0) / 3.0 + 1.0 / 3.0,\n 3.0 - (15.0 / 6.0) / 3.0 - 2.0 * 1.0 / 3.0,\n -2.0 + (9.0 / 6.0) / 3.0 + 1.0 / 3.0\n ),\n vec4<f32>(\n 0.0,\n 0.0,\n -1.0 / 3.0,\n (1.0 / 6.0) / 3.0 + 1.0 / 3.0\n )\n);\n\n// The four cubic weights for a single fractional value.\nfn cubic_weights(fract: f32) -> vec4<f32> {\n return vec4<f32>(\n single_weight(fract, MF[0][0], MF[0][1], MF[0][2], MF[0][3]),\n single_weight(fract, MF[1][0], MF[1][1], MF[1][2], MF[1][3]),\n single_weight(fract, MF[2][0], MF[2][1], MF[2][2], MF[2][3]),\n single_weight(fract, MF[3][0], MF[3][1], MF[3][2], MF[3][3])\n );\n}\n\n// One weight from the fractional value t and the cubic coefficients.\nfn single_weight(t: f32, a: f32, b: f32, c: f32, d: f32) -> f32 {\n return t * (t * (t * d + c) + b) + a;\n}\n\n// -----------------------------------------------------------------------------\n// External-texture sampling\n// -----------------------------------------------------------------------------\n\n// The atlas-array samplers above, restated for a plain 2D texture: an\n// externally bound image is a whole texture rather than a page of the array.\n\nfn external_bilinear_sample(\n tex: texture_2d<f32>,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n let image_max = image_offset + image_size - vec2(1.0);\n let clamped_coords = clamp(coords, image_offset, image_max);\n let coord_quad = vec4(floor(clamped_coords), ceil(clamped_coords));\n let coord_frac = fract(coords);\n let a = textureLoad(tex, vec2<i32>(coord_quad.xy), 0);\n let b = textureLoad(tex, vec2<i32>(coord_quad.xw), 0);\n let c = textureLoad(tex, vec2<i32>(coord_quad.zy), 0);\n let d = textureLoad(tex, vec2<i32>(coord_quad.zw), 0);\n return mix(mix(a, b, coord_frac.y), mix(c, d, coord_frac.y), coord_frac.x);\n}\n\nfn external_bicubic_sample(\n tex: texture_2d<f32>,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n let image_max = image_offset + image_size - vec2(1.0);\n let frac_coords = fract(coords + 0.5);\n let cx = cubic_weights(frac_coords.x);\n let cy = cubic_weights(frac_coords.y);\n\n let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, image_max)), 0);\n let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, image_max)), 0);\n let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, image_max)), 0);\n let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, image_max)), 0);\n\n let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, image_max)), 0);\n let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, image_max)), 0);\n let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, image_max)), 0);\n let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, image_max)), 0);\n\n let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, image_max)), 0);\n let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, image_max)), 0);\n let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, image_max)), 0);\n let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, image_max)), 0);\n\n let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, image_max)), 0);\n let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, image_max)), 0);\n let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, image_max)), 0);\n let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, image_max)), 0);\n\n let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n // Clamp alpha first, then clamp the premultiplied color channels against it.\n let a = clamp(result.a, 0.0, 1.0);\n return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\nfn sample_external_image(\n tex: texture_2d<f32>,\n quality: u32,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n if quality == IMAGE_QUALITY_HIGH {\n return external_bicubic_sample(tex, coords, image_offset, image_size);\n }\n if quality == IMAGE_QUALITY_MEDIUM {\n return external_bilinear_sample(\n tex,\n coords - vec2(0.5),\n image_offset,\n image_size,\n );\n }\n return textureLoad(tex, vec2<u32>(coords), 0);\n}\n// Copyright 2026 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/filter.wesl`, the\n// `downscale`/`upscale`/`convolve` half the reference splits out as\n// `filters/scale.wesl` and `filters/blur.wesl`). Its\n// `import package::helpers::...` lines are resolved by prepending\n// `helpers.wgsl` at load time (`crate::filters::FILTER`).\n//\n// The Gaussian-blur kernels: the two rescaling steps a decimated blur is built\n// from, and the separable convolution run between them. Each one is a pure\n// function over the textures it is handed.\n//\n// Like `helpers.wgsl`, this file declares no `@group`/`@binding` global \u{2014} a\n// kernel that reads a texture takes the texture and the sampler as parameters,\n// so prepending it to `filter.wgsl` cannot disturb that module\'s derived\n// bind-group layout.\n//\n// All three kernels sample through GPU-native bilinear filtering rather than\n// averaging texel loads, which is what lets a decimation step cost four samples\n// instead of sixteen and a convolution tap cost one sample instead of two.\n// Every one of those samples goes through `sample_region_bilinear`, which is\n// what holds the kernels\' one invariant: a tap outside the region being\n// filtered reads transparent black, whatever the sampler\'s address mode would\n// have given it there.\n//\n// `textureSampleLevel` rather than `textureSample` throughout: the convolution\n// loop runs a dynamic number of iterations, and an implicit-derivative sample\n// inside non-uniform control flow is not accepted by the Direct3D backend.\n\n// One bilinear sample of the source region, with every tap outside the region\n// reading transparent black.\n//\n// `rel` is the sample position in texels, relative to the region\'s own origin,\n// at texel centres; the region spans `[0, source_size)` from there.\n//\n// The kernels cannot leave this to the sampler. The far side of a region is\n// already transparent \u{2014} a filter round clears its whole destination page, and\n// a page is sized a padding border wider than the region it holds \u{2014} but the\n// *near* side is not: a filter layer\'s region sits at its page\'s own origin,\n// so a tap below zero leaves the texture entirely and `ClampToEdge` answers it\n// with the region\'s own edge texel. That is the region\'s real content whenever\n// the layer\'s expanded bounds were cut off at device zero (a full-screen\n// backdrop blur), and replicating it there paints a smear the CPU rasterizer\'s\n// `EdgeMode::None` never produces.\n//\n// Both halves are exact rather than approximate, which is what lets the same\n// helper serve a tap that straddles the boundary as well as one wholly outside\n// it. Clamping the sample to the region\'s last texel centre turns a straddling\n// tap into a plain texel read of the one texel of the pair that is inside the\n// region; the per-axis coverage below is that texel\'s own bilinear weight, so\n// the product is what the sample would have been had the outside texel held\n// transparent black. Inside the region both factors are exactly one and the\n// clamp is a no-op, so an interior tap is bit-for-bit the sample it always was.\nfn sample_region_bilinear(\n source_texture: texture_2d<f32>,\n linear_sampler: sampler,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n rel: vec2<f32>,\n) -> vec4<f32> {\n let extent = vec2<f32>(source_size);\n // `max(.., 0)` for the degenerate empty region, whose last texel centre\n // would otherwise sit before its first; the coverage below is zero there\n // anyway, so the clamp only has to stay in range.\n let last = max(extent - vec2(1.0), vec2(0.0));\n let clamped = clamp(rel, vec2(0.0), last);\n let coverage = clamp(rel + vec2(1.0), vec2(0.0), vec2(1.0))\n * clamp(extent - rel, vec2(0.0), vec2(1.0));\n\n let source_texel = vec2<f32>(source_origin) + clamped;\n let source_texture_size = vec2<f32>(textureDimensions(source_texture));\n let sampled = textureSampleLevel(source_texture, linear_sampler, (source_texel + 0.5) / source_texture_size, 0.0);\n\n return sampled * (coverage.x * coverage.y);\n}\n\n// The most linear-sampling tap pairs per side a kernel can carry.\n//\n// Keep in sync with `MAX_TAPS_PER_SIDE` in `crate::filters::blur`: the\n// `vec3<f32>` weight and offset accessors below assume exactly this many.\nconst MAX_TAPS_PER_SIDE: u32 = 3u;\n\n// Halve one axis pair of the source, writing the texel of `dest_origin`\'s\n// region that `position` names.\n//\n// This follows the CPU rasterizer, which downscales by applying a [1,3,3,1]/8\n// binomial filter along each axis. On the GPU the two axes collapse into one\n// pass: four bilinear samples placed a quarter-texel outside each pair of\n// source texels reproduce the same [1,3,3,1] weighting in both directions, so\n// sixteen texel reads become four samples.\nfn filter_downscale(\n source_texture: texture_2d<f32>,\n linear_sampler: sampler,\n position: vec4<f32>,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n dest_origin: vec2<u32>,\n) -> vec4<f32> {\n let frag_coord = vec2<u32>(position.xy);\n let rel = vec2<i32>(frag_coord - dest_origin);\n let source_rel = vec2<f32>(rel * 2);\n\n // The four sample points are [src - 1, src, src + 1, src + 2]. To weight\n // them [1,3,3,1], the low sample shifts 0.25 texels towards the low side\n // and the high sample 1.25 towards the high side; bilinear filtering then\n // performs the 1:3 blend of each pair for free.\n let lo = vec2<f32>(-0.25);\n let hi = vec2<f32>(1.25);\n\n let s00 = sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel + vec2(lo.x, lo.y));\n let s01 = sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel + vec2(lo.x, hi.y));\n let s10 = sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel + vec2(hi.x, lo.y));\n let s11 = sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel + vec2(hi.x, hi.y));\n\n return (s00 + s01 + s10 + s11) * 0.25;\n}\n\n// Double one axis pair of the source, writing the texel of `dest_origin`\'s\n// region that `position` names.\n//\n// The reconstruction the downscale above is paired with: each output texel is\n// 75% of the decimated texel covering it and 25% of the neighbour on the side\n// it sits, which one bilinear sample placed a quarter-texel off centre gives\n// directly.\nfn filter_upscale(\n source_texture: texture_2d<f32>,\n linear_sampler: sampler,\n position: vec4<f32>,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n dest_origin: vec2<u32>,\n) -> vec4<f32> {\n let frag_coord = vec2<u32>(position.xy);\n let rel = vec2<i32>(frag_coord - dest_origin);\n let source_base = vec2<f32>(rel / 2);\n let phase = vec2<f32>(rel % 2);\n\n // Even phase leans towards the low neighbour, odd phase towards the high\n // one.\n let sample_offset = select(vec2(-0.25), vec2(0.25), phase == vec2(1.0));\n\n return sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_base + sample_offset);\n}\n\n// One separable Gaussian pass along `dir` (`(1, 0)` horizontal, `(0, 1)`\n// vertical), reading the source texel `source_rel` names inside the source\n// region at `source_origin`.\n//\n// The discrete kernel the CPU rasterizer convolves with is pre-merged on the\n// host into a centre weight plus up to `MAX_TAPS_PER_SIDE` bilinear tap pairs\n// (see `crate::filters::blur::LinearKernel`): each pair of adjacent kernel taps\n// becomes one sample at a fractional offset between them, weighted by their\n// sum. The result is the same Gaussian with roughly half the samples.\nfn filter_convolve(\n source_texture: texture_2d<f32>,\n linear_sampler: sampler,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n source_rel: vec2<f32>,\n dir: vec2<f32>,\n n_linear_taps: u32,\n center_weight: f32,\n weights: vec3<f32>,\n offsets: vec3<f32>,\n) -> vec4<f32> {\n var color = sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel) * center_weight;\n\n // Indexing a `vec3` dynamically is not expressible, so the merged weights\n // and offsets are copied into arrays the loop can index.\n var weights_arr: array<f32, 3>;\n weights_arr[0] = weights.x;\n weights_arr[1] = weights.y;\n weights_arr[2] = weights.z;\n\n var offsets_arr: array<f32, 3>;\n offsets_arr[0] = offsets.x;\n offsets_arr[1] = offsets.y;\n offsets_arr[2] = offsets.z;\n\n // The kernel is symmetric, so each tap contributes on both sides of the\n // centre at the same weight.\n for (var i = 0u; i < n_linear_taps; i++) {\n let w = weights_arr[i];\n let d = dir * offsets_arr[i];\n color += sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel + d) * w;\n color += sample_region_bilinear(source_texture, linear_sampler, source_origin, source_size, source_rel - d) * w;\n }\n\n return color;\n}\n// Copyright 2026 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/filters/drop_shadow.wesl`),\n// narrowed to the shadow-only shape `crate::filters::drop_shadow` plans for:\n// no `original_texture` binding and no `composite_drop_shadow` pass, because\n// this engine never composites a filter layer\'s unfiltered content back over\n// its shadow (see that module\'s own doc for why).\n//\n// The two passes a drop shadow needs beyond a plain blur\'s: a shift by the\n// shadow\'s own device-space offset, and a recolour of the blurred, offset\n// alpha mask into the shadow\'s own premultiplied colour. Both are pure\n// functions over the texture and parameters they are handed, on the same terms\n// `filters_blur.wgsl`\'s kernels are: no `@group`/`@binding` global, so\n// prepending this file disturbs no module\'s derived bind-group layout.\n//\n// The blur passes of a shadow\'s own sequence need nothing from here: a\n// `GpuDropShadow` block packs its header, centre weight, linear weights and\n// linear offsets at the identical texel offsets a blur\'s own block does (see\n// `crate::filters::drop_shadow::GpuDropShadow`), so `filter.wgsl`\'s existing\n// `PASS_BLUR_H`/`PASS_BLUR_V` cases and this prelude\'s own `filters_blur.wgsl`\n// kernels read a shadow\'s blur exactly as they read a plain blur\'s, unmodified.\n\n// The shadow\'s own device-space offset, packed in the third texel of its\n// parameter block (`GpuDropShadow::dx`/`dy`).\nfn get_drop_shadow_offset(texel2: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel2.x), bitcast<f32>(texel2.y));\n}\n\n// The shadow\'s own premultiplied colour, packed as RGBA8 in the third texel\n// (`GpuDropShadow::color`).\nfn get_drop_shadow_color(texel2: vec4<u32>) -> vec4<f32> {\n return unpack4x8unorm(texel2.z);\n}\n\n// One texel of `source_texture`, relative to `source_origin`, or transparent\n// black outside `[0, source_size)` \u{2014} the same terms every kernel here samples\n// past the region it filters on.\n//\n// The check matters most on the negative side: a shadow offset can shift a\n// read below zero, which an unchecked `vec2<i32>` -> `vec2<u32>` cast would\n// wrap into a large, unrelated texel address rather than the transparent read\n// the CPU reference gives it (`EdgeMode::None`, the mode every filter this\n// engine serves is prepared with).\nfn drop_shadow_load_checked(\n source_texture: texture_2d<f32>,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n coord: vec2<f32>,\n) -> vec4<f32> {\n if coord.x < 0.0 || coord.y < 0.0 || coord.x >= f32(source_size.x) || coord.y >= f32(source_size.y) {\n return vec4<f32>(0.0);\n }\n\n let texel = vec2<u32>(vec2<i32>(source_origin) + vec2<i32>(coord));\n return textureLoad(source_texture, texel, 0);\n}\n\n// Shift `source_texture` by the shadow\'s own `(dx, dy)`, rounded to the\n// nearest whole texel.\n//\n// `floor(x + 0.5)` rather than WGSL\'s `round()`, which ties to even: the CPU\n// reference this is compared against rounds ties away from zero, and the two\n// disagree at exact `.5` offsets.\nfn offset_drop_shadow(\n source_texture: texture_2d<f32>,\n source_origin: vec2<u32>,\n source_size: vec2<u32>,\n rel_coord: vec2<f32>,\n dxdy: vec2<f32>,\n) -> vec4<f32> {\n let shifted = rel_coord - floor(dxdy + 0.5);\n return drop_shadow_load_checked(source_texture, source_origin, source_size, shifted);\n}\n\n// Recolour a blurred, offset alpha mask into the shadow\'s own premultiplied\n// colour.\n//\n// The mask\'s own colour channels are discarded \u{2014} a drop shadow paints the\n// shape the blur produced, not the colour of the content that cast it.\n// `rel_coord` is inside the source region by construction: the fragment stage\n// this is called from already returned for a texel of the padding border.\nfn colorize_drop_shadow(\n source_texture: texture_2d<f32>,\n source_origin: vec2<u32>,\n rel_coord: vec2<f32>,\n color: vec4<f32>,\n) -> vec4<f32> {\n let texel = vec2<u32>(vec2<i32>(source_origin) + vec2<i32>(rel_coord));\n let blurred = textureLoad(source_texture, texel, 0);\n return color * blurred.a;\n}\n// Copyright 2026 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/filter.wesl`); its\n// `import package::helpers::...` lines are resolved by prepending\n// `helpers.wgsl`, the blur kernels it calls by prepending `filters_blur.wgsl`,\n// and the drop-shadow passes it calls by prepending\n// `filters_drop_shadow.wgsl`, at load time (`crate::filters::FILTER`).\n//\n// One filter pass over one destination page: the vertex stage places the quad\n// the pass writes, the fragment stage dispatches on the pass kind and returns\n// that texel\'s filtered colour. A whole Gaussian blur, or a whole shadow-only\n// drop shadow, is a sequence of these passes, each its own render pass over\n// the page the previous one did not write, planned host-side by\n// `crate::filters::blur`/`crate::filters::drop_shadow`.\n//\n// The blur half of the reference\'s pass set, plus the offset and colourize\n// halves a shadow-only drop shadow needs, are implemented here. Flood (1) and\n// the drop-shadow composite that reads a layer\'s own unfiltered content back\n// (7) are still reserved rather than implemented \u{2014} this engine never\n// composites a filter layer\'s original content back over its shadow (see\n// `crate::filters::drop_shadow`\'s own doc for why) \u{2014} so the numbering of\n// every pass kind is kept and those two arms can be added later without\n// renumbering the wire format; a pass kind this module does not implement can\n// never reach it, because the scheduler refuses every filter but a blur or a\n// shadow-only drop shadow before a pass is ever planned.\n\n// The texture holding the encoded parameters of every filter in the frame.\n@group(0) @binding(0)\nvar filter_data: texture_2d<u32>;\n// The page holding this pass\'s source, one half of the ping-pong pair.\n@group(1) @binding(0)\nvar source_texture: texture_2d<f32>;\n// A bilinear sampler over `source_texture`.\n@group(1) @binding(1)\nvar linear_sampler: sampler;\n\n// Keep every constant and layout below in sync with `crate::filters::blur`.\n\n// Every filter\'s parameter block is this many bytes, whatever its kind, which\n// is what lets one texel offset address any of them. Declared here rather than\n// used here: `load_filter_texel` is handed an offset the host already expressed\n// in texels, and these are the numbers that offset was computed from.\nconst FILTER_SIZE_BYTES: u32 = 48u;\nconst FILTER_SIZE_U32: u32 = FILTER_SIZE_BYTES / 4u;\nconst TEXELS_PER_FILTER: u32 = FILTER_SIZE_U32 / 4u;\n\n// Pass kind 1 (flood) and 7 (the drop-shadow composite that reads a layer\'s\n// own unfiltered content back) are the reference\'s; they are reserved rather\n// than implemented (see this file\'s header).\nconst PASS_COPY: u32 = 0u;\nconst PASS_OFFSET: u32 = 2u;\nconst PASS_DOWNSCALE: u32 = 3u;\nconst PASS_BLUR_H: u32 = 4u;\nconst PASS_BLUR_V: u32 = 5u;\nconst PASS_UPSCALE: u32 = 6u;\nconst PASS_COLORIZE: u32 = 8u;\n\n// Transparent border a decimated pass overdraws around the region it writes.\n//\n// Half a kernel is as far as a bilinear tap can reach past the region it is\n// filtering, so a border of transparent black that wide is what surrounds a\n// decimated region with the transparency the next pass\'s kernel expects there.\n// It is a second line rather than the first one: the kernels bound their own\n// taps against the source region (`sample_region_bilinear` in\n// `filters_blur.wgsl`, `drop_shadow_load_checked` in\n// `filters_drop_shadow.wgsl`), so what a tap would have read past the region\n// no longer decides the result. The scheduler clears a filter round\'s whole\n// destination page besides.\n//\n// Keep in sync with `FILTER_ATLAS_PADDING` in `crate::filters::blur`.\nconst FILTER_ATLAS_PADDING: u32 = 6u;\n\n// The packed header in the first word of a filter\'s parameter block:\n// bits [0:4] filter_type (5 bits)\n// bits [5:6] edge_mode (2 bits, blur only; ignored by this module)\n// bits [7:10] n_decimations (4 bits, blur only; read host-side only)\n// bits [11:12] n_linear_taps (2 bits, blur only)\n// bit [13] composite_original (drop shadow only; read host-side only)\n// bits [14:31] reserved\n\n// One texel of the filter parameter block starting at `texel_offset`.\nfn load_filter_texel(texel_offset: u32, texel_index: u32) -> vec4<u32> {\n let w = textureDimensions(filter_data).x;\n let flat_index = texel_offset + texel_index;\n return textureLoad(filter_data, flat_index_to_texture_coord(flat_index, w), 0);\n}\n\n// How many merged bilinear tap pairs per side the blur kernel carries.\nfn get_filter_header_n_linear_taps(texel0: vec4<u32>) -> u32 { return (texel0.x >> 11u) & 0x3u; }\n\n// The blur kernel\'s centre weight.\nfn get_blur_center_weight(texel0: vec4<u32>) -> f32 { return bitcast<f32>(texel0.y); }\n\n// The blur kernel\'s merged tap weights. Assumes `MAX_TAPS_PER_SIDE` is 3.\nfn get_blur_linear_weights(texel0: vec4<u32>, texel1: vec4<u32>) -> vec3<f32> {\n return vec3<f32>(\n bitcast<f32>(texel0.z),\n bitcast<f32>(texel0.w),\n bitcast<f32>(texel1.x),\n );\n}\n\n// The blur kernel\'s merged tap offsets. Assumes `MAX_TAPS_PER_SIDE` is 3.\nfn get_blur_linear_offsets(texel1: vec4<u32>) -> vec3<f32> {\n return vec3<f32>(\n bitcast<f32>(texel1.y),\n bitcast<f32>(texel1.z),\n bitcast<f32>(texel1.w),\n );\n}\n\n// Must stay byte-compatible with `crate::filters::blur::FilterInstanceData`.\nstruct FilterInstanceData {\n // Origin of this pass\'s source region in the source page, packed as u16s.\n @location(0)\n source_origin: u32,\n // Extent of this pass\'s source region, packed as u16s \u{2014} the bound every\n // kernel holds its taps inside (`sample_region_bilinear`,\n // `drop_shadow_load_checked`).\n @location(1)\n source_size: u32,\n // Origin of this pass\'s destination region in the destination page, packed\n // as u16s.\n @location(2)\n dest_origin: u32,\n // Extent of this pass\'s destination region, packed as u16s.\n @location(3)\n dest_size: u32,\n // Extent of the whole destination page, packed as u16s.\n @location(4)\n dest_texture_size: u32,\n // Texel offset of this filter\'s parameter block in `filter_data`.\n @location(5)\n filter_data_offset: u32,\n // Extent of the filter layer\'s unscaled region, packed as u16s.\n @location(6)\n original_size: u32,\n // Which pass of the filter\'s sequence this instance runs.\n @location(7)\n filter_pass_kind: u32,\n}\n\nstruct FilterVertexOutput {\n @builtin(position)\n position: vec4<f32>,\n @location(0) @interpolate(flat)\n filter_data_offset: u32,\n @location(1) @interpolate(flat)\n source_origin: vec2<u32>,\n @location(2) @interpolate(flat)\n source_size: vec2<u32>,\n @location(3) @interpolate(flat)\n dest_origin: vec2<u32>,\n @location(4) @interpolate(flat)\n dest_size: vec2<u32>,\n @location(5) @interpolate(flat)\n filter_pass_kind: u32,\n}\n\n@vertex\nfn vs_main(\n @builtin(vertex_index) vertex_index: u32,\n instance: FilterInstanceData,\n) -> FilterVertexOutput {\n let source_origin = unpack_u16_pair(instance.source_origin);\n let source_size = unpack_u16_pair(instance.source_size);\n let dest_origin = unpack_u16_pair(instance.dest_origin);\n let dest_size = unpack_u16_pair(instance.dest_size);\n let dest_texture_size = vec2<f32>(unpack_u16_pair(instance.dest_texture_size));\n let original_size = unpack_u16_pair(instance.original_size);\n\n // A decimated pass writes fewer texels than the pass before it did, so the\n // quad is drawn a padding border wider than the region and the fragment\n // stage writes transparent black over the difference \u{2014} bounded by the\n // layer\'s own unscaled extent, past which nothing is ever sampled.\n let render_size = min(original_size, dest_size + vec2(FILTER_ATLAS_PADDING));\n let corner = quad_corner(vertex_index);\n let dest_xy = vec2<f32>(dest_origin) + corner * vec2<f32>(render_size);\n\n var out: FilterVertexOutput;\n out.position = vec4<f32>(pixel_to_ndc(dest_xy, dest_texture_size), 0.0, 1.0);\n out.filter_data_offset = instance.filter_data_offset;\n out.source_origin = source_origin;\n out.source_size = source_size;\n out.dest_origin = dest_origin;\n out.dest_size = dest_size;\n out.filter_pass_kind = instance.filter_pass_kind;\n return out;\n}\n\n// One texel of the source region, addressed relative to its origin.\n//\n// `rel_coord` is inside the region by construction: the fragment stage returns\n// before reaching here for a texel of the padding border.\nfn sample_source(source_origin: vec2<u32>, rel_coord: vec2<f32>) -> vec4<f32> {\n let source_coord = vec2<u32>(vec2<i32>(source_origin) + vec2<i32>(rel_coord));\n return textureLoad(source_texture, source_coord, 0);\n}\n\nconst HORIZONTAL: vec2<f32> = vec2<f32>(1.0, 0.0);\nconst VERTICAL: vec2<f32> = vec2<f32>(0.0, 1.0);\n\n@fragment\nfn fs_main(\n @location(0) @interpolate(flat) filter_data_offset: u32,\n @location(1) @interpolate(flat) source_origin: vec2<u32>,\n @location(2) @interpolate(flat) source_size: vec2<u32>,\n @location(3) @interpolate(flat) dest_origin: vec2<u32>,\n @location(4) @interpolate(flat) dest_size: vec2<u32>,\n @location(5) @interpolate(flat) filter_pass_kind: u32,\n @builtin(position) position: vec4<f32>,\n) -> @location(0) vec4<f32> {\n let frag_coord = vec2<u32>(position.xy);\n let rel_coord = vec2<f32>(frag_coord - dest_origin);\n // The padding border the vertex stage added: cleared, not filtered.\n if rel_coord.x >= f32(dest_size.x) || rel_coord.y >= f32(dest_size.y) {\n return vec4<f32>(0.0);\n }\n\n switch filter_pass_kind {\n case PASS_COPY: {\n return sample_source(source_origin, rel_coord);\n }\n case PASS_OFFSET: {\n let filter_texel2 = load_filter_texel(filter_data_offset, 2u);\n let dxdy = get_drop_shadow_offset(filter_texel2);\n return offset_drop_shadow(source_texture, source_origin, source_size, rel_coord, dxdy);\n }\n case PASS_DOWNSCALE: {\n return filter_downscale(source_texture, linear_sampler, position, source_origin, source_size, dest_origin);\n }\n case PASS_BLUR_H: {\n let filter_texel0 = load_filter_texel(filter_data_offset, 0u);\n let filter_texel1 = load_filter_texel(filter_data_offset, 1u);\n return filter_convolve(\n source_texture,\n linear_sampler,\n source_origin,\n source_size,\n rel_coord,\n HORIZONTAL,\n get_filter_header_n_linear_taps(filter_texel0),\n get_blur_center_weight(filter_texel0),\n get_blur_linear_weights(filter_texel0, filter_texel1),\n get_blur_linear_offsets(filter_texel1),\n );\n }\n case PASS_BLUR_V: {\n let filter_texel0 = load_filter_texel(filter_data_offset, 0u);\n let filter_texel1 = load_filter_texel(filter_data_offset, 1u);\n return filter_convolve(\n source_texture,\n linear_sampler,\n source_origin,\n source_size,\n rel_coord,\n VERTICAL,\n get_filter_header_n_linear_taps(filter_texel0),\n get_blur_center_weight(filter_texel0),\n get_blur_linear_weights(filter_texel0, filter_texel1),\n get_blur_linear_offsets(filter_texel1),\n );\n }\n case PASS_UPSCALE: {\n return filter_upscale(source_texture, linear_sampler, position, source_origin, source_size, dest_origin);\n }\n case PASS_COLORIZE: {\n let filter_texel2 = load_filter_texel(filter_data_offset, 2u);\n let color = get_drop_shadow_color(filter_texel2);\n return colorize_drop_shadow(source_texture, source_origin, rel_coord, color);\n }\n // A pass kind this module does not implement; unreachable, because the\n // scheduler plans none.\n default: {\n return vec4<f32>(0.0);\n }\n }\n}\n";Expand description
One filter pass over one destination page: vs_main + fs_main, one
pipeline (see crate::gpu::pipelines::EnginePipeline::Filter).
The only module assembled from more than one prelude: the binding-free
helpers every module gets, then FILTER_KERNELS, then
DROP_SHADOW_KERNELS, then the entry-point module that declares the
bindings and dispatches on the pass kind. The layouts and constants it
reads are crate::filters::blur’s and crate::filters::drop_shadow’s;
those modules’ own tests pin all three sides against each other.