pub const COPY: &str = "// Copyright 2024 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/helpers/*.wesl`).\n//\n// Shared, binding-free helper functions: packing, quad geometry, extend\n// modes, encoded-paint accessors, gradient/blur evaluation and atlas\n// sampling. Every entry-point module (`strip.wgsl`, `clear.wgsl`,\n// `copy.wgsl`) is compiled with this file prepended, so the WESL reference\'s\n// `import package::helpers::...` lines are resolved by concatenation rather\n// than by a resolver at build time.\n//\n// Two consequences of that flattening are visible below:\n//\n// 1. Nothing here declares a `@group`/`@binding` global. A texture a helper\n// reads is always a function parameter, so prepending this file never\n// changes an entry-point module\'s derived bind-group layout.\n// 2. Parameters that named a texture or an extend mode in the reference are\n// renamed (`tex`, `mode`) where the reference\'s own module boundary was\n// the only thing keeping them from shadowing an entry-point global or a\n// function declared here.\n\n// Mathematical constants.\nconst PI: f32 = 3.1415926535897932384626433832795028;\nconst TWO_PI: f32 = 2.0 * PI;\n// Tolerance for nearly-zero comparisons. Must match `SCALAR_NEARLY_ZERO` in\n// `vello_common::math`, which the CPU-side rasterizer compares against.\nconst NEARLY_ZERO_TOLERANCE: f32 = 1.0 / 4096.0;\n\n// Extend modes, shared by images and gradients.\nconst EXTEND_PAD: u32 = 0u;\nconst EXTEND_REPEAT: u32 = 1u;\nconst EXTEND_REFLECT: u32 = 2u;\n\n// Image rendering quality.\nconst IMAGE_QUALITY_LOW: u32 = 0u;\nconst IMAGE_QUALITY_MEDIUM: u32 = 1u;\nconst IMAGE_QUALITY_HIGH: u32 = 2u;\n\n// Image source kinds.\nconst IMAGE_SOURCE_ATLAS: u32 = 0u;\nconst IMAGE_SOURCE_EXTERNAL: u32 = 1u;\n\n// Tint modes.\nconst TINT_MODE_ALPHA_MASK: u32 = 0u;\nconst TINT_MODE_MULTIPLY: u32 = 1u;\n\n// Gradient types.\nconst GRADIENT_TYPE_LINEAR: u32 = 0u;\nconst GRADIENT_TYPE_RADIAL: u32 = 1u;\nconst GRADIENT_TYPE_SWEEP: u32 = 2u;\n\n// Radial gradient types.\nconst RADIAL_GRADIENT_TYPE_STANDARD: u32 = 0u;\nconst RADIAL_GRADIENT_TYPE_STRIP: u32 = 1u;\nconst RADIAL_GRADIENT_TYPE_FOCAL: u32 = 2u;\n\n// -----------------------------------------------------------------------------\n// Packing\n// -----------------------------------------------------------------------------\n\nfn unpack_u16_pair(value: u32) -> vec2<u32> {\n return vec2<u32>(value & 0xffffu, value >> 16u);\n}\n\n// -----------------------------------------------------------------------------\n// Texture addressing\n// -----------------------------------------------------------------------------\n\nfn flat_index_to_texture_coord(index: u32, width: u32) -> vec2<u32> {\n return vec2<u32>(index % width, index / width);\n}\n\n// -----------------------------------------------------------------------------\n// Quad geometry\n// -----------------------------------------------------------------------------\n\nfn quad_corner(vertex_index: u32) -> vec2<f32> {\n return vec2<f32>(\n f32(vertex_index & 1u),\n f32(vertex_index >> 1u),\n );\n}\n\nfn pixel_to_ndc(pixel: vec2<f32>, target_size: vec2<f32>) -> vec2<f32> {\n return vec2<f32>(\n pixel.x * 2.0 / target_size.x - 1.0,\n 1.0 - pixel.y * 2.0 / target_size.y,\n );\n}\n\n// -----------------------------------------------------------------------------\n// Strip alpha unpacking\n// -----------------------------------------------------------------------------\n\n// Alpha textures store 16 1-byte alpha values per texel, with each color\n// channel packing the 4 alpha values of a single strip column.\nfn unpack_alphas_from_channel(rgba: vec4<u32>, channel_index: u32) -> u32 {\n switch channel_index {\n case 0u: { return rgba.x; }\n case 1u: { return rgba.y; }\n case 2u: { return rgba.z; }\n case 3u: { return rgba.w; }\n // Fallback, should never happen.\n default: { return rgba.x; }\n }\n}\n\n// -----------------------------------------------------------------------------\n// Extend modes\n// -----------------------------------------------------------------------------\n\nfn extend_mode(t: f32, mode: u32, max: f32) -> f32 {\n switch mode {\n case EXTEND_PAD: {\n return clamp(t, 0.0, max - 1.0);\n }\n case EXTEND_REPEAT: {\n return extend_mode_normalized(t / max, mode) * max;\n }\n case EXTEND_REFLECT, default: {\n return extend_mode_normalized(t / max, mode) * max;\n }\n }\n}\n\nfn extend_mode_normalized(t: f32, mode: u32) -> f32 {\n switch mode {\n case EXTEND_PAD: {\n return clamp(t, 0.0, 1.0);\n }\n case EXTEND_REPEAT: {\n return fract(t);\n }\n case EXTEND_REFLECT, default: {\n return abs(t - 2.0 * round(0.5 * t));\n }\n }\n}\n\n// -----------------------------------------------------------------------------\n// Encoded gradient accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// Sample from the gradient LUT texture at the calculated position.\nfn sample_gradient_lut(\n tex: texture_2d<f32>,\n t_value: f32,\n mode: u32,\n gradient_start: u32,\n texture_width: u32,\n) -> vec4<f32> {\n // Apply the extend mode to t_value.\n let clamped_t = extend_mode_normalized(t_value, mode);\n // Convert t_value to a texture coordinate.\n let t_offset = u32(clamped_t * f32(texture_width - 1u));\n // Absolute position in the flat gradient texture.\n let flat_coord = gradient_start + t_offset;\n let gradient_tex_width = textureDimensions(tex).x;\n let texture_coord = flat_index_to_texture_coord(flat_coord, gradient_tex_width);\n return textureLoad(tex, texture_coord, 0);\n}\n\n// Width of the gradient\'s own ramp, in texels.\nfn get_gradient_texture_width(texel0: vec4<u32>) -> u32 { return texel0.x & 0x0FFFFFFFu; }\n\n// The extend mode for the gradient.\nfn get_gradient_extend_mode(texel0: vec4<u32>) -> u32 { return (texel0.x >> 30u) & 3u; }\n\n// Start coordinate in the flat gradient texture.\nfn get_gradient_start(texel0: vec4<u32>) -> u32 { return texel0.y; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_gradient_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w)),\n vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_gradient_translate(texel1: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.z), bitcast<f32>(texel1.w));\n}\n\nfn apply_gradient_transform(\n texel0: vec4<u32>,\n texel1: vec4<u32>,\n fragment_pos: vec2<f32>,\n) -> vec2<f32> {\n return get_gradient_transform(texel0, texel1) * fragment_pos + get_gradient_translate(texel1);\n}\n\n// Kind of radial gradient (0=Radial, 1=Strip, 2=Focal).\nfn get_radial_kind(texel2: vec4<u32>) -> u32 { return texel2.x & 0x3u; }\n\n// Whether the focal point is swapped for the radial gradient (0=false, 1=true).\nfn get_radial_f_is_swapped(texel2: vec4<u32>) -> u32 { return (texel2.x >> 2u) & 1u; }\n\n// Bias value for radial gradient calculation.\nfn get_radial_bias(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Scale factor for radial gradient calculation.\nfn get_radial_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\n// Focal point 0 parameter for radial gradient.\nfn get_radial_fp0(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\n// Focal point 1 parameter for radial gradient.\nfn get_radial_fp1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\n// Focal radius 1 parameter for radial gradient.\nfn get_radial_fr1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\n// Focal X coordinate for radial gradient.\nfn get_radial_f_focal_x(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\n// Scaled radius 0 squared parameter for the radial gradient strip kind.\nfn get_radial_scaled_r0_squared(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\n// Starting angle for sweep gradient (in radians).\nfn get_sweep_start_angle(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\n// Inverse of angle delta for sweep gradient.\nfn get_sweep_inv_angle_delta(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Fast polynomial approximation for xy_to_unit_angle from Skia.\n// Returns an angle in the [0, 1) range representing [0, 2*PI).\n// See: https://github.com/google/skia/blob/30bba741989865c157c7a997a0caebe94921276b/src/opts/SkRasterPipeline_opts.h#L5859\nfn xy_to_unit_angle(x: f32, y: f32) -> f32 {\n let xabs = abs(x);\n let yabs = abs(y);\n let slope = min(xabs, yabs) / max(xabs, yabs);\n let s = slope * slope;\n // A 7th degree polynomial approximating atan, generated with\n // sollya.gforge.inria.fr via\n // P1 = fpminimax((1/(2*Pi))*atan(x),[|1,3,5,7|],[|24...|],[2^(-40),1],relative);\n var phi = slope * (0.15912117063999176025390625 + s * (-5.185396969318389892578125e-2 + s * (2.476101927459239959716796875e-2 + s * (-7.0547382347285747528076171875e-3))));\n // Map from the first octant to the full circle using quadrant information.\n // Handle the [0, 90] degree range.\n phi = select(phi, 0.25 - phi, xabs < yabs);\n // Handle the [90, 180] degree range.\n phi = select(phi, 0.5 - phi, x < 0.0);\n // Handle the [180, 360] degree range.\n phi = select(phi, 1.0 - phi, y < 0.0);\n // Handle NaN cases (using the property that NaN != NaN).\n phi = select(phi, 0.0, phi != phi);\n return phi;\n}\n\n// Calculate a radial gradient; matches the CPU rasterizer\'s implementation.\n// Returns [t_value, validity], where validity is 0.0 for a sample that has no\n// gradient coverage at all.\nfn calculate_radial_gradient(\n grad_pos: vec2<f32>,\n texel2: vec4<u32>,\n texel3: vec4<u32>,\n) -> vec2<f32> {\n let x_pos = grad_pos.x;\n let y_pos = grad_pos.y;\n\n var t_value: f32;\n var is_valid: bool;\n let kind = get_radial_kind(texel2);\n\n switch kind {\n case RADIAL_GRADIENT_TYPE_STANDARD: {\n // Standard radial gradient: bias + scale * sqrt(x^2 + y^2).\n let radius = sqrt(x_pos * x_pos + y_pos * y_pos);\n t_value = get_radial_bias(texel2) + get_radial_scale(texel2) * radius;\n // Radial gradients are always valid.\n is_valid = true;\n }\n case RADIAL_GRADIENT_TYPE_STRIP: {\n // Strip gradient: x + sqrt(scaled_r0_squared - y^2).\n let p1 = get_radial_scaled_r0_squared(texel3) - y_pos * y_pos;\n // Invalid if negative under the square root.\n is_valid = p1 >= 0.0;\n if is_valid {\n t_value = x_pos + sqrt(p1);\n } else {\n // Value doesn\'t matter when invalid.\n t_value = 0.0;\n }\n }\n case RADIAL_GRADIENT_TYPE_FOCAL, default: {\n var t = 0.0;\n let fp0 = get_radial_fp0(texel2);\n let fp1 = get_radial_fp1(texel3);\n let fr1 = get_radial_fr1(texel3);\n let f_focal_x = get_radial_f_focal_x(texel3);\n let is_swapped = get_radial_f_is_swapped(texel2);\n\n // Focal flags, derived from the encoded field values.\n let is_focal_on_circle = abs(1.0 - fr1) <= NEARLY_ZERO_TOLERANCE;\n let is_well_behaved = !is_focal_on_circle && fr1 > 1.0;\n let is_natively_focal = abs(f_focal_x) <= NEARLY_ZERO_TOLERANCE;\n\n // Start with the valid assumption.\n is_valid = true;\n\n if is_focal_on_circle {\n t = x_pos + y_pos * y_pos / x_pos;\n // Check for division by zero and negative t.\n is_valid = t >= 0.0 && x_pos != 0.0;\n } else if is_well_behaved {\n t = sqrt(x_pos * x_pos + y_pos * y_pos) - x_pos * fp0;\n } else {\n // For non-well-behaved gradients, check whether the\n // calculation is valid.\n let xx = x_pos * x_pos;\n let yy = y_pos * y_pos;\n let discriminant = xx - yy;\n\n if is_swapped != 0u || (1.0 - f_focal_x < 0.0) {\n t = -sqrt(discriminant) - x_pos * fp0;\n } else {\n t = sqrt(discriminant) - x_pos * fp0;\n }\n\n // Invalid if the discriminant is negative or t is negative.\n is_valid = discriminant >= 0.0 && t >= 0.0;\n }\n\n // Apply the additional focal transforms only if still valid.\n if is_valid {\n if 1.0 - f_focal_x < 0.0 {\n t = -t;\n }\n\n if !is_natively_focal {\n t = t + fp1;\n }\n\n if is_swapped != 0u {\n t = 1.0 - t;\n }\n }\n\n t_value = t;\n }\n }\n\n return vec2<f32>(t_value, select(0.0, 1.0, is_valid));\n}\n\n// -----------------------------------------------------------------------------\n// Encoded blurred-rounded-rect accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_blurred_rounded_rect_transform(texel0: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.x), bitcast<f32>(texel0.y)),\n vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_blurred_rounded_rect_translate(texel1: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y));\n}\n\n// Premultiplied rectangle color.\nfn get_blurred_rounded_rect_color(texel1: vec4<u32>) -> vec4<f32> { return unpack4x8unorm(texel1.z); }\n\n// Whether to paint the inverse (`1 - alpha`) of the blur coverage.\nfn get_blurred_rounded_rect_invert(texel1: vec4<u32>) -> u32 { return texel1.w; }\n\nfn get_blurred_rounded_rect_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\nfn get_blurred_rounded_rect_recip_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\nfn get_blurred_rounded_rect_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\nfn get_blurred_rounded_rect_std_dev_inv(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\nfn get_blurred_rounded_rect_min_edge(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\nfn get_blurred_rounded_rect_w(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\nfn get_blurred_rounded_rect_h(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\nfn get_blurred_rounded_rect_r1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\nfn get_blurred_rounded_rect_width(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.x); }\n\nfn get_blurred_rounded_rect_height(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.y); }\n\n// Approximation to erf, matching the CPU rasterizer\'s blur painter.\nfn erf7(x: f32) -> f32 {\n let y = clamp(x * 1.1283791671, -100.0, 100.0);\n let yy = y * y;\n let z = y + (0.24295 + (0.03395 + 0.0104 * yy) * yy) * (y * yy);\n return z / sqrt(1.0 + z * z);\n}\n\n// Approximation for the convolution of a gaussian filter with a rounded\n// rectangle, modelled after the CPU rasterizer\'s blurred-rounded-rect painter.\nfn calculate_blurred_rounded_rect(\n fragment_pos: vec2<f32>,\n texel0: vec4<u32>,\n texel1: vec4<u32>,\n texel2: vec4<u32>,\n texel3: vec4<u32>,\n texel4: vec4<u32>,\n) -> vec4<f32> {\n let transform = get_blurred_rounded_rect_transform(texel0);\n let translate = get_blurred_rounded_rect_translate(texel1);\n let color = get_blurred_rounded_rect_color(texel1);\n let invert = get_blurred_rounded_rect_invert(texel1);\n let exponent = get_blurred_rounded_rect_exponent(texel2);\n let recip_exponent = get_blurred_rounded_rect_recip_exponent(texel2);\n let scale = get_blurred_rounded_rect_scale(texel2);\n let std_dev_inv = get_blurred_rounded_rect_std_dev_inv(texel2);\n let min_edge = get_blurred_rounded_rect_min_edge(texel3);\n let w = get_blurred_rounded_rect_w(texel3);\n let h = get_blurred_rounded_rect_h(texel3);\n let r1 = get_blurred_rounded_rect_r1(texel3);\n let width = get_blurred_rounded_rect_width(texel4);\n let height = get_blurred_rounded_rect_height(texel4);\n\n let local_xy = transform * fragment_pos + translate;\n // The 0.5 and 0.0 constants correspond to the CPU painter\'s v1 and v0.\n let y = local_xy.y - 0.5 * height;\n let y0 = r1 + abs(y) - 0.5 * h;\n let y1 = max(y0, 0.0);\n\n let x = local_xy.x - 0.5 * width;\n let x0 = r1 + abs(x) - 0.5 * w;\n let x1 = max(x0, 0.0);\n\n let d_pos = pow(\n pow(x1, exponent) + pow(y1, exponent),\n recip_exponent,\n );\n let d_neg = min(max(x0, y0), 0.0);\n let d = d_pos + d_neg - r1;\n let blur_coverage = scale * (\n erf7(std_dev_inv * (min_edge + d)) -\n erf7(std_dev_inv * d)\n );\n\n // Invert alpha when the `invert` flag is set.\n let blur_alpha = select(blur_coverage, 1.0 - blur_coverage, invert != 0u);\n\n return color * blur_alpha;\n}\n\n// -----------------------------------------------------------------------------\n// Encoded image accessors\n// -----------------------------------------------------------------------------\n\n// Encoded image layout. Must match `GpuEncodedImage` in `gpu::paint_texture`.\n//\n// texel0.x: image_params\n// bits 0-1: quality\n// bits 2-3: extend_x\n// bits 4-5: extend_y\n// bits 6-13: atlas_index\n// bit 14: source_kind (0=atlas, 1=external texture)\n// texel0.y: image_size, packed as [width:16, height:16]\n// texel0.z: image_offset, packed as [x:16, y:16]\n// texel0.w/texel1.x/texel1.y/texel1.z: transform matrix [a, b, c, d]\n// texel1.w/texel2.x: translation [tx, ty]\n// texel2.y: premultiplied tint color packed as RGBA8 unorm\n// texel2.z: tint mode\n// texel2.w: transparent padding pixels around the image in the atlas\n\n// The rendering quality of the image.\nfn get_image_quality(texel0: vec4<u32>) -> u32 { return texel0.x & 0x3u; }\n\n// The extend modes in the horizontal and vertical direction.\nfn get_image_extend_modes(texel0: vec4<u32>) -> vec2<u32> {\n return vec2<u32>((texel0.x >> 2u) & 0x3u, (texel0.x >> 4u) & 0x3u);\n}\n\n// The size of the image in pixels.\nfn get_image_size(texel0: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(f32(texel0.y >> 16u), f32(texel0.y & 0xFFFFu));\n}\n\n// The offset of the image in pixels.\nfn get_image_offset(texel0: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(f32(texel0.z >> 16u), f32(texel0.z & 0xFFFFu));\n}\n\n// The atlas index containing this image.\nfn get_image_atlas_index(texel0: vec4<u32>) -> u32 { return (texel0.x >> 6u) & 0xFFu; }\n\n// Whether the image is sourced from the atlas or the externally bound texture.\nfn get_image_source_kind(texel0: vec4<u32>) -> u32 { return (texel0.x >> 14u) & 0x1u; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_image_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n return mat2x2<f32>(\n vec2<f32>(bitcast<f32>(texel0.w), bitcast<f32>(texel1.x)),\n vec2<f32>(bitcast<f32>(texel1.y), bitcast<f32>(texel1.z))\n );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_image_translate(texel1: vec4<u32>, texel2: vec4<u32>) -> vec2<f32> {\n return vec2<f32>(bitcast<f32>(texel1.w), bitcast<f32>(texel2.x));\n}\n\n// Number of transparent padding pixels around the image in the atlas.\nfn get_image_padding(texel2: vec4<u32>) -> f32 { return f32(texel2.w); }\n\n// -----------------------------------------------------------------------------\n// Atlas-array sampling\n// -----------------------------------------------------------------------------\n\n// Bilinear filtering: sample the 4 surrounding texels of the target point and\n// interpolate them with a bilinear filter.\nfn bilinear_sample(\n tex: texture_2d_array<f32>,\n coords: vec2<f32>,\n atlas_idx: i32,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n _extend_modes: vec2<u32>,\n _image_padding: f32,\n) -> vec4<f32> {\n let atlas_max = image_offset + image_size - vec2(1.0);\n let atlas_uv_clamped = clamp(coords, image_offset, atlas_max);\n let uv_quad = vec4(floor(atlas_uv_clamped), ceil(atlas_uv_clamped));\n let uv_frac = fract(coords);\n let a = textureLoad(tex, vec2<i32>(uv_quad.xy), atlas_idx, 0);\n let b = textureLoad(tex, vec2<i32>(uv_quad.xw), atlas_idx, 0);\n let c = textureLoad(tex, vec2<i32>(uv_quad.zy), atlas_idx, 0);\n let d = textureLoad(tex, vec2<i32>(uv_quad.zw), atlas_idx, 0);\n return mix(mix(a, b, uv_frac.y), mix(c, d, uv_frac.y), uv_frac.x);\n}\n\n// Bicubic filtering with a Mitchell filter (B=1/3, C=1/3): sample the 16\n// surrounding texels of the target point and interpolate them with a cubic\n// filter. The 4x4 matrix holds the coefficients of the cubic function used to\n// derive the weights from the fractional part of the sample location.\nfn bicubic_sample(\n tex: texture_2d_array<f32>,\n coords: vec2<f32>,\n atlas_idx: i32,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n _extend_modes: vec2<u32>,\n _image_padding: f32,\n) -> vec4<f32> {\n let atlas_max = image_offset + image_size - vec2(1.0);\n let frac_coords = fract(coords + 0.5);\n // Cubic weights for the x and y directions.\n let cx = cubic_weights(frac_coords.x);\n let cy = cubic_weights(frac_coords.y);\n\n // Sample the 4x4 grid around `coords`.\n let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n // Interpolate in the x direction for each row.\n let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n // Interpolate in the y direction.\n let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n // Clamp alpha first, then clamp the premultiplied color channels against it.\n let a = clamp(result.a, 0.0, 1.0);\n return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\n// Mitchell-Netravali cubic filter coefficients with B=1/3 and C=1/3, matching\n// the CPU rasterizer\'s cubic resampler.\nconst MF: array<vec4<f32>, 4> = array<vec4<f32>, 4>(\n vec4<f32>(\n (1.0 / 6.0) / 3.0,\n -(3.0 / 6.0) / 3.0 - 1.0 / 3.0,\n (3.0 / 6.0) / 3.0 + 2.0 * 1.0 / 3.0,\n -(1.0 / 6.0) / 3.0 - 1.0 / 3.0\n ),\n vec4<f32>(\n 1.0 - (2.0 / 6.0) / 3.0,\n 0.0,\n -3.0 + (12.0 / 6.0) / 3.0 + 1.0 / 3.0,\n 2.0 - (9.0 / 6.0) / 3.0 - 1.0 / 3.0\n ),\n vec4<f32>(\n (1.0 / 6.0) / 3.0,\n (3.0 / 6.0) / 3.0 + 1.0 / 3.0,\n 3.0 - (15.0 / 6.0) / 3.0 - 2.0 * 1.0 / 3.0,\n -2.0 + (9.0 / 6.0) / 3.0 + 1.0 / 3.0\n ),\n vec4<f32>(\n 0.0,\n 0.0,\n -1.0 / 3.0,\n (1.0 / 6.0) / 3.0 + 1.0 / 3.0\n )\n);\n\n// The four cubic weights for a single fractional value.\nfn cubic_weights(fract: f32) -> vec4<f32> {\n return vec4<f32>(\n single_weight(fract, MF[0][0], MF[0][1], MF[0][2], MF[0][3]),\n single_weight(fract, MF[1][0], MF[1][1], MF[1][2], MF[1][3]),\n single_weight(fract, MF[2][0], MF[2][1], MF[2][2], MF[2][3]),\n single_weight(fract, MF[3][0], MF[3][1], MF[3][2], MF[3][3])\n );\n}\n\n// One weight from the fractional value t and the cubic coefficients.\nfn single_weight(t: f32, a: f32, b: f32, c: f32, d: f32) -> f32 {\n return t * (t * (t * d + c) + b) + a;\n}\n\n// -----------------------------------------------------------------------------\n// External-texture sampling\n// -----------------------------------------------------------------------------\n\n// The atlas-array samplers above, restated for a plain 2D texture: an\n// externally bound image is a whole texture rather than a page of the array.\n\nfn external_bilinear_sample(\n tex: texture_2d<f32>,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n let image_max = image_offset + image_size - vec2(1.0);\n let clamped_coords = clamp(coords, image_offset, image_max);\n let coord_quad = vec4(floor(clamped_coords), ceil(clamped_coords));\n let coord_frac = fract(coords);\n let a = textureLoad(tex, vec2<i32>(coord_quad.xy), 0);\n let b = textureLoad(tex, vec2<i32>(coord_quad.xw), 0);\n let c = textureLoad(tex, vec2<i32>(coord_quad.zy), 0);\n let d = textureLoad(tex, vec2<i32>(coord_quad.zw), 0);\n return mix(mix(a, b, coord_frac.y), mix(c, d, coord_frac.y), coord_frac.x);\n}\n\nfn external_bicubic_sample(\n tex: texture_2d<f32>,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n let image_max = image_offset + image_size - vec2(1.0);\n let frac_coords = fract(coords + 0.5);\n let cx = cubic_weights(frac_coords.x);\n let cy = cubic_weights(frac_coords.y);\n\n let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, image_max)), 0);\n let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, image_max)), 0);\n let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, image_max)), 0);\n let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, image_max)), 0);\n\n let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, image_max)), 0);\n let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, image_max)), 0);\n let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, image_max)), 0);\n let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, image_max)), 0);\n\n let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, image_max)), 0);\n let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, image_max)), 0);\n let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, image_max)), 0);\n let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, image_max)), 0);\n\n let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, image_max)), 0);\n let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, image_max)), 0);\n let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, image_max)), 0);\n let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, image_max)), 0);\n\n let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n // Clamp alpha first, then clamp the premultiplied color channels against it.\n let a = clamp(result.a, 0.0, 1.0);\n return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\nfn sample_external_image(\n tex: texture_2d<f32>,\n quality: u32,\n coords: vec2<f32>,\n image_offset: vec2<f32>,\n image_size: vec2<f32>,\n) -> vec4<f32> {\n if quality == IMAGE_QUALITY_HIGH {\n return external_bicubic_sample(tex, coords, image_offset, image_size);\n }\n if quality == IMAGE_QUALITY_MEDIUM {\n return external_bilinear_sample(\n tex,\n coords - vec2(0.5),\n image_offset,\n image_size,\n );\n }\n return textureLoad(tex, vec2<u32>(coords), 0);\n}\n// Copyright 2026 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/copy.wesl`); its\n// `import package::helpers::...` lines are resolved by prepending\n// `helpers.wgsl` at load time (`gpu::shader_src`).\n//\n// Copies rectangular regions between intermediate textures, one instance per\n// region.\n\n// Must stay byte-compatible with the copy-instance vertex layout in\n// `gpu::pipelines`.\nstruct CopyInstance {\n // Origin in the destination texture page, packed as u16s.\n @location(0)\n dest_texture_origin: u32,\n // Origin in the source texture page, packed as u16s.\n @location(1)\n source_texture_origin: u32,\n // Width and height of the copied region, packed as u16s.\n @location(2)\n copy_rect_size: u32,\n // Width and height of the destination texture page, packed as u16s.\n @location(3)\n dest_texture_size: u32,\n}\n\nstruct VertexOutput {\n @location(0)\n source_texture_xy: vec2<f32>,\n @builtin(position)\n position: vec4<f32>,\n}\n\n@group(0) @binding(0)\nvar source_texture: texture_2d<f32>;\n\n@vertex\nfn vs_main(\n @builtin(vertex_index) vertex_index: u32,\n instance: CopyInstance,\n) -> VertexOutput {\n let dest_texture_origin = unpack_u16_pair(instance.dest_texture_origin);\n let source_texture_origin = unpack_u16_pair(instance.source_texture_origin);\n let copy_rect_size = unpack_u16_pair(instance.copy_rect_size);\n let dest_texture_size = unpack_u16_pair(instance.dest_texture_size);\n\n let local = quad_corner(vertex_index) * vec2<f32>(copy_rect_size);\n let dest_texture_xy = vec2<f32>(dest_texture_origin) + local;\n\n var out: VertexOutput;\n out.source_texture_xy = vec2<f32>(source_texture_origin) + local;\n\n let ndc = pixel_to_ndc(dest_texture_xy, vec2<f32>(dest_texture_size));\n out.position = vec4<f32>(ndc, 0.0, 1.0);\n return out;\n}\n\n@fragment\nfn fs_main(\n @location(0) source_texture_xy: vec2<f32>,\n) -> @location(0) vec4<f32> {\n return textureLoad(source_texture, vec2<i32>(source_texture_xy), 0);\n}\n";Expand description
Rectangular region copies between intermediate textures: vs_main +
fs_main.