Skip to main content

COPY

Constant COPY 

Source
pub const COPY: &str = "// Copyright 2024 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/helpers/*.wesl`).\n//\n// Shared, binding-free helper functions: packing, quad geometry, extend\n// modes, encoded-paint accessors, gradient/blur evaluation and atlas\n// sampling. Every entry-point module (`strip.wgsl`, `clear.wgsl`,\n// `copy.wgsl`) is compiled with this file prepended, so the WESL reference\'s\n// `import package::helpers::...` lines are resolved by concatenation rather\n// than by a resolver at build time.\n//\n// Two consequences of that flattening are visible below:\n//\n// 1. Nothing here declares a `@group`/`@binding` global. A texture a helper\n//    reads is always a function parameter, so prepending this file never\n//    changes an entry-point module\'s derived bind-group layout.\n// 2. Parameters that named a texture or an extend mode in the reference are\n//    renamed (`tex`, `mode`) where the reference\'s own module boundary was\n//    the only thing keeping them from shadowing an entry-point global or a\n//    function declared here.\n\n// Mathematical constants.\nconst PI: f32 = 3.1415926535897932384626433832795028;\nconst TWO_PI: f32 = 2.0 * PI;\n// Tolerance for nearly-zero comparisons. Must match `SCALAR_NEARLY_ZERO` in\n// `vello_common::math`, which the CPU-side rasterizer compares against.\nconst NEARLY_ZERO_TOLERANCE: f32 = 1.0 / 4096.0;\n\n// Extend modes, shared by images and gradients.\nconst EXTEND_PAD: u32 = 0u;\nconst EXTEND_REPEAT: u32 = 1u;\nconst EXTEND_REFLECT: u32 = 2u;\n\n// Image rendering quality.\nconst IMAGE_QUALITY_LOW: u32 = 0u;\nconst IMAGE_QUALITY_MEDIUM: u32 = 1u;\nconst IMAGE_QUALITY_HIGH: u32 = 2u;\n\n// Image source kinds.\nconst IMAGE_SOURCE_ATLAS: u32 = 0u;\nconst IMAGE_SOURCE_EXTERNAL: u32 = 1u;\n\n// Tint modes.\nconst TINT_MODE_ALPHA_MASK: u32 = 0u;\nconst TINT_MODE_MULTIPLY: u32 = 1u;\n\n// Gradient types.\nconst GRADIENT_TYPE_LINEAR: u32 = 0u;\nconst GRADIENT_TYPE_RADIAL: u32 = 1u;\nconst GRADIENT_TYPE_SWEEP: u32 = 2u;\n\n// Radial gradient types.\nconst RADIAL_GRADIENT_TYPE_STANDARD: u32 = 0u;\nconst RADIAL_GRADIENT_TYPE_STRIP: u32 = 1u;\nconst RADIAL_GRADIENT_TYPE_FOCAL: u32 = 2u;\n\n// -----------------------------------------------------------------------------\n// Packing\n// -----------------------------------------------------------------------------\n\nfn unpack_u16_pair(value: u32) -> vec2<u32> {\n    return vec2<u32>(value & 0xffffu, value >> 16u);\n}\n\n// -----------------------------------------------------------------------------\n// Texture addressing\n// -----------------------------------------------------------------------------\n\nfn flat_index_to_texture_coord(index: u32, width: u32) -> vec2<u32> {\n    return vec2<u32>(index % width, index / width);\n}\n\n// -----------------------------------------------------------------------------\n// Quad geometry\n// -----------------------------------------------------------------------------\n\nfn quad_corner(vertex_index: u32) -> vec2<f32> {\n    return vec2<f32>(\n        f32(vertex_index & 1u),\n        f32(vertex_index >> 1u),\n    );\n}\n\nfn pixel_to_ndc(pixel: vec2<f32>, target_size: vec2<f32>) -> vec2<f32> {\n    return vec2<f32>(\n        pixel.x * 2.0 / target_size.x - 1.0,\n        1.0 - pixel.y * 2.0 / target_size.y,\n    );\n}\n\n// -----------------------------------------------------------------------------\n// Strip alpha unpacking\n// -----------------------------------------------------------------------------\n\n// Alpha textures store 16 1-byte alpha values per texel, with each color\n// channel packing the 4 alpha values of a single strip column.\nfn unpack_alphas_from_channel(rgba: vec4<u32>, channel_index: u32) -> u32 {\n    switch channel_index {\n        case 0u: { return rgba.x; }\n        case 1u: { return rgba.y; }\n        case 2u: { return rgba.z; }\n        case 3u: { return rgba.w; }\n        // Fallback, should never happen.\n        default: { return rgba.x; }\n    }\n}\n\n// -----------------------------------------------------------------------------\n// Extend modes\n// -----------------------------------------------------------------------------\n\nfn extend_mode(t: f32, mode: u32, max: f32) -> f32 {\n    switch mode {\n        case EXTEND_PAD: {\n            return clamp(t, 0.0, max - 1.0);\n        }\n        case EXTEND_REPEAT: {\n            return extend_mode_normalized(t / max, mode) * max;\n        }\n        case EXTEND_REFLECT, default: {\n            return extend_mode_normalized(t / max, mode) * max;\n        }\n    }\n}\n\nfn extend_mode_normalized(t: f32, mode: u32) -> f32 {\n    switch mode {\n        case EXTEND_PAD: {\n            return clamp(t, 0.0, 1.0);\n        }\n        case EXTEND_REPEAT: {\n            return fract(t);\n        }\n        case EXTEND_REFLECT, default: {\n            return abs(t - 2.0 * round(0.5 * t));\n        }\n    }\n}\n\n// -----------------------------------------------------------------------------\n// Encoded gradient accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// Sample from the gradient LUT texture at the calculated position.\nfn sample_gradient_lut(\n    tex: texture_2d<f32>,\n    t_value: f32,\n    mode: u32,\n    gradient_start: u32,\n    texture_width: u32,\n) -> vec4<f32> {\n    // Apply the extend mode to t_value.\n    let clamped_t = extend_mode_normalized(t_value, mode);\n    // Convert t_value to a texture coordinate.\n    let t_offset = u32(clamped_t * f32(texture_width - 1u));\n    // Absolute position in the flat gradient texture.\n    let flat_coord = gradient_start + t_offset;\n    let gradient_tex_width = textureDimensions(tex).x;\n    let texture_coord = flat_index_to_texture_coord(flat_coord, gradient_tex_width);\n    return textureLoad(tex, texture_coord, 0);\n}\n\n// Width of the gradient\'s own ramp, in texels.\nfn get_gradient_texture_width(texel0: vec4<u32>) -> u32 { return texel0.x & 0x0FFFFFFFu; }\n\n// The extend mode for the gradient.\nfn get_gradient_extend_mode(texel0: vec4<u32>) -> u32 { return (texel0.x >> 30u) & 3u; }\n\n// Start coordinate in the flat gradient texture.\nfn get_gradient_start(texel0: vec4<u32>) -> u32 { return texel0.y; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_gradient_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n    return mat2x2<f32>(\n        vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w)),\n        vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y))\n    );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_gradient_translate(texel1: vec4<u32>) -> vec2<f32> {\n    return vec2<f32>(bitcast<f32>(texel1.z), bitcast<f32>(texel1.w));\n}\n\nfn apply_gradient_transform(\n    texel0: vec4<u32>,\n    texel1: vec4<u32>,\n    fragment_pos: vec2<f32>,\n) -> vec2<f32> {\n    return get_gradient_transform(texel0, texel1) * fragment_pos + get_gradient_translate(texel1);\n}\n\n// Kind of radial gradient (0=Radial, 1=Strip, 2=Focal).\nfn get_radial_kind(texel2: vec4<u32>) -> u32 { return texel2.x & 0x3u; }\n\n// Whether the focal point is swapped for the radial gradient (0=false, 1=true).\nfn get_radial_f_is_swapped(texel2: vec4<u32>) -> u32 { return (texel2.x >> 2u) & 1u; }\n\n// Bias value for radial gradient calculation.\nfn get_radial_bias(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Scale factor for radial gradient calculation.\nfn get_radial_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\n// Focal point 0 parameter for radial gradient.\nfn get_radial_fp0(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\n// Focal point 1 parameter for radial gradient.\nfn get_radial_fp1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\n// Focal radius 1 parameter for radial gradient.\nfn get_radial_fr1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\n// Focal X coordinate for radial gradient.\nfn get_radial_f_focal_x(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\n// Scaled radius 0 squared parameter for the radial gradient strip kind.\nfn get_radial_scaled_r0_squared(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\n// Starting angle for sweep gradient (in radians).\nfn get_sweep_start_angle(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\n// Inverse of angle delta for sweep gradient.\nfn get_sweep_inv_angle_delta(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\n// Fast polynomial approximation for xy_to_unit_angle from Skia.\n// Returns an angle in the [0, 1) range representing [0, 2*PI).\n// See: https://github.com/google/skia/blob/30bba741989865c157c7a997a0caebe94921276b/src/opts/SkRasterPipeline_opts.h#L5859\nfn xy_to_unit_angle(x: f32, y: f32) -> f32 {\n    let xabs = abs(x);\n    let yabs = abs(y);\n    let slope = min(xabs, yabs) / max(xabs, yabs);\n    let s = slope * slope;\n    // A 7th degree polynomial approximating atan, generated with\n    // sollya.gforge.inria.fr via\n    // P1 = fpminimax((1/(2*Pi))*atan(x),[|1,3,5,7|],[|24...|],[2^(-40),1],relative);\n    var phi = slope * (0.15912117063999176025390625 + s * (-5.185396969318389892578125e-2 + s * (2.476101927459239959716796875e-2 + s * (-7.0547382347285747528076171875e-3))));\n    // Map from the first octant to the full circle using quadrant information.\n    // Handle the [0, 90] degree range.\n    phi = select(phi, 0.25 - phi, xabs < yabs);\n    // Handle the [90, 180] degree range.\n    phi = select(phi, 0.5 - phi, x < 0.0);\n    // Handle the [180, 360] degree range.\n    phi = select(phi, 1.0 - phi, y < 0.0);\n    // Handle NaN cases (using the property that NaN != NaN).\n    phi = select(phi, 0.0, phi != phi);\n    return phi;\n}\n\n// Calculate a radial gradient; matches the CPU rasterizer\'s implementation.\n// Returns [t_value, validity], where validity is 0.0 for a sample that has no\n// gradient coverage at all.\nfn calculate_radial_gradient(\n    grad_pos: vec2<f32>,\n    texel2: vec4<u32>,\n    texel3: vec4<u32>,\n) -> vec2<f32> {\n    let x_pos = grad_pos.x;\n    let y_pos = grad_pos.y;\n\n    var t_value: f32;\n    var is_valid: bool;\n    let kind = get_radial_kind(texel2);\n\n    switch kind {\n        case RADIAL_GRADIENT_TYPE_STANDARD: {\n            // Standard radial gradient: bias + scale * sqrt(x^2 + y^2).\n            let radius = sqrt(x_pos * x_pos + y_pos * y_pos);\n            t_value = get_radial_bias(texel2) + get_radial_scale(texel2) * radius;\n            // Radial gradients are always valid.\n            is_valid = true;\n        }\n        case RADIAL_GRADIENT_TYPE_STRIP: {\n            // Strip gradient: x + sqrt(scaled_r0_squared - y^2).\n            let p1 = get_radial_scaled_r0_squared(texel3) - y_pos * y_pos;\n            // Invalid if negative under the square root.\n            is_valid = p1 >= 0.0;\n            if is_valid {\n                t_value = x_pos + sqrt(p1);\n            } else {\n                // Value doesn\'t matter when invalid.\n                t_value = 0.0;\n            }\n        }\n        case RADIAL_GRADIENT_TYPE_FOCAL, default: {\n            var t = 0.0;\n            let fp0 = get_radial_fp0(texel2);\n            let fp1 = get_radial_fp1(texel3);\n            let fr1 = get_radial_fr1(texel3);\n            let f_focal_x = get_radial_f_focal_x(texel3);\n            let is_swapped = get_radial_f_is_swapped(texel2);\n\n            // Focal flags, derived from the encoded field values.\n            let is_focal_on_circle = abs(1.0 - fr1) <= NEARLY_ZERO_TOLERANCE;\n            let is_well_behaved = !is_focal_on_circle && fr1 > 1.0;\n            let is_natively_focal = abs(f_focal_x) <= NEARLY_ZERO_TOLERANCE;\n\n            // Start with the valid assumption.\n            is_valid = true;\n\n            if is_focal_on_circle {\n                t = x_pos + y_pos * y_pos / x_pos;\n                // Check for division by zero and negative t.\n                is_valid = t >= 0.0 && x_pos != 0.0;\n            } else if is_well_behaved {\n                t = sqrt(x_pos * x_pos + y_pos * y_pos) - x_pos * fp0;\n            } else {\n                // For non-well-behaved gradients, check whether the\n                // calculation is valid.\n                let xx = x_pos * x_pos;\n                let yy = y_pos * y_pos;\n                let discriminant = xx - yy;\n\n                if is_swapped != 0u || (1.0 - f_focal_x < 0.0) {\n                    t = -sqrt(discriminant) - x_pos * fp0;\n                } else {\n                    t = sqrt(discriminant) - x_pos * fp0;\n                }\n\n                // Invalid if the discriminant is negative or t is negative.\n                is_valid = discriminant >= 0.0 && t >= 0.0;\n            }\n\n            // Apply the additional focal transforms only if still valid.\n            if is_valid {\n                if 1.0 - f_focal_x < 0.0 {\n                    t = -t;\n                }\n\n                if !is_natively_focal {\n                    t = t + fp1;\n                }\n\n                if is_swapped != 0u {\n                    t = 1.0 - t;\n                }\n            }\n\n            t_value = t;\n        }\n    }\n\n    return vec2<f32>(t_value, select(0.0, 1.0, is_valid));\n}\n\n// -----------------------------------------------------------------------------\n// Encoded blurred-rounded-rect accessors and evaluation\n// -----------------------------------------------------------------------------\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_blurred_rounded_rect_transform(texel0: vec4<u32>) -> mat2x2<f32> {\n    return mat2x2<f32>(\n        vec2<f32>(bitcast<f32>(texel0.x), bitcast<f32>(texel0.y)),\n        vec2<f32>(bitcast<f32>(texel0.z), bitcast<f32>(texel0.w))\n    );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_blurred_rounded_rect_translate(texel1: vec4<u32>) -> vec2<f32> {\n    return vec2<f32>(bitcast<f32>(texel1.x), bitcast<f32>(texel1.y));\n}\n\n// Premultiplied rectangle color.\nfn get_blurred_rounded_rect_color(texel1: vec4<u32>) -> vec4<f32> { return unpack4x8unorm(texel1.z); }\n\n// Whether to paint the inverse (`1 - alpha`) of the blur coverage.\nfn get_blurred_rounded_rect_invert(texel1: vec4<u32>) -> u32 { return texel1.w; }\n\nfn get_blurred_rounded_rect_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.x); }\n\nfn get_blurred_rounded_rect_recip_exponent(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.y); }\n\nfn get_blurred_rounded_rect_scale(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.z); }\n\nfn get_blurred_rounded_rect_std_dev_inv(texel2: vec4<u32>) -> f32 { return bitcast<f32>(texel2.w); }\n\nfn get_blurred_rounded_rect_min_edge(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.x); }\n\nfn get_blurred_rounded_rect_w(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.y); }\n\nfn get_blurred_rounded_rect_h(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.z); }\n\nfn get_blurred_rounded_rect_r1(texel3: vec4<u32>) -> f32 { return bitcast<f32>(texel3.w); }\n\nfn get_blurred_rounded_rect_width(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.x); }\n\nfn get_blurred_rounded_rect_height(texel4: vec4<u32>) -> f32 { return bitcast<f32>(texel4.y); }\n\n// Approximation to erf, matching the CPU rasterizer\'s blur painter.\nfn erf7(x: f32) -> f32 {\n    let y = clamp(x * 1.1283791671, -100.0, 100.0);\n    let yy = y * y;\n    let z = y + (0.24295 + (0.03395 + 0.0104 * yy) * yy) * (y * yy);\n    return z / sqrt(1.0 + z * z);\n}\n\n// Approximation for the convolution of a gaussian filter with a rounded\n// rectangle, modelled after the CPU rasterizer\'s blurred-rounded-rect painter.\nfn calculate_blurred_rounded_rect(\n    fragment_pos: vec2<f32>,\n    texel0: vec4<u32>,\n    texel1: vec4<u32>,\n    texel2: vec4<u32>,\n    texel3: vec4<u32>,\n    texel4: vec4<u32>,\n) -> vec4<f32> {\n    let transform = get_blurred_rounded_rect_transform(texel0);\n    let translate = get_blurred_rounded_rect_translate(texel1);\n    let color = get_blurred_rounded_rect_color(texel1);\n    let invert = get_blurred_rounded_rect_invert(texel1);\n    let exponent = get_blurred_rounded_rect_exponent(texel2);\n    let recip_exponent = get_blurred_rounded_rect_recip_exponent(texel2);\n    let scale = get_blurred_rounded_rect_scale(texel2);\n    let std_dev_inv = get_blurred_rounded_rect_std_dev_inv(texel2);\n    let min_edge = get_blurred_rounded_rect_min_edge(texel3);\n    let w = get_blurred_rounded_rect_w(texel3);\n    let h = get_blurred_rounded_rect_h(texel3);\n    let r1 = get_blurred_rounded_rect_r1(texel3);\n    let width = get_blurred_rounded_rect_width(texel4);\n    let height = get_blurred_rounded_rect_height(texel4);\n\n    let local_xy = transform * fragment_pos + translate;\n    // The 0.5 and 0.0 constants correspond to the CPU painter\'s v1 and v0.\n    let y = local_xy.y - 0.5 * height;\n    let y0 = r1 + abs(y) - 0.5 * h;\n    let y1 = max(y0, 0.0);\n\n    let x = local_xy.x - 0.5 * width;\n    let x0 = r1 + abs(x) - 0.5 * w;\n    let x1 = max(x0, 0.0);\n\n    let d_pos = pow(\n        pow(x1, exponent) + pow(y1, exponent),\n        recip_exponent,\n    );\n    let d_neg = min(max(x0, y0), 0.0);\n    let d = d_pos + d_neg - r1;\n    let blur_coverage = scale * (\n        erf7(std_dev_inv * (min_edge + d)) -\n        erf7(std_dev_inv * d)\n    );\n\n    // Invert alpha when the `invert` flag is set.\n    let blur_alpha = select(blur_coverage, 1.0 - blur_coverage, invert != 0u);\n\n    return color * blur_alpha;\n}\n\n// -----------------------------------------------------------------------------\n// Encoded image accessors\n// -----------------------------------------------------------------------------\n\n// Encoded image layout. Must match `GpuEncodedImage` in `gpu::paint_texture`.\n//\n// texel0.x: image_params\n//   bits 0-1: quality\n//   bits 2-3: extend_x\n//   bits 4-5: extend_y\n//   bits 6-13: atlas_index\n//   bit 14: source_kind (0=atlas, 1=external texture)\n// texel0.y: image_size, packed as [width:16, height:16]\n// texel0.z: image_offset, packed as [x:16, y:16]\n// texel0.w/texel1.x/texel1.y/texel1.z: transform matrix [a, b, c, d]\n// texel1.w/texel2.x: translation [tx, ty]\n// texel2.y: premultiplied tint color packed as RGBA8 unorm\n// texel2.z: tint mode\n// texel2.w: transparent padding pixels around the image in the atlas\n\n// The rendering quality of the image.\nfn get_image_quality(texel0: vec4<u32>) -> u32 { return texel0.x & 0x3u; }\n\n// The extend modes in the horizontal and vertical direction.\nfn get_image_extend_modes(texel0: vec4<u32>) -> vec2<u32> {\n    return vec2<u32>((texel0.x >> 2u) & 0x3u, (texel0.x >> 4u) & 0x3u);\n}\n\n// The size of the image in pixels.\nfn get_image_size(texel0: vec4<u32>) -> vec2<f32> {\n    return vec2<f32>(f32(texel0.y >> 16u), f32(texel0.y & 0xFFFFu));\n}\n\n// The offset of the image in pixels.\nfn get_image_offset(texel0: vec4<u32>) -> vec2<f32> {\n    return vec2<f32>(f32(texel0.z >> 16u), f32(texel0.z & 0xFFFFu));\n}\n\n// The atlas index containing this image.\nfn get_image_atlas_index(texel0: vec4<u32>) -> u32 { return (texel0.x >> 6u) & 0xFFu; }\n\n// Whether the image is sourced from the atlas or the externally bound texture.\nfn get_image_source_kind(texel0: vec4<u32>) -> u32 { return (texel0.x >> 14u) & 0x1u; }\n\n// 2x2 linear part of the affine transform (columns [a,b] and [c,d]).\nfn get_image_transform(texel0: vec4<u32>, texel1: vec4<u32>) -> mat2x2<f32> {\n    return mat2x2<f32>(\n        vec2<f32>(bitcast<f32>(texel0.w), bitcast<f32>(texel1.x)),\n        vec2<f32>(bitcast<f32>(texel1.y), bitcast<f32>(texel1.z))\n    );\n}\n\n// Translation part of the affine transform [tx, ty].\nfn get_image_translate(texel1: vec4<u32>, texel2: vec4<u32>) -> vec2<f32> {\n    return vec2<f32>(bitcast<f32>(texel1.w), bitcast<f32>(texel2.x));\n}\n\n// Number of transparent padding pixels around the image in the atlas.\nfn get_image_padding(texel2: vec4<u32>) -> f32 { return f32(texel2.w); }\n\n// -----------------------------------------------------------------------------\n// Atlas-array sampling\n// -----------------------------------------------------------------------------\n\n// Bilinear filtering: sample the 4 surrounding texels of the target point and\n// interpolate them with a bilinear filter.\nfn bilinear_sample(\n    tex: texture_2d_array<f32>,\n    coords: vec2<f32>,\n    atlas_idx: i32,\n    image_offset: vec2<f32>,\n    image_size: vec2<f32>,\n    _extend_modes: vec2<u32>,\n    _image_padding: f32,\n) -> vec4<f32> {\n    let atlas_max = image_offset + image_size - vec2(1.0);\n    let atlas_uv_clamped = clamp(coords, image_offset, atlas_max);\n    let uv_quad = vec4(floor(atlas_uv_clamped), ceil(atlas_uv_clamped));\n    let uv_frac = fract(coords);\n    let a = textureLoad(tex, vec2<i32>(uv_quad.xy), atlas_idx, 0);\n    let b = textureLoad(tex, vec2<i32>(uv_quad.xw), atlas_idx, 0);\n    let c = textureLoad(tex, vec2<i32>(uv_quad.zy), atlas_idx, 0);\n    let d = textureLoad(tex, vec2<i32>(uv_quad.zw), atlas_idx, 0);\n    return mix(mix(a, b, uv_frac.y), mix(c, d, uv_frac.y), uv_frac.x);\n}\n\n// Bicubic filtering with a Mitchell filter (B=1/3, C=1/3): sample the 16\n// surrounding texels of the target point and interpolate them with a cubic\n// filter. The 4x4 matrix holds the coefficients of the cubic function used to\n// derive the weights from the fractional part of the sample location.\nfn bicubic_sample(\n    tex: texture_2d_array<f32>,\n    coords: vec2<f32>,\n    atlas_idx: i32,\n    image_offset: vec2<f32>,\n    image_size: vec2<f32>,\n    _extend_modes: vec2<u32>,\n    _image_padding: f32,\n) -> vec4<f32> {\n    let atlas_max = image_offset + image_size - vec2(1.0);\n    let frac_coords = fract(coords + 0.5);\n    // Cubic weights for the x and y directions.\n    let cx = cubic_weights(frac_coords.x);\n    let cy = cubic_weights(frac_coords.y);\n\n    // Sample the 4x4 grid around `coords`.\n    let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n    let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n    let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, atlas_max)), atlas_idx, 0);\n\n    let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n    let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, atlas_max)), atlas_idx, 0);\n\n    // Interpolate in the x direction for each row.\n    let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n    let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n    let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n    let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n    // Interpolate in the y direction.\n    let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n    // Clamp alpha first, then clamp the premultiplied color channels against it.\n    let a = clamp(result.a, 0.0, 1.0);\n    return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\n// Mitchell-Netravali cubic filter coefficients with B=1/3 and C=1/3, matching\n// the CPU rasterizer\'s cubic resampler.\nconst MF: array<vec4<f32>, 4> = array<vec4<f32>, 4>(\n    vec4<f32>(\n        (1.0 / 6.0) / 3.0,\n        -(3.0 / 6.0) / 3.0 - 1.0 / 3.0,\n        (3.0 / 6.0) / 3.0 + 2.0 * 1.0 / 3.0,\n        -(1.0 / 6.0) / 3.0 - 1.0 / 3.0\n    ),\n    vec4<f32>(\n        1.0 - (2.0 / 6.0) / 3.0,\n        0.0,\n        -3.0 + (12.0 / 6.0) / 3.0 + 1.0 / 3.0,\n        2.0 - (9.0 / 6.0) / 3.0 - 1.0 / 3.0\n    ),\n    vec4<f32>(\n        (1.0 / 6.0) / 3.0,\n        (3.0 / 6.0) / 3.0 + 1.0 / 3.0,\n        3.0 - (15.0 / 6.0) / 3.0 - 2.0 * 1.0 / 3.0,\n        -2.0 + (9.0 / 6.0) / 3.0 + 1.0 / 3.0\n    ),\n    vec4<f32>(\n        0.0,\n        0.0,\n        -1.0 / 3.0,\n        (1.0 / 6.0) / 3.0 + 1.0 / 3.0\n    )\n);\n\n// The four cubic weights for a single fractional value.\nfn cubic_weights(fract: f32) -> vec4<f32> {\n    return vec4<f32>(\n        single_weight(fract, MF[0][0], MF[0][1], MF[0][2], MF[0][3]),\n        single_weight(fract, MF[1][0], MF[1][1], MF[1][2], MF[1][3]),\n        single_weight(fract, MF[2][0], MF[2][1], MF[2][2], MF[2][3]),\n        single_weight(fract, MF[3][0], MF[3][1], MF[3][2], MF[3][3])\n    );\n}\n\n// One weight from the fractional value t and the cubic coefficients.\nfn single_weight(t: f32, a: f32, b: f32, c: f32, d: f32) -> f32 {\n    return t * (t * (t * d + c) + b) + a;\n}\n\n// -----------------------------------------------------------------------------\n// External-texture sampling\n// -----------------------------------------------------------------------------\n\n// The atlas-array samplers above, restated for a plain 2D texture: an\n// externally bound image is a whole texture rather than a page of the array.\n\nfn external_bilinear_sample(\n    tex: texture_2d<f32>,\n    coords: vec2<f32>,\n    image_offset: vec2<f32>,\n    image_size: vec2<f32>,\n) -> vec4<f32> {\n    let image_max = image_offset + image_size - vec2(1.0);\n    let clamped_coords = clamp(coords, image_offset, image_max);\n    let coord_quad = vec4(floor(clamped_coords), ceil(clamped_coords));\n    let coord_frac = fract(coords);\n    let a = textureLoad(tex, vec2<i32>(coord_quad.xy), 0);\n    let b = textureLoad(tex, vec2<i32>(coord_quad.xw), 0);\n    let c = textureLoad(tex, vec2<i32>(coord_quad.zy), 0);\n    let d = textureLoad(tex, vec2<i32>(coord_quad.zw), 0);\n    return mix(mix(a, b, coord_frac.y), mix(c, d, coord_frac.y), coord_frac.x);\n}\n\nfn external_bicubic_sample(\n    tex: texture_2d<f32>,\n    coords: vec2<f32>,\n    image_offset: vec2<f32>,\n    image_size: vec2<f32>,\n) -> vec4<f32> {\n    let image_max = image_offset + image_size - vec2(1.0);\n    let frac_coords = fract(coords + 0.5);\n    let cx = cubic_weights(frac_coords.x);\n    let cy = cubic_weights(frac_coords.y);\n\n    let s00 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -1.5), image_offset, image_max)), 0);\n    let s10 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -1.5), image_offset, image_max)), 0);\n    let s20 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -1.5), image_offset, image_max)), 0);\n    let s30 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -1.5), image_offset, image_max)), 0);\n\n    let s01 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, -0.5), image_offset, image_max)), 0);\n    let s11 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, -0.5), image_offset, image_max)), 0);\n    let s21 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, -0.5), image_offset, image_max)), 0);\n    let s31 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, -0.5), image_offset, image_max)), 0);\n\n    let s02 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 0.5), image_offset, image_max)), 0);\n    let s12 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 0.5), image_offset, image_max)), 0);\n    let s22 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 0.5), image_offset, image_max)), 0);\n    let s32 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 0.5), image_offset, image_max)), 0);\n\n    let s03 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-1.5, 1.5), image_offset, image_max)), 0);\n    let s13 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(-0.5, 1.5), image_offset, image_max)), 0);\n    let s23 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(0.5, 1.5), image_offset, image_max)), 0);\n    let s33 = textureLoad(tex, vec2<i32>(clamp(coords + vec2(1.5, 1.5), image_offset, image_max)), 0);\n\n    let row0 = cx.x * s00 + cx.y * s10 + cx.z * s20 + cx.w * s30;\n    let row1 = cx.x * s01 + cx.y * s11 + cx.z * s21 + cx.w * s31;\n    let row2 = cx.x * s02 + cx.y * s12 + cx.z * s22 + cx.w * s32;\n    let row3 = cx.x * s03 + cx.y * s13 + cx.z * s23 + cx.w * s33;\n    let result = cy.x * row0 + cy.y * row1 + cy.z * row2 + cy.w * row3;\n\n    // Clamp alpha first, then clamp the premultiplied color channels against it.\n    let a = clamp(result.a, 0.0, 1.0);\n    return vec4<f32>(clamp(result.rgb, vec3(0.0), vec3(a)), a);\n}\n\nfn sample_external_image(\n    tex: texture_2d<f32>,\n    quality: u32,\n    coords: vec2<f32>,\n    image_offset: vec2<f32>,\n    image_size: vec2<f32>,\n) -> vec4<f32> {\n    if quality == IMAGE_QUALITY_HIGH {\n        return external_bicubic_sample(tex, coords, image_offset, image_size);\n    }\n    if quality == IMAGE_QUALITY_MEDIUM {\n        return external_bilinear_sample(\n            tex,\n            coords - vec2(0.5),\n            image_offset,\n            image_size,\n        );\n    }\n    return textureLoad(tex, vec2<u32>(coords), 0);\n}\n// Copyright 2026 the Vello Authors\n// SPDX-License-Identifier: Apache-2.0 OR MIT\n\n// Derived from vello_sparse_shaders 0.2.0 (`shaders/copy.wesl`); its\n// `import package::helpers::...` lines are resolved by prepending\n// `helpers.wgsl` at load time (`gpu::shader_src`).\n//\n// Copies rectangular regions between intermediate textures, one instance per\n// region.\n\n// Must stay byte-compatible with the copy-instance vertex layout in\n// `gpu::pipelines`.\nstruct CopyInstance {\n    // Origin in the destination texture page, packed as u16s.\n    @location(0)\n    dest_texture_origin: u32,\n    // Origin in the source texture page, packed as u16s.\n    @location(1)\n    source_texture_origin: u32,\n    // Width and height of the copied region, packed as u16s.\n    @location(2)\n    copy_rect_size: u32,\n    // Width and height of the destination texture page, packed as u16s.\n    @location(3)\n    dest_texture_size: u32,\n}\n\nstruct VertexOutput {\n    @location(0)\n    source_texture_xy: vec2<f32>,\n    @builtin(position)\n    position: vec4<f32>,\n}\n\n@group(0) @binding(0)\nvar source_texture: texture_2d<f32>;\n\n@vertex\nfn vs_main(\n    @builtin(vertex_index) vertex_index: u32,\n    instance: CopyInstance,\n) -> VertexOutput {\n    let dest_texture_origin = unpack_u16_pair(instance.dest_texture_origin);\n    let source_texture_origin = unpack_u16_pair(instance.source_texture_origin);\n    let copy_rect_size = unpack_u16_pair(instance.copy_rect_size);\n    let dest_texture_size = unpack_u16_pair(instance.dest_texture_size);\n\n    let local = quad_corner(vertex_index) * vec2<f32>(copy_rect_size);\n    let dest_texture_xy = vec2<f32>(dest_texture_origin) + local;\n\n    var out: VertexOutput;\n    out.source_texture_xy = vec2<f32>(source_texture_origin) + local;\n\n    let ndc = pixel_to_ndc(dest_texture_xy, vec2<f32>(dest_texture_size));\n    out.position = vec4<f32>(ndc, 0.0, 1.0);\n    return out;\n}\n\n@fragment\nfn fs_main(\n    @location(0) source_texture_xy: vec2<f32>,\n) -> @location(0) vec4<f32> {\n    return textureLoad(source_texture, vec2<i32>(source_texture_xy), 0);\n}\n";
Expand description

Rectangular region copies between intermediate textures: vs_main + fs_main.