Skip to main content

frust_engine/gpu/
present.rs

1//! The present pass: the engine's premultiplied frame, converted for a
2//! swapchain that stores STRAIGHT alpha.
3//!
4//! Every engine pipeline blends and writes premultiplied alpha (see
5//! [`super::pipelines`]), which is exactly what an opaque swapchain — or a
6//! premultiplied-expecting translucent one, Android's `Inherit` — wants. The
7//! one composite alpha mode that disagrees is iOS's `PostMultiplied`: it reads
8//! the frame's `(C·a, a)` as `(C, a)`, so every partial-alpha pixel presents
9//! too dark. [`UnpremultiplyPass`] is what serves that surface — the frame is
10//! rendered into an intermediate of the host's own and one full-screen
11//! triangle writes `(rgb / max(a, ALPHA_FLOOR), a)` into the swapchain, in
12//! `unpremultiply.wgsl`.
13//!
14//! **A render pass, never a compute one.** The conversion is a fragment write
15//! into an ordinary `RENDER_ATTACHMENT` colour target: neither a compute stage
16//! nor a storage texture (E1/E2), both of which the engine refuses by design
17//! and neither of which a downlevel target offers. A swapchain configured for
18//! this tier carries `RENDER_ATTACHMENT` and nothing else, so a storage write
19//! into it is not merely disallowed by the rules — it is unavailable.
20//!
21//! **The host decides, the pass states the rule.** Which surfaces need it is
22//! [`UnpremultiplyPass::selected_by`]: an [`OutputAlpha::Straight`] *swapchain*
23//! does, an [`OutputAlpha::Premultiplied`] one does not. Note the split that
24//! implies for a host on this arm — the intermediate the engine renders into
25//! really does hold premultiplied pixels, so it is described to
26//! [`crate::EngineRenderer::encode`] as [`OutputAlpha::Premultiplied`] (which
27//! is what makes the frame's base colour clear in the same convention as
28//! everything drawn over it); `Straight` describes the *swapchain* this pass
29//! then writes, not the buffer it reads.
30//!
31//! **The caller owns the encoder.** [`UnpremultiplyPass::record`] records into
32//! a `wgpu::CommandEncoder` it is handed and never submits, the same
33//! single-submit contract `frust_gpu::encoder` states and
34//! [`crate::EngineRenderer::encode`] keeps: the frame's own passes and this one
35//! land in one command buffer, in order, against the texture about to be
36//! presented.
37
38use std::sync::Arc;
39
40use frust_gpu::lint::PipelineLayoutDesc;
41use frust_gpu::{PipelineCache, RenderPipelineDesc, ShaderId, ShaderLibrary};
42
43use super::pipelines::{FS_MAIN, VS_MAIN};
44use crate::OutputAlpha;
45
46/// The conversion program, loaded from the shader directory like every other
47/// engine module so `frust_gpu::lint::lint_wgsl_dir` sees every line the GPU
48/// compiles.
49///
50/// Not assembled in [`super::shader_src`] with the pipeline modules: this is
51/// the host's *present* pass rather than one of the frame's own, it needs no
52/// helper prelude, and it is compiled per surface by [`UnpremultiplyPass::new`]
53/// rather than registered in the renderer's shared library.
54pub const UNPREMULTIPLY: &str = include_str!("../../shaders/unpremultiply.wgsl");
55
56/// The name [`UNPREMULTIPLY`] is registered under in this pass's own shader
57/// library.
58pub const UNPREMULTIPLY_NAME: &str = "frust-engine unpremultiply";
59
60/// The floor the shader divides by, mirroring its own `ALPHA_FLOOR` constant.
61///
62/// Stated here so a host can document the guard it is buying without reading
63/// WGSL. A test below pins the two against each other, since a divisor guard
64/// that drifted would only show up as NaN pixels on a device.
65pub const ALPHA_FLOOR: f32 = 1e-4;
66
67/// Vertices the full-screen triangle draws. One primitive, no vertex buffer.
68const FULLSCREEN_VERTICES: u32 = 3;
69
70/// Bind groups the program declares: its source texture, and nothing else.
71const UNPREMULTIPLY_BIND_GROUPS: usize = 1;
72
73/// The straight-alpha present pass for one surface: a pipeline built for that
74/// surface's own format plus the bind-group layout its program derived.
75///
76/// Built once per surface configure (a rare event) — a pipeline object, a
77/// bind-group layout and nothing per frame but one bind group.
78#[derive(Debug)]
79pub struct UnpremultiplyPass {
80    pipeline: wgpu::RenderPipeline,
81    /// wgpu's default layout, derived from the program — the only layout
82    /// `frust_gpu::RenderPipelineDesc` builds with, so it is read back off the
83    /// pipeline rather than declared twice.
84    bind_group_layout: wgpu::BindGroupLayout,
85    format: wgpu::TextureFormat,
86}
87
88impl UnpremultiplyPass {
89    /// Whether a swapchain interpreting its alpha as `output` needs this pass.
90    ///
91    /// The whole routing rule, in one place: `Straight` does, `Premultiplied`
92    /// does not — the engine already writes the latter's convention, and a
93    /// conversion there would divide every partial-alpha pixel by its own alpha
94    /// for nothing.
95    #[must_use]
96    pub fn selected_by(output: OutputAlpha) -> bool {
97        matches!(output, OutputAlpha::Straight)
98    }
99
100    /// Builds the pass for a `format` target.
101    ///
102    /// `driver_cache` is the host's persisted `wgpu::PipelineCache` when it has
103    /// one — `None` on every backend but Vulkan, which is every backend this
104    /// arm actually runs on today (it exists for iOS/Metal), so the parameter
105    /// is about keeping the seam honest rather than about a cache hit.
106    ///
107    /// The pipeline is built through a `frust_gpu::PipelineCache` of this
108    /// pass's own, over a one-module library: the substrate's rule is that a
109    /// frame is never the first place a pipeline is compiled, and a surface
110    /// configure is not a frame. The cache is not kept — the compiled pipeline
111    /// is a reference-counted handle that outlives it, and there is exactly one
112    /// variant to ask for.
113    #[must_use]
114    pub fn new(
115        device: &wgpu::Device,
116        format: wgpu::TextureFormat,
117        driver_cache: Option<&wgpu::PipelineCache>,
118    ) -> Self {
119        let mut library = ShaderLibrary::new();
120        let shader = library.insert_wgsl(device, UNPREMULTIPLY_NAME, UNPREMULTIPLY);
121        let mut pipelines = PipelineCache::new(Arc::new(library), driver_cache.cloned());
122        let pipeline = pipelines
123            .get_or_create(device, &Self::desc(shader, format))
124            .clone();
125        let bind_group_layout = pipeline.get_bind_group_layout(0);
126        Self {
127            pipeline,
128            bind_group_layout,
129            format,
130        }
131    }
132
133    /// The target format this pass was built for — the surface's own, since it
134    /// writes the swapchain directly.
135    #[must_use]
136    pub fn format(&self) -> wgpu::TextureFormat {
137        self.format
138    }
139
140    /// Records the conversion into `encoder`: reads `source_view` (the
141    /// premultiplied intermediate the frame was rendered into) and writes
142    /// straight-alpha pixels into `target_view` (the acquired swapchain
143    /// texture). The caller submits `encoder`.
144    ///
145    /// No extent is taken: the triangle covers the whole destination whatever
146    /// its size, and the fragment stage reads the source at its own position,
147    /// so the two views only have to agree with each other — which the surface
148    /// that owns both guarantees by recreating them together.
149    ///
150    /// The attachment is CLEARED rather than loaded even though every pixel of
151    /// it is then written: the destination is a fresh swapchain image whose
152    /// prior contents mean nothing, and a load would cost a tile read per tile
153    /// on exactly the mobile GPUs this tier targets.
154    ///
155    /// The bind group is built per call rather than cached, mirroring
156    /// `frust-render`'s own premultiply pass: it names the source view, which a
157    /// resize replaces, so caching it would need an invalidation seam to buy one
158    /// allocation a frame.
159    ///
160    /// `timestamp_writes` is the pass's own `timestamp_writes`, verbatim — the
161    /// host asks its `frust_gpu::diag::TimestampRing` (via
162    /// `frust_engine::FrameTimestamps`) for one fresh pair charged to
163    /// [`crate::diag::EngineSpan::Blit`] and hands it straight through, `None`
164    /// on every arm that does not time this pass (an inert ring, a tier build
165    /// with no `perf-trace`, or the `EngineDirect` arm this pass never runs on
166    /// at all).
167    pub fn record(
168        &self,
169        device: &wgpu::Device,
170        encoder: &mut wgpu::CommandEncoder,
171        source_view: &wgpu::TextureView,
172        target_view: &wgpu::TextureView,
173        timestamp_writes: Option<wgpu::RenderPassTimestampWrites<'_>>,
174    ) {
175        let bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
176            label: Some("frust-engine unpremultiply bind group"),
177            layout: &self.bind_group_layout,
178            entries: &[wgpu::BindGroupEntry {
179                binding: 0,
180                resource: wgpu::BindingResource::TextureView(source_view),
181            }],
182        });
183        let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
184            label: Some("frust-engine unpremultiply pass"),
185            color_attachments: &[Some(wgpu::RenderPassColorAttachment {
186                view: target_view,
187                depth_slice: None,
188                resolve_target: None,
189                ops: wgpu::Operations {
190                    load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT),
191                    store: wgpu::StoreOp::Store,
192                },
193            })],
194            depth_stencil_attachment: None,
195            timestamp_writes,
196            occlusion_query_set: None,
197            multiview_mask: None,
198        });
199        pass.set_pipeline(&self.pipeline);
200        pass.set_bind_group(0, &bind_group, &[]);
201        pass.draw(0..FULLSCREEN_VERTICES, 0..1);
202    }
203
204    /// This pass's shape as the downlevel lint reads it, so the bind-group,
205    /// vertex-topology and sample-count ceilings
206    /// (`frust_gpu::lint::lint_pipeline_layout`) are checked against the real
207    /// description rather than restated in prose.
208    #[must_use]
209    pub fn layout_desc() -> PipelineLayoutDesc {
210        PipelineLayoutDesc {
211            bind_group_count: UNPREMULTIPLY_BIND_GROUPS,
212            // The program generates its own geometry, so there is no vertex
213            // buffer to declare and nothing to step over.
214            max_vertex_buffers: 0,
215            total_vertex_attributes: 0,
216            max_vertex_buffer_stride: 0,
217            sample_count: 1,
218            uniform_buffer_sizes: Vec::new(),
219        }
220    }
221
222    /// The pipeline description, which is `RenderPipelineDesc::new`'s minimal
223    /// variant unchanged — every one of its defaults is what this pass wants:
224    /// no vertex layouts (the program generates its own geometry), no blend
225    /// (the pass REPLACES every pixel of the destination — a blend would
226    /// composite the frame onto whatever the swapchain image last held), no
227    /// depth, one sample, and a triangle list, whose single primitive is the
228    /// full-screen triangle itself.
229    fn desc(shader: ShaderId, format: wgpu::TextureFormat) -> RenderPipelineDesc {
230        RenderPipelineDesc::new(shader, VS_MAIN, FS_MAIN, format)
231    }
232}
233
234#[cfg(test)]
235mod tests {
236    use super::*;
237
238    /// The pass is selected by the swapchain's own alpha convention, and by
239    /// nothing else.
240    #[test]
241    fn only_a_straight_alpha_target_selects_the_pass() {
242        assert!(UnpremultiplyPass::selected_by(OutputAlpha::Straight));
243        assert!(!UnpremultiplyPass::selected_by(OutputAlpha::Premultiplied));
244    }
245
246    /// The value the shader's own `ALPHA_FLOOR` declaration carries, read out
247    /// of the source the GPU compiles rather than restated here.
248    fn shader_alpha_floor() -> f32 {
249        let declaration = UNPREMULTIPLY
250            .lines()
251            .find(|line| line.trim_start().starts_with("const ALPHA_FLOOR"))
252            .expect("the shader declares an ALPHA_FLOOR constant");
253        declaration
254            .split('=')
255            .nth(1)
256            .expect("the declaration assigns a value")
257            .trim()
258            .trim_end_matches(';')
259            .parse()
260            .expect("the shader's alpha floor is a float literal")
261    }
262
263    /// The divisor guard is one decision spelled in two languages; a drift
264    /// between them would surface only as NaN pixels on a device.
265    #[test]
266    fn the_shader_and_rust_alpha_floors_agree() {
267        let floor = shader_alpha_floor();
268        assert_eq!(
269            floor, ALPHA_FLOOR,
270            "the shader's floor must stay in step with `ALPHA_FLOOR`"
271        );
272        assert!(
273            floor > 0.0,
274            "a floor of zero would leave `0 / 0` NaNs in the swapchain"
275        );
276        assert!(
277            floor < 1.0 / 255.0,
278            "the floor must sit below the smallest representable 8-bit alpha, so it never clamps \
279             a pixel that carries colour"
280        );
281    }
282
283    /// The program the host compiles is the file the WGSL lint scans, not an
284    /// inline string — the shader directory's own rule.
285    #[test]
286    fn the_program_is_loaded_from_the_shader_directory() {
287        assert!(UNPREMULTIPLY.contains("fn vs_main"));
288        assert!(UNPREMULTIPLY.contains("fn fs_main"));
289        assert!(
290            !UNPREMULTIPLY.contains("@compute"),
291            "the conversion is a render pass, never a compute one (E1)"
292        );
293        assert!(
294            !UNPREMULTIPLY.contains("texture_storage_"),
295            "a swapchain on this tier is RENDER_ATTACHMENT-only (E2)"
296        );
297    }
298
299    #[test]
300    fn the_pass_layout_passes_the_downlevel_lint() {
301        let violations = frust_gpu::lint_pipeline_layout(&UnpremultiplyPass::layout_desc());
302        assert!(
303            violations.is_empty(),
304            "the present pass violates a downlevel design rule: {violations:?}"
305        );
306        assert_eq!(
307            UnpremultiplyPass::layout_desc().bind_group_count,
308            UNPREMULTIPLY_BIND_GROUPS,
309            "one bind group: the source texture"
310        );
311    }
312}