frust_engine/gpu/present.rs
1//! The present pass: the engine's premultiplied frame, converted for a
2//! swapchain that stores STRAIGHT alpha.
3//!
4//! Every engine pipeline blends and writes premultiplied alpha (see
5//! [`super::pipelines`]), which is exactly what an opaque swapchain — or a
6//! premultiplied-expecting translucent one, Android's `Inherit` — wants. The
7//! one composite alpha mode that disagrees is iOS's `PostMultiplied`: it reads
8//! the frame's `(C·a, a)` as `(C, a)`, so every partial-alpha pixel presents
9//! too dark. [`UnpremultiplyPass`] is what serves that surface — the frame is
10//! rendered into an intermediate of the host's own and one full-screen
11//! triangle writes `(rgb / max(a, ALPHA_FLOOR), a)` into the swapchain, in
12//! `unpremultiply.wgsl`.
13//!
14//! **A render pass, never a compute one.** The conversion is a fragment write
15//! into an ordinary `RENDER_ATTACHMENT` colour target: neither a compute stage
16//! nor a storage texture (E1/E2), both of which the engine refuses by design
17//! and neither of which a downlevel target offers. A swapchain configured for
18//! this tier carries `RENDER_ATTACHMENT` and nothing else, so a storage write
19//! into it is not merely disallowed by the rules — it is unavailable.
20//!
21//! **The host decides, the pass states the rule.** Which surfaces need it is
22//! [`UnpremultiplyPass::selected_by`]: an [`OutputAlpha::Straight`] *swapchain*
23//! does, an [`OutputAlpha::Premultiplied`] one does not. Note the split that
24//! implies for a host on this arm — the intermediate the engine renders into
25//! really does hold premultiplied pixels, so it is described to
26//! [`crate::EngineRenderer::encode`] as [`OutputAlpha::Premultiplied`] (which
27//! is what makes the frame's base colour clear in the same convention as
28//! everything drawn over it); `Straight` describes the *swapchain* this pass
29//! then writes, not the buffer it reads.
30//!
31//! **The caller owns the encoder.** [`UnpremultiplyPass::record`] records into
32//! a `wgpu::CommandEncoder` it is handed and never submits, the same
33//! single-submit contract `frust_gpu::encoder` states and
34//! [`crate::EngineRenderer::encode`] keeps: the frame's own passes and this one
35//! land in one command buffer, in order, against the texture about to be
36//! presented.
37
38use std::sync::Arc;
39
40use frust_gpu::lint::PipelineLayoutDesc;
41use frust_gpu::{PipelineCache, RenderPipelineDesc, ShaderId, ShaderLibrary};
42
43use super::pipelines::{FS_MAIN, VS_MAIN};
44use crate::OutputAlpha;
45
46/// The conversion program, loaded from the shader directory like every other
47/// engine module so `frust_gpu::lint::lint_wgsl_dir` sees every line the GPU
48/// compiles.
49///
50/// Not assembled in [`super::shader_src`] with the pipeline modules: this is
51/// the host's *present* pass rather than one of the frame's own, it needs no
52/// helper prelude, and it is compiled per surface by [`UnpremultiplyPass::new`]
53/// rather than registered in the renderer's shared library.
54pub const UNPREMULTIPLY: &str = include_str!("../../shaders/unpremultiply.wgsl");
55
56/// The name [`UNPREMULTIPLY`] is registered under in this pass's own shader
57/// library.
58pub const UNPREMULTIPLY_NAME: &str = "frust-engine unpremultiply";
59
60/// The floor the shader divides by, mirroring its own `ALPHA_FLOOR` constant.
61///
62/// Stated here so a host can document the guard it is buying without reading
63/// WGSL. A test below pins the two against each other, since a divisor guard
64/// that drifted would only show up as NaN pixels on a device.
65pub const ALPHA_FLOOR: f32 = 1e-4;
66
67/// Vertices the full-screen triangle draws. One primitive, no vertex buffer.
68const FULLSCREEN_VERTICES: u32 = 3;
69
70/// Bind groups the program declares: its source texture, and nothing else.
71const UNPREMULTIPLY_BIND_GROUPS: usize = 1;
72
73/// The straight-alpha present pass for one surface: a pipeline built for that
74/// surface's own format plus the bind-group layout its program derived.
75///
76/// Built once per surface configure (a rare event) — a pipeline object, a
77/// bind-group layout and nothing per frame but one bind group.
78#[derive(Debug)]
79pub struct UnpremultiplyPass {
80 pipeline: wgpu::RenderPipeline,
81 /// wgpu's default layout, derived from the program — the only layout
82 /// `frust_gpu::RenderPipelineDesc` builds with, so it is read back off the
83 /// pipeline rather than declared twice.
84 bind_group_layout: wgpu::BindGroupLayout,
85 format: wgpu::TextureFormat,
86}
87
88impl UnpremultiplyPass {
89 /// Whether a swapchain interpreting its alpha as `output` needs this pass.
90 ///
91 /// The whole routing rule, in one place: `Straight` does, `Premultiplied`
92 /// does not — the engine already writes the latter's convention, and a
93 /// conversion there would divide every partial-alpha pixel by its own alpha
94 /// for nothing.
95 #[must_use]
96 pub fn selected_by(output: OutputAlpha) -> bool {
97 matches!(output, OutputAlpha::Straight)
98 }
99
100 /// Builds the pass for a `format` target.
101 ///
102 /// `driver_cache` is the host's persisted `wgpu::PipelineCache` when it has
103 /// one — `None` on every backend but Vulkan, which is every backend this
104 /// arm actually runs on today (it exists for iOS/Metal), so the parameter
105 /// is about keeping the seam honest rather than about a cache hit.
106 ///
107 /// The pipeline is built through a `frust_gpu::PipelineCache` of this
108 /// pass's own, over a one-module library: the substrate's rule is that a
109 /// frame is never the first place a pipeline is compiled, and a surface
110 /// configure is not a frame. The cache is not kept — the compiled pipeline
111 /// is a reference-counted handle that outlives it, and there is exactly one
112 /// variant to ask for.
113 #[must_use]
114 pub fn new(
115 device: &wgpu::Device,
116 format: wgpu::TextureFormat,
117 driver_cache: Option<&wgpu::PipelineCache>,
118 ) -> Self {
119 let mut library = ShaderLibrary::new();
120 let shader = library.insert_wgsl(device, UNPREMULTIPLY_NAME, UNPREMULTIPLY);
121 let mut pipelines = PipelineCache::new(Arc::new(library), driver_cache.cloned());
122 let pipeline = pipelines
123 .get_or_create(device, &Self::desc(shader, format))
124 .clone();
125 let bind_group_layout = pipeline.get_bind_group_layout(0);
126 Self {
127 pipeline,
128 bind_group_layout,
129 format,
130 }
131 }
132
133 /// The target format this pass was built for — the surface's own, since it
134 /// writes the swapchain directly.
135 #[must_use]
136 pub fn format(&self) -> wgpu::TextureFormat {
137 self.format
138 }
139
140 /// Records the conversion into `encoder`: reads `source_view` (the
141 /// premultiplied intermediate the frame was rendered into) and writes
142 /// straight-alpha pixels into `target_view` (the acquired swapchain
143 /// texture). The caller submits `encoder`.
144 ///
145 /// No extent is taken: the triangle covers the whole destination whatever
146 /// its size, and the fragment stage reads the source at its own position,
147 /// so the two views only have to agree with each other — which the surface
148 /// that owns both guarantees by recreating them together.
149 ///
150 /// The attachment is CLEARED rather than loaded even though every pixel of
151 /// it is then written: the destination is a fresh swapchain image whose
152 /// prior contents mean nothing, and a load would cost a tile read per tile
153 /// on exactly the mobile GPUs this tier targets.
154 ///
155 /// The bind group is built per call rather than cached, mirroring
156 /// `frust-render`'s own premultiply pass: it names the source view, which a
157 /// resize replaces, so caching it would need an invalidation seam to buy one
158 /// allocation a frame.
159 ///
160 /// `timestamp_writes` is the pass's own `timestamp_writes`, verbatim — the
161 /// host asks its `frust_gpu::diag::TimestampRing` (via
162 /// `frust_engine::FrameTimestamps`) for one fresh pair charged to
163 /// [`crate::diag::EngineSpan::Blit`] and hands it straight through, `None`
164 /// on every arm that does not time this pass (an inert ring, a tier build
165 /// with no `perf-trace`, or the `EngineDirect` arm this pass never runs on
166 /// at all).
167 pub fn record(
168 &self,
169 device: &wgpu::Device,
170 encoder: &mut wgpu::CommandEncoder,
171 source_view: &wgpu::TextureView,
172 target_view: &wgpu::TextureView,
173 timestamp_writes: Option<wgpu::RenderPassTimestampWrites<'_>>,
174 ) {
175 let bind_group = device.create_bind_group(&wgpu::BindGroupDescriptor {
176 label: Some("frust-engine unpremultiply bind group"),
177 layout: &self.bind_group_layout,
178 entries: &[wgpu::BindGroupEntry {
179 binding: 0,
180 resource: wgpu::BindingResource::TextureView(source_view),
181 }],
182 });
183 let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
184 label: Some("frust-engine unpremultiply pass"),
185 color_attachments: &[Some(wgpu::RenderPassColorAttachment {
186 view: target_view,
187 depth_slice: None,
188 resolve_target: None,
189 ops: wgpu::Operations {
190 load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT),
191 store: wgpu::StoreOp::Store,
192 },
193 })],
194 depth_stencil_attachment: None,
195 timestamp_writes,
196 occlusion_query_set: None,
197 multiview_mask: None,
198 });
199 pass.set_pipeline(&self.pipeline);
200 pass.set_bind_group(0, &bind_group, &[]);
201 pass.draw(0..FULLSCREEN_VERTICES, 0..1);
202 }
203
204 /// This pass's shape as the downlevel lint reads it, so the bind-group,
205 /// vertex-topology and sample-count ceilings
206 /// (`frust_gpu::lint::lint_pipeline_layout`) are checked against the real
207 /// description rather than restated in prose.
208 #[must_use]
209 pub fn layout_desc() -> PipelineLayoutDesc {
210 PipelineLayoutDesc {
211 bind_group_count: UNPREMULTIPLY_BIND_GROUPS,
212 // The program generates its own geometry, so there is no vertex
213 // buffer to declare and nothing to step over.
214 max_vertex_buffers: 0,
215 total_vertex_attributes: 0,
216 max_vertex_buffer_stride: 0,
217 sample_count: 1,
218 uniform_buffer_sizes: Vec::new(),
219 }
220 }
221
222 /// The pipeline description, which is `RenderPipelineDesc::new`'s minimal
223 /// variant unchanged — every one of its defaults is what this pass wants:
224 /// no vertex layouts (the program generates its own geometry), no blend
225 /// (the pass REPLACES every pixel of the destination — a blend would
226 /// composite the frame onto whatever the swapchain image last held), no
227 /// depth, one sample, and a triangle list, whose single primitive is the
228 /// full-screen triangle itself.
229 fn desc(shader: ShaderId, format: wgpu::TextureFormat) -> RenderPipelineDesc {
230 RenderPipelineDesc::new(shader, VS_MAIN, FS_MAIN, format)
231 }
232}
233
234#[cfg(test)]
235mod tests {
236 use super::*;
237
238 /// The pass is selected by the swapchain's own alpha convention, and by
239 /// nothing else.
240 #[test]
241 fn only_a_straight_alpha_target_selects_the_pass() {
242 assert!(UnpremultiplyPass::selected_by(OutputAlpha::Straight));
243 assert!(!UnpremultiplyPass::selected_by(OutputAlpha::Premultiplied));
244 }
245
246 /// The value the shader's own `ALPHA_FLOOR` declaration carries, read out
247 /// of the source the GPU compiles rather than restated here.
248 fn shader_alpha_floor() -> f32 {
249 let declaration = UNPREMULTIPLY
250 .lines()
251 .find(|line| line.trim_start().starts_with("const ALPHA_FLOOR"))
252 .expect("the shader declares an ALPHA_FLOOR constant");
253 declaration
254 .split('=')
255 .nth(1)
256 .expect("the declaration assigns a value")
257 .trim()
258 .trim_end_matches(';')
259 .parse()
260 .expect("the shader's alpha floor is a float literal")
261 }
262
263 /// The divisor guard is one decision spelled in two languages; a drift
264 /// between them would surface only as NaN pixels on a device.
265 #[test]
266 fn the_shader_and_rust_alpha_floors_agree() {
267 let floor = shader_alpha_floor();
268 assert_eq!(
269 floor, ALPHA_FLOOR,
270 "the shader's floor must stay in step with `ALPHA_FLOOR`"
271 );
272 assert!(
273 floor > 0.0,
274 "a floor of zero would leave `0 / 0` NaNs in the swapchain"
275 );
276 assert!(
277 floor < 1.0 / 255.0,
278 "the floor must sit below the smallest representable 8-bit alpha, so it never clamps \
279 a pixel that carries colour"
280 );
281 }
282
283 /// The program the host compiles is the file the WGSL lint scans, not an
284 /// inline string — the shader directory's own rule.
285 #[test]
286 fn the_program_is_loaded_from_the_shader_directory() {
287 assert!(UNPREMULTIPLY.contains("fn vs_main"));
288 assert!(UNPREMULTIPLY.contains("fn fs_main"));
289 assert!(
290 !UNPREMULTIPLY.contains("@compute"),
291 "the conversion is a render pass, never a compute one (E1)"
292 );
293 assert!(
294 !UNPREMULTIPLY.contains("texture_storage_"),
295 "a swapchain on this tier is RENDER_ATTACHMENT-only (E2)"
296 );
297 }
298
299 #[test]
300 fn the_pass_layout_passes_the_downlevel_lint() {
301 let violations = frust_gpu::lint_pipeline_layout(&UnpremultiplyPass::layout_desc());
302 assert!(
303 violations.is_empty(),
304 "the present pass violates a downlevel design rule: {violations:?}"
305 );
306 assert_eq!(
307 UnpremultiplyPass::layout_desc().bind_group_count,
308 UNPREMULTIPLY_BIND_GROUPS,
309 "one bind group: the source texture"
310 );
311 }
312}