Skip to main content

kui_wgpu/
lib.rs

1//! wgpu renderer for kui: draws a [`kui_core::DisplayList`] with one instanced pipeline, a single draw call per frame.
2//!
3//! kui splits a UI into a model that lays out and paints into a display
4//! list (`kui-core`), a renderer that puts that list on screen (this
5//! crate) and a runner that owns the window and the event loop
6//! (`kui-native`, the one most apps use). Reach for `kui-wgpu` directly
7//! when you are writing your own runner: you already have a window, or an
8//! event loop, that `kui-native` does not fit.
9//!
10//! A [`Renderer`] owns the swapchain of one window and a GPU copy of the
11//! core's glyph atlas, kept in step with the atlas it is handed each
12//! frame. Rounded rectangles, borders, shadows, glyphs and images are all
13//! instances of the same quad, so an ordinary frame is one draw call;
14//! only a texture-backed image or a custom fragment shader splits it.
15//! Several windows share one device through [`Gpu`].
16//!
17//! # Example
18//!
19//! A runner's whole life with the renderer: open it on a window, tell the
20//! core whether subpixel text will render, then build, draw and present
21//! one frame at a time. `window` is anything wgpu can make a surface
22//! from, such as a `winit` window.
23//!
24//! ```rust,no_run
25//! use kui_core::{Core, Size, TextStyle};
26//! use kui_wgpu::{RenderError, Renderer};
27//!
28//! fn run(
29//!     window: impl Into<kui_wgpu::wgpu::SurfaceTarget<'static>>,
30//! ) -> Result<(), Box<dyn std::error::Error>> {
31//!     let (width, height) = (800u32, 600u32);
32//!     let mut renderer = pollster::block_on(Renderer::new(window, width, height))?;
33//!     let mut core = Core::new();
34//!     core.set_subpixel_text(renderer.subpixel_text());
35//!
36//!     loop {
37//!         // When the windowing library reports a new size:
38//!         // renderer.resize(new_width, new_height);
39//!
40//!         // Build the frame through the core, in logical pixels.
41//!         let scale = 1.0;
42//!         let viewport = Size::new(width as f32 / scale, height as f32 / scale);
43//!         let mut ui = core.frame(viewport, scale);
44//!         ui.text("Hello from a custom runner", TextStyle::new(24.0));
45//!         ui.finish();
46//!
47//!         // Draw it. The atlas is `&mut` so the renderer can clear its dirty flag.
48//!         let (list, atlas) = core.output();
49//!         match renderer.render(list, atlas) {
50//!             Ok(report) => {
51//!                 let _blocked_on_vsync_ms = report.vsync_wait_ms;
52//!             }
53//!             Err(RenderError::Reconfigure | RenderError::Validation) => {
54//!                 renderer.resize(width, height);
55//!             }
56//!             Err(RenderError::Skip) => {}
57//!             Err(RenderError::DeviceLost) => {
58//!                 // Open a new `Renderer` (and a new device) and carry on.
59//!                 break;
60//!             }
61//!         }
62//!     }
63//!     Ok(())
64//! }
65//! ```
66//!
67//! # Where to look
68//!
69//! - [`Renderer`]: one window's swapchain, pipelines and atlas texture.
70//! - [`Renderer::render`]: a display list in, a presented frame (or a
71//!   [`RenderError`]) out.
72//! - [`Renderer::resize`]: reconfigure after the window changed size.
73//! - [`Renderer::subpixel_text`]: what to pass to `Core::set_subpixel_text`.
74//! - [`Gpu`]: the device, queue and adapter that windows share;
75//!   [`Renderer::new_in`] opens a second window on it.
76//! - [`RenderError`]: what each failed frame asks the runner to do next.
77//! - [`DEFAULT_FRAME_LATENCY`] and [`Renderer::set_frame_latency`]: how
78//!   many frames may queue ahead of the one on screen.
79//! - [`wgpu`] is re-exported, so a runner builds against the same version
80//!   this crate was.
81//!
82//! # Subpixel text
83//!
84//! Where the device offers dual-source blending (Metal, DX12, most Vulkan)
85//! the pipeline blends per channel, which is what LCD subpixel glyphs need.
86//! Elsewhere it falls back to ordinary alpha blending and the core should
87//! rasterize grayscale masks instead, which is what
88//! [`Renderer::subpixel_text`] tells it.
89//!
90//! The book: <https://kui-book.qxuken.dev>. Repository:
91//! <https://github.com/qxuken/kui>.
92
93pub use wgpu;
94
95mod backdrop;
96
97use kui_core::atlas::GlyphAtlas;
98use kui_core::{Clip, DisplayList, Quad, QuadKind};
99
100#[repr(C)]
101#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
102struct Instance {
103    pos: [f32; 2],
104    size: [f32; 2],
105    color: [f32; 4],
106    border_color: [f32; 4],
107    params: [f32; 4],
108    uv: [f32; 4],
109    clip: [f32; 4],
110    /// Corner radii, clockwise from the top-left.
111    radii: [f32; 4],
112    /// Radii of the clip itself; all zero = a plain rect clip.
113    clip_radii: [f32; 4],
114}
115
116/// The frame's own numbers, at group 0 binding 0 for both pipelines.
117/// `kui_core::fragment::PRELUDE` declares the same bytes as `KuiGlobals`
118/// so an app's fragment can read `time` and `scale`; the padding is what
119/// makes the struct a multiple of sixteen, which a uniform must be.
120#[repr(C)]
121#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
122struct Globals {
123    viewport: [f32; 2],
124    atlas_size: [f32; 2],
125    time: f32,
126    scale: f32,
127    _pad: [f32; 2],
128}
129
130/// One fragment's parameters as the shader takes them — the sixteen
131/// floats and the texel rect of its `image`, laid out as the epilogue's
132/// `KuiFragmentParams` — padded out to the device's dynamic-offset
133/// alignment so a frame's draws can share one buffer and pick their slot
134/// by offset.
135#[repr(C)]
136#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
137struct FragmentParams {
138    params: [f32; 16],
139    /// `FragmentIn::image`: `[x, y, w, h]` in the texture bound at group 0
140    /// for this draw — the atlas, or the image's own; zero with none.
141    image: [f32; 4],
142}
143
144/// The clip is resolved out of the frame's table here rather than read off
145/// the quad: it rides as an index (`kui_core::ClipId`) so the display list
146/// carries it once per distinct clip instead of once per quad.
147fn instance_of(q: &Quad, clips: &[Clip], textures: &[kui_core::display::TextureDraw]) -> Instance {
148    let clip = clips.get(q.clip as usize).copied().unwrap_or(Clip::NONE);
149    let kind = match q.kind {
150        QuadKind::Solid => 0.0,
151        QuadKind::GlyphMask => 1.0,
152        QuadKind::GlyphColor => 2.0,
153        QuadKind::Image => 3.0,
154        QuadKind::GlyphSubpixel => 4.0,
155        QuadKind::Shadow => 5.0,
156        QuadKind::Segment => 6.0,
157        QuadKind::Fragment => 7.0,
158        // Drawn by the image branch with its own texture bound in the
159        // atlas's place.
160        QuadKind::Texture => 3.0,
161        // Never drawn by this pipeline: the pass breaks at it and
162        // `backdrop` blurs what is under it. A solid with no colour, so
163        // one a frame did not plan a blur for draws nothing in a run.
164        QuadKind::Backdrop => 0.0,
165    };
166    // `uv` is atlas texels on every kind but two: a segment carries its
167    // endpoints there as f32 bits, and a texture quad an index into the
168    // side list whose entry holds the texel rect. The shader wants floats.
169    let uv = if q.kind == QuadKind::Segment {
170        q.segment_ends()
171    } else if q.kind == QuadKind::Texture {
172        let uv = textures.get(q.uv[0] as usize).map_or([0; 4], |t| t.uv);
173        [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32]
174    } else {
175        [
176            q.uv[0] as f32,
177            q.uv[1] as f32,
178            q.uv[2] as f32,
179            q.uv[3] as f32,
180        ]
181    };
182    if q.kind == QuadKind::Backdrop {
183        return Instance {
184            pos: [q.rect.x, q.rect.y],
185            size: [0.0, 0.0],
186            color: [0.0; 4],
187            border_color: [0.0; 4],
188            params: [0.0; 4],
189            uv: [0.0; 4],
190            clip: [0.0; 4],
191            radii: [0.0; 4],
192            clip_radii: [0.0; 4],
193        };
194    }
195    Instance {
196        pos: [q.rect.x, q.rect.y],
197        size: [q.rect.w, q.rect.h],
198        color: [q.color.r, q.color.g, q.color.b, q.color.a],
199        border_color: [
200            q.border_color.r,
201            q.border_color.g,
202            q.border_color.b,
203            q.border_color.a,
204        ],
205        params: [q.blur, q.border_w, kind, 0.0],
206        uv,
207        clip: [clip.rect.x, clip.rect.y, clip.rect.w, clip.rect.h],
208        radii: q.radius,
209        clip_radii: clip.radius,
210    }
211}
212
213/// The GPU objects an app's windows share: one instance, adapter, device and queue.
214///
215/// Two devices cannot see each other's buffers or textures, so every
216/// window of an app draws through the same `Gpu`. A single-window app
217/// never names it: [`Renderer::new`] opens a private one. A second window
218/// takes the first renderer's [`Renderer::gpu`] and opens through
219/// [`Renderer::new_in`]. Cloning a `Gpu` clones a handle to the same
220/// device.
221#[derive(Clone)]
222pub struct Gpu(std::sync::Arc<GpuInner>);
223
224struct GpuInner {
225    instance: wgpu::Instance,
226    adapter: wgpu::Adapter,
227    device: wgpu::Device,
228    queue: wgpu::Queue,
229    /// What it was opened for ([`Gpu::new_with`]).
230    options: GpuOptions,
231    dual_source: bool,
232    /// One pipeline per registered fragment per surface format, built the
233    /// first time a frame draws it (about 0.2 ms, paid once) and shared by
234    /// every window on this device, dropped when a frame's list says the
235    /// handle is gone (`dropped_fragments`). A `Mutex` because `Gpu` is a
236    /// shared handle and building is rare; nothing here is touched on a
237    /// frame that draws no new fragment.
238    fragment_pipelines: std::sync::Mutex<
239        std::collections::HashMap<(u64, wgpu::TextureFormat), wgpu::RenderPipeline>,
240    >,
241    /// One texture per texture-backed image, uploaded the first time a
242    /// frame on this device draws it and again when its revision moves,
243    /// shared by every window like the pipelines above, dropped when the
244    /// core says the handle is gone.
245    textures: std::sync::Mutex<std::collections::HashMap<u64, std::sync::Arc<ImageTexture>>>,
246    /// Set by the device's lost callback: a driver update, a GPU reset, a
247    /// hang the OS answered by removing the device. Nothing on it works
248    /// again; a shell opens a new one ([`Gpu::lost`]).
249    lost: std::sync::Arc<std::sync::atomic::AtomicBool>,
250}
251
252/// A texture-backed image on the device: the texture, and what was
253/// uploaded into it. A new `Arc` is made when the size changes, which is
254/// what tells a renderer its bind group is stale.
255struct ImageTexture {
256    texture: wgpu::Texture,
257    view: wgpu::TextureView,
258    width: u32,
259    height: u32,
260    /// The revision the pixels in the texture came from, behind a lock
261    /// because the texture is shared and the upload is per device.
262    rev: std::sync::Mutex<u32>,
263}
264
265/// How a [`Gpu`] is opened: what its surfaces must be able to do, decided
266/// before the first one exists because some of it is the instance's.
267#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
268#[non_exhaustive]
269pub struct GpuOptions {
270    /// Its surfaces may be presented with alpha
271    /// ([`Renderer::set_transparent`]), so a window shows what is behind
272    /// it where a frame paints nothing (backlog F126). On Windows this
273    /// presents D3D12 through DirectComposition
274    /// (`Dx12SwapchainKind::DxgiFromVisual`), the one D3D12 swapchain that
275    /// takes alpha, instead of a swapchain on the window's handle; nothing
276    /// changes elsewhere. Off by default, so an opaque app keeps the
277    /// swapchain it always had.
278    pub transparent: bool,
279}
280
281impl GpuOptions {
282    /// Options whose surfaces may be transparent ([`Self::transparent`]).
283    pub fn transparent(transparent: bool) -> Self {
284        Self { transparent }
285    }
286}
287
288impl Gpu {
289    /// Opens a device that can present to `target`, and returns the
290    /// surface it was chosen for.
291    ///
292    /// The first window's surface has to exist before an adapter can be
293    /// picked, so it comes back with the device; later windows get theirs
294    /// from [`Gpu::create_surface`]. Most runners call [`Renderer::new`]
295    /// instead, which does both and builds the renderer. On Windows only
296    /// the D3D12 backend is enabled unless `WGPU_BACKEND` names another.
297    pub async fn new(
298        target: impl Into<wgpu::SurfaceTarget<'static>>,
299    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
300        Self::new_with(target, GpuOptions::default()).await
301    }
302
303    /// [`Gpu::new`] with `options`: what every surface opened on this
304    /// device must be able to do.
305    pub async fn new_with(
306        target: impl Into<wgpu::SurfaceTarget<'static>>,
307        options: GpuOptions,
308    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
309        report_faults();
310        // Every backend the build has, as wgpu defaults — but on Windows
311        // D3D12 alone unless `WGPU_BACKEND` names another. An instance
312        // keeps every backend it enumerated alive for as long as it lives,
313        // so with all of them the process holds an OpenGL context and a
314        // Vulkan instance it never draws with, both in the driver's
315        // `nvoglv64.dll`; and wgpu, left to choose, took Vulkan over D3D12
316        // here. Under a driver update that DLL faulted in present rather
317        // than answer `DEVICE_LOST`, which ended the process; D3D12's
318        // `nvwgf2umx.dll` reports the removal, and the shell reopens the
319        // device (`Gpu::lost`).
320        let mut desc = wgpu::InstanceDescriptor::new_without_display_handle_from_env();
321        if cfg!(windows) && std::env::var_os("WGPU_BACKEND").is_none() {
322            desc.backends = wgpu::Backends::DX12;
323        }
324        // A swapchain made on the window's handle is opaque whatever it is
325        // configured with; one presented through a composition visual
326        // takes premultiplied alpha. Only when asked, and under
327        // `WGPU_DX12_PRESENTATION_SYSTEM`, which still wins for a look at
328        // either.
329        if options.transparent && std::env::var_os("WGPU_DX12_PRESENTATION_SYSTEM").is_none() {
330            desc.backend_options.dx12.presentation_system = wgpu::Dx12SwapchainKind::DxgiFromVisual;
331        }
332        let instance = wgpu::Instance::new(desc);
333        let surface = instance.create_surface(target)?;
334        let adapter = instance
335            .request_adapter(&wgpu::RequestAdapterOptions {
336                compatible_surface: Some(&surface),
337                ..Default::default()
338            })
339            .await?;
340        // Per-channel blending for LCD subpixel text, when the device has it.
341        let dual_source = adapter
342            .features()
343            .contains(wgpu::Features::DUAL_SOURCE_BLENDING);
344        let (device, queue) = adapter
345            .request_device(&wgpu::DeviceDescriptor {
346                required_features: if dual_source {
347                    wgpu::Features::DUAL_SOURCE_BLENDING
348                } else {
349                    wgpu::Features::empty()
350                },
351                ..Default::default()
352            })
353            .await?;
354        // What goes wrong on the device is said, not swallowed: an error
355        // outside a scope, and the loss of the device itself — remembered
356        // too, so a frame can tell a dead device from a stale swapchain.
357        device.on_uncaptured_error(std::sync::Arc::new(|e| eprintln!("kui: wgpu: {e}")));
358        let lost = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
359        device.set_device_lost_callback({
360            let lost = lost.clone();
361            move |reason, message| {
362                if reason == wgpu::DeviceLostReason::Unknown {
363                    eprintln!("kui: device lost: {message}");
364                    lost.store(true, std::sync::atomic::Ordering::Release);
365                }
366            }
367        });
368        let gpu = Self(std::sync::Arc::new(GpuInner {
369            instance,
370            adapter,
371            device,
372            queue,
373            options,
374            dual_source,
375            fragment_pipelines: Default::default(),
376            textures: Default::default(),
377            lost,
378        }));
379        Ok((gpu, surface))
380    }
381
382    /// Whether the device is gone (a driver update or a GPU reset took it).
383    ///
384    /// Nothing on a lost device works again: open a new `Gpu` and a new
385    /// renderer on it for every window. [`Renderer::render`] reports the
386    /// same condition as [`RenderError::DeviceLost`].
387    pub fn lost(&self) -> bool {
388        self.0.lost.load(std::sync::atomic::Ordering::Acquire)
389    }
390
391    /// Loses the device on purpose, as a driver update or a GPU reset
392    /// would, so a runner can test its reopening path without one.
393    ///
394    /// On D3D12 the device is really removed (`ID3D12Device5::RemoveDevice`)
395    /// and the loss lands on its next use through the lost callback, as a
396    /// real one does; elsewhere the device is only marked lost.
397    pub fn mark_lost(&self) {
398        #[cfg(windows)]
399        {
400            use windows::Win32::Graphics::Direct3D12::ID3D12Device5;
401            use windows::core::Interface;
402            // SAFETY: the hal device is only read for its raw handle, and
403            // `RemoveDevice` is what D3D12 offers for exactly this.
404            let removed = unsafe {
405                self.0
406                    .device
407                    .as_hal::<wgpu::hal::api::Dx12>()
408                    .and_then(|d| d.raw_device().cast::<ID3D12Device5>().ok())
409                    .map(|d| d.RemoveDevice())
410            };
411            if removed.is_some() {
412                // The loss lands on the device's next use, through the
413                // lost callback, as a real one does.
414                return;
415            }
416        }
417        self.0
418            .lost
419            .store(true, std::sync::atomic::Ordering::Release);
420    }
421
422    /// A surface for another window on the same instance, which is what
423    /// [`Renderer::new_in`] draws into.
424    pub fn create_surface(
425        &self,
426        target: impl Into<wgpu::SurfaceTarget<'static>>,
427    ) -> Result<wgpu::Surface<'static>, wgpu::CreateSurfaceError> {
428        self.0.instance.create_surface(target)
429    }
430
431    /// What the device was opened for ([`Gpu::new_with`]).
432    pub fn options(&self) -> GpuOptions {
433        self.0.options
434    }
435
436    /// The wgpu instance the device was opened on.
437    pub fn instance(&self) -> &wgpu::Instance {
438        &self.0.instance
439    }
440
441    /// The adapter the device was requested from.
442    pub fn adapter(&self) -> &wgpu::Adapter {
443        &self.0.adapter
444    }
445
446    /// The device, for a runner that creates resources of its own on it.
447    pub fn device(&self) -> &wgpu::Device {
448        &self.0.device
449    }
450
451    /// The queue the renderer submits to.
452    pub fn queue(&self) -> &wgpu::Queue {
453        &self.0.queue
454    }
455
456    /// Whether this device blends per channel (dual-source blending), so
457    /// LCD subpixel glyphs draw with per-channel coverage rather than
458    /// their union.
459    pub fn dual_source(&self) -> bool {
460        self.0.dual_source
461    }
462
463    /// The texture for one texture-backed image, uploaded on first sight
464    /// and whenever `rev` has moved past what the texture holds; a size
465    /// change makes a new texture. `None` for a degenerate size, which
466    /// draws nothing.
467    fn image_texture(
468        &self,
469        id: u64,
470        px: &kui_core::display::TexturePixels,
471    ) -> Option<std::sync::Arc<ImageTexture>> {
472        // Degenerate, or past what this device can hold in one texture
473        // (8192 on many adapters, 16384 on Metal): draws nothing, which is
474        // what the core says a texture-backed image that cannot be backed
475        // does, rather than a validation error the device turns into a
476        // panic.
477        let max = self.0.device.limits().max_texture_dimension_2d;
478        if px.width == 0 || px.height == 0 || px.width > max || px.height > max {
479            return None;
480        }
481        let mut cache = self.0.textures.lock().unwrap_or_else(|e| e.into_inner());
482        let fresh = match cache.get(&id) {
483            Some(t) if t.width == px.width && t.height == px.height => {
484                let mut rev = t.rev.lock().unwrap_or_else(|e| e.into_inner());
485                if *rev != px.rev {
486                    upload_image(&self.0.queue, &t.texture, px);
487                    *rev = px.rev;
488                }
489                return Some(t.clone());
490            }
491            _ => {
492                let texture = self.0.device.create_texture(&wgpu::TextureDescriptor {
493                    label: Some("kui.image"),
494                    size: wgpu::Extent3d {
495                        width: px.width,
496                        height: px.height,
497                        depth_or_array_layers: 1,
498                    },
499                    mip_level_count: 1,
500                    sample_count: 1,
501                    dimension: wgpu::TextureDimension::D2,
502                    format: wgpu::TextureFormat::Rgba8Unorm,
503                    usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
504                    view_formats: &[],
505                });
506                upload_image(&self.0.queue, &texture, px);
507                let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
508                std::sync::Arc::new(ImageTexture {
509                    texture,
510                    view,
511                    width: px.width,
512                    height: px.height,
513                    rev: std::sync::Mutex::new(px.rev),
514                })
515            }
516        };
517        cache.insert(id, fresh.clone());
518        Some(fresh)
519    }
520
521    /// Forgets a removed fragment's pipelines, one per surface format it
522    /// was ever drawn in; the GPU frees them once no frame in flight
523    /// holds one.
524    fn drop_fragment_pipelines(&self, id: u64) {
525        self.0
526            .fragment_pipelines
527            .lock()
528            .unwrap_or_else(|e| e.into_inner())
529            .retain(|(fid, _), _| *fid != id);
530    }
531
532    /// Forgets a removed image's texture; the GPU frees it once no bind
533    /// group holds it.
534    fn drop_image_texture(&self, id: u64) {
535        self.0
536            .textures
537            .lock()
538            .unwrap_or_else(|e| e.into_inner())
539            .remove(&id);
540    }
541
542    /// Whether the cache still holds exactly this texture for `id` — what
543    /// a renderer asks before keeping a bind group over it, since a
544    /// removal reaches the cache through whichever window's frame carried
545    /// it and the other windows' bind groups would otherwise hold the
546    /// texture for as long as they live.
547    fn holds_image_texture(&self, id: u64, texture: &std::sync::Arc<ImageTexture>) -> bool {
548        self.0
549            .textures
550            .lock()
551            .unwrap_or_else(|e| e.into_inner())
552            .get(&id)
553            .is_some_and(|t| std::sync::Arc::ptr_eq(t, texture))
554    }
555
556    /// The pipeline for one registered fragment, built on first sight and
557    /// then shared by every window on this device. `source` is the app's
558    /// WGSL, which the core already validated; it is wrapped in the same
559    /// prelude and epilogue here, from `kui_core::fragment::module_source`,
560    /// so what compiles is what was validated.
561    ///
562    /// The source is not validated again here: `Core::add_fragment` parsed
563    /// and validated this exact module text with the same naga this wgpu
564    /// carries, and refused a handle for anything that failed. A module
565    /// that still does not compile is a kui bug, and reaches wgpu's own
566    /// error handler like any other.
567    fn fragment_pipeline(
568        &self,
569        id: u64,
570        source: &str,
571        format: wgpu::TextureFormat,
572        layouts: &FragmentLayouts,
573    ) -> wgpu::RenderPipeline {
574        let mut cache = self
575            .0
576            .fragment_pipelines
577            .lock()
578            .unwrap_or_else(|e| e.into_inner());
579        if let Some(p) = cache.get(&(id, format)) {
580            return p.clone();
581        }
582        let device = &self.0.device;
583        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
584            label: Some("kui.fragment"),
585            source: wgpu::ShaderSource::Wgsl(kui_core::fragment::module_source(source).into()),
586        });
587        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
588            label: Some("kui.fragment"),
589            layout: Some(&layouts.pipeline),
590            vertex: wgpu::VertexState {
591                module: &layouts.vertex,
592                entry_point: Some("vs_main"),
593                compilation_options: Default::default(),
594                buffers: &[Some(instance_buffer_layout(&INSTANCE_ATTRS))],
595            },
596            fragment: Some(wgpu::FragmentState {
597                module: &module,
598                entry_point: Some(kui_core::fragment::ENTRY_POINT),
599                compilation_options: Default::default(),
600                targets: &[Some(wgpu::ColorTargetState {
601                    format,
602                    // A fragment returns premultiplied colour, always over.
603                    blend: Some(wgpu::BlendState {
604                        color: wgpu::BlendComponent {
605                            src_factor: wgpu::BlendFactor::One,
606                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
607                            operation: wgpu::BlendOperation::Add,
608                        },
609                        alpha: wgpu::BlendComponent {
610                            src_factor: wgpu::BlendFactor::One,
611                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
612                            operation: wgpu::BlendOperation::Add,
613                        },
614                    }),
615                    write_mask: wgpu::ColorWrites::ALL,
616                })],
617            }),
618            primitive: wgpu::PrimitiveState::default(),
619            depth_stencil: None,
620            multisample: wgpu::MultisampleState::default(),
621            multiview_mask: None,
622            cache: None,
623        });
624        cache.insert((id, format), pipeline.clone());
625        pipeline
626    }
627}
628
629/// What building a fragment pipeline needs besides its own source: kui's
630/// vertex stage, and the layout that puts the globals at group 0 and the
631/// parameters at group 1.
632struct FragmentLayouts {
633    vertex: wgpu::ShaderModule,
634    pipeline: wgpu::PipelineLayout,
635}
636
637/// The instance attributes both pipelines read; one array so the vertex
638/// layout cannot differ between them.
639const INSTANCE_ATTRS: [wgpu::VertexAttribute; 9] = wgpu::vertex_attr_array![
640    0 => Float32x2, 1 => Float32x2, 2 => Float32x4,
641    3 => Float32x4, 4 => Float32x4, 5 => Float32x4,
642    6 => Float32x4, 7 => Float32x4, 8 => Float32x4,
643];
644
645fn instance_buffer_layout(attrs: &[wgpu::VertexAttribute]) -> wgpu::VertexBufferLayout<'_> {
646    wgpu::VertexBufferLayout {
647        array_stride: std::mem::size_of::<Instance>() as u64,
648        step_mode: wgpu::VertexStepMode::Instance,
649        attributes: attrs,
650    }
651}
652
653impl std::fmt::Debug for Gpu {
654    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
655        f.debug_struct("Gpu")
656            .field("adapter", &self.0.adapter.get_info().name)
657            .field("dual_source", &self.0.dual_source)
658            .finish()
659    }
660}
661
662/// One window's renderer: its surface, the pipelines and a GPU copy of the core's glyph atlas.
663///
664/// Open one per window with [`Renderer::new`] (first window, on a device
665/// of its own) or [`Renderer::new_in`] (another window on a shared
666/// [`Gpu`]). Each frame, hand [`Renderer::render`] the display list and
667/// atlas from `Core::output`; call [`Renderer::resize`] when the window
668/// changes size. The crate root has the whole sequence.
669pub struct Renderer {
670    gpu: Gpu,
671    surface: wgpu::Surface<'static>,
672    config: wgpu::SurfaceConfiguration,
673    pipeline: wgpu::RenderPipeline,
674    globals_buf: wgpu::Buffer,
675    bind_group: wgpu::BindGroup,
676    bind_layout: wgpu::BindGroupLayout,
677    /// The two samplers every group-0 bind group carries: linear at
678    /// binding 2, nearest at 3.
679    samplers: Samplers,
680    /// Per texture-backed image this window has drawn: the device's
681    /// texture, and a bind group of this window's own — group 0 with that
682    /// texture in the atlas's place and a globals copy whose `atlas_size`
683    /// is the texture's, rewritten each frame the image is drawn.
684    texture_binds: std::collections::HashMap<u64, TextureBind>,
685    atlas_tex: wgpu::Texture,
686    atlas_size: u32,
687    atlas_epoch: u64,
688    instance_buf: wgpu::Buffer,
689    instance_cap: usize,
690    instances: Vec<Instance>,
691    /// What a fragment pipeline is built against: kui's vertex stage and
692    /// the two-group layout. Built once per renderer, handed to the
693    /// device's shared cache.
694    fragment_layouts: FragmentLayouts,
695    /// One slot per fragment this frame, each padded to the device's
696    /// dynamic-offset alignment.
697    fragment_params_buf: wgpu::Buffer,
698    fragment_params_cap: usize,
699    fragment_bind: wgpu::BindGroup,
700    fragment_bind_layout: wgpu::BindGroupLayout,
701    /// The alignment slots are padded to; `dynamic_offset` steps by it.
702    uniform_align: u32,
703    /// Scratch for one frame's padded parameter slots.
704    fragment_bytes: Vec<u8>,
705    /// The color a frame is cleared to before anything is drawn.
706    ///
707    /// A runner usually sets it to the theme's background each frame. It
708    /// is written straight through: the renderer asks for a non-sRGB
709    /// surface, so each component is the byte it lands as, the same way a
710    /// quad's color is.
711    pub clear_color: wgpu::Color,
712    /// A picture drawn over the clear and under everything the frame
713    /// draws ([`Renderer::set_ground`]), when there is one.
714    ground: Option<Ground>,
715    /// The backdrop blur's pipelines (backlog F129): made the first time a
716    /// frame blurs, kept for the renderer's life.
717    backdrop_pipes: Option<backdrop::Pipes>,
718    /// Its offscreen frame and scratch textures, at the surface's size:
719    /// made when a frame blurs, dropped after `backdrop::IDLE_FRAMES`
720    /// frames that do not.
721    backdrop_targets: Option<backdrop::Targets>,
722    backdrop_idle: u32,
723    /// The frame's blurs, in paint order.
724    blurs: Vec<backdrop::Blur>,
725}
726
727/// The picture under a frame: its texture, the part of it the window
728/// shows (`uv`), and the pipeline that draws it across the viewport.
729struct Ground {
730    pipeline: wgpu::RenderPipeline,
731    bind: wgpu::BindGroup,
732    uv: wgpu::Buffer,
733    /// Kept for the bind group, which holds a view of it.
734    _texture: wgpu::Texture,
735}
736
737/// The ground's shader: one quad over the viewport, sampling the part of
738/// the picture `g.uv` names (`u0, v0, u1, v1`), opaque.
739const GROUND_SHADER: &str = r"
740struct G { uv: vec4<f32> }
741@group(0) @binding(0) var<uniform> g: G;
742@group(0) @binding(1) var t: texture_2d<f32>;
743@group(0) @binding(2) var s: sampler;
744struct V { @builtin(position) pos: vec4<f32>, @location(0) uv: vec2<f32> }
745@vertex fn vs(@builtin(vertex_index) i: u32) -> V {
746    let x = f32(i & 1u);
747    let y = f32((i >> 1u) & 1u);
748    var o: V;
749    o.pos = vec4<f32>(x * 2.0 - 1.0, 1.0 - y * 2.0, 0.0, 1.0);
750    o.uv = vec2<f32>(mix(g.uv.x, g.uv.z, x), mix(g.uv.y, g.uv.w, y));
751    return o;
752}
753@fragment fn fs(v: V) -> @location(0) vec4<f32> {
754    return vec4<f32>(textureSample(t, s, v.uv).rgb, 1.0);
755}
756";
757
758struct Samplers {
759    linear: wgpu::Sampler,
760    nearest: wgpu::Sampler,
761}
762
763struct TextureBind {
764    texture: std::sync::Arc<ImageTexture>,
765    globals: wgpu::Buffer,
766    bind: wgpu::BindGroup,
767}
768
769fn upload_image(
770    queue: &wgpu::Queue,
771    texture: &wgpu::Texture,
772    px: &kui_core::display::TexturePixels,
773) {
774    queue.write_texture(
775        wgpu::TexelCopyTextureInfo {
776            texture,
777            mip_level: 0,
778            origin: wgpu::Origin3d::ZERO,
779            aspect: wgpu::TextureAspect::All,
780        },
781        &px.rgba,
782        wgpu::TexelCopyBufferLayout {
783            offset: 0,
784            bytes_per_row: Some(px.width * 4),
785            rows_per_image: Some(px.height),
786        },
787        wgpu::Extent3d {
788            width: px.width,
789            height: px.height,
790            depth_or_array_layers: 1,
791        },
792    );
793}
794
795/// The alpha mode an opaque window presents with: `Opaque` wherever the
796/// surface offers it — every surface a window gets does — and the
797/// surface's first mode otherwise. The first mode was the choice before
798/// a surface could be transparent, and is `Opaque` everywhere but a
799/// D3D12 composition swapchain, which lists `Auto` first.
800fn opaque_mode(modes: &[wgpu::CompositeAlphaMode]) -> wgpu::CompositeAlphaMode {
801    if modes.contains(&wgpu::CompositeAlphaMode::Opaque) {
802        wgpu::CompositeAlphaMode::Opaque
803    } else {
804        modes
805            .first()
806            .copied()
807            .unwrap_or(wgpu::CompositeAlphaMode::Opaque)
808    }
809}
810
811/// The alpha mode a transparent window presents with, of the ones the
812/// surface offers, or `None` when it offers none that composites the
813/// frame kui draws — which is premultiplied. `PreMultiplied` first; on
814/// Metal, `PostMultiplied`, which is only how wgpu names a layer that is
815/// not opaque, and Core Animation composites a layer's pixels as
816/// premultiplied whatever it is called; then `Inherit`, the window
817/// system's own way, which under Wayland and an ARGB X11 visual is
818/// premultiplied too. Never `PostMultiplied` on Vulkan or D3D12, where it
819/// means straight alpha and every translucent pixel would darken.
820fn transparent_mode(
821    modes: &[wgpu::CompositeAlphaMode],
822    backend: wgpu::Backend,
823) -> Option<wgpu::CompositeAlphaMode> {
824    use wgpu::CompositeAlphaMode as M;
825    let mut wanted = vec![M::PreMultiplied];
826    if backend == wgpu::Backend::Metal {
827        wanted.push(M::PostMultiplied);
828    }
829    wanted.push(M::Inherit);
830    wanted.into_iter().find(|m| modes.contains(m))
831}
832
833/// Picks the `//DUAL:` or `//SINGLE:` lines of the shader template.
834fn preprocess_shader(src: &str, dual: bool) -> String {
835    let (keep, drop) = if dual {
836        ("//DUAL:", "//SINGLE:")
837    } else {
838        ("//SINGLE:", "//DUAL:")
839    };
840    let mut out = String::with_capacity(src.len());
841    for line in src.lines() {
842        if let Some(rest) = line.strip_prefix(keep) {
843            out.push_str(rest);
844        } else if line.starts_with(drop) {
845            continue;
846        } else {
847            out.push_str(line);
848        }
849        out.push('\n');
850    }
851    out
852}
853
854/// How many frames may be queued ahead of the one on screen by default: two, or one on Windows.
855///
856/// With two, a drawable to render into is waiting while the previous
857/// frame's is still out, so a frame whose thread woke a little late still
858/// makes its vsync (on Metal this is triple buffering). The price is that
859/// a frame built the moment a drawable frees reaches the screen a vsync
860/// later than it could; `kui-native` wins that back by starting frames at
861/// the display's vsync, and a runner driven any other way pays it. On
862/// Windows the flip-model swapchain delivers every vsync with a single
863/// queued frame, so the second would be latency for nothing. Change it
864/// per renderer with [`Renderer::set_frame_latency`].
865pub const DEFAULT_FRAME_LATENCY: u32 = if cfg!(target_os = "windows") { 1 } else { 2 };
866
867impl Renderer {
868    /// A renderer for one window, on a device of its own.
869    ///
870    /// `width` and `height` are the window's size in physical pixels.
871    /// Fails when no adapter can present to `target` or the surface cannot
872    /// be configured. Use [`Renderer::new_in`] for every window after the
873    /// first, so they share the device.
874    pub async fn new(
875        target: impl Into<wgpu::SurfaceTarget<'static>>,
876        width: u32,
877        height: u32,
878    ) -> Result<Self, Box<dyn std::error::Error>> {
879        Self::new_with(target, width, height, GpuOptions::default()).await
880    }
881
882    /// [`Renderer::new`] on a device opened with `options` — what every
883    /// window on it must be able to do ([`GpuOptions`]) — and, under
884    /// [`GpuOptions::transparent`], this window's surface presented with
885    /// alpha where it can be ([`Renderer::transparent`]).
886    pub async fn new_with(
887        target: impl Into<wgpu::SurfaceTarget<'static>>,
888        width: u32,
889        height: u32,
890        options: GpuOptions,
891    ) -> Result<Self, Box<dyn std::error::Error>> {
892        let (gpu, surface) = Gpu::new_with(target, options).await?;
893        Self::with_surface(gpu, surface, width, height, options.transparent)
894    }
895
896    /// Whether the surface is presented with alpha, so what is behind the
897    /// window shows where a frame paints nothing and through a colour with
898    /// alpha ([`Renderer::new_with`], [`Renderer::new_in_with`]). The frame
899    /// is premultiplied, which is what every quad and glyph already blends
900    /// to; clear it to a transparent [`Renderer::clear_color`] for the
901    /// window to show through.
902    pub fn transparent(&self) -> bool {
903        !matches!(
904            self.config.alpha_mode,
905            wgpu::CompositeAlphaMode::Opaque | wgpu::CompositeAlphaMode::Auto
906        )
907    }
908
909    /// A renderer for another window on an existing device, the one every
910    /// window of the app shares. Get `gpu` from the first renderer's
911    /// [`Renderer::gpu`].
912    pub fn new_in(
913        gpu: &Gpu,
914        target: impl Into<wgpu::SurfaceTarget<'static>>,
915        width: u32,
916        height: u32,
917    ) -> Result<Self, Box<dyn std::error::Error>> {
918        Self::new_in_with(gpu, target, width, height, false)
919    }
920
921    /// [`Renderer::new_in`] for a window whose surface is presented with
922    /// alpha (`transparent`, backlog F126). Decided here, once, and not by
923    /// a reconfigure: a D3D12 swapchain keeps the alpha mode it was made
924    /// with, and resizing it later changes nothing — measured on Windows
925    /// 11, where a surface reconfigured to premultiplied after it was made
926    /// opaque went on compositing over black. Where the surface cannot
927    /// take alpha — a D3D12 device not opened with
928    /// [`GpuOptions::transparent`], a Vulkan or GL surface that offers
929    /// only opaque — it is opaque, and [`Renderer::transparent`] says so.
930    pub fn new_in_with(
931        gpu: &Gpu,
932        target: impl Into<wgpu::SurfaceTarget<'static>>,
933        width: u32,
934        height: u32,
935        transparent: bool,
936    ) -> Result<Self, Box<dyn std::error::Error>> {
937        let surface = gpu.create_surface(target)?;
938        Self::with_surface(gpu.clone(), surface, width, height, transparent)
939    }
940
941    /// The device this renderer draws with, to open another window on.
942    pub fn gpu(&self) -> &Gpu {
943        &self.gpu
944    }
945
946    /// Sets how many frames may be queued ahead of the one on screen (at
947    /// least one; see [`DEFAULT_FRAME_LATENCY`]). Reconfigures the surface
948    /// when the value changes.
949    pub fn set_frame_latency(&mut self, frames: u32) {
950        let frames = frames.max(1);
951        if self.config.desired_maximum_frame_latency != frames {
952            self.config.desired_maximum_frame_latency = frames;
953            self.surface.configure(self.gpu.device(), &self.config);
954        }
955    }
956
957    /// The frame latency the surface is configured with.
958    pub fn frame_latency(&self) -> u32 {
959        self.config.desired_maximum_frame_latency
960    }
961
962    fn with_surface(
963        gpu: Gpu,
964        surface: wgpu::Surface<'static>,
965        width: u32,
966        height: u32,
967        transparent: bool,
968    ) -> Result<Self, Box<dyn std::error::Error>> {
969        let device = gpu.device();
970        let dual_source = gpu.dual_source();
971
972        let caps = surface.get_capabilities(gpu.adapter());
973        let format = caps
974            .formats
975            .iter()
976            .copied()
977            .find(|f| !f.is_srgb())
978            .unwrap_or(caps.formats[0]);
979        let backend = gpu.adapter().get_info().backend;
980        let alpha_mode = transparent
981            .then(|| transparent_mode(&caps.alpha_modes, backend))
982            .flatten()
983            .unwrap_or_else(|| opaque_mode(&caps.alpha_modes));
984        let config = wgpu::SurfaceConfiguration {
985            usage: wgpu::TextureUsages::RENDER_ATTACHMENT,
986            format,
987            width: width.clamp(1, device.limits().max_texture_dimension_2d),
988            height: height.clamp(1, device.limits().max_texture_dimension_2d),
989            present_mode: wgpu::PresentMode::AutoVsync,
990            alpha_mode,
991            color_space: wgpu::SurfaceColorSpace::Auto,
992            view_formats: vec![],
993            // See `DEFAULT_FRAME_LATENCY`; a runner that wants another
994            // says so through `set_frame_latency`.
995            desired_maximum_frame_latency: DEFAULT_FRAME_LATENCY,
996        };
997        // A configure that fails only reports to the device's error
998        // handler, and the first acquire on the unconfigured surface is a
999        // panic inside wgpu; caught here, it is this constructor's error
1000        // — a window DXGI will not give a second swapchain, say.
1001        let scope = device.push_error_scope(wgpu::ErrorFilter::Validation);
1002        surface.configure(device, &config);
1003        if let Some(err) = pollster::block_on(scope.pop()) {
1004            return Err(format!("configuring the surface: {err}").into());
1005        }
1006
1007        let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1008            label: Some("kui"),
1009            source: wgpu::ShaderSource::Wgsl(
1010                preprocess_shader(include_str!("shader.wgsl"), dual_source).into(),
1011            ),
1012        });
1013        // Dual source: the shader outputs premultiplied color and a
1014        // per-channel coverage; out = src + dst * (1 - coverage). For
1015        // ordinary quads every channel's coverage equals alpha, which is
1016        // exactly premultiplied alpha blending.
1017        let blend = if dual_source {
1018            wgpu::BlendState {
1019                color: wgpu::BlendComponent {
1020                    src_factor: wgpu::BlendFactor::One,
1021                    dst_factor: wgpu::BlendFactor::OneMinusSrc1,
1022                    operation: wgpu::BlendOperation::Add,
1023                },
1024                alpha: wgpu::BlendComponent {
1025                    src_factor: wgpu::BlendFactor::One,
1026                    dst_factor: wgpu::BlendFactor::OneMinusSrc1Alpha,
1027                    operation: wgpu::BlendOperation::Add,
1028                },
1029            }
1030        } else {
1031            wgpu::BlendState::ALPHA_BLENDING
1032        };
1033
1034        let bind_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1035            label: Some("kui.globals"),
1036            entries: &[
1037                wgpu::BindGroupLayoutEntry {
1038                    binding: 0,
1039                    visibility: wgpu::ShaderStages::VERTEX_FRAGMENT,
1040                    ty: wgpu::BindingType::Buffer {
1041                        ty: wgpu::BufferBindingType::Uniform,
1042                        has_dynamic_offset: false,
1043                        min_binding_size: None,
1044                    },
1045                    count: None,
1046                },
1047                wgpu::BindGroupLayoutEntry {
1048                    binding: 1,
1049                    visibility: wgpu::ShaderStages::FRAGMENT,
1050                    ty: wgpu::BindingType::Texture {
1051                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1052                        view_dimension: wgpu::TextureViewDimension::D2,
1053                        multisampled: false,
1054                    },
1055                    count: None,
1056                },
1057                wgpu::BindGroupLayoutEntry {
1058                    binding: 2,
1059                    visibility: wgpu::ShaderStages::FRAGMENT,
1060                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1061                    count: None,
1062                },
1063                wgpu::BindGroupLayoutEntry {
1064                    binding: 3,
1065                    visibility: wgpu::ShaderStages::FRAGMENT,
1066                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1067                    count: None,
1068                },
1069            ],
1070        });
1071
1072        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1073            label: Some("kui"),
1074            bind_group_layouts: &[Some(&bind_layout)],
1075            immediate_size: 0,
1076        });
1077
1078        let instance_attrs = INSTANCE_ATTRS;
1079        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1080            label: Some("kui.quads"),
1081            layout: Some(&pipeline_layout),
1082            vertex: wgpu::VertexState {
1083                module: &shader,
1084                entry_point: Some("vs_main"),
1085                compilation_options: Default::default(),
1086                buffers: &[Some(instance_buffer_layout(&instance_attrs))],
1087            },
1088            fragment: Some(wgpu::FragmentState {
1089                module: &shader,
1090                entry_point: Some("fs_main"),
1091                compilation_options: Default::default(),
1092                targets: &[Some(wgpu::ColorTargetState {
1093                    format,
1094                    blend: Some(blend),
1095                    write_mask: wgpu::ColorWrites::ALL,
1096                })],
1097            }),
1098            primitive: wgpu::PrimitiveState::default(),
1099            depth_stencil: None,
1100            multisample: wgpu::MultisampleState::default(),
1101            multiview_mask: None,
1102            cache: None,
1103        });
1104
1105        let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1106            label: Some("kui.globals"),
1107            size: std::mem::size_of::<Globals>() as u64,
1108            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1109            mapped_at_creation: false,
1110        });
1111
1112        let atlas_size = kui_core::atlas::ATLAS_SIZE;
1113        let atlas_tex = create_atlas_texture(device, atlas_size);
1114        let samplers = Samplers {
1115            linear: device.create_sampler(&wgpu::SamplerDescriptor {
1116                label: Some("kui.linear"),
1117                mag_filter: wgpu::FilterMode::Linear,
1118                min_filter: wgpu::FilterMode::Linear,
1119                ..Default::default()
1120            }),
1121            nearest: device.create_sampler(&wgpu::SamplerDescriptor {
1122                label: Some("kui.nearest"),
1123                mag_filter: wgpu::FilterMode::Nearest,
1124                min_filter: wgpu::FilterMode::Nearest,
1125                ..Default::default()
1126            }),
1127        };
1128        let atlas_view = atlas_tex.create_view(&wgpu::TextureViewDescriptor::default());
1129        let bind_group =
1130            create_bind_group(device, &bind_layout, &globals_buf, &atlas_view, &samplers);
1131
1132        let instance_cap = 4096;
1133        let instance_buf = create_instance_buffer(device, instance_cap);
1134
1135        // Fragments: one uniform slot per draw, picked by dynamic offset,
1136        // and the layout their pipelines are built against. All of it is
1137        // built whether or not a frame ever draws one — a bind group
1138        // layout and an empty buffer, not a pipeline, which is the part
1139        // that costs and is built on first sight.
1140        let fragment_bind_layout =
1141            device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1142                label: Some("kui.fragment.params"),
1143                entries: &[wgpu::BindGroupLayoutEntry {
1144                    binding: 0,
1145                    visibility: wgpu::ShaderStages::FRAGMENT,
1146                    ty: wgpu::BindingType::Buffer {
1147                        ty: wgpu::BufferBindingType::Uniform,
1148                        has_dynamic_offset: true,
1149                        min_binding_size: std::num::NonZeroU64::new(std::mem::size_of::<
1150                            FragmentParams,
1151                        >()
1152                            as u64),
1153                    },
1154                    count: None,
1155                }],
1156            });
1157        let fragment_layouts = FragmentLayouts {
1158            vertex: shader,
1159            pipeline: device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1160                label: Some("kui.fragment"),
1161                bind_group_layouts: &[Some(&bind_layout), Some(&fragment_bind_layout)],
1162                immediate_size: 0,
1163            }),
1164        };
1165        // The *device's* limit, not the adapter's: the device is opened with
1166        // `Limits::default()`, whose `min_uniform_buffer_offset_alignment` is
1167        // 256, and validation holds a dynamic offset to what the device asked
1168        // for rather than to what the hardware could have done. An adapter
1169        // reporting the smaller 64 — which DX12 does — then gave 64-byte slots
1170        // and a validation error on the frame's second fragment.
1171        let uniform_align = device.limits().min_uniform_buffer_offset_alignment;
1172        let fragment_params_cap = 16;
1173        let fragment_params_buf =
1174            create_fragment_params_buffer(device, fragment_params_cap, uniform_align);
1175        let fragment_bind =
1176            create_fragment_bind_group(device, &fragment_bind_layout, &fragment_params_buf);
1177
1178        Ok(Self {
1179            gpu,
1180            surface,
1181            config,
1182            pipeline,
1183            globals_buf,
1184            bind_group,
1185            bind_layout,
1186            samplers,
1187            texture_binds: Default::default(),
1188            atlas_tex,
1189            atlas_size,
1190            atlas_epoch: u64::MAX,
1191            instance_buf,
1192            instance_cap,
1193            instances: Vec::new(),
1194            fragment_layouts,
1195            fragment_params_buf,
1196            fragment_params_cap,
1197            fragment_bind,
1198            fragment_bind_layout,
1199            uniform_align,
1200            fragment_bytes: Vec::new(),
1201            clear_color: wgpu::Color {
1202                r: 0.06,
1203                g: 0.065,
1204                b: 0.08,
1205                a: 1.0,
1206            },
1207            ground: None,
1208            backdrop_pipes: None,
1209            backdrop_targets: None,
1210            backdrop_idle: 0,
1211            blurs: Vec::new(),
1212        })
1213    }
1214
1215    /// Draws `rgba` (`width` by `height` pixels, four bytes each, row by
1216    /// row from the top left, the bytes the surface takes as they are) over
1217    /// the clear and under everything a frame draws, opaque, stretched
1218    /// across the viewport and sampled with linear filtering — so a small
1219    /// picture reads as a soft one. What a runner draws as a window's
1220    /// ground where the OS has no material to put behind it: the
1221    /// wallpaper, scaled down and blurred once (backlog F126). Which part
1222    /// of the picture shows is [`Renderer::set_ground_uv`]; all of it
1223    /// until that is called. A degenerate size, or pixels that are not
1224    /// that size, clear it.
1225    pub fn set_ground(&mut self, rgba: &[u8], width: u32, height: u32) {
1226        let max = self.gpu.device().limits().max_texture_dimension_2d;
1227        if width == 0
1228            || height == 0
1229            || width > max
1230            || height > max
1231            || rgba.len() != width as usize * height as usize * 4
1232        {
1233            self.ground = None;
1234            return;
1235        }
1236        let device = self.gpu.device();
1237        let size = wgpu::Extent3d {
1238            width,
1239            height,
1240            depth_or_array_layers: 1,
1241        };
1242        let texture = device.create_texture(&wgpu::TextureDescriptor {
1243            label: Some("kui.ground"),
1244            size,
1245            mip_level_count: 1,
1246            sample_count: 1,
1247            dimension: wgpu::TextureDimension::D2,
1248            format: wgpu::TextureFormat::Rgba8Unorm,
1249            usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1250            view_formats: &[],
1251        });
1252        self.gpu.queue().write_texture(
1253            wgpu::TexelCopyTextureInfo {
1254                texture: &texture,
1255                mip_level: 0,
1256                origin: wgpu::Origin3d::ZERO,
1257                aspect: wgpu::TextureAspect::All,
1258            },
1259            rgba,
1260            wgpu::TexelCopyBufferLayout {
1261                offset: 0,
1262                bytes_per_row: Some(width * 4),
1263                rows_per_image: Some(height),
1264            },
1265            size,
1266        );
1267        let uv = device.create_buffer(&wgpu::BufferDescriptor {
1268            label: Some("kui.ground.uv"),
1269            size: 16,
1270            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1271            mapped_at_creation: false,
1272        });
1273        self.gpu
1274            .queue()
1275            .write_buffer(&uv, 0, bytemuck::cast_slice(&[0.0f32, 0.0, 1.0, 1.0]));
1276        let layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1277            label: Some("kui.ground"),
1278            entries: &[
1279                wgpu::BindGroupLayoutEntry {
1280                    binding: 0,
1281                    visibility: wgpu::ShaderStages::VERTEX,
1282                    ty: wgpu::BindingType::Buffer {
1283                        ty: wgpu::BufferBindingType::Uniform,
1284                        has_dynamic_offset: false,
1285                        min_binding_size: None,
1286                    },
1287                    count: None,
1288                },
1289                wgpu::BindGroupLayoutEntry {
1290                    binding: 1,
1291                    visibility: wgpu::ShaderStages::FRAGMENT,
1292                    ty: wgpu::BindingType::Texture {
1293                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1294                        view_dimension: wgpu::TextureViewDimension::D2,
1295                        multisampled: false,
1296                    },
1297                    count: None,
1298                },
1299                wgpu::BindGroupLayoutEntry {
1300                    binding: 2,
1301                    visibility: wgpu::ShaderStages::FRAGMENT,
1302                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1303                    count: None,
1304                },
1305            ],
1306        });
1307        let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
1308        let bind = device.create_bind_group(&wgpu::BindGroupDescriptor {
1309            label: Some("kui.ground"),
1310            layout: &layout,
1311            entries: &[
1312                wgpu::BindGroupEntry {
1313                    binding: 0,
1314                    resource: uv.as_entire_binding(),
1315                },
1316                wgpu::BindGroupEntry {
1317                    binding: 1,
1318                    resource: wgpu::BindingResource::TextureView(&view),
1319                },
1320                wgpu::BindGroupEntry {
1321                    binding: 2,
1322                    resource: wgpu::BindingResource::Sampler(&self.samplers.linear),
1323                },
1324            ],
1325        });
1326        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1327            label: Some("kui.ground"),
1328            source: wgpu::ShaderSource::Wgsl(GROUND_SHADER.into()),
1329        });
1330        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1331            label: Some("kui.ground"),
1332            bind_group_layouts: &[Some(&layout)],
1333            immediate_size: 0,
1334        });
1335        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1336            label: Some("kui.ground"),
1337            layout: Some(&pipeline_layout),
1338            vertex: wgpu::VertexState {
1339                module: &module,
1340                entry_point: Some("vs"),
1341                compilation_options: Default::default(),
1342                buffers: &[],
1343            },
1344            fragment: Some(wgpu::FragmentState {
1345                module: &module,
1346                entry_point: Some("fs"),
1347                compilation_options: Default::default(),
1348                targets: &[Some(wgpu::ColorTargetState {
1349                    format: self.config.format,
1350                    blend: None,
1351                    write_mask: wgpu::ColorWrites::ALL,
1352                })],
1353            }),
1354            primitive: wgpu::PrimitiveState {
1355                topology: wgpu::PrimitiveTopology::TriangleStrip,
1356                ..Default::default()
1357            },
1358            depth_stencil: None,
1359            multisample: wgpu::MultisampleState::default(),
1360            multiview_mask: None,
1361            cache: None,
1362        });
1363        self.ground = Some(Ground {
1364            pipeline,
1365            bind,
1366            uv,
1367            _texture: texture,
1368        });
1369    }
1370
1371    /// Which part of the ground shows across the viewport: `[u0, v0, u1,
1372    /// v1]` in the picture's own 0..1 coordinates, the top-left corner
1373    /// first. Outside 0..1 the picture's edge is stretched. Nothing without
1374    /// a ground.
1375    pub fn set_ground_uv(&mut self, uv: [f32; 4]) {
1376        if let Some(g) = &self.ground {
1377            self.gpu
1378                .queue()
1379                .write_buffer(&g.uv, 0, bytemuck::cast_slice(&uv));
1380        }
1381    }
1382
1383    /// Takes the ground away ([`Renderer::set_ground`]).
1384    pub fn clear_ground(&mut self) {
1385        self.ground = None;
1386    }
1387
1388    /// Whether a ground is drawn under the frame.
1389    pub fn has_ground(&self) -> bool {
1390        self.ground.is_some()
1391    }
1392
1393    /// Whether this device can draw LCD subpixel glyphs
1394    /// (`QuadKind::GlyphSubpixel`) as intended.
1395    ///
1396    /// Pass it to `Core::set_subpixel_text` once after opening the
1397    /// renderer; when it is `false` the core keeps rasterizing grayscale
1398    /// masks, which every device blends correctly.
1399    pub fn subpixel_text(&self) -> bool {
1400        self.gpu.dual_source()
1401    }
1402
1403    /// Reconfigures the swapchain for a new window size, in physical pixels.
1404    ///
1405    /// The size is clamped to at least 1 (a minimized window reports zero,
1406    /// and a zero-sized surface is a validation error) and to the device's
1407    /// largest texture (Windows hands out nonsense sizes mid-resize, and
1408    /// configuring past the limit panics inside wgpu). A clamped frame is
1409    /// one wrong picture; the next real size fixes it.
1410    pub fn resize(&mut self, width: u32, height: u32) {
1411        let max = self.gpu.device().limits().max_texture_dimension_2d;
1412        self.config.width = width.clamp(1, max);
1413        self.config.height = height.clamp(1, max);
1414        self.surface.configure(self.gpu.device(), &self.config);
1415    }
1416
1417    fn sync_atlas(&mut self, atlas: &mut GlyphAtlas) {
1418        if atlas.size != self.atlas_size {
1419            self.atlas_size = atlas.size;
1420            self.atlas_tex = create_atlas_texture(self.gpu.device(), atlas.size);
1421            let view = self
1422                .atlas_tex
1423                .create_view(&wgpu::TextureViewDescriptor::default());
1424            self.bind_group = create_bind_group(
1425                self.gpu.device(),
1426                &self.bind_layout,
1427                &self.globals_buf,
1428                &view,
1429                &self.samplers,
1430            );
1431            self.atlas_epoch = u64::MAX;
1432        }
1433        if atlas.dirty || self.atlas_epoch != atlas.epoch {
1434            self.gpu.queue().write_texture(
1435                wgpu::TexelCopyTextureInfo {
1436                    texture: &self.atlas_tex,
1437                    mip_level: 0,
1438                    origin: wgpu::Origin3d::ZERO,
1439                    aspect: wgpu::TextureAspect::All,
1440                },
1441                &atlas.pixels,
1442                wgpu::TexelCopyBufferLayout {
1443                    offset: 0,
1444                    bytes_per_row: Some(atlas.size * 4),
1445                    rows_per_image: Some(atlas.size),
1446                },
1447                wgpu::Extent3d {
1448                    width: atlas.size,
1449                    height: atlas.size,
1450                    depth_or_array_layers: 1,
1451                },
1452            );
1453            atlas.dirty = false;
1454            self.atlas_epoch = atlas.epoch;
1455        }
1456    }
1457
1458    /// Plans the frame's backdrop blurs (backlog F129), and makes or drops
1459    /// what drawing them takes: the pipelines once, the offscreen frame and
1460    /// scratch textures at the surface's size, a parameter slot per blur.
1461    /// `any` is whether the list has a backdrop quad at all, noticed while
1462    /// the instances were written, so a frame without one scans nothing.
1463    fn plan_backdrops(&mut self, dl: &DisplayList, any: bool) {
1464        self.blurs.clear();
1465        let (w, h) = (self.config.width, self.config.height);
1466        if any {
1467            for (i, q) in dl.quads.iter().enumerate() {
1468                if q.kind == QuadKind::Backdrop
1469                    && let Some(b) = backdrop::plan(i as u32, q, dl.clip_of(q), w, h)
1470                {
1471                    self.blurs.push(b);
1472                }
1473            }
1474        }
1475        if self.blurs.is_empty() {
1476            self.backdrop_idle = self.backdrop_idle.saturating_add(1);
1477            if self.backdrop_idle > backdrop::IDLE_FRAMES {
1478                self.backdrop_targets = None;
1479            }
1480            return;
1481        }
1482        self.backdrop_idle = 0;
1483        let device = self.gpu.device();
1484        let align = self.uniform_align;
1485        let format = self.config.format;
1486        let pipes = self
1487            .backdrop_pipes
1488            .get_or_insert_with(|| backdrop::Pipes::new(device, format, align));
1489        let mut rebind = false;
1490        if self.blurs.len() > pipes.params_cap {
1491            pipes.params_cap = self.blurs.len().next_power_of_two();
1492            pipes.params = backdrop::params_buffer(device, pipes.params_cap, align);
1493            rebind = true;
1494        }
1495        if rebind
1496            || self
1497                .backdrop_targets
1498                .as_ref()
1499                .is_none_or(|t| t.size != (w, h))
1500        {
1501            self.backdrop_targets = Some(backdrop::Targets::new(device, pipes, format, w, h));
1502        }
1503        let slot = align as usize;
1504        let mut bytes = vec![0u8; self.blurs.len() * slot];
1505        for (i, b) in self.blurs.iter().enumerate() {
1506            let mut p = b.params;
1507            p.sizes[2] = w as f32;
1508            p.sizes[3] = h as f32;
1509            bytes[i * slot..i * slot + std::mem::size_of::<backdrop::Params>()]
1510                .copy_from_slice(bytemuck::bytes_of(&p));
1511        }
1512        self.gpu.queue().write_buffer(&pipes.params, 0, &bytes);
1513    }
1514
1515    /// Draws the instances in `range` into `pass`: in one instanced draw,
1516    /// or — where a fragment or a texture-backed image interrupts the run
1517    /// — the run before it with the über-pipeline, then that one quad with
1518    /// its own pipeline (a fragment) or its own group 0 (a texture), then
1519    /// on. Consecutive quads of the same handle still take one set each
1520    /// (about 0.6 us); runs of ordinary quads are unbroken.
1521    fn draw_quads(
1522        &self,
1523        pass: &mut wgpu::RenderPass<'_>,
1524        dl: &DisplayList,
1525        range: std::ops::Range<u32>,
1526        fragment_pipelines: &[wgpu::RenderPipeline],
1527        texture_binds: &[Option<u64>],
1528    ) {
1529        if range.is_empty() {
1530            return;
1531        }
1532        pass.set_vertex_buffer(0, self.instance_buf.slice(..));
1533        if fragment_pipelines.is_empty() && texture_binds.is_empty() {
1534            // The whole run in one instanced draw, as a frame has always
1535            // been. Nothing below runs.
1536            pass.set_pipeline(&self.pipeline);
1537            pass.set_bind_group(0, &self.bind_group, &[]);
1538            pass.draw(0..6, range);
1539            return;
1540        }
1541        let mut run_start = range.start;
1542        let mut on_quads = false;
1543        for i in range.clone() {
1544            let q = &dl.quads[i as usize];
1545            if q.kind != QuadKind::Fragment && q.kind != QuadKind::Texture {
1546                continue;
1547            }
1548            if i > run_start {
1549                if !on_quads {
1550                    pass.set_pipeline(&self.pipeline);
1551                    pass.set_bind_group(0, &self.bind_group, &[]);
1552                    on_quads = true;
1553                }
1554                pass.draw(0..6, run_start..i);
1555            }
1556            // `uv[0]` is the index into the side list, which is also this
1557            // fragment's parameter slot, or this texture's bind.
1558            let slot = q.uv[0] as usize;
1559            if q.kind == QuadKind::Texture {
1560                if let Some(Some(id)) = texture_binds.get(slot)
1561                    && let Some(b) = self.texture_binds.get(id)
1562                {
1563                    pass.set_pipeline(&self.pipeline);
1564                    pass.set_bind_group(0, &b.bind, &[]);
1565                    on_quads = false;
1566                    pass.draw(0..6, i..i + 1);
1567                }
1568            } else if let Some(pipeline) = fragment_pipelines.get(slot) {
1569                // A fragment reading a texture-backed image takes that
1570                // image's group 0 — the texture in the atlas's place,
1571                // `atlas_size` its size — exactly as a texture quad does;
1572                // one reading the atlas, or nothing, takes the frame's. A
1573                // texture the device could not make (a degenerate or
1574                // oversized image) draws the fragment against the atlas
1575                // with a zero rect, which `kui_sample` reads as no image.
1576                let group0 = match draw_image_texture(&dl.fragments[slot]) {
1577                    Some(index) => texture_binds
1578                        .get(index)
1579                        .copied()
1580                        .flatten()
1581                        .and_then(|id| self.texture_binds.get(&id))
1582                        .map_or(&self.bind_group, |b| &b.bind),
1583                    None => &self.bind_group,
1584                };
1585                pass.set_pipeline(pipeline);
1586                pass.set_bind_group(0, group0, &[]);
1587                pass.set_bind_group(1, &self.fragment_bind, &[slot as u32 * self.uniform_align]);
1588                on_quads = false;
1589                pass.draw(0..6, i..i + 1);
1590            }
1591            run_start = i + 1;
1592        }
1593        if range.end > run_start {
1594            if !on_quads {
1595                pass.set_pipeline(&self.pipeline);
1596                pass.set_bind_group(0, &self.bind_group, &[]);
1597            }
1598            pass.draw(0..6, run_start..range.end);
1599        }
1600    }
1601
1602    /// Draws one frame and presents it.
1603    ///
1604    /// Uploads the atlas when it changed since the last frame (and clears
1605    /// its `dirty` flag), writes the list's quads to the instance buffer,
1606    /// draws them over [`Renderer::clear_color`] in one render pass, and
1607    /// presents. Acquiring the swapchain image blocks while vsync holds
1608    /// the frame back; that time comes back as
1609    /// [`RenderReport::vsync_wait_ms`] so a runner can tell pacing from
1610    /// work.
1611    ///
1612    /// Fails without presenting when the surface or the device cannot
1613    /// take the frame; each [`RenderError`] says what to do next.
1614    pub fn render(
1615        &mut self,
1616        dl: &DisplayList,
1617        atlas: &mut GlyphAtlas,
1618    ) -> Result<RenderReport, RenderError> {
1619        // A dead device takes no work: everything below would only add
1620        // errors to the one that lost it.
1621        if self.gpu.lost() {
1622            return Err(RenderError::DeviceLost);
1623        }
1624        self.sync_atlas(atlas);
1625
1626        self.instances.clear();
1627        let mut any_backdrop = false;
1628        self.instances.extend(dl.quads.iter().map(|q| {
1629            any_backdrop |= q.kind == QuadKind::Backdrop;
1630            instance_of(q, &dl.clips, &dl.textures)
1631        }));
1632        self.plan_backdrops(dl, any_backdrop);
1633        if self.instances.len() > self.instance_cap {
1634            self.instance_cap = self.instances.len().next_power_of_two();
1635            self.instance_buf = create_instance_buffer(self.gpu.device(), self.instance_cap);
1636        }
1637        if !self.instances.is_empty() {
1638            self.gpu.queue().write_buffer(
1639                &self.instance_buf,
1640                0,
1641                bytemuck::cast_slice(&self.instances),
1642            );
1643        }
1644        let globals = Globals {
1645            viewport: [dl.viewport.w.max(1.0), dl.viewport.h.max(1.0)],
1646            atlas_size: [self.atlas_size as f32, self.atlas_size as f32],
1647            time: dl.time,
1648            scale: dl.scale,
1649            _pad: [0.0; 2],
1650        };
1651        self.gpu
1652            .queue()
1653            .write_buffer(&self.globals_buf, 0, bytemuck::bytes_of(&globals));
1654
1655        // Texture-backed images: drop what the core
1656        // removed, upload what moved, and give each one drawn this frame
1657        // a group-0 bind group of its own with a globals copy whose
1658        // `atlas_size` is the texture's. All skipped on a frame that
1659        // draws none.
1660        for id in &dl.dropped_textures {
1661            self.texture_binds.remove(&id.to_ffi());
1662            self.gpu.drop_image_texture(id.to_ffi());
1663        }
1664        // And the pipelines of removed fragments — built per handle
1665        // and shared by every window, so one window's list carries the
1666        // removal and this is the only eviction they get.
1667        for id in &dl.dropped_fragments {
1668            self.gpu.drop_fragment_pipelines(id.to_ffi());
1669        }
1670        // A drop another window's frame carried: the cache no longer
1671        // holds the texture this bind group does. One lock per frame,
1672        // and only for a window that has ever drawn a texture.
1673        if !self.texture_binds.is_empty() {
1674            let gpu = &self.gpu;
1675            self.texture_binds
1676                .retain(|id, b| gpu.holds_image_texture(*id, &b.texture));
1677        }
1678        let mut texture_binds: Vec<Option<u64>> = Vec::new();
1679        if !dl.textures.is_empty() {
1680            texture_binds.reserve(dl.textures.len());
1681            for (draw, px) in dl.textures.iter().zip(&dl.texture_pixels) {
1682                let id = draw.id.to_ffi();
1683                let Some(texture) = self.gpu.image_texture(id, px) else {
1684                    texture_binds.push(None);
1685                    continue;
1686                };
1687                let stale = self
1688                    .texture_binds
1689                    .get(&id)
1690                    .is_none_or(|b| !std::sync::Arc::ptr_eq(&b.texture, &texture));
1691                if stale {
1692                    let device = self.gpu.device();
1693                    let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1694                        label: Some("kui.image.globals"),
1695                        size: std::mem::size_of::<Globals>() as u64,
1696                        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1697                        mapped_at_creation: false,
1698                    });
1699                    let bind = create_bind_group(
1700                        device,
1701                        &self.bind_layout,
1702                        &globals_buf,
1703                        &texture.view,
1704                        &self.samplers,
1705                    );
1706                    self.texture_binds.insert(
1707                        id,
1708                        TextureBind {
1709                            texture: texture.clone(),
1710                            globals: globals_buf,
1711                            bind,
1712                        },
1713                    );
1714                }
1715                let b = &self.texture_binds[&id];
1716                let mine = Globals {
1717                    atlas_size: [texture.width as f32, texture.height as f32],
1718                    ..globals
1719                };
1720                self.gpu
1721                    .queue()
1722                    .write_buffer(&b.globals, 0, bytemuck::bytes_of(&mine));
1723                texture_binds.push(Some(id));
1724            }
1725        }
1726
1727        // Each fragment's parameters into its own slot, and its pipeline
1728        // built if this device has not seen the handle before. Both are
1729        // skipped whole on a frame that draws no fragment.
1730        let mut fragment_pipelines: Vec<wgpu::RenderPipeline> = Vec::new();
1731        if !dl.fragments.is_empty() {
1732            let align = self.uniform_align as usize;
1733            if dl.fragments.len() > self.fragment_params_cap {
1734                self.fragment_params_cap = dl.fragments.len().next_power_of_two();
1735                self.fragment_params_buf = create_fragment_params_buffer(
1736                    self.gpu.device(),
1737                    self.fragment_params_cap,
1738                    self.uniform_align,
1739                );
1740                self.fragment_bind = create_fragment_bind_group(
1741                    self.gpu.device(),
1742                    &self.fragment_bind_layout,
1743                    &self.fragment_params_buf,
1744                );
1745            }
1746            self.fragment_bytes.clear();
1747            self.fragment_bytes.resize(dl.fragments.len() * align, 0);
1748            for (i, draw) in dl.fragments.iter().enumerate() {
1749                let uv = draw.image.uv();
1750                let slot = FragmentParams {
1751                    params: draw.params,
1752                    image: [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32],
1753                };
1754                let at = i * align;
1755                self.fragment_bytes[at..at + std::mem::size_of::<FragmentParams>()]
1756                    .copy_from_slice(bytemuck::bytes_of(&slot));
1757            }
1758            self.gpu
1759                .queue()
1760                .write_buffer(&self.fragment_params_buf, 0, &self.fragment_bytes);
1761            fragment_pipelines.reserve(dl.fragments.len());
1762            for (draw, source) in dl.fragments.iter().zip(&dl.fragment_sources) {
1763                fragment_pipelines.push(self.gpu.fragment_pipeline(
1764                    draw.id.to_ffi(),
1765                    source,
1766                    self.config.format,
1767                    &self.fragment_layouts,
1768                ));
1769            }
1770        }
1771
1772        // Acquiring the swapchain image is where vsync backpressure blocks;
1773        // report it separately so latency graphs show pacing vs work.
1774        let t_wait = std::time::Instant::now();
1775        let frame = match self.surface.get_current_texture() {
1776            wgpu::CurrentSurfaceTexture::Success(f)
1777            | wgpu::CurrentSurfaceTexture::Suboptimal(f) => f,
1778            wgpu::CurrentSurfaceTexture::Timeout | wgpu::CurrentSurfaceTexture::Occluded => {
1779                return Err(RenderError::Skip);
1780            }
1781            wgpu::CurrentSurfaceTexture::Outdated | wgpu::CurrentSurfaceTexture::Lost => {
1782                return Err(RenderError::Reconfigure);
1783            }
1784            // The acquire's error went to the device's error handler; if
1785            // it was the device itself, the lost callback has run by now.
1786            wgpu::CurrentSurfaceTexture::Validation => {
1787                return Err(if self.gpu.lost() {
1788                    RenderError::DeviceLost
1789                } else {
1790                    RenderError::Validation
1791                });
1792            }
1793        };
1794        let vsync_wait_ms = t_wait.elapsed().as_secs_f32() * 1e3;
1795        let surface_view = frame
1796            .texture
1797            .create_view(&wgpu::TextureViewDescriptor::default());
1798        let mut encoder = self
1799            .gpu
1800            .device()
1801            .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("kui") });
1802        // A frame that blurs draws into the offscreen copy, so what it has
1803        // drawn can be read back, and breaks its pass at each blur
1804        // (backlog F129); one that does not draws to the surface in one
1805        // pass, as it always has.
1806        let blurring = !self.blurs.is_empty();
1807        let offscreen = match (&self.backdrop_targets, blurring) {
1808            (Some(t), true) => Some(t),
1809            _ => None,
1810        };
1811        let target = offscreen.map_or(&surface_view, |t| &t.frame);
1812        let end = self.instances.len() as u32;
1813        let mut start = 0u32;
1814        let mut next = 0usize;
1815        loop {
1816            let stop = self.blurs.get(next).map_or(end, |b| b.quad);
1817            {
1818                let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1819                    label: Some("kui"),
1820                    color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1821                        view: target,
1822                        depth_slice: None,
1823                        resolve_target: None,
1824                        ops: wgpu::Operations {
1825                            load: if next == 0 {
1826                                wgpu::LoadOp::Clear(self.clear_color)
1827                            } else {
1828                                wgpu::LoadOp::Load
1829                            },
1830                            store: wgpu::StoreOp::Store,
1831                        },
1832                    })],
1833                    depth_stencil_attachment: None,
1834                    timestamp_writes: None,
1835                    occlusion_query_set: None,
1836                    multiview_mask: None,
1837                });
1838                // The ground first, opaque over the clear, so everything
1839                // the frame paints with alpha blends over it.
1840                if next == 0
1841                    && let Some(g) = &self.ground
1842                {
1843                    pass.set_pipeline(&g.pipeline);
1844                    pass.set_bind_group(0, &g.bind, &[]);
1845                    pass.draw(0..4, 0..1);
1846                }
1847                self.draw_quads(
1848                    &mut pass,
1849                    dl,
1850                    start..stop,
1851                    &fragment_pipelines,
1852                    &texture_binds,
1853                );
1854            }
1855            let (Some(b), Some(pipes), Some(t)) =
1856                (self.blurs.get(next), &self.backdrop_pipes, offscreen)
1857            else {
1858                break;
1859            };
1860            backdrop::record(&mut encoder, pipes, t, b, next as u32 * self.uniform_align);
1861            start = b.quad + 1;
1862            next += 1;
1863        }
1864        if let (Some(pipes), Some(t)) = (&self.backdrop_pipes, offscreen) {
1865            let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1866                label: Some("kui.backdrop.blit"),
1867                color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1868                    view: &surface_view,
1869                    depth_slice: None,
1870                    resolve_target: None,
1871                    ops: wgpu::Operations {
1872                        load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT),
1873                        store: wgpu::StoreOp::Store,
1874                    },
1875                })],
1876                depth_stencil_attachment: None,
1877                timestamp_writes: None,
1878                occlusion_query_set: None,
1879                multiview_mask: None,
1880            });
1881            pass.set_pipeline(&pipes.blit);
1882            pass.set_bind_group(0, &t.blit, &[0]);
1883            pass.draw(0..3, 0..1);
1884        }
1885        self.gpu.queue().submit([encoder.finish()]);
1886        self.gpu.queue().present(frame);
1887        Ok(RenderReport { vsync_wait_ms })
1888    }
1889}
1890
1891/// The `textures` entry a fragment draw reads its image from, if its
1892/// image has a texture of its own.
1893fn draw_image_texture(draw: &kui_core::FragmentDraw) -> Option<usize> {
1894    match draw.image {
1895        kui_core::FragmentImage::Texture { index, .. } => Some(index as usize),
1896        _ => None,
1897    }
1898}
1899
1900/// Timing details from one [`Renderer::render`] call.
1901#[derive(Clone, Copy, Debug, Default)]
1902pub struct RenderReport {
1903    /// Milliseconds spent blocked acquiring the swapchain image, which is
1904    /// where vsync backpressure shows up.
1905    pub vsync_wait_ms: f32,
1906}
1907
1908/// A frame that produced no image, and what to do about it.
1909///
1910/// Mapped from wgpu's `CurrentSurfaceTexture`; the crate root's example
1911/// handles every variant.
1912#[derive(Clone, Copy, Debug)]
1913pub enum RenderError {
1914    /// The surface is outdated or lost: call [`Renderer::resize`] with the
1915    /// window's size and draw again.
1916    Reconfigure,
1917    /// Nothing can be presented right now (the window is occluded, or the
1918    /// acquire timed out): try again next frame.
1919    Skip,
1920    /// The surface is configured wrong for the window: call
1921    /// [`Renderer::resize`] with the window's size and draw again. A
1922    /// surface that stays wrong is best given up with its device.
1923    Validation,
1924    /// The device is gone ([`Gpu::lost`]): open a new one, and a renderer
1925    /// on it for every window.
1926    DeviceLost,
1927}
1928
1929impl std::fmt::Display for RenderError {
1930    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1931        match self {
1932            Self::Reconfigure => write!(f, "surface outdated or lost; reconfigure"),
1933            Self::Skip => write!(f, "no frame available; skip"),
1934            Self::Validation => write!(f, "surface texture validation error"),
1935            Self::DeviceLost => write!(f, "device lost; reopen"),
1936        }
1937    }
1938}
1939
1940impl std::error::Error for RenderError {}
1941
1942fn create_atlas_texture(device: &wgpu::Device, size: u32) -> wgpu::Texture {
1943    device.create_texture(&wgpu::TextureDescriptor {
1944        label: Some("kui.atlas"),
1945        size: wgpu::Extent3d {
1946            width: size,
1947            height: size,
1948            depth_or_array_layers: 1,
1949        },
1950        mip_level_count: 1,
1951        sample_count: 1,
1952        dimension: wgpu::TextureDimension::D2,
1953        format: wgpu::TextureFormat::Rgba8Unorm,
1954        usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1955        view_formats: &[],
1956    })
1957}
1958
1959/// Group 0: the globals, a texture — the atlas, or a texture-backed image
1960/// in its place — and the two samplers.
1961fn create_bind_group(
1962    device: &wgpu::Device,
1963    layout: &wgpu::BindGroupLayout,
1964    globals: &wgpu::Buffer,
1965    view: &wgpu::TextureView,
1966    samplers: &Samplers,
1967) -> wgpu::BindGroup {
1968    device.create_bind_group(&wgpu::BindGroupDescriptor {
1969        label: Some("kui"),
1970        layout,
1971        entries: &[
1972            wgpu::BindGroupEntry {
1973                binding: 0,
1974                resource: globals.as_entire_binding(),
1975            },
1976            wgpu::BindGroupEntry {
1977                binding: 1,
1978                resource: wgpu::BindingResource::TextureView(view),
1979            },
1980            wgpu::BindGroupEntry {
1981                binding: 2,
1982                resource: wgpu::BindingResource::Sampler(&samplers.linear),
1983            },
1984            wgpu::BindGroupEntry {
1985                binding: 3,
1986                resource: wgpu::BindingResource::Sampler(&samplers.nearest),
1987            },
1988        ],
1989    })
1990}
1991
1992fn create_instance_buffer(device: &wgpu::Device, cap: usize) -> wgpu::Buffer {
1993    device.create_buffer(&wgpu::BufferDescriptor {
1994        label: Some("kui.instances"),
1995        size: (cap * std::mem::size_of::<Instance>()) as u64,
1996        usage: wgpu::BufferUsages::VERTEX | wgpu::BufferUsages::COPY_DST,
1997        mapped_at_creation: false,
1998    })
1999}
2000
2001fn create_fragment_params_buffer(device: &wgpu::Device, cap: usize, align: u32) -> wgpu::Buffer {
2002    device.create_buffer(&wgpu::BufferDescriptor {
2003        label: Some("kui.fragment.params"),
2004        size: (cap.max(1) * align as usize) as u64,
2005        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
2006        mapped_at_creation: false,
2007    })
2008}
2009
2010fn create_fragment_bind_group(
2011    device: &wgpu::Device,
2012    layout: &wgpu::BindGroupLayout,
2013    buf: &wgpu::Buffer,
2014) -> wgpu::BindGroup {
2015    device.create_bind_group(&wgpu::BindGroupDescriptor {
2016        label: Some("kui.fragment.params"),
2017        layout,
2018        entries: &[wgpu::BindGroupEntry {
2019            binding: 0,
2020            resource: wgpu::BindingResource::Buffer(wgpu::BufferBinding {
2021                buffer: buf,
2022                offset: 0,
2023                size: std::num::NonZeroU64::new(std::mem::size_of::<FragmentParams>() as u64),
2024            }),
2025        }],
2026    })
2027}
2028
2029/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2030///
2031/// A fault in a GPU driver, such as one being replaced under the app, ends
2032/// the process with no line from anyone: it is not a panic, and Windows
2033/// reports only `0xC000041D` for an exception in a window callback. This
2034/// installs an unhandled-exception filter that names the exception code
2035/// and the module the faulting address is in, then lets the crash go on;
2036/// it is a diagnostic, not a recovery. A crash reporter the host installed
2037/// first is still called, with its answer returned; one installed after
2038/// replaces this filter.
2039///
2040/// [`Gpu::new`] calls it, so a runner rarely needs to. Installing it more
2041/// than once is harmless.
2042#[cfg(windows)]
2043pub fn report_faults() {
2044    use std::cell::Cell;
2045    use windows::Win32::Foundation::{
2046        EXCEPTION_ACCESS_VIOLATION, EXCEPTION_ILLEGAL_INSTRUCTION, EXCEPTION_IN_PAGE_ERROR,
2047        EXCEPTION_STACK_OVERFLOW, HMODULE, NTSTATUS, STATUS_FATAL_USER_CALLBACK_EXCEPTION,
2048    };
2049    use windows::Win32::Storage::FileSystem::WriteFile;
2050    use windows::Win32::System::Console::{GetStdHandle, STD_ERROR_HANDLE};
2051    use windows::Win32::System::Diagnostics::Debug::{
2052        AddVectoredExceptionHandler, EXCEPTION_POINTERS, EXCEPTION_RECORD,
2053        LPTOP_LEVEL_EXCEPTION_FILTER, SetUnhandledExceptionFilter,
2054    };
2055    use windows::Win32::System::LibraryLoader::{
2056        GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS, GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2057        GetModuleFileNameW, GetModuleHandleExW,
2058    };
2059
2060    const CONTINUE_SEARCH: i32 = 0;
2061    /// The faults a driver ends a process with: the ones remembered for
2062    /// a callback's `0xC000041D` to be read by.
2063    const FAULTS: [NTSTATUS; 4] = [
2064        EXCEPTION_ACCESS_VIOLATION,
2065        EXCEPTION_ILLEGAL_INSTRUCTION,
2066        EXCEPTION_IN_PAGE_ERROR,
2067        EXCEPTION_STACK_OVERFLOW,
2068    ];
2069    /// The filter this one replaced, called after it; set once, with the
2070    /// two handlers, by the one call that installs them.
2071    static PREVIOUS: std::sync::OnceLock<LPTOP_LEVEL_EXCEPTION_FILTER> = std::sync::OnceLock::new();
2072    thread_local! {
2073        /// The last fault this thread saw, code and address, handled or
2074        /// not. A `const` cell with no destructor: a plain thread-local
2075        /// slot, read and written without allocating or registering
2076        /// anything, from inside an exception.
2077        static LAST: Cell<Option<(i32, usize)>> = const { Cell::new(None) };
2078    }
2079
2080    /// Remembers a fault; says nothing and handles nothing.
2081    unsafe extern "system" fn remember(info: *mut EXCEPTION_POINTERS) -> i32 {
2082        // SAFETY: the system hands a valid record for the exception.
2083        if let Some(record) =
2084            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2085            && FAULTS.contains(&record.ExceptionCode)
2086        {
2087            let seen = (record.ExceptionCode.0, record.ExceptionAddress as usize);
2088            let _ = LAST.try_with(|l| l.set(Some(seen)));
2089        }
2090        CONTINUE_SEARCH
2091    }
2092
2093    /// The first fault on a record's chain of nested exceptions, past the
2094    /// record itself; a few links, since a chain is one or two long and a
2095    /// broken one is not worth following further.
2096    fn nested(record: &EXCEPTION_RECORD) -> Option<(i32, usize)> {
2097        let mut at = record.ExceptionRecord;
2098        for _ in 0..4 {
2099            // SAFETY: a nested record the system chained to this one.
2100            let inner = unsafe { at.as_ref() }?;
2101            if FAULTS.contains(&inner.ExceptionCode) {
2102                return Some((inner.ExceptionCode.0, inner.ExceptionAddress as usize));
2103            }
2104            at = inner.ExceptionRecord;
2105        }
2106        None
2107    }
2108
2109    /// Writes one crash's line to stderr, straight to the handle: no
2110    /// `eprintln!`, which takes a lock and, on a console, converts
2111    /// through a stack buffer eight kilobytes deep — more than a stack
2112    /// overflow leaves.
2113    fn say(code: i32, at: usize, escaped: Option<i32>) {
2114        let mut module = HMODULE::default();
2115        let mut name = [0u16; 260];
2116        // SAFETY: `at` is only looked up, never read; the buffers are ours.
2117        let found = unsafe {
2118            GetModuleHandleExW(
2119                GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS
2120                    | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2121                windows::core::PCWSTR(at as *const u16),
2122                &mut module,
2123            )
2124        }
2125        .is_ok();
2126        let path = found.then(|| {
2127            // SAFETY: the module was just found; the buffer is ours.
2128            let n = unsafe { GetModuleFileNameW(Some(module), &mut name) } as usize;
2129            &name[..n.min(name.len())]
2130        });
2131        let line = FaultLine::new(code as u32, at, path, escaped.map(|c| c as u32));
2132        // SAFETY: a handle the process was given, written from our buffer.
2133        if let Ok(err) = unsafe { GetStdHandle(STD_ERROR_HANDLE) } {
2134            let mut written = 0u32;
2135            let _ = unsafe { WriteFile(err, Some(line.bytes()), Some(&mut written), None) };
2136        }
2137    }
2138
2139    /// The crash: said, then handed to the filter before this one.
2140    unsafe extern "system" fn filter(info: *const EXCEPTION_POINTERS) -> i32 {
2141        // SAFETY: the system hands a valid record for the exception.
2142        if let Some(record) =
2143            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2144        {
2145            let (code, at) = (record.ExceptionCode, record.ExceptionAddress as usize);
2146            let inner = if code == STATUS_FATAL_USER_CALLBACK_EXCEPTION {
2147                nested(record).or_else(|| LAST.try_with(Cell::get).ok().flatten())
2148            } else {
2149                None
2150            };
2151            match inner {
2152                Some((fault, fault_at)) => say(fault, fault_at, Some(code.0)),
2153                None => say(code.0, at, None),
2154            }
2155        }
2156        match PREVIOUS.get().copied().flatten() {
2157            // SAFETY: the filter the system held before ours, called as
2158            // the system would have called it.
2159            Some(previous) => unsafe { previous(info) },
2160            None => CONTINUE_SEARCH,
2161        }
2162    }
2163
2164    PREVIOUS.get_or_init(|| {
2165        // SAFETY: both handlers read only what the system gives them and
2166        // write only their own thread-local slot and stderr.
2167        unsafe {
2168            AddVectoredExceptionHandler(0, Some(remember));
2169            SetUnhandledExceptionFilter(Some(filter))
2170        }
2171    });
2172}
2173
2174/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2175///
2176/// On Windows a fault in a GPU driver ends the process with no line from
2177/// anyone, and this installs the exception filter that names it. On every
2178/// other platform it does nothing. [`Gpu::new`] calls it, so a runner
2179/// rarely needs to.
2180#[cfg(not(windows))]
2181pub fn report_faults() {}
2182
2183/// One crash's line for `report_faults`, written into a buffer on the
2184/// stack: it is said with whatever stack the crash left (a stack
2185/// overflow leaves the few pages the thread reserved for its handlers)
2186/// and in a process whose heap may be what faulted, so nothing here
2187/// allocates. A line too long for it is cut, at a character, and still
2188/// ends in a newline. Built on every platform so it is tested on every
2189/// platform; only Windows says one.
2190#[cfg_attr(not(windows), allow(dead_code))]
2191struct FaultLine {
2192    buf: [u8; 640],
2193    len: usize,
2194}
2195
2196#[cfg_attr(not(windows), allow(dead_code))]
2197impl FaultLine {
2198    /// `kui: fault <code> at <address> in <module>`, and for a fault that
2199    /// escaped a window callback, the code it escaped as. `module` is the
2200    /// UTF-16 path Windows gives; none is a fault outside any module.
2201    fn new(code: u32, at: usize, module: Option<&[u16]>, escaped: Option<u32>) -> Self {
2202        use std::fmt::Write;
2203        let mut line = Self {
2204            buf: [0; 640],
2205            len: 0,
2206        };
2207        let _ = write!(line, "kui: fault {code:#010x} at {at:#x} in ");
2208        match module {
2209            Some(path) => {
2210                for c in char::decode_utf16(path.iter().copied()) {
2211                    let _ = line.write_char(c.unwrap_or(char::REPLACEMENT_CHARACTER));
2212                }
2213            }
2214            None => {
2215                let _ = line.write_str("no module (jit or freed code)");
2216            }
2217        }
2218        if let Some(escaped) = escaped {
2219            let _ = write!(line, ", escaped from a window callback as {escaped:#010x}");
2220        }
2221        // The newline has its byte kept for it (`write_str`).
2222        line.buf[line.len] = b'\n';
2223        line.len += 1;
2224        line
2225    }
2226
2227    fn bytes(&self) -> &[u8] {
2228        &self.buf[..self.len]
2229    }
2230}
2231
2232impl std::fmt::Write for FaultLine {
2233    fn write_str(&mut self, s: &str) -> std::fmt::Result {
2234        // One byte short of the buffer, for the newline.
2235        let room = self.buf.len() - 1 - self.len;
2236        let mut n = s.len().min(room);
2237        while !s.is_char_boundary(n) {
2238            n -= 1;
2239        }
2240        self.buf[self.len..self.len + n].copy_from_slice(&s.as_bytes()[..n]);
2241        self.len += n;
2242        Ok(())
2243    }
2244}
2245
2246#[cfg(test)]
2247mod tests {
2248    use super::*;
2249
2250    /// An opaque window keeps the mode it always had, `Opaque` — also on
2251    /// a composition swapchain, which lists `Auto` first and would have
2252    /// presented `DXGI_ALPHA_MODE_UNSPECIFIED` (backlog F126).
2253    #[test]
2254    fn an_opaque_window_presents_opaque_wherever_it_can() {
2255        use wgpu::CompositeAlphaMode as M;
2256        assert_eq!(opaque_mode(&[M::Opaque]), M::Opaque);
2257        assert_eq!(opaque_mode(&[M::Opaque, M::PostMultiplied]), M::Opaque);
2258        assert_eq!(
2259            opaque_mode(&[
2260                M::Auto,
2261                M::Inherit,
2262                M::Opaque,
2263                M::PostMultiplied,
2264                M::PreMultiplied
2265            ]),
2266            M::Opaque
2267        );
2268        assert_eq!(opaque_mode(&[M::Inherit]), M::Inherit);
2269    }
2270
2271    /// A transparent one takes a mode that composites premultiplied
2272    /// pixels, and none where only straight alpha or opaque is offered.
2273    #[test]
2274    fn a_transparent_window_presents_premultiplied_or_not_at_all() {
2275        use wgpu::{Backend, CompositeAlphaMode as M};
2276        // D3D12 through a composition visual.
2277        let visual = [
2278            M::Auto,
2279            M::Inherit,
2280            M::Opaque,
2281            M::PostMultiplied,
2282            M::PreMultiplied,
2283        ];
2284        assert_eq!(
2285            transparent_mode(&visual, Backend::Dx12),
2286            Some(M::PreMultiplied)
2287        );
2288        // D3D12 on the window's handle: opaque only.
2289        assert_eq!(transparent_mode(&[M::Opaque], Backend::Dx12), None);
2290        // Metal names its non-opaque layer post-multiplied.
2291        let metal = [M::Opaque, M::PostMultiplied];
2292        assert_eq!(
2293            transparent_mode(&metal, Backend::Metal),
2294            Some(M::PostMultiplied)
2295        );
2296        // Vulkan's post-multiplied is straight alpha: not that.
2297        assert_eq!(
2298            transparent_mode(&[M::Opaque, M::PostMultiplied], Backend::Vulkan),
2299            None
2300        );
2301        assert_eq!(
2302            transparent_mode(&[M::Opaque, M::Inherit], Backend::Vulkan),
2303            Some(M::Inherit)
2304        );
2305    }
2306
2307    /// The globals are one buffer read by two pipelines whose modules
2308    /// declare it separately: this crate's `shader.wgsl` for quads, and
2309    /// `kui_core::fragment::PRELUDE` for every fragment. If the two
2310    /// declarations drift, a fragment reads the wrong bytes and there is
2311    /// nothing to catch it at runtime — the buffer is the right size and
2312    /// the numbers are just wrong. So: same field names, same order, and
2313    /// the size the Rust struct actually is.
2314    #[test]
2315    fn globals_layout_matches() {
2316        let fields = ["viewport", "atlas_size", "time", "scale", "_pad"];
2317        let of = |src: &str, name: &str| {
2318            let start = src
2319                .find(name)
2320                .unwrap_or_else(|| panic!("{name} is not declared in\n{src}"));
2321            let body = &src[start..];
2322            let end = body.find('}').expect("a closing brace");
2323            body[..end].to_string()
2324        };
2325        let quads = of(include_str!("shader.wgsl"), "struct Globals {");
2326        let frags = of(kui_core::fragment::PRELUDE, "struct KuiGlobals {");
2327        let read = |body: &str| -> Vec<String> {
2328            body.lines()
2329                .filter_map(|l| l.split_once(':'))
2330                .map(|(name, ty)| format!("{}: {}", name.trim(), ty.trim().trim_end_matches(',')))
2331                .collect()
2332        };
2333        let (a, b) = (read(&quads), read(&frags));
2334        assert_eq!(a, b, "shader.wgsl and the fragment prelude disagree");
2335        assert_eq!(
2336            a.len(),
2337            fields.len(),
2338            "a field was added to the globals without this test being told"
2339        );
2340        for (row, want) in a.iter().zip(fields) {
2341            assert!(row.starts_with(want), "expected {want}, got {row}");
2342        }
2343        // vec2 + vec2 + f32 + f32 + vec2 = 32 bytes, and a uniform's size
2344        // must be a multiple of sixteen, which is what `_pad` is for.
2345        assert_eq!(std::mem::size_of::<Globals>(), 32);
2346    }
2347
2348    /// The fragment parameter slot is the other buffer two declarations
2349    /// read: `FragmentParams` here and `KuiFragmentParams` in the
2350    /// epilogue. Sixteen floats then the image's rect, 80 bytes.
2351    #[test]
2352    fn fragment_params_layout_matches() {
2353        assert_eq!(std::mem::size_of::<FragmentParams>(), 80);
2354        assert_eq!(std::mem::offset_of!(FragmentParams, image), 64);
2355        let epilogue = kui_core::fragment::EPILOGUE;
2356        assert!(
2357            epilogue
2358                .contains("struct KuiFragmentParams { p: array<vec4<f32>, 4>, image: vec4<f32> };"),
2359            "the epilogue's params struct moved without this test being told"
2360        );
2361    }
2362
2363    /// The bindings the prelude declares at group 0 are this crate's, by
2364    /// number and kind, since a fragment pipeline binds the quad
2365    /// pipeline's group 0 layout as it is.
2366    #[test]
2367    fn prelude_bindings_match_group_zero() {
2368        let prelude = kui_core::fragment::PRELUDE;
2369        for line in [
2370            "@group(0) @binding(0) var<uniform> kui_globals: KuiGlobals;",
2371            "@group(0) @binding(1) var kui_atlas: texture_2d<f32>;",
2372            "@group(0) @binding(2) var kui_sampler: sampler;",
2373            "@group(0) @binding(3) var kui_sampler_nearest: sampler;",
2374        ] {
2375            assert!(prelude.contains(line), "prelude lacks `{line}`");
2376        }
2377        let quads = include_str!("shader.wgsl");
2378        for line in [
2379            "@group(0) @binding(1) var atlas_tex: texture_2d<f32>;",
2380            "@group(0) @binding(2) var atlas_smp: sampler;",
2381            "@group(0) @binding(3) var nearest_smp: sampler;",
2382        ] {
2383            assert!(quads.contains(line), "shader.wgsl lacks `{line}`");
2384        }
2385    }
2386
2387    /// Both preprocessed variants of the shader must parse and validate
2388    /// (pipeline creation would otherwise fail at runtime, in a window).
2389    #[test]
2390    fn shader_variants_validate() {
2391        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2392        for dual in [false, true] {
2393            let src = preprocess_shader(include_str!("shader.wgsl"), dual);
2394            let module = wgpu::naga::front::wgsl::parse_str(&src)
2395                .unwrap_or_else(|e| panic!("dual={dual}: {}", e.emit_to_string(&src)));
2396            let caps = if dual {
2397                Capabilities::DUAL_SOURCE_BLENDING
2398            } else {
2399                Capabilities::empty()
2400            };
2401            Validator::new(ValidationFlags::all(), caps)
2402                .validate(&module)
2403                .unwrap_or_else(|e| panic!("dual={dual}: {e:?}"));
2404        }
2405    }
2406
2407    /// The ground's shader parses and validates, as the pipeline that
2408    /// draws a window's wallpaper would otherwise fail in a window
2409    /// (backlog F126).
2410    #[test]
2411    fn the_ground_shader_validates() {
2412        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2413        let module = wgpu::naga::front::wgsl::parse_str(GROUND_SHADER)
2414            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(GROUND_SHADER)));
2415        Validator::new(ValidationFlags::all(), Capabilities::empty())
2416            .validate(&module)
2417            .unwrap_or_else(|e| panic!("{e:?}"));
2418    }
2419
2420    /// The backdrop blur's shader (backlog F129), every entry point, and
2421    /// its `Params` the size the Rust struct writes.
2422    #[test]
2423    fn the_backdrop_shader_validates() {
2424        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2425        let src = include_str!("backdrop.wgsl");
2426        let module = wgpu::naga::front::wgsl::parse_str(src)
2427            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(src)));
2428        Validator::new(ValidationFlags::all(), Capabilities::empty())
2429            .validate(&module)
2430            .unwrap_or_else(|e| panic!("{e:?}"));
2431        let names: Vec<&str> = module
2432            .entry_points
2433            .iter()
2434            .map(|e| e.name.as_str())
2435            .collect();
2436        for want in [
2437            "vs",
2438            "fs_down",
2439            "fs_blur_h",
2440            "fs_blur_v",
2441            "fs_composite",
2442            "fs_blit",
2443        ] {
2444            assert!(names.contains(&want), "{want} in {names:?}");
2445        }
2446        let params = module
2447            .types
2448            .iter()
2449            .find(|(_, t)| t.name.as_deref() == Some("Params"))
2450            .expect("Params")
2451            .1;
2452        let wgpu::naga::TypeInner::Struct { span, .. } = params.inner else {
2453            panic!("Params is a struct");
2454        };
2455        assert_eq!(span as usize, std::mem::size_of::<backdrop::Params>());
2456    }
2457
2458    /// The line `report_faults` says, built without the heap (RG31): the
2459    /// code, the address and the module's UTF-16 path decoded, and for a
2460    /// fault that escaped a window callback the code it escaped as.
2461    #[test]
2462    fn a_fault_line_names_the_code_the_address_and_the_module() {
2463        let path: Vec<u16> = r"C:\Windows\System32\nvoglv64.dll".encode_utf16().collect();
2464        let line = FaultLine::new(0xC000_0005, 0x7ff6_1234, Some(&path), None);
2465        assert_eq!(
2466            std::str::from_utf8(line.bytes()).unwrap(),
2467            "kui: fault 0xc0000005 at 0x7ff61234 in C:\\Windows\\System32\\nvoglv64.dll\n"
2468        );
2469        let line = FaultLine::new(0xC000_0005, 0x10, None, Some(0xC000_041D));
2470        assert_eq!(
2471            std::str::from_utf8(line.bytes()).unwrap(),
2472            "kui: fault 0xc0000005 at 0x10 in no module (jit or freed code), \
2473             escaped from a window callback as 0xc000041d\n"
2474        );
2475        // A path that is not UTF-16 is said, not refused.
2476        let line = FaultLine::new(0xC000_001D, 0x20, Some(&[0x44, 0xD800, 0x45]), None);
2477        assert_eq!(
2478            std::str::from_utf8(line.bytes()).unwrap(),
2479            "kui: fault 0xc000001d at 0x20 in D\u{FFFD}E\n"
2480        );
2481    }
2482
2483    /// A line longer than its stack buffer is cut, between characters,
2484    /// and still ends in its newline.
2485    #[test]
2486    fn a_fault_line_too_long_is_cut_at_a_character() {
2487        let path: Vec<u16> = "é".repeat(1000).encode_utf16().collect();
2488        let line = FaultLine::new(0xC000_00FD, 0x30, Some(&path), Some(0xC000_041D));
2489        let text = std::str::from_utf8(line.bytes()).expect("cut at a character");
2490        assert!(text.ends_with("é\n"), "{text:?}");
2491        assert!(text.len() <= 640 && text.len() >= 638, "{}", text.len());
2492        assert!(text.starts_with("kui: fault 0xc00000fd at 0x30 in é"));
2493    }
2494}