Skip to main content

kui_wgpu/
lib.rs

1//! wgpu renderer for kui: draws a [`kui_core::DisplayList`] with one instanced pipeline, a single draw call per frame.
2//!
3//! kui splits a UI into a model that lays out and paints into a display
4//! list (`kui-core`), a renderer that puts that list on screen (this
5//! crate) and a runner that owns the window and the event loop
6//! (`kui-native`, the one most apps use). Reach for `kui-wgpu` directly
7//! when you are writing your own runner: you already have a window, or an
8//! event loop, that `kui-native` does not fit.
9//!
10//! A [`Renderer`] owns the swapchain of one window and a GPU copy of the
11//! core's glyph atlas, kept in step with the atlas it is handed each
12//! frame. Rounded rectangles, borders, shadows, glyphs and images are all
13//! instances of the same quad, so an ordinary frame is one draw call;
14//! only a texture-backed image or a custom fragment shader splits it.
15//! Several windows share one device through [`Gpu`].
16//!
17//! # Example
18//!
19//! A runner's whole life with the renderer: open it on a window, tell the
20//! core whether subpixel text will render, then build, draw and present
21//! one frame at a time. `window` is anything wgpu can make a surface
22//! from, such as a `winit` window.
23//!
24//! ```rust,no_run
25//! use kui_core::{Core, Size, TextStyle};
26//! use kui_wgpu::{RenderError, Renderer};
27//!
28//! fn run(
29//!     window: impl Into<kui_wgpu::wgpu::SurfaceTarget<'static>>,
30//! ) -> Result<(), Box<dyn std::error::Error>> {
31//!     let (width, height) = (800u32, 600u32);
32//!     let mut renderer = pollster::block_on(Renderer::new(window, width, height))?;
33//!     let mut core = Core::new();
34//!     core.set_subpixel_text(renderer.subpixel_text());
35//!
36//!     loop {
37//!         // When the windowing library reports a new size:
38//!         // renderer.resize(new_width, new_height);
39//!
40//!         // Build the frame through the core, in logical pixels.
41//!         let scale = 1.0;
42//!         let viewport = Size::new(width as f32 / scale, height as f32 / scale);
43//!         let mut ui = core.frame(viewport, scale);
44//!         ui.text("Hello from a custom runner", TextStyle::new(24.0));
45//!         ui.finish();
46//!
47//!         // Draw it. The atlas is `&mut` so the renderer can clear its dirty flag.
48//!         let (list, atlas) = core.output();
49//!         match renderer.render(list, atlas) {
50//!             Ok(report) => {
51//!                 let _blocked_on_vsync_ms = report.vsync_wait_ms;
52//!             }
53//!             Err(RenderError::Reconfigure | RenderError::Validation) => {
54//!                 renderer.resize(width, height);
55//!             }
56//!             Err(RenderError::Skip) => {}
57//!             Err(RenderError::DeviceLost) => {
58//!                 // Open a new `Renderer` (and a new device) and carry on.
59//!                 break;
60//!             }
61//!         }
62//!     }
63//!     Ok(())
64//! }
65//! ```
66//!
67//! # Where to look
68//!
69//! - [`Renderer`]: one window's swapchain, pipelines and atlas texture.
70//! - [`Renderer::render`]: a display list in, a presented frame (or a
71//!   [`RenderError`]) out.
72//! - [`Renderer::resize`]: reconfigure after the window changed size.
73//! - [`Renderer::subpixel_text`]: what to pass to `Core::set_subpixel_text`.
74//! - [`Gpu`]: the device, queue and adapter that windows share;
75//!   [`Renderer::new_in`] opens a second window on it.
76//! - [`RenderError`]: what each failed frame asks the runner to do next.
77//! - [`DEFAULT_FRAME_LATENCY`] and [`Renderer::set_frame_latency`]: how
78//!   many frames may queue ahead of the one on screen.
79//! - [`wgpu`] is re-exported, so a runner builds against the same version
80//!   this crate was.
81//!
82//! # Subpixel text
83//!
84//! Where the device offers dual-source blending (Metal, DX12, most Vulkan)
85//! the pipeline blends per channel, which is what LCD subpixel glyphs need.
86//! Elsewhere it falls back to ordinary alpha blending and the core should
87//! rasterize grayscale masks instead, which is what
88//! [`Renderer::subpixel_text`] tells it.
89//!
90//! The book: <https://kui-book.qxuken.dev>. Repository:
91//! <https://github.com/qxuken/kui>.
92
93pub use wgpu;
94
95mod backdrop;
96
97use kui_core::atlas::GlyphAtlas;
98use kui_core::{Clip, DisplayList, Quad, QuadKind};
99
100#[repr(C)]
101#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
102struct Instance {
103    pos: [f32; 2],
104    size: [f32; 2],
105    color: [f32; 4],
106    border_color: [f32; 4],
107    params: [f32; 4],
108    uv: [f32; 4],
109    clip: [f32; 4],
110    /// Corner radii, clockwise from the top-left.
111    radii: [f32; 4],
112    /// Radii of the clip itself; all zero = a plain rect clip.
113    clip_radii: [f32; 4],
114}
115
116/// The frame's own numbers, at group 0 binding 0 for both pipelines.
117/// `kui_core::fragment::PRELUDE` declares the same bytes as `KuiGlobals`
118/// so an app's fragment can read `time` and `scale`; the padding is what
119/// makes the struct a multiple of sixteen, which a uniform must be.
120#[repr(C)]
121#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
122struct Globals {
123    viewport: [f32; 2],
124    atlas_size: [f32; 2],
125    time: f32,
126    scale: f32,
127    _pad: [f32; 2],
128}
129
130/// One fragment's parameters as the shader takes them — the sixteen
131/// floats and the texel rect of its `image`, laid out as the epilogue's
132/// `KuiFragmentParams` — padded out to the device's dynamic-offset
133/// alignment so a frame's draws can share one buffer and pick their slot
134/// by offset.
135#[repr(C)]
136#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
137struct FragmentParams {
138    params: [f32; 16],
139    /// `FragmentIn::image`: `[x, y, w, h]` in the texture bound at group 0
140    /// for this draw — the atlas, or the image's own; zero with none.
141    image: [f32; 4],
142}
143
144/// The clip is resolved out of the frame's table here rather than read off
145/// the quad: it rides as an index (`kui_core::ClipId`) so the display list
146/// carries it once per distinct clip instead of once per quad.
147fn instance_of(q: &Quad, clips: &[Clip], textures: &[kui_core::display::TextureDraw]) -> Instance {
148    let clip = clips.get(q.clip as usize).copied().unwrap_or(Clip::NONE);
149    let kind = match q.kind {
150        QuadKind::Solid => 0.0,
151        QuadKind::GlyphMask => 1.0,
152        QuadKind::GlyphColor => 2.0,
153        QuadKind::Image => 3.0,
154        QuadKind::GlyphSubpixel => 4.0,
155        QuadKind::Shadow => 5.0,
156        QuadKind::Segment => 6.0,
157        QuadKind::Fragment => 7.0,
158        // Drawn by the image branch with its own texture bound in the
159        // atlas's place.
160        QuadKind::Texture => 3.0,
161        // Never drawn by this pipeline: the pass breaks at it and
162        // `backdrop` blurs what is under it. A solid with no colour, so
163        // one a frame did not plan a blur for draws nothing in a run.
164        QuadKind::Backdrop => 0.0,
165    };
166    // `uv` is atlas texels on every kind but two: a segment carries its
167    // endpoints there as f32 bits, and a texture quad an index into the
168    // side list whose entry holds the texel rect. The shader wants floats.
169    let uv = if q.kind == QuadKind::Segment {
170        q.segment_ends()
171    } else if q.kind == QuadKind::Texture {
172        let uv = textures.get(q.uv[0] as usize).map_or([0; 4], |t| t.uv);
173        [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32]
174    } else {
175        [
176            q.uv[0] as f32,
177            q.uv[1] as f32,
178            q.uv[2] as f32,
179            q.uv[3] as f32,
180        ]
181    };
182    if q.kind == QuadKind::Backdrop {
183        return Instance {
184            pos: [q.rect.x, q.rect.y],
185            size: [0.0, 0.0],
186            color: [0.0; 4],
187            border_color: [0.0; 4],
188            params: [0.0; 4],
189            uv: [0.0; 4],
190            clip: [0.0; 4],
191            radii: [0.0; 4],
192            clip_radii: [0.0; 4],
193        };
194    }
195    Instance {
196        pos: [q.rect.x, q.rect.y],
197        size: [q.rect.w, q.rect.h],
198        color: [q.color.r, q.color.g, q.color.b, q.color.a],
199        border_color: [
200            q.border_color.r,
201            q.border_color.g,
202            q.border_color.b,
203            q.border_color.a,
204        ],
205        params: [q.blur, q.border_w, kind, 0.0],
206        uv,
207        clip: [clip.rect.x, clip.rect.y, clip.rect.w, clip.rect.h],
208        radii: q.radius,
209        clip_radii: clip.radius,
210    }
211}
212
213/// The GPU objects an app's windows share: one instance, adapter, device and queue.
214///
215/// Two devices cannot see each other's buffers or textures, so every
216/// window of an app draws through the same `Gpu`. A single-window app
217/// never names it: [`Renderer::new`] opens a private one. A second window
218/// takes the first renderer's [`Renderer::gpu`] and opens through
219/// [`Renderer::new_in`]. Cloning a `Gpu` clones a handle to the same
220/// device.
221#[derive(Clone)]
222pub struct Gpu(std::sync::Arc<GpuInner>);
223
224struct GpuInner {
225    instance: wgpu::Instance,
226    adapter: wgpu::Adapter,
227    device: wgpu::Device,
228    queue: wgpu::Queue,
229    /// What it was opened for ([`Gpu::new_with`]).
230    options: GpuOptions,
231    /// Whether a surface on this device can be presented with alpha at
232    /// all: everywhere but Windows, and there only on D3D12 presenting
233    /// through a composition visual ([`see_through_by_visual`]).
234    alpha: bool,
235    dual_source: bool,
236    /// One pipeline per registered fragment per surface format, built the
237    /// first time a frame draws it (about 0.2 ms, paid once) and shared by
238    /// every window on this device, dropped when a frame's list says the
239    /// handle is gone (`dropped_fragments`). A `Mutex` because `Gpu` is a
240    /// shared handle and building is rare; nothing here is touched on a
241    /// frame that draws no new fragment.
242    fragment_pipelines: std::sync::Mutex<
243        std::collections::HashMap<(u64, wgpu::TextureFormat), wgpu::RenderPipeline>,
244    >,
245    /// One texture per texture-backed image, uploaded the first time a
246    /// frame on this device draws it and again when its revision moves,
247    /// shared by every window like the pipelines above, dropped when the
248    /// core says the handle is gone.
249    textures: std::sync::Mutex<std::collections::HashMap<u64, std::sync::Arc<ImageTexture>>>,
250    /// Set by the device's lost callback: a driver update, a GPU reset, a
251    /// hang the OS answered by removing the device. Nothing on it works
252    /// again; a shell opens a new one ([`Gpu::lost`]).
253    lost: std::sync::Arc<std::sync::atomic::AtomicBool>,
254}
255
256/// A texture-backed image on the device: the texture, and what was
257/// uploaded into it. A new `Arc` is made when the size changes, which is
258/// what tells a renderer its bind group is stale.
259struct ImageTexture {
260    texture: wgpu::Texture,
261    view: wgpu::TextureView,
262    width: u32,
263    height: u32,
264    /// The revision the pixels in the texture came from, behind a lock
265    /// because the texture is shared and the upload is per device.
266    rev: std::sync::Mutex<u32>,
267}
268
269/// How a [`Gpu`] is opened: what its surfaces must be able to do, decided
270/// before the first one exists because some of it is the instance's.
271#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
272#[non_exhaustive]
273pub struct GpuOptions {
274    /// Its surfaces may be presented with alpha ([`Renderer::new_in_with`],
275    /// [`Renderer::transparent`]), so a window shows what is behind it
276    /// where a frame paints nothing (backlog F126). On Windows this
277    /// presents D3D12 through DirectComposition
278    /// (`Dx12SwapchainKind::DxgiFromVisual`), the one D3D12 swapchain that
279    /// takes alpha, instead of a swapchain on the window's handle — where
280    /// [`see_through_by_visual`] says the window was made for it, and
281    /// nowhere else, so there the surfaces stay opaque otherwise; nothing
282    /// changes on other platforms. Off by default, so an opaque app keeps
283    /// the swapchain it always had.
284    pub transparent: bool,
285}
286
287impl GpuOptions {
288    /// Options whose surfaces may be transparent ([`Self::transparent`]).
289    pub fn transparent(transparent: bool) -> Self {
290        Self { transparent }
291    }
292}
293
294/// Whether a see-through window on Windows is presented by D3D12 through
295/// a DirectComposition visual — the one way its translucent pixels show
296/// what is behind it (backlog F126, RG150). Read from the process's
297/// environment once, so the two things it decides cannot disagree: the
298/// runner creates the window with no GDI surface of its own
299/// (`WS_EX_NOREDIRECTIONBITMAP`) only when it is true, and
300/// [`Gpu::new_with`] under [`GpuOptions::transparent`] presents through a
301/// visual only when it is true. A window with a GDI surface under a
302/// visual composites its translucent pixels over that surface's black;
303/// a window with no surface and a swapchain on its handle draws nothing.
304/// When it is false a transparent window is opaque, and
305/// [`Renderer::transparent`] says so. Only Windows reads it.
306///
307/// See [`see_through_by_visual_with`] for what the two variables say.
308pub fn see_through_by_visual() -> bool {
309    static DECIDED: std::sync::OnceLock<bool> = std::sync::OnceLock::new();
310    *DECIDED.get_or_init(|| {
311        // `var`, not `var_os`: wgpu reads both this way, so a value that
312        // is not Unicode is unset here as it is there.
313        see_through_by_visual_with(
314            std::env::var("WGPU_BACKEND").ok().as_deref(),
315            std::env::var("WGPU_DX12_PRESENTATION_SYSTEM")
316                .ok()
317                .as_deref(),
318        )
319    })
320}
321
322/// [`see_through_by_visual`] from the values of `WGPU_BACKEND` and
323/// `WGPU_DX12_PRESENTATION_SYSTEM` (`None` for unset), each parsed the
324/// way wgpu parses it:
325///
326/// - `WGPU_BACKEND` unset is D3D12 alone, which is what kui opens on
327///   Windows; set, it must list D3D12 (`dx12` or `d3d12`, in wgpu's
328///   comma-separated, case-insensitive list). A list naming others too
329///   is narrowed to D3D12 for a transparent device ([`Gpu::new_with`]),
330///   so the window this decided on is the one the device presents to.
331/// - `WGPU_DX12_PRESENTATION_SYSTEM` naming the handle (`hwnd`,
332///   `DxgiFromHwnd`) is a swapchain on the window's handle, which takes
333///   no alpha; unset, naming the visual (`visual`, `DxgiFromVisual`), or
334///   a value wgpu does not recognise leaves kui's choice, the visual.
335pub fn see_through_by_visual_with(backend: Option<&str>, presentation: Option<&str>) -> bool {
336    let d3d12 =
337        backend.is_none_or(|b| wgpu::Backends::from_comma_list(b).contains(wgpu::Backends::DX12));
338    // Untrimmed, as wgpu matches it (`Dx12SwapchainKind::from_env`).
339    let hwnd =
340        presentation.is_some_and(|p| matches!(p.to_lowercase().as_str(), "dxgifromhwnd" | "hwnd"));
341    d3d12 && !hwnd
342}
343
344impl Gpu {
345    /// Opens a device that can present to `target`, and returns the
346    /// surface it was chosen for.
347    ///
348    /// The first window's surface has to exist before an adapter can be
349    /// picked, so it comes back with the device; later windows get theirs
350    /// from [`Gpu::create_surface`]. Most runners call [`Renderer::new`]
351    /// instead, which does both and builds the renderer. On Windows only
352    /// the D3D12 backend is enabled unless `WGPU_BACKEND` names another;
353    /// for [`GpuOptions::transparent`], a list naming D3D12 among others
354    /// is narrowed to it ([`see_through_by_visual_with`]).
355    pub async fn new(
356        target: impl Into<wgpu::SurfaceTarget<'static>>,
357    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
358        Self::new_with(target, GpuOptions::default()).await
359    }
360
361    /// [`Gpu::new`] with `options`: what every surface opened on this
362    /// device must be able to do.
363    pub async fn new_with(
364        target: impl Into<wgpu::SurfaceTarget<'static>>,
365        options: GpuOptions,
366    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
367        report_faults();
368        // Every backend the build has, as wgpu defaults — but on Windows
369        // D3D12 alone unless `WGPU_BACKEND` names another. An instance
370        // keeps every backend it enumerated alive for as long as it lives,
371        // so with all of them the process holds an OpenGL context and a
372        // Vulkan instance it never draws with, both in the driver's
373        // `nvoglv64.dll`; and wgpu, left to choose, took Vulkan over D3D12
374        // here. Under a driver update that DLL faulted in present rather
375        // than answer `DEVICE_LOST`, which ended the process; D3D12's
376        // `nvwgf2umx.dll` reports the removal, and the shell reopens the
377        // device (`Gpu::lost`).
378        let mut desc = wgpu::InstanceDescriptor::new_without_display_handle_from_env();
379        // A swapchain made on the window's handle is opaque whatever it is
380        // configured with; one presented through a composition visual
381        // takes premultiplied alpha. Only when asked, and only where the
382        // runner made the window for it: `see_through_by_visual` decides
383        // both, from `WGPU_BACKEND` and `WGPU_DX12_PRESENTATION_SYSTEM`.
384        let by_visual = cfg!(windows) && options.transparent && see_through_by_visual();
385        if cfg!(windows) {
386            if std::env::var("WGPU_BACKEND").is_err() {
387                desc.backends = wgpu::Backends::DX12;
388            } else if by_visual {
389                // A list that names D3D12 among others, narrowed to it:
390                // the window was made with no surface of its own, and
391                // only D3D12 presents to such a window. Left to choose,
392                // wgpu takes Vulkan first.
393                desc.backends &= wgpu::Backends::DX12;
394            }
395        }
396        if by_visual {
397            desc.backend_options.dx12.presentation_system = wgpu::Dx12SwapchainKind::DxgiFromVisual;
398        }
399        let instance = wgpu::Instance::new(desc);
400        let surface = instance.create_surface(target)?;
401        let adapter = instance
402            .request_adapter(&wgpu::RequestAdapterOptions {
403                compatible_surface: Some(&surface),
404                ..Default::default()
405            })
406            .await?;
407        // On Windows a surface takes alpha only through the visual, and
408        // only D3D12 presents through one (the backends were narrowed to
409        // it above; this is the check that they held).
410        let alpha =
411            !cfg!(windows) || (by_visual && adapter.get_info().backend == wgpu::Backend::Dx12);
412        // Per-channel blending for LCD subpixel text, when the device has it.
413        let dual_source = adapter
414            .features()
415            .contains(wgpu::Features::DUAL_SOURCE_BLENDING);
416        let (device, queue) = adapter
417            .request_device(&wgpu::DeviceDescriptor {
418                required_features: if dual_source {
419                    wgpu::Features::DUAL_SOURCE_BLENDING
420                } else {
421                    wgpu::Features::empty()
422                },
423                ..Default::default()
424            })
425            .await?;
426        // What goes wrong on the device is said, not swallowed: an error
427        // outside a scope, and the loss of the device itself — remembered
428        // too, so a frame can tell a dead device from a stale swapchain.
429        device.on_uncaptured_error(std::sync::Arc::new(|e| eprintln!("kui: wgpu: {e}")));
430        let lost = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
431        device.set_device_lost_callback({
432            let lost = lost.clone();
433            move |reason, message| {
434                if reason == wgpu::DeviceLostReason::Unknown {
435                    eprintln!("kui: device lost: {message}");
436                    lost.store(true, std::sync::atomic::Ordering::Release);
437                }
438            }
439        });
440        let gpu = Self(std::sync::Arc::new(GpuInner {
441            instance,
442            adapter,
443            device,
444            queue,
445            options,
446            alpha,
447            dual_source,
448            fragment_pipelines: Default::default(),
449            textures: Default::default(),
450            lost,
451        }));
452        Ok((gpu, surface))
453    }
454
455    /// Whether the device is gone (a driver update or a GPU reset took it).
456    ///
457    /// Nothing on a lost device works again: open a new `Gpu` and a new
458    /// renderer on it for every window. [`Renderer::render`] reports the
459    /// same condition as [`RenderError::DeviceLost`].
460    pub fn lost(&self) -> bool {
461        self.0.lost.load(std::sync::atomic::Ordering::Acquire)
462    }
463
464    /// Loses the device on purpose, as a driver update or a GPU reset
465    /// would, so a runner can test its reopening path without one.
466    ///
467    /// On D3D12 the device is really removed (`ID3D12Device5::RemoveDevice`)
468    /// and the loss lands on its next use through the lost callback, as a
469    /// real one does; elsewhere the device is only marked lost.
470    pub fn mark_lost(&self) {
471        #[cfg(windows)]
472        {
473            use windows::Win32::Graphics::Direct3D12::ID3D12Device5;
474            use windows::core::Interface;
475            // SAFETY: the hal device is only read for its raw handle, and
476            // `RemoveDevice` is what D3D12 offers for exactly this.
477            let removed = unsafe {
478                self.0
479                    .device
480                    .as_hal::<wgpu::hal::api::Dx12>()
481                    .and_then(|d| d.raw_device().cast::<ID3D12Device5>().ok())
482                    .map(|d| d.RemoveDevice())
483            };
484            if removed.is_some() {
485                // The loss lands on the device's next use, through the
486                // lost callback, as a real one does.
487                return;
488            }
489        }
490        self.0
491            .lost
492            .store(true, std::sync::atomic::Ordering::Release);
493    }
494
495    /// A surface for another window on the same instance, which is what
496    /// [`Renderer::new_in`] draws into.
497    pub fn create_surface(
498        &self,
499        target: impl Into<wgpu::SurfaceTarget<'static>>,
500    ) -> Result<wgpu::Surface<'static>, wgpu::CreateSurfaceError> {
501        self.0.instance.create_surface(target)
502    }
503
504    /// What the device was opened for ([`Gpu::new_with`]).
505    pub fn options(&self) -> GpuOptions {
506        self.0.options
507    }
508
509    /// The wgpu instance the device was opened on.
510    pub fn instance(&self) -> &wgpu::Instance {
511        &self.0.instance
512    }
513
514    /// The adapter the device was requested from.
515    pub fn adapter(&self) -> &wgpu::Adapter {
516        &self.0.adapter
517    }
518
519    /// The device, for a runner that creates resources of its own on it.
520    pub fn device(&self) -> &wgpu::Device {
521        &self.0.device
522    }
523
524    /// The queue the renderer submits to.
525    pub fn queue(&self) -> &wgpu::Queue {
526        &self.0.queue
527    }
528
529    /// Whether this device blends per channel (dual-source blending), so
530    /// LCD subpixel glyphs draw with per-channel coverage rather than
531    /// their union.
532    pub fn dual_source(&self) -> bool {
533        self.0.dual_source
534    }
535
536    /// The texture for one texture-backed image, uploaded on first sight
537    /// and whenever `rev` has moved past what the texture holds; a size
538    /// change makes a new texture. `None` for a degenerate size, which
539    /// draws nothing.
540    fn image_texture(
541        &self,
542        id: u64,
543        px: &kui_core::display::TexturePixels,
544    ) -> Option<std::sync::Arc<ImageTexture>> {
545        // Degenerate, or past what this device can hold in one texture
546        // (8192 on many adapters, 16384 on Metal): draws nothing, which is
547        // what the core says a texture-backed image that cannot be backed
548        // does, rather than a validation error the device turns into a
549        // panic.
550        let max = self.0.device.limits().max_texture_dimension_2d;
551        if px.width == 0 || px.height == 0 || px.width > max || px.height > max {
552            return None;
553        }
554        let mut cache = self.0.textures.lock().unwrap_or_else(|e| e.into_inner());
555        let fresh = match cache.get(&id) {
556            Some(t) if t.width == px.width && t.height == px.height => {
557                let mut rev = t.rev.lock().unwrap_or_else(|e| e.into_inner());
558                if *rev != px.rev {
559                    upload_image(&self.0.queue, &t.texture, px);
560                    *rev = px.rev;
561                }
562                return Some(t.clone());
563            }
564            _ => {
565                let texture = self.0.device.create_texture(&wgpu::TextureDescriptor {
566                    label: Some("kui.image"),
567                    size: wgpu::Extent3d {
568                        width: px.width,
569                        height: px.height,
570                        depth_or_array_layers: 1,
571                    },
572                    mip_level_count: 1,
573                    sample_count: 1,
574                    dimension: wgpu::TextureDimension::D2,
575                    format: wgpu::TextureFormat::Rgba8Unorm,
576                    usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
577                    view_formats: &[],
578                });
579                upload_image(&self.0.queue, &texture, px);
580                let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
581                std::sync::Arc::new(ImageTexture {
582                    texture,
583                    view,
584                    width: px.width,
585                    height: px.height,
586                    rev: std::sync::Mutex::new(px.rev),
587                })
588            }
589        };
590        cache.insert(id, fresh.clone());
591        Some(fresh)
592    }
593
594    /// Forgets a removed fragment's pipelines, one per surface format it
595    /// was ever drawn in; the GPU frees them once no frame in flight
596    /// holds one.
597    fn drop_fragment_pipelines(&self, id: u64) {
598        self.0
599            .fragment_pipelines
600            .lock()
601            .unwrap_or_else(|e| e.into_inner())
602            .retain(|(fid, _), _| *fid != id);
603    }
604
605    /// Forgets a removed image's texture; the GPU frees it once no bind
606    /// group holds it.
607    fn drop_image_texture(&self, id: u64) {
608        self.0
609            .textures
610            .lock()
611            .unwrap_or_else(|e| e.into_inner())
612            .remove(&id);
613    }
614
615    /// Whether the cache still holds exactly this texture for `id` — what
616    /// a renderer asks before keeping a bind group over it, since a
617    /// removal reaches the cache through whichever window's frame carried
618    /// it and the other windows' bind groups would otherwise hold the
619    /// texture for as long as they live.
620    fn holds_image_texture(&self, id: u64, texture: &std::sync::Arc<ImageTexture>) -> bool {
621        self.0
622            .textures
623            .lock()
624            .unwrap_or_else(|e| e.into_inner())
625            .get(&id)
626            .is_some_and(|t| std::sync::Arc::ptr_eq(t, texture))
627    }
628
629    /// The pipeline for one registered fragment, built on first sight and
630    /// then shared by every window on this device. `source` is the app's
631    /// WGSL, which the core already validated; it is wrapped in the same
632    /// prelude and epilogue here, from `kui_core::fragment::module_source`,
633    /// so what compiles is what was validated.
634    ///
635    /// The source is not validated again here: `Core::add_fragment` parsed
636    /// and validated this exact module text with the same naga this wgpu
637    /// carries, and refused a handle for anything that failed. A module
638    /// that still does not compile is a kui bug, and reaches wgpu's own
639    /// error handler like any other.
640    fn fragment_pipeline(
641        &self,
642        id: u64,
643        source: &str,
644        format: wgpu::TextureFormat,
645        layouts: &FragmentLayouts,
646    ) -> wgpu::RenderPipeline {
647        let mut cache = self
648            .0
649            .fragment_pipelines
650            .lock()
651            .unwrap_or_else(|e| e.into_inner());
652        if let Some(p) = cache.get(&(id, format)) {
653            return p.clone();
654        }
655        let device = &self.0.device;
656        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
657            label: Some("kui.fragment"),
658            source: wgpu::ShaderSource::Wgsl(kui_core::fragment::module_source(source).into()),
659        });
660        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
661            label: Some("kui.fragment"),
662            layout: Some(&layouts.pipeline),
663            vertex: wgpu::VertexState {
664                module: &layouts.vertex,
665                entry_point: Some("vs_main"),
666                compilation_options: Default::default(),
667                buffers: &[Some(instance_buffer_layout(&INSTANCE_ATTRS))],
668            },
669            fragment: Some(wgpu::FragmentState {
670                module: &module,
671                entry_point: Some(kui_core::fragment::ENTRY_POINT),
672                compilation_options: Default::default(),
673                targets: &[Some(wgpu::ColorTargetState {
674                    format,
675                    // A fragment returns premultiplied colour, always over.
676                    blend: Some(wgpu::BlendState {
677                        color: wgpu::BlendComponent {
678                            src_factor: wgpu::BlendFactor::One,
679                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
680                            operation: wgpu::BlendOperation::Add,
681                        },
682                        alpha: wgpu::BlendComponent {
683                            src_factor: wgpu::BlendFactor::One,
684                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
685                            operation: wgpu::BlendOperation::Add,
686                        },
687                    }),
688                    write_mask: wgpu::ColorWrites::ALL,
689                })],
690            }),
691            primitive: wgpu::PrimitiveState::default(),
692            depth_stencil: None,
693            multisample: wgpu::MultisampleState::default(),
694            multiview_mask: None,
695            cache: None,
696        });
697        cache.insert((id, format), pipeline.clone());
698        pipeline
699    }
700}
701
702/// What building a fragment pipeline needs besides its own source: kui's
703/// vertex stage, and the layout that puts the globals at group 0 and the
704/// parameters at group 1.
705struct FragmentLayouts {
706    vertex: wgpu::ShaderModule,
707    pipeline: wgpu::PipelineLayout,
708}
709
710/// The instance attributes both pipelines read; one array so the vertex
711/// layout cannot differ between them.
712const INSTANCE_ATTRS: [wgpu::VertexAttribute; 9] = wgpu::vertex_attr_array![
713    0 => Float32x2, 1 => Float32x2, 2 => Float32x4,
714    3 => Float32x4, 4 => Float32x4, 5 => Float32x4,
715    6 => Float32x4, 7 => Float32x4, 8 => Float32x4,
716];
717
718fn instance_buffer_layout(attrs: &[wgpu::VertexAttribute]) -> wgpu::VertexBufferLayout<'_> {
719    wgpu::VertexBufferLayout {
720        array_stride: std::mem::size_of::<Instance>() as u64,
721        step_mode: wgpu::VertexStepMode::Instance,
722        attributes: attrs,
723    }
724}
725
726impl std::fmt::Debug for Gpu {
727    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
728        f.debug_struct("Gpu")
729            .field("adapter", &self.0.adapter.get_info().name)
730            .field("dual_source", &self.0.dual_source)
731            .finish()
732    }
733}
734
735/// One window's renderer: its surface, the pipelines and a GPU copy of the core's glyph atlas.
736///
737/// Open one per window with [`Renderer::new`] (first window, on a device
738/// of its own) or [`Renderer::new_in`] (another window on a shared
739/// [`Gpu`]). Each frame, hand [`Renderer::render`] the display list and
740/// atlas from `Core::output`; call [`Renderer::resize`] when the window
741/// changes size. The crate root has the whole sequence.
742pub struct Renderer {
743    gpu: Gpu,
744    surface: wgpu::Surface<'static>,
745    config: wgpu::SurfaceConfiguration,
746    pipeline: wgpu::RenderPipeline,
747    globals_buf: wgpu::Buffer,
748    bind_group: wgpu::BindGroup,
749    bind_layout: wgpu::BindGroupLayout,
750    /// The two samplers every group-0 bind group carries: linear at
751    /// binding 2, nearest at 3.
752    samplers: Samplers,
753    /// Per texture-backed image this window has drawn: the device's
754    /// texture, and a bind group of this window's own — group 0 with that
755    /// texture in the atlas's place and a globals copy whose `atlas_size`
756    /// is the texture's, rewritten each frame the image is drawn.
757    texture_binds: std::collections::HashMap<u64, TextureBind>,
758    atlas_tex: wgpu::Texture,
759    atlas_size: u32,
760    atlas_epoch: u64,
761    instance_buf: wgpu::Buffer,
762    instance_cap: usize,
763    instances: Vec<Instance>,
764    /// What a fragment pipeline is built against: kui's vertex stage and
765    /// the two-group layout. Built once per renderer, handed to the
766    /// device's shared cache.
767    fragment_layouts: FragmentLayouts,
768    /// One slot per fragment this frame, each padded to the device's
769    /// dynamic-offset alignment.
770    fragment_params_buf: wgpu::Buffer,
771    fragment_params_cap: usize,
772    fragment_bind: wgpu::BindGroup,
773    fragment_bind_layout: wgpu::BindGroupLayout,
774    /// The alignment slots are padded to; `dynamic_offset` steps by it.
775    uniform_align: u32,
776    /// Scratch for one frame's padded parameter slots.
777    fragment_bytes: Vec<u8>,
778    /// The color a frame is cleared to before anything is drawn.
779    ///
780    /// A runner usually sets it to the theme's background each frame. It
781    /// is written straight through: the renderer asks for a non-sRGB
782    /// surface, so each component is the byte it lands as, the same way a
783    /// quad's color is.
784    pub clear_color: wgpu::Color,
785    /// A picture drawn over the clear and under everything the frame
786    /// draws ([`Renderer::set_ground`]), when there is one.
787    ground: Option<Ground>,
788    /// The backdrop blur's pipelines (backlog F129): made the first time a
789    /// frame blurs, kept for the renderer's life.
790    backdrop_pipes: Option<backdrop::Pipes>,
791    /// Its offscreen frame, at the surface's size, and its scratch, at the
792    /// largest blur's (backlog RG150): made when a frame blurs, dropped by
793    /// the first frame that does not.
794    backdrop: backdrop::Targets,
795    /// The frame's blurs, in paint order.
796    blurs: Vec<backdrop::Blur>,
797}
798
799/// The picture under a frame: its texture, the part of it the window
800/// shows (`uv`), and the pipeline that draws it across the viewport.
801struct Ground {
802    pipeline: wgpu::RenderPipeline,
803    bind: wgpu::BindGroup,
804    uv: wgpu::Buffer,
805    /// Kept for the bind group, which holds a view of it.
806    _texture: wgpu::Texture,
807}
808
809/// The ground's shader: one quad over the viewport, sampling the part of
810/// the picture `g.uv` names (`u0, v0, u1, v1`), opaque.
811const GROUND_SHADER: &str = r"
812struct G { uv: vec4<f32> }
813@group(0) @binding(0) var<uniform> g: G;
814@group(0) @binding(1) var t: texture_2d<f32>;
815@group(0) @binding(2) var s: sampler;
816struct V { @builtin(position) pos: vec4<f32>, @location(0) uv: vec2<f32> }
817@vertex fn vs(@builtin(vertex_index) i: u32) -> V {
818    let x = f32(i & 1u);
819    let y = f32((i >> 1u) & 1u);
820    var o: V;
821    o.pos = vec4<f32>(x * 2.0 - 1.0, 1.0 - y * 2.0, 0.0, 1.0);
822    o.uv = vec2<f32>(mix(g.uv.x, g.uv.z, x), mix(g.uv.y, g.uv.w, y));
823    return o;
824}
825@fragment fn fs(v: V) -> @location(0) vec4<f32> {
826    return vec4<f32>(textureSample(t, s, v.uv).rgb, 1.0);
827}
828";
829
830struct Samplers {
831    linear: wgpu::Sampler,
832    nearest: wgpu::Sampler,
833}
834
835struct TextureBind {
836    texture: std::sync::Arc<ImageTexture>,
837    globals: wgpu::Buffer,
838    bind: wgpu::BindGroup,
839}
840
841fn upload_image(
842    queue: &wgpu::Queue,
843    texture: &wgpu::Texture,
844    px: &kui_core::display::TexturePixels,
845) {
846    queue.write_texture(
847        wgpu::TexelCopyTextureInfo {
848            texture,
849            mip_level: 0,
850            origin: wgpu::Origin3d::ZERO,
851            aspect: wgpu::TextureAspect::All,
852        },
853        &px.rgba,
854        wgpu::TexelCopyBufferLayout {
855            offset: 0,
856            bytes_per_row: Some(px.width * 4),
857            rows_per_image: Some(px.height),
858        },
859        wgpu::Extent3d {
860            width: px.width,
861            height: px.height,
862            depth_or_array_layers: 1,
863        },
864    );
865}
866
867/// The alpha mode an opaque window presents with: `Opaque` wherever the
868/// surface offers it — every surface a window gets does — and the
869/// surface's first mode otherwise. The first mode was the choice before
870/// a surface could be transparent, and is `Opaque` everywhere but a
871/// D3D12 composition swapchain, which lists `Auto` first.
872fn opaque_mode(modes: &[wgpu::CompositeAlphaMode]) -> wgpu::CompositeAlphaMode {
873    if modes.contains(&wgpu::CompositeAlphaMode::Opaque) {
874        wgpu::CompositeAlphaMode::Opaque
875    } else {
876        modes
877            .first()
878            .copied()
879            .unwrap_or(wgpu::CompositeAlphaMode::Opaque)
880    }
881}
882
883/// The alpha mode a transparent window presents with, of the ones the
884/// surface offers, or `None` when it offers none that composites the
885/// frame kui draws — which is premultiplied. `PreMultiplied` first; on
886/// Metal, `PostMultiplied`, which is only how wgpu names a layer that is
887/// not opaque, and Core Animation composites a layer's pixels as
888/// premultiplied whatever it is called; then `Inherit`, the window
889/// system's own way, which under Wayland and an ARGB X11 visual is
890/// premultiplied too. Never `PostMultiplied` on Vulkan or D3D12, where it
891/// means straight alpha and every translucent pixel would darken.
892fn transparent_mode(
893    modes: &[wgpu::CompositeAlphaMode],
894    backend: wgpu::Backend,
895) -> Option<wgpu::CompositeAlphaMode> {
896    use wgpu::CompositeAlphaMode as M;
897    let mut wanted = vec![M::PreMultiplied];
898    if backend == wgpu::Backend::Metal {
899        wanted.push(M::PostMultiplied);
900    }
901    wanted.push(M::Inherit);
902    wanted.into_iter().find(|m| modes.contains(m))
903}
904
905/// Picks the `//DUAL:` or `//SINGLE:` lines of the shader template.
906fn preprocess_shader(src: &str, dual: bool) -> String {
907    let (keep, drop) = if dual {
908        ("//DUAL:", "//SINGLE:")
909    } else {
910        ("//SINGLE:", "//DUAL:")
911    };
912    let mut out = String::with_capacity(src.len());
913    for line in src.lines() {
914        if let Some(rest) = line.strip_prefix(keep) {
915            out.push_str(rest);
916        } else if line.starts_with(drop) {
917            continue;
918        } else {
919            out.push_str(line);
920        }
921        out.push('\n');
922    }
923    out
924}
925
926/// How many frames may be queued ahead of the one on screen by default: two, or one on Windows.
927///
928/// With two, a drawable to render into is waiting while the previous
929/// frame's is still out, so a frame whose thread woke a little late still
930/// makes its vsync (on Metal this is triple buffering). The price is that
931/// a frame built the moment a drawable frees reaches the screen a vsync
932/// later than it could; `kui-native` wins that back by starting frames at
933/// the display's vsync, and a runner driven any other way pays it. On
934/// Windows the flip-model swapchain delivers every vsync with a single
935/// queued frame, so the second would be latency for nothing. Change it
936/// per renderer with [`Renderer::set_frame_latency`].
937pub const DEFAULT_FRAME_LATENCY: u32 = if cfg!(target_os = "windows") { 1 } else { 2 };
938
939impl Renderer {
940    /// A renderer for one window, on a device of its own.
941    ///
942    /// `width` and `height` are the window's size in physical pixels.
943    /// Fails when no adapter can present to `target` or the surface cannot
944    /// be configured. Use [`Renderer::new_in`] for every window after the
945    /// first, so they share the device.
946    pub async fn new(
947        target: impl Into<wgpu::SurfaceTarget<'static>>,
948        width: u32,
949        height: u32,
950    ) -> Result<Self, Box<dyn std::error::Error>> {
951        Self::new_with(target, width, height, GpuOptions::default()).await
952    }
953
954    /// [`Renderer::new`] on a device opened with `options` — what every
955    /// window on it must be able to do ([`GpuOptions`]) — and, under
956    /// [`GpuOptions::transparent`], this window's surface presented with
957    /// alpha where it can be ([`Renderer::transparent`]).
958    pub async fn new_with(
959        target: impl Into<wgpu::SurfaceTarget<'static>>,
960        width: u32,
961        height: u32,
962        options: GpuOptions,
963    ) -> Result<Self, Box<dyn std::error::Error>> {
964        let (gpu, surface) = Gpu::new_with(target, options).await?;
965        Self::with_surface(gpu, surface, width, height, options.transparent)
966    }
967
968    /// Whether the surface is presented with alpha, so what is behind the
969    /// window shows where a frame paints nothing and through a colour with
970    /// alpha ([`Renderer::new_with`], [`Renderer::new_in_with`]). The frame
971    /// is premultiplied, which is what every quad and glyph already blends
972    /// to; clear it to a transparent [`Renderer::clear_color`] for the
973    /// window to show through.
974    pub fn transparent(&self) -> bool {
975        !matches!(
976            self.config.alpha_mode,
977            wgpu::CompositeAlphaMode::Opaque | wgpu::CompositeAlphaMode::Auto
978        )
979    }
980
981    /// A renderer for another window on an existing device, the one every
982    /// window of the app shares. Get `gpu` from the first renderer's
983    /// [`Renderer::gpu`].
984    pub fn new_in(
985        gpu: &Gpu,
986        target: impl Into<wgpu::SurfaceTarget<'static>>,
987        width: u32,
988        height: u32,
989    ) -> Result<Self, Box<dyn std::error::Error>> {
990        Self::new_in_with(gpu, target, width, height, false)
991    }
992
993    /// [`Renderer::new_in`] for a window whose surface is presented with
994    /// alpha (`transparent`, backlog F126). Decided here, once, and not by
995    /// a reconfigure: a D3D12 swapchain keeps the alpha mode it was made
996    /// with, and resizing it later changes nothing — measured on Windows
997    /// 11, where a surface reconfigured to premultiplied after it was made
998    /// opaque went on compositing over black. Where the surface cannot
999    /// take alpha — a D3D12 device not opened with
1000    /// [`GpuOptions::transparent`], a Vulkan or GL surface that offers
1001    /// only opaque — it is opaque, and [`Renderer::transparent`] says so.
1002    pub fn new_in_with(
1003        gpu: &Gpu,
1004        target: impl Into<wgpu::SurfaceTarget<'static>>,
1005        width: u32,
1006        height: u32,
1007        transparent: bool,
1008    ) -> Result<Self, Box<dyn std::error::Error>> {
1009        let surface = gpu.create_surface(target)?;
1010        Self::with_surface(gpu.clone(), surface, width, height, transparent)
1011    }
1012
1013    /// The device this renderer draws with, to open another window on.
1014    pub fn gpu(&self) -> &Gpu {
1015        &self.gpu
1016    }
1017
1018    /// Sets how many frames may be queued ahead of the one on screen (at
1019    /// least one; see [`DEFAULT_FRAME_LATENCY`]). Reconfigures the surface
1020    /// when the value changes.
1021    pub fn set_frame_latency(&mut self, frames: u32) {
1022        let frames = frames.max(1);
1023        if self.config.desired_maximum_frame_latency != frames {
1024            self.config.desired_maximum_frame_latency = frames;
1025            self.surface.configure(self.gpu.device(), &self.config);
1026        }
1027    }
1028
1029    /// The frame latency the surface is configured with.
1030    pub fn frame_latency(&self) -> u32 {
1031        self.config.desired_maximum_frame_latency
1032    }
1033
1034    fn with_surface(
1035        gpu: Gpu,
1036        surface: wgpu::Surface<'static>,
1037        width: u32,
1038        height: u32,
1039        transparent: bool,
1040    ) -> Result<Self, Box<dyn std::error::Error>> {
1041        let device = gpu.device();
1042        let dual_source = gpu.dual_source();
1043
1044        let caps = surface.get_capabilities(gpu.adapter());
1045        let format = caps
1046            .formats
1047            .iter()
1048            .copied()
1049            .find(|f| !f.is_srgb())
1050            .unwrap_or(caps.formats[0]);
1051        let backend = gpu.adapter().get_info().backend;
1052        // Not on a device whose surfaces cannot show what is behind the
1053        // window (Windows, unless `see_through_by_visual`): there a mode
1054        // with alpha would composite over black and still say transparent.
1055        let alpha_mode = (transparent && gpu.0.alpha)
1056            .then(|| transparent_mode(&caps.alpha_modes, backend))
1057            .flatten()
1058            .unwrap_or_else(|| opaque_mode(&caps.alpha_modes));
1059        let config = wgpu::SurfaceConfiguration {
1060            usage: wgpu::TextureUsages::RENDER_ATTACHMENT,
1061            format,
1062            width: width.clamp(1, device.limits().max_texture_dimension_2d),
1063            height: height.clamp(1, device.limits().max_texture_dimension_2d),
1064            present_mode: wgpu::PresentMode::AutoVsync,
1065            alpha_mode,
1066            color_space: wgpu::SurfaceColorSpace::Auto,
1067            view_formats: vec![],
1068            // See `DEFAULT_FRAME_LATENCY`; a runner that wants another
1069            // says so through `set_frame_latency`.
1070            desired_maximum_frame_latency: DEFAULT_FRAME_LATENCY,
1071        };
1072        // A configure that fails only reports to the device's error
1073        // handler, and the first acquire on the unconfigured surface is a
1074        // panic inside wgpu; caught here, it is this constructor's error
1075        // — a window DXGI will not give a second swapchain, say.
1076        let scope = device.push_error_scope(wgpu::ErrorFilter::Validation);
1077        surface.configure(device, &config);
1078        if let Some(err) = pollster::block_on(scope.pop()) {
1079            return Err(format!("configuring the surface: {err}").into());
1080        }
1081
1082        let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1083            label: Some("kui"),
1084            source: wgpu::ShaderSource::Wgsl(
1085                preprocess_shader(include_str!("shader.wgsl"), dual_source).into(),
1086            ),
1087        });
1088        // Dual source: the shader outputs premultiplied color and a
1089        // per-channel coverage; out = src + dst * (1 - coverage). For
1090        // ordinary quads every channel's coverage equals alpha, which is
1091        // exactly premultiplied alpha blending.
1092        let blend = if dual_source {
1093            wgpu::BlendState {
1094                color: wgpu::BlendComponent {
1095                    src_factor: wgpu::BlendFactor::One,
1096                    dst_factor: wgpu::BlendFactor::OneMinusSrc1,
1097                    operation: wgpu::BlendOperation::Add,
1098                },
1099                alpha: wgpu::BlendComponent {
1100                    src_factor: wgpu::BlendFactor::One,
1101                    dst_factor: wgpu::BlendFactor::OneMinusSrc1Alpha,
1102                    operation: wgpu::BlendOperation::Add,
1103                },
1104            }
1105        } else {
1106            wgpu::BlendState::ALPHA_BLENDING
1107        };
1108
1109        let bind_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1110            label: Some("kui.globals"),
1111            entries: &[
1112                wgpu::BindGroupLayoutEntry {
1113                    binding: 0,
1114                    visibility: wgpu::ShaderStages::VERTEX_FRAGMENT,
1115                    ty: wgpu::BindingType::Buffer {
1116                        ty: wgpu::BufferBindingType::Uniform,
1117                        has_dynamic_offset: false,
1118                        min_binding_size: None,
1119                    },
1120                    count: None,
1121                },
1122                wgpu::BindGroupLayoutEntry {
1123                    binding: 1,
1124                    visibility: wgpu::ShaderStages::FRAGMENT,
1125                    ty: wgpu::BindingType::Texture {
1126                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1127                        view_dimension: wgpu::TextureViewDimension::D2,
1128                        multisampled: false,
1129                    },
1130                    count: None,
1131                },
1132                wgpu::BindGroupLayoutEntry {
1133                    binding: 2,
1134                    visibility: wgpu::ShaderStages::FRAGMENT,
1135                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1136                    count: None,
1137                },
1138                wgpu::BindGroupLayoutEntry {
1139                    binding: 3,
1140                    visibility: wgpu::ShaderStages::FRAGMENT,
1141                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1142                    count: None,
1143                },
1144            ],
1145        });
1146
1147        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1148            label: Some("kui"),
1149            bind_group_layouts: &[Some(&bind_layout)],
1150            immediate_size: 0,
1151        });
1152
1153        let instance_attrs = INSTANCE_ATTRS;
1154        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1155            label: Some("kui.quads"),
1156            layout: Some(&pipeline_layout),
1157            vertex: wgpu::VertexState {
1158                module: &shader,
1159                entry_point: Some("vs_main"),
1160                compilation_options: Default::default(),
1161                buffers: &[Some(instance_buffer_layout(&instance_attrs))],
1162            },
1163            fragment: Some(wgpu::FragmentState {
1164                module: &shader,
1165                entry_point: Some("fs_main"),
1166                compilation_options: Default::default(),
1167                targets: &[Some(wgpu::ColorTargetState {
1168                    format,
1169                    blend: Some(blend),
1170                    write_mask: wgpu::ColorWrites::ALL,
1171                })],
1172            }),
1173            primitive: wgpu::PrimitiveState::default(),
1174            depth_stencil: None,
1175            multisample: wgpu::MultisampleState::default(),
1176            multiview_mask: None,
1177            cache: None,
1178        });
1179
1180        let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1181            label: Some("kui.globals"),
1182            size: std::mem::size_of::<Globals>() as u64,
1183            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1184            mapped_at_creation: false,
1185        });
1186
1187        let atlas_size = kui_core::atlas::ATLAS_SIZE;
1188        let atlas_tex = create_atlas_texture(device, atlas_size);
1189        let samplers = Samplers {
1190            linear: device.create_sampler(&wgpu::SamplerDescriptor {
1191                label: Some("kui.linear"),
1192                mag_filter: wgpu::FilterMode::Linear,
1193                min_filter: wgpu::FilterMode::Linear,
1194                ..Default::default()
1195            }),
1196            nearest: device.create_sampler(&wgpu::SamplerDescriptor {
1197                label: Some("kui.nearest"),
1198                mag_filter: wgpu::FilterMode::Nearest,
1199                min_filter: wgpu::FilterMode::Nearest,
1200                ..Default::default()
1201            }),
1202        };
1203        let atlas_view = atlas_tex.create_view(&wgpu::TextureViewDescriptor::default());
1204        let bind_group =
1205            create_bind_group(device, &bind_layout, &globals_buf, &atlas_view, &samplers);
1206
1207        let instance_cap = 4096;
1208        let instance_buf = create_instance_buffer(device, instance_cap);
1209
1210        // Fragments: one uniform slot per draw, picked by dynamic offset,
1211        // and the layout their pipelines are built against. All of it is
1212        // built whether or not a frame ever draws one — a bind group
1213        // layout and an empty buffer, not a pipeline, which is the part
1214        // that costs and is built on first sight.
1215        let fragment_bind_layout =
1216            device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1217                label: Some("kui.fragment.params"),
1218                entries: &[wgpu::BindGroupLayoutEntry {
1219                    binding: 0,
1220                    visibility: wgpu::ShaderStages::FRAGMENT,
1221                    ty: wgpu::BindingType::Buffer {
1222                        ty: wgpu::BufferBindingType::Uniform,
1223                        has_dynamic_offset: true,
1224                        min_binding_size: std::num::NonZeroU64::new(std::mem::size_of::<
1225                            FragmentParams,
1226                        >()
1227                            as u64),
1228                    },
1229                    count: None,
1230                }],
1231            });
1232        let fragment_layouts = FragmentLayouts {
1233            vertex: shader,
1234            pipeline: device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1235                label: Some("kui.fragment"),
1236                bind_group_layouts: &[Some(&bind_layout), Some(&fragment_bind_layout)],
1237                immediate_size: 0,
1238            }),
1239        };
1240        // The *device's* limit, not the adapter's: the device is opened with
1241        // `Limits::default()`, whose `min_uniform_buffer_offset_alignment` is
1242        // 256, and validation holds a dynamic offset to what the device asked
1243        // for rather than to what the hardware could have done. An adapter
1244        // reporting the smaller 64 — which DX12 does — then gave 64-byte slots
1245        // and a validation error on the frame's second fragment.
1246        let uniform_align = device.limits().min_uniform_buffer_offset_alignment;
1247        let fragment_params_cap = 16;
1248        let fragment_params_buf =
1249            create_fragment_params_buffer(device, fragment_params_cap, uniform_align);
1250        let fragment_bind =
1251            create_fragment_bind_group(device, &fragment_bind_layout, &fragment_params_buf);
1252
1253        Ok(Self {
1254            gpu,
1255            surface,
1256            config,
1257            pipeline,
1258            globals_buf,
1259            bind_group,
1260            bind_layout,
1261            samplers,
1262            texture_binds: Default::default(),
1263            atlas_tex,
1264            atlas_size,
1265            atlas_epoch: u64::MAX,
1266            instance_buf,
1267            instance_cap,
1268            instances: Vec::new(),
1269            fragment_layouts,
1270            fragment_params_buf,
1271            fragment_params_cap,
1272            fragment_bind,
1273            fragment_bind_layout,
1274            uniform_align,
1275            fragment_bytes: Vec::new(),
1276            clear_color: wgpu::Color {
1277                r: 0.06,
1278                g: 0.065,
1279                b: 0.08,
1280                a: 1.0,
1281            },
1282            ground: None,
1283            backdrop_pipes: None,
1284            backdrop: backdrop::Targets::default(),
1285            blurs: Vec::new(),
1286        })
1287    }
1288
1289    /// Draws `rgba` (`width` by `height` pixels, four bytes each, row by
1290    /// row from the top left, the bytes the surface takes as they are) over
1291    /// the clear and under everything a frame draws, opaque, stretched
1292    /// across the viewport and sampled with linear filtering — so a small
1293    /// picture reads as a soft one. What a runner draws as a window's
1294    /// ground where the OS has no material to put behind it: the
1295    /// wallpaper, scaled down and blurred once (backlog F126). Which part
1296    /// of the picture shows is [`Renderer::set_ground_uv`]; all of it
1297    /// until that is called. A degenerate size, or pixels that are not
1298    /// that size, clear it.
1299    pub fn set_ground(&mut self, rgba: &[u8], width: u32, height: u32) {
1300        let max = self.gpu.device().limits().max_texture_dimension_2d;
1301        if width == 0
1302            || height == 0
1303            || width > max
1304            || height > max
1305            || rgba.len() != width as usize * height as usize * 4
1306        {
1307            self.ground = None;
1308            return;
1309        }
1310        let device = self.gpu.device();
1311        let size = wgpu::Extent3d {
1312            width,
1313            height,
1314            depth_or_array_layers: 1,
1315        };
1316        let texture = device.create_texture(&wgpu::TextureDescriptor {
1317            label: Some("kui.ground"),
1318            size,
1319            mip_level_count: 1,
1320            sample_count: 1,
1321            dimension: wgpu::TextureDimension::D2,
1322            format: wgpu::TextureFormat::Rgba8Unorm,
1323            usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1324            view_formats: &[],
1325        });
1326        self.gpu.queue().write_texture(
1327            wgpu::TexelCopyTextureInfo {
1328                texture: &texture,
1329                mip_level: 0,
1330                origin: wgpu::Origin3d::ZERO,
1331                aspect: wgpu::TextureAspect::All,
1332            },
1333            rgba,
1334            wgpu::TexelCopyBufferLayout {
1335                offset: 0,
1336                bytes_per_row: Some(width * 4),
1337                rows_per_image: Some(height),
1338            },
1339            size,
1340        );
1341        let uv = device.create_buffer(&wgpu::BufferDescriptor {
1342            label: Some("kui.ground.uv"),
1343            size: 16,
1344            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1345            mapped_at_creation: false,
1346        });
1347        self.gpu
1348            .queue()
1349            .write_buffer(&uv, 0, bytemuck::cast_slice(&[0.0f32, 0.0, 1.0, 1.0]));
1350        let layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1351            label: Some("kui.ground"),
1352            entries: &[
1353                wgpu::BindGroupLayoutEntry {
1354                    binding: 0,
1355                    visibility: wgpu::ShaderStages::VERTEX,
1356                    ty: wgpu::BindingType::Buffer {
1357                        ty: wgpu::BufferBindingType::Uniform,
1358                        has_dynamic_offset: false,
1359                        min_binding_size: None,
1360                    },
1361                    count: None,
1362                },
1363                wgpu::BindGroupLayoutEntry {
1364                    binding: 1,
1365                    visibility: wgpu::ShaderStages::FRAGMENT,
1366                    ty: wgpu::BindingType::Texture {
1367                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1368                        view_dimension: wgpu::TextureViewDimension::D2,
1369                        multisampled: false,
1370                    },
1371                    count: None,
1372                },
1373                wgpu::BindGroupLayoutEntry {
1374                    binding: 2,
1375                    visibility: wgpu::ShaderStages::FRAGMENT,
1376                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1377                    count: None,
1378                },
1379            ],
1380        });
1381        let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
1382        let bind = device.create_bind_group(&wgpu::BindGroupDescriptor {
1383            label: Some("kui.ground"),
1384            layout: &layout,
1385            entries: &[
1386                wgpu::BindGroupEntry {
1387                    binding: 0,
1388                    resource: uv.as_entire_binding(),
1389                },
1390                wgpu::BindGroupEntry {
1391                    binding: 1,
1392                    resource: wgpu::BindingResource::TextureView(&view),
1393                },
1394                wgpu::BindGroupEntry {
1395                    binding: 2,
1396                    resource: wgpu::BindingResource::Sampler(&self.samplers.linear),
1397                },
1398            ],
1399        });
1400        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1401            label: Some("kui.ground"),
1402            source: wgpu::ShaderSource::Wgsl(GROUND_SHADER.into()),
1403        });
1404        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1405            label: Some("kui.ground"),
1406            bind_group_layouts: &[Some(&layout)],
1407            immediate_size: 0,
1408        });
1409        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1410            label: Some("kui.ground"),
1411            layout: Some(&pipeline_layout),
1412            vertex: wgpu::VertexState {
1413                module: &module,
1414                entry_point: Some("vs"),
1415                compilation_options: Default::default(),
1416                buffers: &[],
1417            },
1418            fragment: Some(wgpu::FragmentState {
1419                module: &module,
1420                entry_point: Some("fs"),
1421                compilation_options: Default::default(),
1422                targets: &[Some(wgpu::ColorTargetState {
1423                    format: self.config.format,
1424                    blend: None,
1425                    write_mask: wgpu::ColorWrites::ALL,
1426                })],
1427            }),
1428            primitive: wgpu::PrimitiveState {
1429                topology: wgpu::PrimitiveTopology::TriangleStrip,
1430                ..Default::default()
1431            },
1432            depth_stencil: None,
1433            multisample: wgpu::MultisampleState::default(),
1434            multiview_mask: None,
1435            cache: None,
1436        });
1437        self.ground = Some(Ground {
1438            pipeline,
1439            bind,
1440            uv,
1441            _texture: texture,
1442        });
1443    }
1444
1445    /// Which part of the ground shows across the viewport: `[u0, v0, u1,
1446    /// v1]` in the picture's own 0..1 coordinates, the top-left corner
1447    /// first. Outside 0..1 the picture's edge is stretched. Nothing without
1448    /// a ground.
1449    pub fn set_ground_uv(&mut self, uv: [f32; 4]) {
1450        if let Some(g) = &self.ground {
1451            self.gpu
1452                .queue()
1453                .write_buffer(&g.uv, 0, bytemuck::cast_slice(&uv));
1454        }
1455    }
1456
1457    /// Takes the ground away ([`Renderer::set_ground`]).
1458    pub fn clear_ground(&mut self) {
1459        self.ground = None;
1460    }
1461
1462    /// Whether a ground is drawn under the frame.
1463    pub fn has_ground(&self) -> bool {
1464        self.ground.is_some()
1465    }
1466
1467    /// Whether this device can draw LCD subpixel glyphs
1468    /// (`QuadKind::GlyphSubpixel`) as intended.
1469    ///
1470    /// Pass it to `Core::set_subpixel_text` once after opening the
1471    /// renderer; when it is `false` the core keeps rasterizing grayscale
1472    /// masks, which every device blends correctly.
1473    pub fn subpixel_text(&self) -> bool {
1474        self.gpu.dual_source()
1475    }
1476
1477    /// Reconfigures the swapchain for a new window size, in physical pixels.
1478    ///
1479    /// The size is clamped to at least 1 (a minimized window reports zero,
1480    /// and a zero-sized surface is a validation error) and to the device's
1481    /// largest texture (Windows hands out nonsense sizes mid-resize, and
1482    /// configuring past the limit panics inside wgpu). A clamped frame is
1483    /// one wrong picture; the next real size fixes it.
1484    pub fn resize(&mut self, width: u32, height: u32) {
1485        let max = self.gpu.device().limits().max_texture_dimension_2d;
1486        self.config.width = width.clamp(1, max);
1487        self.config.height = height.clamp(1, max);
1488        self.surface.configure(self.gpu.device(), &self.config);
1489    }
1490
1491    fn sync_atlas(&mut self, atlas: &mut GlyphAtlas) {
1492        if atlas.size != self.atlas_size {
1493            self.atlas_size = atlas.size;
1494            self.atlas_tex = create_atlas_texture(self.gpu.device(), atlas.size);
1495            let view = self
1496                .atlas_tex
1497                .create_view(&wgpu::TextureViewDescriptor::default());
1498            self.bind_group = create_bind_group(
1499                self.gpu.device(),
1500                &self.bind_layout,
1501                &self.globals_buf,
1502                &view,
1503                &self.samplers,
1504            );
1505            self.atlas_epoch = u64::MAX;
1506        }
1507        if atlas.dirty || self.atlas_epoch != atlas.epoch {
1508            self.gpu.queue().write_texture(
1509                wgpu::TexelCopyTextureInfo {
1510                    texture: &self.atlas_tex,
1511                    mip_level: 0,
1512                    origin: wgpu::Origin3d::ZERO,
1513                    aspect: wgpu::TextureAspect::All,
1514                },
1515                &atlas.pixels,
1516                wgpu::TexelCopyBufferLayout {
1517                    offset: 0,
1518                    bytes_per_row: Some(atlas.size * 4),
1519                    rows_per_image: Some(atlas.size),
1520                },
1521                wgpu::Extent3d {
1522                    width: atlas.size,
1523                    height: atlas.size,
1524                    depth_or_array_layers: 1,
1525                },
1526            );
1527            atlas.dirty = false;
1528            self.atlas_epoch = atlas.epoch;
1529        }
1530    }
1531
1532    /// Plans the frame's backdrop blurs (backlog F129), and makes or drops
1533    /// what drawing them takes: the pipelines once, the offscreen frame at
1534    /// the surface's size, the scratch at what the largest blur needs, a
1535    /// parameter slot per blur. `any` is whether the list has a backdrop
1536    /// quad at all, noticed while the instances were written, so a frame
1537    /// without one scans nothing.
1538    ///
1539    /// A frame without a blur drops the frame and the scratch there and
1540    /// then (backlog RG150; [`backdrop::Targets::fit`] says why).
1541    fn plan_backdrops(&mut self, dl: &DisplayList, any: bool) {
1542        self.blurs.clear();
1543        let (w, h) = (self.config.width, self.config.height);
1544        if any {
1545            for (i, q) in dl.quads.iter().enumerate() {
1546                if q.kind == QuadKind::Backdrop
1547                    && let Some(b) = backdrop::plan(i as u32, q, dl.clip_of(q), w, h)
1548                {
1549                    self.blurs.push(b);
1550                }
1551            }
1552        }
1553        if self.blurs.is_empty() {
1554            self.backdrop = backdrop::Targets::default();
1555            return;
1556        }
1557        let device = self.gpu.device();
1558        let align = self.uniform_align;
1559        let format = self.config.format;
1560        let pipes = self
1561            .backdrop_pipes
1562            .get_or_insert_with(|| backdrop::Pipes::new(device, format, align));
1563        if self.blurs.len() > pipes.params_cap {
1564            pipes.params_cap = self.blurs.len().next_power_of_two();
1565            pipes.params = backdrop::params_buffer(device, pipes.params_cap, align);
1566            // Every bind group names the old buffer.
1567            self.backdrop = backdrop::Targets::default();
1568        }
1569        self.backdrop
1570            .fit(device, pipes, format, (w, h), &self.blurs);
1571        let Some((_, scratch)) = self.backdrop.get() else {
1572            return;
1573        };
1574        let slot = align as usize;
1575        let mut bytes = vec![0u8; self.blurs.len() * slot];
1576        for (i, b) in self.blurs.iter().enumerate() {
1577            bytes[i * slot..i * slot + std::mem::size_of::<backdrop::Params>()]
1578                .copy_from_slice(bytemuck::bytes_of(&scratch.params(b)));
1579        }
1580        self.gpu.queue().write_buffer(&pipes.params, 0, &bytes);
1581    }
1582
1583    /// Draws the instances in `range` into `pass`: in one instanced draw,
1584    /// or — where a fragment or a texture-backed image interrupts the run
1585    /// — the run before it with the über-pipeline, then that one quad with
1586    /// its own pipeline (a fragment) or its own group 0 (a texture), then
1587    /// on. Consecutive quads of the same handle still take one set each
1588    /// (about 0.6 us); runs of ordinary quads are unbroken.
1589    fn draw_quads(
1590        &self,
1591        pass: &mut wgpu::RenderPass<'_>,
1592        dl: &DisplayList,
1593        range: std::ops::Range<u32>,
1594        fragment_pipelines: &[wgpu::RenderPipeline],
1595        texture_binds: &[Option<u64>],
1596    ) {
1597        if range.is_empty() {
1598            return;
1599        }
1600        pass.set_vertex_buffer(0, self.instance_buf.slice(..));
1601        if fragment_pipelines.is_empty() && texture_binds.is_empty() {
1602            // The whole run in one instanced draw, as a frame has always
1603            // been. Nothing below runs.
1604            pass.set_pipeline(&self.pipeline);
1605            pass.set_bind_group(0, &self.bind_group, &[]);
1606            pass.draw(0..6, range);
1607            return;
1608        }
1609        let mut run_start = range.start;
1610        let mut on_quads = false;
1611        for i in range.clone() {
1612            let q = &dl.quads[i as usize];
1613            if q.kind != QuadKind::Fragment && q.kind != QuadKind::Texture {
1614                continue;
1615            }
1616            if i > run_start {
1617                if !on_quads {
1618                    pass.set_pipeline(&self.pipeline);
1619                    pass.set_bind_group(0, &self.bind_group, &[]);
1620                    on_quads = true;
1621                }
1622                pass.draw(0..6, run_start..i);
1623            }
1624            // `uv[0]` is the index into the side list, which is also this
1625            // fragment's parameter slot, or this texture's bind.
1626            let slot = q.uv[0] as usize;
1627            if q.kind == QuadKind::Texture {
1628                if let Some(Some(id)) = texture_binds.get(slot)
1629                    && let Some(b) = self.texture_binds.get(id)
1630                {
1631                    pass.set_pipeline(&self.pipeline);
1632                    pass.set_bind_group(0, &b.bind, &[]);
1633                    on_quads = false;
1634                    pass.draw(0..6, i..i + 1);
1635                }
1636            } else if let Some(pipeline) = fragment_pipelines.get(slot) {
1637                // A fragment reading a texture-backed image takes that
1638                // image's group 0 — the texture in the atlas's place,
1639                // `atlas_size` its size — exactly as a texture quad does;
1640                // one reading the atlas, or nothing, takes the frame's. A
1641                // texture the device could not make (a degenerate or
1642                // oversized image) draws the fragment against the atlas
1643                // with a zero rect, which `kui_sample` reads as no image.
1644                let group0 = match draw_image_texture(&dl.fragments[slot]) {
1645                    Some(index) => texture_binds
1646                        .get(index)
1647                        .copied()
1648                        .flatten()
1649                        .and_then(|id| self.texture_binds.get(&id))
1650                        .map_or(&self.bind_group, |b| &b.bind),
1651                    None => &self.bind_group,
1652                };
1653                pass.set_pipeline(pipeline);
1654                pass.set_bind_group(0, group0, &[]);
1655                pass.set_bind_group(1, &self.fragment_bind, &[slot as u32 * self.uniform_align]);
1656                on_quads = false;
1657                pass.draw(0..6, i..i + 1);
1658            }
1659            run_start = i + 1;
1660        }
1661        if range.end > run_start {
1662            if !on_quads {
1663                pass.set_pipeline(&self.pipeline);
1664                pass.set_bind_group(0, &self.bind_group, &[]);
1665            }
1666            pass.draw(0..6, run_start..range.end);
1667        }
1668    }
1669
1670    /// Draws one frame and presents it.
1671    ///
1672    /// Uploads the atlas when it changed since the last frame (and clears
1673    /// its `dirty` flag), writes the list's quads to the instance buffer,
1674    /// draws them over [`Renderer::clear_color`] in one render pass, and
1675    /// presents. Acquiring the swapchain image blocks while vsync holds
1676    /// the frame back; that time comes back as
1677    /// [`RenderReport::vsync_wait_ms`] so a runner can tell pacing from
1678    /// work.
1679    ///
1680    /// Fails without presenting when the surface or the device cannot
1681    /// take the frame; each [`RenderError`] says what to do next.
1682    pub fn render(
1683        &mut self,
1684        dl: &DisplayList,
1685        atlas: &mut GlyphAtlas,
1686    ) -> Result<RenderReport, RenderError> {
1687        // A dead device takes no work: everything below would only add
1688        // errors to the one that lost it.
1689        if self.gpu.lost() {
1690            return Err(RenderError::DeviceLost);
1691        }
1692        self.sync_atlas(atlas);
1693
1694        self.instances.clear();
1695        let mut any_backdrop = false;
1696        self.instances.extend(dl.quads.iter().map(|q| {
1697            any_backdrop |= q.kind == QuadKind::Backdrop;
1698            instance_of(q, &dl.clips, &dl.textures)
1699        }));
1700        self.plan_backdrops(dl, any_backdrop);
1701        if self.instances.len() > self.instance_cap {
1702            self.instance_cap = self.instances.len().next_power_of_two();
1703            self.instance_buf = create_instance_buffer(self.gpu.device(), self.instance_cap);
1704        }
1705        if !self.instances.is_empty() {
1706            self.gpu.queue().write_buffer(
1707                &self.instance_buf,
1708                0,
1709                bytemuck::cast_slice(&self.instances),
1710            );
1711        }
1712        let globals = Globals {
1713            viewport: [dl.viewport.w.max(1.0), dl.viewport.h.max(1.0)],
1714            atlas_size: [self.atlas_size as f32, self.atlas_size as f32],
1715            time: dl.time,
1716            scale: dl.scale,
1717            _pad: [0.0; 2],
1718        };
1719        self.gpu
1720            .queue()
1721            .write_buffer(&self.globals_buf, 0, bytemuck::bytes_of(&globals));
1722
1723        // Texture-backed images: drop what the core
1724        // removed, upload what moved, and give each one drawn this frame
1725        // a group-0 bind group of its own with a globals copy whose
1726        // `atlas_size` is the texture's. All skipped on a frame that
1727        // draws none.
1728        for id in &dl.dropped_textures {
1729            self.texture_binds.remove(&id.to_ffi());
1730            self.gpu.drop_image_texture(id.to_ffi());
1731        }
1732        // And the pipelines of removed fragments — built per handle
1733        // and shared by every window, so one window's list carries the
1734        // removal and this is the only eviction they get.
1735        for id in &dl.dropped_fragments {
1736            self.gpu.drop_fragment_pipelines(id.to_ffi());
1737        }
1738        // A drop another window's frame carried: the cache no longer
1739        // holds the texture this bind group does. One lock per frame,
1740        // and only for a window that has ever drawn a texture.
1741        if !self.texture_binds.is_empty() {
1742            let gpu = &self.gpu;
1743            self.texture_binds
1744                .retain(|id, b| gpu.holds_image_texture(*id, &b.texture));
1745        }
1746        let mut texture_binds: Vec<Option<u64>> = Vec::new();
1747        if !dl.textures.is_empty() {
1748            texture_binds.reserve(dl.textures.len());
1749            for (draw, px) in dl.textures.iter().zip(&dl.texture_pixels) {
1750                let id = draw.id.to_ffi();
1751                let Some(texture) = self.gpu.image_texture(id, px) else {
1752                    texture_binds.push(None);
1753                    continue;
1754                };
1755                let stale = self
1756                    .texture_binds
1757                    .get(&id)
1758                    .is_none_or(|b| !std::sync::Arc::ptr_eq(&b.texture, &texture));
1759                if stale {
1760                    let device = self.gpu.device();
1761                    let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1762                        label: Some("kui.image.globals"),
1763                        size: std::mem::size_of::<Globals>() as u64,
1764                        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1765                        mapped_at_creation: false,
1766                    });
1767                    let bind = create_bind_group(
1768                        device,
1769                        &self.bind_layout,
1770                        &globals_buf,
1771                        &texture.view,
1772                        &self.samplers,
1773                    );
1774                    self.texture_binds.insert(
1775                        id,
1776                        TextureBind {
1777                            texture: texture.clone(),
1778                            globals: globals_buf,
1779                            bind,
1780                        },
1781                    );
1782                }
1783                let b = &self.texture_binds[&id];
1784                let mine = Globals {
1785                    atlas_size: [texture.width as f32, texture.height as f32],
1786                    ..globals
1787                };
1788                self.gpu
1789                    .queue()
1790                    .write_buffer(&b.globals, 0, bytemuck::bytes_of(&mine));
1791                texture_binds.push(Some(id));
1792            }
1793        }
1794
1795        // Each fragment's parameters into its own slot, and its pipeline
1796        // built if this device has not seen the handle before. Both are
1797        // skipped whole on a frame that draws no fragment.
1798        let mut fragment_pipelines: Vec<wgpu::RenderPipeline> = Vec::new();
1799        if !dl.fragments.is_empty() {
1800            let align = self.uniform_align as usize;
1801            if dl.fragments.len() > self.fragment_params_cap {
1802                self.fragment_params_cap = dl.fragments.len().next_power_of_two();
1803                self.fragment_params_buf = create_fragment_params_buffer(
1804                    self.gpu.device(),
1805                    self.fragment_params_cap,
1806                    self.uniform_align,
1807                );
1808                self.fragment_bind = create_fragment_bind_group(
1809                    self.gpu.device(),
1810                    &self.fragment_bind_layout,
1811                    &self.fragment_params_buf,
1812                );
1813            }
1814            self.fragment_bytes.clear();
1815            self.fragment_bytes.resize(dl.fragments.len() * align, 0);
1816            for (i, draw) in dl.fragments.iter().enumerate() {
1817                let uv = draw.image.uv();
1818                let slot = FragmentParams {
1819                    params: draw.params,
1820                    image: [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32],
1821                };
1822                let at = i * align;
1823                self.fragment_bytes[at..at + std::mem::size_of::<FragmentParams>()]
1824                    .copy_from_slice(bytemuck::bytes_of(&slot));
1825            }
1826            self.gpu
1827                .queue()
1828                .write_buffer(&self.fragment_params_buf, 0, &self.fragment_bytes);
1829            fragment_pipelines.reserve(dl.fragments.len());
1830            for (draw, source) in dl.fragments.iter().zip(&dl.fragment_sources) {
1831                fragment_pipelines.push(self.gpu.fragment_pipeline(
1832                    draw.id.to_ffi(),
1833                    source,
1834                    self.config.format,
1835                    &self.fragment_layouts,
1836                ));
1837            }
1838        }
1839
1840        // Acquiring the swapchain image is where vsync backpressure blocks;
1841        // report it separately so latency graphs show pacing vs work.
1842        let t_wait = std::time::Instant::now();
1843        let frame = match self.surface.get_current_texture() {
1844            wgpu::CurrentSurfaceTexture::Success(f)
1845            | wgpu::CurrentSurfaceTexture::Suboptimal(f) => f,
1846            wgpu::CurrentSurfaceTexture::Timeout | wgpu::CurrentSurfaceTexture::Occluded => {
1847                return Err(RenderError::Skip);
1848            }
1849            wgpu::CurrentSurfaceTexture::Outdated | wgpu::CurrentSurfaceTexture::Lost => {
1850                return Err(RenderError::Reconfigure);
1851            }
1852            // The acquire's error went to the device's error handler; if
1853            // it was the device itself, the lost callback has run by now.
1854            wgpu::CurrentSurfaceTexture::Validation => {
1855                return Err(if self.gpu.lost() {
1856                    RenderError::DeviceLost
1857                } else {
1858                    RenderError::Validation
1859                });
1860            }
1861        };
1862        let vsync_wait_ms = t_wait.elapsed().as_secs_f32() * 1e3;
1863        let surface_view = frame
1864            .texture
1865            .create_view(&wgpu::TextureViewDescriptor::default());
1866        let mut encoder = self
1867            .gpu
1868            .device()
1869            .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("kui") });
1870        // A frame that blurs draws into the offscreen copy, so what it has
1871        // drawn can be read back, and breaks its pass at each blur
1872        // (backlog F129); one that does not draws to the surface in one
1873        // pass, as it always has.
1874        let blurring = !self.blurs.is_empty();
1875        let offscreen = match (&self.backdrop_pipes, self.backdrop.get(), blurring) {
1876            (Some(p), Some((f, s)), true) => Some((p, f, s)),
1877            _ => None,
1878        };
1879        let target = offscreen.map_or(&surface_view, |(_, f, _)| &f.view);
1880        let end = self.instances.len() as u32;
1881        let mut start = 0u32;
1882        let mut next = 0usize;
1883        loop {
1884            let stop = self.blurs.get(next).map_or(end, |b| b.quad);
1885            {
1886                let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1887                    label: Some("kui"),
1888                    color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1889                        view: target,
1890                        depth_slice: None,
1891                        resolve_target: None,
1892                        ops: wgpu::Operations {
1893                            load: if next == 0 {
1894                                wgpu::LoadOp::Clear(self.clear_color)
1895                            } else {
1896                                wgpu::LoadOp::Load
1897                            },
1898                            store: wgpu::StoreOp::Store,
1899                        },
1900                    })],
1901                    depth_stencil_attachment: None,
1902                    timestamp_writes: None,
1903                    occlusion_query_set: None,
1904                    multiview_mask: None,
1905                });
1906                // The ground first, opaque over the clear, so everything
1907                // the frame paints with alpha blends over it.
1908                if next == 0
1909                    && let Some(g) = &self.ground
1910                {
1911                    pass.set_pipeline(&g.pipeline);
1912                    pass.set_bind_group(0, &g.bind, &[]);
1913                    pass.draw(0..4, 0..1);
1914                }
1915                // The blur the last pass stopped for, written back before
1916                // anything after its quad draws over it.
1917                if let Some(((pipes, frame, scratch), b)) =
1918                    offscreen.zip(next.checked_sub(1).and_then(|i| self.blurs.get(i)))
1919                {
1920                    let slot = (next - 1) as u32 * self.uniform_align;
1921                    backdrop::composite(&mut pass, pipes, scratch, b, slot);
1922                    pass.set_scissor_rect(0, 0, frame.size.0, frame.size.1);
1923                }
1924                self.draw_quads(
1925                    &mut pass,
1926                    dl,
1927                    start..stop,
1928                    &fragment_pipelines,
1929                    &texture_binds,
1930                );
1931            }
1932            let (Some(b), Some((pipes, frame, scratch))) = (self.blurs.get(next), offscreen) else {
1933                break;
1934            };
1935            let slot = next as u32 * self.uniform_align;
1936            backdrop::record(&mut encoder, pipes, frame, scratch, b, slot);
1937            start = b.quad + 1;
1938            next += 1;
1939        }
1940        if let Some((pipes, frame, _)) = offscreen {
1941            let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1942                label: Some("kui.backdrop.blit"),
1943                color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1944                    view: &surface_view,
1945                    depth_slice: None,
1946                    resolve_target: None,
1947                    ops: wgpu::Operations {
1948                        load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT),
1949                        store: wgpu::StoreOp::Store,
1950                    },
1951                })],
1952                depth_stencil_attachment: None,
1953                timestamp_writes: None,
1954                occlusion_query_set: None,
1955                multiview_mask: None,
1956            });
1957            pass.set_pipeline(&pipes.blit);
1958            pass.set_bind_group(0, &frame.blit, &[0]);
1959            pass.draw(0..3, 0..1);
1960        }
1961        self.gpu.queue().submit([encoder.finish()]);
1962        self.gpu.queue().present(frame);
1963        Ok(RenderReport { vsync_wait_ms })
1964    }
1965}
1966
1967/// The `textures` entry a fragment draw reads its image from, if its
1968/// image has a texture of its own.
1969fn draw_image_texture(draw: &kui_core::FragmentDraw) -> Option<usize> {
1970    match draw.image {
1971        kui_core::FragmentImage::Texture { index, .. } => Some(index as usize),
1972        _ => None,
1973    }
1974}
1975
1976/// Timing details from one [`Renderer::render`] call.
1977#[derive(Clone, Copy, Debug, Default)]
1978pub struct RenderReport {
1979    /// Milliseconds spent blocked acquiring the swapchain image, which is
1980    /// where vsync backpressure shows up.
1981    pub vsync_wait_ms: f32,
1982}
1983
1984/// A frame that produced no image, and what to do about it.
1985///
1986/// Mapped from wgpu's `CurrentSurfaceTexture`; the crate root's example
1987/// handles every variant.
1988#[derive(Clone, Copy, Debug)]
1989pub enum RenderError {
1990    /// The surface is outdated or lost: call [`Renderer::resize`] with the
1991    /// window's size and draw again.
1992    Reconfigure,
1993    /// Nothing can be presented right now (the window is occluded, or the
1994    /// acquire timed out): try again next frame.
1995    Skip,
1996    /// The surface is configured wrong for the window: call
1997    /// [`Renderer::resize`] with the window's size and draw again. A
1998    /// surface that stays wrong is best given up with its device.
1999    Validation,
2000    /// The device is gone ([`Gpu::lost`]): open a new one, and a renderer
2001    /// on it for every window.
2002    DeviceLost,
2003}
2004
2005impl std::fmt::Display for RenderError {
2006    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2007        match self {
2008            Self::Reconfigure => write!(f, "surface outdated or lost; reconfigure"),
2009            Self::Skip => write!(f, "no frame available; skip"),
2010            Self::Validation => write!(f, "surface texture validation error"),
2011            Self::DeviceLost => write!(f, "device lost; reopen"),
2012        }
2013    }
2014}
2015
2016impl std::error::Error for RenderError {}
2017
2018fn create_atlas_texture(device: &wgpu::Device, size: u32) -> wgpu::Texture {
2019    device.create_texture(&wgpu::TextureDescriptor {
2020        label: Some("kui.atlas"),
2021        size: wgpu::Extent3d {
2022            width: size,
2023            height: size,
2024            depth_or_array_layers: 1,
2025        },
2026        mip_level_count: 1,
2027        sample_count: 1,
2028        dimension: wgpu::TextureDimension::D2,
2029        format: wgpu::TextureFormat::Rgba8Unorm,
2030        usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
2031        view_formats: &[],
2032    })
2033}
2034
2035/// Group 0: the globals, a texture — the atlas, or a texture-backed image
2036/// in its place — and the two samplers.
2037fn create_bind_group(
2038    device: &wgpu::Device,
2039    layout: &wgpu::BindGroupLayout,
2040    globals: &wgpu::Buffer,
2041    view: &wgpu::TextureView,
2042    samplers: &Samplers,
2043) -> wgpu::BindGroup {
2044    device.create_bind_group(&wgpu::BindGroupDescriptor {
2045        label: Some("kui"),
2046        layout,
2047        entries: &[
2048            wgpu::BindGroupEntry {
2049                binding: 0,
2050                resource: globals.as_entire_binding(),
2051            },
2052            wgpu::BindGroupEntry {
2053                binding: 1,
2054                resource: wgpu::BindingResource::TextureView(view),
2055            },
2056            wgpu::BindGroupEntry {
2057                binding: 2,
2058                resource: wgpu::BindingResource::Sampler(&samplers.linear),
2059            },
2060            wgpu::BindGroupEntry {
2061                binding: 3,
2062                resource: wgpu::BindingResource::Sampler(&samplers.nearest),
2063            },
2064        ],
2065    })
2066}
2067
2068fn create_instance_buffer(device: &wgpu::Device, cap: usize) -> wgpu::Buffer {
2069    device.create_buffer(&wgpu::BufferDescriptor {
2070        label: Some("kui.instances"),
2071        size: (cap * std::mem::size_of::<Instance>()) as u64,
2072        usage: wgpu::BufferUsages::VERTEX | wgpu::BufferUsages::COPY_DST,
2073        mapped_at_creation: false,
2074    })
2075}
2076
2077fn create_fragment_params_buffer(device: &wgpu::Device, cap: usize, align: u32) -> wgpu::Buffer {
2078    device.create_buffer(&wgpu::BufferDescriptor {
2079        label: Some("kui.fragment.params"),
2080        size: (cap.max(1) * align as usize) as u64,
2081        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
2082        mapped_at_creation: false,
2083    })
2084}
2085
2086fn create_fragment_bind_group(
2087    device: &wgpu::Device,
2088    layout: &wgpu::BindGroupLayout,
2089    buf: &wgpu::Buffer,
2090) -> wgpu::BindGroup {
2091    device.create_bind_group(&wgpu::BindGroupDescriptor {
2092        label: Some("kui.fragment.params"),
2093        layout,
2094        entries: &[wgpu::BindGroupEntry {
2095            binding: 0,
2096            resource: wgpu::BindingResource::Buffer(wgpu::BufferBinding {
2097                buffer: buf,
2098                offset: 0,
2099                size: std::num::NonZeroU64::new(std::mem::size_of::<FragmentParams>() as u64),
2100            }),
2101        }],
2102    })
2103}
2104
2105/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2106///
2107/// A fault in a GPU driver, such as one being replaced under the app, ends
2108/// the process with no line from anyone: it is not a panic, and Windows
2109/// reports only `0xC000041D` for an exception in a window callback. This
2110/// installs an unhandled-exception filter that names the exception code
2111/// and the module the faulting address is in, then lets the crash go on;
2112/// it is a diagnostic, not a recovery. A crash reporter the host installed
2113/// first is still called, with its answer returned; one installed after
2114/// replaces this filter.
2115///
2116/// [`Gpu::new`] calls it, so a runner rarely needs to. Installing it more
2117/// than once is harmless.
2118#[cfg(windows)]
2119pub fn report_faults() {
2120    use std::cell::Cell;
2121    use windows::Win32::Foundation::{
2122        EXCEPTION_ACCESS_VIOLATION, EXCEPTION_ILLEGAL_INSTRUCTION, EXCEPTION_IN_PAGE_ERROR,
2123        EXCEPTION_STACK_OVERFLOW, HMODULE, NTSTATUS, STATUS_FATAL_USER_CALLBACK_EXCEPTION,
2124    };
2125    use windows::Win32::Storage::FileSystem::WriteFile;
2126    use windows::Win32::System::Console::{GetStdHandle, STD_ERROR_HANDLE};
2127    use windows::Win32::System::Diagnostics::Debug::{
2128        AddVectoredExceptionHandler, EXCEPTION_POINTERS, EXCEPTION_RECORD,
2129        LPTOP_LEVEL_EXCEPTION_FILTER, SetUnhandledExceptionFilter,
2130    };
2131    use windows::Win32::System::LibraryLoader::{
2132        GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS, GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2133        GetModuleFileNameW, GetModuleHandleExW,
2134    };
2135
2136    const CONTINUE_SEARCH: i32 = 0;
2137    /// The faults a driver ends a process with: the ones remembered for
2138    /// a callback's `0xC000041D` to be read by.
2139    const FAULTS: [NTSTATUS; 4] = [
2140        EXCEPTION_ACCESS_VIOLATION,
2141        EXCEPTION_ILLEGAL_INSTRUCTION,
2142        EXCEPTION_IN_PAGE_ERROR,
2143        EXCEPTION_STACK_OVERFLOW,
2144    ];
2145    /// The filter this one replaced, called after it; set once, with the
2146    /// two handlers, by the one call that installs them.
2147    static PREVIOUS: std::sync::OnceLock<LPTOP_LEVEL_EXCEPTION_FILTER> = std::sync::OnceLock::new();
2148    thread_local! {
2149        /// The last fault this thread saw, code and address, handled or
2150        /// not. A `const` cell with no destructor: a plain thread-local
2151        /// slot, read and written without allocating or registering
2152        /// anything, from inside an exception.
2153        static LAST: Cell<Option<(i32, usize)>> = const { Cell::new(None) };
2154    }
2155
2156    /// Remembers a fault; says nothing and handles nothing.
2157    unsafe extern "system" fn remember(info: *mut EXCEPTION_POINTERS) -> i32 {
2158        // SAFETY: the system hands a valid record for the exception.
2159        if let Some(record) =
2160            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2161            && FAULTS.contains(&record.ExceptionCode)
2162        {
2163            let seen = (record.ExceptionCode.0, record.ExceptionAddress as usize);
2164            let _ = LAST.try_with(|l| l.set(Some(seen)));
2165        }
2166        CONTINUE_SEARCH
2167    }
2168
2169    /// The first fault on a record's chain of nested exceptions, past the
2170    /// record itself; a few links, since a chain is one or two long and a
2171    /// broken one is not worth following further.
2172    fn nested(record: &EXCEPTION_RECORD) -> Option<(i32, usize)> {
2173        let mut at = record.ExceptionRecord;
2174        for _ in 0..4 {
2175            // SAFETY: a nested record the system chained to this one.
2176            let inner = unsafe { at.as_ref() }?;
2177            if FAULTS.contains(&inner.ExceptionCode) {
2178                return Some((inner.ExceptionCode.0, inner.ExceptionAddress as usize));
2179            }
2180            at = inner.ExceptionRecord;
2181        }
2182        None
2183    }
2184
2185    /// Writes one crash's line to stderr, straight to the handle: no
2186    /// `eprintln!`, which takes a lock and, on a console, converts
2187    /// through a stack buffer eight kilobytes deep — more than a stack
2188    /// overflow leaves.
2189    fn say(code: i32, at: usize, escaped: Option<i32>) {
2190        let mut module = HMODULE::default();
2191        let mut name = [0u16; 260];
2192        // SAFETY: `at` is only looked up, never read; the buffers are ours.
2193        let found = unsafe {
2194            GetModuleHandleExW(
2195                GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS
2196                    | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2197                windows::core::PCWSTR(at as *const u16),
2198                &mut module,
2199            )
2200        }
2201        .is_ok();
2202        let path = found.then(|| {
2203            // SAFETY: the module was just found; the buffer is ours.
2204            let n = unsafe { GetModuleFileNameW(Some(module), &mut name) } as usize;
2205            &name[..n.min(name.len())]
2206        });
2207        let line = FaultLine::new(code as u32, at, path, escaped.map(|c| c as u32));
2208        // SAFETY: a handle the process was given, written from our buffer.
2209        if let Ok(err) = unsafe { GetStdHandle(STD_ERROR_HANDLE) } {
2210            let mut written = 0u32;
2211            let _ = unsafe { WriteFile(err, Some(line.bytes()), Some(&mut written), None) };
2212        }
2213    }
2214
2215    /// The crash: said, then handed to the filter before this one.
2216    unsafe extern "system" fn filter(info: *const EXCEPTION_POINTERS) -> i32 {
2217        // SAFETY: the system hands a valid record for the exception.
2218        if let Some(record) =
2219            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2220        {
2221            let (code, at) = (record.ExceptionCode, record.ExceptionAddress as usize);
2222            let inner = if code == STATUS_FATAL_USER_CALLBACK_EXCEPTION {
2223                nested(record).or_else(|| LAST.try_with(Cell::get).ok().flatten())
2224            } else {
2225                None
2226            };
2227            match inner {
2228                Some((fault, fault_at)) => say(fault, fault_at, Some(code.0)),
2229                None => say(code.0, at, None),
2230            }
2231        }
2232        match PREVIOUS.get().copied().flatten() {
2233            // SAFETY: the filter the system held before ours, called as
2234            // the system would have called it.
2235            Some(previous) => unsafe { previous(info) },
2236            None => CONTINUE_SEARCH,
2237        }
2238    }
2239
2240    PREVIOUS.get_or_init(|| {
2241        // SAFETY: both handlers read only what the system gives them and
2242        // write only their own thread-local slot and stderr.
2243        unsafe {
2244            AddVectoredExceptionHandler(0, Some(remember));
2245            SetUnhandledExceptionFilter(Some(filter))
2246        }
2247    });
2248}
2249
2250/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2251///
2252/// On Windows a fault in a GPU driver ends the process with no line from
2253/// anyone, and this installs the exception filter that names it. On every
2254/// other platform it does nothing. [`Gpu::new`] calls it, so a runner
2255/// rarely needs to.
2256#[cfg(not(windows))]
2257pub fn report_faults() {}
2258
2259/// One crash's line for `report_faults`, written into a buffer on the
2260/// stack: it is said with whatever stack the crash left (a stack
2261/// overflow leaves the few pages the thread reserved for its handlers)
2262/// and in a process whose heap may be what faulted, so nothing here
2263/// allocates. A line too long for it is cut, at a character, and still
2264/// ends in a newline. Built on every platform so it is tested on every
2265/// platform; only Windows says one.
2266#[cfg_attr(not(windows), allow(dead_code))]
2267struct FaultLine {
2268    buf: [u8; 640],
2269    len: usize,
2270}
2271
2272#[cfg_attr(not(windows), allow(dead_code))]
2273impl FaultLine {
2274    /// `kui: fault <code> at <address> in <module>`, and for a fault that
2275    /// escaped a window callback, the code it escaped as. `module` is the
2276    /// UTF-16 path Windows gives; none is a fault outside any module.
2277    fn new(code: u32, at: usize, module: Option<&[u16]>, escaped: Option<u32>) -> Self {
2278        use std::fmt::Write;
2279        let mut line = Self {
2280            buf: [0; 640],
2281            len: 0,
2282        };
2283        let _ = write!(line, "kui: fault {code:#010x} at {at:#x} in ");
2284        match module {
2285            Some(path) => {
2286                for c in char::decode_utf16(path.iter().copied()) {
2287                    let _ = line.write_char(c.unwrap_or(char::REPLACEMENT_CHARACTER));
2288                }
2289            }
2290            None => {
2291                let _ = line.write_str("no module (jit or freed code)");
2292            }
2293        }
2294        if let Some(escaped) = escaped {
2295            let _ = write!(line, ", escaped from a window callback as {escaped:#010x}");
2296        }
2297        // The newline has its byte kept for it (`write_str`).
2298        line.buf[line.len] = b'\n';
2299        line.len += 1;
2300        line
2301    }
2302
2303    fn bytes(&self) -> &[u8] {
2304        &self.buf[..self.len]
2305    }
2306}
2307
2308impl std::fmt::Write for FaultLine {
2309    fn write_str(&mut self, s: &str) -> std::fmt::Result {
2310        // One byte short of the buffer, for the newline.
2311        let room = self.buf.len() - 1 - self.len;
2312        let mut n = s.len().min(room);
2313        while !s.is_char_boundary(n) {
2314            n -= 1;
2315        }
2316        self.buf[self.len..self.len + n].copy_from_slice(&s.as_bytes()[..n]);
2317        self.len += n;
2318        Ok(())
2319    }
2320}
2321
2322#[cfg(test)]
2323mod tests {
2324    use super::*;
2325
2326    /// An opaque window keeps the mode it always had, `Opaque` — also on
2327    /// a composition swapchain, which lists `Auto` first and would have
2328    /// presented `DXGI_ALPHA_MODE_UNSPECIFIED` (backlog F126).
2329    #[test]
2330    fn an_opaque_window_presents_opaque_wherever_it_can() {
2331        use wgpu::CompositeAlphaMode as M;
2332        assert_eq!(opaque_mode(&[M::Opaque]), M::Opaque);
2333        assert_eq!(opaque_mode(&[M::Opaque, M::PostMultiplied]), M::Opaque);
2334        assert_eq!(
2335            opaque_mode(&[
2336                M::Auto,
2337                M::Inherit,
2338                M::Opaque,
2339                M::PostMultiplied,
2340                M::PreMultiplied
2341            ]),
2342            M::Opaque
2343        );
2344        assert_eq!(opaque_mode(&[M::Inherit]), M::Inherit);
2345    }
2346
2347    /// One answer for the window's style and the swapchain (backlog
2348    /// RG150): D3D12 through a visual unless the backend list leaves
2349    /// D3D12 out or the presentation system names the handle — each
2350    /// variable read as wgpu reads it.
2351    #[test]
2352    fn the_window_and_the_swapchain_are_decided_together() {
2353        let by = see_through_by_visual_with;
2354        // Nothing set: kui opens D3D12 alone, through the visual.
2355        assert!(by(None, None));
2356        // `WGPU_BACKEND` naming D3D12, alone or in a list, in either
2357        // spelling and any case, spaces stripped.
2358        for b in ["dx12", "D3D12", "DX12", "vulkan,dx12", " gl , d3d12 "] {
2359            assert!(by(Some(b), None), "{b}");
2360        }
2361        // Naming only others, or nothing wgpu knows.
2362        for b in ["vulkan", "gl", "vk,gl", "", "directx", "dx11"] {
2363            assert!(!by(Some(b), None), "{b}");
2364        }
2365        // The presentation system: the handle's swapchain takes no alpha;
2366        // the visual's, or a value wgpu does not read, leave kui's choice.
2367        for p in ["hwnd", "Hwnd", "DxgiFromHwnd", "dxgifromhwnd"] {
2368            assert!(!by(None, Some(p)), "{p}");
2369            assert!(!by(Some("dx12"), Some(p)), "{p}");
2370        }
2371        for p in ["visual", "DxgiFromVisual", "", "nonsense", " hwnd"] {
2372            assert!(by(None, Some(p)), "{p}");
2373        }
2374        // Both must allow it.
2375        assert!(!by(Some("vulkan"), Some("visual")));
2376    }
2377
2378    /// A transparent one takes a mode that composites premultiplied
2379    /// pixels, and none where only straight alpha or opaque is offered.
2380    #[test]
2381    fn a_transparent_window_presents_premultiplied_or_not_at_all() {
2382        use wgpu::{Backend, CompositeAlphaMode as M};
2383        // D3D12 through a composition visual.
2384        let visual = [
2385            M::Auto,
2386            M::Inherit,
2387            M::Opaque,
2388            M::PostMultiplied,
2389            M::PreMultiplied,
2390        ];
2391        assert_eq!(
2392            transparent_mode(&visual, Backend::Dx12),
2393            Some(M::PreMultiplied)
2394        );
2395        // D3D12 on the window's handle: opaque only.
2396        assert_eq!(transparent_mode(&[M::Opaque], Backend::Dx12), None);
2397        // Metal names its non-opaque layer post-multiplied.
2398        let metal = [M::Opaque, M::PostMultiplied];
2399        assert_eq!(
2400            transparent_mode(&metal, Backend::Metal),
2401            Some(M::PostMultiplied)
2402        );
2403        // Vulkan's post-multiplied is straight alpha: not that.
2404        assert_eq!(
2405            transparent_mode(&[M::Opaque, M::PostMultiplied], Backend::Vulkan),
2406            None
2407        );
2408        assert_eq!(
2409            transparent_mode(&[M::Opaque, M::Inherit], Backend::Vulkan),
2410            Some(M::Inherit)
2411        );
2412    }
2413
2414    /// The globals are one buffer read by two pipelines whose modules
2415    /// declare it separately: this crate's `shader.wgsl` for quads, and
2416    /// `kui_core::fragment::PRELUDE` for every fragment. If the two
2417    /// declarations drift, a fragment reads the wrong bytes and there is
2418    /// nothing to catch it at runtime — the buffer is the right size and
2419    /// the numbers are just wrong. So: same field names, same order, and
2420    /// the size the Rust struct actually is.
2421    #[test]
2422    fn globals_layout_matches() {
2423        let fields = ["viewport", "atlas_size", "time", "scale", "_pad"];
2424        let of = |src: &str, name: &str| {
2425            let start = src
2426                .find(name)
2427                .unwrap_or_else(|| panic!("{name} is not declared in\n{src}"));
2428            let body = &src[start..];
2429            let end = body.find('}').expect("a closing brace");
2430            body[..end].to_string()
2431        };
2432        let quads = of(include_str!("shader.wgsl"), "struct Globals {");
2433        let frags = of(kui_core::fragment::PRELUDE, "struct KuiGlobals {");
2434        let read = |body: &str| -> Vec<String> {
2435            body.lines()
2436                .filter_map(|l| l.split_once(':'))
2437                .map(|(name, ty)| format!("{}: {}", name.trim(), ty.trim().trim_end_matches(',')))
2438                .collect()
2439        };
2440        let (a, b) = (read(&quads), read(&frags));
2441        assert_eq!(a, b, "shader.wgsl and the fragment prelude disagree");
2442        assert_eq!(
2443            a.len(),
2444            fields.len(),
2445            "a field was added to the globals without this test being told"
2446        );
2447        for (row, want) in a.iter().zip(fields) {
2448            assert!(row.starts_with(want), "expected {want}, got {row}");
2449        }
2450        // vec2 + vec2 + f32 + f32 + vec2 = 32 bytes, and a uniform's size
2451        // must be a multiple of sixteen, which is what `_pad` is for.
2452        assert_eq!(std::mem::size_of::<Globals>(), 32);
2453    }
2454
2455    /// The fragment parameter slot is the other buffer two declarations
2456    /// read: `FragmentParams` here and `KuiFragmentParams` in the
2457    /// epilogue. Sixteen floats then the image's rect, 80 bytes.
2458    #[test]
2459    fn fragment_params_layout_matches() {
2460        assert_eq!(std::mem::size_of::<FragmentParams>(), 80);
2461        assert_eq!(std::mem::offset_of!(FragmentParams, image), 64);
2462        let epilogue = kui_core::fragment::EPILOGUE;
2463        assert!(
2464            epilogue
2465                .contains("struct KuiFragmentParams { p: array<vec4<f32>, 4>, image: vec4<f32> };"),
2466            "the epilogue's params struct moved without this test being told"
2467        );
2468    }
2469
2470    /// The bindings the prelude declares at group 0 are this crate's, by
2471    /// number and kind, since a fragment pipeline binds the quad
2472    /// pipeline's group 0 layout as it is.
2473    #[test]
2474    fn prelude_bindings_match_group_zero() {
2475        let prelude = kui_core::fragment::PRELUDE;
2476        for line in [
2477            "@group(0) @binding(0) var<uniform> kui_globals: KuiGlobals;",
2478            "@group(0) @binding(1) var kui_atlas: texture_2d<f32>;",
2479            "@group(0) @binding(2) var kui_sampler: sampler;",
2480            "@group(0) @binding(3) var kui_sampler_nearest: sampler;",
2481        ] {
2482            assert!(prelude.contains(line), "prelude lacks `{line}`");
2483        }
2484        let quads = include_str!("shader.wgsl");
2485        for line in [
2486            "@group(0) @binding(1) var atlas_tex: texture_2d<f32>;",
2487            "@group(0) @binding(2) var atlas_smp: sampler;",
2488            "@group(0) @binding(3) var nearest_smp: sampler;",
2489        ] {
2490            assert!(quads.contains(line), "shader.wgsl lacks `{line}`");
2491        }
2492    }
2493
2494    /// Both preprocessed variants of the shader must parse and validate
2495    /// (pipeline creation would otherwise fail at runtime, in a window).
2496    #[test]
2497    fn shader_variants_validate() {
2498        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2499        for dual in [false, true] {
2500            let src = preprocess_shader(include_str!("shader.wgsl"), dual);
2501            let module = wgpu::naga::front::wgsl::parse_str(&src)
2502                .unwrap_or_else(|e| panic!("dual={dual}: {}", e.emit_to_string(&src)));
2503            let caps = if dual {
2504                Capabilities::DUAL_SOURCE_BLENDING
2505            } else {
2506                Capabilities::empty()
2507            };
2508            Validator::new(ValidationFlags::all(), caps)
2509                .validate(&module)
2510                .unwrap_or_else(|e| panic!("dual={dual}: {e:?}"));
2511        }
2512    }
2513
2514    /// The ground's shader parses and validates, as the pipeline that
2515    /// draws a window's wallpaper would otherwise fail in a window
2516    /// (backlog F126).
2517    #[test]
2518    fn the_ground_shader_validates() {
2519        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2520        let module = wgpu::naga::front::wgsl::parse_str(GROUND_SHADER)
2521            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(GROUND_SHADER)));
2522        Validator::new(ValidationFlags::all(), Capabilities::empty())
2523            .validate(&module)
2524            .unwrap_or_else(|e| panic!("{e:?}"));
2525    }
2526
2527    /// The backdrop blur's shader (backlog F129), every entry point, and
2528    /// its `Params` the size the Rust struct writes.
2529    #[test]
2530    fn the_backdrop_shader_validates() {
2531        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2532        let src = include_str!("backdrop.wgsl");
2533        let module = wgpu::naga::front::wgsl::parse_str(src)
2534            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(src)));
2535        Validator::new(ValidationFlags::all(), Capabilities::empty())
2536            .validate(&module)
2537            .unwrap_or_else(|e| panic!("{e:?}"));
2538        let names: Vec<&str> = module
2539            .entry_points
2540            .iter()
2541            .map(|e| e.name.as_str())
2542            .collect();
2543        for want in [
2544            "vs",
2545            "fs_down",
2546            "fs_blur_h",
2547            "fs_blur_v",
2548            "fs_composite",
2549            "fs_blit",
2550        ] {
2551            assert!(names.contains(&want), "{want} in {names:?}");
2552        }
2553        let params = module
2554            .types
2555            .iter()
2556            .find(|(_, t)| t.name.as_deref() == Some("Params"))
2557            .expect("Params")
2558            .1;
2559        let wgpu::naga::TypeInner::Struct { span, .. } = params.inner else {
2560            panic!("Params is a struct");
2561        };
2562        assert_eq!(span as usize, std::mem::size_of::<backdrop::Params>());
2563    }
2564
2565    /// The line `report_faults` says, built without the heap (RG31): the
2566    /// code, the address and the module's UTF-16 path decoded, and for a
2567    /// fault that escaped a window callback the code it escaped as.
2568    #[test]
2569    fn a_fault_line_names_the_code_the_address_and_the_module() {
2570        let path: Vec<u16> = r"C:\Windows\System32\nvoglv64.dll".encode_utf16().collect();
2571        let line = FaultLine::new(0xC000_0005, 0x7ff6_1234, Some(&path), None);
2572        assert_eq!(
2573            std::str::from_utf8(line.bytes()).unwrap(),
2574            "kui: fault 0xc0000005 at 0x7ff61234 in C:\\Windows\\System32\\nvoglv64.dll\n"
2575        );
2576        let line = FaultLine::new(0xC000_0005, 0x10, None, Some(0xC000_041D));
2577        assert_eq!(
2578            std::str::from_utf8(line.bytes()).unwrap(),
2579            "kui: fault 0xc0000005 at 0x10 in no module (jit or freed code), \
2580             escaped from a window callback as 0xc000041d\n"
2581        );
2582        // A path that is not UTF-16 is said, not refused.
2583        let line = FaultLine::new(0xC000_001D, 0x20, Some(&[0x44, 0xD800, 0x45]), None);
2584        assert_eq!(
2585            std::str::from_utf8(line.bytes()).unwrap(),
2586            "kui: fault 0xc000001d at 0x20 in D\u{FFFD}E\n"
2587        );
2588    }
2589
2590    /// A line longer than its stack buffer is cut, between characters,
2591    /// and still ends in its newline.
2592    #[test]
2593    fn a_fault_line_too_long_is_cut_at_a_character() {
2594        let path: Vec<u16> = "é".repeat(1000).encode_utf16().collect();
2595        let line = FaultLine::new(0xC000_00FD, 0x30, Some(&path), Some(0xC000_041D));
2596        let text = std::str::from_utf8(line.bytes()).expect("cut at a character");
2597        assert!(text.ends_with("é\n"), "{text:?}");
2598        assert!(text.len() <= 640 && text.len() >= 638, "{}", text.len());
2599        assert!(text.starts_with("kui: fault 0xc00000fd at 0x30 in é"));
2600    }
2601}