Skip to main content

kui_wgpu/
lib.rs

1//! wgpu renderer for kui: draws a [`kui_core::DisplayList`] with one instanced pipeline, a single draw call per frame.
2//!
3//! kui splits a UI into a model that lays out and paints into a display
4//! list (`kui-core`), a renderer that puts that list on screen (this
5//! crate) and a runner that owns the window and the event loop
6//! (`kui-native`, the one most apps use). Reach for `kui-wgpu` directly
7//! when you are writing your own runner: you already have a window, or an
8//! event loop, that `kui-native` does not fit.
9//!
10//! A [`Renderer`] owns the swapchain of one window and a GPU copy of the
11//! core's glyph atlas, kept in step with the atlas it is handed each
12//! frame. Rounded rectangles, borders, shadows, glyphs and images are all
13//! instances of the same quad, so an ordinary frame is one draw call;
14//! only a texture-backed image or a custom fragment shader splits it.
15//! Several windows share one device through [`Gpu`].
16//!
17//! # Example
18//!
19//! A runner's whole life with the renderer: open it on a window, tell the
20//! core whether subpixel text will render, then build, draw and present
21//! one frame at a time. `window` is anything wgpu can make a surface
22//! from, such as a `winit` window.
23//!
24//! ```rust,no_run
25//! use kui_core::{Core, Size, TextStyle};
26//! use kui_wgpu::{RenderError, Renderer};
27//!
28//! fn run(
29//!     window: impl Into<kui_wgpu::wgpu::SurfaceTarget<'static>>,
30//! ) -> Result<(), Box<dyn std::error::Error>> {
31//!     let (width, height) = (800u32, 600u32);
32//!     let mut renderer = pollster::block_on(Renderer::new(window, width, height))?;
33//!     let mut core = Core::new();
34//!     core.set_subpixel_text(renderer.subpixel_text());
35//!
36//!     loop {
37//!         // When the windowing library reports a new size:
38//!         // renderer.resize(new_width, new_height);
39//!
40//!         // Build the frame through the core, in logical pixels.
41//!         let scale = 1.0;
42//!         let viewport = Size::new(width as f32 / scale, height as f32 / scale);
43//!         let mut ui = core.frame(viewport, scale);
44//!         ui.text("Hello from a custom runner", TextStyle::new(24.0));
45//!         ui.finish();
46//!
47//!         // Draw it. The atlas is `&mut` so the renderer can clear its dirty flag.
48//!         let (list, atlas) = core.output();
49//!         match renderer.render(list, atlas) {
50//!             Ok(report) => {
51//!                 let _blocked_on_vsync_ms = report.vsync_wait_ms;
52//!             }
53//!             Err(RenderError::Reconfigure | RenderError::Validation) => {
54//!                 renderer.resize(width, height);
55//!             }
56//!             Err(RenderError::Skip) => {}
57//!             Err(RenderError::DeviceLost) => {
58//!                 // Open a new `Renderer` (and a new device) and carry on.
59//!                 break;
60//!             }
61//!         }
62//!     }
63//!     Ok(())
64//! }
65//! ```
66//!
67//! # Where to look
68//!
69//! - [`Renderer`]: one window's swapchain, pipelines and atlas texture.
70//! - [`Renderer::render`]: a display list in, a presented frame (or a
71//!   [`RenderError`]) out.
72//! - [`Renderer::resize`]: reconfigure after the window changed size.
73//! - [`Renderer::subpixel_text`]: what to pass to `Core::set_subpixel_text`.
74//! - [`Gpu`]: the device, queue and adapter that windows share;
75//!   [`Renderer::new_in`] opens a second window on it.
76//! - [`RenderError`]: what each failed frame asks the runner to do next.
77//! - [`DEFAULT_FRAME_LATENCY`] and [`Renderer::set_frame_latency`]: how
78//!   many frames may queue ahead of the one on screen.
79//! - [`wgpu`] is re-exported, so a runner builds against the same version
80//!   this crate was.
81//!
82//! # Subpixel text
83//!
84//! Where the device offers dual-source blending (Metal, DX12, most Vulkan)
85//! the pipeline blends per channel, which is what LCD subpixel glyphs need.
86//! Elsewhere it falls back to ordinary alpha blending and the core should
87//! rasterize grayscale masks instead, which is what
88//! [`Renderer::subpixel_text`] tells it.
89//!
90//! The book: <https://kui-book.qxuken.dev>. Repository:
91//! <https://github.com/qxuken/kui>.
92
93pub use wgpu;
94
95mod backdrop;
96
97use kui_core::atlas::GlyphAtlas;
98use kui_core::{Clip, DisplayList, Quad, QuadKind};
99
100#[repr(C)]
101#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
102struct Instance {
103    pos: [f32; 2],
104    size: [f32; 2],
105    color: [f32; 4],
106    border_color: [f32; 4],
107    params: [f32; 4],
108    uv: [f32; 4],
109    clip: [f32; 4],
110    /// Corner radii, clockwise from the top-left.
111    radii: [f32; 4],
112    /// Radii of the clip itself; all zero = a plain rect clip.
113    clip_radii: [f32; 4],
114    /// The turn the quad is drawn through (ADR 0043): angle in radians,
115    /// scale, tx, ty — the clip entry's `transform`. The identity on
116    /// every quad of a frame with no `rotate` or `scale`.
117    xform: [f32; 4],
118    /// The clip in the quad's own space, before the turn: x, y, w, h.
119    inner: [f32; 4],
120    /// Its radii, as `clip_radii` are the outer clip's.
121    inner_radii: [f32; 4],
122}
123
124/// The frame's own numbers, at group 0 binding 0 for both pipelines.
125/// `kui_core::fragment::PRELUDE` declares the same bytes as `KuiGlobals`
126/// so an app's fragment can read `time` and `scale`; the padding is what
127/// makes the struct a multiple of sixteen, which a uniform must be.
128#[repr(C)]
129#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
130struct Globals {
131    viewport: [f32; 2],
132    atlas_size: [f32; 2],
133    time: f32,
134    scale: f32,
135    _pad: [f32; 2],
136}
137
138/// One fragment's parameters as the shader takes them — the sixteen
139/// floats and the texel rect of its `image`, laid out as the epilogue's
140/// `KuiFragmentParams` — padded out to the device's dynamic-offset
141/// alignment so a frame's draws can share one buffer and pick their slot
142/// by offset.
143#[repr(C)]
144#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
145struct FragmentParams {
146    params: [f32; 16],
147    /// `FragmentIn::image`: `[x, y, w, h]` in the texture bound at group 0
148    /// for this draw — the atlas, or the image's own; zero with none.
149    image: [f32; 4],
150}
151
152/// The clip is resolved out of the frame's table here rather than read off
153/// the quad: it rides as an index (`kui_core::ClipId`) so the display list
154/// carries it once per distinct clip instead of once per quad.
155fn instance_of(q: &Quad, clips: &[Clip], textures: &[kui_core::display::TextureDraw]) -> Instance {
156    let clip = clips.get(q.clip as usize).copied().unwrap_or(Clip::NONE);
157    let kind = match q.kind {
158        QuadKind::Solid => 0.0,
159        QuadKind::GlyphMask => 1.0,
160        QuadKind::GlyphColor => 2.0,
161        QuadKind::Image => 3.0,
162        QuadKind::GlyphSubpixel => 4.0,
163        QuadKind::Shadow => 5.0,
164        QuadKind::Segment => 6.0,
165        QuadKind::Fragment => 7.0,
166        // Drawn by the image branch with its own texture bound in the
167        // atlas's place.
168        QuadKind::Texture => 3.0,
169        // Never drawn by this pipeline: the pass breaks at it and
170        // `backdrop` blurs what is under it. A solid with no colour, so
171        // one a frame did not plan a blur for draws nothing in a run.
172        QuadKind::Backdrop => 0.0,
173    };
174    // `uv` is atlas texels on every kind but two: a segment carries its
175    // endpoints there as f32 bits, and a texture quad an index into the
176    // side list whose entry holds the texel rect. The shader wants floats.
177    let uv = if q.kind == QuadKind::Segment {
178        q.segment_ends()
179    } else if q.kind == QuadKind::Texture {
180        let uv = textures.get(q.uv[0] as usize).map_or([0; 4], |t| t.uv);
181        [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32]
182    } else {
183        [
184            q.uv[0] as f32,
185            q.uv[1] as f32,
186            q.uv[2] as f32,
187            q.uv[3] as f32,
188        ]
189    };
190    if q.kind == QuadKind::Backdrop {
191        return Instance {
192            pos: [q.rect.x, q.rect.y],
193            size: [0.0, 0.0],
194            color: [0.0; 4],
195            border_color: [0.0; 4],
196            params: [0.0; 4],
197            uv: [0.0; 4],
198            clip: [0.0; 4],
199            radii: [0.0; 4],
200            clip_radii: [0.0; 4],
201            xform: [0.0, 1.0, 0.0, 0.0],
202            inner: [0.0; 4],
203            inner_radii: [0.0; 4],
204        };
205    }
206    Instance {
207        pos: [q.rect.x, q.rect.y],
208        size: [q.rect.w, q.rect.h],
209        color: [q.color.r, q.color.g, q.color.b, q.color.a],
210        border_color: [
211            q.border_color.r,
212            q.border_color.g,
213            q.border_color.b,
214            q.border_color.a,
215        ],
216        params: [q.blur, q.border_w, kind, 0.0],
217        uv,
218        clip: [clip.rect.x, clip.rect.y, clip.rect.w, clip.rect.h],
219        radii: q.radius,
220        clip_radii: clip.radius,
221        xform: [
222            clip.transform.angle,
223            clip.transform.scale,
224            clip.transform.tx,
225            clip.transform.ty,
226        ],
227        inner: [clip.inner.x, clip.inner.y, clip.inner.w, clip.inner.h],
228        inner_radii: clip.inner_radius,
229    }
230}
231
232/// The GPU objects an app's windows share: one instance, adapter, device and queue.
233///
234/// Two devices cannot see each other's buffers or textures, so every
235/// window of an app draws through the same `Gpu`. A single-window app
236/// never names it: [`Renderer::new`] opens a private one. A second window
237/// takes the first renderer's [`Renderer::gpu`] and opens through
238/// [`Renderer::new_in`]. Cloning a `Gpu` clones a handle to the same
239/// device.
240#[derive(Clone)]
241pub struct Gpu(std::sync::Arc<GpuInner>);
242
243struct GpuInner {
244    instance: wgpu::Instance,
245    adapter: wgpu::Adapter,
246    device: wgpu::Device,
247    queue: wgpu::Queue,
248    /// What it was opened for ([`Gpu::new_with`]).
249    options: GpuOptions,
250    /// Whether a surface on this device can be presented with alpha at
251    /// all: everywhere but Windows, and there only on D3D12 presenting
252    /// through a composition visual ([`see_through_by_visual`]).
253    alpha: bool,
254    dual_source: bool,
255    /// One pipeline per registered fragment per surface format, built the
256    /// first time a frame draws it (about 0.2 ms, paid once) and shared by
257    /// every window on this device, dropped when a frame's list says the
258    /// handle is gone (`dropped_fragments`). A `Mutex` because `Gpu` is a
259    /// shared handle and building is rare; nothing here is touched on a
260    /// frame that draws no new fragment.
261    fragment_pipelines: std::sync::Mutex<
262        std::collections::HashMap<(u64, wgpu::TextureFormat), wgpu::RenderPipeline>,
263    >,
264    /// One texture per texture-backed image, uploaded the first time a
265    /// frame on this device draws it and again when its revision moves,
266    /// shared by every window like the pipelines above, dropped when the
267    /// core says the handle is gone.
268    textures: std::sync::Mutex<std::collections::HashMap<u64, std::sync::Arc<ImageTexture>>>,
269    /// Set by the device's lost callback: a driver update, a GPU reset, a
270    /// hang the OS answered by removing the device. Nothing on it works
271    /// again; a shell opens a new one ([`Gpu::lost`]).
272    lost: std::sync::Arc<std::sync::atomic::AtomicBool>,
273}
274
275/// A texture-backed image on the device: the texture, and what was
276/// uploaded into it. A new `Arc` is made when the size changes, which is
277/// what tells a renderer its bind group is stale.
278struct ImageTexture {
279    texture: wgpu::Texture,
280    view: wgpu::TextureView,
281    width: u32,
282    height: u32,
283    /// The revision the pixels in the texture came from, behind a lock
284    /// because the texture is shared and the upload is per device.
285    rev: std::sync::Mutex<u32>,
286}
287
288/// How a [`Gpu`] is opened: what its surfaces must be able to do, decided
289/// before the first one exists because some of it is the instance's.
290#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
291#[non_exhaustive]
292pub struct GpuOptions {
293    /// Its surfaces may be presented with alpha ([`Renderer::new_in_with`],
294    /// [`Renderer::transparent`]), so a window shows what is behind it
295    /// where a frame paints nothing (backlog F126). On Windows this
296    /// presents D3D12 through DirectComposition
297    /// (`Dx12SwapchainKind::DxgiFromVisual`), the one D3D12 swapchain that
298    /// takes alpha, instead of a swapchain on the window's handle — where
299    /// [`see_through_by_visual`] says the window was made for it, and
300    /// nowhere else, so there the surfaces stay opaque otherwise; nothing
301    /// changes on other platforms. Off by default, so an opaque app keeps
302    /// the swapchain it always had.
303    pub transparent: bool,
304}
305
306impl GpuOptions {
307    /// Options whose surfaces may be transparent ([`Self::transparent`]).
308    pub fn transparent(transparent: bool) -> Self {
309        Self { transparent }
310    }
311}
312
313/// Whether a see-through window on Windows is presented by D3D12 through
314/// a DirectComposition visual — the one way its translucent pixels show
315/// what is behind it (backlog F126, RG150). Read from the process's
316/// environment once, so the two things it decides cannot disagree: the
317/// runner creates the window with no GDI surface of its own
318/// (`WS_EX_NOREDIRECTIONBITMAP`) only when it is true, and
319/// [`Gpu::new_with`] under [`GpuOptions::transparent`] presents through a
320/// visual only when it is true. A window with a GDI surface under a
321/// visual composites its translucent pixels over that surface's black;
322/// a window with no surface and a swapchain on its handle draws nothing.
323/// When it is false a transparent window is opaque, and
324/// [`Renderer::transparent`] says so. Only Windows reads it.
325///
326/// See [`see_through_by_visual_with`] for what the two variables say.
327pub fn see_through_by_visual() -> bool {
328    static DECIDED: std::sync::OnceLock<bool> = std::sync::OnceLock::new();
329    *DECIDED.get_or_init(|| {
330        // `var`, not `var_os`: wgpu reads both this way, so a value that
331        // is not Unicode is unset here as it is there.
332        see_through_by_visual_with(
333            std::env::var("WGPU_BACKEND").ok().as_deref(),
334            std::env::var("WGPU_DX12_PRESENTATION_SYSTEM")
335                .ok()
336                .as_deref(),
337        )
338    })
339}
340
341/// [`see_through_by_visual`] from the values of `WGPU_BACKEND` and
342/// `WGPU_DX12_PRESENTATION_SYSTEM` (`None` for unset), each parsed the
343/// way wgpu parses it:
344///
345/// - `WGPU_BACKEND` unset is D3D12 alone, which is what kui opens on
346///   Windows; set, it must list D3D12 (`dx12` or `d3d12`, in wgpu's
347///   comma-separated, case-insensitive list). A list naming others too
348///   is narrowed to D3D12 for a transparent device ([`Gpu::new_with`]),
349///   so the window this decided on is the one the device presents to.
350/// - `WGPU_DX12_PRESENTATION_SYSTEM` naming the handle (`hwnd`,
351///   `DxgiFromHwnd`) is a swapchain on the window's handle, which takes
352///   no alpha; unset, naming the visual (`visual`, `DxgiFromVisual`), or
353///   a value wgpu does not recognise leaves kui's choice, the visual.
354pub fn see_through_by_visual_with(backend: Option<&str>, presentation: Option<&str>) -> bool {
355    let d3d12 =
356        backend.is_none_or(|b| wgpu::Backends::from_comma_list(b).contains(wgpu::Backends::DX12));
357    // Untrimmed, as wgpu matches it (`Dx12SwapchainKind::from_env`).
358    let hwnd =
359        presentation.is_some_and(|p| matches!(p.to_lowercase().as_str(), "dxgifromhwnd" | "hwnd"));
360    d3d12 && !hwnd
361}
362
363impl Gpu {
364    /// Opens a device that can present to `target`, and returns the
365    /// surface it was chosen for.
366    ///
367    /// The first window's surface has to exist before an adapter can be
368    /// picked, so it comes back with the device; later windows get theirs
369    /// from [`Gpu::create_surface`]. Most runners call [`Renderer::new`]
370    /// instead, which does both and builds the renderer. On Windows only
371    /// the D3D12 backend is enabled unless `WGPU_BACKEND` names another;
372    /// for [`GpuOptions::transparent`], a list naming D3D12 among others
373    /// is narrowed to it ([`see_through_by_visual_with`]).
374    pub async fn new(
375        target: impl Into<wgpu::SurfaceTarget<'static>>,
376    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
377        Self::new_with(target, GpuOptions::default()).await
378    }
379
380    /// [`Gpu::new`] with `options`: what every surface opened on this
381    /// device must be able to do.
382    pub async fn new_with(
383        target: impl Into<wgpu::SurfaceTarget<'static>>,
384        options: GpuOptions,
385    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
386        report_faults();
387        // Every backend the build has, as wgpu defaults — but on Windows
388        // D3D12 alone unless `WGPU_BACKEND` names another. An instance
389        // keeps every backend it enumerated alive for as long as it lives,
390        // so with all of them the process holds an OpenGL context and a
391        // Vulkan instance it never draws with, both in the driver's
392        // `nvoglv64.dll`; and wgpu, left to choose, took Vulkan over D3D12
393        // here. Under a driver update that DLL faulted in present rather
394        // than answer `DEVICE_LOST`, which ended the process; D3D12's
395        // `nvwgf2umx.dll` reports the removal, and the shell reopens the
396        // device (`Gpu::lost`).
397        let mut desc = wgpu::InstanceDescriptor::new_without_display_handle_from_env();
398        // A swapchain made on the window's handle is opaque whatever it is
399        // configured with; one presented through a composition visual
400        // takes premultiplied alpha. Only when asked, and only where the
401        // runner made the window for it: `see_through_by_visual` decides
402        // both, from `WGPU_BACKEND` and `WGPU_DX12_PRESENTATION_SYSTEM`.
403        let by_visual = cfg!(windows) && options.transparent && see_through_by_visual();
404        if cfg!(windows) {
405            if std::env::var("WGPU_BACKEND").is_err() {
406                desc.backends = wgpu::Backends::DX12;
407            } else if by_visual {
408                // A list that names D3D12 among others, narrowed to it:
409                // the window was made with no surface of its own, and
410                // only D3D12 presents to such a window. Left to choose,
411                // wgpu takes Vulkan first.
412                desc.backends &= wgpu::Backends::DX12;
413            }
414        }
415        if by_visual {
416            desc.backend_options.dx12.presentation_system = wgpu::Dx12SwapchainKind::DxgiFromVisual;
417        }
418        let instance = wgpu::Instance::new(desc);
419        let surface = instance.create_surface(target)?;
420        let adapter = instance
421            .request_adapter(&wgpu::RequestAdapterOptions {
422                compatible_surface: Some(&surface),
423                ..Default::default()
424            })
425            .await?;
426        // On Windows a surface takes alpha only through the visual, and
427        // only D3D12 presents through one (the backends were narrowed to
428        // it above; this is the check that they held).
429        let alpha =
430            !cfg!(windows) || (by_visual && adapter.get_info().backend == wgpu::Backend::Dx12);
431        // Per-channel blending for LCD subpixel text, when the device has it.
432        let dual_source = adapter
433            .features()
434            .contains(wgpu::Features::DUAL_SOURCE_BLENDING);
435        let (device, queue) = adapter
436            .request_device(&wgpu::DeviceDescriptor {
437                required_features: if dual_source {
438                    wgpu::Features::DUAL_SOURCE_BLENDING
439                } else {
440                    wgpu::Features::empty()
441                },
442                ..Default::default()
443            })
444            .await?;
445        // What goes wrong on the device is said, not swallowed: an error
446        // outside a scope, and the loss of the device itself — remembered
447        // too, so a frame can tell a dead device from a stale swapchain.
448        device.on_uncaptured_error(std::sync::Arc::new(|e| eprintln!("kui: wgpu: {e}")));
449        let lost = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
450        device.set_device_lost_callback({
451            let lost = lost.clone();
452            move |reason, message| {
453                if reason == wgpu::DeviceLostReason::Unknown {
454                    eprintln!("kui: device lost: {message}");
455                    lost.store(true, std::sync::atomic::Ordering::Release);
456                }
457            }
458        });
459        let gpu = Self(std::sync::Arc::new(GpuInner {
460            instance,
461            adapter,
462            device,
463            queue,
464            options,
465            alpha,
466            dual_source,
467            fragment_pipelines: Default::default(),
468            textures: Default::default(),
469            lost,
470        }));
471        Ok((gpu, surface))
472    }
473
474    /// Whether the device is gone (a driver update or a GPU reset took it).
475    ///
476    /// Nothing on a lost device works again: open a new `Gpu` and a new
477    /// renderer on it for every window. [`Renderer::render`] reports the
478    /// same condition as [`RenderError::DeviceLost`].
479    pub fn lost(&self) -> bool {
480        self.0.lost.load(std::sync::atomic::Ordering::Acquire)
481    }
482
483    /// Loses the device on purpose, as a driver update or a GPU reset
484    /// would, so a runner can test its reopening path without one.
485    ///
486    /// On D3D12 the device is really removed (`ID3D12Device5::RemoveDevice`)
487    /// and the loss lands on its next use through the lost callback, as a
488    /// real one does; elsewhere the device is only marked lost.
489    pub fn mark_lost(&self) {
490        #[cfg(windows)]
491        {
492            use windows::Win32::Graphics::Direct3D12::ID3D12Device5;
493            use windows::core::Interface;
494            // SAFETY: the hal device is only read for its raw handle, and
495            // `RemoveDevice` is what D3D12 offers for exactly this.
496            let removed = unsafe {
497                self.0
498                    .device
499                    .as_hal::<wgpu::hal::api::Dx12>()
500                    .and_then(|d| d.raw_device().cast::<ID3D12Device5>().ok())
501                    .map(|d| d.RemoveDevice())
502            };
503            if removed.is_some() {
504                // The loss lands on the device's next use, through the
505                // lost callback, as a real one does.
506                return;
507            }
508        }
509        self.0
510            .lost
511            .store(true, std::sync::atomic::Ordering::Release);
512    }
513
514    /// A surface for another window on the same instance, which is what
515    /// [`Renderer::new_in`] draws into.
516    pub fn create_surface(
517        &self,
518        target: impl Into<wgpu::SurfaceTarget<'static>>,
519    ) -> Result<wgpu::Surface<'static>, wgpu::CreateSurfaceError> {
520        self.0.instance.create_surface(target)
521    }
522
523    /// What the device was opened for ([`Gpu::new_with`]).
524    pub fn options(&self) -> GpuOptions {
525        self.0.options
526    }
527
528    /// The wgpu instance the device was opened on.
529    pub fn instance(&self) -> &wgpu::Instance {
530        &self.0.instance
531    }
532
533    /// The adapter the device was requested from.
534    pub fn adapter(&self) -> &wgpu::Adapter {
535        &self.0.adapter
536    }
537
538    /// The device, for a runner that creates resources of its own on it.
539    pub fn device(&self) -> &wgpu::Device {
540        &self.0.device
541    }
542
543    /// The queue the renderer submits to.
544    pub fn queue(&self) -> &wgpu::Queue {
545        &self.0.queue
546    }
547
548    /// Whether this device blends per channel (dual-source blending), so
549    /// LCD subpixel glyphs draw with per-channel coverage rather than
550    /// their union.
551    pub fn dual_source(&self) -> bool {
552        self.0.dual_source
553    }
554
555    /// The texture for one texture-backed image, uploaded on first sight
556    /// and whenever `rev` has moved past what the texture holds; a size
557    /// change makes a new texture. `None` for a degenerate size, which
558    /// draws nothing.
559    fn image_texture(
560        &self,
561        id: u64,
562        px: &kui_core::display::TexturePixels,
563    ) -> Option<std::sync::Arc<ImageTexture>> {
564        // Degenerate, or past what this device can hold in one texture
565        // (8192 on many adapters, 16384 on Metal): draws nothing, which is
566        // what the core says a texture-backed image that cannot be backed
567        // does, rather than a validation error the device turns into a
568        // panic.
569        let max = self.0.device.limits().max_texture_dimension_2d;
570        if px.width == 0 || px.height == 0 || px.width > max || px.height > max {
571            return None;
572        }
573        let mut cache = self.0.textures.lock().unwrap_or_else(|e| e.into_inner());
574        let fresh = match cache.get(&id) {
575            Some(t) if t.width == px.width && t.height == px.height => {
576                let mut rev = t.rev.lock().unwrap_or_else(|e| e.into_inner());
577                if *rev != px.rev {
578                    upload_image(&self.0.queue, &t.texture, px);
579                    *rev = px.rev;
580                }
581                return Some(t.clone());
582            }
583            _ => {
584                let texture = self.0.device.create_texture(&wgpu::TextureDescriptor {
585                    label: Some("kui.image"),
586                    size: wgpu::Extent3d {
587                        width: px.width,
588                        height: px.height,
589                        depth_or_array_layers: 1,
590                    },
591                    mip_level_count: 1,
592                    sample_count: 1,
593                    dimension: wgpu::TextureDimension::D2,
594                    format: wgpu::TextureFormat::Rgba8Unorm,
595                    usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
596                    view_formats: &[],
597                });
598                upload_image(&self.0.queue, &texture, px);
599                let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
600                std::sync::Arc::new(ImageTexture {
601                    texture,
602                    view,
603                    width: px.width,
604                    height: px.height,
605                    rev: std::sync::Mutex::new(px.rev),
606                })
607            }
608        };
609        cache.insert(id, fresh.clone());
610        Some(fresh)
611    }
612
613    /// Forgets a removed fragment's pipelines, one per surface format it
614    /// was ever drawn in; the GPU frees them once no frame in flight
615    /// holds one.
616    fn drop_fragment_pipelines(&self, id: u64) {
617        self.0
618            .fragment_pipelines
619            .lock()
620            .unwrap_or_else(|e| e.into_inner())
621            .retain(|(fid, _), _| *fid != id);
622    }
623
624    /// Forgets a removed image's texture; the GPU frees it once no bind
625    /// group holds it.
626    fn drop_image_texture(&self, id: u64) {
627        self.0
628            .textures
629            .lock()
630            .unwrap_or_else(|e| e.into_inner())
631            .remove(&id);
632    }
633
634    /// Whether the cache still holds exactly this texture for `id` — what
635    /// a renderer asks before keeping a bind group over it, since a
636    /// removal reaches the cache through whichever window's frame carried
637    /// it and the other windows' bind groups would otherwise hold the
638    /// texture for as long as they live.
639    fn holds_image_texture(&self, id: u64, texture: &std::sync::Arc<ImageTexture>) -> bool {
640        self.0
641            .textures
642            .lock()
643            .unwrap_or_else(|e| e.into_inner())
644            .get(&id)
645            .is_some_and(|t| std::sync::Arc::ptr_eq(t, texture))
646    }
647
648    /// The pipeline for one registered fragment, built on first sight and
649    /// then shared by every window on this device. `source` is the app's
650    /// WGSL, which the core already validated; it is wrapped in the same
651    /// prelude and epilogue here, from `kui_core::fragment::module_source`,
652    /// so what compiles is what was validated.
653    ///
654    /// The source is not validated again here: `Core::add_fragment` parsed
655    /// and validated this exact module text with the same naga this wgpu
656    /// carries, and refused a handle for anything that failed. A module
657    /// that still does not compile is a kui bug, and reaches wgpu's own
658    /// error handler like any other.
659    fn fragment_pipeline(
660        &self,
661        id: u64,
662        source: &str,
663        format: wgpu::TextureFormat,
664        layouts: &FragmentLayouts,
665    ) -> wgpu::RenderPipeline {
666        let mut cache = self
667            .0
668            .fragment_pipelines
669            .lock()
670            .unwrap_or_else(|e| e.into_inner());
671        if let Some(p) = cache.get(&(id, format)) {
672            return p.clone();
673        }
674        let device = &self.0.device;
675        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
676            label: Some("kui.fragment"),
677            source: wgpu::ShaderSource::Wgsl(kui_core::fragment::module_source(source).into()),
678        });
679        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
680            label: Some("kui.fragment"),
681            layout: Some(&layouts.pipeline),
682            vertex: wgpu::VertexState {
683                module: &layouts.vertex,
684                entry_point: Some("vs_main"),
685                compilation_options: Default::default(),
686                buffers: &[Some(instance_buffer_layout(&INSTANCE_ATTRS))],
687            },
688            fragment: Some(wgpu::FragmentState {
689                module: &module,
690                entry_point: Some(kui_core::fragment::ENTRY_POINT),
691                compilation_options: Default::default(),
692                targets: &[Some(wgpu::ColorTargetState {
693                    format,
694                    // A fragment returns premultiplied colour, always over.
695                    blend: Some(wgpu::BlendState {
696                        color: wgpu::BlendComponent {
697                            src_factor: wgpu::BlendFactor::One,
698                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
699                            operation: wgpu::BlendOperation::Add,
700                        },
701                        alpha: wgpu::BlendComponent {
702                            src_factor: wgpu::BlendFactor::One,
703                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
704                            operation: wgpu::BlendOperation::Add,
705                        },
706                    }),
707                    write_mask: wgpu::ColorWrites::ALL,
708                })],
709            }),
710            primitive: wgpu::PrimitiveState::default(),
711            depth_stencil: None,
712            multisample: wgpu::MultisampleState::default(),
713            multiview_mask: None,
714            cache: None,
715        });
716        cache.insert((id, format), pipeline.clone());
717        pipeline
718    }
719}
720
721/// What building a fragment pipeline needs besides its own source: kui's
722/// vertex stage, and the layout that puts the globals at group 0 and the
723/// parameters at group 1.
724struct FragmentLayouts {
725    vertex: wgpu::ShaderModule,
726    pipeline: wgpu::PipelineLayout,
727}
728
729/// The instance attributes both pipelines read; one array so the vertex
730/// layout cannot differ between them.
731const INSTANCE_ATTRS: [wgpu::VertexAttribute; 12] = wgpu::vertex_attr_array![
732    0 => Float32x2, 1 => Float32x2, 2 => Float32x4,
733    3 => Float32x4, 4 => Float32x4, 5 => Float32x4,
734    6 => Float32x4, 7 => Float32x4, 8 => Float32x4,
735    9 => Float32x4, 10 => Float32x4, 11 => Float32x4,
736];
737
738fn instance_buffer_layout(attrs: &[wgpu::VertexAttribute]) -> wgpu::VertexBufferLayout<'_> {
739    wgpu::VertexBufferLayout {
740        array_stride: std::mem::size_of::<Instance>() as u64,
741        step_mode: wgpu::VertexStepMode::Instance,
742        attributes: attrs,
743    }
744}
745
746impl std::fmt::Debug for Gpu {
747    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
748        f.debug_struct("Gpu")
749            .field("adapter", &self.0.adapter.get_info().name)
750            .field("dual_source", &self.0.dual_source)
751            .finish()
752    }
753}
754
755/// One window's renderer: its surface, the pipelines and a GPU copy of the core's glyph atlas.
756///
757/// Open one per window with [`Renderer::new`] (first window, on a device
758/// of its own) or [`Renderer::new_in`] (another window on a shared
759/// [`Gpu`]). Each frame, hand [`Renderer::render`] the display list and
760/// atlas from `Core::output`; call [`Renderer::resize`] when the window
761/// changes size. The crate root has the whole sequence.
762pub struct Renderer {
763    gpu: Gpu,
764    surface: wgpu::Surface<'static>,
765    config: wgpu::SurfaceConfiguration,
766    pipeline: wgpu::RenderPipeline,
767    globals_buf: wgpu::Buffer,
768    bind_group: wgpu::BindGroup,
769    bind_layout: wgpu::BindGroupLayout,
770    /// The two samplers every group-0 bind group carries: linear at
771    /// binding 2, nearest at 3.
772    samplers: Samplers,
773    /// Per texture-backed image this window has drawn: the device's
774    /// texture, and a bind group of this window's own — group 0 with that
775    /// texture in the atlas's place and a globals copy whose `atlas_size`
776    /// is the texture's, rewritten each frame the image is drawn.
777    texture_binds: std::collections::HashMap<u64, TextureBind>,
778    atlas_tex: wgpu::Texture,
779    atlas_size: u32,
780    atlas_epoch: u64,
781    instance_buf: wgpu::Buffer,
782    instance_cap: usize,
783    instances: Vec<Instance>,
784    /// What a fragment pipeline is built against: kui's vertex stage and
785    /// the two-group layout. Built once per renderer, handed to the
786    /// device's shared cache.
787    fragment_layouts: FragmentLayouts,
788    /// One slot per fragment this frame, each padded to the device's
789    /// dynamic-offset alignment.
790    fragment_params_buf: wgpu::Buffer,
791    fragment_params_cap: usize,
792    fragment_bind: wgpu::BindGroup,
793    fragment_bind_layout: wgpu::BindGroupLayout,
794    /// The alignment slots are padded to; `dynamic_offset` steps by it.
795    uniform_align: u32,
796    /// Scratch for one frame's padded parameter slots.
797    fragment_bytes: Vec<u8>,
798    /// The color a frame is cleared to before anything is drawn.
799    ///
800    /// A runner usually sets it to the theme's background each frame. It
801    /// is written straight through: the renderer asks for a non-sRGB
802    /// surface, so each component is the byte it lands as, the same way a
803    /// quad's color is.
804    pub clear_color: wgpu::Color,
805    /// A picture drawn over the clear and under everything the frame
806    /// draws ([`Renderer::set_ground`]), when there is one.
807    ground: Option<Ground>,
808    /// The backdrop blur's pipelines (backlog F129): made the first time a
809    /// frame blurs, kept for the renderer's life.
810    backdrop_pipes: Option<backdrop::Pipes>,
811    /// Its offscreen frame, at the surface's size, and its scratch, at the
812    /// largest blur's (backlog RG150): made when a frame blurs, dropped by
813    /// the first frame that does not.
814    backdrop: backdrop::Targets,
815    /// The frame's blurs, in paint order.
816    blurs: Vec<backdrop::Blur>,
817}
818
819/// The picture under a frame: its texture, the part of it the window
820/// shows (`uv`), and the pipeline that draws it across the viewport.
821struct Ground {
822    pipeline: wgpu::RenderPipeline,
823    bind: wgpu::BindGroup,
824    uv: wgpu::Buffer,
825    /// Kept for the bind group, which holds a view of it.
826    _texture: wgpu::Texture,
827}
828
829/// The ground's shader: one quad over the viewport, sampling the part of
830/// the picture `g.uv` names (`u0, v0, u1, v1`), opaque.
831const GROUND_SHADER: &str = r"
832struct G { uv: vec4<f32> }
833@group(0) @binding(0) var<uniform> g: G;
834@group(0) @binding(1) var t: texture_2d<f32>;
835@group(0) @binding(2) var s: sampler;
836struct V { @builtin(position) pos: vec4<f32>, @location(0) uv: vec2<f32> }
837@vertex fn vs(@builtin(vertex_index) i: u32) -> V {
838    let x = f32(i & 1u);
839    let y = f32((i >> 1u) & 1u);
840    var o: V;
841    o.pos = vec4<f32>(x * 2.0 - 1.0, 1.0 - y * 2.0, 0.0, 1.0);
842    o.uv = vec2<f32>(mix(g.uv.x, g.uv.z, x), mix(g.uv.y, g.uv.w, y));
843    return o;
844}
845@fragment fn fs(v: V) -> @location(0) vec4<f32> {
846    return vec4<f32>(textureSample(t, s, v.uv).rgb, 1.0);
847}
848";
849
850struct Samplers {
851    linear: wgpu::Sampler,
852    nearest: wgpu::Sampler,
853}
854
855struct TextureBind {
856    texture: std::sync::Arc<ImageTexture>,
857    globals: wgpu::Buffer,
858    bind: wgpu::BindGroup,
859}
860
861fn upload_image(
862    queue: &wgpu::Queue,
863    texture: &wgpu::Texture,
864    px: &kui_core::display::TexturePixels,
865) {
866    queue.write_texture(
867        wgpu::TexelCopyTextureInfo {
868            texture,
869            mip_level: 0,
870            origin: wgpu::Origin3d::ZERO,
871            aspect: wgpu::TextureAspect::All,
872        },
873        &px.rgba,
874        wgpu::TexelCopyBufferLayout {
875            offset: 0,
876            bytes_per_row: Some(px.width * 4),
877            rows_per_image: Some(px.height),
878        },
879        wgpu::Extent3d {
880            width: px.width,
881            height: px.height,
882            depth_or_array_layers: 1,
883        },
884    );
885}
886
887/// The alpha mode an opaque window presents with: `Opaque` wherever the
888/// surface offers it — every surface a window gets does — and the
889/// surface's first mode otherwise. The first mode was the choice before
890/// a surface could be transparent, and is `Opaque` everywhere but a
891/// D3D12 composition swapchain, which lists `Auto` first.
892fn opaque_mode(modes: &[wgpu::CompositeAlphaMode]) -> wgpu::CompositeAlphaMode {
893    if modes.contains(&wgpu::CompositeAlphaMode::Opaque) {
894        wgpu::CompositeAlphaMode::Opaque
895    } else {
896        modes
897            .first()
898            .copied()
899            .unwrap_or(wgpu::CompositeAlphaMode::Opaque)
900    }
901}
902
903/// The alpha mode a transparent window presents with, of the ones the
904/// surface offers, or `None` when it offers none that composites the
905/// frame kui draws — which is premultiplied. `PreMultiplied` first; on
906/// Metal, `PostMultiplied`, which is only how wgpu names a layer that is
907/// not opaque, and Core Animation composites a layer's pixels as
908/// premultiplied whatever it is called; then `Inherit`, the window
909/// system's own way, which under Wayland and an ARGB X11 visual is
910/// premultiplied too. Never `PostMultiplied` on Vulkan or D3D12, where it
911/// means straight alpha and every translucent pixel would darken.
912fn transparent_mode(
913    modes: &[wgpu::CompositeAlphaMode],
914    backend: wgpu::Backend,
915) -> Option<wgpu::CompositeAlphaMode> {
916    use wgpu::CompositeAlphaMode as M;
917    let mut wanted = vec![M::PreMultiplied];
918    if backend == wgpu::Backend::Metal {
919        wanted.push(M::PostMultiplied);
920    }
921    wanted.push(M::Inherit);
922    wanted.into_iter().find(|m| modes.contains(m))
923}
924
925/// Picks the `//DUAL:` or `//SINGLE:` lines of the shader template.
926fn preprocess_shader(src: &str, dual: bool) -> String {
927    let (keep, drop) = if dual {
928        ("//DUAL:", "//SINGLE:")
929    } else {
930        ("//SINGLE:", "//DUAL:")
931    };
932    let mut out = String::with_capacity(src.len());
933    for line in src.lines() {
934        if let Some(rest) = line.strip_prefix(keep) {
935            out.push_str(rest);
936        } else if line.starts_with(drop) {
937            continue;
938        } else {
939            out.push_str(line);
940        }
941        out.push('\n');
942    }
943    out
944}
945
946/// How many frames may be queued ahead of the one on screen by default: two, or one on Windows.
947///
948/// With two, a drawable to render into is waiting while the previous
949/// frame's is still out, so a frame whose thread woke a little late still
950/// makes its vsync (on Metal this is triple buffering). The price is that
951/// a frame built the moment a drawable frees reaches the screen a vsync
952/// later than it could; `kui-native` wins that back by starting frames at
953/// the display's vsync, and a runner driven any other way pays it. On
954/// Windows the flip-model swapchain delivers every vsync with a single
955/// queued frame, so the second would be latency for nothing. Change it
956/// per renderer with [`Renderer::set_frame_latency`].
957pub const DEFAULT_FRAME_LATENCY: u32 = if cfg!(target_os = "windows") { 1 } else { 2 };
958
959impl Renderer {
960    /// A renderer for one window, on a device of its own.
961    ///
962    /// `width` and `height` are the window's size in physical pixels.
963    /// Fails when no adapter can present to `target` or the surface cannot
964    /// be configured. Use [`Renderer::new_in`] for every window after the
965    /// first, so they share the device.
966    pub async fn new(
967        target: impl Into<wgpu::SurfaceTarget<'static>>,
968        width: u32,
969        height: u32,
970    ) -> Result<Self, Box<dyn std::error::Error>> {
971        Self::new_with(target, width, height, GpuOptions::default()).await
972    }
973
974    /// [`Renderer::new`] on a device opened with `options` — what every
975    /// window on it must be able to do ([`GpuOptions`]) — and, under
976    /// [`GpuOptions::transparent`], this window's surface presented with
977    /// alpha where it can be ([`Renderer::transparent`]).
978    pub async fn new_with(
979        target: impl Into<wgpu::SurfaceTarget<'static>>,
980        width: u32,
981        height: u32,
982        options: GpuOptions,
983    ) -> Result<Self, Box<dyn std::error::Error>> {
984        let (gpu, surface) = Gpu::new_with(target, options).await?;
985        Self::with_surface(gpu, surface, width, height, options.transparent)
986    }
987
988    /// Whether the surface is presented with alpha, so what is behind the
989    /// window shows where a frame paints nothing and through a colour with
990    /// alpha ([`Renderer::new_with`], [`Renderer::new_in_with`]). The frame
991    /// is premultiplied, which is what every quad and glyph already blends
992    /// to; clear it to a transparent [`Renderer::clear_color`] for the
993    /// window to show through.
994    pub fn transparent(&self) -> bool {
995        !matches!(
996            self.config.alpha_mode,
997            wgpu::CompositeAlphaMode::Opaque | wgpu::CompositeAlphaMode::Auto
998        )
999    }
1000
1001    /// A renderer for another window on an existing device, the one every
1002    /// window of the app shares. Get `gpu` from the first renderer's
1003    /// [`Renderer::gpu`].
1004    pub fn new_in(
1005        gpu: &Gpu,
1006        target: impl Into<wgpu::SurfaceTarget<'static>>,
1007        width: u32,
1008        height: u32,
1009    ) -> Result<Self, Box<dyn std::error::Error>> {
1010        Self::new_in_with(gpu, target, width, height, false)
1011    }
1012
1013    /// [`Renderer::new_in`] for a window whose surface is presented with
1014    /// alpha (`transparent`, backlog F126). Decided here, once, and not by
1015    /// a reconfigure: a D3D12 swapchain keeps the alpha mode it was made
1016    /// with, and resizing it later changes nothing — measured on Windows
1017    /// 11, where a surface reconfigured to premultiplied after it was made
1018    /// opaque went on compositing over black. Where the surface cannot
1019    /// take alpha — a D3D12 device not opened with
1020    /// [`GpuOptions::transparent`], a Vulkan or GL surface that offers
1021    /// only opaque — it is opaque, and [`Renderer::transparent`] says so.
1022    pub fn new_in_with(
1023        gpu: &Gpu,
1024        target: impl Into<wgpu::SurfaceTarget<'static>>,
1025        width: u32,
1026        height: u32,
1027        transparent: bool,
1028    ) -> Result<Self, Box<dyn std::error::Error>> {
1029        let surface = gpu.create_surface(target)?;
1030        Self::with_surface(gpu.clone(), surface, width, height, transparent)
1031    }
1032
1033    /// The device this renderer draws with, to open another window on.
1034    pub fn gpu(&self) -> &Gpu {
1035        &self.gpu
1036    }
1037
1038    /// Sets how many frames may be queued ahead of the one on screen (at
1039    /// least one; see [`DEFAULT_FRAME_LATENCY`]). Reconfigures the surface
1040    /// when the value changes.
1041    pub fn set_frame_latency(&mut self, frames: u32) {
1042        let frames = frames.max(1);
1043        if self.config.desired_maximum_frame_latency != frames {
1044            self.config.desired_maximum_frame_latency = frames;
1045            self.surface.configure(self.gpu.device(), &self.config);
1046        }
1047    }
1048
1049    /// The frame latency the surface is configured with.
1050    pub fn frame_latency(&self) -> u32 {
1051        self.config.desired_maximum_frame_latency
1052    }
1053
1054    fn with_surface(
1055        gpu: Gpu,
1056        surface: wgpu::Surface<'static>,
1057        width: u32,
1058        height: u32,
1059        transparent: bool,
1060    ) -> Result<Self, Box<dyn std::error::Error>> {
1061        let device = gpu.device();
1062        let dual_source = gpu.dual_source();
1063
1064        let caps = surface.get_capabilities(gpu.adapter());
1065        let format = caps
1066            .formats
1067            .iter()
1068            .copied()
1069            .find(|f| !f.is_srgb())
1070            .unwrap_or(caps.formats[0]);
1071        let backend = gpu.adapter().get_info().backend;
1072        // Not on a device whose surfaces cannot show what is behind the
1073        // window (Windows, unless `see_through_by_visual`): there a mode
1074        // with alpha would composite over black and still say transparent.
1075        let alpha_mode = (transparent && gpu.0.alpha)
1076            .then(|| transparent_mode(&caps.alpha_modes, backend))
1077            .flatten()
1078            .unwrap_or_else(|| opaque_mode(&caps.alpha_modes));
1079        let config = wgpu::SurfaceConfiguration {
1080            usage: wgpu::TextureUsages::RENDER_ATTACHMENT,
1081            format,
1082            width: width.clamp(1, device.limits().max_texture_dimension_2d),
1083            height: height.clamp(1, device.limits().max_texture_dimension_2d),
1084            present_mode: wgpu::PresentMode::AutoVsync,
1085            alpha_mode,
1086            color_space: wgpu::SurfaceColorSpace::Auto,
1087            view_formats: vec![],
1088            // See `DEFAULT_FRAME_LATENCY`; a runner that wants another
1089            // says so through `set_frame_latency`.
1090            desired_maximum_frame_latency: DEFAULT_FRAME_LATENCY,
1091        };
1092        // A configure that fails only reports to the device's error
1093        // handler, and the first acquire on the unconfigured surface is a
1094        // panic inside wgpu; caught here, it is this constructor's error
1095        // — a window DXGI will not give a second swapchain, say.
1096        let scope = device.push_error_scope(wgpu::ErrorFilter::Validation);
1097        surface.configure(device, &config);
1098        if let Some(err) = pollster::block_on(scope.pop()) {
1099            return Err(format!("configuring the surface: {err}").into());
1100        }
1101
1102        let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1103            label: Some("kui"),
1104            source: wgpu::ShaderSource::Wgsl(
1105                preprocess_shader(include_str!("shader.wgsl"), dual_source).into(),
1106            ),
1107        });
1108        // Dual source: the shader outputs premultiplied color and a
1109        // per-channel coverage; out = src + dst * (1 - coverage). For
1110        // ordinary quads every channel's coverage equals alpha, which is
1111        // exactly premultiplied alpha blending.
1112        let blend = if dual_source {
1113            wgpu::BlendState {
1114                color: wgpu::BlendComponent {
1115                    src_factor: wgpu::BlendFactor::One,
1116                    dst_factor: wgpu::BlendFactor::OneMinusSrc1,
1117                    operation: wgpu::BlendOperation::Add,
1118                },
1119                alpha: wgpu::BlendComponent {
1120                    src_factor: wgpu::BlendFactor::One,
1121                    dst_factor: wgpu::BlendFactor::OneMinusSrc1Alpha,
1122                    operation: wgpu::BlendOperation::Add,
1123                },
1124            }
1125        } else {
1126            wgpu::BlendState::ALPHA_BLENDING
1127        };
1128
1129        let bind_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1130            label: Some("kui.globals"),
1131            entries: &[
1132                wgpu::BindGroupLayoutEntry {
1133                    binding: 0,
1134                    visibility: wgpu::ShaderStages::VERTEX_FRAGMENT,
1135                    ty: wgpu::BindingType::Buffer {
1136                        ty: wgpu::BufferBindingType::Uniform,
1137                        has_dynamic_offset: false,
1138                        min_binding_size: None,
1139                    },
1140                    count: None,
1141                },
1142                wgpu::BindGroupLayoutEntry {
1143                    binding: 1,
1144                    visibility: wgpu::ShaderStages::FRAGMENT,
1145                    ty: wgpu::BindingType::Texture {
1146                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1147                        view_dimension: wgpu::TextureViewDimension::D2,
1148                        multisampled: false,
1149                    },
1150                    count: None,
1151                },
1152                wgpu::BindGroupLayoutEntry {
1153                    binding: 2,
1154                    visibility: wgpu::ShaderStages::FRAGMENT,
1155                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1156                    count: None,
1157                },
1158                wgpu::BindGroupLayoutEntry {
1159                    binding: 3,
1160                    visibility: wgpu::ShaderStages::FRAGMENT,
1161                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1162                    count: None,
1163                },
1164            ],
1165        });
1166
1167        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1168            label: Some("kui"),
1169            bind_group_layouts: &[Some(&bind_layout)],
1170            immediate_size: 0,
1171        });
1172
1173        let instance_attrs = INSTANCE_ATTRS;
1174        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1175            label: Some("kui.quads"),
1176            layout: Some(&pipeline_layout),
1177            vertex: wgpu::VertexState {
1178                module: &shader,
1179                entry_point: Some("vs_main"),
1180                compilation_options: Default::default(),
1181                buffers: &[Some(instance_buffer_layout(&instance_attrs))],
1182            },
1183            fragment: Some(wgpu::FragmentState {
1184                module: &shader,
1185                entry_point: Some("fs_main"),
1186                compilation_options: Default::default(),
1187                targets: &[Some(wgpu::ColorTargetState {
1188                    format,
1189                    blend: Some(blend),
1190                    write_mask: wgpu::ColorWrites::ALL,
1191                })],
1192            }),
1193            primitive: wgpu::PrimitiveState::default(),
1194            depth_stencil: None,
1195            multisample: wgpu::MultisampleState::default(),
1196            multiview_mask: None,
1197            cache: None,
1198        });
1199
1200        let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1201            label: Some("kui.globals"),
1202            size: std::mem::size_of::<Globals>() as u64,
1203            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1204            mapped_at_creation: false,
1205        });
1206
1207        let atlas_size = kui_core::atlas::ATLAS_SIZE;
1208        let atlas_tex = create_atlas_texture(device, atlas_size);
1209        let samplers = Samplers {
1210            linear: device.create_sampler(&wgpu::SamplerDescriptor {
1211                label: Some("kui.linear"),
1212                mag_filter: wgpu::FilterMode::Linear,
1213                min_filter: wgpu::FilterMode::Linear,
1214                ..Default::default()
1215            }),
1216            nearest: device.create_sampler(&wgpu::SamplerDescriptor {
1217                label: Some("kui.nearest"),
1218                mag_filter: wgpu::FilterMode::Nearest,
1219                min_filter: wgpu::FilterMode::Nearest,
1220                ..Default::default()
1221            }),
1222        };
1223        let atlas_view = atlas_tex.create_view(&wgpu::TextureViewDescriptor::default());
1224        let bind_group =
1225            create_bind_group(device, &bind_layout, &globals_buf, &atlas_view, &samplers);
1226
1227        let instance_cap = 4096;
1228        let instance_buf = create_instance_buffer(device, instance_cap);
1229
1230        // Fragments: one uniform slot per draw, picked by dynamic offset,
1231        // and the layout their pipelines are built against. All of it is
1232        // built whether or not a frame ever draws one — a bind group
1233        // layout and an empty buffer, not a pipeline, which is the part
1234        // that costs and is built on first sight.
1235        let fragment_bind_layout =
1236            device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1237                label: Some("kui.fragment.params"),
1238                entries: &[wgpu::BindGroupLayoutEntry {
1239                    binding: 0,
1240                    visibility: wgpu::ShaderStages::FRAGMENT,
1241                    ty: wgpu::BindingType::Buffer {
1242                        ty: wgpu::BufferBindingType::Uniform,
1243                        has_dynamic_offset: true,
1244                        min_binding_size: std::num::NonZeroU64::new(std::mem::size_of::<
1245                            FragmentParams,
1246                        >()
1247                            as u64),
1248                    },
1249                    count: None,
1250                }],
1251            });
1252        let fragment_layouts = FragmentLayouts {
1253            vertex: shader,
1254            pipeline: device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1255                label: Some("kui.fragment"),
1256                bind_group_layouts: &[Some(&bind_layout), Some(&fragment_bind_layout)],
1257                immediate_size: 0,
1258            }),
1259        };
1260        // The *device's* limit, not the adapter's: the device is opened with
1261        // `Limits::default()`, whose `min_uniform_buffer_offset_alignment` is
1262        // 256, and validation holds a dynamic offset to what the device asked
1263        // for rather than to what the hardware could have done. An adapter
1264        // reporting the smaller 64 — which DX12 does — then gave 64-byte slots
1265        // and a validation error on the frame's second fragment.
1266        let uniform_align = device.limits().min_uniform_buffer_offset_alignment;
1267        let fragment_params_cap = 16;
1268        let fragment_params_buf =
1269            create_fragment_params_buffer(device, fragment_params_cap, uniform_align);
1270        let fragment_bind =
1271            create_fragment_bind_group(device, &fragment_bind_layout, &fragment_params_buf);
1272
1273        Ok(Self {
1274            gpu,
1275            surface,
1276            config,
1277            pipeline,
1278            globals_buf,
1279            bind_group,
1280            bind_layout,
1281            samplers,
1282            texture_binds: Default::default(),
1283            atlas_tex,
1284            atlas_size,
1285            atlas_epoch: u64::MAX,
1286            instance_buf,
1287            instance_cap,
1288            instances: Vec::new(),
1289            fragment_layouts,
1290            fragment_params_buf,
1291            fragment_params_cap,
1292            fragment_bind,
1293            fragment_bind_layout,
1294            uniform_align,
1295            fragment_bytes: Vec::new(),
1296            clear_color: wgpu::Color {
1297                r: 0.06,
1298                g: 0.065,
1299                b: 0.08,
1300                a: 1.0,
1301            },
1302            ground: None,
1303            backdrop_pipes: None,
1304            backdrop: backdrop::Targets::default(),
1305            blurs: Vec::new(),
1306        })
1307    }
1308
1309    /// Draws `rgba` (`width` by `height` pixels, four bytes each, row by
1310    /// row from the top left, the bytes the surface takes as they are) over
1311    /// the clear and under everything a frame draws, opaque, stretched
1312    /// across the viewport and sampled with linear filtering — so a small
1313    /// picture reads as a soft one. What a runner draws as a window's
1314    /// ground where the OS has no material to put behind it: the
1315    /// wallpaper, scaled down and blurred once (backlog F126). Which part
1316    /// of the picture shows is [`Renderer::set_ground_uv`]; all of it
1317    /// until that is called. A degenerate size, or pixels that are not
1318    /// that size, clear it.
1319    pub fn set_ground(&mut self, rgba: &[u8], width: u32, height: u32) {
1320        let max = self.gpu.device().limits().max_texture_dimension_2d;
1321        if width == 0
1322            || height == 0
1323            || width > max
1324            || height > max
1325            || rgba.len() != width as usize * height as usize * 4
1326        {
1327            self.ground = None;
1328            return;
1329        }
1330        let device = self.gpu.device();
1331        let size = wgpu::Extent3d {
1332            width,
1333            height,
1334            depth_or_array_layers: 1,
1335        };
1336        let texture = device.create_texture(&wgpu::TextureDescriptor {
1337            label: Some("kui.ground"),
1338            size,
1339            mip_level_count: 1,
1340            sample_count: 1,
1341            dimension: wgpu::TextureDimension::D2,
1342            format: wgpu::TextureFormat::Rgba8Unorm,
1343            usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1344            view_formats: &[],
1345        });
1346        self.gpu.queue().write_texture(
1347            wgpu::TexelCopyTextureInfo {
1348                texture: &texture,
1349                mip_level: 0,
1350                origin: wgpu::Origin3d::ZERO,
1351                aspect: wgpu::TextureAspect::All,
1352            },
1353            rgba,
1354            wgpu::TexelCopyBufferLayout {
1355                offset: 0,
1356                bytes_per_row: Some(width * 4),
1357                rows_per_image: Some(height),
1358            },
1359            size,
1360        );
1361        let uv = device.create_buffer(&wgpu::BufferDescriptor {
1362            label: Some("kui.ground.uv"),
1363            size: 16,
1364            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1365            mapped_at_creation: false,
1366        });
1367        self.gpu
1368            .queue()
1369            .write_buffer(&uv, 0, bytemuck::cast_slice(&[0.0f32, 0.0, 1.0, 1.0]));
1370        let layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
1371            label: Some("kui.ground"),
1372            entries: &[
1373                wgpu::BindGroupLayoutEntry {
1374                    binding: 0,
1375                    visibility: wgpu::ShaderStages::VERTEX,
1376                    ty: wgpu::BindingType::Buffer {
1377                        ty: wgpu::BufferBindingType::Uniform,
1378                        has_dynamic_offset: false,
1379                        min_binding_size: None,
1380                    },
1381                    count: None,
1382                },
1383                wgpu::BindGroupLayoutEntry {
1384                    binding: 1,
1385                    visibility: wgpu::ShaderStages::FRAGMENT,
1386                    ty: wgpu::BindingType::Texture {
1387                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
1388                        view_dimension: wgpu::TextureViewDimension::D2,
1389                        multisampled: false,
1390                    },
1391                    count: None,
1392                },
1393                wgpu::BindGroupLayoutEntry {
1394                    binding: 2,
1395                    visibility: wgpu::ShaderStages::FRAGMENT,
1396                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
1397                    count: None,
1398                },
1399            ],
1400        });
1401        let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
1402        let bind = device.create_bind_group(&wgpu::BindGroupDescriptor {
1403            label: Some("kui.ground"),
1404            layout: &layout,
1405            entries: &[
1406                wgpu::BindGroupEntry {
1407                    binding: 0,
1408                    resource: uv.as_entire_binding(),
1409                },
1410                wgpu::BindGroupEntry {
1411                    binding: 1,
1412                    resource: wgpu::BindingResource::TextureView(&view),
1413                },
1414                wgpu::BindGroupEntry {
1415                    binding: 2,
1416                    resource: wgpu::BindingResource::Sampler(&self.samplers.linear),
1417                },
1418            ],
1419        });
1420        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
1421            label: Some("kui.ground"),
1422            source: wgpu::ShaderSource::Wgsl(GROUND_SHADER.into()),
1423        });
1424        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
1425            label: Some("kui.ground"),
1426            bind_group_layouts: &[Some(&layout)],
1427            immediate_size: 0,
1428        });
1429        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
1430            label: Some("kui.ground"),
1431            layout: Some(&pipeline_layout),
1432            vertex: wgpu::VertexState {
1433                module: &module,
1434                entry_point: Some("vs"),
1435                compilation_options: Default::default(),
1436                buffers: &[],
1437            },
1438            fragment: Some(wgpu::FragmentState {
1439                module: &module,
1440                entry_point: Some("fs"),
1441                compilation_options: Default::default(),
1442                targets: &[Some(wgpu::ColorTargetState {
1443                    format: self.config.format,
1444                    blend: None,
1445                    write_mask: wgpu::ColorWrites::ALL,
1446                })],
1447            }),
1448            primitive: wgpu::PrimitiveState {
1449                topology: wgpu::PrimitiveTopology::TriangleStrip,
1450                ..Default::default()
1451            },
1452            depth_stencil: None,
1453            multisample: wgpu::MultisampleState::default(),
1454            multiview_mask: None,
1455            cache: None,
1456        });
1457        self.ground = Some(Ground {
1458            pipeline,
1459            bind,
1460            uv,
1461            _texture: texture,
1462        });
1463    }
1464
1465    /// Which part of the ground shows across the viewport: `[u0, v0, u1,
1466    /// v1]` in the picture's own 0..1 coordinates, the top-left corner
1467    /// first. Outside 0..1 the picture's edge is stretched. Nothing without
1468    /// a ground.
1469    pub fn set_ground_uv(&mut self, uv: [f32; 4]) {
1470        if let Some(g) = &self.ground {
1471            self.gpu
1472                .queue()
1473                .write_buffer(&g.uv, 0, bytemuck::cast_slice(&uv));
1474        }
1475    }
1476
1477    /// Takes the ground away ([`Renderer::set_ground`]).
1478    pub fn clear_ground(&mut self) {
1479        self.ground = None;
1480    }
1481
1482    /// Whether a ground is drawn under the frame.
1483    pub fn has_ground(&self) -> bool {
1484        self.ground.is_some()
1485    }
1486
1487    /// Whether this device can draw LCD subpixel glyphs
1488    /// (`QuadKind::GlyphSubpixel`) as intended.
1489    ///
1490    /// Pass it to `Core::set_subpixel_text` once after opening the
1491    /// renderer; when it is `false` the core keeps rasterizing grayscale
1492    /// masks, which every device blends correctly.
1493    pub fn subpixel_text(&self) -> bool {
1494        self.gpu.dual_source()
1495    }
1496
1497    /// Reconfigures the swapchain for a new window size, in physical pixels.
1498    ///
1499    /// The size is clamped to at least 1 (a minimized window reports zero,
1500    /// and a zero-sized surface is a validation error) and to the device's
1501    /// largest texture (Windows hands out nonsense sizes mid-resize, and
1502    /// configuring past the limit panics inside wgpu). A clamped frame is
1503    /// one wrong picture; the next real size fixes it.
1504    pub fn resize(&mut self, width: u32, height: u32) {
1505        let max = self.gpu.device().limits().max_texture_dimension_2d;
1506        self.config.width = width.clamp(1, max);
1507        self.config.height = height.clamp(1, max);
1508        self.surface.configure(self.gpu.device(), &self.config);
1509    }
1510
1511    fn sync_atlas(&mut self, atlas: &mut GlyphAtlas) {
1512        if atlas.size != self.atlas_size {
1513            self.atlas_size = atlas.size;
1514            self.atlas_tex = create_atlas_texture(self.gpu.device(), atlas.size);
1515            let view = self
1516                .atlas_tex
1517                .create_view(&wgpu::TextureViewDescriptor::default());
1518            self.bind_group = create_bind_group(
1519                self.gpu.device(),
1520                &self.bind_layout,
1521                &self.globals_buf,
1522                &view,
1523                &self.samplers,
1524            );
1525            self.atlas_epoch = u64::MAX;
1526        }
1527        if atlas.dirty || self.atlas_epoch != atlas.epoch {
1528            self.gpu.queue().write_texture(
1529                wgpu::TexelCopyTextureInfo {
1530                    texture: &self.atlas_tex,
1531                    mip_level: 0,
1532                    origin: wgpu::Origin3d::ZERO,
1533                    aspect: wgpu::TextureAspect::All,
1534                },
1535                &atlas.pixels,
1536                wgpu::TexelCopyBufferLayout {
1537                    offset: 0,
1538                    bytes_per_row: Some(atlas.size * 4),
1539                    rows_per_image: Some(atlas.size),
1540                },
1541                wgpu::Extent3d {
1542                    width: atlas.size,
1543                    height: atlas.size,
1544                    depth_or_array_layers: 1,
1545                },
1546            );
1547            atlas.dirty = false;
1548            self.atlas_epoch = atlas.epoch;
1549        }
1550    }
1551
1552    /// Plans the frame's backdrop blurs (backlog F129), and makes or drops
1553    /// what drawing them takes: the pipelines once, the offscreen frame at
1554    /// the surface's size, the scratch at what the largest blur needs, a
1555    /// parameter slot per blur. `any` is whether the list has a backdrop
1556    /// quad at all, noticed while the instances were written, so a frame
1557    /// without one scans nothing.
1558    ///
1559    /// A frame without a blur drops the frame and the scratch there and
1560    /// then (backlog RG150; [`backdrop::Targets::fit`] says why).
1561    fn plan_backdrops(&mut self, dl: &DisplayList, any: bool) {
1562        self.blurs.clear();
1563        let (w, h) = (self.config.width, self.config.height);
1564        if any {
1565            for (i, q) in dl.quads.iter().enumerate() {
1566                if q.kind == QuadKind::Backdrop
1567                    && let Some(b) = backdrop::plan(i as u32, q, dl.clip_of(q), w, h)
1568                {
1569                    self.blurs.push(b);
1570                }
1571            }
1572        }
1573        if self.blurs.is_empty() {
1574            self.backdrop = backdrop::Targets::default();
1575            return;
1576        }
1577        let device = self.gpu.device();
1578        let align = self.uniform_align;
1579        let format = self.config.format;
1580        let pipes = self
1581            .backdrop_pipes
1582            .get_or_insert_with(|| backdrop::Pipes::new(device, format, align));
1583        if self.blurs.len() > pipes.params_cap {
1584            pipes.params_cap = self.blurs.len().next_power_of_two();
1585            pipes.params = backdrop::params_buffer(device, pipes.params_cap, align);
1586            // Every bind group names the old buffer.
1587            self.backdrop = backdrop::Targets::default();
1588        }
1589        self.backdrop
1590            .fit(device, pipes, format, (w, h), &self.blurs);
1591        let Some((_, scratch)) = self.backdrop.get() else {
1592            return;
1593        };
1594        let slot = align as usize;
1595        let mut bytes = vec![0u8; self.blurs.len() * slot];
1596        for (i, b) in self.blurs.iter().enumerate() {
1597            bytes[i * slot..i * slot + std::mem::size_of::<backdrop::Params>()]
1598                .copy_from_slice(bytemuck::bytes_of(&scratch.params(b)));
1599        }
1600        self.gpu.queue().write_buffer(&pipes.params, 0, &bytes);
1601    }
1602
1603    /// Draws the instances in `range` into `pass`: in one instanced draw,
1604    /// or — where a fragment or a texture-backed image interrupts the run
1605    /// — the run before it with the über-pipeline, then that one quad with
1606    /// its own pipeline (a fragment) or its own group 0 (a texture), then
1607    /// on. Consecutive quads of the same handle still take one set each
1608    /// (about 0.6 us); runs of ordinary quads are unbroken.
1609    fn draw_quads(
1610        &self,
1611        pass: &mut wgpu::RenderPass<'_>,
1612        dl: &DisplayList,
1613        range: std::ops::Range<u32>,
1614        fragment_pipelines: &[wgpu::RenderPipeline],
1615        texture_binds: &[Option<u64>],
1616    ) {
1617        if range.is_empty() {
1618            return;
1619        }
1620        pass.set_vertex_buffer(0, self.instance_buf.slice(..));
1621        if fragment_pipelines.is_empty() && texture_binds.is_empty() {
1622            // The whole run in one instanced draw, as a frame has always
1623            // been. Nothing below runs.
1624            pass.set_pipeline(&self.pipeline);
1625            pass.set_bind_group(0, &self.bind_group, &[]);
1626            pass.draw(0..6, range);
1627            return;
1628        }
1629        let mut run_start = range.start;
1630        let mut on_quads = false;
1631        for i in range.clone() {
1632            let q = &dl.quads[i as usize];
1633            if q.kind != QuadKind::Fragment && q.kind != QuadKind::Texture {
1634                continue;
1635            }
1636            if i > run_start {
1637                if !on_quads {
1638                    pass.set_pipeline(&self.pipeline);
1639                    pass.set_bind_group(0, &self.bind_group, &[]);
1640                    on_quads = true;
1641                }
1642                pass.draw(0..6, run_start..i);
1643            }
1644            // `uv[0]` is the index into the side list, which is also this
1645            // fragment's parameter slot, or this texture's bind.
1646            let slot = q.uv[0] as usize;
1647            if q.kind == QuadKind::Texture {
1648                if let Some(Some(id)) = texture_binds.get(slot)
1649                    && let Some(b) = self.texture_binds.get(id)
1650                {
1651                    pass.set_pipeline(&self.pipeline);
1652                    pass.set_bind_group(0, &b.bind, &[]);
1653                    on_quads = false;
1654                    pass.draw(0..6, i..i + 1);
1655                }
1656            } else if let Some(pipeline) = fragment_pipelines.get(slot) {
1657                // A fragment reading a texture-backed image takes that
1658                // image's group 0 — the texture in the atlas's place,
1659                // `atlas_size` its size — exactly as a texture quad does;
1660                // one reading the atlas, or nothing, takes the frame's. A
1661                // texture the device could not make (a degenerate or
1662                // oversized image) draws the fragment against the atlas
1663                // with a zero rect, which `kui_sample` reads as no image.
1664                let group0 = match draw_image_texture(&dl.fragments[slot]) {
1665                    Some(index) => texture_binds
1666                        .get(index)
1667                        .copied()
1668                        .flatten()
1669                        .and_then(|id| self.texture_binds.get(&id))
1670                        .map_or(&self.bind_group, |b| &b.bind),
1671                    None => &self.bind_group,
1672                };
1673                pass.set_pipeline(pipeline);
1674                pass.set_bind_group(0, group0, &[]);
1675                pass.set_bind_group(1, &self.fragment_bind, &[slot as u32 * self.uniform_align]);
1676                on_quads = false;
1677                pass.draw(0..6, i..i + 1);
1678            }
1679            run_start = i + 1;
1680        }
1681        if range.end > run_start {
1682            if !on_quads {
1683                pass.set_pipeline(&self.pipeline);
1684                pass.set_bind_group(0, &self.bind_group, &[]);
1685            }
1686            pass.draw(0..6, run_start..range.end);
1687        }
1688    }
1689
1690    /// Draws one frame and presents it.
1691    ///
1692    /// Uploads the atlas when it changed since the last frame (and clears
1693    /// its `dirty` flag), writes the list's quads to the instance buffer,
1694    /// draws them over [`Renderer::clear_color`] in one render pass, and
1695    /// presents. Acquiring the swapchain image blocks while vsync holds
1696    /// the frame back; that time comes back as
1697    /// [`RenderReport::vsync_wait_ms`] so a runner can tell pacing from
1698    /// work.
1699    ///
1700    /// Fails without presenting when the surface or the device cannot
1701    /// take the frame; each [`RenderError`] says what to do next.
1702    pub fn render(
1703        &mut self,
1704        dl: &DisplayList,
1705        atlas: &mut GlyphAtlas,
1706    ) -> Result<RenderReport, RenderError> {
1707        // A dead device takes no work: everything below would only add
1708        // errors to the one that lost it.
1709        if self.gpu.lost() {
1710            return Err(RenderError::DeviceLost);
1711        }
1712        self.sync_atlas(atlas);
1713
1714        self.instances.clear();
1715        let mut any_backdrop = false;
1716        self.instances.extend(dl.quads.iter().map(|q| {
1717            any_backdrop |= q.kind == QuadKind::Backdrop;
1718            instance_of(q, &dl.clips, &dl.textures)
1719        }));
1720        self.plan_backdrops(dl, any_backdrop);
1721        if self.instances.len() > self.instance_cap {
1722            self.instance_cap = self.instances.len().next_power_of_two();
1723            self.instance_buf = create_instance_buffer(self.gpu.device(), self.instance_cap);
1724        }
1725        if !self.instances.is_empty() {
1726            self.gpu.queue().write_buffer(
1727                &self.instance_buf,
1728                0,
1729                bytemuck::cast_slice(&self.instances),
1730            );
1731        }
1732        let globals = Globals {
1733            viewport: [dl.viewport.w.max(1.0), dl.viewport.h.max(1.0)],
1734            atlas_size: [self.atlas_size as f32, self.atlas_size as f32],
1735            time: dl.time,
1736            scale: dl.scale,
1737            _pad: [0.0; 2],
1738        };
1739        self.gpu
1740            .queue()
1741            .write_buffer(&self.globals_buf, 0, bytemuck::bytes_of(&globals));
1742
1743        // Texture-backed images: drop what the core
1744        // removed, upload what moved, and give each one drawn this frame
1745        // a group-0 bind group of its own with a globals copy whose
1746        // `atlas_size` is the texture's. All skipped on a frame that
1747        // draws none.
1748        for id in &dl.dropped_textures {
1749            self.texture_binds.remove(&id.to_ffi());
1750            self.gpu.drop_image_texture(id.to_ffi());
1751        }
1752        // And the pipelines of removed fragments — built per handle
1753        // and shared by every window, so one window's list carries the
1754        // removal and this is the only eviction they get.
1755        for id in &dl.dropped_fragments {
1756            self.gpu.drop_fragment_pipelines(id.to_ffi());
1757        }
1758        // A drop another window's frame carried: the cache no longer
1759        // holds the texture this bind group does. One lock per frame,
1760        // and only for a window that has ever drawn a texture.
1761        if !self.texture_binds.is_empty() {
1762            let gpu = &self.gpu;
1763            self.texture_binds
1764                .retain(|id, b| gpu.holds_image_texture(*id, &b.texture));
1765        }
1766        let mut texture_binds: Vec<Option<u64>> = Vec::new();
1767        if !dl.textures.is_empty() {
1768            texture_binds.reserve(dl.textures.len());
1769            for (draw, px) in dl.textures.iter().zip(&dl.texture_pixels) {
1770                let id = draw.id.to_ffi();
1771                let Some(texture) = self.gpu.image_texture(id, px) else {
1772                    texture_binds.push(None);
1773                    continue;
1774                };
1775                let stale = self
1776                    .texture_binds
1777                    .get(&id)
1778                    .is_none_or(|b| !std::sync::Arc::ptr_eq(&b.texture, &texture));
1779                if stale {
1780                    let device = self.gpu.device();
1781                    let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1782                        label: Some("kui.image.globals"),
1783                        size: std::mem::size_of::<Globals>() as u64,
1784                        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1785                        mapped_at_creation: false,
1786                    });
1787                    let bind = create_bind_group(
1788                        device,
1789                        &self.bind_layout,
1790                        &globals_buf,
1791                        &texture.view,
1792                        &self.samplers,
1793                    );
1794                    self.texture_binds.insert(
1795                        id,
1796                        TextureBind {
1797                            texture: texture.clone(),
1798                            globals: globals_buf,
1799                            bind,
1800                        },
1801                    );
1802                }
1803                let b = &self.texture_binds[&id];
1804                let mine = Globals {
1805                    atlas_size: [texture.width as f32, texture.height as f32],
1806                    ..globals
1807                };
1808                self.gpu
1809                    .queue()
1810                    .write_buffer(&b.globals, 0, bytemuck::bytes_of(&mine));
1811                texture_binds.push(Some(id));
1812            }
1813        }
1814
1815        // Each fragment's parameters into its own slot, and its pipeline
1816        // built if this device has not seen the handle before. Both are
1817        // skipped whole on a frame that draws no fragment.
1818        let mut fragment_pipelines: Vec<wgpu::RenderPipeline> = Vec::new();
1819        if !dl.fragments.is_empty() {
1820            let align = self.uniform_align as usize;
1821            if dl.fragments.len() > self.fragment_params_cap {
1822                self.fragment_params_cap = dl.fragments.len().next_power_of_two();
1823                self.fragment_params_buf = create_fragment_params_buffer(
1824                    self.gpu.device(),
1825                    self.fragment_params_cap,
1826                    self.uniform_align,
1827                );
1828                self.fragment_bind = create_fragment_bind_group(
1829                    self.gpu.device(),
1830                    &self.fragment_bind_layout,
1831                    &self.fragment_params_buf,
1832                );
1833            }
1834            self.fragment_bytes.clear();
1835            self.fragment_bytes.resize(dl.fragments.len() * align, 0);
1836            for (i, draw) in dl.fragments.iter().enumerate() {
1837                let uv = draw.image.uv();
1838                let slot = FragmentParams {
1839                    params: draw.params,
1840                    image: [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32],
1841                };
1842                let at = i * align;
1843                self.fragment_bytes[at..at + std::mem::size_of::<FragmentParams>()]
1844                    .copy_from_slice(bytemuck::bytes_of(&slot));
1845            }
1846            self.gpu
1847                .queue()
1848                .write_buffer(&self.fragment_params_buf, 0, &self.fragment_bytes);
1849            fragment_pipelines.reserve(dl.fragments.len());
1850            for (draw, source) in dl.fragments.iter().zip(&dl.fragment_sources) {
1851                fragment_pipelines.push(self.gpu.fragment_pipeline(
1852                    draw.id.to_ffi(),
1853                    source,
1854                    self.config.format,
1855                    &self.fragment_layouts,
1856                ));
1857            }
1858        }
1859
1860        // Acquiring the swapchain image is where vsync backpressure blocks;
1861        // report it separately so latency graphs show pacing vs work.
1862        let t_wait = std::time::Instant::now();
1863        let frame = match self.surface.get_current_texture() {
1864            wgpu::CurrentSurfaceTexture::Success(f)
1865            | wgpu::CurrentSurfaceTexture::Suboptimal(f) => f,
1866            wgpu::CurrentSurfaceTexture::Timeout | wgpu::CurrentSurfaceTexture::Occluded => {
1867                return Err(RenderError::Skip);
1868            }
1869            wgpu::CurrentSurfaceTexture::Outdated | wgpu::CurrentSurfaceTexture::Lost => {
1870                return Err(RenderError::Reconfigure);
1871            }
1872            // The acquire's error went to the device's error handler; if
1873            // it was the device itself, the lost callback has run by now.
1874            wgpu::CurrentSurfaceTexture::Validation => {
1875                return Err(if self.gpu.lost() {
1876                    RenderError::DeviceLost
1877                } else {
1878                    RenderError::Validation
1879                });
1880            }
1881        };
1882        let vsync_wait_ms = t_wait.elapsed().as_secs_f32() * 1e3;
1883        let surface_view = frame
1884            .texture
1885            .create_view(&wgpu::TextureViewDescriptor::default());
1886        let mut encoder = self
1887            .gpu
1888            .device()
1889            .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("kui") });
1890        // A frame that blurs draws into the offscreen copy, so what it has
1891        // drawn can be read back, and breaks its pass at each blur
1892        // (backlog F129); one that does not draws to the surface in one
1893        // pass, as it always has.
1894        let blurring = !self.blurs.is_empty();
1895        let offscreen = match (&self.backdrop_pipes, self.backdrop.get(), blurring) {
1896            (Some(p), Some((f, s)), true) => Some((p, f, s)),
1897            _ => None,
1898        };
1899        let target = offscreen.map_or(&surface_view, |(_, f, _)| &f.view);
1900        let end = self.instances.len() as u32;
1901        let mut start = 0u32;
1902        let mut next = 0usize;
1903        loop {
1904            let stop = self.blurs.get(next).map_or(end, |b| b.quad);
1905            {
1906                let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1907                    label: Some("kui"),
1908                    color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1909                        view: target,
1910                        depth_slice: None,
1911                        resolve_target: None,
1912                        ops: wgpu::Operations {
1913                            load: if next == 0 {
1914                                wgpu::LoadOp::Clear(self.clear_color)
1915                            } else {
1916                                wgpu::LoadOp::Load
1917                            },
1918                            store: wgpu::StoreOp::Store,
1919                        },
1920                    })],
1921                    depth_stencil_attachment: None,
1922                    timestamp_writes: None,
1923                    occlusion_query_set: None,
1924                    multiview_mask: None,
1925                });
1926                // The ground first, opaque over the clear, so everything
1927                // the frame paints with alpha blends over it.
1928                if next == 0
1929                    && let Some(g) = &self.ground
1930                {
1931                    pass.set_pipeline(&g.pipeline);
1932                    pass.set_bind_group(0, &g.bind, &[]);
1933                    pass.draw(0..4, 0..1);
1934                }
1935                // The blur the last pass stopped for, written back before
1936                // anything after its quad draws over it.
1937                if let Some(((pipes, frame, scratch), b)) =
1938                    offscreen.zip(next.checked_sub(1).and_then(|i| self.blurs.get(i)))
1939                {
1940                    let slot = (next - 1) as u32 * self.uniform_align;
1941                    backdrop::composite(&mut pass, pipes, scratch, b, slot);
1942                    pass.set_scissor_rect(0, 0, frame.size.0, frame.size.1);
1943                }
1944                self.draw_quads(
1945                    &mut pass,
1946                    dl,
1947                    start..stop,
1948                    &fragment_pipelines,
1949                    &texture_binds,
1950                );
1951            }
1952            let (Some(b), Some((pipes, frame, scratch))) = (self.blurs.get(next), offscreen) else {
1953                break;
1954            };
1955            let slot = next as u32 * self.uniform_align;
1956            backdrop::record(&mut encoder, pipes, frame, scratch, b, slot);
1957            start = b.quad + 1;
1958            next += 1;
1959        }
1960        if let Some((pipes, frame, _)) = offscreen {
1961            let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1962                label: Some("kui.backdrop.blit"),
1963                color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1964                    view: &surface_view,
1965                    depth_slice: None,
1966                    resolve_target: None,
1967                    ops: wgpu::Operations {
1968                        load: wgpu::LoadOp::Clear(wgpu::Color::TRANSPARENT),
1969                        store: wgpu::StoreOp::Store,
1970                    },
1971                })],
1972                depth_stencil_attachment: None,
1973                timestamp_writes: None,
1974                occlusion_query_set: None,
1975                multiview_mask: None,
1976            });
1977            pass.set_pipeline(&pipes.blit);
1978            pass.set_bind_group(0, &frame.blit, &[0]);
1979            pass.draw(0..3, 0..1);
1980        }
1981        self.gpu.queue().submit([encoder.finish()]);
1982        self.gpu.queue().present(frame);
1983        Ok(RenderReport { vsync_wait_ms })
1984    }
1985}
1986
1987/// The `textures` entry a fragment draw reads its image from, if its
1988/// image has a texture of its own.
1989fn draw_image_texture(draw: &kui_core::FragmentDraw) -> Option<usize> {
1990    match draw.image {
1991        kui_core::FragmentImage::Texture { index, .. } => Some(index as usize),
1992        _ => None,
1993    }
1994}
1995
1996/// Timing details from one [`Renderer::render`] call.
1997#[derive(Clone, Copy, Debug, Default)]
1998pub struct RenderReport {
1999    /// Milliseconds spent blocked acquiring the swapchain image, which is
2000    /// where vsync backpressure shows up.
2001    pub vsync_wait_ms: f32,
2002}
2003
2004/// A frame that produced no image, and what to do about it.
2005///
2006/// Mapped from wgpu's `CurrentSurfaceTexture`; the crate root's example
2007/// handles every variant.
2008#[derive(Clone, Copy, Debug)]
2009pub enum RenderError {
2010    /// The surface is outdated or lost: call [`Renderer::resize`] with the
2011    /// window's size and draw again.
2012    Reconfigure,
2013    /// Nothing can be presented right now (the window is occluded, or the
2014    /// acquire timed out): try again next frame.
2015    Skip,
2016    /// The surface is configured wrong for the window: call
2017    /// [`Renderer::resize`] with the window's size and draw again. A
2018    /// surface that stays wrong is best given up with its device.
2019    Validation,
2020    /// The device is gone ([`Gpu::lost`]): open a new one, and a renderer
2021    /// on it for every window.
2022    DeviceLost,
2023}
2024
2025impl std::fmt::Display for RenderError {
2026    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
2027        match self {
2028            Self::Reconfigure => write!(f, "surface outdated or lost; reconfigure"),
2029            Self::Skip => write!(f, "no frame available; skip"),
2030            Self::Validation => write!(f, "surface texture validation error"),
2031            Self::DeviceLost => write!(f, "device lost; reopen"),
2032        }
2033    }
2034}
2035
2036impl std::error::Error for RenderError {}
2037
2038fn create_atlas_texture(device: &wgpu::Device, size: u32) -> wgpu::Texture {
2039    device.create_texture(&wgpu::TextureDescriptor {
2040        label: Some("kui.atlas"),
2041        size: wgpu::Extent3d {
2042            width: size,
2043            height: size,
2044            depth_or_array_layers: 1,
2045        },
2046        mip_level_count: 1,
2047        sample_count: 1,
2048        dimension: wgpu::TextureDimension::D2,
2049        format: wgpu::TextureFormat::Rgba8Unorm,
2050        usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
2051        view_formats: &[],
2052    })
2053}
2054
2055/// Group 0: the globals, a texture — the atlas, or a texture-backed image
2056/// in its place — and the two samplers.
2057fn create_bind_group(
2058    device: &wgpu::Device,
2059    layout: &wgpu::BindGroupLayout,
2060    globals: &wgpu::Buffer,
2061    view: &wgpu::TextureView,
2062    samplers: &Samplers,
2063) -> wgpu::BindGroup {
2064    device.create_bind_group(&wgpu::BindGroupDescriptor {
2065        label: Some("kui"),
2066        layout,
2067        entries: &[
2068            wgpu::BindGroupEntry {
2069                binding: 0,
2070                resource: globals.as_entire_binding(),
2071            },
2072            wgpu::BindGroupEntry {
2073                binding: 1,
2074                resource: wgpu::BindingResource::TextureView(view),
2075            },
2076            wgpu::BindGroupEntry {
2077                binding: 2,
2078                resource: wgpu::BindingResource::Sampler(&samplers.linear),
2079            },
2080            wgpu::BindGroupEntry {
2081                binding: 3,
2082                resource: wgpu::BindingResource::Sampler(&samplers.nearest),
2083            },
2084        ],
2085    })
2086}
2087
2088fn create_instance_buffer(device: &wgpu::Device, cap: usize) -> wgpu::Buffer {
2089    device.create_buffer(&wgpu::BufferDescriptor {
2090        label: Some("kui.instances"),
2091        size: (cap * std::mem::size_of::<Instance>()) as u64,
2092        usage: wgpu::BufferUsages::VERTEX | wgpu::BufferUsages::COPY_DST,
2093        mapped_at_creation: false,
2094    })
2095}
2096
2097fn create_fragment_params_buffer(device: &wgpu::Device, cap: usize, align: u32) -> wgpu::Buffer {
2098    device.create_buffer(&wgpu::BufferDescriptor {
2099        label: Some("kui.fragment.params"),
2100        size: (cap.max(1) * align as usize) as u64,
2101        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
2102        mapped_at_creation: false,
2103    })
2104}
2105
2106fn create_fragment_bind_group(
2107    device: &wgpu::Device,
2108    layout: &wgpu::BindGroupLayout,
2109    buf: &wgpu::Buffer,
2110) -> wgpu::BindGroup {
2111    device.create_bind_group(&wgpu::BindGroupDescriptor {
2112        label: Some("kui.fragment.params"),
2113        layout,
2114        entries: &[wgpu::BindGroupEntry {
2115            binding: 0,
2116            resource: wgpu::BindingResource::Buffer(wgpu::BufferBinding {
2117                buffer: buf,
2118                offset: 0,
2119                size: std::num::NonZeroU64::new(std::mem::size_of::<FragmentParams>() as u64),
2120            }),
2121        }],
2122    })
2123}
2124
2125/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2126///
2127/// A fault in a GPU driver, such as one being replaced under the app, ends
2128/// the process with no line from anyone: it is not a panic, and Windows
2129/// reports only `0xC000041D` for an exception in a window callback. This
2130/// installs an unhandled-exception filter that names the exception code
2131/// and the module the faulting address is in, then lets the crash go on;
2132/// it is a diagnostic, not a recovery. A crash reporter the host installed
2133/// first is still called, with its answer returned; one installed after
2134/// replaces this filter.
2135///
2136/// [`Gpu::new`] calls it, so a runner rarely needs to. Installing it more
2137/// than once is harmless.
2138#[cfg(windows)]
2139pub fn report_faults() {
2140    use std::cell::Cell;
2141    use windows::Win32::Foundation::{
2142        EXCEPTION_ACCESS_VIOLATION, EXCEPTION_ILLEGAL_INSTRUCTION, EXCEPTION_IN_PAGE_ERROR,
2143        EXCEPTION_STACK_OVERFLOW, HMODULE, NTSTATUS, STATUS_FATAL_USER_CALLBACK_EXCEPTION,
2144    };
2145    use windows::Win32::Storage::FileSystem::WriteFile;
2146    use windows::Win32::System::Console::{GetStdHandle, STD_ERROR_HANDLE};
2147    use windows::Win32::System::Diagnostics::Debug::{
2148        AddVectoredExceptionHandler, EXCEPTION_POINTERS, EXCEPTION_RECORD,
2149        LPTOP_LEVEL_EXCEPTION_FILTER, SetUnhandledExceptionFilter,
2150    };
2151    use windows::Win32::System::LibraryLoader::{
2152        GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS, GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2153        GetModuleFileNameW, GetModuleHandleExW,
2154    };
2155
2156    const CONTINUE_SEARCH: i32 = 0;
2157    /// The faults a driver ends a process with: the ones remembered for
2158    /// a callback's `0xC000041D` to be read by.
2159    const FAULTS: [NTSTATUS; 4] = [
2160        EXCEPTION_ACCESS_VIOLATION,
2161        EXCEPTION_ILLEGAL_INSTRUCTION,
2162        EXCEPTION_IN_PAGE_ERROR,
2163        EXCEPTION_STACK_OVERFLOW,
2164    ];
2165    /// The filter this one replaced, called after it; set once, with the
2166    /// two handlers, by the one call that installs them.
2167    static PREVIOUS: std::sync::OnceLock<LPTOP_LEVEL_EXCEPTION_FILTER> = std::sync::OnceLock::new();
2168    thread_local! {
2169        /// The last fault this thread saw, code and address, handled or
2170        /// not. A `const` cell with no destructor: a plain thread-local
2171        /// slot, read and written without allocating or registering
2172        /// anything, from inside an exception.
2173        static LAST: Cell<Option<(i32, usize)>> = const { Cell::new(None) };
2174    }
2175
2176    /// Remembers a fault; says nothing and handles nothing.
2177    unsafe extern "system" fn remember(info: *mut EXCEPTION_POINTERS) -> i32 {
2178        // SAFETY: the system hands a valid record for the exception.
2179        if let Some(record) =
2180            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2181            && FAULTS.contains(&record.ExceptionCode)
2182        {
2183            let seen = (record.ExceptionCode.0, record.ExceptionAddress as usize);
2184            let _ = LAST.try_with(|l| l.set(Some(seen)));
2185        }
2186        CONTINUE_SEARCH
2187    }
2188
2189    /// The first fault on a record's chain of nested exceptions, past the
2190    /// record itself; a few links, since a chain is one or two long and a
2191    /// broken one is not worth following further.
2192    fn nested(record: &EXCEPTION_RECORD) -> Option<(i32, usize)> {
2193        let mut at = record.ExceptionRecord;
2194        for _ in 0..4 {
2195            // SAFETY: a nested record the system chained to this one.
2196            let inner = unsafe { at.as_ref() }?;
2197            if FAULTS.contains(&inner.ExceptionCode) {
2198                return Some((inner.ExceptionCode.0, inner.ExceptionAddress as usize));
2199            }
2200            at = inner.ExceptionRecord;
2201        }
2202        None
2203    }
2204
2205    /// Writes one crash's line to stderr, straight to the handle: no
2206    /// `eprintln!`, which takes a lock and, on a console, converts
2207    /// through a stack buffer eight kilobytes deep — more than a stack
2208    /// overflow leaves.
2209    fn say(code: i32, at: usize, escaped: Option<i32>) {
2210        let mut module = HMODULE::default();
2211        let mut name = [0u16; 260];
2212        // SAFETY: `at` is only looked up, never read; the buffers are ours.
2213        let found = unsafe {
2214            GetModuleHandleExW(
2215                GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS
2216                    | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
2217                windows::core::PCWSTR(at as *const u16),
2218                &mut module,
2219            )
2220        }
2221        .is_ok();
2222        let path = found.then(|| {
2223            // SAFETY: the module was just found; the buffer is ours.
2224            let n = unsafe { GetModuleFileNameW(Some(module), &mut name) } as usize;
2225            &name[..n.min(name.len())]
2226        });
2227        let line = FaultLine::new(code as u32, at, path, escaped.map(|c| c as u32));
2228        // SAFETY: a handle the process was given, written from our buffer.
2229        if let Ok(err) = unsafe { GetStdHandle(STD_ERROR_HANDLE) } {
2230            let mut written = 0u32;
2231            let _ = unsafe { WriteFile(err, Some(line.bytes()), Some(&mut written), None) };
2232        }
2233    }
2234
2235    /// The crash: said, then handed to the filter before this one.
2236    unsafe extern "system" fn filter(info: *const EXCEPTION_POINTERS) -> i32 {
2237        // SAFETY: the system hands a valid record for the exception.
2238        if let Some(record) =
2239            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
2240        {
2241            let (code, at) = (record.ExceptionCode, record.ExceptionAddress as usize);
2242            let inner = if code == STATUS_FATAL_USER_CALLBACK_EXCEPTION {
2243                nested(record).or_else(|| LAST.try_with(Cell::get).ok().flatten())
2244            } else {
2245                None
2246            };
2247            match inner {
2248                Some((fault, fault_at)) => say(fault, fault_at, Some(code.0)),
2249                None => say(code.0, at, None),
2250            }
2251        }
2252        match PREVIOUS.get().copied().flatten() {
2253            // SAFETY: the filter the system held before ours, called as
2254            // the system would have called it.
2255            Some(previous) => unsafe { previous(info) },
2256            None => CONTINUE_SEARCH,
2257        }
2258    }
2259
2260    PREVIOUS.get_or_init(|| {
2261        // SAFETY: both handlers read only what the system gives them and
2262        // write only their own thread-local slot and stderr.
2263        unsafe {
2264            AddVectoredExceptionHandler(0, Some(remember));
2265            SetUnhandledExceptionFilter(Some(filter))
2266        }
2267    });
2268}
2269
2270/// Prints the code and module of a crash to stderr before the process dies (Windows only).
2271///
2272/// On Windows a fault in a GPU driver ends the process with no line from
2273/// anyone, and this installs the exception filter that names it. On every
2274/// other platform it does nothing. [`Gpu::new`] calls it, so a runner
2275/// rarely needs to.
2276#[cfg(not(windows))]
2277pub fn report_faults() {}
2278
2279/// One crash's line for `report_faults`, written into a buffer on the
2280/// stack: it is said with whatever stack the crash left (a stack
2281/// overflow leaves the few pages the thread reserved for its handlers)
2282/// and in a process whose heap may be what faulted, so nothing here
2283/// allocates. A line too long for it is cut, at a character, and still
2284/// ends in a newline. Built on every platform so it is tested on every
2285/// platform; only Windows says one.
2286#[cfg_attr(not(windows), allow(dead_code))]
2287struct FaultLine {
2288    buf: [u8; 640],
2289    len: usize,
2290}
2291
2292#[cfg_attr(not(windows), allow(dead_code))]
2293impl FaultLine {
2294    /// `kui: fault <code> at <address> in <module>`, and for a fault that
2295    /// escaped a window callback, the code it escaped as. `module` is the
2296    /// UTF-16 path Windows gives; none is a fault outside any module.
2297    fn new(code: u32, at: usize, module: Option<&[u16]>, escaped: Option<u32>) -> Self {
2298        use std::fmt::Write;
2299        let mut line = Self {
2300            buf: [0; 640],
2301            len: 0,
2302        };
2303        let _ = write!(line, "kui: fault {code:#010x} at {at:#x} in ");
2304        match module {
2305            Some(path) => {
2306                for c in char::decode_utf16(path.iter().copied()) {
2307                    let _ = line.write_char(c.unwrap_or(char::REPLACEMENT_CHARACTER));
2308                }
2309            }
2310            None => {
2311                let _ = line.write_str("no module (jit or freed code)");
2312            }
2313        }
2314        if let Some(escaped) = escaped {
2315            let _ = write!(line, ", escaped from a window callback as {escaped:#010x}");
2316        }
2317        // The newline has its byte kept for it (`write_str`).
2318        line.buf[line.len] = b'\n';
2319        line.len += 1;
2320        line
2321    }
2322
2323    fn bytes(&self) -> &[u8] {
2324        &self.buf[..self.len]
2325    }
2326}
2327
2328impl std::fmt::Write for FaultLine {
2329    fn write_str(&mut self, s: &str) -> std::fmt::Result {
2330        // One byte short of the buffer, for the newline.
2331        let room = self.buf.len() - 1 - self.len;
2332        let mut n = s.len().min(room);
2333        while !s.is_char_boundary(n) {
2334            n -= 1;
2335        }
2336        self.buf[self.len..self.len + n].copy_from_slice(&s.as_bytes()[..n]);
2337        self.len += n;
2338        Ok(())
2339    }
2340}
2341
2342#[cfg(test)]
2343mod tests {
2344    use super::*;
2345
2346    /// An opaque window keeps the mode it always had, `Opaque` — also on
2347    /// a composition swapchain, which lists `Auto` first and would have
2348    /// presented `DXGI_ALPHA_MODE_UNSPECIFIED` (backlog F126).
2349    #[test]
2350    fn an_opaque_window_presents_opaque_wherever_it_can() {
2351        use wgpu::CompositeAlphaMode as M;
2352        assert_eq!(opaque_mode(&[M::Opaque]), M::Opaque);
2353        assert_eq!(opaque_mode(&[M::Opaque, M::PostMultiplied]), M::Opaque);
2354        assert_eq!(
2355            opaque_mode(&[
2356                M::Auto,
2357                M::Inherit,
2358                M::Opaque,
2359                M::PostMultiplied,
2360                M::PreMultiplied
2361            ]),
2362            M::Opaque
2363        );
2364        assert_eq!(opaque_mode(&[M::Inherit]), M::Inherit);
2365    }
2366
2367    /// One answer for the window's style and the swapchain (backlog
2368    /// RG150): D3D12 through a visual unless the backend list leaves
2369    /// D3D12 out or the presentation system names the handle — each
2370    /// variable read as wgpu reads it.
2371    #[test]
2372    fn the_window_and_the_swapchain_are_decided_together() {
2373        let by = see_through_by_visual_with;
2374        // Nothing set: kui opens D3D12 alone, through the visual.
2375        assert!(by(None, None));
2376        // `WGPU_BACKEND` naming D3D12, alone or in a list, in either
2377        // spelling and any case, spaces stripped.
2378        for b in ["dx12", "D3D12", "DX12", "vulkan,dx12", " gl , d3d12 "] {
2379            assert!(by(Some(b), None), "{b}");
2380        }
2381        // Naming only others, or nothing wgpu knows.
2382        for b in ["vulkan", "gl", "vk,gl", "", "directx", "dx11"] {
2383            assert!(!by(Some(b), None), "{b}");
2384        }
2385        // The presentation system: the handle's swapchain takes no alpha;
2386        // the visual's, or a value wgpu does not read, leave kui's choice.
2387        for p in ["hwnd", "Hwnd", "DxgiFromHwnd", "dxgifromhwnd"] {
2388            assert!(!by(None, Some(p)), "{p}");
2389            assert!(!by(Some("dx12"), Some(p)), "{p}");
2390        }
2391        for p in ["visual", "DxgiFromVisual", "", "nonsense", " hwnd"] {
2392            assert!(by(None, Some(p)), "{p}");
2393        }
2394        // Both must allow it.
2395        assert!(!by(Some("vulkan"), Some("visual")));
2396    }
2397
2398    /// A transparent one takes a mode that composites premultiplied
2399    /// pixels, and none where only straight alpha or opaque is offered.
2400    #[test]
2401    fn a_transparent_window_presents_premultiplied_or_not_at_all() {
2402        use wgpu::{Backend, CompositeAlphaMode as M};
2403        // D3D12 through a composition visual.
2404        let visual = [
2405            M::Auto,
2406            M::Inherit,
2407            M::Opaque,
2408            M::PostMultiplied,
2409            M::PreMultiplied,
2410        ];
2411        assert_eq!(
2412            transparent_mode(&visual, Backend::Dx12),
2413            Some(M::PreMultiplied)
2414        );
2415        // D3D12 on the window's handle: opaque only.
2416        assert_eq!(transparent_mode(&[M::Opaque], Backend::Dx12), None);
2417        // Metal names its non-opaque layer post-multiplied.
2418        let metal = [M::Opaque, M::PostMultiplied];
2419        assert_eq!(
2420            transparent_mode(&metal, Backend::Metal),
2421            Some(M::PostMultiplied)
2422        );
2423        // Vulkan's post-multiplied is straight alpha: not that.
2424        assert_eq!(
2425            transparent_mode(&[M::Opaque, M::PostMultiplied], Backend::Vulkan),
2426            None
2427        );
2428        assert_eq!(
2429            transparent_mode(&[M::Opaque, M::Inherit], Backend::Vulkan),
2430            Some(M::Inherit)
2431        );
2432    }
2433
2434    /// The globals are one buffer read by two pipelines whose modules
2435    /// declare it separately: this crate's `shader.wgsl` for quads, and
2436    /// `kui_core::fragment::PRELUDE` for every fragment. If the two
2437    /// declarations drift, a fragment reads the wrong bytes and there is
2438    /// nothing to catch it at runtime — the buffer is the right size and
2439    /// the numbers are just wrong. So: same field names, same order, and
2440    /// the size the Rust struct actually is.
2441    #[test]
2442    fn globals_layout_matches() {
2443        let fields = ["viewport", "atlas_size", "time", "scale", "_pad"];
2444        let of = |src: &str, name: &str| {
2445            let start = src
2446                .find(name)
2447                .unwrap_or_else(|| panic!("{name} is not declared in\n{src}"));
2448            let body = &src[start..];
2449            let end = body.find('}').expect("a closing brace");
2450            body[..end].to_string()
2451        };
2452        let quads = of(include_str!("shader.wgsl"), "struct Globals {");
2453        let frags = of(kui_core::fragment::PRELUDE, "struct KuiGlobals {");
2454        let read = |body: &str| -> Vec<String> {
2455            body.lines()
2456                .filter_map(|l| l.split_once(':'))
2457                .map(|(name, ty)| format!("{}: {}", name.trim(), ty.trim().trim_end_matches(',')))
2458                .collect()
2459        };
2460        let (a, b) = (read(&quads), read(&frags));
2461        assert_eq!(a, b, "shader.wgsl and the fragment prelude disagree");
2462        assert_eq!(
2463            a.len(),
2464            fields.len(),
2465            "a field was added to the globals without this test being told"
2466        );
2467        for (row, want) in a.iter().zip(fields) {
2468            assert!(row.starts_with(want), "expected {want}, got {row}");
2469        }
2470        // vec2 + vec2 + f32 + f32 + vec2 = 32 bytes, and a uniform's size
2471        // must be a multiple of sixteen, which is what `_pad` is for.
2472        assert_eq!(std::mem::size_of::<Globals>(), 32);
2473    }
2474
2475    /// The fragment parameter slot is the other buffer two declarations
2476    /// read: `FragmentParams` here and `KuiFragmentParams` in the
2477    /// epilogue. Sixteen floats then the image's rect, 80 bytes.
2478    #[test]
2479    fn fragment_params_layout_matches() {
2480        assert_eq!(std::mem::size_of::<FragmentParams>(), 80);
2481        assert_eq!(std::mem::offset_of!(FragmentParams, image), 64);
2482        let epilogue = kui_core::fragment::EPILOGUE;
2483        assert!(
2484            epilogue
2485                .contains("struct KuiFragmentParams { p: array<vec4<f32>, 4>, image: vec4<f32> };"),
2486            "the epilogue's params struct moved without this test being told"
2487        );
2488    }
2489
2490    /// The bindings the prelude declares at group 0 are this crate's, by
2491    /// number and kind, since a fragment pipeline binds the quad
2492    /// pipeline's group 0 layout as it is.
2493    #[test]
2494    fn prelude_bindings_match_group_zero() {
2495        let prelude = kui_core::fragment::PRELUDE;
2496        for line in [
2497            "@group(0) @binding(0) var<uniform> kui_globals: KuiGlobals;",
2498            "@group(0) @binding(1) var kui_atlas: texture_2d<f32>;",
2499            "@group(0) @binding(2) var kui_sampler: sampler;",
2500            "@group(0) @binding(3) var kui_sampler_nearest: sampler;",
2501        ] {
2502            assert!(prelude.contains(line), "prelude lacks `{line}`");
2503        }
2504        let quads = include_str!("shader.wgsl");
2505        for line in [
2506            "@group(0) @binding(1) var atlas_tex: texture_2d<f32>;",
2507            "@group(0) @binding(2) var atlas_smp: sampler;",
2508            "@group(0) @binding(3) var nearest_smp: sampler;",
2509        ] {
2510            assert!(quads.contains(line), "shader.wgsl lacks `{line}`");
2511        }
2512    }
2513
2514    /// Both preprocessed variants of the shader must parse and validate
2515    /// (pipeline creation would otherwise fail at runtime, in a window).
2516    #[test]
2517    fn shader_variants_validate() {
2518        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2519        for dual in [false, true] {
2520            let src = preprocess_shader(include_str!("shader.wgsl"), dual);
2521            let module = wgpu::naga::front::wgsl::parse_str(&src)
2522                .unwrap_or_else(|e| panic!("dual={dual}: {}", e.emit_to_string(&src)));
2523            let caps = if dual {
2524                Capabilities::DUAL_SOURCE_BLENDING
2525            } else {
2526                Capabilities::empty()
2527            };
2528            Validator::new(ValidationFlags::all(), caps)
2529                .validate(&module)
2530                .unwrap_or_else(|e| panic!("dual={dual}: {e:?}"));
2531        }
2532    }
2533
2534    /// The ground's shader parses and validates, as the pipeline that
2535    /// draws a window's wallpaper would otherwise fail in a window
2536    /// (backlog F126).
2537    #[test]
2538    fn the_ground_shader_validates() {
2539        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2540        let module = wgpu::naga::front::wgsl::parse_str(GROUND_SHADER)
2541            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(GROUND_SHADER)));
2542        Validator::new(ValidationFlags::all(), Capabilities::empty())
2543            .validate(&module)
2544            .unwrap_or_else(|e| panic!("{e:?}"));
2545    }
2546
2547    /// The backdrop blur's shader (backlog F129), every entry point, and
2548    /// its `Params` the size the Rust struct writes.
2549    #[test]
2550    fn the_backdrop_shader_validates() {
2551        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
2552        let src = include_str!("backdrop.wgsl");
2553        let module = wgpu::naga::front::wgsl::parse_str(src)
2554            .unwrap_or_else(|e| panic!("{}", e.emit_to_string(src)));
2555        Validator::new(ValidationFlags::all(), Capabilities::empty())
2556            .validate(&module)
2557            .unwrap_or_else(|e| panic!("{e:?}"));
2558        let names: Vec<&str> = module
2559            .entry_points
2560            .iter()
2561            .map(|e| e.name.as_str())
2562            .collect();
2563        for want in [
2564            "vs",
2565            "fs_down",
2566            "fs_blur_h",
2567            "fs_blur_v",
2568            "fs_composite",
2569            "fs_blit",
2570        ] {
2571            assert!(names.contains(&want), "{want} in {names:?}");
2572        }
2573        let params = module
2574            .types
2575            .iter()
2576            .find(|(_, t)| t.name.as_deref() == Some("Params"))
2577            .expect("Params")
2578            .1;
2579        let wgpu::naga::TypeInner::Struct { span, .. } = params.inner else {
2580            panic!("Params is a struct");
2581        };
2582        assert_eq!(span as usize, std::mem::size_of::<backdrop::Params>());
2583    }
2584
2585    /// The line `report_faults` says, built without the heap (RG31): the
2586    /// code, the address and the module's UTF-16 path decoded, and for a
2587    /// fault that escaped a window callback the code it escaped as.
2588    #[test]
2589    fn a_fault_line_names_the_code_the_address_and_the_module() {
2590        let path: Vec<u16> = r"C:\Windows\System32\nvoglv64.dll".encode_utf16().collect();
2591        let line = FaultLine::new(0xC000_0005, 0x7ff6_1234, Some(&path), None);
2592        assert_eq!(
2593            std::str::from_utf8(line.bytes()).unwrap(),
2594            "kui: fault 0xc0000005 at 0x7ff61234 in C:\\Windows\\System32\\nvoglv64.dll\n"
2595        );
2596        let line = FaultLine::new(0xC000_0005, 0x10, None, Some(0xC000_041D));
2597        assert_eq!(
2598            std::str::from_utf8(line.bytes()).unwrap(),
2599            "kui: fault 0xc0000005 at 0x10 in no module (jit or freed code), \
2600             escaped from a window callback as 0xc000041d\n"
2601        );
2602        // A path that is not UTF-16 is said, not refused.
2603        let line = FaultLine::new(0xC000_001D, 0x20, Some(&[0x44, 0xD800, 0x45]), None);
2604        assert_eq!(
2605            std::str::from_utf8(line.bytes()).unwrap(),
2606            "kui: fault 0xc000001d at 0x20 in D\u{FFFD}E\n"
2607        );
2608    }
2609
2610    /// A line longer than its stack buffer is cut, between characters,
2611    /// and still ends in its newline.
2612    #[test]
2613    fn a_fault_line_too_long_is_cut_at_a_character() {
2614        let path: Vec<u16> = "é".repeat(1000).encode_utf16().collect();
2615        let line = FaultLine::new(0xC000_00FD, 0x30, Some(&path), Some(0xC000_041D));
2616        let text = std::str::from_utf8(line.bytes()).expect("cut at a character");
2617        assert!(text.ends_with("é\n"), "{text:?}");
2618        assert!(text.len() <= 640 && text.len() >= 638, "{}", text.len());
2619        assert!(text.starts_with("kui: fault 0xc00000fd at 0x30 in é"));
2620    }
2621}