Skip to main content

kui_wgpu/
lib.rs

1//! wgpu renderer for kui: draws a [`kui_core::DisplayList`] with one instanced pipeline, a single draw call per frame.
2//!
3//! kui splits a UI into a model that lays out and paints into a display
4//! list (`kui-core`), a renderer that puts that list on screen (this
5//! crate) and a runner that owns the window and the event loop
6//! (`kui-native`, the one most apps use). Reach for `kui-wgpu` directly
7//! when you are writing your own runner: you already have a window, or an
8//! event loop, that `kui-native` does not fit.
9//!
10//! A [`Renderer`] owns the swapchain of one window and a GPU copy of the
11//! core's glyph atlas, kept in step with the atlas it is handed each
12//! frame. Rounded rectangles, borders, shadows, glyphs and images are all
13//! instances of the same quad, so an ordinary frame is one draw call;
14//! only a texture-backed image or a custom fragment shader splits it.
15//! Several windows share one device through [`Gpu`].
16//!
17//! # Example
18//!
19//! A runner's whole life with the renderer: open it on a window, tell the
20//! core whether subpixel text will render, then build, draw and present
21//! one frame at a time. `window` is anything wgpu can make a surface
22//! from, such as a `winit` window.
23//!
24//! ```rust,no_run
25//! use kui_core::{Core, Size, TextStyle};
26//! use kui_wgpu::{RenderError, Renderer};
27//!
28//! fn run(
29//!     window: impl Into<kui_wgpu::wgpu::SurfaceTarget<'static>>,
30//! ) -> Result<(), Box<dyn std::error::Error>> {
31//!     let (width, height) = (800u32, 600u32);
32//!     let mut renderer = pollster::block_on(Renderer::new(window, width, height))?;
33//!     let mut core = Core::new();
34//!     core.set_subpixel_text(renderer.subpixel_text());
35//!
36//!     loop {
37//!         // When the windowing library reports a new size:
38//!         // renderer.resize(new_width, new_height);
39//!
40//!         // Build the frame through the core, in logical pixels.
41//!         let scale = 1.0;
42//!         let viewport = Size::new(width as f32 / scale, height as f32 / scale);
43//!         let mut ui = core.frame(viewport, scale);
44//!         ui.text("Hello from a custom runner", TextStyle::new(24.0));
45//!         ui.finish();
46//!
47//!         // Draw it. The atlas is `&mut` so the renderer can clear its dirty flag.
48//!         let (list, atlas) = core.output();
49//!         match renderer.render(list, atlas) {
50//!             Ok(report) => {
51//!                 let _blocked_on_vsync_ms = report.vsync_wait_ms;
52//!             }
53//!             Err(RenderError::Reconfigure | RenderError::Validation) => {
54//!                 renderer.resize(width, height);
55//!             }
56//!             Err(RenderError::Skip) => {}
57//!             Err(RenderError::DeviceLost) => {
58//!                 // Open a new `Renderer` (and a new device) and carry on.
59//!                 break;
60//!             }
61//!         }
62//!     }
63//!     Ok(())
64//! }
65//! ```
66//!
67//! # Where to look
68//!
69//! - [`Renderer`]: one window's swapchain, pipelines and atlas texture.
70//! - [`Renderer::render`]: a display list in, a presented frame (or a
71//!   [`RenderError`]) out.
72//! - [`Renderer::resize`]: reconfigure after the window changed size.
73//! - [`Renderer::subpixel_text`]: what to pass to `Core::set_subpixel_text`.
74//! - [`Gpu`]: the device, queue and adapter that windows share;
75//!   [`Renderer::new_in`] opens a second window on it.
76//! - [`RenderError`]: what each failed frame asks the runner to do next.
77//! - [`DEFAULT_FRAME_LATENCY`] and [`Renderer::set_frame_latency`]: how
78//!   many frames may queue ahead of the one on screen.
79//! - [`wgpu`] is re-exported, so a runner builds against the same version
80//!   this crate was.
81//!
82//! # Subpixel text
83//!
84//! Where the device offers dual-source blending (Metal, DX12, most Vulkan)
85//! the pipeline blends per channel, which is what LCD subpixel glyphs need.
86//! Elsewhere it falls back to ordinary alpha blending and the core should
87//! rasterize grayscale masks instead, which is what
88//! [`Renderer::subpixel_text`] tells it.
89//!
90//! The book: <https://kui-book.qxuken.dev>. Repository:
91//! <https://github.com/qxuken/kui>.
92
93pub use wgpu;
94
95use kui_core::atlas::GlyphAtlas;
96use kui_core::{Clip, DisplayList, Quad, QuadKind};
97
98#[repr(C)]
99#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
100struct Instance {
101    pos: [f32; 2],
102    size: [f32; 2],
103    color: [f32; 4],
104    border_color: [f32; 4],
105    params: [f32; 4],
106    uv: [f32; 4],
107    clip: [f32; 4],
108    /// Corner radii, clockwise from the top-left.
109    radii: [f32; 4],
110    /// Radii of the clip itself; all zero = a plain rect clip.
111    clip_radii: [f32; 4],
112}
113
114/// The frame's own numbers, at group 0 binding 0 for both pipelines.
115/// `kui_core::fragment::PRELUDE` declares the same bytes as `KuiGlobals`
116/// so an app's fragment can read `time` and `scale`; the padding is what
117/// makes the struct a multiple of sixteen, which a uniform must be.
118#[repr(C)]
119#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
120struct Globals {
121    viewport: [f32; 2],
122    atlas_size: [f32; 2],
123    time: f32,
124    scale: f32,
125    _pad: [f32; 2],
126}
127
128/// One fragment's parameters as the shader takes them — the sixteen
129/// floats and the texel rect of its `image`, laid out as the epilogue's
130/// `KuiFragmentParams` — padded out to the device's dynamic-offset
131/// alignment so a frame's draws can share one buffer and pick their slot
132/// by offset.
133#[repr(C)]
134#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
135struct FragmentParams {
136    params: [f32; 16],
137    /// `FragmentIn::image`: `[x, y, w, h]` in the texture bound at group 0
138    /// for this draw — the atlas, or the image's own; zero with none.
139    image: [f32; 4],
140}
141
142/// The clip is resolved out of the frame's table here rather than read off
143/// the quad: it rides as an index (`kui_core::ClipId`) so the display list
144/// carries it once per distinct clip instead of once per quad.
145fn instance_of(q: &Quad, clips: &[Clip], textures: &[kui_core::display::TextureDraw]) -> Instance {
146    let clip = clips.get(q.clip as usize).copied().unwrap_or(Clip::NONE);
147    let kind = match q.kind {
148        QuadKind::Solid => 0.0,
149        QuadKind::GlyphMask => 1.0,
150        QuadKind::GlyphColor => 2.0,
151        QuadKind::Image => 3.0,
152        QuadKind::GlyphSubpixel => 4.0,
153        QuadKind::Shadow => 5.0,
154        QuadKind::Segment => 6.0,
155        QuadKind::Fragment => 7.0,
156        // Drawn by the image branch with its own texture bound in the
157        // atlas's place.
158        QuadKind::Texture => 3.0,
159    };
160    // `uv` is atlas texels on every kind but two: a segment carries its
161    // endpoints there as f32 bits, and a texture quad an index into the
162    // side list whose entry holds the texel rect. The shader wants floats.
163    let uv = if q.kind == QuadKind::Segment {
164        q.segment_ends()
165    } else if q.kind == QuadKind::Texture {
166        let uv = textures.get(q.uv[0] as usize).map_or([0; 4], |t| t.uv);
167        [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32]
168    } else {
169        [
170            q.uv[0] as f32,
171            q.uv[1] as f32,
172            q.uv[2] as f32,
173            q.uv[3] as f32,
174        ]
175    };
176    Instance {
177        pos: [q.rect.x, q.rect.y],
178        size: [q.rect.w, q.rect.h],
179        color: [q.color.r, q.color.g, q.color.b, q.color.a],
180        border_color: [
181            q.border_color.r,
182            q.border_color.g,
183            q.border_color.b,
184            q.border_color.a,
185        ],
186        params: [q.blur, q.border_w, kind, 0.0],
187        uv,
188        clip: [clip.rect.x, clip.rect.y, clip.rect.w, clip.rect.h],
189        radii: q.radius,
190        clip_radii: clip.radius,
191    }
192}
193
194/// The GPU objects an app's windows share: one instance, adapter, device and queue.
195///
196/// Two devices cannot see each other's buffers or textures, so every
197/// window of an app draws through the same `Gpu`. A single-window app
198/// never names it: [`Renderer::new`] opens a private one. A second window
199/// takes the first renderer's [`Renderer::gpu`] and opens through
200/// [`Renderer::new_in`]. Cloning a `Gpu` clones a handle to the same
201/// device.
202#[derive(Clone)]
203pub struct Gpu(std::sync::Arc<GpuInner>);
204
205struct GpuInner {
206    instance: wgpu::Instance,
207    adapter: wgpu::Adapter,
208    device: wgpu::Device,
209    queue: wgpu::Queue,
210    dual_source: bool,
211    /// One pipeline per registered fragment per surface format, built the
212    /// first time a frame draws it (about 0.2 ms, paid once) and shared by
213    /// every window on this device, dropped when a frame's list says the
214    /// handle is gone (`dropped_fragments`). A `Mutex` because `Gpu` is a
215    /// shared handle and building is rare; nothing here is touched on a
216    /// frame that draws no new fragment.
217    fragment_pipelines: std::sync::Mutex<
218        std::collections::HashMap<(u64, wgpu::TextureFormat), wgpu::RenderPipeline>,
219    >,
220    /// One texture per texture-backed image, uploaded the first time a
221    /// frame on this device draws it and again when its revision moves,
222    /// shared by every window like the pipelines above, dropped when the
223    /// core says the handle is gone.
224    textures: std::sync::Mutex<std::collections::HashMap<u64, std::sync::Arc<ImageTexture>>>,
225    /// Set by the device's lost callback: a driver update, a GPU reset, a
226    /// hang the OS answered by removing the device. Nothing on it works
227    /// again; a shell opens a new one ([`Gpu::lost`]).
228    lost: std::sync::Arc<std::sync::atomic::AtomicBool>,
229}
230
231/// A texture-backed image on the device: the texture, and what was
232/// uploaded into it. A new `Arc` is made when the size changes, which is
233/// what tells a renderer its bind group is stale.
234struct ImageTexture {
235    texture: wgpu::Texture,
236    view: wgpu::TextureView,
237    width: u32,
238    height: u32,
239    /// The revision the pixels in the texture came from, behind a lock
240    /// because the texture is shared and the upload is per device.
241    rev: std::sync::Mutex<u32>,
242}
243
244impl Gpu {
245    /// Opens a device that can present to `target`, and returns the
246    /// surface it was chosen for.
247    ///
248    /// The first window's surface has to exist before an adapter can be
249    /// picked, so it comes back with the device; later windows get theirs
250    /// from [`Gpu::create_surface`]. Most runners call [`Renderer::new`]
251    /// instead, which does both and builds the renderer. On Windows only
252    /// the D3D12 backend is enabled unless `WGPU_BACKEND` names another.
253    pub async fn new(
254        target: impl Into<wgpu::SurfaceTarget<'static>>,
255    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
256        report_faults();
257        // Every backend the build has, as wgpu defaults — but on Windows
258        // D3D12 alone unless `WGPU_BACKEND` names another. An instance
259        // keeps every backend it enumerated alive for as long as it lives,
260        // so with all of them the process holds an OpenGL context and a
261        // Vulkan instance it never draws with, both in the driver's
262        // `nvoglv64.dll`; and wgpu, left to choose, took Vulkan over D3D12
263        // here. Under a driver update that DLL faulted in present rather
264        // than answer `DEVICE_LOST`, which ended the process; D3D12's
265        // `nvwgf2umx.dll` reports the removal, and the shell reopens the
266        // device (`Gpu::lost`).
267        let mut desc = wgpu::InstanceDescriptor::new_without_display_handle_from_env();
268        if cfg!(windows) && std::env::var_os("WGPU_BACKEND").is_none() {
269            desc.backends = wgpu::Backends::DX12;
270        }
271        let instance = wgpu::Instance::new(desc);
272        let surface = instance.create_surface(target)?;
273        let adapter = instance
274            .request_adapter(&wgpu::RequestAdapterOptions {
275                compatible_surface: Some(&surface),
276                ..Default::default()
277            })
278            .await?;
279        // Per-channel blending for LCD subpixel text, when the device has it.
280        let dual_source = adapter
281            .features()
282            .contains(wgpu::Features::DUAL_SOURCE_BLENDING);
283        let (device, queue) = adapter
284            .request_device(&wgpu::DeviceDescriptor {
285                required_features: if dual_source {
286                    wgpu::Features::DUAL_SOURCE_BLENDING
287                } else {
288                    wgpu::Features::empty()
289                },
290                ..Default::default()
291            })
292            .await?;
293        // What goes wrong on the device is said, not swallowed: an error
294        // outside a scope, and the loss of the device itself — remembered
295        // too, so a frame can tell a dead device from a stale swapchain.
296        device.on_uncaptured_error(std::sync::Arc::new(|e| eprintln!("kui: wgpu: {e}")));
297        let lost = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
298        device.set_device_lost_callback({
299            let lost = lost.clone();
300            move |reason, message| {
301                if reason == wgpu::DeviceLostReason::Unknown {
302                    eprintln!("kui: device lost: {message}");
303                    lost.store(true, std::sync::atomic::Ordering::Release);
304                }
305            }
306        });
307        let gpu = Self(std::sync::Arc::new(GpuInner {
308            instance,
309            adapter,
310            device,
311            queue,
312            dual_source,
313            fragment_pipelines: Default::default(),
314            textures: Default::default(),
315            lost,
316        }));
317        Ok((gpu, surface))
318    }
319
320    /// Whether the device is gone (a driver update or a GPU reset took it).
321    ///
322    /// Nothing on a lost device works again: open a new `Gpu` and a new
323    /// renderer on it for every window. [`Renderer::render`] reports the
324    /// same condition as [`RenderError::DeviceLost`].
325    pub fn lost(&self) -> bool {
326        self.0.lost.load(std::sync::atomic::Ordering::Acquire)
327    }
328
329    /// Loses the device on purpose, as a driver update or a GPU reset
330    /// would, so a runner can test its reopening path without one.
331    ///
332    /// On D3D12 the device is really removed (`ID3D12Device5::RemoveDevice`)
333    /// and the loss lands on its next use through the lost callback, as a
334    /// real one does; elsewhere the device is only marked lost.
335    pub fn mark_lost(&self) {
336        #[cfg(windows)]
337        {
338            use windows::Win32::Graphics::Direct3D12::ID3D12Device5;
339            use windows::core::Interface;
340            // SAFETY: the hal device is only read for its raw handle, and
341            // `RemoveDevice` is what D3D12 offers for exactly this.
342            let removed = unsafe {
343                self.0
344                    .device
345                    .as_hal::<wgpu::hal::api::Dx12>()
346                    .and_then(|d| d.raw_device().cast::<ID3D12Device5>().ok())
347                    .map(|d| d.RemoveDevice())
348            };
349            if removed.is_some() {
350                // The loss lands on the device's next use, through the
351                // lost callback, as a real one does.
352                return;
353            }
354        }
355        self.0
356            .lost
357            .store(true, std::sync::atomic::Ordering::Release);
358    }
359
360    /// A surface for another window on the same instance, which is what
361    /// [`Renderer::new_in`] draws into.
362    pub fn create_surface(
363        &self,
364        target: impl Into<wgpu::SurfaceTarget<'static>>,
365    ) -> Result<wgpu::Surface<'static>, wgpu::CreateSurfaceError> {
366        self.0.instance.create_surface(target)
367    }
368
369    /// The wgpu instance the device was opened on.
370    pub fn instance(&self) -> &wgpu::Instance {
371        &self.0.instance
372    }
373
374    /// The adapter the device was requested from.
375    pub fn adapter(&self) -> &wgpu::Adapter {
376        &self.0.adapter
377    }
378
379    /// The device, for a runner that creates resources of its own on it.
380    pub fn device(&self) -> &wgpu::Device {
381        &self.0.device
382    }
383
384    /// The queue the renderer submits to.
385    pub fn queue(&self) -> &wgpu::Queue {
386        &self.0.queue
387    }
388
389    /// Whether this device blends per channel (dual-source blending), so
390    /// LCD subpixel glyphs draw with per-channel coverage rather than
391    /// their union.
392    pub fn dual_source(&self) -> bool {
393        self.0.dual_source
394    }
395
396    /// The texture for one texture-backed image, uploaded on first sight
397    /// and whenever `rev` has moved past what the texture holds; a size
398    /// change makes a new texture. `None` for a degenerate size, which
399    /// draws nothing.
400    fn image_texture(
401        &self,
402        id: u64,
403        px: &kui_core::display::TexturePixels,
404    ) -> Option<std::sync::Arc<ImageTexture>> {
405        // Degenerate, or past what this device can hold in one texture
406        // (8192 on many adapters, 16384 on Metal): draws nothing, which is
407        // what the core says a texture-backed image that cannot be backed
408        // does, rather than a validation error the device turns into a
409        // panic.
410        let max = self.0.device.limits().max_texture_dimension_2d;
411        if px.width == 0 || px.height == 0 || px.width > max || px.height > max {
412            return None;
413        }
414        let mut cache = self.0.textures.lock().unwrap_or_else(|e| e.into_inner());
415        let fresh = match cache.get(&id) {
416            Some(t) if t.width == px.width && t.height == px.height => {
417                let mut rev = t.rev.lock().unwrap_or_else(|e| e.into_inner());
418                if *rev != px.rev {
419                    upload_image(&self.0.queue, &t.texture, px);
420                    *rev = px.rev;
421                }
422                return Some(t.clone());
423            }
424            _ => {
425                let texture = self.0.device.create_texture(&wgpu::TextureDescriptor {
426                    label: Some("kui.image"),
427                    size: wgpu::Extent3d {
428                        width: px.width,
429                        height: px.height,
430                        depth_or_array_layers: 1,
431                    },
432                    mip_level_count: 1,
433                    sample_count: 1,
434                    dimension: wgpu::TextureDimension::D2,
435                    format: wgpu::TextureFormat::Rgba8Unorm,
436                    usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
437                    view_formats: &[],
438                });
439                upload_image(&self.0.queue, &texture, px);
440                let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
441                std::sync::Arc::new(ImageTexture {
442                    texture,
443                    view,
444                    width: px.width,
445                    height: px.height,
446                    rev: std::sync::Mutex::new(px.rev),
447                })
448            }
449        };
450        cache.insert(id, fresh.clone());
451        Some(fresh)
452    }
453
454    /// Forgets a removed fragment's pipelines, one per surface format it
455    /// was ever drawn in; the GPU frees them once no frame in flight
456    /// holds one.
457    fn drop_fragment_pipelines(&self, id: u64) {
458        self.0
459            .fragment_pipelines
460            .lock()
461            .unwrap_or_else(|e| e.into_inner())
462            .retain(|(fid, _), _| *fid != id);
463    }
464
465    /// Forgets a removed image's texture; the GPU frees it once no bind
466    /// group holds it.
467    fn drop_image_texture(&self, id: u64) {
468        self.0
469            .textures
470            .lock()
471            .unwrap_or_else(|e| e.into_inner())
472            .remove(&id);
473    }
474
475    /// Whether the cache still holds exactly this texture for `id` — what
476    /// a renderer asks before keeping a bind group over it, since a
477    /// removal reaches the cache through whichever window's frame carried
478    /// it and the other windows' bind groups would otherwise hold the
479    /// texture for as long as they live.
480    fn holds_image_texture(&self, id: u64, texture: &std::sync::Arc<ImageTexture>) -> bool {
481        self.0
482            .textures
483            .lock()
484            .unwrap_or_else(|e| e.into_inner())
485            .get(&id)
486            .is_some_and(|t| std::sync::Arc::ptr_eq(t, texture))
487    }
488
489    /// The pipeline for one registered fragment, built on first sight and
490    /// then shared by every window on this device. `source` is the app's
491    /// WGSL, which the core already validated; it is wrapped in the same
492    /// prelude and epilogue here, from `kui_core::fragment::module_source`,
493    /// so what compiles is what was validated.
494    ///
495    /// The source is not validated again here: `Core::add_fragment` parsed
496    /// and validated this exact module text with the same naga this wgpu
497    /// carries, and refused a handle for anything that failed. A module
498    /// that still does not compile is a kui bug, and reaches wgpu's own
499    /// error handler like any other.
500    fn fragment_pipeline(
501        &self,
502        id: u64,
503        source: &str,
504        format: wgpu::TextureFormat,
505        layouts: &FragmentLayouts,
506    ) -> wgpu::RenderPipeline {
507        let mut cache = self
508            .0
509            .fragment_pipelines
510            .lock()
511            .unwrap_or_else(|e| e.into_inner());
512        if let Some(p) = cache.get(&(id, format)) {
513            return p.clone();
514        }
515        let device = &self.0.device;
516        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
517            label: Some("kui.fragment"),
518            source: wgpu::ShaderSource::Wgsl(kui_core::fragment::module_source(source).into()),
519        });
520        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
521            label: Some("kui.fragment"),
522            layout: Some(&layouts.pipeline),
523            vertex: wgpu::VertexState {
524                module: &layouts.vertex,
525                entry_point: Some("vs_main"),
526                compilation_options: Default::default(),
527                buffers: &[Some(instance_buffer_layout(&INSTANCE_ATTRS))],
528            },
529            fragment: Some(wgpu::FragmentState {
530                module: &module,
531                entry_point: Some(kui_core::fragment::ENTRY_POINT),
532                compilation_options: Default::default(),
533                targets: &[Some(wgpu::ColorTargetState {
534                    format,
535                    // A fragment returns premultiplied colour, always over.
536                    blend: Some(wgpu::BlendState {
537                        color: wgpu::BlendComponent {
538                            src_factor: wgpu::BlendFactor::One,
539                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
540                            operation: wgpu::BlendOperation::Add,
541                        },
542                        alpha: wgpu::BlendComponent {
543                            src_factor: wgpu::BlendFactor::One,
544                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
545                            operation: wgpu::BlendOperation::Add,
546                        },
547                    }),
548                    write_mask: wgpu::ColorWrites::ALL,
549                })],
550            }),
551            primitive: wgpu::PrimitiveState::default(),
552            depth_stencil: None,
553            multisample: wgpu::MultisampleState::default(),
554            multiview_mask: None,
555            cache: None,
556        });
557        cache.insert((id, format), pipeline.clone());
558        pipeline
559    }
560}
561
562/// What building a fragment pipeline needs besides its own source: kui's
563/// vertex stage, and the layout that puts the globals at group 0 and the
564/// parameters at group 1.
565struct FragmentLayouts {
566    vertex: wgpu::ShaderModule,
567    pipeline: wgpu::PipelineLayout,
568}
569
570/// The instance attributes both pipelines read; one array so the vertex
571/// layout cannot differ between them.
572const INSTANCE_ATTRS: [wgpu::VertexAttribute; 9] = wgpu::vertex_attr_array![
573    0 => Float32x2, 1 => Float32x2, 2 => Float32x4,
574    3 => Float32x4, 4 => Float32x4, 5 => Float32x4,
575    6 => Float32x4, 7 => Float32x4, 8 => Float32x4,
576];
577
578fn instance_buffer_layout(attrs: &[wgpu::VertexAttribute]) -> wgpu::VertexBufferLayout<'_> {
579    wgpu::VertexBufferLayout {
580        array_stride: std::mem::size_of::<Instance>() as u64,
581        step_mode: wgpu::VertexStepMode::Instance,
582        attributes: attrs,
583    }
584}
585
586impl std::fmt::Debug for Gpu {
587    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
588        f.debug_struct("Gpu")
589            .field("adapter", &self.0.adapter.get_info().name)
590            .field("dual_source", &self.0.dual_source)
591            .finish()
592    }
593}
594
595/// One window's renderer: its surface, the pipelines and a GPU copy of the core's glyph atlas.
596///
597/// Open one per window with [`Renderer::new`] (first window, on a device
598/// of its own) or [`Renderer::new_in`] (another window on a shared
599/// [`Gpu`]). Each frame, hand [`Renderer::render`] the display list and
600/// atlas from `Core::output`; call [`Renderer::resize`] when the window
601/// changes size. The crate root has the whole sequence.
602pub struct Renderer {
603    gpu: Gpu,
604    surface: wgpu::Surface<'static>,
605    config: wgpu::SurfaceConfiguration,
606    pipeline: wgpu::RenderPipeline,
607    globals_buf: wgpu::Buffer,
608    bind_group: wgpu::BindGroup,
609    bind_layout: wgpu::BindGroupLayout,
610    /// The two samplers every group-0 bind group carries: linear at
611    /// binding 2, nearest at 3.
612    samplers: Samplers,
613    /// Per texture-backed image this window has drawn: the device's
614    /// texture, and a bind group of this window's own — group 0 with that
615    /// texture in the atlas's place and a globals copy whose `atlas_size`
616    /// is the texture's, rewritten each frame the image is drawn.
617    texture_binds: std::collections::HashMap<u64, TextureBind>,
618    atlas_tex: wgpu::Texture,
619    atlas_size: u32,
620    atlas_epoch: u64,
621    instance_buf: wgpu::Buffer,
622    instance_cap: usize,
623    instances: Vec<Instance>,
624    /// What a fragment pipeline is built against: kui's vertex stage and
625    /// the two-group layout. Built once per renderer, handed to the
626    /// device's shared cache.
627    fragment_layouts: FragmentLayouts,
628    /// One slot per fragment this frame, each padded to the device's
629    /// dynamic-offset alignment.
630    fragment_params_buf: wgpu::Buffer,
631    fragment_params_cap: usize,
632    fragment_bind: wgpu::BindGroup,
633    fragment_bind_layout: wgpu::BindGroupLayout,
634    /// The alignment slots are padded to; `dynamic_offset` steps by it.
635    uniform_align: u32,
636    /// Scratch for one frame's padded parameter slots.
637    fragment_bytes: Vec<u8>,
638    /// The color a frame is cleared to before anything is drawn.
639    ///
640    /// A runner usually sets it to the theme's background each frame. It
641    /// is written straight through: the renderer asks for a non-sRGB
642    /// surface, so each component is the byte it lands as, the same way a
643    /// quad's color is.
644    pub clear_color: wgpu::Color,
645}
646
647struct Samplers {
648    linear: wgpu::Sampler,
649    nearest: wgpu::Sampler,
650}
651
652struct TextureBind {
653    texture: std::sync::Arc<ImageTexture>,
654    globals: wgpu::Buffer,
655    bind: wgpu::BindGroup,
656}
657
658fn upload_image(
659    queue: &wgpu::Queue,
660    texture: &wgpu::Texture,
661    px: &kui_core::display::TexturePixels,
662) {
663    queue.write_texture(
664        wgpu::TexelCopyTextureInfo {
665            texture,
666            mip_level: 0,
667            origin: wgpu::Origin3d::ZERO,
668            aspect: wgpu::TextureAspect::All,
669        },
670        &px.rgba,
671        wgpu::TexelCopyBufferLayout {
672            offset: 0,
673            bytes_per_row: Some(px.width * 4),
674            rows_per_image: Some(px.height),
675        },
676        wgpu::Extent3d {
677            width: px.width,
678            height: px.height,
679            depth_or_array_layers: 1,
680        },
681    );
682}
683
684/// Picks the `//DUAL:` or `//SINGLE:` lines of the shader template.
685fn preprocess_shader(src: &str, dual: bool) -> String {
686    let (keep, drop) = if dual {
687        ("//DUAL:", "//SINGLE:")
688    } else {
689        ("//SINGLE:", "//DUAL:")
690    };
691    let mut out = String::with_capacity(src.len());
692    for line in src.lines() {
693        if let Some(rest) = line.strip_prefix(keep) {
694            out.push_str(rest);
695        } else if line.starts_with(drop) {
696            continue;
697        } else {
698            out.push_str(line);
699        }
700        out.push('\n');
701    }
702    out
703}
704
705/// How many frames may be queued ahead of the one on screen by default: two, or one on Windows.
706///
707/// With two, a drawable to render into is waiting while the previous
708/// frame's is still out, so a frame whose thread woke a little late still
709/// makes its vsync (on Metal this is triple buffering). The price is that
710/// a frame built the moment a drawable frees reaches the screen a vsync
711/// later than it could; `kui-native` wins that back by starting frames at
712/// the display's vsync, and a runner driven any other way pays it. On
713/// Windows the flip-model swapchain delivers every vsync with a single
714/// queued frame, so the second would be latency for nothing. Change it
715/// per renderer with [`Renderer::set_frame_latency`].
716pub const DEFAULT_FRAME_LATENCY: u32 = if cfg!(target_os = "windows") { 1 } else { 2 };
717
718impl Renderer {
719    /// A renderer for one window, on a device of its own.
720    ///
721    /// `width` and `height` are the window's size in physical pixels.
722    /// Fails when no adapter can present to `target` or the surface cannot
723    /// be configured. Use [`Renderer::new_in`] for every window after the
724    /// first, so they share the device.
725    pub async fn new(
726        target: impl Into<wgpu::SurfaceTarget<'static>>,
727        width: u32,
728        height: u32,
729    ) -> Result<Self, Box<dyn std::error::Error>> {
730        let (gpu, surface) = Gpu::new(target).await?;
731        Self::with_surface(gpu, surface, width, height)
732    }
733
734    /// A renderer for another window on an existing device, the one every
735    /// window of the app shares. Get `gpu` from the first renderer's
736    /// [`Renderer::gpu`].
737    pub fn new_in(
738        gpu: &Gpu,
739        target: impl Into<wgpu::SurfaceTarget<'static>>,
740        width: u32,
741        height: u32,
742    ) -> Result<Self, Box<dyn std::error::Error>> {
743        let surface = gpu.create_surface(target)?;
744        Self::with_surface(gpu.clone(), surface, width, height)
745    }
746
747    /// The device this renderer draws with, to open another window on.
748    pub fn gpu(&self) -> &Gpu {
749        &self.gpu
750    }
751
752    /// Sets how many frames may be queued ahead of the one on screen (at
753    /// least one; see [`DEFAULT_FRAME_LATENCY`]). Reconfigures the surface
754    /// when the value changes.
755    pub fn set_frame_latency(&mut self, frames: u32) {
756        let frames = frames.max(1);
757        if self.config.desired_maximum_frame_latency != frames {
758            self.config.desired_maximum_frame_latency = frames;
759            self.surface.configure(self.gpu.device(), &self.config);
760        }
761    }
762
763    /// The frame latency the surface is configured with.
764    pub fn frame_latency(&self) -> u32 {
765        self.config.desired_maximum_frame_latency
766    }
767
768    fn with_surface(
769        gpu: Gpu,
770        surface: wgpu::Surface<'static>,
771        width: u32,
772        height: u32,
773    ) -> Result<Self, Box<dyn std::error::Error>> {
774        let device = gpu.device();
775        let dual_source = gpu.dual_source();
776
777        let caps = surface.get_capabilities(gpu.adapter());
778        let format = caps
779            .formats
780            .iter()
781            .copied()
782            .find(|f| !f.is_srgb())
783            .unwrap_or(caps.formats[0]);
784        let config = wgpu::SurfaceConfiguration {
785            usage: wgpu::TextureUsages::RENDER_ATTACHMENT,
786            format,
787            width: width.clamp(1, device.limits().max_texture_dimension_2d),
788            height: height.clamp(1, device.limits().max_texture_dimension_2d),
789            present_mode: wgpu::PresentMode::AutoVsync,
790            alpha_mode: caps.alpha_modes[0],
791            color_space: wgpu::SurfaceColorSpace::Auto,
792            view_formats: vec![],
793            // See `DEFAULT_FRAME_LATENCY`; a runner that wants another
794            // says so through `set_frame_latency`.
795            desired_maximum_frame_latency: DEFAULT_FRAME_LATENCY,
796        };
797        // A configure that fails only reports to the device's error
798        // handler, and the first acquire on the unconfigured surface is a
799        // panic inside wgpu; caught here, it is this constructor's error
800        // — a window DXGI will not give a second swapchain, say.
801        let scope = device.push_error_scope(wgpu::ErrorFilter::Validation);
802        surface.configure(device, &config);
803        if let Some(err) = pollster::block_on(scope.pop()) {
804            return Err(format!("configuring the surface: {err}").into());
805        }
806
807        let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
808            label: Some("kui"),
809            source: wgpu::ShaderSource::Wgsl(
810                preprocess_shader(include_str!("shader.wgsl"), dual_source).into(),
811            ),
812        });
813        // Dual source: the shader outputs premultiplied color and a
814        // per-channel coverage; out = src + dst * (1 - coverage). For
815        // ordinary quads every channel's coverage equals alpha, which is
816        // exactly premultiplied alpha blending.
817        let blend = if dual_source {
818            wgpu::BlendState {
819                color: wgpu::BlendComponent {
820                    src_factor: wgpu::BlendFactor::One,
821                    dst_factor: wgpu::BlendFactor::OneMinusSrc1,
822                    operation: wgpu::BlendOperation::Add,
823                },
824                alpha: wgpu::BlendComponent {
825                    src_factor: wgpu::BlendFactor::One,
826                    dst_factor: wgpu::BlendFactor::OneMinusSrc1Alpha,
827                    operation: wgpu::BlendOperation::Add,
828                },
829            }
830        } else {
831            wgpu::BlendState::ALPHA_BLENDING
832        };
833
834        let bind_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
835            label: Some("kui.globals"),
836            entries: &[
837                wgpu::BindGroupLayoutEntry {
838                    binding: 0,
839                    visibility: wgpu::ShaderStages::VERTEX_FRAGMENT,
840                    ty: wgpu::BindingType::Buffer {
841                        ty: wgpu::BufferBindingType::Uniform,
842                        has_dynamic_offset: false,
843                        min_binding_size: None,
844                    },
845                    count: None,
846                },
847                wgpu::BindGroupLayoutEntry {
848                    binding: 1,
849                    visibility: wgpu::ShaderStages::FRAGMENT,
850                    ty: wgpu::BindingType::Texture {
851                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
852                        view_dimension: wgpu::TextureViewDimension::D2,
853                        multisampled: false,
854                    },
855                    count: None,
856                },
857                wgpu::BindGroupLayoutEntry {
858                    binding: 2,
859                    visibility: wgpu::ShaderStages::FRAGMENT,
860                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
861                    count: None,
862                },
863                wgpu::BindGroupLayoutEntry {
864                    binding: 3,
865                    visibility: wgpu::ShaderStages::FRAGMENT,
866                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
867                    count: None,
868                },
869            ],
870        });
871
872        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
873            label: Some("kui"),
874            bind_group_layouts: &[Some(&bind_layout)],
875            immediate_size: 0,
876        });
877
878        let instance_attrs = INSTANCE_ATTRS;
879        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
880            label: Some("kui.quads"),
881            layout: Some(&pipeline_layout),
882            vertex: wgpu::VertexState {
883                module: &shader,
884                entry_point: Some("vs_main"),
885                compilation_options: Default::default(),
886                buffers: &[Some(instance_buffer_layout(&instance_attrs))],
887            },
888            fragment: Some(wgpu::FragmentState {
889                module: &shader,
890                entry_point: Some("fs_main"),
891                compilation_options: Default::default(),
892                targets: &[Some(wgpu::ColorTargetState {
893                    format,
894                    blend: Some(blend),
895                    write_mask: wgpu::ColorWrites::ALL,
896                })],
897            }),
898            primitive: wgpu::PrimitiveState::default(),
899            depth_stencil: None,
900            multisample: wgpu::MultisampleState::default(),
901            multiview_mask: None,
902            cache: None,
903        });
904
905        let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
906            label: Some("kui.globals"),
907            size: std::mem::size_of::<Globals>() as u64,
908            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
909            mapped_at_creation: false,
910        });
911
912        let atlas_size = kui_core::atlas::ATLAS_SIZE;
913        let atlas_tex = create_atlas_texture(device, atlas_size);
914        let samplers = Samplers {
915            linear: device.create_sampler(&wgpu::SamplerDescriptor {
916                label: Some("kui.linear"),
917                mag_filter: wgpu::FilterMode::Linear,
918                min_filter: wgpu::FilterMode::Linear,
919                ..Default::default()
920            }),
921            nearest: device.create_sampler(&wgpu::SamplerDescriptor {
922                label: Some("kui.nearest"),
923                mag_filter: wgpu::FilterMode::Nearest,
924                min_filter: wgpu::FilterMode::Nearest,
925                ..Default::default()
926            }),
927        };
928        let atlas_view = atlas_tex.create_view(&wgpu::TextureViewDescriptor::default());
929        let bind_group =
930            create_bind_group(device, &bind_layout, &globals_buf, &atlas_view, &samplers);
931
932        let instance_cap = 4096;
933        let instance_buf = create_instance_buffer(device, instance_cap);
934
935        // Fragments: one uniform slot per draw, picked by dynamic offset,
936        // and the layout their pipelines are built against. All of it is
937        // built whether or not a frame ever draws one — a bind group
938        // layout and an empty buffer, not a pipeline, which is the part
939        // that costs and is built on first sight.
940        let fragment_bind_layout =
941            device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
942                label: Some("kui.fragment.params"),
943                entries: &[wgpu::BindGroupLayoutEntry {
944                    binding: 0,
945                    visibility: wgpu::ShaderStages::FRAGMENT,
946                    ty: wgpu::BindingType::Buffer {
947                        ty: wgpu::BufferBindingType::Uniform,
948                        has_dynamic_offset: true,
949                        min_binding_size: std::num::NonZeroU64::new(std::mem::size_of::<
950                            FragmentParams,
951                        >()
952                            as u64),
953                    },
954                    count: None,
955                }],
956            });
957        let fragment_layouts = FragmentLayouts {
958            vertex: shader,
959            pipeline: device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
960                label: Some("kui.fragment"),
961                bind_group_layouts: &[Some(&bind_layout), Some(&fragment_bind_layout)],
962                immediate_size: 0,
963            }),
964        };
965        // The *device's* limit, not the adapter's: the device is opened with
966        // `Limits::default()`, whose `min_uniform_buffer_offset_alignment` is
967        // 256, and validation holds a dynamic offset to what the device asked
968        // for rather than to what the hardware could have done. An adapter
969        // reporting the smaller 64 — which DX12 does — then gave 64-byte slots
970        // and a validation error on the frame's second fragment.
971        let uniform_align = device.limits().min_uniform_buffer_offset_alignment;
972        let fragment_params_cap = 16;
973        let fragment_params_buf =
974            create_fragment_params_buffer(device, fragment_params_cap, uniform_align);
975        let fragment_bind =
976            create_fragment_bind_group(device, &fragment_bind_layout, &fragment_params_buf);
977
978        Ok(Self {
979            gpu,
980            surface,
981            config,
982            pipeline,
983            globals_buf,
984            bind_group,
985            bind_layout,
986            samplers,
987            texture_binds: Default::default(),
988            atlas_tex,
989            atlas_size,
990            atlas_epoch: u64::MAX,
991            instance_buf,
992            instance_cap,
993            instances: Vec::new(),
994            fragment_layouts,
995            fragment_params_buf,
996            fragment_params_cap,
997            fragment_bind,
998            fragment_bind_layout,
999            uniform_align,
1000            fragment_bytes: Vec::new(),
1001            clear_color: wgpu::Color {
1002                r: 0.06,
1003                g: 0.065,
1004                b: 0.08,
1005                a: 1.0,
1006            },
1007        })
1008    }
1009
1010    /// Whether this device can draw LCD subpixel glyphs
1011    /// (`QuadKind::GlyphSubpixel`) as intended.
1012    ///
1013    /// Pass it to `Core::set_subpixel_text` once after opening the
1014    /// renderer; when it is `false` the core keeps rasterizing grayscale
1015    /// masks, which every device blends correctly.
1016    pub fn subpixel_text(&self) -> bool {
1017        self.gpu.dual_source()
1018    }
1019
1020    /// Reconfigures the swapchain for a new window size, in physical pixels.
1021    ///
1022    /// The size is clamped to at least 1 (a minimized window reports zero,
1023    /// and a zero-sized surface is a validation error) and to the device's
1024    /// largest texture (Windows hands out nonsense sizes mid-resize, and
1025    /// configuring past the limit panics inside wgpu). A clamped frame is
1026    /// one wrong picture; the next real size fixes it.
1027    pub fn resize(&mut self, width: u32, height: u32) {
1028        let max = self.gpu.device().limits().max_texture_dimension_2d;
1029        self.config.width = width.clamp(1, max);
1030        self.config.height = height.clamp(1, max);
1031        self.surface.configure(self.gpu.device(), &self.config);
1032    }
1033
1034    fn sync_atlas(&mut self, atlas: &mut GlyphAtlas) {
1035        if atlas.size != self.atlas_size {
1036            self.atlas_size = atlas.size;
1037            self.atlas_tex = create_atlas_texture(self.gpu.device(), atlas.size);
1038            let view = self
1039                .atlas_tex
1040                .create_view(&wgpu::TextureViewDescriptor::default());
1041            self.bind_group = create_bind_group(
1042                self.gpu.device(),
1043                &self.bind_layout,
1044                &self.globals_buf,
1045                &view,
1046                &self.samplers,
1047            );
1048            self.atlas_epoch = u64::MAX;
1049        }
1050        if atlas.dirty || self.atlas_epoch != atlas.epoch {
1051            self.gpu.queue().write_texture(
1052                wgpu::TexelCopyTextureInfo {
1053                    texture: &self.atlas_tex,
1054                    mip_level: 0,
1055                    origin: wgpu::Origin3d::ZERO,
1056                    aspect: wgpu::TextureAspect::All,
1057                },
1058                &atlas.pixels,
1059                wgpu::TexelCopyBufferLayout {
1060                    offset: 0,
1061                    bytes_per_row: Some(atlas.size * 4),
1062                    rows_per_image: Some(atlas.size),
1063                },
1064                wgpu::Extent3d {
1065                    width: atlas.size,
1066                    height: atlas.size,
1067                    depth_or_array_layers: 1,
1068                },
1069            );
1070            atlas.dirty = false;
1071            self.atlas_epoch = atlas.epoch;
1072        }
1073    }
1074
1075    /// Draws one frame and presents it.
1076    ///
1077    /// Uploads the atlas when it changed since the last frame (and clears
1078    /// its `dirty` flag), writes the list's quads to the instance buffer,
1079    /// draws them over [`Renderer::clear_color`] in one render pass, and
1080    /// presents. Acquiring the swapchain image blocks while vsync holds
1081    /// the frame back; that time comes back as
1082    /// [`RenderReport::vsync_wait_ms`] so a runner can tell pacing from
1083    /// work.
1084    ///
1085    /// Fails without presenting when the surface or the device cannot
1086    /// take the frame; each [`RenderError`] says what to do next.
1087    pub fn render(
1088        &mut self,
1089        dl: &DisplayList,
1090        atlas: &mut GlyphAtlas,
1091    ) -> Result<RenderReport, RenderError> {
1092        // A dead device takes no work: everything below would only add
1093        // errors to the one that lost it.
1094        if self.gpu.lost() {
1095            return Err(RenderError::DeviceLost);
1096        }
1097        self.sync_atlas(atlas);
1098
1099        self.instances.clear();
1100        self.instances.extend(
1101            dl.quads
1102                .iter()
1103                .map(|q| instance_of(q, &dl.clips, &dl.textures)),
1104        );
1105        if self.instances.len() > self.instance_cap {
1106            self.instance_cap = self.instances.len().next_power_of_two();
1107            self.instance_buf = create_instance_buffer(self.gpu.device(), self.instance_cap);
1108        }
1109        if !self.instances.is_empty() {
1110            self.gpu.queue().write_buffer(
1111                &self.instance_buf,
1112                0,
1113                bytemuck::cast_slice(&self.instances),
1114            );
1115        }
1116        let globals = Globals {
1117            viewport: [dl.viewport.w.max(1.0), dl.viewport.h.max(1.0)],
1118            atlas_size: [self.atlas_size as f32, self.atlas_size as f32],
1119            time: dl.time,
1120            scale: dl.scale,
1121            _pad: [0.0; 2],
1122        };
1123        self.gpu
1124            .queue()
1125            .write_buffer(&self.globals_buf, 0, bytemuck::bytes_of(&globals));
1126
1127        // Texture-backed images: drop what the core
1128        // removed, upload what moved, and give each one drawn this frame
1129        // a group-0 bind group of its own with a globals copy whose
1130        // `atlas_size` is the texture's. All skipped on a frame that
1131        // draws none.
1132        for id in &dl.dropped_textures {
1133            self.texture_binds.remove(&id.to_ffi());
1134            self.gpu.drop_image_texture(id.to_ffi());
1135        }
1136        // And the pipelines of removed fragments — built per handle
1137        // and shared by every window, so one window's list carries the
1138        // removal and this is the only eviction they get.
1139        for id in &dl.dropped_fragments {
1140            self.gpu.drop_fragment_pipelines(id.to_ffi());
1141        }
1142        // A drop another window's frame carried: the cache no longer
1143        // holds the texture this bind group does. One lock per frame,
1144        // and only for a window that has ever drawn a texture.
1145        if !self.texture_binds.is_empty() {
1146            let gpu = &self.gpu;
1147            self.texture_binds
1148                .retain(|id, b| gpu.holds_image_texture(*id, &b.texture));
1149        }
1150        let mut texture_binds: Vec<Option<u64>> = Vec::new();
1151        if !dl.textures.is_empty() {
1152            texture_binds.reserve(dl.textures.len());
1153            for (draw, px) in dl.textures.iter().zip(&dl.texture_pixels) {
1154                let id = draw.id.to_ffi();
1155                let Some(texture) = self.gpu.image_texture(id, px) else {
1156                    texture_binds.push(None);
1157                    continue;
1158                };
1159                let stale = self
1160                    .texture_binds
1161                    .get(&id)
1162                    .is_none_or(|b| !std::sync::Arc::ptr_eq(&b.texture, &texture));
1163                if stale {
1164                    let device = self.gpu.device();
1165                    let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1166                        label: Some("kui.image.globals"),
1167                        size: std::mem::size_of::<Globals>() as u64,
1168                        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1169                        mapped_at_creation: false,
1170                    });
1171                    let bind = create_bind_group(
1172                        device,
1173                        &self.bind_layout,
1174                        &globals_buf,
1175                        &texture.view,
1176                        &self.samplers,
1177                    );
1178                    self.texture_binds.insert(
1179                        id,
1180                        TextureBind {
1181                            texture: texture.clone(),
1182                            globals: globals_buf,
1183                            bind,
1184                        },
1185                    );
1186                }
1187                let b = &self.texture_binds[&id];
1188                let mine = Globals {
1189                    atlas_size: [texture.width as f32, texture.height as f32],
1190                    ..globals
1191                };
1192                self.gpu
1193                    .queue()
1194                    .write_buffer(&b.globals, 0, bytemuck::bytes_of(&mine));
1195                texture_binds.push(Some(id));
1196            }
1197        }
1198
1199        // Each fragment's parameters into its own slot, and its pipeline
1200        // built if this device has not seen the handle before. Both are
1201        // skipped whole on a frame that draws no fragment.
1202        let mut fragment_pipelines: Vec<wgpu::RenderPipeline> = Vec::new();
1203        if !dl.fragments.is_empty() {
1204            let align = self.uniform_align as usize;
1205            if dl.fragments.len() > self.fragment_params_cap {
1206                self.fragment_params_cap = dl.fragments.len().next_power_of_two();
1207                self.fragment_params_buf = create_fragment_params_buffer(
1208                    self.gpu.device(),
1209                    self.fragment_params_cap,
1210                    self.uniform_align,
1211                );
1212                self.fragment_bind = create_fragment_bind_group(
1213                    self.gpu.device(),
1214                    &self.fragment_bind_layout,
1215                    &self.fragment_params_buf,
1216                );
1217            }
1218            self.fragment_bytes.clear();
1219            self.fragment_bytes.resize(dl.fragments.len() * align, 0);
1220            for (i, draw) in dl.fragments.iter().enumerate() {
1221                let uv = draw.image.uv();
1222                let slot = FragmentParams {
1223                    params: draw.params,
1224                    image: [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32],
1225                };
1226                let at = i * align;
1227                self.fragment_bytes[at..at + std::mem::size_of::<FragmentParams>()]
1228                    .copy_from_slice(bytemuck::bytes_of(&slot));
1229            }
1230            self.gpu
1231                .queue()
1232                .write_buffer(&self.fragment_params_buf, 0, &self.fragment_bytes);
1233            fragment_pipelines.reserve(dl.fragments.len());
1234            for (draw, source) in dl.fragments.iter().zip(&dl.fragment_sources) {
1235                fragment_pipelines.push(self.gpu.fragment_pipeline(
1236                    draw.id.to_ffi(),
1237                    source,
1238                    self.config.format,
1239                    &self.fragment_layouts,
1240                ));
1241            }
1242        }
1243
1244        // Acquiring the swapchain image is where vsync backpressure blocks;
1245        // report it separately so latency graphs show pacing vs work.
1246        let t_wait = std::time::Instant::now();
1247        let frame = match self.surface.get_current_texture() {
1248            wgpu::CurrentSurfaceTexture::Success(f)
1249            | wgpu::CurrentSurfaceTexture::Suboptimal(f) => f,
1250            wgpu::CurrentSurfaceTexture::Timeout | wgpu::CurrentSurfaceTexture::Occluded => {
1251                return Err(RenderError::Skip);
1252            }
1253            wgpu::CurrentSurfaceTexture::Outdated | wgpu::CurrentSurfaceTexture::Lost => {
1254                return Err(RenderError::Reconfigure);
1255            }
1256            // The acquire's error went to the device's error handler; if
1257            // it was the device itself, the lost callback has run by now.
1258            wgpu::CurrentSurfaceTexture::Validation => {
1259                return Err(if self.gpu.lost() {
1260                    RenderError::DeviceLost
1261                } else {
1262                    RenderError::Validation
1263                });
1264            }
1265        };
1266        let vsync_wait_ms = t_wait.elapsed().as_secs_f32() * 1e3;
1267        let view = frame
1268            .texture
1269            .create_view(&wgpu::TextureViewDescriptor::default());
1270        let mut encoder = self
1271            .gpu
1272            .device()
1273            .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("kui") });
1274        {
1275            let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1276                label: Some("kui"),
1277                color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1278                    view: &view,
1279                    depth_slice: None,
1280                    resolve_target: None,
1281                    ops: wgpu::Operations {
1282                        load: wgpu::LoadOp::Clear(self.clear_color),
1283                        store: wgpu::StoreOp::Store,
1284                    },
1285                })],
1286                depth_stencil_attachment: None,
1287                timestamp_writes: None,
1288                occlusion_query_set: None,
1289                multiview_mask: None,
1290            });
1291            if !self.instances.is_empty() {
1292                pass.set_vertex_buffer(0, self.instance_buf.slice(..));
1293                if fragment_pipelines.is_empty() && texture_binds.is_empty() {
1294                    // The whole frame in one instanced draw, as it has
1295                    // always been. Nothing below runs.
1296                    pass.set_pipeline(&self.pipeline);
1297                    pass.set_bind_group(0, &self.bind_group, &[]);
1298                    pass.draw(0..6, 0..self.instances.len() as u32);
1299                } else {
1300                    // A fragment or a texture-backed image interrupts the
1301                    // run: draw what came before with the über-pipeline,
1302                    // then that one quad with its own pipeline (a
1303                    // fragment) or its own group 0 (a texture), then
1304                    // carry on. Consecutive quads of the same handle
1305                    // still take one set each (about 0.6 us); runs of
1306                    // ordinary quads are unbroken.
1307                    let mut run_start = 0u32;
1308                    let mut on_quads = false;
1309                    for (i, q) in dl.quads.iter().enumerate() {
1310                        if q.kind != QuadKind::Fragment && q.kind != QuadKind::Texture {
1311                            continue;
1312                        }
1313                        let i = i as u32;
1314                        if i > run_start {
1315                            if !on_quads {
1316                                pass.set_pipeline(&self.pipeline);
1317                                pass.set_bind_group(0, &self.bind_group, &[]);
1318                                on_quads = true;
1319                            }
1320                            pass.draw(0..6, run_start..i);
1321                        }
1322                        // `uv[0]` is the index into the side list, which
1323                        // is also this fragment's parameter slot, or this
1324                        // texture's bind.
1325                        let slot = q.uv[0] as usize;
1326                        if q.kind == QuadKind::Texture {
1327                            if let Some(Some(id)) = texture_binds.get(slot)
1328                                && let Some(b) = self.texture_binds.get(id)
1329                            {
1330                                pass.set_pipeline(&self.pipeline);
1331                                pass.set_bind_group(0, &b.bind, &[]);
1332                                on_quads = false;
1333                                pass.draw(0..6, i..i + 1);
1334                            }
1335                        } else if let Some(pipeline) = fragment_pipelines.get(slot) {
1336                            // A fragment reading a texture-backed image
1337                            // takes that image's group 0 — the texture in
1338                            // the atlas's place, `atlas_size` its size —
1339                            // exactly as a texture quad does; one reading
1340                            // the atlas, or nothing, takes the frame's.
1341                            // A texture the device could not make (a
1342                            // degenerate or oversized image) draws the
1343                            // fragment against the atlas with a zero rect,
1344                            // which `kui_sample` reads as no image.
1345                            let group0 = match draw_image_texture(&dl.fragments[slot]) {
1346                                Some(index) => texture_binds
1347                                    .get(index)
1348                                    .copied()
1349                                    .flatten()
1350                                    .and_then(|id| self.texture_binds.get(&id))
1351                                    .map_or(&self.bind_group, |b| &b.bind),
1352                                None => &self.bind_group,
1353                            };
1354                            pass.set_pipeline(pipeline);
1355                            pass.set_bind_group(0, group0, &[]);
1356                            pass.set_bind_group(
1357                                1,
1358                                &self.fragment_bind,
1359                                &[slot as u32 * self.uniform_align],
1360                            );
1361                            on_quads = false;
1362                            pass.draw(0..6, i..i + 1);
1363                        }
1364                        run_start = i + 1;
1365                    }
1366                    let end = self.instances.len() as u32;
1367                    if end > run_start {
1368                        if !on_quads {
1369                            pass.set_pipeline(&self.pipeline);
1370                            pass.set_bind_group(0, &self.bind_group, &[]);
1371                        }
1372                        pass.draw(0..6, run_start..end);
1373                    }
1374                }
1375            }
1376        }
1377        self.gpu.queue().submit([encoder.finish()]);
1378        self.gpu.queue().present(frame);
1379        Ok(RenderReport { vsync_wait_ms })
1380    }
1381}
1382
1383/// The `textures` entry a fragment draw reads its image from, if its
1384/// image has a texture of its own.
1385fn draw_image_texture(draw: &kui_core::FragmentDraw) -> Option<usize> {
1386    match draw.image {
1387        kui_core::FragmentImage::Texture { index, .. } => Some(index as usize),
1388        _ => None,
1389    }
1390}
1391
1392/// Timing details from one [`Renderer::render`] call.
1393#[derive(Clone, Copy, Debug, Default)]
1394pub struct RenderReport {
1395    /// Milliseconds spent blocked acquiring the swapchain image, which is
1396    /// where vsync backpressure shows up.
1397    pub vsync_wait_ms: f32,
1398}
1399
1400/// A frame that produced no image, and what to do about it.
1401///
1402/// Mapped from wgpu's `CurrentSurfaceTexture`; the crate root's example
1403/// handles every variant.
1404#[derive(Clone, Copy, Debug)]
1405pub enum RenderError {
1406    /// The surface is outdated or lost: call [`Renderer::resize`] with the
1407    /// window's size and draw again.
1408    Reconfigure,
1409    /// Nothing can be presented right now (the window is occluded, or the
1410    /// acquire timed out): try again next frame.
1411    Skip,
1412    /// The surface is configured wrong for the window: call
1413    /// [`Renderer::resize`] with the window's size and draw again. A
1414    /// surface that stays wrong is best given up with its device.
1415    Validation,
1416    /// The device is gone ([`Gpu::lost`]): open a new one, and a renderer
1417    /// on it for every window.
1418    DeviceLost,
1419}
1420
1421impl std::fmt::Display for RenderError {
1422    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1423        match self {
1424            Self::Reconfigure => write!(f, "surface outdated or lost; reconfigure"),
1425            Self::Skip => write!(f, "no frame available; skip"),
1426            Self::Validation => write!(f, "surface texture validation error"),
1427            Self::DeviceLost => write!(f, "device lost; reopen"),
1428        }
1429    }
1430}
1431
1432impl std::error::Error for RenderError {}
1433
1434fn create_atlas_texture(device: &wgpu::Device, size: u32) -> wgpu::Texture {
1435    device.create_texture(&wgpu::TextureDescriptor {
1436        label: Some("kui.atlas"),
1437        size: wgpu::Extent3d {
1438            width: size,
1439            height: size,
1440            depth_or_array_layers: 1,
1441        },
1442        mip_level_count: 1,
1443        sample_count: 1,
1444        dimension: wgpu::TextureDimension::D2,
1445        format: wgpu::TextureFormat::Rgba8Unorm,
1446        usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1447        view_formats: &[],
1448    })
1449}
1450
1451/// Group 0: the globals, a texture — the atlas, or a texture-backed image
1452/// in its place — and the two samplers.
1453fn create_bind_group(
1454    device: &wgpu::Device,
1455    layout: &wgpu::BindGroupLayout,
1456    globals: &wgpu::Buffer,
1457    view: &wgpu::TextureView,
1458    samplers: &Samplers,
1459) -> wgpu::BindGroup {
1460    device.create_bind_group(&wgpu::BindGroupDescriptor {
1461        label: Some("kui"),
1462        layout,
1463        entries: &[
1464            wgpu::BindGroupEntry {
1465                binding: 0,
1466                resource: globals.as_entire_binding(),
1467            },
1468            wgpu::BindGroupEntry {
1469                binding: 1,
1470                resource: wgpu::BindingResource::TextureView(view),
1471            },
1472            wgpu::BindGroupEntry {
1473                binding: 2,
1474                resource: wgpu::BindingResource::Sampler(&samplers.linear),
1475            },
1476            wgpu::BindGroupEntry {
1477                binding: 3,
1478                resource: wgpu::BindingResource::Sampler(&samplers.nearest),
1479            },
1480        ],
1481    })
1482}
1483
1484fn create_instance_buffer(device: &wgpu::Device, cap: usize) -> wgpu::Buffer {
1485    device.create_buffer(&wgpu::BufferDescriptor {
1486        label: Some("kui.instances"),
1487        size: (cap * std::mem::size_of::<Instance>()) as u64,
1488        usage: wgpu::BufferUsages::VERTEX | wgpu::BufferUsages::COPY_DST,
1489        mapped_at_creation: false,
1490    })
1491}
1492
1493fn create_fragment_params_buffer(device: &wgpu::Device, cap: usize, align: u32) -> wgpu::Buffer {
1494    device.create_buffer(&wgpu::BufferDescriptor {
1495        label: Some("kui.fragment.params"),
1496        size: (cap.max(1) * align as usize) as u64,
1497        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1498        mapped_at_creation: false,
1499    })
1500}
1501
1502fn create_fragment_bind_group(
1503    device: &wgpu::Device,
1504    layout: &wgpu::BindGroupLayout,
1505    buf: &wgpu::Buffer,
1506) -> wgpu::BindGroup {
1507    device.create_bind_group(&wgpu::BindGroupDescriptor {
1508        label: Some("kui.fragment.params"),
1509        layout,
1510        entries: &[wgpu::BindGroupEntry {
1511            binding: 0,
1512            resource: wgpu::BindingResource::Buffer(wgpu::BufferBinding {
1513                buffer: buf,
1514                offset: 0,
1515                size: std::num::NonZeroU64::new(std::mem::size_of::<FragmentParams>() as u64),
1516            }),
1517        }],
1518    })
1519}
1520
1521/// Prints the code and module of a crash to stderr before the process dies (Windows only).
1522///
1523/// A fault in a GPU driver, such as one being replaced under the app, ends
1524/// the process with no line from anyone: it is not a panic, and Windows
1525/// reports only `0xC000041D` for an exception in a window callback. This
1526/// installs an unhandled-exception filter that names the exception code
1527/// and the module the faulting address is in, then lets the crash go on;
1528/// it is a diagnostic, not a recovery. A crash reporter the host installed
1529/// first is still called, with its answer returned; one installed after
1530/// replaces this filter.
1531///
1532/// [`Gpu::new`] calls it, so a runner rarely needs to. Installing it more
1533/// than once is harmless.
1534#[cfg(windows)]
1535pub fn report_faults() {
1536    use std::cell::Cell;
1537    use windows::Win32::Foundation::{
1538        EXCEPTION_ACCESS_VIOLATION, EXCEPTION_ILLEGAL_INSTRUCTION, EXCEPTION_IN_PAGE_ERROR,
1539        EXCEPTION_STACK_OVERFLOW, HMODULE, NTSTATUS, STATUS_FATAL_USER_CALLBACK_EXCEPTION,
1540    };
1541    use windows::Win32::Storage::FileSystem::WriteFile;
1542    use windows::Win32::System::Console::{GetStdHandle, STD_ERROR_HANDLE};
1543    use windows::Win32::System::Diagnostics::Debug::{
1544        AddVectoredExceptionHandler, EXCEPTION_POINTERS, EXCEPTION_RECORD,
1545        LPTOP_LEVEL_EXCEPTION_FILTER, SetUnhandledExceptionFilter,
1546    };
1547    use windows::Win32::System::LibraryLoader::{
1548        GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS, GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
1549        GetModuleFileNameW, GetModuleHandleExW,
1550    };
1551
1552    const CONTINUE_SEARCH: i32 = 0;
1553    /// The faults a driver ends a process with: the ones remembered for
1554    /// a callback's `0xC000041D` to be read by.
1555    const FAULTS: [NTSTATUS; 4] = [
1556        EXCEPTION_ACCESS_VIOLATION,
1557        EXCEPTION_ILLEGAL_INSTRUCTION,
1558        EXCEPTION_IN_PAGE_ERROR,
1559        EXCEPTION_STACK_OVERFLOW,
1560    ];
1561    /// The filter this one replaced, called after it; set once, with the
1562    /// two handlers, by the one call that installs them.
1563    static PREVIOUS: std::sync::OnceLock<LPTOP_LEVEL_EXCEPTION_FILTER> = std::sync::OnceLock::new();
1564    thread_local! {
1565        /// The last fault this thread saw, code and address, handled or
1566        /// not. A `const` cell with no destructor: a plain thread-local
1567        /// slot, read and written without allocating or registering
1568        /// anything, from inside an exception.
1569        static LAST: Cell<Option<(i32, usize)>> = const { Cell::new(None) };
1570    }
1571
1572    /// Remembers a fault; says nothing and handles nothing.
1573    unsafe extern "system" fn remember(info: *mut EXCEPTION_POINTERS) -> i32 {
1574        // SAFETY: the system hands a valid record for the exception.
1575        if let Some(record) =
1576            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
1577            && FAULTS.contains(&record.ExceptionCode)
1578        {
1579            let seen = (record.ExceptionCode.0, record.ExceptionAddress as usize);
1580            let _ = LAST.try_with(|l| l.set(Some(seen)));
1581        }
1582        CONTINUE_SEARCH
1583    }
1584
1585    /// The first fault on a record's chain of nested exceptions, past the
1586    /// record itself; a few links, since a chain is one or two long and a
1587    /// broken one is not worth following further.
1588    fn nested(record: &EXCEPTION_RECORD) -> Option<(i32, usize)> {
1589        let mut at = record.ExceptionRecord;
1590        for _ in 0..4 {
1591            // SAFETY: a nested record the system chained to this one.
1592            let inner = unsafe { at.as_ref() }?;
1593            if FAULTS.contains(&inner.ExceptionCode) {
1594                return Some((inner.ExceptionCode.0, inner.ExceptionAddress as usize));
1595            }
1596            at = inner.ExceptionRecord;
1597        }
1598        None
1599    }
1600
1601    /// Writes one crash's line to stderr, straight to the handle: no
1602    /// `eprintln!`, which takes a lock and, on a console, converts
1603    /// through a stack buffer eight kilobytes deep — more than a stack
1604    /// overflow leaves.
1605    fn say(code: i32, at: usize, escaped: Option<i32>) {
1606        let mut module = HMODULE::default();
1607        let mut name = [0u16; 260];
1608        // SAFETY: `at` is only looked up, never read; the buffers are ours.
1609        let found = unsafe {
1610            GetModuleHandleExW(
1611                GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS
1612                    | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
1613                windows::core::PCWSTR(at as *const u16),
1614                &mut module,
1615            )
1616        }
1617        .is_ok();
1618        let path = found.then(|| {
1619            // SAFETY: the module was just found; the buffer is ours.
1620            let n = unsafe { GetModuleFileNameW(Some(module), &mut name) } as usize;
1621            &name[..n.min(name.len())]
1622        });
1623        let line = FaultLine::new(code as u32, at, path, escaped.map(|c| c as u32));
1624        // SAFETY: a handle the process was given, written from our buffer.
1625        if let Ok(err) = unsafe { GetStdHandle(STD_ERROR_HANDLE) } {
1626            let mut written = 0u32;
1627            let _ = unsafe { WriteFile(err, Some(line.bytes()), Some(&mut written), None) };
1628        }
1629    }
1630
1631    /// The crash: said, then handed to the filter before this one.
1632    unsafe extern "system" fn filter(info: *const EXCEPTION_POINTERS) -> i32 {
1633        // SAFETY: the system hands a valid record for the exception.
1634        if let Some(record) =
1635            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
1636        {
1637            let (code, at) = (record.ExceptionCode, record.ExceptionAddress as usize);
1638            let inner = if code == STATUS_FATAL_USER_CALLBACK_EXCEPTION {
1639                nested(record).or_else(|| LAST.try_with(Cell::get).ok().flatten())
1640            } else {
1641                None
1642            };
1643            match inner {
1644                Some((fault, fault_at)) => say(fault, fault_at, Some(code.0)),
1645                None => say(code.0, at, None),
1646            }
1647        }
1648        match PREVIOUS.get().copied().flatten() {
1649            // SAFETY: the filter the system held before ours, called as
1650            // the system would have called it.
1651            Some(previous) => unsafe { previous(info) },
1652            None => CONTINUE_SEARCH,
1653        }
1654    }
1655
1656    PREVIOUS.get_or_init(|| {
1657        // SAFETY: both handlers read only what the system gives them and
1658        // write only their own thread-local slot and stderr.
1659        unsafe {
1660            AddVectoredExceptionHandler(0, Some(remember));
1661            SetUnhandledExceptionFilter(Some(filter))
1662        }
1663    });
1664}
1665
1666/// Prints the code and module of a crash to stderr before the process dies (Windows only).
1667///
1668/// On Windows a fault in a GPU driver ends the process with no line from
1669/// anyone, and this installs the exception filter that names it. On every
1670/// other platform it does nothing. [`Gpu::new`] calls it, so a runner
1671/// rarely needs to.
1672#[cfg(not(windows))]
1673pub fn report_faults() {}
1674
1675/// One crash's line for `report_faults`, written into a buffer on the
1676/// stack: it is said with whatever stack the crash left (a stack
1677/// overflow leaves the few pages the thread reserved for its handlers)
1678/// and in a process whose heap may be what faulted, so nothing here
1679/// allocates. A line too long for it is cut, at a character, and still
1680/// ends in a newline. Built on every platform so it is tested on every
1681/// platform; only Windows says one.
1682#[cfg_attr(not(windows), allow(dead_code))]
1683struct FaultLine {
1684    buf: [u8; 640],
1685    len: usize,
1686}
1687
1688#[cfg_attr(not(windows), allow(dead_code))]
1689impl FaultLine {
1690    /// `kui: fault <code> at <address> in <module>`, and for a fault that
1691    /// escaped a window callback, the code it escaped as. `module` is the
1692    /// UTF-16 path Windows gives; none is a fault outside any module.
1693    fn new(code: u32, at: usize, module: Option<&[u16]>, escaped: Option<u32>) -> Self {
1694        use std::fmt::Write;
1695        let mut line = Self {
1696            buf: [0; 640],
1697            len: 0,
1698        };
1699        let _ = write!(line, "kui: fault {code:#010x} at {at:#x} in ");
1700        match module {
1701            Some(path) => {
1702                for c in char::decode_utf16(path.iter().copied()) {
1703                    let _ = line.write_char(c.unwrap_or(char::REPLACEMENT_CHARACTER));
1704                }
1705            }
1706            None => {
1707                let _ = line.write_str("no module (jit or freed code)");
1708            }
1709        }
1710        if let Some(escaped) = escaped {
1711            let _ = write!(line, ", escaped from a window callback as {escaped:#010x}");
1712        }
1713        // The newline has its byte kept for it (`write_str`).
1714        line.buf[line.len] = b'\n';
1715        line.len += 1;
1716        line
1717    }
1718
1719    fn bytes(&self) -> &[u8] {
1720        &self.buf[..self.len]
1721    }
1722}
1723
1724impl std::fmt::Write for FaultLine {
1725    fn write_str(&mut self, s: &str) -> std::fmt::Result {
1726        // One byte short of the buffer, for the newline.
1727        let room = self.buf.len() - 1 - self.len;
1728        let mut n = s.len().min(room);
1729        while !s.is_char_boundary(n) {
1730            n -= 1;
1731        }
1732        self.buf[self.len..self.len + n].copy_from_slice(&s.as_bytes()[..n]);
1733        self.len += n;
1734        Ok(())
1735    }
1736}
1737
1738#[cfg(test)]
1739mod tests {
1740    use super::*;
1741
1742    /// The globals are one buffer read by two pipelines whose modules
1743    /// declare it separately: this crate's `shader.wgsl` for quads, and
1744    /// `kui_core::fragment::PRELUDE` for every fragment. If the two
1745    /// declarations drift, a fragment reads the wrong bytes and there is
1746    /// nothing to catch it at runtime — the buffer is the right size and
1747    /// the numbers are just wrong. So: same field names, same order, and
1748    /// the size the Rust struct actually is.
1749    #[test]
1750    fn globals_layout_matches() {
1751        let fields = ["viewport", "atlas_size", "time", "scale", "_pad"];
1752        let of = |src: &str, name: &str| {
1753            let start = src
1754                .find(name)
1755                .unwrap_or_else(|| panic!("{name} is not declared in\n{src}"));
1756            let body = &src[start..];
1757            let end = body.find('}').expect("a closing brace");
1758            body[..end].to_string()
1759        };
1760        let quads = of(include_str!("shader.wgsl"), "struct Globals {");
1761        let frags = of(kui_core::fragment::PRELUDE, "struct KuiGlobals {");
1762        let read = |body: &str| -> Vec<String> {
1763            body.lines()
1764                .filter_map(|l| l.split_once(':'))
1765                .map(|(name, ty)| format!("{}: {}", name.trim(), ty.trim().trim_end_matches(',')))
1766                .collect()
1767        };
1768        let (a, b) = (read(&quads), read(&frags));
1769        assert_eq!(a, b, "shader.wgsl and the fragment prelude disagree");
1770        assert_eq!(
1771            a.len(),
1772            fields.len(),
1773            "a field was added to the globals without this test being told"
1774        );
1775        for (row, want) in a.iter().zip(fields) {
1776            assert!(row.starts_with(want), "expected {want}, got {row}");
1777        }
1778        // vec2 + vec2 + f32 + f32 + vec2 = 32 bytes, and a uniform's size
1779        // must be a multiple of sixteen, which is what `_pad` is for.
1780        assert_eq!(std::mem::size_of::<Globals>(), 32);
1781    }
1782
1783    /// The fragment parameter slot is the other buffer two declarations
1784    /// read: `FragmentParams` here and `KuiFragmentParams` in the
1785    /// epilogue. Sixteen floats then the image's rect, 80 bytes.
1786    #[test]
1787    fn fragment_params_layout_matches() {
1788        assert_eq!(std::mem::size_of::<FragmentParams>(), 80);
1789        assert_eq!(std::mem::offset_of!(FragmentParams, image), 64);
1790        let epilogue = kui_core::fragment::EPILOGUE;
1791        assert!(
1792            epilogue
1793                .contains("struct KuiFragmentParams { p: array<vec4<f32>, 4>, image: vec4<f32> };"),
1794            "the epilogue's params struct moved without this test being told"
1795        );
1796    }
1797
1798    /// The bindings the prelude declares at group 0 are this crate's, by
1799    /// number and kind, since a fragment pipeline binds the quad
1800    /// pipeline's group 0 layout as it is.
1801    #[test]
1802    fn prelude_bindings_match_group_zero() {
1803        let prelude = kui_core::fragment::PRELUDE;
1804        for line in [
1805            "@group(0) @binding(0) var<uniform> kui_globals: KuiGlobals;",
1806            "@group(0) @binding(1) var kui_atlas: texture_2d<f32>;",
1807            "@group(0) @binding(2) var kui_sampler: sampler;",
1808            "@group(0) @binding(3) var kui_sampler_nearest: sampler;",
1809        ] {
1810            assert!(prelude.contains(line), "prelude lacks `{line}`");
1811        }
1812        let quads = include_str!("shader.wgsl");
1813        for line in [
1814            "@group(0) @binding(1) var atlas_tex: texture_2d<f32>;",
1815            "@group(0) @binding(2) var atlas_smp: sampler;",
1816            "@group(0) @binding(3) var nearest_smp: sampler;",
1817        ] {
1818            assert!(quads.contains(line), "shader.wgsl lacks `{line}`");
1819        }
1820    }
1821
1822    /// Both preprocessed variants of the shader must parse and validate
1823    /// (pipeline creation would otherwise fail at runtime, in a window).
1824    #[test]
1825    fn shader_variants_validate() {
1826        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
1827        for dual in [false, true] {
1828            let src = preprocess_shader(include_str!("shader.wgsl"), dual);
1829            let module = wgpu::naga::front::wgsl::parse_str(&src)
1830                .unwrap_or_else(|e| panic!("dual={dual}: {}", e.emit_to_string(&src)));
1831            let caps = if dual {
1832                Capabilities::DUAL_SOURCE_BLENDING
1833            } else {
1834                Capabilities::empty()
1835            };
1836            Validator::new(ValidationFlags::all(), caps)
1837                .validate(&module)
1838                .unwrap_or_else(|e| panic!("dual={dual}: {e:?}"));
1839        }
1840    }
1841
1842    /// The line `report_faults` says, built without the heap (RG31): the
1843    /// code, the address and the module's UTF-16 path decoded, and for a
1844    /// fault that escaped a window callback the code it escaped as.
1845    #[test]
1846    fn a_fault_line_names_the_code_the_address_and_the_module() {
1847        let path: Vec<u16> = r"C:\Windows\System32\nvoglv64.dll".encode_utf16().collect();
1848        let line = FaultLine::new(0xC000_0005, 0x7ff6_1234, Some(&path), None);
1849        assert_eq!(
1850            std::str::from_utf8(line.bytes()).unwrap(),
1851            "kui: fault 0xc0000005 at 0x7ff61234 in C:\\Windows\\System32\\nvoglv64.dll\n"
1852        );
1853        let line = FaultLine::new(0xC000_0005, 0x10, None, Some(0xC000_041D));
1854        assert_eq!(
1855            std::str::from_utf8(line.bytes()).unwrap(),
1856            "kui: fault 0xc0000005 at 0x10 in no module (jit or freed code), \
1857             escaped from a window callback as 0xc000041d\n"
1858        );
1859        // A path that is not UTF-16 is said, not refused.
1860        let line = FaultLine::new(0xC000_001D, 0x20, Some(&[0x44, 0xD800, 0x45]), None);
1861        assert_eq!(
1862            std::str::from_utf8(line.bytes()).unwrap(),
1863            "kui: fault 0xc000001d at 0x20 in D\u{FFFD}E\n"
1864        );
1865    }
1866
1867    /// A line longer than its stack buffer is cut, between characters,
1868    /// and still ends in its newline.
1869    #[test]
1870    fn a_fault_line_too_long_is_cut_at_a_character() {
1871        let path: Vec<u16> = "é".repeat(1000).encode_utf16().collect();
1872        let line = FaultLine::new(0xC000_00FD, 0x30, Some(&path), Some(0xC000_041D));
1873        let text = std::str::from_utf8(line.bytes()).expect("cut at a character");
1874        assert!(text.ends_with("é\n"), "{text:?}");
1875        assert!(text.len() <= 640 && text.len() >= 638, "{}", text.len());
1876        assert!(text.starts_with("kui: fault 0xc00000fd at 0x30 in é"));
1877    }
1878}