Skip to main content

kui_wgpu/
lib.rs

1//! wgpu backend: one über-pipeline drawing instanced quads (rounded rects,
2//! borders, and atlas glyphs unified), so a whole UI is a single draw call.
3//! Consumes `kui_core::DisplayList` and mirrors the core's glyph atlas.
4//!
5//! Where the device offers dual-source blending (Metal, DX12, most Vulkan)
6//! the pipeline blends per channel, which is what LCD subpixel text needs:
7//! the fragment shader emits premultiplied color plus a per-channel
8//! coverage, and the blend is `src + dst * (1 - coverage)` channel-wise.
9//! Otherwise it falls back to ordinary alpha blending and subpixel glyphs
10//! draw from their union coverage (grayscale).
11
12pub use wgpu;
13
14use kui_core::atlas::GlyphAtlas;
15use kui_core::{Clip, DisplayList, Quad, QuadKind};
16
17#[repr(C)]
18#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
19struct Instance {
20    pos: [f32; 2],
21    size: [f32; 2],
22    color: [f32; 4],
23    border_color: [f32; 4],
24    params: [f32; 4],
25    uv: [f32; 4],
26    clip: [f32; 4],
27    /// Corner radii, clockwise from the top-left.
28    radii: [f32; 4],
29    /// Radii of the clip itself; all zero = a plain rect clip.
30    clip_radii: [f32; 4],
31}
32
33/// The frame's own numbers, at group 0 binding 0 for both pipelines.
34/// `kui_core::fragment::PRELUDE` declares the same bytes as `KuiGlobals`
35/// so an app's fragment can read `time` and `scale`; the padding is what
36/// makes the struct a multiple of sixteen, which a uniform must be.
37#[repr(C)]
38#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
39struct Globals {
40    viewport: [f32; 2],
41    atlas_size: [f32; 2],
42    time: f32,
43    scale: f32,
44    _pad: [f32; 2],
45}
46
47/// One fragment's parameters as the shader takes them — the sixteen
48/// floats and the texel rect of its `image`, laid out as the epilogue's
49/// `KuiFragmentParams` — padded out to the device's dynamic-offset
50/// alignment so a frame's draws can share one buffer and pick their slot
51/// by offset.
52#[repr(C)]
53#[derive(Clone, Copy, bytemuck::Pod, bytemuck::Zeroable)]
54struct FragmentParams {
55    params: [f32; 16],
56    /// `FragmentIn::image`: `[x, y, w, h]` in the texture bound at group 0
57    /// for this draw — the atlas, or the image's own; zero with none.
58    image: [f32; 4],
59}
60
61/// The clip is resolved out of the frame's table here rather than read off
62/// the quad: it rides as an index (`kui_core::ClipId`) so the display list
63/// carries it once per distinct clip instead of once per quad.
64fn instance_of(q: &Quad, clips: &[Clip], textures: &[kui_core::display::TextureDraw]) -> Instance {
65    let clip = clips.get(q.clip as usize).copied().unwrap_or(Clip::NONE);
66    let kind = match q.kind {
67        QuadKind::Solid => 0.0,
68        QuadKind::GlyphMask => 1.0,
69        QuadKind::GlyphColor => 2.0,
70        QuadKind::Image => 3.0,
71        QuadKind::GlyphSubpixel => 4.0,
72        QuadKind::Shadow => 5.0,
73        QuadKind::Segment => 6.0,
74        QuadKind::Fragment => 7.0,
75        // Drawn by the image branch with its own texture bound in the
76        // atlas's place (ADR 0025, decision 3).
77        QuadKind::Texture => 3.0,
78    };
79    // `uv` is atlas texels on every kind but two: a segment carries its
80    // endpoints there as f32 bits, and a texture quad an index into the
81    // side list whose entry holds the texel rect. The shader wants floats.
82    let uv = if q.kind == QuadKind::Segment {
83        q.segment_ends()
84    } else if q.kind == QuadKind::Texture {
85        let uv = textures.get(q.uv[0] as usize).map_or([0; 4], |t| t.uv);
86        [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32]
87    } else {
88        [
89            q.uv[0] as f32,
90            q.uv[1] as f32,
91            q.uv[2] as f32,
92            q.uv[3] as f32,
93        ]
94    };
95    Instance {
96        pos: [q.rect.x, q.rect.y],
97        size: [q.rect.w, q.rect.h],
98        color: [q.color.r, q.color.g, q.color.b, q.color.a],
99        border_color: [
100            q.border_color.r,
101            q.border_color.g,
102            q.border_color.b,
103            q.border_color.a,
104        ],
105        params: [q.blur, q.border_w, kind, 0.0],
106        uv,
107        clip: [clip.rect.x, clip.rect.y, clip.rect.w, clip.rect.h],
108        radii: q.radius,
109        clip_radii: clip.radius,
110    }
111}
112
113/// The GPU objects a session's windows share: one instance, one adapter,
114/// one device, one queue. Windows must share a device before a second one
115/// is worth opening — two devices cannot see each other's buffers or
116/// textures, and each costs a driver context — so a `Renderer` holds a
117/// handle to one rather than making its own. Cloning a `Gpu` clones the
118/// handle; `Renderer::new` makes a private one for its window, which is
119/// what a single-window app gets and never has to name.
120#[derive(Clone)]
121pub struct Gpu(std::sync::Arc<GpuInner>);
122
123struct GpuInner {
124    instance: wgpu::Instance,
125    adapter: wgpu::Adapter,
126    device: wgpu::Device,
127    queue: wgpu::Queue,
128    dual_source: bool,
129    /// One pipeline per registered fragment per surface format, built the
130    /// first time a frame draws it and shared by every window on this
131    /// device — the cost the ADR measured at about 0.2 ms, paid once —
132    /// and dropped when a frame's list says the handle is gone
133    /// (`dropped_fragments`). A `Mutex` because `Gpu` is a shared handle
134    /// and building is rare; nothing here is touched on a frame that
135    /// draws no new fragment.
136    fragment_pipelines: std::sync::Mutex<
137        std::collections::HashMap<(u64, wgpu::TextureFormat), wgpu::RenderPipeline>,
138    >,
139    /// One texture per texture-backed image, uploaded the first time a
140    /// frame on this device draws it and again when its revision moves,
141    /// shared by every window like the pipelines above, dropped when the
142    /// core says the handle is gone (ADR 0025, decisions 2 and 3).
143    textures: std::sync::Mutex<std::collections::HashMap<u64, std::sync::Arc<ImageTexture>>>,
144    /// Set by the device's lost callback: a driver update, a GPU reset, a
145    /// hang the OS answered by removing the device. Nothing on it works
146    /// again; a shell opens a new one ([`Gpu::lost`]).
147    lost: std::sync::Arc<std::sync::atomic::AtomicBool>,
148}
149
150/// A texture-backed image on the device: the texture, and what was
151/// uploaded into it. A new `Arc` is made when the size changes, which is
152/// what tells a renderer its bind group is stale.
153struct ImageTexture {
154    texture: wgpu::Texture,
155    view: wgpu::TextureView,
156    width: u32,
157    height: u32,
158    /// The revision the pixels in the texture came from, behind a lock
159    /// because the texture is shared and the upload is per device.
160    rev: std::sync::Mutex<u32>,
161}
162
163impl Gpu {
164    /// Opens the shared device, choosing an adapter that can present to
165    /// `target`'s surface — the first window's, whose surface comes back
166    /// with it because it has to exist before the adapter can be picked.
167    /// Every later window's surface comes from [`Gpu::create_surface`].
168    pub async fn new(
169        target: impl Into<wgpu::SurfaceTarget<'static>>,
170    ) -> Result<(Self, wgpu::Surface<'static>), Box<dyn std::error::Error>> {
171        report_faults();
172        // Every backend the build has, as wgpu defaults — but on Windows
173        // D3D12 alone unless `WGPU_BACKEND` names another. An instance
174        // keeps every backend it enumerated alive for as long as it lives,
175        // so with all of them the process holds an OpenGL context and a
176        // Vulkan instance it never draws with, both in the driver's
177        // `nvoglv64.dll`; and wgpu, left to choose, took Vulkan over D3D12
178        // here. Under a driver update that DLL faulted in present rather
179        // than answer `DEVICE_LOST`, which ended the process; D3D12's
180        // `nvwgf2umx.dll` reports the removal, and the shell reopens the
181        // device (`Gpu::lost`).
182        let mut desc = wgpu::InstanceDescriptor::new_without_display_handle_from_env();
183        if cfg!(windows) && std::env::var_os("WGPU_BACKEND").is_none() {
184            desc.backends = wgpu::Backends::DX12;
185        }
186        let instance = wgpu::Instance::new(desc);
187        let surface = instance.create_surface(target)?;
188        let adapter = instance
189            .request_adapter(&wgpu::RequestAdapterOptions {
190                compatible_surface: Some(&surface),
191                ..Default::default()
192            })
193            .await?;
194        // Per-channel blending for LCD subpixel text, when the device has it.
195        let dual_source = adapter
196            .features()
197            .contains(wgpu::Features::DUAL_SOURCE_BLENDING);
198        let (device, queue) = adapter
199            .request_device(&wgpu::DeviceDescriptor {
200                required_features: if dual_source {
201                    wgpu::Features::DUAL_SOURCE_BLENDING
202                } else {
203                    wgpu::Features::empty()
204                },
205                ..Default::default()
206            })
207            .await?;
208        // What goes wrong on the device is said, not swallowed: an error
209        // outside a scope, and the loss of the device itself — remembered
210        // too, so a frame can tell a dead device from a stale swapchain.
211        device.on_uncaptured_error(std::sync::Arc::new(|e| eprintln!("kui: wgpu: {e}")));
212        let lost = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
213        device.set_device_lost_callback({
214            let lost = lost.clone();
215            move |reason, message| {
216                if reason == wgpu::DeviceLostReason::Unknown {
217                    eprintln!("kui: device lost: {message}");
218                    lost.store(true, std::sync::atomic::Ordering::Release);
219                }
220            }
221        });
222        let gpu = Self(std::sync::Arc::new(GpuInner {
223            instance,
224            adapter,
225            device,
226            queue,
227            dual_source,
228            fragment_pipelines: Default::default(),
229            textures: Default::default(),
230            lost,
231        }));
232        Ok((gpu, surface))
233    }
234
235    /// Whether the device is gone — a driver update or a GPU reset took
236    /// it — so every renderer on it is to be opened again on a new one
237    /// ([`Renderer::render`] says so with [`RenderError::DeviceLost`]).
238    pub fn lost(&self) -> bool {
239        self.0.lost.load(std::sync::atomic::Ordering::Acquire)
240    }
241
242    /// Loses the device on purpose, as a driver update or a GPU reset
243    /// would — for a shell to see its reopening happen without one. On
244    /// D3D12 the device is really removed (`ID3D12Device5::RemoveDevice`),
245    /// so every resource on it dies as it does then and the lost callback
246    /// runs as it does then; elsewhere the device is only treated as lost.
247    pub fn mark_lost(&self) {
248        #[cfg(windows)]
249        {
250            use windows::Win32::Graphics::Direct3D12::ID3D12Device5;
251            use windows::core::Interface;
252            // SAFETY: the hal device is only read for its raw handle, and
253            // `RemoveDevice` is what D3D12 offers for exactly this.
254            let removed = unsafe {
255                self.0
256                    .device
257                    .as_hal::<wgpu::hal::api::Dx12>()
258                    .and_then(|d| d.raw_device().cast::<ID3D12Device5>().ok())
259                    .map(|d| d.RemoveDevice())
260            };
261            if removed.is_some() {
262                // The loss lands on the device's next use, through the
263                // lost callback, as a real one does.
264                return;
265            }
266        }
267        self.0
268            .lost
269            .store(true, std::sync::atomic::Ordering::Release);
270    }
271
272    /// A surface for another window on the same instance — what
273    /// [`Renderer::new_in`] draws into.
274    pub fn create_surface(
275        &self,
276        target: impl Into<wgpu::SurfaceTarget<'static>>,
277    ) -> Result<wgpu::Surface<'static>, wgpu::CreateSurfaceError> {
278        self.0.instance.create_surface(target)
279    }
280
281    pub fn instance(&self) -> &wgpu::Instance {
282        &self.0.instance
283    }
284
285    pub fn adapter(&self) -> &wgpu::Adapter {
286        &self.0.adapter
287    }
288
289    pub fn device(&self) -> &wgpu::Device {
290        &self.0.device
291    }
292
293    pub fn queue(&self) -> &wgpu::Queue {
294        &self.0.queue
295    }
296
297    /// Whether this device blends per channel, i.e. LCD subpixel glyphs
298    /// draw with per-channel coverage rather than their union.
299    pub fn dual_source(&self) -> bool {
300        self.0.dual_source
301    }
302
303    /// The texture for one texture-backed image, uploaded on first sight
304    /// and whenever `rev` has moved past what the texture holds; a size
305    /// change makes a new texture. `None` for a degenerate size, which
306    /// draws nothing.
307    fn image_texture(
308        &self,
309        id: u64,
310        px: &kui_core::display::TexturePixels,
311    ) -> Option<std::sync::Arc<ImageTexture>> {
312        // Degenerate, or past what this device can hold in one texture
313        // (8192 on many adapters, 16384 on Metal): draws nothing, which is
314        // what the core says a texture-backed image that cannot be backed
315        // does, rather than a validation error the device turns into a
316        // panic.
317        let max = self.0.device.limits().max_texture_dimension_2d;
318        if px.width == 0 || px.height == 0 || px.width > max || px.height > max {
319            return None;
320        }
321        let mut cache = self.0.textures.lock().unwrap_or_else(|e| e.into_inner());
322        let fresh = match cache.get(&id) {
323            Some(t) if t.width == px.width && t.height == px.height => {
324                let mut rev = t.rev.lock().unwrap_or_else(|e| e.into_inner());
325                if *rev != px.rev {
326                    upload_image(&self.0.queue, &t.texture, px);
327                    *rev = px.rev;
328                }
329                return Some(t.clone());
330            }
331            _ => {
332                let texture = self.0.device.create_texture(&wgpu::TextureDescriptor {
333                    label: Some("kui.image"),
334                    size: wgpu::Extent3d {
335                        width: px.width,
336                        height: px.height,
337                        depth_or_array_layers: 1,
338                    },
339                    mip_level_count: 1,
340                    sample_count: 1,
341                    dimension: wgpu::TextureDimension::D2,
342                    format: wgpu::TextureFormat::Rgba8Unorm,
343                    usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
344                    view_formats: &[],
345                });
346                upload_image(&self.0.queue, &texture, px);
347                let view = texture.create_view(&wgpu::TextureViewDescriptor::default());
348                std::sync::Arc::new(ImageTexture {
349                    texture,
350                    view,
351                    width: px.width,
352                    height: px.height,
353                    rev: std::sync::Mutex::new(px.rev),
354                })
355            }
356        };
357        cache.insert(id, fresh.clone());
358        Some(fresh)
359    }
360
361    /// Forgets a removed fragment's pipelines, one per surface format it
362    /// was ever drawn in; the GPU frees them once no frame in flight
363    /// holds one.
364    fn drop_fragment_pipelines(&self, id: u64) {
365        self.0
366            .fragment_pipelines
367            .lock()
368            .unwrap_or_else(|e| e.into_inner())
369            .retain(|(fid, _), _| *fid != id);
370    }
371
372    /// Forgets a removed image's texture; the GPU frees it once no bind
373    /// group holds it.
374    fn drop_image_texture(&self, id: u64) {
375        self.0
376            .textures
377            .lock()
378            .unwrap_or_else(|e| e.into_inner())
379            .remove(&id);
380    }
381
382    /// Whether the cache still holds exactly this texture for `id` — what
383    /// a renderer asks before keeping a bind group over it, since a
384    /// removal reaches the cache through whichever window's frame carried
385    /// it and the other windows' bind groups would otherwise hold the
386    /// texture for as long as they live.
387    fn holds_image_texture(&self, id: u64, texture: &std::sync::Arc<ImageTexture>) -> bool {
388        self.0
389            .textures
390            .lock()
391            .unwrap_or_else(|e| e.into_inner())
392            .get(&id)
393            .is_some_and(|t| std::sync::Arc::ptr_eq(t, texture))
394    }
395
396    /// The pipeline for one registered fragment, built on first sight and
397    /// then shared by every window on this device. `source` is the app's
398    /// WGSL, which the core already validated; it is wrapped in the same
399    /// prelude and epilogue here, from `kui_core::fragment::module_source`,
400    /// so what compiles is what was validated.
401    ///
402    /// The source is not validated again here: `Core::add_fragment` parsed
403    /// and validated this exact module text with the same naga this wgpu
404    /// carries, and refused a handle for anything that failed. A module
405    /// that still does not compile is a kui bug, and reaches wgpu's own
406    /// error handler like any other.
407    fn fragment_pipeline(
408        &self,
409        id: u64,
410        source: &str,
411        format: wgpu::TextureFormat,
412        layouts: &FragmentLayouts,
413    ) -> wgpu::RenderPipeline {
414        let mut cache = self
415            .0
416            .fragment_pipelines
417            .lock()
418            .unwrap_or_else(|e| e.into_inner());
419        if let Some(p) = cache.get(&(id, format)) {
420            return p.clone();
421        }
422        let device = &self.0.device;
423        let module = device.create_shader_module(wgpu::ShaderModuleDescriptor {
424            label: Some("kui.fragment"),
425            source: wgpu::ShaderSource::Wgsl(kui_core::fragment::module_source(source).into()),
426        });
427        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
428            label: Some("kui.fragment"),
429            layout: Some(&layouts.pipeline),
430            vertex: wgpu::VertexState {
431                module: &layouts.vertex,
432                entry_point: Some("vs_main"),
433                compilation_options: Default::default(),
434                buffers: &[Some(instance_buffer_layout(&INSTANCE_ATTRS))],
435            },
436            fragment: Some(wgpu::FragmentState {
437                module: &module,
438                entry_point: Some(kui_core::fragment::ENTRY_POINT),
439                compilation_options: Default::default(),
440                targets: &[Some(wgpu::ColorTargetState {
441                    format,
442                    // A fragment returns premultiplied colour, always over.
443                    blend: Some(wgpu::BlendState {
444                        color: wgpu::BlendComponent {
445                            src_factor: wgpu::BlendFactor::One,
446                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
447                            operation: wgpu::BlendOperation::Add,
448                        },
449                        alpha: wgpu::BlendComponent {
450                            src_factor: wgpu::BlendFactor::One,
451                            dst_factor: wgpu::BlendFactor::OneMinusSrcAlpha,
452                            operation: wgpu::BlendOperation::Add,
453                        },
454                    }),
455                    write_mask: wgpu::ColorWrites::ALL,
456                })],
457            }),
458            primitive: wgpu::PrimitiveState::default(),
459            depth_stencil: None,
460            multisample: wgpu::MultisampleState::default(),
461            multiview_mask: None,
462            cache: None,
463        });
464        cache.insert((id, format), pipeline.clone());
465        pipeline
466    }
467}
468
469/// What building a fragment pipeline needs besides its own source: kui's
470/// vertex stage, and the layout that puts the globals at group 0 and the
471/// parameters at group 1.
472struct FragmentLayouts {
473    vertex: wgpu::ShaderModule,
474    pipeline: wgpu::PipelineLayout,
475}
476
477/// The instance attributes both pipelines read; one array so the vertex
478/// layout cannot differ between them.
479const INSTANCE_ATTRS: [wgpu::VertexAttribute; 9] = wgpu::vertex_attr_array![
480    0 => Float32x2, 1 => Float32x2, 2 => Float32x4,
481    3 => Float32x4, 4 => Float32x4, 5 => Float32x4,
482    6 => Float32x4, 7 => Float32x4, 8 => Float32x4,
483];
484
485fn instance_buffer_layout(attrs: &[wgpu::VertexAttribute]) -> wgpu::VertexBufferLayout<'_> {
486    wgpu::VertexBufferLayout {
487        array_stride: std::mem::size_of::<Instance>() as u64,
488        step_mode: wgpu::VertexStepMode::Instance,
489        attributes: attrs,
490    }
491}
492
493impl std::fmt::Debug for Gpu {
494    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
495        f.debug_struct("Gpu")
496            .field("adapter", &self.0.adapter.get_info().name)
497            .field("dual_source", &self.0.dual_source)
498            .finish()
499    }
500}
501
502pub struct Renderer {
503    gpu: Gpu,
504    surface: wgpu::Surface<'static>,
505    config: wgpu::SurfaceConfiguration,
506    pipeline: wgpu::RenderPipeline,
507    globals_buf: wgpu::Buffer,
508    bind_group: wgpu::BindGroup,
509    bind_layout: wgpu::BindGroupLayout,
510    /// The two samplers every group-0 bind group carries: linear at
511    /// binding 2, nearest at 3 (ADR 0025, decision 4).
512    samplers: Samplers,
513    /// Per texture-backed image this window has drawn: the device's
514    /// texture, and a bind group of this window's own — group 0 with that
515    /// texture in the atlas's place and a globals copy whose `atlas_size`
516    /// is the texture's, rewritten each frame the image is drawn.
517    texture_binds: std::collections::HashMap<u64, TextureBind>,
518    atlas_tex: wgpu::Texture,
519    atlas_size: u32,
520    atlas_epoch: u64,
521    instance_buf: wgpu::Buffer,
522    instance_cap: usize,
523    instances: Vec<Instance>,
524    /// What a fragment pipeline is built against: kui's vertex stage and
525    /// the two-group layout. Built once per renderer, handed to the
526    /// device's shared cache.
527    fragment_layouts: FragmentLayouts,
528    /// One slot per fragment this frame, each padded to the device's
529    /// dynamic-offset alignment.
530    fragment_params_buf: wgpu::Buffer,
531    fragment_params_cap: usize,
532    fragment_bind: wgpu::BindGroup,
533    fragment_bind_layout: wgpu::BindGroupLayout,
534    /// The alignment slots are padded to; `dynamic_offset` steps by it.
535    uniform_align: u32,
536    /// Scratch for one frame's padded parameter slots.
537    fragment_bytes: Vec<u8>,
538    pub clear_color: wgpu::Color,
539}
540
541struct Samplers {
542    linear: wgpu::Sampler,
543    nearest: wgpu::Sampler,
544}
545
546struct TextureBind {
547    texture: std::sync::Arc<ImageTexture>,
548    globals: wgpu::Buffer,
549    bind: wgpu::BindGroup,
550}
551
552fn upload_image(
553    queue: &wgpu::Queue,
554    texture: &wgpu::Texture,
555    px: &kui_core::display::TexturePixels,
556) {
557    queue.write_texture(
558        wgpu::TexelCopyTextureInfo {
559            texture,
560            mip_level: 0,
561            origin: wgpu::Origin3d::ZERO,
562            aspect: wgpu::TextureAspect::All,
563        },
564        &px.rgba,
565        wgpu::TexelCopyBufferLayout {
566            offset: 0,
567            bytes_per_row: Some(px.width * 4),
568            rows_per_image: Some(px.height),
569        },
570        wgpu::Extent3d {
571            width: px.width,
572            height: px.height,
573            depth_or_array_layers: 1,
574        },
575    );
576}
577
578/// Picks the `//DUAL:` or `//SINGLE:` lines of the shader template.
579fn preprocess_shader(src: &str, dual: bool) -> String {
580    let (keep, drop) = if dual {
581        ("//DUAL:", "//SINGLE:")
582    } else {
583        ("//SINGLE:", "//DUAL:")
584    };
585    let mut out = String::with_capacity(src.len());
586    for line in src.lines() {
587        if let Some(rest) = line.strip_prefix(keep) {
588            out.push_str(rest);
589        } else if line.starts_with(drop) {
590            continue;
591        } else {
592            out.push_str(line);
593        }
594        out.push('\n');
595    }
596    out
597}
598
599/// How many frames may be queued ahead of the one on screen, by default:
600/// two, so a drawable to render into is waiting when the previous frame's
601/// is still out (backlog C47). On Metal wgpu makes this the layer's
602/// `maximumDrawableCount` less one, so two is triple buffering — what gpui
603/// runs with. With one, a frame whose thread woke a little late at light
604/// load found no free drawable and missed its vsync: 1–6% of them on an
605/// M3 Pro under macOS 27 at 100 and 2,500 boxes, none under heavy load,
606/// where there is no idle gap to wake late from. Two delivered 1198–1201
607/// of ~1200 vsyncs in every run. Queued behind a frame, though, a frame
608/// built as soon as a drawable frees reaches the screen a vsync later
609/// while frames run back to back — 27.6 ms sampling-to-photon against
610/// 19.3 — so `kui-native`'s runner starts such frames at the display's vsync
611/// instead (its `pacer`, macOS 14+), where the extra drawable is slack
612/// and not a queue: 17.5–19.2 ms, every vsync delivered. A renderer
613/// driven any other way pays the frame.
614///
615/// One on Windows (backlog RG46), where there is no pacer to win the frame
616/// back and nothing to win it for: D3D12's flip-model swapchain waits on
617/// its frame-latency object, and with one queued frame an RTX 5080 at
618/// 240 Hz delivered 2,400 of 2,400 vsyncs in 10 s at 100, 2,500 and
619/// 10,000 boxes rebuilt every frame, the same as with two — so two was a
620/// vsync of latency, 4.2 ms there, for nothing.
621pub const DEFAULT_FRAME_LATENCY: u32 = if cfg!(target_os = "windows") { 1 } else { 2 };
622
623impl Renderer {
624    /// A renderer for one window, on a device of its own.
625    pub async fn new(
626        target: impl Into<wgpu::SurfaceTarget<'static>>,
627        width: u32,
628        height: u32,
629    ) -> Result<Self, Box<dyn std::error::Error>> {
630        let (gpu, surface) = Gpu::new(target).await?;
631        Self::with_surface(gpu, surface, width, height)
632    }
633
634    /// A renderer for another window on an existing device — the one every
635    /// window of a session shares. Get it from [`Renderer::gpu`].
636    pub fn new_in(
637        gpu: &Gpu,
638        target: impl Into<wgpu::SurfaceTarget<'static>>,
639        width: u32,
640        height: u32,
641    ) -> Result<Self, Box<dyn std::error::Error>> {
642        let surface = gpu.create_surface(target)?;
643        Self::with_surface(gpu.clone(), surface, width, height)
644    }
645
646    /// The device this renderer draws with, to open another window on.
647    pub fn gpu(&self) -> &Gpu {
648        &self.gpu
649    }
650
651    /// How many frames may be queued ahead of the one on screen (at least
652    /// one); see [`DEFAULT_FRAME_LATENCY`]. Reconfigures the surface when
653    /// it changes.
654    pub fn set_frame_latency(&mut self, frames: u32) {
655        let frames = frames.max(1);
656        if self.config.desired_maximum_frame_latency != frames {
657            self.config.desired_maximum_frame_latency = frames;
658            self.surface.configure(self.gpu.device(), &self.config);
659        }
660    }
661
662    /// The frame latency the surface is configured with.
663    pub fn frame_latency(&self) -> u32 {
664        self.config.desired_maximum_frame_latency
665    }
666
667    fn with_surface(
668        gpu: Gpu,
669        surface: wgpu::Surface<'static>,
670        width: u32,
671        height: u32,
672    ) -> Result<Self, Box<dyn std::error::Error>> {
673        let device = gpu.device();
674        let dual_source = gpu.dual_source();
675
676        let caps = surface.get_capabilities(gpu.adapter());
677        let format = caps
678            .formats
679            .iter()
680            .copied()
681            .find(|f| !f.is_srgb())
682            .unwrap_or(caps.formats[0]);
683        let config = wgpu::SurfaceConfiguration {
684            usage: wgpu::TextureUsages::RENDER_ATTACHMENT,
685            format,
686            width: width.clamp(1, device.limits().max_texture_dimension_2d),
687            height: height.clamp(1, device.limits().max_texture_dimension_2d),
688            present_mode: wgpu::PresentMode::AutoVsync,
689            alpha_mode: caps.alpha_modes[0],
690            color_space: wgpu::SurfaceColorSpace::Auto,
691            view_formats: vec![],
692            // See `DEFAULT_FRAME_LATENCY`; a runner that wants another
693            // says so through `set_frame_latency`.
694            desired_maximum_frame_latency: DEFAULT_FRAME_LATENCY,
695        };
696        // A configure that fails only reports to the device's error
697        // handler, and the first acquire on the unconfigured surface is a
698        // panic inside wgpu; caught here, it is this constructor's error
699        // — a window DXGI will not give a second swapchain, say.
700        let scope = device.push_error_scope(wgpu::ErrorFilter::Validation);
701        surface.configure(device, &config);
702        if let Some(err) = pollster::block_on(scope.pop()) {
703            return Err(format!("configuring the surface: {err}").into());
704        }
705
706        let shader = device.create_shader_module(wgpu::ShaderModuleDescriptor {
707            label: Some("kui"),
708            source: wgpu::ShaderSource::Wgsl(
709                preprocess_shader(include_str!("shader.wgsl"), dual_source).into(),
710            ),
711        });
712        // Dual source: the shader outputs premultiplied color and a
713        // per-channel coverage; out = src + dst * (1 - coverage). For
714        // ordinary quads every channel's coverage equals alpha, which is
715        // exactly premultiplied alpha blending.
716        let blend = if dual_source {
717            wgpu::BlendState {
718                color: wgpu::BlendComponent {
719                    src_factor: wgpu::BlendFactor::One,
720                    dst_factor: wgpu::BlendFactor::OneMinusSrc1,
721                    operation: wgpu::BlendOperation::Add,
722                },
723                alpha: wgpu::BlendComponent {
724                    src_factor: wgpu::BlendFactor::One,
725                    dst_factor: wgpu::BlendFactor::OneMinusSrc1Alpha,
726                    operation: wgpu::BlendOperation::Add,
727                },
728            }
729        } else {
730            wgpu::BlendState::ALPHA_BLENDING
731        };
732
733        let bind_layout = device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
734            label: Some("kui.globals"),
735            entries: &[
736                wgpu::BindGroupLayoutEntry {
737                    binding: 0,
738                    visibility: wgpu::ShaderStages::VERTEX_FRAGMENT,
739                    ty: wgpu::BindingType::Buffer {
740                        ty: wgpu::BufferBindingType::Uniform,
741                        has_dynamic_offset: false,
742                        min_binding_size: None,
743                    },
744                    count: None,
745                },
746                wgpu::BindGroupLayoutEntry {
747                    binding: 1,
748                    visibility: wgpu::ShaderStages::FRAGMENT,
749                    ty: wgpu::BindingType::Texture {
750                        sample_type: wgpu::TextureSampleType::Float { filterable: true },
751                        view_dimension: wgpu::TextureViewDimension::D2,
752                        multisampled: false,
753                    },
754                    count: None,
755                },
756                wgpu::BindGroupLayoutEntry {
757                    binding: 2,
758                    visibility: wgpu::ShaderStages::FRAGMENT,
759                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
760                    count: None,
761                },
762                wgpu::BindGroupLayoutEntry {
763                    binding: 3,
764                    visibility: wgpu::ShaderStages::FRAGMENT,
765                    ty: wgpu::BindingType::Sampler(wgpu::SamplerBindingType::Filtering),
766                    count: None,
767                },
768            ],
769        });
770
771        let pipeline_layout = device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
772            label: Some("kui"),
773            bind_group_layouts: &[Some(&bind_layout)],
774            immediate_size: 0,
775        });
776
777        let instance_attrs = INSTANCE_ATTRS;
778        let pipeline = device.create_render_pipeline(&wgpu::RenderPipelineDescriptor {
779            label: Some("kui.quads"),
780            layout: Some(&pipeline_layout),
781            vertex: wgpu::VertexState {
782                module: &shader,
783                entry_point: Some("vs_main"),
784                compilation_options: Default::default(),
785                buffers: &[Some(instance_buffer_layout(&instance_attrs))],
786            },
787            fragment: Some(wgpu::FragmentState {
788                module: &shader,
789                entry_point: Some("fs_main"),
790                compilation_options: Default::default(),
791                targets: &[Some(wgpu::ColorTargetState {
792                    format,
793                    blend: Some(blend),
794                    write_mask: wgpu::ColorWrites::ALL,
795                })],
796            }),
797            primitive: wgpu::PrimitiveState::default(),
798            depth_stencil: None,
799            multisample: wgpu::MultisampleState::default(),
800            multiview_mask: None,
801            cache: None,
802        });
803
804        let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
805            label: Some("kui.globals"),
806            size: std::mem::size_of::<Globals>() as u64,
807            usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
808            mapped_at_creation: false,
809        });
810
811        let atlas_size = kui_core::atlas::ATLAS_SIZE;
812        let atlas_tex = create_atlas_texture(device, atlas_size);
813        let samplers = Samplers {
814            linear: device.create_sampler(&wgpu::SamplerDescriptor {
815                label: Some("kui.linear"),
816                mag_filter: wgpu::FilterMode::Linear,
817                min_filter: wgpu::FilterMode::Linear,
818                ..Default::default()
819            }),
820            nearest: device.create_sampler(&wgpu::SamplerDescriptor {
821                label: Some("kui.nearest"),
822                mag_filter: wgpu::FilterMode::Nearest,
823                min_filter: wgpu::FilterMode::Nearest,
824                ..Default::default()
825            }),
826        };
827        let atlas_view = atlas_tex.create_view(&wgpu::TextureViewDescriptor::default());
828        let bind_group =
829            create_bind_group(device, &bind_layout, &globals_buf, &atlas_view, &samplers);
830
831        let instance_cap = 4096;
832        let instance_buf = create_instance_buffer(device, instance_cap);
833
834        // Fragments: one uniform slot per draw, picked by dynamic offset,
835        // and the layout their pipelines are built against. All of it is
836        // built whether or not a frame ever draws one — a bind group
837        // layout and an empty buffer, not a pipeline, which is the part
838        // that costs and is built on first sight.
839        let fragment_bind_layout =
840            device.create_bind_group_layout(&wgpu::BindGroupLayoutDescriptor {
841                label: Some("kui.fragment.params"),
842                entries: &[wgpu::BindGroupLayoutEntry {
843                    binding: 0,
844                    visibility: wgpu::ShaderStages::FRAGMENT,
845                    ty: wgpu::BindingType::Buffer {
846                        ty: wgpu::BufferBindingType::Uniform,
847                        has_dynamic_offset: true,
848                        min_binding_size: std::num::NonZeroU64::new(std::mem::size_of::<
849                            FragmentParams,
850                        >()
851                            as u64),
852                    },
853                    count: None,
854                }],
855            });
856        let fragment_layouts = FragmentLayouts {
857            vertex: shader,
858            pipeline: device.create_pipeline_layout(&wgpu::PipelineLayoutDescriptor {
859                label: Some("kui.fragment"),
860                bind_group_layouts: &[Some(&bind_layout), Some(&fragment_bind_layout)],
861                immediate_size: 0,
862            }),
863        };
864        // The *device's* limit, not the adapter's: the device is opened with
865        // `Limits::default()`, whose `min_uniform_buffer_offset_alignment` is
866        // 256, and validation holds a dynamic offset to what the device asked
867        // for rather than to what the hardware could have done. An adapter
868        // reporting the smaller 64 — which DX12 does — then gave 64-byte slots
869        // and a validation error on the frame's second fragment.
870        let uniform_align = device.limits().min_uniform_buffer_offset_alignment;
871        let fragment_params_cap = 16;
872        let fragment_params_buf =
873            create_fragment_params_buffer(device, fragment_params_cap, uniform_align);
874        let fragment_bind =
875            create_fragment_bind_group(device, &fragment_bind_layout, &fragment_params_buf);
876
877        Ok(Self {
878            gpu,
879            surface,
880            config,
881            pipeline,
882            globals_buf,
883            bind_group,
884            bind_layout,
885            samplers,
886            texture_binds: Default::default(),
887            atlas_tex,
888            atlas_size,
889            atlas_epoch: u64::MAX,
890            instance_buf,
891            instance_cap,
892            instances: Vec::new(),
893            fragment_layouts,
894            fragment_params_buf,
895            fragment_params_cap,
896            fragment_bind,
897            fragment_bind_layout,
898            uniform_align,
899            fragment_bytes: Vec::new(),
900            clear_color: wgpu::Color {
901                r: 0.06,
902                g: 0.065,
903                b: 0.08,
904                a: 1.0,
905            },
906        })
907    }
908
909    /// Whether this device blends per channel, i.e. LCD subpixel glyphs
910    /// (`QuadKind::GlyphSubpixel`) render as intended. Drivers feed this to
911    /// `Core::set_subpixel_text`; without it the core should keep
912    /// rasterizing alpha masks.
913    pub fn subpixel_text(&self) -> bool {
914        self.gpu.dual_source()
915    }
916
917    /// Reconfigures the swapchain for a new window size.
918    ///
919    /// Both bounds are the platform's, not ours. `max(1)` because a
920    /// minimized window reports zero and a zero-sized surface is a
921    /// validation error; `min(max_texture_dimension_2d)` because Windows
922    /// hands out a nonsense size mid-resize — a 2600x1500 move on Windows
923    /// 11 arrived as 2578x32711 — and configuring a surface larger than
924    /// the device can hold panics inside wgpu, taking the app with it. A
925    /// clamped frame is one wrong picture; the next real size fixes it.
926    pub fn resize(&mut self, width: u32, height: u32) {
927        let max = self.gpu.device().limits().max_texture_dimension_2d;
928        self.config.width = width.clamp(1, max);
929        self.config.height = height.clamp(1, max);
930        self.surface.configure(self.gpu.device(), &self.config);
931    }
932
933    fn sync_atlas(&mut self, atlas: &mut GlyphAtlas) {
934        if atlas.size != self.atlas_size {
935            self.atlas_size = atlas.size;
936            self.atlas_tex = create_atlas_texture(self.gpu.device(), atlas.size);
937            let view = self
938                .atlas_tex
939                .create_view(&wgpu::TextureViewDescriptor::default());
940            self.bind_group = create_bind_group(
941                self.gpu.device(),
942                &self.bind_layout,
943                &self.globals_buf,
944                &view,
945                &self.samplers,
946            );
947            self.atlas_epoch = u64::MAX;
948        }
949        if atlas.dirty || self.atlas_epoch != atlas.epoch {
950            self.gpu.queue().write_texture(
951                wgpu::TexelCopyTextureInfo {
952                    texture: &self.atlas_tex,
953                    mip_level: 0,
954                    origin: wgpu::Origin3d::ZERO,
955                    aspect: wgpu::TextureAspect::All,
956                },
957                &atlas.pixels,
958                wgpu::TexelCopyBufferLayout {
959                    offset: 0,
960                    bytes_per_row: Some(atlas.size * 4),
961                    rows_per_image: Some(atlas.size),
962                },
963                wgpu::Extent3d {
964                    width: atlas.size,
965                    height: atlas.size,
966                    depth_or_array_layers: 1,
967                },
968            );
969            atlas.dirty = false;
970            self.atlas_epoch = atlas.epoch;
971        }
972    }
973
974    pub fn render(
975        &mut self,
976        dl: &DisplayList,
977        atlas: &mut GlyphAtlas,
978    ) -> Result<RenderReport, RenderError> {
979        // A dead device takes no work: everything below would only add
980        // errors to the one that lost it.
981        if self.gpu.lost() {
982            return Err(RenderError::DeviceLost);
983        }
984        self.sync_atlas(atlas);
985
986        self.instances.clear();
987        self.instances.extend(
988            dl.quads
989                .iter()
990                .map(|q| instance_of(q, &dl.clips, &dl.textures)),
991        );
992        if self.instances.len() > self.instance_cap {
993            self.instance_cap = self.instances.len().next_power_of_two();
994            self.instance_buf = create_instance_buffer(self.gpu.device(), self.instance_cap);
995        }
996        if !self.instances.is_empty() {
997            self.gpu.queue().write_buffer(
998                &self.instance_buf,
999                0,
1000                bytemuck::cast_slice(&self.instances),
1001            );
1002        }
1003        let globals = Globals {
1004            viewport: [dl.viewport.w.max(1.0), dl.viewport.h.max(1.0)],
1005            atlas_size: [self.atlas_size as f32, self.atlas_size as f32],
1006            time: dl.time,
1007            scale: dl.scale,
1008            _pad: [0.0; 2],
1009        };
1010        self.gpu
1011            .queue()
1012            .write_buffer(&self.globals_buf, 0, bytemuck::bytes_of(&globals));
1013
1014        // Texture-backed images (ADR 0025, decision 3): drop what the core
1015        // removed, upload what moved, and give each one drawn this frame
1016        // a group-0 bind group of its own with a globals copy whose
1017        // `atlas_size` is the texture's. All skipped on a frame that
1018        // draws none.
1019        for id in &dl.dropped_textures {
1020            self.texture_binds.remove(&id.to_ffi());
1021            self.gpu.drop_image_texture(id.to_ffi());
1022        }
1023        // And the pipelines of removed fragments (AR8) — built per handle
1024        // and shared by every window, so one window's list carries the
1025        // removal and this is the only eviction they get.
1026        for id in &dl.dropped_fragments {
1027            self.gpu.drop_fragment_pipelines(id.to_ffi());
1028        }
1029        // A drop another window's frame carried: the cache no longer
1030        // holds the texture this bind group does. One lock per frame,
1031        // and only for a window that has ever drawn a texture.
1032        if !self.texture_binds.is_empty() {
1033            let gpu = &self.gpu;
1034            self.texture_binds
1035                .retain(|id, b| gpu.holds_image_texture(*id, &b.texture));
1036        }
1037        let mut texture_binds: Vec<Option<u64>> = Vec::new();
1038        if !dl.textures.is_empty() {
1039            texture_binds.reserve(dl.textures.len());
1040            for (draw, px) in dl.textures.iter().zip(&dl.texture_pixels) {
1041                let id = draw.id.to_ffi();
1042                let Some(texture) = self.gpu.image_texture(id, px) else {
1043                    texture_binds.push(None);
1044                    continue;
1045                };
1046                let stale = self
1047                    .texture_binds
1048                    .get(&id)
1049                    .is_none_or(|b| !std::sync::Arc::ptr_eq(&b.texture, &texture));
1050                if stale {
1051                    let device = self.gpu.device();
1052                    let globals_buf = device.create_buffer(&wgpu::BufferDescriptor {
1053                        label: Some("kui.image.globals"),
1054                        size: std::mem::size_of::<Globals>() as u64,
1055                        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1056                        mapped_at_creation: false,
1057                    });
1058                    let bind = create_bind_group(
1059                        device,
1060                        &self.bind_layout,
1061                        &globals_buf,
1062                        &texture.view,
1063                        &self.samplers,
1064                    );
1065                    self.texture_binds.insert(
1066                        id,
1067                        TextureBind {
1068                            texture: texture.clone(),
1069                            globals: globals_buf,
1070                            bind,
1071                        },
1072                    );
1073                }
1074                let b = &self.texture_binds[&id];
1075                let mine = Globals {
1076                    atlas_size: [texture.width as f32, texture.height as f32],
1077                    ..globals
1078                };
1079                self.gpu
1080                    .queue()
1081                    .write_buffer(&b.globals, 0, bytemuck::bytes_of(&mine));
1082                texture_binds.push(Some(id));
1083            }
1084        }
1085
1086        // Each fragment's parameters into its own slot, and its pipeline
1087        // built if this device has not seen the handle before. Both are
1088        // skipped whole on a frame that draws no fragment.
1089        let mut fragment_pipelines: Vec<wgpu::RenderPipeline> = Vec::new();
1090        if !dl.fragments.is_empty() {
1091            let align = self.uniform_align as usize;
1092            if dl.fragments.len() > self.fragment_params_cap {
1093                self.fragment_params_cap = dl.fragments.len().next_power_of_two();
1094                self.fragment_params_buf = create_fragment_params_buffer(
1095                    self.gpu.device(),
1096                    self.fragment_params_cap,
1097                    self.uniform_align,
1098                );
1099                self.fragment_bind = create_fragment_bind_group(
1100                    self.gpu.device(),
1101                    &self.fragment_bind_layout,
1102                    &self.fragment_params_buf,
1103                );
1104            }
1105            self.fragment_bytes.clear();
1106            self.fragment_bytes.resize(dl.fragments.len() * align, 0);
1107            for (i, draw) in dl.fragments.iter().enumerate() {
1108                let uv = draw.image.uv();
1109                let slot = FragmentParams {
1110                    params: draw.params,
1111                    image: [uv[0] as f32, uv[1] as f32, uv[2] as f32, uv[3] as f32],
1112                };
1113                let at = i * align;
1114                self.fragment_bytes[at..at + std::mem::size_of::<FragmentParams>()]
1115                    .copy_from_slice(bytemuck::bytes_of(&slot));
1116            }
1117            self.gpu
1118                .queue()
1119                .write_buffer(&self.fragment_params_buf, 0, &self.fragment_bytes);
1120            fragment_pipelines.reserve(dl.fragments.len());
1121            for (draw, source) in dl.fragments.iter().zip(&dl.fragment_sources) {
1122                fragment_pipelines.push(self.gpu.fragment_pipeline(
1123                    draw.id.to_ffi(),
1124                    source,
1125                    self.config.format,
1126                    &self.fragment_layouts,
1127                ));
1128            }
1129        }
1130
1131        // Acquiring the swapchain image is where vsync backpressure blocks;
1132        // report it separately so latency graphs show pacing vs work.
1133        let t_wait = std::time::Instant::now();
1134        let frame = match self.surface.get_current_texture() {
1135            wgpu::CurrentSurfaceTexture::Success(f)
1136            | wgpu::CurrentSurfaceTexture::Suboptimal(f) => f,
1137            wgpu::CurrentSurfaceTexture::Timeout | wgpu::CurrentSurfaceTexture::Occluded => {
1138                return Err(RenderError::Skip);
1139            }
1140            wgpu::CurrentSurfaceTexture::Outdated | wgpu::CurrentSurfaceTexture::Lost => {
1141                return Err(RenderError::Reconfigure);
1142            }
1143            // The acquire's error went to the device's error handler; if
1144            // it was the device itself, the lost callback has run by now.
1145            wgpu::CurrentSurfaceTexture::Validation => {
1146                return Err(if self.gpu.lost() {
1147                    RenderError::DeviceLost
1148                } else {
1149                    RenderError::Validation
1150                });
1151            }
1152        };
1153        let vsync_wait_ms = t_wait.elapsed().as_secs_f32() * 1e3;
1154        let view = frame
1155            .texture
1156            .create_view(&wgpu::TextureViewDescriptor::default());
1157        let mut encoder = self
1158            .gpu
1159            .device()
1160            .create_command_encoder(&wgpu::CommandEncoderDescriptor { label: Some("kui") });
1161        {
1162            let mut pass = encoder.begin_render_pass(&wgpu::RenderPassDescriptor {
1163                label: Some("kui"),
1164                color_attachments: &[Some(wgpu::RenderPassColorAttachment {
1165                    view: &view,
1166                    depth_slice: None,
1167                    resolve_target: None,
1168                    ops: wgpu::Operations {
1169                        load: wgpu::LoadOp::Clear(self.clear_color),
1170                        store: wgpu::StoreOp::Store,
1171                    },
1172                })],
1173                depth_stencil_attachment: None,
1174                timestamp_writes: None,
1175                occlusion_query_set: None,
1176                multiview_mask: None,
1177            });
1178            if !self.instances.is_empty() {
1179                pass.set_vertex_buffer(0, self.instance_buf.slice(..));
1180                if fragment_pipelines.is_empty() && texture_binds.is_empty() {
1181                    // The whole frame in one instanced draw, as it has
1182                    // always been. Nothing below runs.
1183                    pass.set_pipeline(&self.pipeline);
1184                    pass.set_bind_group(0, &self.bind_group, &[]);
1185                    pass.draw(0..6, 0..self.instances.len() as u32);
1186                } else {
1187                    // A fragment or a texture-backed image interrupts the
1188                    // run: draw what came before with the über-pipeline,
1189                    // then that one quad with its own pipeline (a
1190                    // fragment) or its own group 0 (a texture), then
1191                    // carry on (`docs/adr/0015-…` decision 5, ADR 0025
1192                    // decision 3). Consecutive quads of the same handle
1193                    // still take one set each, which is the 0.6 us the
1194                    // first ADR measured; runs of ordinary quads are
1195                    // unbroken.
1196                    let mut run_start = 0u32;
1197                    let mut on_quads = false;
1198                    for (i, q) in dl.quads.iter().enumerate() {
1199                        if q.kind != QuadKind::Fragment && q.kind != QuadKind::Texture {
1200                            continue;
1201                        }
1202                        let i = i as u32;
1203                        if i > run_start {
1204                            if !on_quads {
1205                                pass.set_pipeline(&self.pipeline);
1206                                pass.set_bind_group(0, &self.bind_group, &[]);
1207                                on_quads = true;
1208                            }
1209                            pass.draw(0..6, run_start..i);
1210                        }
1211                        // `uv[0]` is the index into the side list, which
1212                        // is also this fragment's parameter slot, or this
1213                        // texture's bind.
1214                        let slot = q.uv[0] as usize;
1215                        if q.kind == QuadKind::Texture {
1216                            if let Some(Some(id)) = texture_binds.get(slot)
1217                                && let Some(b) = self.texture_binds.get(id)
1218                            {
1219                                pass.set_pipeline(&self.pipeline);
1220                                pass.set_bind_group(0, &b.bind, &[]);
1221                                on_quads = false;
1222                                pass.draw(0..6, i..i + 1);
1223                            }
1224                        } else if let Some(pipeline) = fragment_pipelines.get(slot) {
1225                            // A fragment reading a texture-backed image
1226                            // takes that image's group 0 — the texture in
1227                            // the atlas's place, `atlas_size` its size —
1228                            // exactly as a texture quad does; one reading
1229                            // the atlas, or nothing, takes the frame's.
1230                            // A texture the device could not make (a
1231                            // degenerate or oversized image) draws the
1232                            // fragment against the atlas with a zero rect,
1233                            // which `kui_sample` reads as no image.
1234                            let group0 = match draw_image_texture(&dl.fragments[slot]) {
1235                                Some(index) => texture_binds
1236                                    .get(index)
1237                                    .copied()
1238                                    .flatten()
1239                                    .and_then(|id| self.texture_binds.get(&id))
1240                                    .map_or(&self.bind_group, |b| &b.bind),
1241                                None => &self.bind_group,
1242                            };
1243                            pass.set_pipeline(pipeline);
1244                            pass.set_bind_group(0, group0, &[]);
1245                            pass.set_bind_group(
1246                                1,
1247                                &self.fragment_bind,
1248                                &[slot as u32 * self.uniform_align],
1249                            );
1250                            on_quads = false;
1251                            pass.draw(0..6, i..i + 1);
1252                        }
1253                        run_start = i + 1;
1254                    }
1255                    let end = self.instances.len() as u32;
1256                    if end > run_start {
1257                        if !on_quads {
1258                            pass.set_pipeline(&self.pipeline);
1259                            pass.set_bind_group(0, &self.bind_group, &[]);
1260                        }
1261                        pass.draw(0..6, run_start..end);
1262                    }
1263                }
1264            }
1265        }
1266        self.gpu.queue().submit([encoder.finish()]);
1267        self.gpu.queue().present(frame);
1268        Ok(RenderReport { vsync_wait_ms })
1269    }
1270}
1271
1272/// The `textures` entry a fragment draw reads its image from, if its
1273/// image has a texture of its own.
1274fn draw_image_texture(draw: &kui_core::FragmentDraw) -> Option<usize> {
1275    match draw.image {
1276        kui_core::FragmentImage::Texture { index, .. } => Some(index as usize),
1277        _ => None,
1278    }
1279}
1280
1281/// Timing details from one `render` call.
1282#[derive(Clone, Copy, Debug, Default)]
1283pub struct RenderReport {
1284    /// Time blocked acquiring the swapchain image (vsync backpressure).
1285    pub vsync_wait_ms: f32,
1286}
1287
1288/// A frame that produced no image, mapped from `CurrentSurfaceTexture`.
1289#[derive(Clone, Copy, Debug)]
1290pub enum RenderError {
1291    /// Surface outdated/lost: `resize` (reconfigure) and redraw.
1292    Reconfigure,
1293    /// Nothing to present right now (occluded/timeout): try next frame.
1294    Skip,
1295    /// Validation error acquiring the surface texture: the surface is
1296    /// configured wrong for the window — `resize` to its size and redraw.
1297    Validation,
1298    /// The device is gone ([`Gpu::lost`]): open a new one, and a renderer
1299    /// on it for every window.
1300    DeviceLost,
1301}
1302
1303impl std::fmt::Display for RenderError {
1304    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
1305        match self {
1306            Self::Reconfigure => write!(f, "surface outdated or lost; reconfigure"),
1307            Self::Skip => write!(f, "no frame available; skip"),
1308            Self::Validation => write!(f, "surface texture validation error"),
1309            Self::DeviceLost => write!(f, "device lost; reopen"),
1310        }
1311    }
1312}
1313
1314impl std::error::Error for RenderError {}
1315
1316fn create_atlas_texture(device: &wgpu::Device, size: u32) -> wgpu::Texture {
1317    device.create_texture(&wgpu::TextureDescriptor {
1318        label: Some("kui.atlas"),
1319        size: wgpu::Extent3d {
1320            width: size,
1321            height: size,
1322            depth_or_array_layers: 1,
1323        },
1324        mip_level_count: 1,
1325        sample_count: 1,
1326        dimension: wgpu::TextureDimension::D2,
1327        format: wgpu::TextureFormat::Rgba8Unorm,
1328        usage: wgpu::TextureUsages::TEXTURE_BINDING | wgpu::TextureUsages::COPY_DST,
1329        view_formats: &[],
1330    })
1331}
1332
1333/// Group 0: the globals, a texture — the atlas, or a texture-backed image
1334/// in its place — and the two samplers.
1335fn create_bind_group(
1336    device: &wgpu::Device,
1337    layout: &wgpu::BindGroupLayout,
1338    globals: &wgpu::Buffer,
1339    view: &wgpu::TextureView,
1340    samplers: &Samplers,
1341) -> wgpu::BindGroup {
1342    device.create_bind_group(&wgpu::BindGroupDescriptor {
1343        label: Some("kui"),
1344        layout,
1345        entries: &[
1346            wgpu::BindGroupEntry {
1347                binding: 0,
1348                resource: globals.as_entire_binding(),
1349            },
1350            wgpu::BindGroupEntry {
1351                binding: 1,
1352                resource: wgpu::BindingResource::TextureView(view),
1353            },
1354            wgpu::BindGroupEntry {
1355                binding: 2,
1356                resource: wgpu::BindingResource::Sampler(&samplers.linear),
1357            },
1358            wgpu::BindGroupEntry {
1359                binding: 3,
1360                resource: wgpu::BindingResource::Sampler(&samplers.nearest),
1361            },
1362        ],
1363    })
1364}
1365
1366fn create_instance_buffer(device: &wgpu::Device, cap: usize) -> wgpu::Buffer {
1367    device.create_buffer(&wgpu::BufferDescriptor {
1368        label: Some("kui.instances"),
1369        size: (cap * std::mem::size_of::<Instance>()) as u64,
1370        usage: wgpu::BufferUsages::VERTEX | wgpu::BufferUsages::COPY_DST,
1371        mapped_at_creation: false,
1372    })
1373}
1374
1375fn create_fragment_params_buffer(device: &wgpu::Device, cap: usize, align: u32) -> wgpu::Buffer {
1376    device.create_buffer(&wgpu::BufferDescriptor {
1377        label: Some("kui.fragment.params"),
1378        size: (cap.max(1) * align as usize) as u64,
1379        usage: wgpu::BufferUsages::UNIFORM | wgpu::BufferUsages::COPY_DST,
1380        mapped_at_creation: false,
1381    })
1382}
1383
1384fn create_fragment_bind_group(
1385    device: &wgpu::Device,
1386    layout: &wgpu::BindGroupLayout,
1387    buf: &wgpu::Buffer,
1388) -> wgpu::BindGroup {
1389    device.create_bind_group(&wgpu::BindGroupDescriptor {
1390        label: Some("kui.fragment.params"),
1391        layout,
1392        entries: &[wgpu::BindGroupEntry {
1393            binding: 0,
1394            resource: wgpu::BindingResource::Buffer(wgpu::BufferBinding {
1395                buffer: buf,
1396                offset: 0,
1397                size: std::num::NonZeroU64::new(std::mem::size_of::<FragmentParams>() as u64),
1398            }),
1399        }],
1400    })
1401}
1402
1403/// Says where a crash is. A fault in a driver — a GPU whose driver is
1404/// being replaced under the app — ends the process with no line from
1405/// anyone: not a panic, so nothing of ours prints, and Windows reports
1406/// only `0xC000041D` for an exception in a window callback. This prints
1407/// the code and the module the faulting address is in, then lets the
1408/// crash go on as it would have; a diagnostic, not a recovery.
1409///
1410/// Only a crash: the line is said from the process's unhandled-exception
1411/// filter, which runs once every frame handler has declined the
1412/// exception — not from a vectored handler, which sees each exception
1413/// first, before anyone has had the chance to handle it, and so sees the
1414/// faults that are part of normal running: a driver probing memory under
1415/// its own `__try`, V8's WebAssembly bounds checks under `node.exe`. Said
1416/// from there, those spent the one report each kind had on something
1417/// harmless, and the crash that followed was never named. A filter set
1418/// before this one is called after it, with its answer returned, so a
1419/// crash reporter the host installed first still gets the crash; one set
1420/// after replaces this one, as it would any filter.
1421///
1422/// A vectored handler is still installed, last among them, but it says
1423/// nothing: it remembers the last fault each thread saw. An exception
1424/// that escapes a window callback reaches the filter as `0xC000041D`, not
1425/// as itself, and the fault inside it is found on the record's own chain
1426/// or, failing that, in what the thread last remembered.
1427#[cfg(windows)]
1428pub fn report_faults() {
1429    use std::cell::Cell;
1430    use windows::Win32::Foundation::{
1431        EXCEPTION_ACCESS_VIOLATION, EXCEPTION_ILLEGAL_INSTRUCTION, EXCEPTION_IN_PAGE_ERROR,
1432        EXCEPTION_STACK_OVERFLOW, HMODULE, NTSTATUS, STATUS_FATAL_USER_CALLBACK_EXCEPTION,
1433    };
1434    use windows::Win32::Storage::FileSystem::WriteFile;
1435    use windows::Win32::System::Console::{GetStdHandle, STD_ERROR_HANDLE};
1436    use windows::Win32::System::Diagnostics::Debug::{
1437        AddVectoredExceptionHandler, EXCEPTION_POINTERS, EXCEPTION_RECORD,
1438        LPTOP_LEVEL_EXCEPTION_FILTER, SetUnhandledExceptionFilter,
1439    };
1440    use windows::Win32::System::LibraryLoader::{
1441        GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS, GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
1442        GetModuleFileNameW, GetModuleHandleExW,
1443    };
1444
1445    const CONTINUE_SEARCH: i32 = 0;
1446    /// The faults a driver ends a process with: the ones remembered for
1447    /// a callback's `0xC000041D` to be read by.
1448    const FAULTS: [NTSTATUS; 4] = [
1449        EXCEPTION_ACCESS_VIOLATION,
1450        EXCEPTION_ILLEGAL_INSTRUCTION,
1451        EXCEPTION_IN_PAGE_ERROR,
1452        EXCEPTION_STACK_OVERFLOW,
1453    ];
1454    /// The filter this one replaced, called after it; set once, with the
1455    /// two handlers, by the one call that installs them.
1456    static PREVIOUS: std::sync::OnceLock<LPTOP_LEVEL_EXCEPTION_FILTER> = std::sync::OnceLock::new();
1457    thread_local! {
1458        /// The last fault this thread saw, code and address, handled or
1459        /// not. A `const` cell with no destructor: a plain thread-local
1460        /// slot, read and written without allocating or registering
1461        /// anything, from inside an exception.
1462        static LAST: Cell<Option<(i32, usize)>> = const { Cell::new(None) };
1463    }
1464
1465    /// Remembers a fault; says nothing and handles nothing.
1466    unsafe extern "system" fn remember(info: *mut EXCEPTION_POINTERS) -> i32 {
1467        // SAFETY: the system hands a valid record for the exception.
1468        if let Some(record) =
1469            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
1470            && FAULTS.contains(&record.ExceptionCode)
1471        {
1472            let seen = (record.ExceptionCode.0, record.ExceptionAddress as usize);
1473            let _ = LAST.try_with(|l| l.set(Some(seen)));
1474        }
1475        CONTINUE_SEARCH
1476    }
1477
1478    /// The first fault on a record's chain of nested exceptions, past the
1479    /// record itself; a few links, since a chain is one or two long and a
1480    /// broken one is not worth following further.
1481    fn nested(record: &EXCEPTION_RECORD) -> Option<(i32, usize)> {
1482        let mut at = record.ExceptionRecord;
1483        for _ in 0..4 {
1484            // SAFETY: a nested record the system chained to this one.
1485            let inner = unsafe { at.as_ref() }?;
1486            if FAULTS.contains(&inner.ExceptionCode) {
1487                return Some((inner.ExceptionCode.0, inner.ExceptionAddress as usize));
1488            }
1489            at = inner.ExceptionRecord;
1490        }
1491        None
1492    }
1493
1494    /// Writes one crash's line to stderr, straight to the handle: no
1495    /// `eprintln!`, which takes a lock and, on a console, converts
1496    /// through a stack buffer eight kilobytes deep — more than a stack
1497    /// overflow leaves.
1498    fn say(code: i32, at: usize, escaped: Option<i32>) {
1499        let mut module = HMODULE::default();
1500        let mut name = [0u16; 260];
1501        // SAFETY: `at` is only looked up, never read; the buffers are ours.
1502        let found = unsafe {
1503            GetModuleHandleExW(
1504                GET_MODULE_HANDLE_EX_FLAG_FROM_ADDRESS
1505                    | GET_MODULE_HANDLE_EX_FLAG_UNCHANGED_REFCOUNT,
1506                windows::core::PCWSTR(at as *const u16),
1507                &mut module,
1508            )
1509        }
1510        .is_ok();
1511        let path = found.then(|| {
1512            // SAFETY: the module was just found; the buffer is ours.
1513            let n = unsafe { GetModuleFileNameW(Some(module), &mut name) } as usize;
1514            &name[..n.min(name.len())]
1515        });
1516        let line = FaultLine::new(code as u32, at, path, escaped.map(|c| c as u32));
1517        // SAFETY: a handle the process was given, written from our buffer.
1518        if let Ok(err) = unsafe { GetStdHandle(STD_ERROR_HANDLE) } {
1519            let mut written = 0u32;
1520            let _ = unsafe { WriteFile(err, Some(line.bytes()), Some(&mut written), None) };
1521        }
1522    }
1523
1524    /// The crash: said, then handed to the filter before this one.
1525    unsafe extern "system" fn filter(info: *const EXCEPTION_POINTERS) -> i32 {
1526        // SAFETY: the system hands a valid record for the exception.
1527        if let Some(record) =
1528            (unsafe { info.as_ref() }).and_then(|i| unsafe { i.ExceptionRecord.as_ref() })
1529        {
1530            let (code, at) = (record.ExceptionCode, record.ExceptionAddress as usize);
1531            let inner = if code == STATUS_FATAL_USER_CALLBACK_EXCEPTION {
1532                nested(record).or_else(|| LAST.try_with(Cell::get).ok().flatten())
1533            } else {
1534                None
1535            };
1536            match inner {
1537                Some((fault, fault_at)) => say(fault, fault_at, Some(code.0)),
1538                None => say(code.0, at, None),
1539            }
1540        }
1541        match PREVIOUS.get().copied().flatten() {
1542            // SAFETY: the filter the system held before ours, called as
1543            // the system would have called it.
1544            Some(previous) => unsafe { previous(info) },
1545            None => CONTINUE_SEARCH,
1546        }
1547    }
1548
1549    PREVIOUS.get_or_init(|| {
1550        // SAFETY: both handlers read only what the system gives them and
1551        // write only their own thread-local slot and stderr.
1552        unsafe {
1553            AddVectoredExceptionHandler(0, Some(remember));
1554            SetUnhandledExceptionFilter(Some(filter))
1555        }
1556    });
1557}
1558
1559#[cfg(not(windows))]
1560pub fn report_faults() {}
1561
1562/// One crash's line for `report_faults`, written into a buffer on the
1563/// stack: it is said with whatever stack the crash left (a stack
1564/// overflow leaves the few pages the thread reserved for its handlers)
1565/// and in a process whose heap may be what faulted, so nothing here
1566/// allocates. A line too long for it is cut, at a character, and still
1567/// ends in a newline. Built on every platform so it is tested on every
1568/// platform; only Windows says one.
1569#[cfg_attr(not(windows), allow(dead_code))]
1570struct FaultLine {
1571    buf: [u8; 640],
1572    len: usize,
1573}
1574
1575#[cfg_attr(not(windows), allow(dead_code))]
1576impl FaultLine {
1577    /// `kui: fault <code> at <address> in <module>`, and for a fault that
1578    /// escaped a window callback, the code it escaped as. `module` is the
1579    /// UTF-16 path Windows gives; none is a fault outside any module.
1580    fn new(code: u32, at: usize, module: Option<&[u16]>, escaped: Option<u32>) -> Self {
1581        use std::fmt::Write;
1582        let mut line = Self {
1583            buf: [0; 640],
1584            len: 0,
1585        };
1586        let _ = write!(line, "kui: fault {code:#010x} at {at:#x} in ");
1587        match module {
1588            Some(path) => {
1589                for c in char::decode_utf16(path.iter().copied()) {
1590                    let _ = line.write_char(c.unwrap_or(char::REPLACEMENT_CHARACTER));
1591                }
1592            }
1593            None => {
1594                let _ = line.write_str("no module (jit or freed code)");
1595            }
1596        }
1597        if let Some(escaped) = escaped {
1598            let _ = write!(line, ", escaped from a window callback as {escaped:#010x}");
1599        }
1600        // The newline has its byte kept for it (`write_str`).
1601        line.buf[line.len] = b'\n';
1602        line.len += 1;
1603        line
1604    }
1605
1606    fn bytes(&self) -> &[u8] {
1607        &self.buf[..self.len]
1608    }
1609}
1610
1611impl std::fmt::Write for FaultLine {
1612    fn write_str(&mut self, s: &str) -> std::fmt::Result {
1613        // One byte short of the buffer, for the newline.
1614        let room = self.buf.len() - 1 - self.len;
1615        let mut n = s.len().min(room);
1616        while !s.is_char_boundary(n) {
1617            n -= 1;
1618        }
1619        self.buf[self.len..self.len + n].copy_from_slice(&s.as_bytes()[..n]);
1620        self.len += n;
1621        Ok(())
1622    }
1623}
1624
1625#[cfg(test)]
1626mod tests {
1627    use super::*;
1628
1629    /// The globals are one buffer read by two pipelines whose modules
1630    /// declare it separately: this crate's `shader.wgsl` for quads, and
1631    /// `kui_core::fragment::PRELUDE` for every fragment. If the two
1632    /// declarations drift, a fragment reads the wrong bytes and there is
1633    /// nothing to catch it at runtime — the buffer is the right size and
1634    /// the numbers are just wrong. So: same field names, same order, and
1635    /// the size the Rust struct actually is.
1636    #[test]
1637    fn globals_layout_matches() {
1638        let fields = ["viewport", "atlas_size", "time", "scale", "_pad"];
1639        let of = |src: &str, name: &str| {
1640            let start = src
1641                .find(name)
1642                .unwrap_or_else(|| panic!("{name} is not declared in\n{src}"));
1643            let body = &src[start..];
1644            let end = body.find('}').expect("a closing brace");
1645            body[..end].to_string()
1646        };
1647        let quads = of(include_str!("shader.wgsl"), "struct Globals {");
1648        let frags = of(kui_core::fragment::PRELUDE, "struct KuiGlobals {");
1649        let read = |body: &str| -> Vec<String> {
1650            body.lines()
1651                .filter_map(|l| l.split_once(':'))
1652                .map(|(name, ty)| format!("{}: {}", name.trim(), ty.trim().trim_end_matches(',')))
1653                .collect()
1654        };
1655        let (a, b) = (read(&quads), read(&frags));
1656        assert_eq!(a, b, "shader.wgsl and the fragment prelude disagree");
1657        assert_eq!(
1658            a.len(),
1659            fields.len(),
1660            "a field was added to the globals without this test being told"
1661        );
1662        for (row, want) in a.iter().zip(fields) {
1663            assert!(row.starts_with(want), "expected {want}, got {row}");
1664        }
1665        // vec2 + vec2 + f32 + f32 + vec2 = 32 bytes, and a uniform's size
1666        // must be a multiple of sixteen, which is what `_pad` is for.
1667        assert_eq!(std::mem::size_of::<Globals>(), 32);
1668    }
1669
1670    /// The fragment parameter slot is the other buffer two declarations
1671    /// read: `FragmentParams` here and `KuiFragmentParams` in the
1672    /// epilogue. Sixteen floats then the image's rect, 80 bytes.
1673    #[test]
1674    fn fragment_params_layout_matches() {
1675        assert_eq!(std::mem::size_of::<FragmentParams>(), 80);
1676        assert_eq!(std::mem::offset_of!(FragmentParams, image), 64);
1677        let epilogue = kui_core::fragment::EPILOGUE;
1678        assert!(
1679            epilogue
1680                .contains("struct KuiFragmentParams { p: array<vec4<f32>, 4>, image: vec4<f32> };"),
1681            "the epilogue's params struct moved without this test being told"
1682        );
1683    }
1684
1685    /// The bindings the prelude declares at group 0 are this crate's, by
1686    /// number and kind, since a fragment pipeline binds the quad
1687    /// pipeline's group 0 layout as it is.
1688    #[test]
1689    fn prelude_bindings_match_group_zero() {
1690        let prelude = kui_core::fragment::PRELUDE;
1691        for line in [
1692            "@group(0) @binding(0) var<uniform> kui_globals: KuiGlobals;",
1693            "@group(0) @binding(1) var kui_atlas: texture_2d<f32>;",
1694            "@group(0) @binding(2) var kui_sampler: sampler;",
1695            "@group(0) @binding(3) var kui_sampler_nearest: sampler;",
1696        ] {
1697            assert!(prelude.contains(line), "prelude lacks `{line}`");
1698        }
1699        let quads = include_str!("shader.wgsl");
1700        for line in [
1701            "@group(0) @binding(1) var atlas_tex: texture_2d<f32>;",
1702            "@group(0) @binding(2) var atlas_smp: sampler;",
1703            "@group(0) @binding(3) var nearest_smp: sampler;",
1704        ] {
1705            assert!(quads.contains(line), "shader.wgsl lacks `{line}`");
1706        }
1707    }
1708
1709    /// Both preprocessed variants of the shader must parse and validate
1710    /// (pipeline creation would otherwise fail at runtime, in a window).
1711    #[test]
1712    fn shader_variants_validate() {
1713        use wgpu::naga::valid::{Capabilities, ValidationFlags, Validator};
1714        for dual in [false, true] {
1715            let src = preprocess_shader(include_str!("shader.wgsl"), dual);
1716            let module = wgpu::naga::front::wgsl::parse_str(&src)
1717                .unwrap_or_else(|e| panic!("dual={dual}: {}", e.emit_to_string(&src)));
1718            let caps = if dual {
1719                Capabilities::DUAL_SOURCE_BLENDING
1720            } else {
1721                Capabilities::empty()
1722            };
1723            Validator::new(ValidationFlags::all(), caps)
1724                .validate(&module)
1725                .unwrap_or_else(|e| panic!("dual={dual}: {e:?}"));
1726        }
1727    }
1728
1729    /// The line `report_faults` says, built without the heap (RG31): the
1730    /// code, the address and the module's UTF-16 path decoded, and for a
1731    /// fault that escaped a window callback the code it escaped as.
1732    #[test]
1733    fn a_fault_line_names_the_code_the_address_and_the_module() {
1734        let path: Vec<u16> = r"C:\Windows\System32\nvoglv64.dll".encode_utf16().collect();
1735        let line = FaultLine::new(0xC000_0005, 0x7ff6_1234, Some(&path), None);
1736        assert_eq!(
1737            std::str::from_utf8(line.bytes()).unwrap(),
1738            "kui: fault 0xc0000005 at 0x7ff61234 in C:\\Windows\\System32\\nvoglv64.dll\n"
1739        );
1740        let line = FaultLine::new(0xC000_0005, 0x10, None, Some(0xC000_041D));
1741        assert_eq!(
1742            std::str::from_utf8(line.bytes()).unwrap(),
1743            "kui: fault 0xc0000005 at 0x10 in no module (jit or freed code), \
1744             escaped from a window callback as 0xc000041d\n"
1745        );
1746        // A path that is not UTF-16 is said, not refused.
1747        let line = FaultLine::new(0xC000_001D, 0x20, Some(&[0x44, 0xD800, 0x45]), None);
1748        assert_eq!(
1749            std::str::from_utf8(line.bytes()).unwrap(),
1750            "kui: fault 0xc000001d at 0x20 in D\u{FFFD}E\n"
1751        );
1752    }
1753
1754    /// A line longer than its stack buffer is cut, between characters,
1755    /// and still ends in its newline.
1756    #[test]
1757    fn a_fault_line_too_long_is_cut_at_a_character() {
1758        let path: Vec<u16> = "é".repeat(1000).encode_utf16().collect();
1759        let line = FaultLine::new(0xC000_00FD, 0x30, Some(&path), Some(0xC000_041D));
1760        let text = std::str::from_utf8(line.bytes()).expect("cut at a character");
1761        assert!(text.ends_with("é\n"), "{text:?}");
1762        assert!(text.len() <= 640 && text.len() >= 638, "{}", text.len());
1763        assert!(text.starts_with("kui: fault 0xc00000fd at 0x30 in é"));
1764    }
1765}