emblema_hal/lib.rs
1//! Rendering HAL trait: the backend-agnostic contract every rendering backend
2//! implements.
3//!
4//! The HAL answers how draw commands become pixels in a GPU image.
5//! Presentation — how a finished image reaches a display, and how the frame
6//! loop is paced — is a separate, orthogonal axis and lives in
7//! `emblema-present`.
8//!
9//! # Shape of the trait
10//!
11//! Deliberately Vulkan-leaning. Recording commands and replaying them as GL
12//! calls works; extracting Vulkan-grade explicitness from an immediate-mode
13//! abstraction does not. GLES pays a small CPU cost for command recording, and
14//! that is the accepted trade. The trait is never reduced to a lowest common
15//! denominator to accommodate a weaker backend: backends implement, emulate,
16//! or capability-gate.
17//!
18//! # Two requirements present from day one
19//!
20//! Both exist for the sake of the DRM presentation path, and both are here
21//! before any DRM code, because retrofitting either is a breaking redesign:
22//!
23//! 1. **External render targets** — [`ExternalImageDesc`] lets the renderer
24//! draw into images it did not allocate, carrying dma-buf fd, DRM fourcc,
25//! and format modifier.
26//! 2. **Exportable sync** — [`HalFence::export_sync_file`] converts GPU
27//! completion into a `sync_file` fd that can ride an atomic commit as
28//! `IN_FENCE_FD`.
29//!
30//! # Branch on capabilities, never on backend identity
31//!
32//! [`Capabilities`] is the only thing layers above the HAL are allowed to
33//! condition on. A Mali GPU lacking fence export and a desktop Vulkan driver
34//! lacking `VK_KHR_external_fence_fd` are the same problem, and code asking
35//! "is this GLES?" instead of "can this export a fence?" gets both wrong.
36
37pub mod batch;
38pub mod blend;
39pub mod capabilities;
40pub mod error;
41pub mod format;
42pub mod material;
43pub mod occlusion;
44pub mod resource;
45pub mod scissor;
46pub mod sync;
47
48pub use batch::{Batch, BatchDraw, ClipRole, ClipState, Vertex};
49pub use blend::{BlendFactor, BlendFactors, BlendMode};
50pub use capabilities::{
51 Capabilities, Capability, DmaBufSupport, SampleCounts, SyncSupport, Withheld,
52};
53pub use error::{Error, Result};
54pub use format::{Extent2D, FormatModifierSet, Fourcc, Modifier, PixelFormat};
55/// A caller's fragment program, in the forms each backend can take one.
56///
57/// Both are given because a program is registered once and drawn on whichever
58/// backend the context happens to be, and neither payload can be derived from
59/// the other here: nothing in this workspace translates shaders at run time,
60/// by the decision recorded in the architecture document. A caller's build
61/// produces both, which is what the reference implementation's toolchain does
62/// and what this project's own build does.
63#[derive(Debug, Clone)]
64pub struct RuntimeProgram {
65 /// A SPIR-V fragment module, for Vulkan.
66 pub spirv: Vec<u32>,
67 /// A GLSL ES 300 fragment source, for GLES.
68 pub glsl_es: String,
69}
70
71pub use material::{
72 ColorFilter, ColorForm, Gamma, Material, MaterialVariant, Sampling, Stop, TileMode,
73 MATERIAL_FLOATS, MAX_EFFECT_TEXTURES, MAX_STOPS, MORPHOLOGY_TAPS, RUNTIME_FLOATS,
74};
75pub use resource::{
76 mip_levels_for, BufferDescriptor, BufferUsage, TextureDescriptor, TextureUsage,
77};
78pub use scissor::Scissor;
79pub use sync::{HalFence, FRAME_WAIT_TIMEOUT};
80
81#[cfg(unix)]
82pub use resource::{DmaBufPlane, ExternalImageDesc};
83
84/// A rendering backend, as a family of associated types.
85///
86/// Implementors are zero-sized markers; the types they name carry the work.
87pub trait Hal: 'static {
88 type Context: HalContext<Hal = Self>;
89 type Texture: HalTexture;
90 /// GPU completion, for callers that submit without waiting.
91 ///
92 /// Named here because the DRM presentation path consumes one: it needs
93 /// something to hand the kernel and something to decide when a frame slot
94 /// is free again.
95 type Fence: HalFence;
96
97 /// Backend name, for logs and report fingerprints.
98 const NAME: &'static str;
99}
100
101/// What a target must be able to tell the layers above it.
102pub trait HalTexture {
103 fn extent(&self) -> Extent2D;
104 fn format(&self) -> PixelFormat;
105}
106
107/// A device, and the resources created from it.
108///
109/// # Why a batch rather than a command buffer
110///
111/// An earlier shape of this trait had callers record incrementally — begin a
112/// pass, bind a pipeline, draw, finish — mirroring how Vulkan itself works.
113/// Building a real backend showed that to be the wrong seam. What a backend
114/// actually wants is the whole [`Batch`] at once, because the useful decisions
115/// are all global to it: which draws can share a pipeline binding, how to lay
116/// out one shared vertex buffer, what to upload in a single copy. Handing over
117/// a stream of calls forces each backend to reconstruct that shape, and a
118/// record-and-replay backend would have to buffer the stream anyway just to see
119/// what it was given.
120///
121/// This stays explicit in the sense that matters — nothing is discovered at
122/// draw time and the caller states its whole intent up front — while leaving
123/// each backend free to realize it natively.
124///
125/// # Threading
126///
127/// These take `&mut self`, so a context is not yet usable for resource
128/// creation from several threads at once. The intended design is thread-safe
129/// creation behind `&self`, which needs interior mutability around the
130/// allocator. That is deferred rather than decided against: nothing creates
131/// resources off the recording thread yet, and adding the synchronization
132/// before there is a caller to shape it around would be guesswork.
133pub trait HalContext {
134 type Hal: Hal;
135
136 /// What this device can do. The only thing callers branch on.
137 fn capabilities(&self) -> &Capabilities;
138
139 /// Allocate a texture, or import an external image when the descriptor
140 /// carries one.
141 fn create_texture(&mut self, desc: &TextureDescriptor) -> Result<<Self::Hal as Hal>::Texture>;
142
143 /// Release a texture and its memory.
144 fn destroy_texture(&mut self, texture: <Self::Hal as Hal>::Texture);
145
146 /// Register a caller's fragment program and return the name for it.
147 ///
148 /// Registering the same payload twice gives the same name back rather than
149 /// a second entry. That is not a convenience: a program leads to pipelines
150 /// keyed by it, so registering one repeatedly would multiply the pipeline
151 /// cache by the number of times a caller happened to ask -- and a caller
152 /// with no place to cache an index, which is every caller that renders a
153 /// list of scenes, would do exactly that.
154 fn register_program(&mut self, program: &RuntimeProgram) -> Result<u32>;
155
156 /// Draw a batch into a target.
157 ///
158 /// A descriptor that preserves rather than clears composes several batches
159 /// onto one target. A multisampled descriptor renders to a transient
160 /// multisample buffer and resolves into the target, so the target stays
161 /// single-sampled and readable either way.
162 fn submit_batch(
163 &mut self,
164 target: &mut <Self::Hal as Hal>::Texture,
165 batch: &Batch,
166 pass: PassDescriptor,
167 ) -> Result<()> {
168 self.submit_batch_textured(target, batch, pass, &[])
169 }
170
171 /// Draw a batch whose materials sample textures.
172 ///
173 /// `textures` is the table a [`Material::Image`] slot indexes. It is passed
174 /// alongside the batch rather than held inside it because a batch is a
175 /// description a recorder produces without touching the device, and a
176 /// backend texture handle is not something it can name.
177 ///
178 /// The target is borrowed mutably and the table immutably, so a batch
179 /// cannot sample the target it draws into. That restriction is real rather
180 /// than incidental — reading an attachment being written in the same pass
181 /// needs machinery this does not have — and having the borrow checker state
182 /// it is better than discovering it as a driver-dependent picture.
183 ///
184 /// A slot with no entry is an error rather than a fallback: a paint
185 /// silently drawn as something else is the failure nobody debugs from.
186 fn submit_batch_textured(
187 &mut self,
188 target: &mut <Self::Hal as Hal>::Texture,
189 batch: &Batch,
190 pass: PassDescriptor,
191 textures: &[&<Self::Hal as Hal>::Texture],
192 ) -> Result<()>;
193
194 /// Copy a target back to host memory, tightly packed.
195 ///
196 /// Part of the trait rather than a backend extra because the offscreen
197 /// target is a first-class citizen: the entire golden and conformance
198 /// apparatus is built on rendering to one and reading it back.
199 fn read_texture(&mut self, texture: &mut <Self::Hal as Hal>::Texture) -> Result<Vec<u8>>;
200
201 /// Fill a texture from host memory, tightly packed and top row first.
202 ///
203 /// The exact inverse of [`Self::read_texture`], stated in the same layout,
204 /// so a round trip through the pair is the identity on every backend. That
205 /// is what lets it be checked without a decoder: write known bytes, read
206 /// them back, compare.
207 ///
208 /// Color must be **premultiplied**, matching what a render target holds
209 /// and what a paint sampling this expects. A decoder usually produces
210 /// straight alpha, so converting is the caller's job. The distinction is
211 /// invisible for an opaque image, which is what makes it worth stating
212 /// here rather than leaving to be discovered.
213 ///
214 /// Decoding images is out of scope for this project; getting already
215 /// decoded pixels onto the device is not. Without this, an image shader
216 /// could sample nothing but what the renderer itself had drawn.
217 fn write_texture(
218 &mut self,
219 texture: &mut <Self::Hal as Hal>::Texture,
220 pixels: &[u8],
221 ) -> Result<()>;
222
223 /// Allocate a target that can be shared with a display controller.
224 ///
225 /// `modifiers` are the layouts the other side accepts, in preference order.
226 /// The default reports the capability as absent, which is the honest answer
227 /// for a backend that cannot do it: callers check
228 /// [`DmaBufSupport::can_allocate_scanout`] and take the GBM-allocated path
229 /// instead. This is capability gating rather than a stub — a backend that
230 /// answered every method this way would be useless, but one that answers
231 /// only the optional ones is correctly describing itself.
232 fn create_exportable_texture(
233 &mut self,
234 _extent: Extent2D,
235 _format: PixelFormat,
236 _modifiers: &[Modifier],
237 ) -> Result<<Self::Hal as Hal>::Texture> {
238 Err(Error::Unsupported("allocating exportable images"))
239 }
240
241 /// Export a texture as a dma-buf for scanout or cross-device sharing.
242 ///
243 /// Returns [`Error::Unsupported`] where [`DmaBufSupport::export`] is false;
244 /// the DRM path then allocates through GBM and imports instead.
245 #[cfg(unix)]
246 fn export_texture(
247 &mut self,
248 _texture: &<Self::Hal as Hal>::Texture,
249 ) -> Result<ExternalImageDesc> {
250 Err(Error::Unsupported("dma-buf export"))
251 }
252
253 /// Submit a batch without waiting, returning something that signals when
254 /// the GPU has finished.
255 ///
256 /// A frame loop needs this rather than the waiting form: the returned
257 /// fence is what gets handed to a display commit, and what decides when a
258 /// frame slot may be reused.
259 ///
260 /// Ordering between two of these is the caller's, and nothing here checks
261 /// it. Submitting twice into one target without waiting for the first is a
262 /// write-after-write hazard: both calls are well formed, both succeed, and
263 /// what lands is whichever the device finished last. A frame loop avoids it
264 /// by construction, since a ring hands out a different slot each frame and
265 /// will not reuse one until its fence has retired.
266 fn submit_batch_deferred(
267 &mut self,
268 target: &mut <Self::Hal as Hal>::Texture,
269 batch: &Batch,
270 pass: PassDescriptor,
271 ) -> Result<<Self::Hal as Hal>::Fence> {
272 self.submit_batch_deferred_textured(target, batch, pass, &[])
273 }
274
275 /// The same, for a batch whose materials sample textures.
276 ///
277 /// What a frame with layers needs. The pass that lands in the image being
278 /// presented is the one that composites the layers, so it samples the
279 /// targets they were rendered into — and a deferred submission that could
280 /// not sample anything meant such a frame could be rendered offscreen and
281 /// never displayed.
282 ///
283 /// The textures must outlive the submission, which the caller arranges:
284 /// they cannot travel with the fence, because a fence is handed to a
285 /// display commit and has to stay sendable while a texture tracks mutable
286 /// state of its own.
287 fn submit_batch_deferred_textured(
288 &mut self,
289 _target: &mut <Self::Hal as Hal>::Texture,
290 _batch: &Batch,
291 _pass: PassDescriptor,
292 _textures: &[&<Self::Hal as Hal>::Texture],
293 ) -> Result<<Self::Hal as Hal>::Fence> {
294 Err(Error::Unsupported("deferred submission"))
295 }
296
297 /// Release a deferred submission once its work has completed.
298 fn retire_fence(&mut self, _fence: <Self::Hal as Hal>::Fence) {}
299}
300
301/// How a pass is configured, beyond the draws themselves.
302///
303/// Grouped rather than passed as loose parameters so later pass-level state —
304/// stencil, depth, damage regions — extends this without changing every
305/// backend's signature.
306#[derive(Debug, Clone, Copy, PartialEq)]
307pub struct PassDescriptor {
308 /// Clear to this color first, or preserve the target's contents.
309 pub clear: Option<[f32; 4]>,
310 /// MSAA sample count. 1 disables multisampling.
311 ///
312 /// Rendering is multisampled and resolved into the target, so the target
313 /// itself stays single-sampled and directly readable.
314 pub samples: u32,
315 /// Where this pass's clip space lands in the target, if not the whole of it.
316 ///
317 /// `None` maps clip space onto the target exactly, which is what every
318 /// pass wants whose geometry was recorded against the target it renders
319 /// into. A pass whose geometry was recorded against a *larger* space sets
320 /// this to crop instead of scale — see [`PassViewport`].
321 pub viewport: Option<PassViewport>,
322}
323
324/// A clip space larger than the target it lands in, and where it lands.
325///
326/// A layer records its draws before anything is known about how much of the
327/// target they will cover. The geometry and every material go into clip space
328/// as they are recorded — `docs/architecture.md` states that a recording is
329/// tessellated geometry rather than a command list — so by the time the extent
330/// could be narrowed, narrowing it would mean rewriting all of that.
331///
332/// This narrows the *target* instead and leaves the recording alone. The
333/// viewport keeps the extent the geometry was recorded against, and `offset`
334/// slides it so that the wanted sub-rectangle lands on the target. Clip space
335/// maps onto the viewport, not onto the target, so the picture is cropped
336/// rather than scaled and every recorded coordinate still means what it meant.
337///
338/// `offset` is normally negative: a target holding the region starting at
339/// device (x, y) of the recorded space offsets by (-x, -y).
340#[derive(Debug, Clone, Copy, PartialEq)]
341pub struct PassViewport {
342 /// Where the recorded space's origin sits, in target pixels.
343 pub offset: [f32; 2],
344 /// The extent the geometry was recorded against.
345 pub extent: Extent2D,
346}
347
348impl Default for PassDescriptor {
349 fn default() -> Self {
350 Self {
351 clear: None,
352 samples: 1,
353 viewport: None,
354 }
355 }
356}
357
358impl PassDescriptor {
359 pub fn clear(color: [f32; 4]) -> Self {
360 Self {
361 clear: Some(color),
362 samples: 1,
363 viewport: None,
364 }
365 }
366
367 /// Land this pass's clip space on part of the target rather than all of it.
368 pub fn with_viewport(mut self, viewport: PassViewport) -> Self {
369 self.viewport = Some(viewport);
370 self
371 }
372
373 pub fn preserve() -> Self {
374 Self::default()
375 }
376
377 pub fn with_samples(mut self, samples: u32) -> Self {
378 self.samples = samples;
379 self
380 }
381
382 pub fn is_multisampled(&self) -> bool {
383 self.samples > 1
384 }
385}