emblema_hal/capabilities.rs
1//! What a device can actually do.
2//!
3//! Everything above the HAL branches on these values, never on backend
4//! identity. A Mali GPU without `EGL_ANDROID_native_fence_sync` and a desktop
5//! Vulkan driver without `VK_KHR_external_fence_fd` present the same problem,
6//! and code that asks "is this GLES?" instead of "can this export a fence?"
7//! gets both cases wrong.
8
9use crate::format::FormatModifierSet;
10
11/// A capability a context can be asked to do without.
12///
13/// One variant per [`Capabilities`] field it withholds, named for that field.
14/// Exists so a test can obtain a context that lacks something the device has,
15/// and so exercise a refusal on any machine rather than only on hardware that
16/// happens to lack it. Several tests here could previously only run where the
17/// capability was genuinely absent, and one of them said so: "on a machine where
18/// both have it there is nothing here to check".
19///
20/// **Withholding only.** There is no way to grant a capability a device has not
21/// got, and that is a soundness property rather than a matter of taste: a Vulkan
22/// device is created without
23/// `VkPhysicalDeviceBlendOperationAdvancedFeaturesEXT` when the capability is
24/// false, so a granted flag would have pipelines built with advanced blend
25/// operations against a device that never enabled the feature. See [`Withheld`],
26/// whose name is the other half of saying this.
27///
28/// Deliberately not `#[non_exhaustive]`. Each backend maps this with an
29/// exhaustive `match`, and since the backends are separate crates a wildcard arm
30/// would let them drift apart -- one honoring a new variant and the other
31/// silently ignoring it. The cost is that adding a variant is a breaking change
32/// for anything downstream that matches on it, which is accepted because this is
33/// test-support vocabulary.
34#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
35pub enum Capability {
36 /// [`Capabilities::advanced_blend`].
37 AdvancedBlend,
38 /// [`Capabilities::float_render_targets`].
39 FloatRenderTargets,
40}
41
42impl Capability {
43 /// Which bit of a [`Withheld`] set stands for this one.
44 const fn bit(self) -> u32 {
45 match self {
46 Self::AdvancedBlend => 1 << 0,
47 Self::FloatRenderTargets => 1 << 1,
48 }
49 }
50}
51
52/// Capabilities a context is to be built without, as a bitmask.
53///
54/// A set rather than one capability because a test may want a device short of
55/// two things at once, and a bitmask rather than a `Vec` because
56/// `ContextConfig` is `Copy` and a heap field would take that away from a
57/// published type. Hand-rolled over a `u32` like [`SampleCounts`] above, for the
58/// same reason: this workspace has no bitflags dependency and does not need one.
59///
60/// The name is the type's only job beyond storage. Nothing here can grant a
61/// capability, and a neutral name like `CapabilitySet` would invite someone to
62/// add that direction; `Withheld` makes it unsayable. [`Self::apply_to`] can
63/// clear a field and can do nothing else.
64#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
65pub struct Withheld(u32);
66
67impl Withheld {
68 /// Withhold nothing, which is what every ordinary caller wants and what
69 /// `Default` gives.
70 pub const fn none() -> Self {
71 Self(0)
72 }
73
74 /// Withhold exactly one.
75 pub const fn of(capability: Capability) -> Self {
76 Self(capability.bit())
77 }
78
79 /// The same set with one more withheld.
80 pub const fn with(self, capability: Capability) -> Self {
81 Self(self.0 | capability.bit())
82 }
83
84 /// Whether this set withholds `capability`.
85 pub const fn contains(self, capability: Capability) -> bool {
86 (self.0 & capability.bit()) != 0
87 }
88
89 /// Whether this set withholds nothing.
90 pub const fn is_empty(self) -> bool {
91 self.0 == 0
92 }
93
94 /// Clear every withheld capability from `capabilities`.
95 ///
96 /// Clearing only. A capability already false stays false, so applying this
97 /// to a device that never had the thing is a no-op rather than a promotion --
98 /// which is what makes the set safe to apply unconditionally at the end of
99 /// detection.
100 ///
101 /// A backend that can express a withholding more honestly should do that
102 /// *as well*: dropping the extension the capability is backed by means the
103 /// probe reports false on its own merits and the device is genuinely built
104 /// without it. This is what covers the fields backed by a format query
105 /// rather than an extension, and it is the belt to that braces.
106 pub fn apply_to(self, capabilities: &mut Capabilities) {
107 for capability in [Capability::AdvancedBlend, Capability::FloatRenderTargets] {
108 if !self.contains(capability) {
109 continue;
110 }
111 match capability {
112 Capability::AdvancedBlend => capabilities.advanced_blend = false,
113 Capability::FloatRenderTargets => capabilities.float_render_targets = false,
114 }
115 }
116 }
117}
118
119impl From<Capability> for Withheld {
120 fn from(capability: Capability) -> Self {
121 Self::of(capability)
122 }
123}
124
125/// Supported MSAA sample counts, as a bitmask of powers of two.
126#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
127pub struct SampleCounts(u32);
128
129impl SampleCounts {
130 pub const fn from_mask(mask: u32) -> Self {
131 Self(mask)
132 }
133
134 /// Whether `count` is supported. Non-power-of-two counts are never valid.
135 pub const fn supports(self, count: u32) -> bool {
136 count.is_power_of_two() && (self.0 & count) != 0
137 }
138
139 /// The highest supported count, or 1 when only single-sampled rendering
140 /// is available.
141 pub const fn max(self) -> u32 {
142 if self.0 == 0 {
143 1
144 } else {
145 // Highest set bit.
146 1 << (u32::BITS - 1 - self.0.leading_zeros())
147 }
148 }
149}
150
151/// How a device can move images across process or device boundaries.
152///
153/// Both halves matter independently. A render-only GPU paired with a separate
154/// display controller — the common ARM SoC topology — needs export on one
155/// device and import on the other.
156#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
157pub struct DmaBufSupport {
158 /// Can import a dma-buf as a texture.
159 pub import: bool,
160 /// Can export one of its own images as a dma-buf.
161 pub export: bool,
162 /// Can allocate and import with an explicit format modifier.
163 ///
164 /// Without this, the only safe shared layout is linear, and scanout gives
165 /// up whatever bandwidth a vendor tiled or compressed layout would have
166 /// saved.
167 pub modifiers: bool,
168}
169
170impl DmaBufSupport {
171 /// Whether this device can allocate its own scanout buffers.
172 ///
173 /// When false, the DRM presentation path must allocate through GBM and
174 /// import instead. Both paths are supported; this is what picks between
175 /// them.
176 pub const fn can_allocate_scanout(self) -> bool {
177 self.export && self.modifiers
178 }
179}
180
181/// Whether GPU completion can be handed to another component as a fence.
182#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
183pub struct SyncSupport {
184 /// Can export a `sync_file` fd from a fence, for use as `IN_FENCE_FD` on
185 /// an atomic commit.
186 pub export_sync_file: bool,
187 /// Can import a `sync_file` fd and wait on it on the GPU timeline.
188 pub import_sync_file: bool,
189}
190
191impl SyncSupport {
192 /// Whether the frame loop can stay fully explicit on the DRM path.
193 ///
194 /// When false, the DRM target must wait on the CPU before committing.
195 /// That is correct but slower, and on Tier-1 hardware it is a driver bug
196 /// to chase rather than a state to settle into, so callers are expected
197 /// to log it loudly.
198 pub const fn supports_explicit_scanout(self) -> bool {
199 self.export_sync_file
200 }
201}
202
203/// Everything the layers above the HAL are allowed to branch on.
204#[derive(Debug, Clone, Default)]
205pub struct Capabilities {
206 /// Maximum width or height of a 2D texture.
207 pub max_texture_size: u32,
208 /// Supported MSAA sample counts for render targets.
209 pub sample_counts: SampleCounts,
210 /// Cross-device image sharing.
211 pub dma_buf: DmaBufSupport,
212 /// Cross-component synchronization.
213 pub sync: SyncSupport,
214 /// Whether the advanced blend modes are available.
215 ///
216 /// Both kinds: the separable ones from `Multiply` through `Exclusion` and
217 /// the four non-separable ones. What they share is the thing that matters
218 /// here -- [`BlendMode::factors`] has no pair to return for any of them --
219 /// and what this does not gate is Porter-Duff, `Plus` or `Modulate`, which
220 /// are factors and available everywhere.
221 ///
222 /// They need a hardware extension and cannot be emulated with blend
223 /// factors, so a device without it refuses [`BlendMode::is_advanced`] modes
224 /// rather than substituting the nearest expressible one. Callers check this
225 /// before using one; nothing branches on which backend is in play.
226 ///
227 /// [`BlendMode::factors`]: crate::BlendMode::factors
228 ///
229 /// [`BlendMode::is_advanced`]: crate::BlendMode::is_advanced
230 pub advanced_blend: bool,
231 /// Whether a floating-point color attachment can be rendered into.
232 ///
233 /// A color outside the sRGB primaries' triangle has a component outside
234 /// zero to one, and `Rgba16Float` is the only format here that can hold
235 /// one. Whether a device will let it be a target is a separate question
236 /// from whether it will sample one: half-float is filterable in core ES
237 /// 3.0 and renderable only with an extension, and plenty of shipping
238 /// drivers have the first and not the second. Vulkan answers per format
239 /// from its format properties.
240 ///
241 /// Distinct from [`Self::render_formats`], which is the scanout list keyed
242 /// by DRM fourcc and is empty on GLES. A format with no fourcc is
243 /// deliberately absent from that list, so it cannot answer this.
244 pub float_render_targets: bool,
245 /// Formats and layouts this device can render into and export.
246 ///
247 /// One half of format negotiation; the presentation target supplies the
248 /// other.
249 pub render_formats: Vec<FormatModifierSet>,
250 /// Human-readable device and driver identification, for report
251 /// fingerprints and bug reports.
252 pub device_name: String,
253 pub driver_name: String,
254 /// Whether rendering happens on the CPU rather than on a GPU.
255 ///
256 /// Not a performance hint. It marks the device properties that are
257 /// consequences of having no graphics hardware rather than defects: a CPU
258 /// rasterizer has no tiling to describe, so advertising only a linear
259 /// layout is the correct answer for it and a sign of a missing modifier
260 /// query on anything else. A test that cannot tell those apart has to
261 /// choose between failing on software and not checking hardware, and both
262 /// are worse than asking.
263 ///
264 /// Vulkan takes this from the device type, which is authoritative. GLES has
265 /// no equivalent query and it is recognized from the renderer string, which
266 /// is not; a software implementation this does not know the name of reports
267 /// false, so treat a true as reliable and a false as merely unremarkable.
268 pub software: bool,
269}
270
271impl Capabilities {
272 /// Whether this device can drive a KMS plane directly, by either
273 /// allocation strategy.
274 ///
275 /// Note this says nothing about sync: a device can be scanout-capable and
276 /// still lack fence export, in which case the path works with a CPU wait.
277 pub fn supports_scanout(&self) -> bool {
278 self.dma_buf.can_allocate_scanout() || self.dma_buf.import
279 }
280
281 /// Whether every blend mode a batch uses is available on this device.
282 ///
283 /// Lives here rather than in each backend so the two refuse the same batch
284 /// for the same reason: a Vulkan device without the advanced-blend
285 /// extension and a GLES context without it are the same problem, and a
286 /// check written twice is a check that eventually disagrees with itself.
287 /// Refusing is deliberate — the alternative is substituting the nearest
288 /// expressible mode, which produces a picture nobody can debug from.
289 pub fn check_blend_modes(&self, batch: &crate::Batch) -> crate::Result<()> {
290 if !self.advanced_blend && batch.draws().iter().any(|draw| draw.blend.is_advanced()) {
291 return Err(crate::Error::Unsupported(
292 "advanced blend modes; this device has no advanced-blend extension",
293 ));
294 }
295 Ok(())
296 }
297
298 /// Whether this device can create the texture a descriptor asks for.
299 ///
300 /// Here rather than in each backend for the reason
301 /// [`Self::check_blend_modes`] is: two backends refusing the same thing for
302 /// the same reason, in one place, rather than a check written twice that
303 /// eventually disagrees with itself. Without it the failure arrives as a
304 /// framebuffer-incomplete number on one backend and a driver error on the
305 /// other, neither of which names what was missing.
306 pub fn check_texture(&self, desc: &crate::TextureDescriptor) -> crate::Result<()> {
307 // Refused for every device rather than for some, because this is not a
308 // capability: the pipeline carries sRGB-encoded components, so a target
309 // that applies the transfer on write applies it twice and the picture
310 // comes back over a third too bright. See `PixelFormat::is_drawable`.
311 //
312 // Only a render target. Sampling one decodes, which is also not what an
313 // encoded pipeline wants, but a caller may have data this is right for
314 // and `Context::create_image` says which format a picture wants. Drawing
315 // into one has no reading under which it is correct.
316 if desc.usage.render_target && !desc.format.is_drawable() {
317 return Err(crate::Error::Unsupported(
318 "an sRGB render target; this pipeline already holds encoded \
319 color and the format would encode it a second time",
320 ));
321 }
322 if desc.usage.render_target
323 && desc.format == crate::PixelFormat::Rgba16Float
324 && !self.float_render_targets
325 {
326 return Err(crate::Error::Unsupported(
327 "a floating-point render target; this device can sample one but not draw into it",
328 ));
329 }
330 Ok(())
331 }
332
333 /// Whether an extent fits within the device's texture limit.
334 pub fn can_allocate(&self, extent: crate::format::Extent2D) -> bool {
335 extent.width <= self.max_texture_size && extent.height <= self.max_texture_size
336 }
337}
338
339#[cfg(test)]
340mod tests {
341 use super::*;
342 use crate::format::Extent2D;
343
344 /// Withholding clears the field it names and touches nothing else.
345 ///
346 /// Field by field rather than by comparing two whole structs, so a failure
347 /// says which field moved. `Capabilities` holds a `Vec` and two `String`s and
348 /// is not `PartialEq`, which is the other reason.
349 #[test]
350 fn withholding_clears_its_own_field_and_no_other() {
351 let full = || Capabilities {
352 max_texture_size: 8192,
353 sample_counts: SampleCounts::from_mask(0b101),
354 dma_buf: DmaBufSupport {
355 import: true,
356 export: true,
357 modifiers: true,
358 },
359 sync: SyncSupport {
360 export_sync_file: true,
361 import_sync_file: true,
362 },
363 advanced_blend: true,
364 float_render_targets: true,
365 render_formats: Vec::new(),
366 device_name: String::from("a device"),
367 driver_name: String::from("a driver"),
368 software: true,
369 };
370
371 let mut caps = full();
372 Withheld::of(Capability::AdvancedBlend).apply_to(&mut caps);
373 assert!(!caps.advanced_blend, "the named field is not cleared");
374 assert!(caps.float_render_targets, "an unnamed field was cleared");
375 assert_eq!(caps.max_texture_size, 8192);
376 assert_eq!(caps.sample_counts, SampleCounts::from_mask(0b101));
377 assert!(caps.dma_buf.import && caps.dma_buf.export && caps.dma_buf.modifiers);
378 assert!(caps.sync.export_sync_file && caps.sync.import_sync_file);
379 assert!(caps.software);
380
381 let mut caps = full();
382 Withheld::of(Capability::FloatRenderTargets).apply_to(&mut caps);
383 assert!(!caps.float_render_targets);
384 assert!(caps.advanced_blend, "an unnamed field was cleared");
385
386 let mut caps = full();
387 Withheld::of(Capability::AdvancedBlend)
388 .with(Capability::FloatRenderTargets)
389 .apply_to(&mut caps);
390 assert!(!caps.advanced_blend && !caps.float_render_targets);
391 assert_eq!(caps.max_texture_size, 8192, "a set of two reached further");
392 }
393
394 /// It can clear and it can do nothing else.
395 ///
396 /// The whole soundness argument in one assertion: applied to a device that has
397 /// nothing, every field is still false afterwards. A mechanism that could
398 /// grant would have pipelines built with advanced blend operations against a
399 /// Vulkan device created without the feature enabled, which is undefined
400 /// behavior reachable from safe published API rather than a wrong picture.
401 #[test]
402 fn withholding_can_never_grant() {
403 let mut caps = Capabilities::default();
404 assert!(!caps.advanced_blend && !caps.float_render_targets);
405
406 Withheld::of(Capability::AdvancedBlend)
407 .with(Capability::FloatRenderTargets)
408 .apply_to(&mut caps);
409
410 assert!(
411 !caps.advanced_blend && !caps.float_render_targets,
412 "withholding a capability a device has not got turned it on"
413 );
414 }
415
416 /// Withholding nothing is the identity, and withholding twice is withholding
417 /// once.
418 ///
419 /// The first is what every ordinary caller gets from `Default`, so it has to
420 /// be free of effect. The second is what lets a backend apply the set at the
421 /// end of detection without knowing whether something earlier already took
422 /// the capability away -- which both backends do, since dropping an extension
423 /// makes the probe report false before this runs.
424 #[test]
425 fn withholding_nothing_changes_nothing_and_applying_twice_is_the_same() {
426 let mut caps = Capabilities {
427 advanced_blend: true,
428 float_render_targets: true,
429 ..Default::default()
430 };
431 Withheld::none().apply_to(&mut caps);
432 assert!(caps.advanced_blend && caps.float_render_targets);
433 assert!(Withheld::none().is_empty());
434 assert!(!Withheld::of(Capability::AdvancedBlend).is_empty());
435
436 let withheld = Withheld::of(Capability::AdvancedBlend);
437 withheld.apply_to(&mut caps);
438 let once = caps.advanced_blend;
439 withheld.apply_to(&mut caps);
440 assert_eq!(once, caps.advanced_blend, "a second application differed");
441 assert!(caps.float_render_targets, "the other field followed along");
442 }
443
444 /// A set says what it holds and nothing more.
445 #[test]
446 fn a_set_reports_what_it_withholds() {
447 let one = Withheld::of(Capability::AdvancedBlend);
448 assert!(one.contains(Capability::AdvancedBlend));
449 assert!(!one.contains(Capability::FloatRenderTargets));
450
451 let both = one.with(Capability::FloatRenderTargets);
452 assert!(both.contains(Capability::AdvancedBlend));
453 assert!(both.contains(Capability::FloatRenderTargets));
454
455 // The conversion the test-facing constructors lean on, so a caller can
456 // pass one capability where a set is wanted.
457 assert_eq!(Withheld::from(Capability::AdvancedBlend), one);
458 }
459
460 /// The advanced-blend refusal, which had no test while the texture refusal
461 /// beside it did.
462 ///
463 /// Three cases rather than one, because a check that refuses is only
464 /// correct if it also lets things through: a refusal that fired on every
465 /// batch would pass a test asserting only that an advanced mode is refused,
466 /// and this renderer draws almost nothing in an advanced mode.
467 ///
468 /// This is the whole of what a device lacking the extension gets. There is
469 /// no second implementation to fall back to -- upstream has one, reached
470 /// through framebuffer fetch, and the comment above says why substituting
471 /// is refused instead. So a caller asking for `Multiply` on such a device
472 /// gets this error and never a different picture.
473 #[test]
474 fn an_advanced_mode_is_refused_only_where_the_extension_is_missing() {
475 const TRI: [[f32; 2]; 3] = [[0.0, 0.0], [1.0, 0.0], [0.0, 1.0]];
476
477 let batch = |blend| {
478 let mut batch = crate::Batch::new();
479 batch
480 .push(&TRI, &[0, 1, 2], crate::Material::solid([1.0; 4]), blend)
481 .expect("one triangle is within any batch's limits");
482 batch
483 };
484 let advanced = batch(crate::BlendMode::Multiply);
485 let ordinary = batch(crate::BlendMode::SrcOver);
486 assert!(
487 crate::BlendMode::Multiply.is_advanced() && !crate::BlendMode::SrcOver.is_advanced(),
488 "the two modes chosen here have to sit on opposite sides of the \
489 thing being checked, or this test checks nothing"
490 );
491
492 let without = Capabilities {
493 advanced_blend: false,
494 ..Default::default()
495 };
496 let with = Capabilities {
497 advanced_blend: true,
498 ..Default::default()
499 };
500
501 assert!(
502 without.check_blend_modes(&advanced).is_err(),
503 "an advanced mode was allowed on a device with no advanced-blend \
504 extension, which reaches the driver as a mode it cannot express"
505 );
506 assert!(
507 with.check_blend_modes(&advanced).is_ok(),
508 "an advanced mode was refused on a device that has the extension"
509 );
510 assert!(
511 without.check_blend_modes(&ordinary).is_ok(),
512 "an ordinary mode was refused for want of an extension it does not \
513 need -- the check is looking at the batch rather than at the mode"
514 );
515 }
516
517 #[test]
518 fn sample_counts_reject_non_powers_of_two() {
519 let counts = SampleCounts::from_mask(1 | 2 | 4);
520 assert!(counts.supports(1));
521 assert!(counts.supports(4));
522 assert!(!counts.supports(8));
523 // 3 has bits in common with the mask but is not a valid count.
524 assert!(!counts.supports(3));
525 }
526
527 #[test]
528 fn sample_counts_report_max_and_degrade_to_single_sampled() {
529 assert_eq!(SampleCounts::from_mask(1 | 2 | 4).max(), 4);
530 assert_eq!(SampleCounts::default().max(), 1);
531 }
532
533 #[test]
534 fn self_allocation_requires_both_export_and_modifiers() {
535 let full = DmaBufSupport {
536 import: true,
537 export: true,
538 modifiers: true,
539 };
540 assert!(full.can_allocate_scanout());
541
542 // Export without modifier support cannot negotiate a scanout-capable
543 // layout, so allocation has to go through GBM instead.
544 let no_modifiers = DmaBufSupport {
545 modifiers: false,
546 ..full
547 };
548 assert!(!no_modifiers.can_allocate_scanout());
549 }
550
551 #[test]
552 fn import_only_devices_still_support_scanout() {
553 // The GBM-allocated path: cannot allocate its own scanout buffers but
554 // can import ones GBM made.
555 let caps = Capabilities {
556 dma_buf: DmaBufSupport {
557 import: true,
558 export: false,
559 modifiers: false,
560 },
561 ..Default::default()
562 };
563 assert!(!caps.dma_buf.can_allocate_scanout());
564 assert!(caps.supports_scanout());
565 }
566
567 #[test]
568 fn scanout_capability_is_independent_of_fence_export() {
569 let caps = Capabilities {
570 dma_buf: DmaBufSupport {
571 import: true,
572 export: true,
573 modifiers: true,
574 },
575 sync: SyncSupport::default(),
576 ..Default::default()
577 };
578 // Scanout works; it just cannot stay explicit, which is the caller's
579 // cue to log the CPU-wait fallback rather than to disable the path.
580 assert!(caps.supports_scanout());
581 assert!(!caps.sync.supports_explicit_scanout());
582 }
583
584 #[test]
585 fn allocation_limit_is_checked_on_both_axes() {
586 let caps = Capabilities {
587 max_texture_size: 4096,
588 ..Default::default()
589 };
590 assert!(caps.can_allocate(Extent2D::new(4096, 4096)));
591 assert!(!caps.can_allocate(Extent2D::new(4097, 16)));
592 assert!(!caps.can_allocate(Extent2D::new(16, 4097)));
593 }
594
595 /// Sampling a half-float texture and drawing into one are separate
596 /// permissions, and a device may offer the first without the second.
597 #[test]
598 fn a_float_target_is_refused_where_only_sampling_one_is_offered() {
599 let mut caps = Capabilities {
600 float_render_targets: false,
601 ..Capabilities::default()
602 };
603 let extent = Extent2D::new(4, 4);
604
605 let target = crate::TextureDescriptor::offscreen(extent, crate::PixelFormat::Rgba16Float);
606 assert!(
607 caps.check_texture(&target).is_err(),
608 "a float attachment was allowed where the device offers none"
609 );
610
611 // Sampling one is a different request and is not refused, which is the
612 // whole reason the two are separate: a gradient ramp wants exactly this
613 // and would otherwise be unavailable on the same devices.
614 let sampled = crate::TextureDescriptor::sampled(extent, crate::PixelFormat::Rgba16Float);
615 assert!(caps.check_texture(&sampled).is_ok());
616
617 // And an eight-bit target is never the question.
618 let ordinary = crate::TextureDescriptor::offscreen(extent, crate::PixelFormat::Rgba8Unorm);
619 assert!(caps.check_texture(&ordinary).is_ok());
620
621 caps.float_render_targets = true;
622 assert!(caps.check_texture(&target).is_ok());
623 }
624}