use super::scanline::render_frame_scanlines;
use crate::settings::model::LocalComputeTarget;
use glam::Vec3;
use indicatrix::{
geometry::plane::GpuFacetPlane,
optics::{
materials::GemMaterial,
raytracer::{Camera, EnvironmentSource, FacetFinish},
},
renderer::gpu_backend::{GpuBackend, GpuSceneRef},
};
#[derive(Clone, Copy)]
pub(super) struct BackendFrame<'a> {
pub(super) width: u32,
pub(super) height: u32,
pub(super) yaw: f32,
pub(super) pitch: f32,
pub(super) distance: f32,
pub(super) camera: &'a Camera,
pub(super) planes: &'a [GpuFacetPlane],
pub(super) facet_finishes: &'a [FacetFinish],
pub(super) material: &'a GemMaterial,
pub(super) max_bounces: u32,
pub(super) environment: EnvironmentSource<'a>,
pub(super) spp: u32,
pub(super) sample_offset: u32,
}
pub(super) struct FrameOutputs<'a> {
pub(super) accum: &'a mut [Vec3],
pub(super) depth: &'a mut [f32],
pub(super) normal: &'a mut [Vec3],
pub(super) facet_id: &'a mut [i32],
}
pub(super) struct ViewportGpu {
gpu: GpuBackend,
guides: crate::bridge::frame_cache::guide_pass::GuideCache,
applied_guide_key: Option<crate::bridge::frame_cache::guide_pass::GuideKey>,
gpu_retired: bool,
status_message: Option<String>,
}
impl ViewportGpu {
pub(super) fn acquire() -> Self {
Self {
gpu: GpuBackend::acquire(),
guides: crate::bridge::frame_cache::guide_pass::GuideCache::new(),
applied_guide_key: None,
gpu_retired: false,
status_message: None,
}
}
fn retire(&mut self) {
self.gpu_retired = true;
self.status_message = Some("CPU fallback (GPU render thread panicked)".to_string());
}
pub(super) fn status_message(&self) -> Option<&str> {
self.status_message.as_deref()
}
pub(super) const fn invalidate_guide_cache(&mut self) {
self.applied_guide_key = None;
}
fn try_accumulate(&mut self, frame: &BackendFrame<'_>, out: &mut FrameOutputs<'_>) -> bool {
if self.gpu_retired {
return false;
}
let scene = GpuSceneRef {
camera: frame.camera,
width: frame.width,
height: frame.height,
planes: frame.planes,
facet_finishes: frame.facet_finishes,
material: frame.material,
max_bounces: frame.max_bounces,
environment: frame.environment,
};
if self
.gpu
.try_accumulate(&scene, frame.sample_offset, frame.spp, out.accum)
{
self.status_message = None;
self.copy_guides(frame, out);
return true;
}
if !self.gpu.is_lost() {
return false;
}
self.note_device_loss();
false
}
fn note_device_loss(&mut self) {
if self.status_message.is_none() {
let reason = self
.gpu
.last_lost_reason()
.unwrap_or_else(|| "unknown reason".to_string());
tracing::warn!(%reason, "GPU renderer lost; rendering on the CPU tracer until it recovers");
self.status_message = Some(format!("CPU fallback (GPU lost: {reason})"));
}
}
fn copy_guides(&mut self, frame: &BackendFrame<'_>, out: &mut FrameOutputs<'_>) {
let key = crate::bridge::frame_cache::guide_pass::GuideCache::key_for(
frame.width,
frame.height,
frame.yaw,
frame.pitch,
frame.distance,
frame.planes,
);
if self.applied_guide_key.as_ref() != Some(&key) {
let guides = self.guides.ensure(
frame.width,
frame.height,
frame.yaw,
frame.pitch,
frame.distance,
frame.planes,
);
out.depth.copy_from_slice(&guides.depth);
out.normal.copy_from_slice(&guides.normal);
out.facet_id.copy_from_slice(&guides.facet_id);
self.applied_guide_key = Some(key);
}
}
}
pub(super) fn accumulate_frame_samples(
backend: &mut ViewportGpu,
frame: &BackendFrame<'_>,
outputs: &mut FrameOutputs<'_>,
hybrid: &mut HybridPacing,
local_compute_target: LocalComputeTarget,
) {
if local_compute_target == LocalComputeTarget::Cpu {
let start = std::time::Instant::now();
render_frame_scanlines(frame, frame.spp, frame.sample_offset + frame.spp, outputs);
backend.invalidate_guide_cache();
hybrid.observe_cpu_only(frame.spp, start.elapsed());
return;
}
if local_compute_target == LocalComputeTarget::CpuGpu
&& let Some(gpu_share) = hybrid.gpu_share(frame.spp)
{
let cpu_share = frame.spp - gpu_share;
if gpu_share > 0 && cpu_share > 0 {
hybrid_frame(backend, frame, outputs, hybrid, gpu_share);
return;
}
}
let start = std::time::Instant::now();
if backend.try_accumulate(frame, outputs) {
hybrid.observe_gpu_only(frame.spp, start.elapsed());
return;
}
let start = std::time::Instant::now();
render_frame_scanlines(frame, frame.spp, frame.sample_offset + frame.spp, outputs);
backend.invalidate_guide_cache();
hybrid.observe_cpu_only(frame.spp, start.elapsed());
}
fn hybrid_frame(
backend: &mut ViewportGpu,
frame: &BackendFrame<'_>,
outputs: &mut FrameOutputs<'_>,
hybrid: &mut HybridPacing,
gpu_share: u32,
) {
let cpu_share = frame.spp - gpu_share;
let pixel_count = (frame.width as usize) * (frame.height as usize);
hybrid.prepare_scratch(pixel_count);
let (gpu_ok, gpu_time, cpu_time, gpu_panicked) = {
let cpu_scratch = &mut hybrid.cpu_scratch;
let cpu_depth = &mut hybrid.scratch_depth;
let cpu_normal = &mut hybrid.scratch_normal;
let cpu_facet = &mut hybrid.scratch_facet;
std::thread::scope(|scope| {
let gpu_task = scope.spawn(|| {
let start = std::time::Instant::now();
let gpu_frame = BackendFrame {
spp: gpu_share,
..*frame
};
let ok = backend.try_accumulate(&gpu_frame, outputs);
(ok, start.elapsed())
});
let start = std::time::Instant::now();
render_frame_scanlines(
frame,
cpu_share,
frame.sample_offset + frame.spp,
&mut FrameOutputs {
accum: cpu_scratch,
depth: cpu_depth,
normal: cpu_normal,
facet_id: cpu_facet,
},
);
let cpu_time = start.elapsed();
match gpu_task.join() {
Ok((gpu_ok, gpu_time)) => (gpu_ok, gpu_time, cpu_time, false),
Err(_) => (false, cpu_time, cpu_time, true),
}
})
};
for (px, extra) in outputs.accum.iter_mut().zip(&hybrid.cpu_scratch) {
*px += *extra;
}
if gpu_ok {
hybrid.observe(gpu_share, gpu_time, cpu_share, cpu_time);
return;
}
hybrid.gpu_dead = true;
if gpu_panicked {
tracing::error!(
"GPU render thread panicked mid-frame; retiring the GPU backend for the \
rest of this session"
);
backend.retire();
}
render_frame_scanlines(frame, gpu_share, frame.sample_offset + gpu_share, outputs);
backend.invalidate_guide_cache();
}
pub(super) struct HybridPacing {
gpu_rate: Option<f64>,
cpu_rate: Option<f64>,
cpu_rate_seeded: bool,
gpu_dead: bool,
cpu_scratch: Vec<Vec3>,
scratch_depth: Vec<f32>,
scratch_normal: Vec<Vec3>,
scratch_facet: Vec<i32>,
}
impl HybridPacing {
pub(super) const fn new() -> Self {
Self {
gpu_rate: None,
cpu_rate: None,
cpu_rate_seeded: false,
gpu_dead: false,
cpu_scratch: Vec::new(),
scratch_depth: Vec::new(),
scratch_normal: Vec::new(),
scratch_facet: Vec::new(),
}
}
fn gpu_share(&self, spp: u32) -> Option<u32> {
if self.gpu_dead || spp < 2 {
return None;
}
let (gpu, cpu) = (self.gpu_rate?, self.cpu_rate?);
let frac = gpu / (gpu + cpu);
let share = (f64::from(spp) * frac).round() as u32;
let cap = if self.cpu_rate_seeded { spp - 1 } else { spp };
Some(share.min(cap))
}
fn prepare_scratch(&mut self, pixel_count: usize) {
self.cpu_scratch.clear();
self.cpu_scratch.resize(pixel_count, Vec3::ZERO);
self.scratch_depth.resize(pixel_count, 1.0e6);
self.scratch_normal.resize(pixel_count, Vec3::ZERO);
self.scratch_facet.resize(pixel_count, -1);
}
fn observe(
&mut self,
gpu_spp: u32,
gpu_time: std::time::Duration,
cpu_spp: u32,
cpu_time: std::time::Duration,
) {
Self::blend(&mut self.gpu_rate, gpu_spp, gpu_time);
Self::blend(&mut self.cpu_rate, cpu_spp, cpu_time);
self.cpu_rate_seeded = false;
}
fn observe_gpu_only(&mut self, spp: u32, elapsed: std::time::Duration) {
Self::blend(&mut self.gpu_rate, spp, elapsed);
if self.cpu_rate.is_none()
&& let Some(gpu) = self.gpu_rate
{
self.cpu_rate = Some(gpu / 16.0);
self.cpu_rate_seeded = true;
}
}
fn observe_cpu_only(&mut self, spp: u32, elapsed: std::time::Duration) {
Self::blend(&mut self.cpu_rate, spp, elapsed);
self.cpu_rate_seeded = false;
}
fn blend(slot: &mut Option<f64>, spp: u32, elapsed: std::time::Duration) {
if spp == 0 {
return;
}
let rate = f64::from(spp) / elapsed.as_secs_f64().max(1e-9);
*slot = Some(slot.map_or(rate, |prev| prev.mul_add(0.7, rate * 0.3)));
}
}
#[cfg(all(test, feature = "gpu"))]
mod gpu_hardware_tests {
use super::*;
use indicatrix::{
geometry::cuts::StandardGemCuts, optics::raytracer::LightingPreset,
renderer::gpu_backend::GpuPipelineKind,
};
fn assert_twelve_rotated_frames_never_lose_the_device(pipeline_kind: GpuPipelineKind) {
let mut viewport = ViewportGpu::acquire();
if viewport.gpu.adapter_label().is_none() {
println!(
"skipping assert_twelve_rotated_frames_never_lose_the_device({pipeline_kind:?}): \
no GPU adapter"
);
return;
}
viewport.gpu.set_pipeline_kind(pipeline_kind);
let planes = StandardGemCuts::standard_round_brilliant();
let material = GemMaterial::by_name("Spinel").expect("Spinel is a built-in cubic material");
let environment = LightingPreset::Daylight.studio(1.0, 0.4, 0.35);
let (width, height) = (480u32, 360u32);
let spp = 2u32;
let pixel_count = (width * height) as usize;
let mut accum = vec![Vec3::ZERO; pixel_count];
let mut depth = vec![1.0e6; pixel_count];
let mut normal = vec![Vec3::ZERO; pixel_count];
let mut facet_id = vec![-1i32; pixel_count];
for frame in 0..12u32 {
let yaw = (frame as f32).mul_add(0.29, 0.1);
let pitch = (frame as f32).mul_add(0.13, 0.2).clamp(-1.4, 1.4);
let camera = Camera::new(yaw, pitch, 5.0, 18.0);
let backend_frame = BackendFrame {
width,
height,
yaw,
pitch,
distance: 5.0,
camera: &camera,
planes: &planes,
facet_finishes: &[],
material: &material,
max_bounces: 4,
environment,
spp,
sample_offset: frame * spp,
};
let ok = viewport.try_accumulate(
&backend_frame,
&mut FrameOutputs {
accum: &mut accum,
depth: &mut depth,
normal: &mut normal,
facet_id: &mut facet_id,
},
);
assert!(
ok,
"frame {frame} ({pipeline_kind:?}, yaw={yaw}, pitch={pitch}) declined -- \
status: {:?}",
viewport.status_message()
);
assert_eq!(
viewport.status_message(),
None,
"frame {frame} ({pipeline_kind:?}) left a self-healing status behind -- \
the device was lost at some point during this run"
);
}
assert!(
accum.iter().any(|v| v.length_squared() > 0.0),
"a lit studio-rig scene traced over 12 frames must leave SOME nonzero radiance"
);
}
#[test]
fn twelve_rotated_frames_never_lose_the_device_megakernel() {
assert_twelve_rotated_frames_never_lose_the_device(GpuPipelineKind::Megakernel);
}
#[test]
fn twelve_rotated_frames_never_lose_the_device_wavefront() {
assert_twelve_rotated_frames_never_lose_the_device(GpuPipelineKind::Wavefront);
}
}