use super::scanline::render_frame_scanlines;
use crate::settings::model::LocalComputeTarget;
use glam::Vec3;
use indicatrix::{
geometry::plane::GpuFacetPlane,
optics::{
materials::GemMaterial,
raytracer::{Camera, EnvironmentSource, FacetFinish},
},
renderer::gpu_backend::{GpuBackend, GpuSceneRef},
};
#[derive(Clone, Copy)]
pub(super) struct BackendFrame<'a> {
pub(super) width: u32,
pub(super) height: u32,
pub(super) yaw: f32,
pub(super) pitch: f32,
pub(super) distance: f32,
pub(super) camera: &'a Camera,
pub(super) planes: &'a [GpuFacetPlane],
pub(super) facet_finishes: &'a [FacetFinish],
pub(super) material: &'a GemMaterial,
pub(super) max_bounces: u32,
pub(super) environment: EnvironmentSource<'a>,
pub(super) spp: u32,
pub(super) sample_offset: u32,
}
pub(super) struct FrameOutputs<'a> {
pub(super) accum: &'a mut [Vec3],
pub(super) depth: &'a mut [f32],
pub(super) normal: &'a mut [Vec3],
pub(super) facet_id: &'a mut [i32],
}
pub(super) struct ViewportGpu {
gpu: GpuBackend,
guides: crate::bridge::frame_cache::guide_pass::GuideCache,
applied_guide_key: Option<crate::bridge::frame_cache::guide_pass::GuideKey>,
}
impl ViewportGpu {
pub(super) fn acquire() -> Self {
Self {
gpu: GpuBackend::acquire(),
guides: crate::bridge::frame_cache::guide_pass::GuideCache::new(),
applied_guide_key: None,
}
}
fn try_accumulate(&mut self, frame: &BackendFrame<'_>, out: &mut FrameOutputs<'_>) -> bool {
let scene = GpuSceneRef {
camera: frame.camera,
width: frame.width,
height: frame.height,
planes: frame.planes,
facet_finishes: frame.facet_finishes,
material: frame.material,
max_bounces: frame.max_bounces,
environment: frame.environment,
};
if !self
.gpu
.try_accumulate(&scene, frame.sample_offset, frame.spp, out.accum)
{
return false;
}
let key = crate::bridge::frame_cache::guide_pass::GuideCache::key_for(
frame.width,
frame.height,
frame.yaw,
frame.pitch,
frame.distance,
frame.planes,
);
if self.applied_guide_key.as_ref() != Some(&key) {
let guides = self.guides.ensure(
frame.width,
frame.height,
frame.yaw,
frame.pitch,
frame.distance,
frame.planes,
);
out.depth.copy_from_slice(&guides.depth);
out.normal.copy_from_slice(&guides.normal);
out.facet_id.copy_from_slice(&guides.facet_id);
self.applied_guide_key = Some(key);
}
true
}
}
pub(super) fn accumulate_frame_samples(
backend: &mut ViewportGpu,
frame: &BackendFrame<'_>,
outputs: &mut FrameOutputs<'_>,
hybrid: &mut HybridPacing,
local_compute_target: LocalComputeTarget,
) {
if local_compute_target == LocalComputeTarget::Cpu {
let start = std::time::Instant::now();
render_frame_scanlines(frame, frame.spp, frame.sample_offset + frame.spp, outputs);
hybrid.observe_cpu_only(frame.spp, start.elapsed());
return;
}
if local_compute_target == LocalComputeTarget::CpuGpu
&& let Some(gpu_share) = hybrid.gpu_share(frame.spp)
{
let cpu_share = frame.spp - gpu_share;
if gpu_share > 0 && cpu_share > 0 {
hybrid_frame(backend, frame, outputs, hybrid, gpu_share);
return;
}
}
let start = std::time::Instant::now();
if backend.try_accumulate(frame, outputs) {
hybrid.observe_gpu_only(frame.spp, start.elapsed());
return;
}
let start = std::time::Instant::now();
render_frame_scanlines(frame, frame.spp, frame.sample_offset + frame.spp, outputs);
hybrid.observe_cpu_only(frame.spp, start.elapsed());
}
fn hybrid_frame(
backend: &mut ViewportGpu,
frame: &BackendFrame<'_>,
outputs: &mut FrameOutputs<'_>,
hybrid: &mut HybridPacing,
gpu_share: u32,
) {
let cpu_share = frame.spp - gpu_share;
let pixel_count = (frame.width as usize) * (frame.height as usize);
hybrid.reset_scratch(pixel_count);
let (gpu_ok, gpu_time, cpu_time) = {
let cpu_scratch = &mut hybrid.cpu_scratch;
let cpu_depth = &mut hybrid.scratch_depth;
let cpu_normal = &mut hybrid.scratch_normal;
let cpu_facet = &mut hybrid.scratch_facet;
std::thread::scope(|scope| {
let gpu_task = scope.spawn(|| {
let start = std::time::Instant::now();
let gpu_frame = BackendFrame {
spp: gpu_share,
..*frame
};
let ok = backend.try_accumulate(&gpu_frame, outputs);
(ok, start.elapsed())
});
let start = std::time::Instant::now();
render_frame_scanlines(
frame,
cpu_share,
frame.sample_offset + frame.spp,
&mut FrameOutputs {
accum: cpu_scratch,
depth: cpu_depth,
normal: cpu_normal,
facet_id: cpu_facet,
},
);
let cpu_time = start.elapsed();
let (gpu_ok, gpu_time) = gpu_task.join().unwrap_or((false, cpu_time));
(gpu_ok, gpu_time, cpu_time)
})
};
for (px, extra) in outputs.accum.iter_mut().zip(&hybrid.cpu_scratch) {
*px += *extra;
}
if gpu_ok {
hybrid.observe(gpu_share, gpu_time, cpu_share, cpu_time);
return;
}
hybrid.gpu_dead = true;
render_frame_scanlines(frame, gpu_share, frame.sample_offset + gpu_share, outputs);
}
pub(super) struct HybridPacing {
gpu_rate: Option<f64>,
cpu_rate: Option<f64>,
cpu_rate_seeded: bool,
gpu_dead: bool,
cpu_scratch: Vec<Vec3>,
scratch_depth: Vec<f32>,
scratch_normal: Vec<Vec3>,
scratch_facet: Vec<i32>,
}
impl HybridPacing {
pub(super) const fn new() -> Self {
Self {
gpu_rate: None,
cpu_rate: None,
cpu_rate_seeded: false,
gpu_dead: false,
cpu_scratch: Vec::new(),
scratch_depth: Vec::new(),
scratch_normal: Vec::new(),
scratch_facet: Vec::new(),
}
}
fn gpu_share(&self, spp: u32) -> Option<u32> {
if self.gpu_dead || spp < 2 {
return None;
}
let (gpu, cpu) = (self.gpu_rate?, self.cpu_rate?);
let frac = gpu / (gpu + cpu);
let share = (f64::from(spp) * frac).round() as u32;
let cap = if self.cpu_rate_seeded { spp - 1 } else { spp };
Some(share.min(cap))
}
fn reset_scratch(&mut self, pixel_count: usize) {
self.cpu_scratch.clear();
self.cpu_scratch.resize(pixel_count, Vec3::ZERO);
self.scratch_depth.clear();
self.scratch_depth.resize(pixel_count, 0.0);
self.scratch_normal.clear();
self.scratch_normal.resize(pixel_count, Vec3::ZERO);
self.scratch_facet.clear();
self.scratch_facet.resize(pixel_count, -1);
}
fn observe(
&mut self,
gpu_spp: u32,
gpu_time: std::time::Duration,
cpu_spp: u32,
cpu_time: std::time::Duration,
) {
Self::blend(&mut self.gpu_rate, gpu_spp, gpu_time);
Self::blend(&mut self.cpu_rate, cpu_spp, cpu_time);
self.cpu_rate_seeded = false;
}
fn observe_gpu_only(&mut self, spp: u32, elapsed: std::time::Duration) {
Self::blend(&mut self.gpu_rate, spp, elapsed);
if self.cpu_rate.is_none()
&& let Some(gpu) = self.gpu_rate
{
self.cpu_rate = Some(gpu / 16.0);
self.cpu_rate_seeded = true;
}
}
fn observe_cpu_only(&mut self, spp: u32, elapsed: std::time::Duration) {
Self::blend(&mut self.cpu_rate, spp, elapsed);
self.cpu_rate_seeded = false;
}
fn blend(slot: &mut Option<f64>, spp: u32, elapsed: std::time::Duration) {
if spp == 0 {
return;
}
let rate = f64::from(spp) / elapsed.as_secs_f64().max(1e-9);
*slot = Some(slot.map_or(rate, |prev| prev.mul_add(0.7, rate * 0.3)));
}
}