concinnity-device 0.19.119

GPU backends (Metal, Vulkan, DirectX) behind a device facade for Concinnity
//! Auto-exposure (EV adaptation) on Metal: a per-frame CPU readback of the
//! previous frame's average log-luminance, an EMA step that updates the adapted
//! EV, and the histogram build + average compute dispatches that produce next
//! frame's average. The compute passes are encoded after the main HDR resolve
//! (where `hdr_resolve` carries this frame's scene color) and read CPU-side at
//! the top of the next frame, so there is one frame of latency between the
//! scene's actual luminance and the exposure applied to it: invisible at
//! human-scale eye-adaptation rates.
//!
//! The readback is a ring of one buffer per frame-in-flight rather than a single
//! shared one. Frame `R` writes slot `R % depth` and reads that same slot before
//! encoding, which the frames-in-flight fence guarantees frame `R - depth` wrote
//! and the GPU has retired -- so the value read is always exactly `depth` frames
//! old. A single shared buffer instead yields whichever frame the GPU happened to
//! have finished, which varies with how far ahead the CPU is running and makes
//! the adaptation jitter. Mirrors `vulkan/auto_exposure.rs` and
//! `directx/auto_exposure.rs`, on the argument `metal/frame_rings.rs` already
//! makes for the transient rings.
#![deny(unsafe_op_in_unsafe_fn)]

use concinnity_core::gfx::auto_exposure;
use concinnity_core::gfx::auto_exposure::ExposureAdaptation;
use concinnity_core::render::error::{RenderError, RenderResult};
use concinnity_core::render::uniforms::*;
use objc2::rc::Retained;
use objc2::runtime::ProtocolObject;
use objc2_foundation::ns_string;
use objc2_metal::{
    MTLBuffer as _, MTLCommandBuffer as _, MTLComputeCommandEncoder as _, MTLComputePassDescriptor,
    MTLComputePipelineState, MTLSize, MTLTexture as _,
};

use super::builtin_shaders::compute_pipeline;
use super::context::*;
use super::encode::ComputeEncode;
use super::scoped_encoder::ScopedEncoder;

// All auto-exposure (EV adaptation) state grouped into one feature unit: the
// resolved tunables, the EMA-tracked adapted EV, the authored bias, the
// histogram-build + average compute pipelines, their shared/readback buffers,
// and the per-frame timing bookkeeping. All `Some` only when the world's
// `PostProcessConfig` turns auto-exposure on; otherwise the static-exposure
// path drives `post_process.exposure` directly and these stay `None`.
pub(crate) struct AutoExposureGpu {
    // The tunables, the authored `exposure_ev` bias and the EMA-tracked adapted
    // EV, updated each frame from the previous frame's GPU-measured average
    // log-luminance.
    pub adaptation: Option<ExposureAdaptation>,
    pub pipelines: Option<AutoExposurePipelines>,
    // 256-bin global histogram the build kernel accumulates into (shared
    // storage so the average kernel can read + clear it in one pass).
    pub histogram: Option<Retained<ProtocolObject<dyn objc2_metal::MTLBuffer>>>,
    // One-float readback buffer per frame-in-flight; the average kernel writes
    // this frame's slot and the CPU reads that same slot `depth` frames later,
    // past the fence that retired its writer. Empty when auto-exposure is off.
    pub outputs: Vec<Retained<ProtocolObject<dyn objc2_metal::MTLBuffer>>>,
    // Last frame's `elapsed`, used to derive a frame `dt` for the EMA step.
    pub last_elapsed: f32,
}

impl MtlContext {
    // Step the auto-exposure EMA from the previous frame's GPU measurement,
    // then push the new exposure multiplier into `self.post_process.exposure`.
    // A no-op when auto-exposure is disabled: the static authored EV then
    // drives `exposure` unchanged.
    //
    // `elapsed` is the total elapsed seconds since startup, the same value
    // `draw_frame` already receives. We diff against the previous call's
    // elapsed to derive a frame `dt`; on the first frame `dt` is 0 so the
    // EMA snaps to the initial state (midpoint of the clamp range).
    pub(super) fn update_auto_exposure(&mut self, elapsed: f32, slot: usize) {
        let Some(adaptation) = self.auto_exposure.adaptation.as_mut() else {
            return;
        };
        let Some(output_buf) = self.auto_exposure.outputs.get(slot) else {
            return;
        };

        // Read this slot's average log-luminance. The frames-in-flight fence
        // has already retired the frame that wrote it, so the value is a
        // completed GPU write exactly `depth` frames old -- deterministic
        // rather than whichever frame the GPU last happened to finish.
        // SAFETY: `output_buf` is shared storage of at least one f32 (the kernel's average
        // log-luminance output), so `contents()` is a live CPU mapping of it. A torn read is
        // impossible for a single aligned f32, and the unwritten value the first `depth` frames
        // see is the zero the buffer was created with.
        let avg_log_lum = unsafe {
            let ptr = output_buf.contents().as_ptr() as *const f32;
            ptr.read()
        };

        let dt = (elapsed - self.auto_exposure.last_elapsed).max(0.0);
        self.auto_exposure.last_elapsed = elapsed;

        // `self.post_process.exposure` is the linear multiplier the post pass
        // and bloom prefilter consume; it already folds in the authored
        // exposure_ev when auto-exposure is off, so we only overwrite it here
        // when the GPU path owns the value.
        self.post_process.exposure = adaptation.step(avg_log_lum, dt);
    }

    // Encode the auto-exposure histogram passes against `hdr_resolve`. The
    // build kernel runs one thread per HDR pixel; the average kernel runs
    // one threadgroup of 256 threads that reduces the histogram, clears it
    // for the next frame, and writes the average log-luminance to the
    // `slot`'s readback buffer, which the CPU reads at the top of the frame
    // that reuses this ring slot. A no-op when auto-exposure is disabled.
    // pub(in crate::metal) so the render-graph executor in
    // metal/graph_exec.rs can dispatch this pass from a CompiledGraph.
    pub(in crate::metal) fn encode_auto_exposure(
        &self,
        cmd_buf: &ProtocolObject<dyn objc2_metal::MTLCommandBuffer>,
        slot: usize,
    ) -> RenderResult<u32> {
        let (Some(pipelines), Some(histogram), Some(output)) = (
            self.auto_exposure.pipelines.as_ref(),
            self.auto_exposure.histogram.as_ref(),
            self.auto_exposure.outputs.get(slot),
        ) else {
            return Ok(0);
        };

        let params = AutoExposureParams::HISTOGRAM;
        let hdr_tex: &ProtocolObject<dyn objc2_metal::MTLTexture> =
            self.targets.hdr.hdr_resolve.as_ref();
        let tex_w = hdr_tex.width();
        let tex_h = hdr_tex.height();
        if tex_w == 0 || tex_h == 0 {
            return Ok(0);
        }

        let ae_desc = MTLComputePassDescriptor::new();
        if let Some(t) = &self.diagnostics.pass_timing {
            t.attach_compute(&ae_desc, super::pass_timing::PassId::AutoExposure);
        }
        let enc = ScopedEncoder::new(
            cmd_buf
                .computeCommandEncoderWithDescriptor(&ae_desc)
                .ok_or_else(|| {
                    RenderError::Other("failed to get auto-exposure compute encoder".into())
                })?,
            ns_string!("auto-exposure"),
        );

        // Build kernel: 16x16 threadgroups, one thread per HDR pixel.
        enc.set_pipeline(&pipelines.build);
        enc.set_texture(hdr_tex, 0);
        enc.set_buffer(histogram, 0, 0);
        enc.set_value(&params, 1);
        let tg = MTLSize {
            width: 16,
            height: 16,
            depth: 1,
        };
        let grid = MTLSize {
            width: tex_w,
            height: tex_h,
            depth: 1,
        };
        enc.dispatchThreads_threadsPerThreadgroup(grid, tg);

        // Average kernel: one threadgroup of 256 threads reduces the
        // histogram and clears it for the next frame. The output buffer at
        // buffer(1) receives the count-weighted average log-luminance.
        enc.set_pipeline(&pipelines.average);
        enc.set_buffer(histogram, 0, 0);
        enc.set_buffer(output, 0, 1);
        enc.set_value(&params, 2);
        let avg_grid = MTLSize {
            width: auto_exposure::HISTOGRAM_BINS,
            height: 1,
            depth: 1,
        };
        let avg_tg = MTLSize {
            width: auto_exposure::HISTOGRAM_BINS,
            height: 1,
            depth: 1,
        };
        enc.dispatchThreads_threadsPerThreadgroup(avg_grid, avg_tg);

        Ok(0)
    }
}

// Pair of compute pipelines driving the auto-exposure histogram path: the
// build kernel (one thread per HDR-resolve pixel) that accumulates a 256-bin
// log-luminance histogram, and the average kernel (one threadgroup of 256
// threads) that reduces the histogram to a single average log-luminance value
// and clears it for the next frame.
pub(super) struct AutoExposurePipelines {
    pub build: Retained<ProtocolObject<dyn MTLComputePipelineState>>,
    pub average: Retained<ProtocolObject<dyn MTLComputePipelineState>>,
}

// Build the auto-exposure compute pipelines from the single-source
// `auto_exposure.hlsl`. Each kernel compiles as its own variant so it declares
// only the resources it binds. Returned only when the world's
// `PostProcessConfig` opts into auto-exposure; otherwise the histogram pass is
// skipped entirely.
pub(super) fn build_auto_exposure_pipelines(
    device: &ProtocolObject<dyn objc2_metal::MTLDevice>,
    hot_reload: bool,
) -> RenderResult<AutoExposurePipelines> {
    Ok(AutoExposurePipelines {
        build: compute_pipeline(
            device,
            &super::builtin_shaders::AUTO_EXPOSURE_BUILD,
            hot_reload,
        )?,
        average: compute_pipeline(
            device,
            &super::builtin_shaders::AUTO_EXPOSURE_AVERAGE,
            hot_reload,
        )?,
    })
}