1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
//! Model-history snapshot compute pass. Once per frame, after the G-buffer
//! pre-pass has read the previous frame's slot, copies this frame's model
//! matrices out of the bindless object buffer into this frame's slot of the
//! model-history ring. Next frame's pre-pass reprojects through what this wrote.
//!
//! The snapshot is a GPU copy rather than a host-built parallel table because
//! the object buffer already carries every model: a host table would write the
//! same 64 bytes per record a second time.
#![deny(unsafe_op_in_unsafe_fn)]
use concinnity_core::render::error::{RenderError, RenderResult};
use concinnity_core::render::uniforms::ModelHistoryParams;
use objc2::rc::Retained;
use objc2::runtime::ProtocolObject;
use objc2_foundation::ns_string;
use objc2_metal::{
MTLBuffer, MTLCommandBuffer, MTLComputeCommandEncoder as _, MTLComputePipelineState, MTLSize,
};
use super::builtin_shaders::compute_pipeline;
use super::context::MtlContext;
use super::encode::ComputeEncode;
use super::scoped_encoder::ScopedEncoder;
// Threads per group, matching `[numthreads(64, 1, 1)]` in model_history.hlsl.
const THREADGROUP: usize = 64;
impl MtlContext {
// Encode one snapshot dispatch per target slot. `targets` is normally this
// frame's slot alone; on the frame a rebuild primes the ring it is every
// slot, so the first pre-pass to read one finds this frame's models rather
// than an unwritten buffer.
pub(in crate::metal) fn encode_model_history(
&self,
cmd_buf: &ProtocolObject<dyn MTLCommandBuffer>,
object_buffer: &Retained<ProtocolObject<dyn MTLBuffer>>,
targets: &[Retained<ProtocolObject<dyn MTLBuffer>>],
record_count: usize,
) -> RenderResult<()> {
let Some(pipeline) = &self.gbuffer.history_pipeline else {
return Ok(());
};
if record_count == 0 || targets.is_empty() {
return Ok(());
}
let params = ModelHistoryParams {
record_count: record_count as u32,
_pad: [0; 3],
};
// No timing attachment: this dispatch is not the `GBufferPrepass`
// pass, and claiming that pass's slot pair here made both encoders
// write it, so the reading was one encoder's start against the other's
// end. The snapshot is a handful of microseconds; the pre-pass it feeds
// is what the profiler reports.
let desc = objc2_metal::MTLComputePassDescriptor::new();
let enc = ScopedEncoder::new(
cmd_buf
.computeCommandEncoderWithDescriptor(&desc)
.ok_or_else(|| {
RenderError::Other("failed to get model-history compute encoder".to_string())
})?,
ns_string!("model history"),
);
enc.set_pipeline(pipeline);
enc.set_value(¶ms, 0);
enc.set_buffer(object_buffer, 0, 1);
let grid = MTLSize {
width: record_count,
height: 1,
depth: 1,
};
let tg = MTLSize {
width: THREADGROUP,
height: 1,
depth: 1,
};
for target in targets {
enc.set_buffer(target, 0, 2);
enc.dispatchThreads_threadsPerThreadgroup(grid, tg);
}
Ok(())
}
}
// Build the model-history compute pipeline from the single-source
// `model_history.hlsl` (params buffer(0), objects buffer(1), history
// buffer(2) -- the same slots the encode above binds).
pub(super) fn build_model_history_pipeline(
device: &ProtocolObject<dyn objc2_metal::MTLDevice>,
hot_reload: bool,
) -> RenderResult<Retained<ProtocolObject<dyn MTLComputePipelineState>>> {
compute_pipeline(device, &super::builtin_shaders::MODEL_HISTORY, hot_reload)
}