1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
//! The stages `VkContext::draw_frame` runs in order around `record_frame`, and
//! the per-frame state `record_frame` advances around its graph dispatch.
use ash::vk;
use concinnity_core::components;
use concinnity_core::profile;
use concinnity_core::profile::PassTiming;
use concinnity_core::render::error::RenderResult;
use concinnity_core::render::hdr_output;
use concinnity_core::render::pass_timing;
use concinnity_core::render::shadow_schedule::{CascadeCamera, CascadeLight};
use concinnity_core::render::view_history::ViewFrame;
use super::upload_shadow_uniforms;
use crate::gpu_wait::GpuWait;
use crate::vulkan::context::VkContext;
use crate::vulkan::error::map_vk_result;
impl VkContext {
// Rebuilds requested since last frame: wireframe pipelines and hot-reloaded shaders.
pub(in crate::vulkan) fn apply_pending_rebuilds(&mut self) -> RenderResult<()> {
// Vulkan polygon mode is pipeline state, so the wireframe view needs its
// own main-pass pipelines; built here on the first wireframe frame.
self.ensure_wireframe_pipelines();
// Shader hot-reload: if the debug `reload-shaders` command set the flag, rebuild every
// built-in pipeline from disk-resident source before this frame's passes start using them.
// The flag is cleared regardless of outcome so a failed rebuild (typo in a shader edit)
// doesn't loop, and the previous pipelines stay live so the session keeps rendering; only a
// device failure propagates. Wait for the GPU to drain first so swapping pipelines out from
// under in-flight command buffers is safe. Mirrors the DirectX `apply_pending_rebuilds`.
if self.shader_reload_requested() {
self.clear_shader_reload_flag();
self.wait_idle();
match self.reload_shaders() {
Ok(()) => tracing::info!("hot-reload: shader pipelines rebuilt"),
Err(e) if e.is_device_failure() => return Err(e),
Err(e) => tracing::error!("hot-reload: shader rebuild failed: {}", e),
}
}
Ok(())
}
// Blocks on this frame slot's previous submission, then runs the ticks it gates.
pub(in crate::vulkan) fn wait_frame_slot(&mut self, frame: usize) -> RenderResult<GpuWait> {
// Wait for this frame's slot to finish. Measured, with the swapchain
// acquire in `acquire_frame`, into the frame's `gpu_wait_us`: both block
// the CPU on the GPU inside `draw_frame`, which the engine times its
// graphics system around.
let mut gpu_wait = crate::gpu_wait::GpuWait::none();
gpu_wait
.measure(|| {
// SAFETY: the fence belongs to this frame slot and was created from this device; the
// slice borrows it for the call.
unsafe {
self.hw.device.wait_for_fences(
std::slice::from_ref(&self.frame_sync.in_flight[frame]),
true,
u64::MAX,
)
}
})
.map_err(|e| map_vk_result(e, "wait fences"))?;
// Streamed texture swaps: re-point this frame slot's bindless pool
// copy at the swapped-in views (legal now -- the fence wait above
// retired every command buffer that binds this slot's set), and free
// the old images / upload transients this slot parked on its previous
// trip (this slot's fence signaling also covers the older frames that
// last sampled them, and every pool copy has been re-pointed since).
self.apply_streamed_texture_rewrites(frame);
// Reclaim this frame slot's shared post-pass descriptor sets. Here for
// the same reason as the two ticks below: the fence wait above is what
// makes reclaiming the previous pass's sets legal.
self.post.arena.begin_frame(&self.hw.device, frame);
// Tick the device allocator: destroy retired handles, reclaim retired
// ranges, release empty blocks. Here because the fence wait above is
// what guarantees a range freed `retire_depth` ticks ago is no longer
// referenced.
self.hw.alloc.begin_frame();
// Same tick for the owned pipeline / layout / render-pass handles a
// rebuild displaced, on the same reasoning.
self.hw.device.begin_frame();
// Same tick for the staged geometry writes' ring and command buffers.
self.geometry_uploads.get_mut().begin_frame();
// Periodic footprint readout, for measuring the pool under streaming
// churn at scale. Inert unless debug logging is enabled.
if self.stream.frame.is_multiple_of(1024) && tracing::enabled!(tracing::Level::DEBUG) {
tracing::debug!("device allocator: {}", self.hw.alloc.stats());
}
Ok(gpu_wait)
}
// Probe bake and auto-exposure steps that need the slot's GPU work retired.
pub(in crate::vulkan) fn service_background_work(&mut self, elapsed: f32, frame: usize) {
// Advance the staggered reflection-probe bake one step. Runs here -- after
// this frame's slot fence wait, before `record_frame` -- so any cube it
// installs (a binding-8 rewrite + `probe.set.count` bump) is picked up by this
// frame's `record_frame` ProbeSet upload + rendering. Non-fatal.
self.bake_pending_probes();
// Auto-exposure: step the EMA from a previous frame's GPU
// measurement before any pipeline reads `post_process.exposure`.
// The `wait_frame_slot` fence wait already gated the
// GPU work that wrote this slot's readback, so the value is
// committed. No-op when auto-exposure is disabled.
self.update_auto_exposure(elapsed, frame);
}
// Whole-frame and per-pass GPU times this slot's last submission resolved.
pub(in crate::vulkan) fn read_gpu_timings(
&self,
frame: usize,
) -> (u32, [PassTiming; profile::MAX_PASS_TIMINGS]) {
let device = &self.hw.device;
// GPU timing for the most-recently completed block on this frame slot:
// the whole-frame pair plus one (start, end) pair per render pass. The
// `wait_frame_slot` fence wait guarantees the previous trip's writes have
// retired, so the available query results are committed. The block is read with
// `WITH_AVAILABILITY` so a pass that did not run this trip (its slots were
// reset but never written) reads back unavailable -> 0, without stalling
// the host (no `WAIT`). Zero before a slot has been visited a second time.
let empty_pass_times = [("", 0u32); profile::MAX_PASS_TIMINGS];
if let Some(pool) = self.hw.timestamp_query_pool {
// One [value, availability] pair per query slot (TYPE_64 +
// WITH_AVAILABILITY -> two u64 per query; ash uses the element size as
// the stride and the slice length as the query count).
let mut results = vec![[0u64; 2]; pass_timing::SLOTS_PER_FRAME];
// SAFETY: a property query on a live handle; it only reads.
let res = unsafe {
device.get_query_pool_results(
pool,
pass_timing::frame_block_base(frame),
&mut results,
vk::QueryResultFlags::TYPE_64 | vk::QueryResultFlags::WITH_AVAILABILITY,
)
};
// WITH_AVAILABILITY fills the buffer + per-query availability bits and
// returns SUCCESS; tolerate NOT_READY defensively (the buffer is still
// written, and the availability bits gate every read).
if matches!(res, Ok(()) | Err(vk::Result::NOT_READY)) {
let period = self.hw.timestamp_period_ns;
let pair_micros = |start_slot: usize, end_slot: usize| -> u32 {
let [s_val, s_avail] = results[start_slot];
let [e_val, e_avail] = results[end_slot];
if s_avail != 0 && e_avail != 0 && e_val > s_val && period > 0.0 {
let nanos = (e_val - s_val) as f64 * period as f64;
((nanos / 1000.0) as u64).min(u32::MAX as u64) as u32
} else {
0
}
};
pass_timing::decode_frame_block(pair_micros)
} else {
(0, empty_pass_times)
}
} else {
(0, empty_pass_times)
}
}
// Publishes this frame's stats before recording; draw calls fill in after.
pub(in crate::vulkan) fn begin_frame_stats(
&self,
gpu_wait: &GpuWait,
(gpu_frame_us, pass_times_us): (u32, [PassTiming; profile::MAX_PASS_TIMINGS]),
) {
// Reset this frame's render stats. `record_frame` accumulates
// `draw_calls` through `inc_draw_calls` (interior-mutability since
// the encoders run through `&self`); the rest is filled here from
// context state.
let counts = crate::object_counts::object_counts(
self.state.draw.objects.len(),
self.instanced.clusters.iter().map(|c| c.instances.len()),
self.state.skinned.draw_objects.iter().map(|o| o.visible),
);
let vram_bytes = self.query_vram_bytes();
let transient_pool_bytes = self.targets.transient_pool.allocated_bytes();
// Reset the parallel-safe draw-call accumulator for this frame; the
// encoders fetch_add into it during recording and `record_frame`
// drains it back into `frame_stats.draw_calls` once recording is done.
self.draw_calls_accum
.store(0, std::sync::atomic::Ordering::Relaxed);
self.frame_stats.set(profile::RenderStats {
draw_calls: 0,
objects: counts.objects,
skinned_visible: counts.skinned_visible,
gpu_frame_us,
// The fence wait alone so far; `acquire_frame` adds to it.
gpu_wait_us: gpu_wait.micros(),
vram_bytes,
transient_pool_bytes,
pass_times_us,
// Adapted auto-exposure EV for the StatHud `EV` chip. `Some` only
// when the world opted into auto-exposure (the EMA state is then
// live); the static-exposure path leaves it `None` so the chip
// stays blank. The value is the EV the most recent
// `update_auto_exposure` EMA step settled on (the multiplier the
// post stack pushes is `2^ev`). Mirrors `DxContext` / `MtlContext`.
auto_exposure_ev: self
.auto_exposure
.adaptation
.as_ref()
.map(|a| a.current_ev()),
// EDR headroom for the StatHud `EDR x.X` chip, taken from the
// `HdrOutputMode` resolved at init. `Some` only on the HDR path
// (Vulkan has no portable max-EDR query, so the value is the
// synthesized placeholder set in `init`); `None` on SDR blanks the
// chip. Mirrors `DxContext` / `MtlContext::render_stats`.
max_edr: match self.hw.hdr_mode {
hdr_output::HdrOutputMode::Hdr { max_edr, .. } => Some(max_edr),
hdr_output::HdrOutputMode::Sdr => None,
},
..profile::RenderStats::default()
});
}
// Acquires the swapchain image, or `None` when the swapchain was rebuilt instead.
pub(in crate::vulkan) fn acquire_frame(
&mut self,
frame: usize,
gpu_wait: &mut GpuWait,
) -> RenderResult<Option<u32>> {
// Acquire swapchain image. Blocks when the presentation engine holds
// every image, so it is the display-paced half of the frame's GPU wait.
let acquire = gpu_wait.measure(|| {
// SAFETY: `self.swapchain.handle` is the live swapchain and `image_available[frame]` is
// an unsignaled semaphore from this device's own pool for this frame slot.
unsafe {
self.swapchain.loader.acquire_next_image(
self.swapchain.handle,
u64::MAX,
self.frame_sync.image_available[frame],
vk::Fence::null(),
)
}
});
// Fold the acquire into the reading `begin_frame_stats` published, which
// the stats snapshot had already captured with the fence wait alone.
let mut waited = self.frame_stats.get();
waited.gpu_wait_us = gpu_wait.micros();
self.frame_stats.set(waited);
let image_index = match acquire {
Ok((idx, suboptimal)) => {
if suboptimal {
self.rebuild_swapchain()?;
return Ok(None);
}
idx
}
Err(vk::Result::ERROR_OUT_OF_DATE_KHR) => {
self.rebuild_swapchain()?;
return Ok(None);
}
Err(e) => return Err(map_vk_result(e, "acquire swapchain image")),
};
let device = &self.hw.device;
// SAFETY: the fence belongs to this frame slot and was just waited on, so it is signaled
// and not in use by a pending submission.
unsafe { device.reset_fences(std::slice::from_ref(&self.frame_sync.in_flight[frame])) }
.map_err(|e| map_vk_result(e, "reset fences"))?;
Ok(Some(image_index))
}
// Submits the recorded buffers and presents, rebuilding the swapchain when out of date.
pub(in crate::vulkan) fn submit_and_present(
&mut self,
frame: usize,
image_index: u32,
submit_bufs: &[vk::CommandBuffer],
) -> RenderResult<()> {
let device = &self.hw.device;
// Submit the whole batch in one call: submission order = GPU order on
// the single graphics queue. The render-finished semaphore is indexed
// by swapchain image (not frame slot) so present never reuses one still
// in flight.
let wait_sems = [self.frame_sync.image_available[frame]];
let wait_stages = [vk::PipelineStageFlags::COLOR_ATTACHMENT_OUTPUT];
let signal_sems = [self.frame_sync.render_finished[image_index as usize]];
let submit_info = vk::SubmitInfo::default()
.wait_semaphores(&wait_sems)
.wait_dst_stage_mask(&wait_stages)
.command_buffers(submit_bufs)
.signal_semaphores(&signal_sems);
// SAFETY: every command buffer in `submit_bufs` was ended and belongs to this frame slot,
// the semaphores and fence were created from this device, and `submit_info` borrows all of
// them for the call.
unsafe {
device
.queue_submit(
self.hw.graphics_queue,
std::slice::from_ref(&submit_info),
self.frame_sync.in_flight[frame],
)
.map_err(|e| map_vk_result(e, "queue submit"))?;
}
// Present.
let swapchains = [self.swapchain.handle];
let image_indices = [image_index];
let present_info = vk::PresentInfoKHR::default()
.wait_semaphores(&signal_sems)
.swapchains(&swapchains)
.image_indices(&image_indices);
// SAFETY: `present_info` borrows the swapchain, image index, and wait semaphore for the
// call; the semaphore is signaled by the submission above.
let present_result = unsafe {
self.swapchain
.loader
.queue_present(self.hw.present_queue, &present_info)
};
if present_result == Err(vk::Result::ERROR_OUT_OF_DATE_KHR) || present_result == Ok(true) {
self.rebuild_swapchain()?;
} else {
present_result.map_err(|e| map_vk_result(e, "present"))?;
// Record which swapchain image now holds a complete, presented frame
// so the `screenshot` debug command can read it back.
self.swapchain.last_present_index = Some(image_index);
}
self.current_frame = (self.current_frame + 1) % self.frames_in_flight;
Ok(())
}
// Cascade light VPs and splits for this camera, the shadow UBO upload, and the spot schedule.
pub(super) fn update_shadow_schedule(
&mut self,
extent: vk::Extent2D,
cam_pos: [f32; 3],
fov_y_radians: f32,
near: f32,
view_distance: Option<f32>,
frame_idx: usize,
) {
// Recompute cascade VPs + splits from the current camera + light, and
// push the result to the shadow UBO so both passes see the same data.
let cascade_aspect = if extent.height == 0 {
1.0
} else {
extent.width as f32 / extent.height as f32
};
if self.shadow.enabled() {
let camera = CascadeCamera {
view: self.state.view.matrix,
position: cam_pos,
fov_y_rad: fov_y_radians,
aspect: cascade_aspect,
near,
view_distance,
};
let shadow = &mut self.shadow;
let light = CascadeLight {
dir_to_source: shadow.light_dir,
map_size: shadow.map_size,
};
// Skipped cascades keep the VP + depth their slice was last rendered
// with; encode_shadow_pass re-rasterizes only the masked slices.
shadow.render_mask =
shadow
.scheduler
.refresh(&mut shadow.uniforms, &shadow.cadence, light, camera);
upload_shadow_uniforms(&self.shadow.ubos[frame_idx], &self.shadow.uniforms);
}
// Spot shadow refresh schedule. Prime-then-round-robin over the slices,
// so N shadowed spots cost one extra depth render per frame rather than
// N. No uniform refresh: the projections are static and were baked at
// init. A no-op (mask stays 0) when the world has no shadowed spot.
self.spot_shadow.advance(matches!(
self.shadow.cadence.update,
components::ShadowUpdate::EveryFrame
));
}
// Drop every accumulated temporal history for a frame that does not
// continue the last: the TAA and SSGI rings, the camera half of the motion
// history, the Hi-Z pyramid the occlusion test would reproject the old view
// through, and the upscaler's history on its next dispatch.
pub(in crate::vulkan) fn reset_temporal_history(&mut self) {
self.cull.hiz_valid = false;
if let Some(taa) = &mut self.taa {
taa.pass.reset_history();
}
if let Some(ssgi) = &mut self.ssgi {
ssgi.reset_history();
}
if let Some(gb) = &mut self.gbuffer {
gb.view_history.reset();
}
if let Some(upscaler) = &self.upscale {
upscaler.request_history_reset();
}
}
// History the next frame reads: TAA jitter and ring, G-buffer VP, Hi-Z VP and validity.
pub(super) fn advance_temporal_state(&mut self, cur: ViewFrame) {
// The Hi-Z reduction that feeds next frame's cull is the graph's terminal
// `HizFinal` pass, so it has already been recorded; `hiz_valid` only
// tracks whether a pyramid at the current resolution now exists.
// The cascade slices rest sampled (SHADER_READ_ONLY_OPTIMAL) between
// frames; next frame's Shadow producer barrier (graph-driven) performs
// the SHADER_READ_ONLY -> DEPTH_STENCIL_ATTACHMENT reset over every
// cascade layer, so no inline end-of-frame restore is needed here.
// Advance the TAA jitter sequence and the accumulation ring that
// validates next frame's history. The motion-vector temporal state lives
// on the unified G-buffer (advanced below); TAA only consumes its
// velocity view.
if let Some(taa) = &mut self.taa {
taa.taa_frame = taa.taa_frame.wrapping_add(1);
// Step the accumulation ring in lockstep: what this frame wrote is
// next frame's history.
taa.pass.advance();
}
// What the SSGI accumulation wrote this frame is next frame's history.
if let Some(ssgi) = &mut self.ssgi {
ssgi.advance();
}
// Advance the unified G-buffer's velocity-channel temporal state in
// lockstep with TAA's: this frame's un-jittered VP, clock and camera
// position become the previous frame next frame reprojects to. The
// per-object half of the same history was snapshotted on the GPU by the
// pre-pass's own dispatch. Owned by `GbufferResources` so the motion
// vector works for any consumer (TAA or FSR), exactly mirroring the TAA
// advance above.
if let Some(gb) = &mut self.gbuffer {
gb.view_history.advance(cur);
}
// Advance Hi-Z temporal state: this frame's un-jittered VP becomes next
// frame's occlusion-test projection, and the pyramid the graph's
// `HizFinal` pass just wrote is now valid for next frame's cull (kept
// independent of TAA, which may be off while Hi-Z is on).
if self.cull.hiz.is_some() {
self.cull.hiz_prev_view_proj = cur.vp;
self.cull.hiz_valid = true;
}
}
pub(super) fn finish_frame_stats(&self) {
// Drain the parallel-safe draw-call accumulator (bumped by every pass
// encoder, including those fanned onto rayon workers) into this frame's
// `frame_stats` for the profiler overlay. All recording is done by here.
let mut stats = self.frame_stats.get();
stats.draw_calls = self
.draw_calls_accum
.load(std::sync::atomic::Ordering::Relaxed);
self.frame_stats.set(stats);
}
}