Skip to main content

molgfx_render/engine/
frame.rs

1//! The frame loop: sync, build, record, one submission, present.
2//!
3//! CPU cost per frame is proportional to the active passes, the placed
4//! structures, the resident semantic tables and the representation states —
5//! never to atom or bond count, because per-primitive work lives on the GPU.
6//! A structure or representation whose revisions are unchanged costs one
7//! comparison and no upload, so a still scene walks a short fixed list rather
8//! than touching its contents. Nothing here allocates on the steady-state path
9//! beyond the first frame's pool construction.
10
11use super::{
12    Engine, FrameCompleteness, FrameDegradation, FrameMetrics, FrameReport, FrameStatus,
13    MotionBlur, QualityTier, RenderMode, TemporalOptions,
14};
15use crate::error::RenderError;
16use crate::graph::{DisplayEncoding, PassContext, ResourceTable, TransientPool, plan_aliases};
17use crate::passes::FrameBindings;
18use molgfx_core::Scene;
19use molgfx_gpu::Queue as _;
20use molgfx_gpu::{Device, Surface as _, SurfaceError, SurfaceFrame as _};
21use molgfx_math::Camera;
22
23/// Returns the number of frame submissions currently pending completion.
24#[inline]
25fn pending_frame_submissions<D: Device>(engine: &Engine<D>) -> u32 {
26    u32::from(engine.frame_submission_pending)
27}
28
29/// Converts a duration to nanoseconds, saturating instead of panicking on
30/// overflow.
31#[inline]
32fn duration_ns(duration: std::time::Duration) -> u64 {
33    u64::try_from(duration.as_nanos()).unwrap_or(u64::MAX)
34}
35
36impl<D: Device> Engine<D> {
37    /// Uploads everything the scene changed since the previous frame.
38    ///
39    /// Returns whether scene data changed and invalidates temporal convergence
40    /// when scene data changed or uploads remain in flight.
41    ///
42    /// # Errors
43    ///
44    /// Returns an error when occupancy preparation, scene synchronization, or
45    /// chunk residency synchronization fails.
46    fn sync_scene(&mut self, scene: &Scene) -> Result<bool, RenderError> {
47        self.ensure_occupancy(scene)?;
48        self.chunk_residency.poll(&self.device, &self.queue)?;
49
50        let scene_changed = self.scene_gpu.sync(crate::scene_gpu::SceneSync {
51            device: &self.device,
52            queue: &self.queue,
53            scene,
54            quality: self.tier() >= QualityTier::Standard,
55            extent: [self.width, self.height],
56            ray_query_layout: self.passes.ambient_occlusion.ray_query_layout(),
57            derived_cache: &mut self.derived_cache,
58            derived_frame: self.derived_frame,
59        })?;
60
61        self.chunk_residency.sync_scene(
62            &mut self.scene_gpu,
63            &self.device,
64            &self.queue,
65            &mut self.derived_cache,
66            self.derived_frame,
67        )?;
68        self.chunk_residency.flush(&self.device, &self.queue)?;
69
70        self.derived_frame = self.derived_frame.wrapping_add(1);
71
72        if scene_changed {
73            self.temporal.invalidate_convergence();
74        }
75
76        Ok(scene_changed)
77    }
78
79    /// Updates the cached temporal scene identity.
80    ///
81    /// Returns `true` when the identity changed. An unchanged scene avoids an
82    /// unnecessary write to the cached identity on the steady-state path.
83    #[inline]
84    fn update_temporal_scene_identity(&mut self, scene: &Scene) -> bool {
85        let identity = scene.cache_identity();
86        let changed = self.temporal_scene_identity != Some(identity);
87
88        if changed {
89            self.temporal_scene_identity = Some(identity);
90        }
91
92        changed
93    }
94
95    /// Returns whether temporal accumulation must restart for the current frame.
96    #[inline]
97    fn temporal_reset_required(
98        &self,
99        scene_reset: bool,
100        pool_rebuilt: bool,
101        camera_changed: bool,
102    ) -> bool {
103        scene_reset || pool_rebuilt || (self.mode == RenderMode::Cinematic && camera_changed)
104    }
105
106    /// Renders one frame of the scene to the presentation surface.
107    ///
108    /// A lost or outdated surface is reconfigured and returns
109    /// `FrameStatus::Skipped`; the next frame recovers. Caller-reachable
110    /// conditions do not panic.
111    ///
112    /// # Errors
113    ///
114    /// Returns an error for unrecoverable GPU errors, scene synchronization
115    /// failures, or render-graph resource reconstruction failures.
116    pub fn render(&mut self, scene: &Scene, camera: &Camera) -> Result<FrameReport, RenderError> {
117        // On native targets the adaptive controller observes this call's CPU
118        // duration — synchronization, recording and submission included. The
119        // submission itself is asynchronous there, so measuring host time is
120        // the honest per-frame cost; the fence poll at the top of the next
121        // call deliberately does not sample a second time. On browser targets
122        // submission returns immediately, so the controller instead samples
123        // submission-to-fence-completion elapsed time in
124        // `poll_pending_submission`; measuring this call there would classify
125        // queued GPU work as free.
126        #[cfg(not(target_arch = "wasm32"))]
127        let render_started_at = self.clock_origin.elapsed();
128
129        if self.poll_pending_submission()? {
130            // Fence-only skip: the previous submission is still pending.
131            // This is not surface/pool work that a retry produces; whether
132            // another frame is needed is decided by the report's upload and
133            // temporal conditions below.
134            #[cfg(not(test))]
135            return Ok(self.frame_report(FrameStatus::Skipped, true));
136        }
137
138        self.prepare_frame(scene, camera)?;
139
140        let Some(frame) = self.acquire_surface_frame()? else {
141            return Ok(self.frame_report(FrameStatus::Skipped, false));
142        };
143
144        let mut encoder = self.device.create_command_encoder();
145
146        if !self.record_frame(&mut encoder, scene, frame.view()) {
147            return Ok(self.frame_report(FrameStatus::Skipped, false));
148        }
149
150        self.submit_frame(encoder);
151        self.device.check_errors()?;
152        frame.present();
153
154        // CPU encoding/submission cost for the native adaptive loop. This is
155        // not device execution time; GPU time needs timestamp queries.
156        #[cfg(not(target_arch = "wasm32"))]
157        self.adaptive.observe(duration_ns(
158            self.clock_origin
159                .elapsed()
160                .saturating_sub(render_started_at),
161        ));
162
163        Ok(self.frame_report(FrameStatus::Presented, false))
164    }
165
166    /// Synchronizes CPU/GPU state and prepares per-frame temporal uniforms.
167    ///
168    /// Scene specializations are settled once here before command recording,
169    /// avoiding redundant specialization work during the presented-frame path.
170    ///
171    /// # Errors
172    ///
173    /// Returns an error for GPU validation or device failures, scene
174    /// synchronization failures, pool reconstruction failures, optics
175    /// resolution failures, or frame-uniform upload failures.
176    fn prepare_frame(&mut self, scene: &Scene, camera: &Camera) -> Result<(), RenderError> {
177        self.sync_quality_tier();
178        self.device.check_errors()?;
179
180        self.chunk_residency.begin_epoch();
181        self.scene_gpu.begin_frame();
182
183        let scene_changed = self.sync_scene(scene)?;
184        let pool_rebuilt = self.rebuild_pool_if_needed()?;
185
186        self.scene_gpu
187            .settle_specializations(&self.device, scene, &self.passes);
188
189        let camera_changed = self.temporal.camera_changed(camera);
190        let scene_reset = self.update_temporal_scene_identity(scene);
191        let cinematic = self.tier() >= QualityTier::Standard;
192
193        let optics = self.resolve_optics(scene, camera)?;
194        let shadow =
195            self.shadow_bound
196                .fit(scene, camera, self.resolved_plan.lighting(), scene_changed);
197
198        let uniforms = self.temporal.prepare(
199            camera,
200            &TemporalOptions {
201                extent: [self.width, self.height],
202                reset: self.temporal_reset_required(scene_reset, pool_rebuilt, camera_changed),
203                quality: cinematic,
204                publication: false,
205                illustration: self.resolved_plan.illustration(),
206                optics,
207                motion_blur: self
208                    .resolved_plan
209                    .motion_blur()
210                    .map_or([0.0; 4], MotionBlur::packed),
211                atmosphere: self
212                    .resolved_plan
213                    .packed_presentation(self.scene_gpu.has_translucency()),
214                lighting: self.resolved_plan.packed_lighting(),
215                shadow_view: shadow.view,
216                shadow_projection: shadow.projection,
217                shadow_view_proj: shadow.view_projection,
218            },
219        );
220
221        self.scene_gpu.write_frame_uniforms(&self.queue, &uniforms)
222    }
223
224    /// Records compute work and render-graph passes into the frame encoder.
225    ///
226    /// Returns `false` when no transient pool is available.
227    ///
228    /// `scene` remains part of the existing signature for compatibility.
229    /// Specialization settlement is performed once by `prepare_frame` before
230    /// this method is reached.
231    fn record_frame(
232        &mut self,
233        encoder: &mut D::CommandEncoder,
234        scene: &Scene,
235        swapchain: &D::TextureView,
236    ) -> bool {
237        // Preserve the existing signature without repeating specialization work.
238        let _ = scene;
239
240        let cinematic = self.tier() >= QualityTier::Standard;
241
242        self.record_scene_compute(encoder, cinematic);
243
244        let Some(pool) = &self.pool else {
245            return false;
246        };
247
248        let table = ResourceTable { pool, swapchain };
249
250        for &index in &self.order {
251            let Some(node) = self.pass_nodes.get(index) else {
252                continue;
253            };
254
255            let mut ctx = PassContext {
256                encoder: &mut *encoder,
257                resources: &table,
258                passes: &self.passes,
259                bindings: self.bindings.as_ref(),
260                scene: &self.scene_gpu,
261                timestamps: None,
262                temporal_write: self.temporal.write_index(),
263                quality: cinematic,
264                display_encoding: self.display_encoding(),
265            };
266
267            (node.record)(&mut ctx);
268        }
269
270        true
271    }
272
273    /// Submits the recorded frame once and updates submission bookkeeping.
274    ///
275    /// The submission timestamp is host time when the encoder reached the
276    /// queue; the completion timestamp is cleared and stays unset until a
277    /// later poll observes the backend fence. That fence fires when submitted
278    /// GPU work is done *executing* on the device, strictly later than host
279    /// queue-completion, and is never GPU execution time itself.
280    fn submit_frame(&mut self, encoder: D::CommandEncoder) {
281        self.last_submission_id = self.last_submission_id.wrapping_add(1);
282        self.last_submission_timestamp_ns = duration_ns(self.clock_origin.elapsed());
283        self.last_completion_timestamp_ns = None;
284
285        self.last_frame_submission = self.queue.submit_tracked(encoder);
286        self.frame_submission_pending = true;
287
288        // Only the browser adaptive loop samples submission-to-completion
289        // elapsed time, so only it needs the submission moment recorded.
290        #[cfg(target_arch = "wasm32")]
291        {
292            self.frame_submitted_at = Some(self.clock_origin.elapsed());
293        }
294    }
295
296    /// Polls the submission fence and updates completion timing.
297    ///
298    /// Returns whether the most recent frame submission remains pending.
299    ///
300    /// A pending-to-complete transition records the host completion
301    /// timestamp. That timestamp is host-observation latency, not exact device
302    /// completion time: the backend fence callback may have fired earlier, and
303    /// the poll that notices it runs on the host clock. GPU execution time is
304    /// not reported here at all; it requires timestamp queries and belongs to
305    /// the profiling path.
306    ///
307    /// On browser targets a completed submission also feeds one
308    /// submission-to-completion elapsed time to the adaptive controller. The
309    /// sample includes host scheduling latency between submit and the fence
310    /// callback, so it is a conservative queue-depth signal, not device time.
311    /// On native targets this poll is a no-op for the controller: `render`
312    /// already measured the frame's full CPU duration before returning, and a
313    /// second observation here would double-count the frame.
314    ///
315    /// # Errors
316    ///
317    /// Returns an error when the device cannot report the completed fence.
318    fn poll_pending_submission(&mut self) -> Result<bool, RenderError> {
319        let was_pending = self.frame_submission_pending;
320        let completed = self.queue.completed_fence(&self.device)?;
321
322        self.frame_submission_pending = completed < self.last_frame_submission;
323
324        if was_pending && !self.frame_submission_pending {
325            self.last_completion_timestamp_ns = Some(duration_ns(self.clock_origin.elapsed()));
326
327            // Only the browser path has an empty controller sample here. On
328            // native, `render` observed the complete CPU duration, so a fence
329            // observation would double-count this frame's budget.
330            #[cfg(target_arch = "wasm32")]
331            if let Some(submitted_at) = self.frame_submitted_at.take() {
332                self.adaptive.observe(duration_ns(
333                    self.clock_origin.elapsed().saturating_sub(submitted_at),
334                ));
335            }
336        }
337
338        Ok(self.frame_submission_pending)
339    }
340
341    /// Records scene-level compute passes required before render-graph passes.
342    ///
343    /// Coordinate-change signals are combined without allocating and drive
344    /// dynamic relation resolution exactly once.
345    fn record_scene_compute(&mut self, encoder: &mut D::CommandEncoder, cinematic: bool) {
346        self.passes
347            .cull
348            .record_attribute_timelines(&self.scene_gpu, encoder);
349
350        self.passes
351            .cull
352            .record_instance_timelines(&self.scene_gpu, encoder);
353
354        let point_coordinates_changed = self
355            .passes
356            .cull
357            .record_point_timelines(&self.scene_gpu, encoder);
358
359        self.scene_gpu
360            .record_particle_motion(encoder, &self.passes.particle_motion);
361
362        let structure_coordinates_changed =
363            self.scene_gpu
364                .record_trajectories(encoder, &self.passes.trajectory, None);
365
366        let paged_coordinates_changed = self
367            .passes
368            .cull
369            .record_paged_trajectories(&self.scene_gpu, encoder);
370
371        let coordinates_changed =
372            structure_coordinates_changed || paged_coordinates_changed || point_coordinates_changed;
373
374        self.scene_gpu.record_dynamic_relations(
375            encoder,
376            &self.passes.relation_resolve,
377            coordinates_changed,
378        );
379
380        self.scene_gpu
381            .record_occupancies(encoder, self.passes.occupancy.as_ref());
382
383        self.scene_gpu.record_surface_fields(
384            encoder,
385            &self.passes.surface_field,
386            &self.passes.surface_components,
387        );
388
389        self.scene_gpu.record_quality_hardware(encoder, cinematic);
390    }
391
392    /// Builds the externally visible report for the current frame state.
393    ///
394    /// The report samples existing counters and resource metrics without
395    /// introducing heap allocation in this layer.
396    ///
397    /// `fence_pending` marks the skip as a poll of a still-pending previous
398    /// submission. Such a skip is not surface or pool work that the next
399    /// frame recovers, so it does not by itself request another frame:
400    /// pending uploads and temporal convergence remain authoritative. A
401    /// surface/pool skip (`fence_pending == false`) keeps the existing
402    /// retry-next-frame contract.
403    fn frame_report(&self, status: FrameStatus, fence_pending: bool) -> FrameReport {
404        let residency = self.chunk_residency.metrics();
405        let derived = self.derived_cache.usage();
406        let physical = self.device.resource_memory();
407        let pending = residency.uploads.active_tickets;
408
409        FrameReport {
410            status,
411            completeness: if pending == 0 {
412                FrameCompleteness::Complete
413            } else {
414                FrameCompleteness::Progressive {
415                    pending_chunks: pending,
416                }
417            },
418            degradation: FrameDegradation::streaming_proxy(
419                self.mode == RenderMode::Realtime && pending != 0,
420            ),
421            metrics: FrameMetrics {
422                tracked_chunks: residency.tracked_chunks,
423                upload_in_flight_bytes: residency.uploads.in_flight_bytes,
424                pending_frame_submissions: pending_frame_submissions(self),
425                last_submission_id: self.last_submission_id,
426                submission_timestamp_ns: self.last_submission_timestamp_ns,
427                completion_timestamp_ns: self.last_completion_timestamp_ns,
428                derived_cache_gpu_bytes: derived.gpu_bytes,
429                derived_cache_peak_gpu_bytes: derived.peak_gpu_bytes,
430                physical_buffer_bytes: physical.buffer_bytes,
431                physical_texture_bytes: physical.texture_bytes,
432                physical_total_bytes: physical.total_bytes(),
433                physical_peak_bytes: physical.peak_bytes,
434            },
435            needs_another_frame: (status == FrameStatus::Skipped && !fence_pending) || pending != 0,
436            quality_tier: self.tier(),
437        }
438    }
439    ///
440    /// Selects a pre-built tonemap pipeline rather than a per-pixel branch, so
441    /// the encoding is fixed for the whole frame by construction.
442    #[inline]
443    pub(super) fn display_encoding(&self) -> DisplayEncoding {
444        let display = self.resolved_plan.display();
445
446        DisplayEncoding {
447            gamut: display.gamut,
448            transfer: display.transfer,
449        }
450    }
451
452    /// Rebuilds transient render resources when the presentation extent changes.
453    ///
454    /// Existing bindings and the old pool are released before allocating the
455    /// replacement pool, minimizing peak RSS during resize.
456    ///
457    /// Returns `true` when a rebuild occurred.
458    ///
459    /// # Errors
460    ///
461    /// Returns an error when the transient pool cannot be built.
462    pub(super) fn rebuild_pool_if_needed(&mut self) -> Result<bool, RenderError> {
463        let rebuild = self
464            .pool
465            .as_ref()
466            .is_none_or(|pool| !pool.matches(self.width, self.height));
467
468        if !rebuild {
469            return Ok(false);
470        }
471
472        let plan = plan_aliases(&self.resources, &self.pass_nodes, &self.order);
473
474        // Old views keep their textures alive. Release bindings first so a
475        // resize only reserves the new pool, including a large-to-small resize.
476        self.bindings = None;
477        self.pool = None;
478        self.temporal.reset();
479
480        self.pool = Some(TransientPool::build(
481            &self.device,
482            &self.resources,
483            plan,
484            self.width,
485            self.height,
486        )?);
487
488        self.bindings = self
489            .pool
490            .as_ref()
491            .and_then(|pool| FrameBindings::new(&self.device, pool, &self.passes));
492
493        Ok(true)
494    }
495
496    /// Acquires the next presentation frame and handles recoverable surfaces.
497    ///
498    /// Lost and outdated surfaces are reconfigured and reported as `None`.
499    /// Timeouts are also reported as `None`, allowing the caller to skip the
500    /// current frame rather than failing or blocking.
501    ///
502    /// # Errors
503    ///
504    /// Returns an error for surface failures other than lost, outdated, or
505    /// timeout conditions.
506    fn acquire_surface_frame(
507        &mut self,
508    ) -> Result<Option<<D::Surface as molgfx_gpu::Surface<D>>::Frame>, RenderError> {
509        let Some(surface) = &mut self.surface else {
510            return Ok(None);
511        };
512
513        match surface.acquire() {
514            Ok(frame) => Ok(Some(frame)),
515
516            Err(SurfaceError::Lost | SurfaceError::Outdated) => {
517                surface.configure(
518                    &self.device,
519                    &molgfx_gpu::SurfaceConfig {
520                        width: self.width,
521                        height: self.height,
522                        format: self.target_format,
523                    },
524                );
525
526                Ok(None)
527            }
528
529            Err(SurfaceError::Timeout) => Ok(None),
530
531            Err(error) => Err(RenderError::Gpu(error.into())),
532        }
533    }
534}