concinnity-device 0.19.119

GPU backends (Metal, Vulkan, DirectX) behind a device facade for Concinnity
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
//! DirectX-side executor for the render graph. `DxContext::execute_graph`
//! walks the `CompiledGraph` produced by the shared
//! `concinnity_core::render::render_graph::build_frame_graph`
//! and dispatches each pass to its `encode_*` method. Mirrors the Metal +
//! Vulkan executors; every backend now drives the same builder.
//!
//! **Per-pass command lists.** Each non-composite pass records into its
//! own `ID3D12GraphicsCommandList` (drawn from the `pass_cmd_lists` pool
//! on `DxContext`). The fan-out runs on `jobs::pool()` via `rayon::scope`
//! so workers encode in parallel; each worker resets its assigned
//! allocator + cmd list, brackets the encode with start/end TIMESTAMP
//! queries, encodes the pass, and closes the cmd list. The main thread
//! then submits every closed cmd list in topological pass order via
//! `ExecuteCommandLists`. The Composite pass keeps using the outer
//! "end" cmd list that `draw_frame` owns (so the final timestamp +
//! `ResolveQueryData` ride the same submission). Mirrors
//! `metal/graph_exec.rs`.
//!
//! Both of a pass's barrier lists are consumed for every resource the registry
//! resolves: `emit_pass_prologue` translates their graph state transitions into
//! `D3D12_RESOURCE_BARRIER` transitions at the start of each pass's own command
//! list -- batched, so a pass costs one `ResourceBarrier` call -- and
//! `emit_graph_restores` returns any that the frame left off their resting state
//! at the end of the outer "end" list. Every other resource still owns its
//! transitions inline in its encoder; `audit/barrier.rs` classifies each
//! remaining site.
//!
//! The registry decides two things per resource: which D3D12 resource backs it,
//! and what state it rests in between frames. Its class -- what a `Write` means --
//! comes from the usage the graph declares, so this executor and the Vulkan one
//! cannot disagree about it. Resting cannot: `shadow_map` and `hdr_depth` are both
//! depth targets, and the first rests sampled while the second rests DEPTH_WRITE.
//!
//! Bundled passes:
//!   * `PassId::SsaoBlur` dispatches the bundled `encode_ssao` (which
//!     internally encodes the GTAO depth copy, kernel and depth-aware
//!     blur). `PassId::SsaoDepth` / `PassId::SsaoKernel` stay timing-only and
//!     the executor rejects them as graph nodes.
//!
//! `PassId::ParticlesSim` and `PassId::ParticlesDraw` are two nodes with their
//! own command lists: the sim integrates every live emitter's pool and the draw
//! reads those pools in its vertex stage, with the transition between them
//! derived from the graph rather than emitted inline.

use concinnity_core::gfx::frustum::Frustum;
use concinnity_core::gfx::render_types::{ClusterParams, TextDrawCall};
use concinnity_core::render::error::{RenderError, RenderResult};
use concinnity_core::render::pass_timing;
use concinnity_core::render::planar_reflection::PlanarFramePlan;
use concinnity_core::render::render_graph;
use concinnity_core::render::render_graph::{
    BarrierOp, CompiledGraph, CompiledPass, GraphResourceClass, PassId, final_states,
};
use concinnity_core::render::uniforms::PassCamera;
use concinnity_core::transform::mat4_inverse;
use concinnity_host::thread::jobs;
use std::sync::Mutex;
use windows::Win32::Graphics::Direct3D12::*;

use super::barrier_translate::{DxBarrier, d3d12_barrier, d3d12_restore};
use super::context::DxContext;
use super::parallel_encoder::{ParallelCtxRef, SendableCmdList, pool_index};
use super::texture::{aliasing_barrier, transition_barrier, uav_barrier};
use crate::directx::descriptor_slot::SrvSlot;
use crate::directx::line::LineUpload;

// One resolved barrier target: the D3D12 resource a graph resource backs, its
// class, and its resting state (created / cross-frame-restored). Built once per
// frame by `build_barrier_registry`.
//
// The resource is borrowed from `DxContext`, not refcount-cloned: the registry is
// a frame-path local that every worker joins before it drops, so a borrow both
// keeps the per-frame path free of `AddRef` / `Release` traffic and makes it
// impossible for a target to outlive the field it was resolved from.
struct DxBarrierTarget<'a> {
    object: DxTargetObject<'a>,
    class: GraphResourceClass,
    resting: D3D12_RESOURCE_STATES,
}

// What a graph resource resolves to: one D3D12 resource, or several under one
// label as a range into the registry's resource arena. `particle_pool` is the
// second case: the graph declares one resource for the emitter set because the
// frame builder has no emitter count, and the executor resolves it to however
// many pools are live this frame. Each gets its own barrier; the arena keeps the
// set from costing an allocation of its own.
enum DxTargetObject<'a> {
    One(&'a ID3D12Resource),
    Set { first: usize, count: usize },
}

// `ResourceId`-indexed table of barrier targets for the graph-driven resources
// (`None` for every resource the executor doesn't graph-drive). A resource is
// graph-driven iff it has a `Some` entry, so this table is the single source of
// truth. Built on the main thread by `build_barrier_registry`, where the only
// field-naming of the graph-driven resources lives; the parallel emit path stays
// field-agnostic.
struct DxBarrierRegistry<'a> {
    targets: Vec<Option<DxBarrierTarget<'a>>>,
    // Backing store for the `DxTargetObject::Set` ranges.
    resources: Vec<&'a ID3D12Resource>,
}

impl<'a> DxBarrierRegistry<'a> {
    fn target(&self, resource_index: usize) -> Option<&DxBarrierTarget<'a>> {
        self.targets.get(resource_index)?.as_ref()
    }

    // The resources one target names: a single object, or the whole set.
    fn objects<'r>(&'r self, target: &'r DxBarrierTarget<'a>) -> &'r [&'a ID3D12Resource] {
        match target.object {
            DxTargetObject::One(ref r) => std::slice::from_ref(r),
            DxTargetObject::Set { first, count } => &self.resources[first..first + count],
        }
    }
}

// SAFETY: same read-only contract as `ParallelCtxRef` / `SendableCmdList` (see
// `parallel_encoder.rs`). The registry holds borrows of D3D12 resource handles
// that workers only read, to record `ResourceBarrier` calls into their own
// command lists; every worker joins before the borrow that built the registry
// ends. D3D12 device-derived objects are thread-safe for shared read per
// Microsoft's free-threading rules.
unsafe impl Sync for DxBarrierRegistry<'_> {}

// One pooled transient a pass reclaims a shared heap region for, resolved once
// per frame on the main thread.
struct DxAliasBarrier<'a> {
    // Topological position of the pass that first-writes it, i.e. where the
    // aliasing barrier and the re-initializing discard belong.
    pass: usize,
    // Its `ResourceId`, so the prologue can recognize the pass's own transition
    // for it and open that from the discard state instead of from rest.
    resource_index: usize,
    resource: &'a ID3D12Resource,
    // Where it sits between frames: the state the discard opens from, and the
    // one it is put back in when the graph drives it no further.
    resting: D3D12_RESOURCE_STATES,
}

// Every aliasing barrier the frame owes, sorted by pass so a pass's own are one
// subslice. Flat rather than a `Vec` per pass: at most a handful of transients
// alias in a frame, and the whole table is one allocation.
struct DxAliasBarriers<'a>(Vec<DxAliasBarrier<'a>>);

// SAFETY: same read-only contract as `DxBarrierRegistry` above.
unsafe impl Sync for DxAliasBarriers<'_> {}

impl<'a> DxAliasBarriers<'a> {
    fn for_pass(&self, pass: usize) -> &[DxAliasBarrier<'a>] {
        let start = self.0.partition_point(|a| a.pass < pass);
        let end = self.0.partition_point(|a| a.pass <= pass);
        &self.0[start..end]
    }
}

// The state a placed resource must be in for `DiscardResource` to re-initialize
// it once an aliasing barrier has claimed its heap region.
const DISCARD_STATE: D3D12_RESOURCE_STATES = D3D12_RESOURCE_STATE_RENDER_TARGET;

// How many barriers one `ResourceBarrier` call carries. A pass declares far
// fewer, so this is a cap rather than a budget: a batch that fills flushes and
// keeps filling, which costs an extra call and stays correct.
const BARRIER_BATCH: usize = 16;

// Barriers accumulated for one `ResourceBarrier` call. D3D12 charges per call
// (each is a driver round trip that can split the command list), so the frame
// path fills one of these per pass instead of issuing a call per resource.
//
// Fixed capacity, so a pass costs no allocation. The barriers hold borrowed
// resource pointers (`com::borrowed`), which is why the batch is a short-lived
// local: it must not outlive the registry the pointers came from.
struct BarrierBatch<'a> {
    cmd: &'a ID3D12GraphicsCommandList,
    entries: [D3D12_RESOURCE_BARRIER; BARRIER_BATCH],
    len: usize,
}

impl<'a> BarrierBatch<'a> {
    fn new(cmd: &'a ID3D12GraphicsCommandList) -> Self {
        Self {
            cmd,
            entries: std::array::from_fn(|_| D3D12_RESOURCE_BARRIER::default()),
            len: 0,
        }
    }

    fn push(&mut self, barrier: D3D12_RESOURCE_BARRIER) {
        if self.len == BARRIER_BATCH {
            self.flush();
        }
        self.entries[self.len] = barrier;
        self.len += 1;
    }

    fn flush(&mut self) {
        if self.len == 0 {
            return;
        }
        // SAFETY: the command list is in the recording state, and every resource, descriptor and
        // slice these commands name is live for the call.
        unsafe {
            self.cmd.ResourceBarrier(&self.entries[..self.len]);
        }
        self.len = 0;
    }
}

// Translate one of a pass's `barriers_before` through the registry and stage it.
// A resource with no registry entry is skipped and keeps its inline barriers, as
// is a transition D3D12 needs nothing for. `opened_at` overrides the state a
// first use (`Undefined` source) starts from, which is how a resource this pass
// has just discarded is taken on from `DISCARD_STATE` rather than from rest.
fn stage_graph_barrier(
    batch: &mut BarrierBatch,
    registry: &DxBarrierRegistry,
    op: &BarrierOp,
    opened_at: Option<D3D12_RESOURCE_STATES>,
) {
    let Some(target) = registry.target(op.resource_index()) else {
        return;
    };
    let Some(barrier) = d3d12_barrier(
        target.class,
        opened_at.unwrap_or(target.resting),
        op.source_state(),
        op.to_state(),
        op.read_stages(),
    ) else {
        return;
    };
    for resource in registry.objects(target) {
        batch.push(match barrier {
            DxBarrier::Transition(before, after) => transition_barrier(resource, before, after),
            DxBarrier::Uav => uav_barrier(resource),
        });
    }
}

// One slot of a per-frame buffer ring, as a barrier target: the slot for the
// frame being recorded. A free function rather than a closure so the borrow it
// returns is tied to the ring, not to the resolver's own frame.
fn frame_slot(
    slots: &[ID3D12Resource],
    frame_idx: usize,
    resting: D3D12_RESOURCE_STATES,
) -> Option<(&ID3D12Resource, D3D12_RESOURCE_STATES)> {
    Some((slots.get(frame_idx)?, resting))
}

// Whether the pass's own graph transition takes a reclaimed resource on from the
// discard, so the prologue does not have to put it back at rest itself.
fn graph_takes_over(
    registry: &DxBarrierRegistry,
    pass: &CompiledPass,
    resource_index: usize,
) -> bool {
    registry.target(resource_index).is_some()
        && pass
            .barriers_before
            .iter()
            .any(|op| op.resource_index() == resource_index)
}

// Everything a pass owes its command list before its body: the aliasing barriers
// for any pooled transient it first-writes and the discard that re-initializes
// it, plus its graph-derived transitions. Vulkan emits the same two halves in the
// same order, for the same reason.
//
// An aliasing barrier leaves the reclaimed memory's contents undefined, and D3D12
// rejects a placed render target's use until a Clear/Discard/Copy initializes it,
// so each reclaimed resource is opened for a `DiscardResource` first. It is left
// in `DISCARD_STATE` afterwards rather than put back at rest: the pass's own
// first-write transition is what takes it on from there, which for a color
// target is the very state it wants and collapses to nothing.
//
// One function because the two recording paths are otherwise asymmetric --
// Composite records into the outer "end" list on the main thread while every
// other pass fans out to a worker -- and that asymmetry is exactly how Composite
// came to record no graph barriers at all, silently dropping any a driven
// resource declared there.
fn emit_pass_prologue(
    cmd: &ID3D12GraphicsCommandList,
    registry: &DxBarrierRegistry,
    alias: &DxAliasBarriers,
    idx: usize,
    pass: &CompiledPass,
) {
    let aliased = alias.for_pass(idx);
    let mut batch = BarrierBatch::new(cmd);
    for a in aliased {
        batch.push(aliasing_barrier(a.resource));
        if a.resting != DISCARD_STATE {
            batch.push(transition_barrier(a.resource, a.resting, DISCARD_STATE));
        }
    }
    for op in &pass.barriers_before {
        if aliased
            .iter()
            .any(|a| a.resource_index == op.resource_index())
        {
            continue;
        }
        stage_graph_barrier(&mut batch, registry, op, None);
    }
    batch.flush();

    if aliased.is_empty() {
        return;
    }
    for a in aliased {
        // SAFETY: the command list is in the recording state, and every resource, descriptor and
        // slice these commands name is live for the call.
        unsafe {
            cmd.DiscardResource(a.resource, None);
        }
    }
    for op in &pass.barriers_before {
        if aliased
            .iter()
            .any(|a| a.resource_index == op.resource_index())
        {
            // The aliasing barrier lands at the resource's first use, so the
            // transition here opens from no prior state, and the discard is what
            // that resolves to instead of the resting state.
            debug_assert_eq!(
                op.source_state(),
                render_graph::ResourceState::Undefined,
                "a reclaimed transient is transitioned from a state the discard replaced"
            );
            stage_graph_barrier(&mut batch, registry, op, Some(DISCARD_STATE));
        }
    }
    // A reclaimed resource the graph drives no further would otherwise sit in the
    // discard state, which is not where the next frame expects to find it.
    for a in aliased {
        if a.resting != DISCARD_STATE && !graph_takes_over(registry, pass, a.resource_index) {
            batch.push(transition_barrier(a.resource, DISCARD_STATE, a.resting));
        }
    }
    batch.flush();
}

// The one thing a pass owes its command list after its body: the transitions the
// graph records on the producing side, for a read run whose readers are split
// across the schedule's two queues. Recorded into the same list as the pass it
// follows, so the flattened serial stream is command-for-command what a
// transition on the run's first reader produced. The transition into a
// compute-legal state belongs here too once a compute queue exists, since it has
// to happen on the direct queue before the signal; nothing creates one yet.
fn emit_pass_epilogue(
    cmd: &ID3D12GraphicsCommandList,
    registry: &DxBarrierRegistry,
    pass: &CompiledPass,
) {
    if pass.barriers_after.is_empty() {
        return;
    }
    let mut batch = BarrierBatch::new(cmd);
    for op in &pass.barriers_after {
        stage_graph_barrier(&mut batch, registry, op, None);
    }
    batch.flush();
}

// Return every driven resource the frame left off its resting state, so the next
// frame's first transition for it names the state the resource is really in (the
// debug layer rejects a mismatch). Recorded last into the outer "end" command
// list, which executes after every pass list. A frame that ends every resource at
// rest emits nothing.
fn emit_graph_restores(
    cmd: &ID3D12GraphicsCommandList,
    registry: &DxBarrierRegistry,
    graph: &CompiledGraph,
) {
    let mut batch = BarrierBatch::new(cmd);
    for (idx, (state, stages)) in final_states(graph).into_iter().enumerate() {
        let Some(target) = registry.target(idx) else {
            continue;
        };
        let Some((before, after)) = d3d12_restore(target.class, target.resting, state, stages)
        else {
            continue;
        };
        for resource in registry.objects(target) {
            batch.push(transition_barrier(resource, before, after));
        }
    }
    batch.flush();
}

// Check the graph's barrier coverage and cross-frame state contract for every
// resource this executor drives, on the frame's real compiled graph. Two
// invariants, both cheap enough to run per frame under `debug_assertions`:
//
//   * every declared read / write of a driven resource is preceded by a
//     transition putting it in the matching state, in the consuming stage;
//   * a driven resource is back in the resting state its registry entry declares
//     once the frame's restores have run, so the next frame's first transition
//     (whose `Undefined` source resolves to that resting state) names the state
//     the resource is really in. The debug layer rejects a mismatch. This is the
//     check the restore pass exists to satisfy; it fires if a resource ends in a
//     state no restore can express.
//
// This is where a registry entry that claims a resource the graph does not fully
// cover shows up; the headless sweep in `render_graph::validate` covers the
// deriver itself. Mirrors the Vulkan executor's `debug_assert_graph_drives`.
#[cfg(debug_assertions)]
fn debug_assert_graph_drives(graph: &CompiledGraph, registry: &DxBarrierRegistry) {
    use super::barrier_translate::d3d12_state;
    use concinnity_core::render::render_graph::{ResourceState, barrier_coverage_gaps_for_driven};

    let driven: Vec<bool> = registry.targets.iter().map(|t| t.is_some()).collect();
    let gaps = barrier_coverage_gaps_for_driven(graph, &driven);
    assert!(
        gaps.is_empty(),
        "render graph (directx): uncovered accesses on graph-driven resources: {}",
        gaps.iter()
            .map(|g| g.to_string())
            .collect::<Vec<_>>()
            .join(", ")
    );

    // The producer-side half of the barrier lists. A pass's `barriers_after` is
    // recorded once its own encode has run, which is only sound for a
    // transition out of the state that encode left the resource in, so every op
    // there has to be a `Write -> Read` on a resource this pass writes. The
    // headless sweep proves that of the graphs it builds; this proves it of the
    // graphs a real session builds.
    for pass in &graph.passes {
        for op in &pass.barriers_after {
            assert_eq!(
                (op.source_state(), op.to_state()),
                (ResourceState::Write, ResourceState::Read),
                "render graph (directx): {:?} records a {:?} -> {:?} on {} after its own work",
                pass.id,
                op.source_state(),
                op.to_state(),
                graph.resources[op.resource_index()].label,
            );
            assert!(
                pass.writes
                    .iter()
                    .any(|w| w.resource_index() == op.resource_index()),
                "render graph (directx): {:?} records a transition on {}, which it does not write",
                pass.id,
                graph.resources[op.resource_index()].label,
            );
        }
    }

    for (idx, (state, stages)) in final_states(graph).into_iter().enumerate() {
        let Some(target) = registry.target(idx) else {
            continue;
        };
        if state == ResourceState::Undefined {
            continue;
        }
        let restored = match d3d12_restore(target.class, target.resting, state, stages) {
            Some((_, after)) => after,
            None => d3d12_state(target.class, state, stages),
        };
        assert_eq!(
            restored.0, target.resting.0,
            "render graph (directx): {} rests in {:?} but the frame leaves it in {:?}",
            graph.resources[idx].label, target.resting, restored,
        );
    }
}

// The back-buffer render target the composite pass writes: the resource (for its
// RENDER_TARGET <-> PRESENT transitions) plus its RTV handle.
#[derive(Clone, Copy)]
pub(in crate::directx) struct CompositeRenderTarget<'a> {
    pub back_buffer: &'a ID3D12Resource,
    pub back_buffer_rtv: D3D12_CPU_DESCRIPTOR_HANDLE,
}

// Output (drawable) resolution the composite pass writes at. Under temporal
// upscaling this differs from the scene render dimensions.
#[derive(Clone, Copy)]
pub(in crate::directx) struct CompositeResolution {
    pub width: u32,
    pub height: u32,
}

// Render-target dimensions of the main pass, in pixels.
#[derive(Clone, Copy)]
pub(in crate::directx) struct MainPassExtent {
    pub width: u32,
    pub height: u32,
}

// GPU virtual addresses of this frame's view / light / shadow constant buffers.
#[derive(Clone, Copy)]
pub(in crate::directx) struct FrameGpuBuffers {
    pub view_gva: u64,
    pub light_gva: u64,
    // GVA of the static per-scene GpuLight storage buffer (root SRV).
    pub local_lights_gva: u64,
    pub shadow_ubo_gva: u64,
}

// Per-frame params the executor threads into each pass's `encode_*`
// method. Mirrors the Vulkan executor's `GraphFrameParams`.
pub(in crate::directx) struct GraphFrameParams<'a> {
    pub cmd: &'a ID3D12GraphicsCommandList,
    pub frame_idx: usize,
    pub back_buffer: &'a ID3D12Resource,
    pub back_buffer_rtv: D3D12_CPU_DESCRIPTOR_HANDLE,
    pub text_calls: &'a [TextDrawCall],
    // This frame's line ribbons, uploaded on the main thread before the fan-out
    // so the Lines pass never touches the upload ring or the allocator. `None`
    // when there is nothing to draw, including whenever the graph carries no
    // Lines node.
    pub lines: Option<LineUpload>,
    // An opaque menu backdrop hides the scene: the Main pass clears its target
    // and skips every draw (the masked graph drops all other world passes), so
    // nothing of the world renders behind the menu.
    pub world_hidden: bool,
    // Scene SRV the composite shader samples: TAA history when TAA is
    // on, SSR output when SSR is on and TAA is off, raw HDR scene SRV
    // otherwise. Computed in `record_frame` once before the dispatch;
    // stable across the whole graph because `taa.frame` only ticks
    // after Composite, so reading `taa.output_index()` upfront points
    // at the same TAA history slot the TaaResolve encoder will write
    // into and the Composite encoder samples.
    pub scene_srv: SrvSlot,
    // Off-screen scene render resolution. Every scene pass (Shadow, Main,
    // SSAO, SSR, Velocity, Fog, Raymarch, Decals, Particles) rasterizes at
    // this size, and the sub-pixel jitter is converted to NDC against it.
    // Equals `output_*` when temporal upscaling is off.
    pub width: u32,
    pub height: u32,
    // Drawable (swapchain) resolution. Only the Composite + text pass uses
    // these; it renders the fullscreen tonemap triangle into the
    // output-sized back buffer and sets the text overlay's window-dim
    // uniform from them. Bloom samples its own output-sized mip extents, so
    // it needs no dim param. Equals `width`/`height` when upscaling is off.
    pub output_width: u32,
    pub output_height: u32,
    // Camera world-space position. Shadow uses it for CSM cascade
    // distance bookkeeping inside `encode_shadow_pass`; Main uses it
    // for per-cluster distance culling and the SSAO bundle's pre-pass.
    pub cam_pos: [f32; 3],
    // GPU virtual address of this frame's `ShadowUniforms` constant
    // buffer (the cached cascade VPs + light direction). Consumed by
    // Shadow and Main.
    pub shadow_ubo_gva: u64,
    // GPU virtual address of this frame's `ViewUniforms` constant
    // buffer. Consumed by Main.
    pub view_gva: u64,
    // GPU virtual address of the shared `LightUniforms` constant
    // buffer. Consumed by Main (and any future pass that lights the
    // scene).
    pub light_gva: u64,
    // GPU virtual address of the static per-scene `GpuLight` storage
    // buffer. Consumed by Main's bindless pass as a root SRV.
    pub local_lights_gva: u64,
    // Jittered camera view-projection matrix (sub-pixel Halton jitter
    // applied when TAA is on). Consumed by Main and Velocity (the
    // jittered VP path).
    pub vp_mat: [[f32; 4]; 4],
    // Un-jittered camera view-projection matrix. Velocity uses it
    // alongside `vp_mat` (jittered) and the prior frame's `prev_vp`
    // stored on the context, so the stored motion vector is free of
    // sub-pixel jitter.
    pub cur_vp: [[f32; 4]; 4],
    // Camera frustum derived from `vp_mat`. Consumed by Main's
    // per-cluster culling and the bundled SSAO pre-pass.
    pub frustum: &'a Frustum,
    // Vertical FOV in radians. Consumed by Main's SSAO pre-pass
    // (depth-reconstruction geometry) and SsrResolve's ray-march
    // projection.
    pub fov_y_radians: f32,
    // Camera aspect ratio (width / height). Consumed by SsrResolve
    // for the ray-march projection.
    pub aspect: f32,
    // Seconds since the engine started. Consumed by ParticlesDraw's
    // bundled compute sim (delta-time computed against the last
    // per-emitter elapsed snapshot stored on `DxContext`).
    pub elapsed: f32,
    // Camera near-plane in view units. Consumed by `FogFroxel` to map the
    // front edge of the froxel volume onto view-space depth, and by
    // `Upscale` (FSR3 dispatch's `cameraNear`).
    pub near: f32,
    // Whether this frame's model-history snapshot fills every ring slot rather
    // than only its own: the rebuild's prime request, taken from the tracker
    // before the fan-out. Consumed by GBufferPrepass.
    pub prime_model_history: bool,
    // Which planar mirrors this frame renders, and the screen rectangle each
    // covers; computed once from `vp_mat` so the mirror pass and the transparent
    // pass that samples it agree.
    pub planar: PlanarFramePlan,
    // The main camera's cluster params, which each mirror render re-aims
    // through its reflected view.
    pub cluster_params: ClusterParams,
    // How the particle and transparent passes treat the reactive mask, and
    // whether TAA, the upscaler or the reactive view reads it.
    pub reactive: concinnity_core::render::reactive_mask::ReactiveMaskPlan,
}

impl DxContext {
    // Walk a compiled render graph and dispatch each non-composite pass
    // to its own freshly-reset per-pass `ID3D12GraphicsCommandList`,
    // fanning the encode work across rayon workers. Composite stays on
    // the outer "end" cmd list (`params.cmd`) the caller provides; the
    // final timestamp + `ResolveQueryData` ride the same submission.
    // Returns the closed per-pass cmd lists in topological pass order
    // (excluding composite); the caller submits them via
    // `ExecuteCommandLists` between the "start" outer cmd list (which
    // holds the timestamp pre-init) and the "end" outer cmd list (which
    // holds composite + post).
    //
    // `&self` mirrors every DirectX `encode_*` method; per-frame mutable
    // state lives behind `RwLock` / `Cell` / `AtomicU32` so the encoders
    // stay sound under the parallel fan-out (see
    // [`super::parallel_encoder`] for the Send/Sync contract).
    pub(in crate::directx) fn execute_graph(
        &self,
        graph: &CompiledGraph,
        params: &GraphFrameParams<'_>,
    ) -> RenderResult<Vec<ID3D12GraphicsCommandList>> {
        // Particle per-frame state (dt / frame index / per-emitter spawn runs
        // and their params-ring slots) is advanced here, once, before any pass
        // encodes, so the sim and draw halves record on separate workers against
        // one consistent frame. `None` when the pass is inert. Mirrors Vulkan's
        // and Metal's `prepare_particle_pass` hoist.
        let particle_frame = self.prepare_particle_pass(params.frame_idx, params.elapsed);
        let particle_ref = particle_frame.as_ref();

        // Find Composite's slot (if any) so we can skip it in the
        // worker fan-out and run it inline on the main thread instead.
        let composite_idx = graph.passes.iter().position(|p| p.id == PassId::Composite);

        // Slot per graph pass: each worker stashes its closed cmd list
        // here on success, indexed by topological position. Main thread
        // collects them into the return Vec in order after the join.
        let worker_slots: Mutex<Vec<Option<SendableCmdList>>> =
            Mutex::new((0..graph.passes.len()).map(|_| None).collect());
        let first_error: Mutex<Option<RenderError>> = Mutex::new(None);

        let ctx_ref = ParallelCtxRef::new(self);
        // Resolve every graph-driven resource's barrier target once, on the main
        // thread, then share the table read-only into the parallel pass workers.
        let registry = self.build_barrier_registry(graph, params.frame_idx);
        #[cfg(debug_assertions)]
        debug_assert_graph_drives(graph, &registry);
        #[cfg(debug_assertions)]
        render_graph::assert_slot_aliasing_sound(
            graph,
            self.targets.transient_pool.slot_labels(),
            "directx",
        );
        // The graph now carries a two-queue schedule (a `PassQueue` per pass plus
        // the cross-queue signal / wait pairs). This executor still records one
        // serial stream in `graph.passes` order, which is the schedule's queues
        // interleaved back into their compiled order: the command order is
        // therefore byte-for-byte what it was before the schedule existed, and
        // no compute queue is created. What has to hold for that flattening to
        // be legal is that the compiled order is a topological order for both
        // queues at once and that every wait names an already-recorded producer,
        // which is what this asserts.
        #[cfg(debug_assertions)]
        render_graph::assert_serial_order_honors_schedule(graph, "directx");
        let registry_ref = &registry;
        // Likewise resolve the per-pass aliasing barriers (which pooled transients
        // reclaim a shared heap region) once, shared read-only into the workers.
        let alias_barriers = self.build_alias_barriers(graph, &registry);
        let alias_barriers_ref = &alias_barriers;
        let frame_idx = params.frame_idx;

        jobs::pool().install(|| {
            rayon::scope(|scope| {
                for (idx, pass) in graph.passes.iter().enumerate() {
                    if Some(idx) == composite_idx {
                        continue;
                    }
                    let pass_id = pass.id;
                    let first_error_ref = &first_error;
                    let worker_slots_ref = &worker_slots;
                    scope.spawn(move |_| {
                        let ctx = ctx_ref.as_ctx();
                        let pool_idx = pool_index(frame_idx, pass_id);
                        let alloc = &ctx.commands.pass_allocators[pool_idx];
                        let cmd = &ctx.commands.pass_cmd_lists[pool_idx];

                        // Reset this pass's allocator + cmd list so we
                        // can record fresh into it. The previous frame's
                        // submission for this same (frame, pass) slot
                        // has already retired by the time we get here
                        // (the FRAMES-deep fence wait at the top of
                        // `draw_frame` gates the entire slot).
                        // SAFETY: the fence for this frame slot was already waited on, so no
                        // submission still references what is being reset.
                        if let Err(e) = unsafe { alloc.Reset() } {
                            let mut lock = first_error_ref.lock().unwrap();
                            if lock.is_none() {
                                *lock = Some(super::error::map_hresult(
                                    e.code(),
                                    &format!("per-pass allocator reset ({})", pass_id.name()),
                                ));
                            }
                            return;
                        }
                        // SAFETY: the fence for this frame slot was already waited on, so no
                        // submission still references what is being reset.
                        if let Err(e) = unsafe { cmd.Reset(alloc, None) } {
                            let mut lock = first_error_ref.lock().unwrap();
                            if lock.is_none() {
                                *lock = Some(super::error::map_hresult(
                                    e.code(),
                                    &format!("per-pass cmd list reset ({})", pass_id.name()),
                                ));
                            }
                            return;
                        }

                        // Per-pass GPU timing: bracket the encoder with
                        // start + end TIMESTAMP `EndQuery` calls into
                        // pre-allocated heap slots. The frame's whole
                        // block is resolved by the "end" outer cmd list
                        // at the end of the frame and read back at the
                        // top of the next frame. See
                        // [`pass_timing`] for the slot layout.
                        if let Some(heap) = ctx.timestamps.query_heap.as_ref() {
                            let (start_slot, _) = pass_timing::pass_pair(frame_idx, pass_id);
                            // SAFETY: the command list is in the recording state, and every
                            // resource, descriptor and slice these commands name is live for the
                            // call.
                            unsafe {
                                cmd.EndQuery(heap, D3D12_QUERY_TYPE_TIMESTAMP, start_slot);
                            }
                        }

                        emit_pass_prologue(cmd, registry_ref, alias_barriers_ref, idx, pass);

                        let encode_result =
                            ctx.encode_pass_into(pass_id, cmd, params, particle_ref);

                        if encode_result.is_ok() {
                            emit_pass_epilogue(cmd, registry_ref, pass);
                        }

                        if let Some(heap) = ctx.timestamps.query_heap.as_ref() {
                            let (_, end_slot) = pass_timing::pass_pair(frame_idx, pass_id);
                            // SAFETY: the command list is in the recording state, and every
                            // resource, descriptor and slice these commands name is live for the
                            // call.
                            unsafe {
                                cmd.EndQuery(heap, D3D12_QUERY_TYPE_TIMESTAMP, end_slot);
                            }
                        }

                        // SAFETY: the command list is live and in the recording state, which is
                        // what `Close` requires.
                        if let Err(e) = unsafe { cmd.Close() } {
                            let mut lock = first_error_ref.lock().unwrap();
                            if lock.is_none() {
                                *lock = Some(super::error::map_hresult(
                                    e.code(),
                                    &format!("per-pass cmd list close ({})", pass_id.name()),
                                ));
                            }
                            return;
                        }

                        match encode_result {
                            Ok(()) => {
                                let mut lock = worker_slots_ref.lock().unwrap();
                                lock[idx] = Some(SendableCmdList(cmd.clone()));
                            }
                            Err(e) => {
                                let mut lock = first_error_ref.lock().unwrap();
                                if lock.is_none() {
                                    *lock = Some(e);
                                }
                            }
                        }
                    });
                }
            });
        });

        if let Some(err) = first_error.into_inner().unwrap_or(None) {
            return Err(err);
        }

        // Composite stays on the outer "end" cmd list (`params.cmd`) the
        // caller supplied. The final timestamp `EndQuery` +
        // `ResolveQueryData` are appended onto the same cmd list by
        // `draw_frame` after this returns, so composite + post-resolve
        // ride one submission.
        if let Some(idx) = composite_idx {
            if let Some(heap) = self.timestamps.query_heap.as_ref() {
                let (start_slot, _) = pass_timing::pass_pair(frame_idx, PassId::Composite);
                // SAFETY: the command list is in the recording state, and every resource,
                // descriptor and slice these commands name is live for the call.
                unsafe {
                    params
                        .cmd
                        .EndQuery(heap, D3D12_QUERY_TYPE_TIMESTAMP, start_slot);
                }
            }
            emit_pass_prologue(
                params.cmd,
                &registry,
                &alias_barriers,
                idx,
                &graph.passes[idx],
            );
            self.encode_composite_and_text(
                params.cmd,
                params.frame_idx,
                CompositeRenderTarget {
                    back_buffer: params.back_buffer,
                    back_buffer_rtv: params.back_buffer_rtv,
                },
                params.text_calls,
                crate::directx::draw::composite::CompositeSources {
                    scene_srv: params.scene_srv,
                    reactive_valid: params.reactive.readable,
                },
                // Composite runs at drawable resolution; it samples the
                // (output-sized) upscaler result / scene SRV through a
                // fullscreen triangle and writes the output-sized back
                // buffer. Under upscaling this differs from the scene
                // render dims in `params.width`/`height`.
                CompositeResolution {
                    width: params.output_width,
                    height: params.output_height,
                },
            )?;
            emit_pass_epilogue(params.cmd, &registry, &graph.passes[idx]);
            if let Some(heap) = self.timestamps.query_heap.as_ref() {
                let (_, end_slot) = pass_timing::pass_pair(frame_idx, PassId::Composite);
                // SAFETY: the command list is in the recording state, and every resource,
                // descriptor and slice these commands name is live for the call.
                unsafe {
                    params
                        .cmd
                        .EndQuery(heap, D3D12_QUERY_TYPE_TIMESTAMP, end_slot);
                }
            }
        }

        // Return every driven resource the frame left off its resting state.
        // Recorded last into the outer "end" list, which executes after every
        // pass list.
        emit_graph_restores(params.cmd, &registry, graph);

        // Collect every worker-encoded cmd list in topological pass
        // order. The empty slots (composite, plus any skipped no-op
        // pass that returned without stashing) drop out; workers only
        // stash on success.
        let slots = worker_slots.into_inner().map_err(|_| {
            RenderError::Other("graph executor (directx): worker slot mutex poisoned".into())
        })?;
        let mut ordered = Vec::with_capacity(graph.passes.len());
        for cb in slots.into_iter().flatten() {
            ordered.push(cb.0);
        }
        Ok(ordered)
    }

    // Resolve every graph-driven resource to its barrier target, indexed by
    // `ResourceId` (its position in `graph.resources`), so the parallel emit path
    // can look a target up by `BarrierOp::resource_index()`. This is the single
    // place that names the graph-driven resources' backing `DxContext` fields;
    // field-grouping re-cuts here, not in the executor. A resource the owning
    // feature disabled gets `None`, and the graph carries no barrier for it either.
    fn build_barrier_registry(
        &self,
        graph: &CompiledGraph,
        frame_idx: usize,
    ) -> DxBarrierRegistry<'_> {
        let mut registry = DxBarrierRegistry {
            targets: Vec::with_capacity(graph.resources.len()),
            resources: Vec::new(),
        };
        // A plain loop rather than a `collect`: a label resolving to a set of
        // resources appends them to the arena while the table is being filled.
        for res in &graph.resources {
            let target = res.class().and_then(|class| {
                let (object, resting) =
                    self.barrier_objects_for_label(res.label, frame_idx, &mut registry.resources)?;
                Some(DxBarrierTarget {
                    object,
                    class,
                    resting,
                })
            });
            registry.targets.push(target);
        }
        registry
    }

    // Resolve, per pass, the pooled transients that reclaim a shared heap region
    // when this pass first-writes them. A resource is aliased iff the pool gives
    // it a slot predecessor; its aliasing barrier lands before the pass at its
    // `lifetime.first`. Empty for every resource the pool does not alias, so the
    // table is empty whenever no slot is shared this frame (e.g. bloom off leaves
    // `ao_output` aliased but `bloom_top` absent from the graph; ssao off leaves
    // `bloom_top` un-aliased). Mirrors the Vulkan executor's `build_alias_barriers`.
    //
    // The resting state comes from the registry when the graph drives the
    // resource, so the discard opens it from where it really sits rather than
    // from an assumed sampled state; a pooled member the graph does not drive
    // falls back to sampled, which is where the pool creates a color target.
    fn build_alias_barriers<'a>(
        &'a self,
        graph: &CompiledGraph,
        registry: &DxBarrierRegistry<'a>,
    ) -> DxAliasBarriers<'a> {
        let mut table = Vec::new();
        for (idx, res) in graph.resources.iter().enumerate() {
            if self
                .targets
                .transient_pool
                .alias_predecessor(res.label)
                .is_none()
            {
                continue;
            }
            if let Some(r) = self.targets.transient_pool.resource_for(res.label)
                && res.lifetime.first < graph.passes.len()
            {
                table.push(DxAliasBarrier {
                    pass: res.lifetime.first,
                    resource_index: idx,
                    resource: r,
                    resting: match registry.target(idx) {
                        Some(target) => target.resting,
                        None => D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE,
                    },
                });
            }
        }
        table.sort_by_key(|a| a.pass);
        DxAliasBarriers(table)
    }

    // Map one graph resource label to the D3D12 resource backing it and its
    // resting state (the state it was created in and returns to at the end of
    // every frame), so a first-use `Undefined` transition names the state the
    // resource is really in. The barrier class is NOT decided here: it follows
    // the usage the graph declares (`CompiledResource::class`), so this backend
    // and Vulkan cannot disagree about what a resource is. Resting state stays
    // per-resource because it is not derivable from usage: `shadow_map` and
    // `hdr_depth` are both depth targets, and one rests sampled while the other
    // rests as a depth attachment. `None` means the owning feature is inactive,
    // and the graph carries no node for it either.
    //
    // The resource is borrowed rather than cloned; see `DxBarrierTarget`.
    fn barrier_objects_for_label<'a>(
        &'a self,
        label: &str,
        frame_idx: usize,
        arena: &mut Vec<&'a ID3D12Resource>,
    ) -> Option<(DxTargetObject<'a>, D3D12_RESOURCE_STATES)> {
        const SAMPLED: D3D12_RESOURCE_STATES = D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE;
        // Every live emitter's particle pool: `ParticlesSim` integrates them and
        // `ParticlesDraw` reads them in its vertex stage. One graph resource
        // covers the whole set, since the frame builder knows only whether
        // particles run this frame. They rest in UNORDERED_ACCESS -- created
        // there, and where the frame's restore returns them -- because that is
        // the binding the simulation writes them through; the draw's SRV read is
        // the only state they leave it for.
        if label == "particle_pool" {
            let first = arena.len();
            arena.extend(
                self.particle
                    .emitter_state
                    .iter()
                    .flatten()
                    .map(|gpu| &gpu.pool),
            );
            let count = arena.len() - first;
            return (count > 0).then_some((
                DxTargetObject::Set { first, count },
                D3D12_RESOURCE_STATE_UNORDERED_ACCESS,
            ));
        }
        let one = |r: Option<(&'a ID3D12Resource, D3D12_RESOURCE_STATES)>| {
            r.map(|(resource, resting)| (DxTargetObject::One(resource), resting))
        };
        one(match label {
            // Indirect commands the cull kernel writes (UAV) and the main pass
            // consumes through `ExecuteIndirect`.
            "draw_args" => frame_slot(
                &self.cull.indirect_cmd_buffers,
                frame_idx,
                D3D12_RESOURCE_STATE_INDIRECT_ARGUMENT,
            ),
            // Phase-2 counterpart, written by `Cull2` and consumed by `Main2`.
            "draw_args2" => frame_slot(
                &self.cull.indirect_cmd_buffers_2,
                frame_idx,
                D3D12_RESOURCE_STATE_INDIRECT_ARGUMENT,
            ),
            // Phase 1 writes it and phase 2 reads it through the same root UAV,
            // so it never leaves `UNORDERED_ACCESS`.
            "cull_status" => frame_slot(
                &self.cull.cull_status_buffers,
                frame_idx,
                D3D12_RESOURCE_STATE_UNORDERED_ACCESS,
            ),
            // Per-cluster light lists and probe masks: the dispatch flips them to UAV and
            // back, so they rest sampled. One buffer, not per-frame.
            "cluster_lists" => Some((&self.light_cull.cluster_buffer, SAMPLED)),
            "ao_output" => self
                .targets
                .transient_pool
                .resource_for("ao_output")
                .map(|r| (r, SAMPLED)),
            // The bloom chain's half-resolution top octave, and the only mip the
            // graph models: the Bloom node writes it, Composite samples it, and
            // the finer octaves in between never leave the node. Pooled, so it
            // resolves through the transient pool exactly like `ao_output`.
            "bloom_top" => self
                .targets
                .transient_pool
                .resource_for("bloom_top")
                .map(|r| (r, SAMPLED)),
            // The cascade array rests sampled: the Shadow producer barrier is the
            // real cross-frame reset for this frame's shadow loop and the Main
            // consumer returns it to sampled. Created sampled, so frame 0's
            // producer starts from the resource's real state.
            "shadow_map" => self
                .shadow
                .resource
                .as_ref()
                .filter(|_| !self.shadow.dsvs.is_empty())
                .map(|s| (&s.resource, SAMPLED)),
            // The spot array rests sampled exactly like the cascades. Only
            // imported when the world has shadowed spots, so the `dsvs` filter
            // matches the graph gate.
            "spot_shadow_map" => self
                .spot_shadow
                .resource
                .as_ref()
                .filter(|_| !self.spot_shadow.dsvs.is_empty())
                .map(|s| (&s.resource, SAMPLED)),
            // The froxel volume rests sampled: the FogFroxel producer opens it for
            // the compute write and the Fog consumer closes it for the sample.
            "fog_froxel_volume" => self
                .fog
                .resources
                .as_ref()
                .map(|f| (&f.volume_resource, SAMPLED)),
            // Main depth rests as the depth attachment: one resource shared by
            // every frame in flight, created in DEPTH_WRITE, and the frame's
            // restore returns it there for the next main pass. This is the case
            // that keeps resting per-resource rather than per-class -- shadow_map
            // is the same class and rests sampled.
            "hdr_depth" => Some((
                &self.targets.depth.resource,
                D3D12_RESOURCE_STATE_DEPTH_WRITE,
            )),
            // The multisample color attachment, which exists only when the
            // world is multisampled -- and so does the graph resource. It rests
            // in RENDER_TARGET and no pass ever samples it, so every derived
            // transition collapses to a no-op and the entry drives nothing. It
            // is registered anyway because that is what makes the MSAA resolve's
            // RENDER_TARGET <-> RESOLVE_SOURCE pair intra-pass rather than a
            // frame-path transition the graph left behind.
            "hdr_color" => self
                .targets
                .hdr
                .resolve
                .is_some()
                .then_some((&self.targets.hdr.color, D3D12_RESOURCE_STATE_RENDER_TARGET)),
            // The single-sample scene spine every decoration blends into. Which
            // object backs it, and where it rests, both follow MSAA: with MSAA
            // on it is the resolve target and rests sampled; with MSAA off there
            // is no resolve step and `hdr.color` *is* the spine, left in
            // RENDER_TARGET for the next frame's main pass. Its class is the
            // same either way.
            "hdr_resolve" => Some(match &self.targets.hdr.resolve {
                Some(resolve) => (resolve, SAMPLED),
                None => (&self.targets.hdr.color, D3D12_RESOURCE_STATE_RENDER_TARGET),
            }),
            // The scene-with-reflections the post stack consumes. Declared by
            // the graph exactly when a reflection resolve runs, which is the
            // same predicate that builds this target, so the entry resolves
            // whenever the resource exists.
            "scene_pre_taa" => self
                .reflection_composite
                .as_ref()
                .map(|rc| (rc.output().resource(), SAMPLED)),
            // The post-TAA scene. Two mutually exclusive writers back it, and
            // only one is driven: the TAA resolve writes this frame's ping-pong
            // history slot, which rests sampled like any other color target,
            // while the temporal upscaler writes a compute output whose
            // between-frames state depends on whether a previous frame
            // dispatched (`output_is_psr`). The graph's resting model has no way
            // to say that, so under upscaling the entry stays `None` and
            // `post/upscale/fsr.rs` keeps owning its transitions.
            "scene_color" => self
                .taa
                .as_ref()
                .filter(|_| self.upscale.backend.is_none())
                .map(|taa| (&taa.output().resource, SAMPLED)),
            // The unified G-buffer pre-pass's color targets, one entry each.
            // One draw writes all three, but their consumers differ -- the
            // reflection resolve reads normal+depth and roughness, the temporal
            // passes read velocity -- so they are separate graph resources with
            // separate lifetimes. All three rest sampled.
            "gbuffer_normal_depth" => self.gbuffer.as_ref().map(|gb| (&gb.normal_depth, SAMPLED)),
            "gbuffer_roughness" => self.gbuffer.as_ref().map(|gb| (&gb.roughness, SAMPLED)),
            "gbuffer_velocity" => self.gbuffer.as_ref().map(|gb| (&gb.velocity, SAMPLED)),
            // The reactive mask: in the graph only on a frame that stores and
            // reads it, and sampled between frames.
            "reactive_mask" => Some((&self.targets.reactive_mask.resource, SAMPLED)),
            // `gbuffer_depth` is deliberately unregistered: it is a depth
            // target rather than a color one, and the only pass that moves it
            // is the upscaler, which borrows it inside its own dispatch.
            // The Hi-Z pyramid rests where the cull kernel samples it, which is a
            // compute stage, so it is the non-pixel shader-resource state rather
            // than the sampled default.
            "hiz_pyramid" => self.cull.hiz.as_ref().map(|h| (&h.texture, h.rest_state)),
            _ => None,
        })
    }

    // The per-frame `RaymarchView` payload. Its VP is the jittered one the Main
    // pass rasterizes with (`vp_mat`), so raymarched surfaces share their NDC
    // depth space with rasterized geometry and take the same TAA jitter.
    fn build_raymarch_view(&self, params: &GraphFrameParams<'_>) -> super::raymarch::RaymarchView {
        super::raymarch::RaymarchView::new(&self.pass_camera(params, params.vp_mat))
    }

    // The per-frame `TransparentView` payload. Its VP is the jittered one
    // (`vp_mat`) the Main pass rasterized with, so the glass quad's clip-space
    // depth matches the stored main-depth the fragment shader tests against.
    fn build_transparent_view(
        &self,
        params: &GraphFrameParams<'_>,
    ) -> super::transparent::TransparentView {
        super::transparent::TransparentView::new(
            &self.pass_camera(params, params.vp_mat),
            &self.uniforms.light_uniforms,
        )
    }

    // The camera a screen-space pass reconstructs from: `vp` at the frame's
    // render size, with this frame's IBL and sky rotation.
    fn pass_camera(&self, params: &GraphFrameParams<'_>, vp: [[f32; 4]; 4]) -> PassCamera {
        PassCamera {
            vp,
            inv_vp: mat4_inverse(vp),
            cam_pos: params.cam_pos,
            viewport: [params.width as f32, params.height as f32],
            time: params.elapsed,
            prefilter_mip_count: self.scene.env_map.prefilter_mip_count as f32,
            sky_rot: self.state.view.sky_rot,
        }
    }

    // Per-pass dispatch, called from both the worker fan-out and the
    // main-thread composite arm. Each arm encodes onto the `cmd` it's
    // given (the worker's per-pass cmd list, or the outer "end" cmd
    // list for composite). Composite is **not** routed through this
    // method; the caller calls `encode_composite_and_text` directly
    // so the trailing timestamp + resolve land on the same cmd list.
    fn encode_pass_into(
        &self,
        pass_id: PassId,
        cmd: &ID3D12GraphicsCommandList,
        params: &GraphFrameParams<'_>,
        particle_frame: Option<&super::particle::ParticleFrame>,
    ) -> RenderResult<()> {
        match pass_id {
            PassId::Cull => {
                self.encode_cull(cmd, params.frame_idx, params.frustum, params.cam_pos);
                // Pose the skinned objects' deformed-vertex buffer for this frame
                // (a no-op when no skinned mesh is folded in). Independent of the
                // cull; both feed Main, which the toposort orders after Cull.
                self.encode_skin(cmd, params.frame_idx);
            }
            PassId::SsaoBlur => {
                self.encode_ssao(cmd, params.frame_idx, params.fov_y_radians, params.aspect)?;
            }
            PassId::SsaoDepth | PassId::SsaoKernel => {
                return Err(RenderError::Other(format!(
                    "graph executor (directx): pass {} is bundled inside SsaoBlur \
                     (encode_ssao encodes the SSAO depth copy, kernel and blur); it \
                     should not appear as its own graph node",
                    pass_id.name()
                )));
            }
            PassId::Sky => {
                // Drawn inline at the tail of Main and of every probe face and
                // mirror render; it only names a timing slot.
                return Err(RenderError::Other(format!(
                    "graph executor (directx): pass {} is drawn inline by the opaque \
                     scene passes and should not appear as a graph node",
                    pass_id.name()
                )));
            }
            PassId::ReflectionComposite => {
                // Metal-only inline pass; never scheduled on DirectX. Handled
                // here only to keep the dispatch match exhaustive.
                return Err(RenderError::Other(format!(
                    "graph executor (directx): pass {} is a Metal-only inline \
                     reflection composite and should not appear as a graph node",
                    pass_id.name()
                )));
            }
            PassId::LightCull => {
                // Bins the local lights and probes into per-cluster light lists
                // and probe masks. The builder emits this node only while either
                // is live (matching `clustering_enabled`), and the RAW
                // edge on `cluster_lists` pins it before Main, which reads
                // the same buffer.
                self.encode_light_cull(
                    cmd,
                    params.frame_idx,
                    self.main_cluster_grid(params.frame_idx),
                );
            }
            PassId::Shadow => {
                // Build the raymarch view only when at least one volume
                // opted in to shadow casting; otherwise pass `None` so
                // the shadow encoder skips the SDF caster sub-pass with
                // zero overhead. The view stays consistent with the
                // matching `PassId::Raymarch` build later this frame:
                // same `cur_vp`, `cam_pos`, `elapsed`, viewport, and
                // prefilter mip count.
                let raymarch_view = self
                    .raymarch
                    .as_ref()
                    .filter(|rm| rm.any_shadow_casters())
                    .map(|_| self.build_raymarch_view(params));
                self.encode_shadow_pass(
                    cmd,
                    params.frame_idx,
                    params.shadow_ubo_gva,
                    params.cam_pos,
                    raymarch_view.as_ref(),
                );
            }
            PassId::SpotShadow => {
                // One depth-only render per scheduled spot slice. The builder
                // emits this node only when the world has shadow-casting spots,
                // and the RAW edge on `spot_shadow_map` pins it before Main.
                self.encode_spot_shadow_pass(cmd, params.frame_idx, params.cam_pos);
            }
            PassId::AutoExposure => {
                self.encode_auto_exposure(cmd, params.frame_idx);
            }
            PassId::Main => {
                self.encode_main_pass(
                    cmd,
                    params.frame_idx,
                    MainPassExtent {
                        width: params.width,
                        height: params.height,
                    },
                    FrameGpuBuffers {
                        view_gva: params.view_gva,
                        light_gva: params.light_gva,
                        local_lights_gva: params.local_lights_gva,
                        shadow_ubo_gva: params.shadow_ubo_gva,
                    },
                    params.world_hidden,
                );
            }
            PassId::Decals => {
                self.encode_decals(cmd, params.frame_idx, params.vp_mat, params.frustum);
            }
            PassId::Lines => {
                self.encode_lines(cmd, params.frame_idx, params.vp_mat, params.lines);
            }
            PassId::Fog => {
                self.encode_fog(cmd, params.frame_idx, params.vp_mat, params.cam_pos);
            }
            PassId::ParticlesSim => {
                // Integrates each live emitter's persistent pool. The graph's only
                // edge out of it is the draw's vertex-stage read of those pools, so
                // the schedule is free to put it on the async queue; this executor
                // still records it in the compiled order. The per-frame particle
                // state was advanced once before the fan-out by `prepare_particle_pass`.
                if let Some(frame) = particle_frame {
                    self.encode_particles_sim(cmd, frame);
                }
            }
            PassId::ParticlesDraw => {
                self.targets
                    .reactive_mask
                    .clear(cmd, params.reactive.particles);
                if let Some(frame) = particle_frame {
                    self.encode_particles_draw(
                        cmd,
                        params.frame_idx,
                        frame,
                        params.vp_mat,
                        params.frustum,
                        params.reactive.particles,
                    );
                }
            }
            PassId::SsrResolve => {
                self.encode_ssr_resolve(
                    cmd,
                    params.frame_idx,
                    params.fov_y_radians,
                    params.aspect,
                    params.cam_pos,
                );
            }
            PassId::TaaResolve => {
                self.encode_taa(cmd, params.frame_idx, params.reactive.readable);
            }
            PassId::Bloom => {
                self.encode_bloom(cmd, params.frame_idx, params.scene_srv)?;
            }
            PassId::Composite => {
                // Composite is run inline on the outer "end" cmd list
                // by `execute_graph` itself so it shares a submission
                // with the trailing timestamp + resolve. This arm is
                // unreachable through the worker fan-out; see the
                // method docstring.
                return Err(RenderError::Other(
                    "graph executor (directx): Composite must run on the outer cmd \
                     list: encode_pass_into is not the right entry point"
                        .to_string(),
                ));
            }
            PassId::Raymarch => {
                let view = self.build_raymarch_view(params);
                self.encode_raymarch(cmd, params.frame_idx, &view)?;
            }
            PassId::FogFroxel => {
                self.encode_fog_froxel(
                    cmd,
                    params.frame_idx,
                    params.near,
                    params.vp_mat,
                    params.cam_pos,
                    params.shadow_ubo_gva,
                );
            }
            PassId::Upscale => {
                // FSR3 temporal upscaler. Driven by the shared graph
                // when `FrameGraphInputs::upscale_enabled` is on (see
                // `record_frame::seed_inputs`). The encoder dispatches
                // FFX against this pass's per-pass cmd list, reading
                // the post-SSR scene + velocity + main depth and
                // writing into the upscaler's output texture (which
                // bloom + composite then sample via `scene_srv_for_post`).
                self.encode_upscale(cmd, params)?;
            }
            PassId::PlanarReflection => {
                // Mirror renders for the flat reflectors in view, each cropped to
                // the screen rectangle its reflectors cover (see `planar.rs`).
                // Recorded ahead of `Transparent` in the frame's submission, each
                // plane's resolve left shader-readable for it.
                self.encode_planar_reflections(cmd, params)?;
            }
            PassId::Transparent => {
                // Generic translucent pass: draws the world's glass panes and
                // water surfaces back-to-front over the post-SSR scene. Gated by
                // `FrameGraphInputs::transparent_enabled`
                // (`DxContext::transparent_enabled`), so it only appears when the
                // world declared a visible `GlassPanel` or `WaterSurface`.
                self.targets
                    .reactive_mask
                    .clear(cmd, params.reactive.transparent);
                let view = self.build_transparent_view(params);
                self.encode_transparent(
                    cmd,
                    params.frame_idx,
                    &view,
                    params.fov_y_radians,
                    params.aspect,
                    params.reactive.transparent,
                )?;
            }
            PassId::HizBuild | PassId::HizFinal => {
                // Two Hi-Z builds share one encoder. `HizBuild` rebuilds the
                // pyramid mid-frame from phase-1 depth so Cull2 re-tests the
                // phase-1 occluded objects against up-to-date depth; `HizFinal`
                // reduces the frame's final depth for the next frame's phase-1
                // cull. Both read the depth the graph has already transitioned.
                self.encode_hiz_build(cmd);
            }
            PassId::Cull2 => {
                self.encode_cull_phase2(cmd, params.frame_idx, params.frustum, params.cur_vp);
            }
            PassId::Main2 => {
                self.encode_main_pass_phase2(
                    cmd,
                    params.frame_idx,
                    params.width,
                    params.height,
                    FrameGpuBuffers {
                        view_gva: params.view_gva,
                        light_gva: params.light_gva,
                        local_lights_gva: params.local_lights_gva,
                        shadow_ubo_gva: params.shadow_ubo_gva,
                    },
                );
            }
            PassId::Ssgi => {
                self.encode_ssgi(cmd, params.frame_idx, params.fov_y_radians, params.aspect);
            }
            PassId::RtReflections => {
                // Hardware ray-traced reflections (DXR inline `RayQuery`). Traces
                // a reflection ray per glossy pixel against the scene TLAS and
                // composites into the RT output target, which `scene_srv_for_post`
                // then feeds the post stack. Occupies the SsrResolve slot; gated
                // by `FrameGraphInputs::rt_reflections_enabled`
                // (`DxContext::rt_reflections_active`). The per-frame TLAS update
                // already ran on the outer "start" cmd list before this trace.
                self.encode_rt_reflections(
                    cmd,
                    params.frame_idx,
                    params.fov_y_radians,
                    params.aspect,
                    params.cam_pos,
                );
            }
            PassId::GBufferPrepass => {
                // Unified geometry pre-pass: one jittered traversal writes
                // normal+depth, roughness, and motion for every screen-space
                // consumer (SSR / SSAO / SSGI / TAA / FSR). `params.vp_mat` is
                // the jittered VP (rasterization, matching the main pass);
                // `params.cur_vp` is the un-jittered VP the shader uses with the
                // previous VP for the motion vector. The velocity channel
                // carries real motion only when a consumer reads it (TAA, FSR
                // or SSGI); otherwise cur == prev and it stays a harmless zero.
                self.encode_gbuffer_prepass(
                    cmd,
                    params.frame_idx,
                    crate::directx::post::gbuffer::GbufferPrepassView {
                        jittered_vp: params.vp_mat,
                        cur_vp: params.cur_vp,
                        elapsed: params.elapsed,
                        cam_pos: params.cam_pos,
                    },
                    crate::directx::post::gbuffer::GbufferPrepassFrame {
                        velocity_active: self.reads_motion(),
                        prime_history: params.prime_model_history,
                    },
                );
            }
        }
        Ok(())
    }
}