rvsim-core 2.0.0

A cycle-level RISC-V 64-bit system simulator.
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
//! The pipeline: backend, widths, queues, functional units and memory
//! dependence prediction.

use super::defaults;
use crate::config::{
    BranchPredictorKind, IttageConfig, LoopConfig, PerceptronConfig, ScConfig, TageConfig,
    TournamentConfig,
};
use crate::isa::encoding::zicboz::CBOZ_BLOCK_SIZE;
use crate::isa::rvv::Vlen;
use serde::Deserialize;

/// The widest unit-stride vector access: one 64-byte line, the smallest
/// line every cache level must have.
pub const MAX_VECTOR_MEM_WIDTH: usize = CBOZ_BLOCK_SIZE as usize;

/// Specifies the memory dependence prediction algorithm used to determine
/// whether loads can bypass older unresolved stores at issue time.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Deserialize)]
#[serde(rename_all = "PascalCase")]
pub enum MemDepPredictorKind {
    /// Blind (conservative) predictor.
    ///
    /// Loads always wait for all older stores to resolve. No speculation,
    /// no violations.
    Blind,
    /// Store-set predictor (Chrysos & Emer 1998), gem5 O3's only predictor.
    ///
    /// Learns load-store dependencies from ordering violations and allows
    /// loads predicted independent to bypass unresolved stores. The default.
    #[default]
    StoreSet,
}

/// Pipeline and branch predictor configuration.
#[derive(Debug, Clone, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct PipelineConfig {
    /// Superscalar width (instructions per cycle)
    #[serde(default = "PipelineConfig::default_width")]
    pub width: usize,

    /// Cycles between commit detecting a trap or interrupt and the pipeline
    /// squashing into its handler.
    #[serde(default = "PipelineConfig::default_trap_latency")]
    pub trap_latency: u64,

    /// Cycles between execute resolving a misprediction, CSR write, fault
    /// or ordering violation and the pipeline squashing into the redirect;
    /// the backend's gem5 value when unset.
    #[serde(default)]
    pub redirect_latency: Option<u64>,

    /// Cycles from a load matching a store in the store buffer to its data
    /// reaching writeback, where a load the L1D answers takes the L1D hit
    /// latency. The L1D hit latency when unset, as a core whose forwarding
    /// shares the load pipeline; `1` is gem5's O3 LSQ, whose forwarded load
    /// writes back the cycle after it executes; `0` writes the load back in
    /// the cycle it matches. A forwarded vector span takes at least one
    /// cycle.
    #[serde(default)]
    pub store_forward_latency: Option<u64>,

    /// Instructions fetched per cycle; `width` when unset.
    #[serde(default)]
    pub fetch_width: Option<usize>,

    /// Instructions decoded per cycle; `width` when unset.
    #[serde(default)]
    pub decode_width: Option<usize>,

    /// Instructions renamed and dispatched per cycle; `width` when unset.
    #[serde(default)]
    pub rename_width: Option<usize>,

    /// Instructions issued to execute per cycle; `width` when unset.
    #[serde(default)]
    pub issue_width: Option<usize>,

    /// Instructions retired per cycle; `width` when unset.
    #[serde(default)]
    pub commit_width: Option<usize>,

    /// Results written back (waking dependents and completing in the ROB)
    /// per cycle in the out-of-order backend; `width` when unset.
    #[serde(default)]
    pub writeback_width: Option<usize>,

    /// Branch predictor type
    #[serde(default)]
    pub branch_predictor: BranchPredictorKind,

    /// Branch Target Buffer size
    #[serde(default = "PipelineConfig::default_btb_size")]
    pub btb_size: usize,

    /// Branch Target Buffer associativity (ways per set)
    #[serde(default = "PipelineConfig::default_btb_ways")]
    pub btb_ways: usize,

    /// Return Address Stack size
    #[serde(default = "PipelineConfig::default_ras_size")]
    pub ras_size: usize,

    /// `misa` from an ISA string such as `"RV64IMAFDC"`, instead of the
    /// default RV64IMAFDC.
    #[serde(default, deserialize_with = "deserialize_misa")]
    pub misa_override: Option<crate::isa::misa::Misa>,

    /// TAGE predictor configuration
    #[serde(default)]
    pub tage: TageConfig,

    /// Perceptron predictor configuration
    #[serde(default)]
    pub perceptron: PerceptronConfig,

    /// Tournament predictor configuration
    #[serde(default)]
    pub tournament: TournamentConfig,

    /// Statistical Corrector configuration (used by SC-L-TAGE)
    #[serde(default)]
    pub sc: ScConfig,

    /// Indirect Target TAGE configuration (used by SC-L-TAGE)
    #[serde(default)]
    pub ittage: IttageConfig,

    /// Loop predictor configuration (used by SC-L-TAGE)
    #[serde(default)]
    pub loop_predictor: LoopConfig,

    /// Backend type (`InOrder` or `OutOfOrder`)
    #[serde(default)]
    pub backend: BackendKind,

    /// Reorder Buffer size
    #[serde(default = "PipelineConfig::default_rob_size")]
    pub rob_size: usize,

    /// Store Buffer size
    #[serde(default = "PipelineConfig::default_store_buffer_size")]
    pub store_buffer_size: usize,

    /// Issue Queue size (for O3 backend)
    #[serde(default = "PipelineConfig::default_issue_queue_size")]
    pub issue_queue_size: usize,

    /// Physical Register File GPR size (O3 backend).
    /// Must satisfy: `prf_gpr_size` >= 32 + `rob_size`.
    #[serde(default = "PipelineConfig::default_prf_gpr_size")]
    pub prf_gpr_size: usize,

    /// Physical Register File FPR size (O3 backend).
    /// Must satisfy: `prf_fpr_size` >= 32 + `rob_size`.
    #[serde(default = "PipelineConfig::default_prf_fpr_size")]
    pub prf_fpr_size: usize,

    /// Load Queue size (O3 backend).
    #[serde(default = "PipelineConfig::default_load_queue_size")]
    pub load_queue_size: usize,

    /// Number of load ports (loads issued per cycle, O3 backend).
    #[serde(default = "PipelineConfig::default_load_ports")]
    pub load_ports: usize,

    /// Number of store ports (stores issued per cycle, O3 backend).
    #[serde(default = "PipelineConfig::default_store_ports")]
    pub store_ports: usize,

    /// Functional unit pool configuration (O3 backend).
    #[serde(default)]
    pub fu_config: FuConfig,

    /// Number of checkpoint slots for O(1) branch recovery (0 = disabled).
    #[serde(default = "PipelineConfig::default_checkpoint_count")]
    pub checkpoint_count: usize,

    /// Reorder-buffer entries commit squashes per cycle after a
    /// misprediction, trap or ordering violation; rename is blocked until
    /// it has finished (gem5's `squashWidth`).
    #[serde(default = "PipelineConfig::default_squash_width")]
    pub squash_width: usize,

    /// Memory dependence predictor type
    #[serde(default)]
    pub mem_dep_predictor: MemDepPredictorKind,

    /// Store-set predictor configuration
    #[serde(default)]
    pub store_set: StoreSetConfig,

    /// Vector register width in bits (VLEN), a power of 2 in [128, 2048].
    #[serde(default)]
    pub vlen: Vlen,

    /// Number of vector execution lanes. Defaults to vlen/64 (min 1).
    #[serde(default)]
    pub num_vec_lanes: Option<usize>,

    /// Bytes one unit-stride vector memory access moves: the vector memory
    /// datapath width. Defaults to one register, VLEN/8, up to a line.
    #[serde(default)]
    pub vector_mem_width: Option<usize>,

    /// Vector Physical Register File size (O3 backend).
    #[serde(default = "PipelineConfig::default_prf_vpr_size")]
    pub prf_vpr_size: usize,

    /// Enable vector operation chaining.
    #[serde(default = "PipelineConfig::default_vec_chaining")]
    pub vec_chaining: bool,

    /// Vector Store Buffer capacity (in-flight vec stores). O3 backend only.
    #[serde(default = "PipelineConfig::default_vec_store_buffer_size")]
    pub vec_store_buffer_size: usize,

    /// Forwarding policy for vector-store→load. O3 backend only. Default
    /// `byte_mask` matches BOOM/Apple/Intel/AMD/ARM. `stall` matches Saturn.
    /// `off` always stalls.
    #[serde(default)]
    pub vec_store_forwarding: crate::config::VecStoreForwarding,
}

impl PipelineConfig {
    /// Instructions fetched per cycle.
    #[must_use]
    pub const fn fetch_width(&self) -> usize {
        Self::stage_width(self.fetch_width, self.width)
    }

    /// Instructions decoded per cycle.
    #[must_use]
    pub const fn decode_width(&self) -> usize {
        Self::stage_width(self.decode_width, self.width)
    }

    /// Instructions renamed and dispatched per cycle.
    #[must_use]
    pub const fn rename_width(&self) -> usize {
        Self::stage_width(self.rename_width, self.width)
    }

    /// Instructions issued to execute per cycle.
    #[must_use]
    pub const fn issue_width(&self) -> usize {
        Self::stage_width(self.issue_width, self.width)
    }

    /// Instructions retired per cycle.
    #[must_use]
    pub const fn commit_width(&self) -> usize {
        Self::stage_width(self.commit_width, self.width)
    }

    /// Results written back per cycle.
    #[must_use]
    pub const fn writeback_width(&self) -> usize {
        Self::stage_width(self.writeback_width, self.width)
    }

    /// Cycles between execute resolving a redirect and the squash into it.
    #[must_use]
    pub const fn redirect_latency(&self) -> u64 {
        match self.redirect_latency {
            Some(latency) => latency,
            None => match self.backend {
                BackendKind::InOrder => defaults::REDIRECT_LATENCY_INORDER,
                BackendKind::OutOfOrder => defaults::REDIRECT_LATENCY_O3,
            },
        }
    }

    const fn stage_width(configured: Option<usize>, width: usize) -> usize {
        match configured {
            Some(stage) => stage,
            None => width,
        }
    }

    /// Returns the default pipeline width (instructions per cycle).
    const fn default_width() -> usize {
        defaults::PIPELINE_WIDTH
    }

    /// Returns the default Branch Target Buffer size.
    const fn default_btb_size() -> usize {
        defaults::BTB_SIZE
    }

    /// Returns the default BTB associativity.
    const fn default_btb_ways() -> usize {
        defaults::BTB_WAYS
    }

    /// Returns the default Return Address Stack size.
    const fn default_ras_size() -> usize {
        defaults::RAS_SIZE
    }

    /// Returns the default ROB size.
    const fn default_rob_size() -> usize {
        defaults::ROB_SIZE
    }

    /// Returns the default store buffer size.
    const fn default_store_buffer_size() -> usize {
        defaults::STORE_BUFFER_SIZE
    }

    /// Returns the default issue queue size.
    const fn default_issue_queue_size() -> usize {
        defaults::ISSUE_QUEUE_SIZE
    }

    /// Returns the default PRF GPR size.
    const fn default_prf_gpr_size() -> usize {
        defaults::PRF_GPR_SIZE
    }

    /// Returns the default PRF FPR size.
    const fn default_prf_fpr_size() -> usize {
        defaults::PRF_FPR_SIZE
    }

    /// Returns the default load queue size.
    const fn default_load_queue_size() -> usize {
        defaults::LOAD_QUEUE_SIZE
    }

    /// Returns the default number of load ports.
    /// Vector execution lanes: `num_vec_lanes`, or one per 64 bits of VLEN.
    #[must_use]
    pub fn vector_lanes(&self) -> usize {
        self.num_vec_lanes.unwrap_or_else(|| (self.vlen.bits() / 64).max(1))
    }

    /// Bytes one unit-stride vector access moves: `vector_mem_width`, or one
    /// register (VLEN/8) up to the widest access a line allows.
    #[must_use]
    pub fn vector_mem_width_bytes(&self) -> usize {
        self.vector_mem_width.unwrap_or_else(|| self.vlen.bytes().min(MAX_VECTOR_MEM_WIDTH))
    }

    const fn default_load_ports() -> usize {
        defaults::LOAD_PORTS
    }

    /// Returns the default number of store ports.
    const fn default_store_ports() -> usize {
        defaults::STORE_PORTS
    }

    /// Returns the default checkpoint count.
    const fn default_checkpoint_count() -> usize {
        defaults::CHECKPOINT_COUNT
    }

    const fn default_squash_width() -> usize {
        defaults::SQUASH_WIDTH
    }

    const fn default_trap_latency() -> u64 {
        defaults::TRAP_LATENCY
    }

    /// Returns the default vector PRF size.
    const fn default_prf_vpr_size() -> usize {
        64
    }

    /// Returns the default vector chaining setting.
    const fn default_vec_chaining() -> bool {
        true
    }

    /// Returns the default Vector Store Buffer size.
    const fn default_vec_store_buffer_size() -> usize {
        defaults::VEC_STORE_BUFFER_SIZE
    }
}

impl Default for PipelineConfig {
    fn default() -> Self {
        Self {
            width: defaults::PIPELINE_WIDTH,
            trap_latency: defaults::TRAP_LATENCY,
            redirect_latency: None,
            store_forward_latency: None,
            fetch_width: None,
            decode_width: None,
            rename_width: None,
            issue_width: None,
            commit_width: None,
            writeback_width: None,
            branch_predictor: BranchPredictorKind::default(),
            btb_size: defaults::BTB_SIZE,
            btb_ways: defaults::BTB_WAYS,
            ras_size: defaults::RAS_SIZE,
            misa_override: None,
            tage: TageConfig::default(),
            perceptron: PerceptronConfig::default(),
            tournament: TournamentConfig::default(),
            sc: ScConfig::default(),
            ittage: IttageConfig::default(),
            loop_predictor: LoopConfig::default(),
            backend: BackendKind::default(),
            rob_size: defaults::ROB_SIZE,
            store_buffer_size: defaults::STORE_BUFFER_SIZE,
            issue_queue_size: defaults::ISSUE_QUEUE_SIZE,
            prf_gpr_size: defaults::PRF_GPR_SIZE,
            prf_fpr_size: defaults::PRF_FPR_SIZE,
            load_queue_size: defaults::LOAD_QUEUE_SIZE,
            load_ports: defaults::LOAD_PORTS,
            store_ports: defaults::STORE_PORTS,
            fu_config: FuConfig::default(),
            checkpoint_count: defaults::CHECKPOINT_COUNT,
            squash_width: defaults::SQUASH_WIDTH,
            mem_dep_predictor: MemDepPredictorKind::default(),
            store_set: StoreSetConfig::default(),
            vlen: Vlen::default(),
            num_vec_lanes: None,
            vector_mem_width: None,
            prf_vpr_size: 64,
            vec_chaining: true,
            vec_store_buffer_size: defaults::VEC_STORE_BUFFER_SIZE,
            vec_store_forwarding: crate::config::VecStoreForwarding::ByteMask,
        }
    }
}

/// Store-set memory dependence predictor configuration.
#[derive(Debug, Clone, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct StoreSetConfig {
    /// SSIT (Store Set ID Table) size — indexed by `(pc >> 2) % ssit_size`.
    #[serde(default = "StoreSetConfig::default_ssit_size")]
    pub ssit_size: usize,

    /// LFST (Last Fetched Store Table) size — indexed by store set ID.
    #[serde(default = "StoreSetConfig::default_lfst_size")]
    pub lfst_size: usize,

    /// Loads and stores dispatched between wipes of both tables (0 = never),
    /// so stale learned dependencies do not throttle a program forever.
    #[serde(default = "StoreSetConfig::default_clear_period")]
    pub clear_period: u64,
}

impl Default for StoreSetConfig {
    fn default() -> Self {
        Self {
            ssit_size: Self::default_ssit_size(),
            lfst_size: Self::default_lfst_size(),
            clear_period: Self::default_clear_period(),
        }
    }
}

impl StoreSetConfig {
    /// gem5 O3's `SSITSize`.
    const fn default_ssit_size() -> usize {
        1024
    }

    /// gem5 O3's `LFSTSize`.
    const fn default_lfst_size() -> usize {
        1024
    }

    /// gem5 O3's `store_set_clear_period`.
    const fn default_clear_period() -> u64 {
        250_000
    }
}

fn deserialize_misa<'de, D>(deserializer: D) -> Result<Option<crate::isa::misa::Misa>, D::Error>
where
    D: serde::Deserializer<'de>,
{
    Option::<String>::deserialize(deserializer)?
        .map(|isa| isa.parse().map_err(serde::de::Error::custom))
        .transpose()
}

/// Configuration for the functional unit pool.
#[derive(Clone, Debug, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct FuConfig {
    /// Number of integer ALU units.
    pub num_int_alu: usize,
    /// Latency of integer ALU operations in cycles.
    pub int_alu_latency: u64,
    /// Number of integer multiplier units.
    pub num_int_mul: usize,
    /// Latency of integer multiply operations in cycles.
    pub int_mul_latency: u64,
    /// Number of integer divider units.
    pub num_int_div: usize,
    /// Latency of integer divide operations in cycles.
    pub int_div_latency: u64,
    /// Number of floating-point adder units.
    pub num_fp_add: usize,
    /// Latency of floating-point add operations in cycles.
    pub fp_add_latency: u64,
    /// Number of floating-point multiplier units.
    pub num_fp_mul: usize,
    /// Latency of floating-point multiply operations in cycles.
    pub fp_mul_latency: u64,
    /// Number of floating-point fused multiply-add units.
    pub num_fp_fma: usize,
    /// Latency of floating-point FMA operations in cycles.
    pub fp_fma_latency: u64,
    /// Number of floating-point divide/sqrt units.
    pub num_fp_div_sqrt: usize,
    /// Latency of floating-point divide/sqrt operations in cycles.
    pub fp_div_sqrt_latency: u64,
    /// Number of branch units.
    pub num_branch: usize,
    /// Latency of branch operations in cycles.
    pub branch_latency: u64,
    /// Number of memory (load/store) units.
    pub num_mem: usize,
    /// Latency of memory operations in cycles.
    pub mem_latency: u64,
    /// Number of vector integer ALU units.
    #[serde(default = "default_num_vec_int_alu")]
    pub num_vec_int_alu: usize,
    /// Startup latency of vector integer ALU operations.
    #[serde(default = "default_vec_int_alu_latency")]
    pub vec_int_alu_latency: u64,
    /// Number of vector integer multiplier units.
    #[serde(default = "default_num_vec_int_mul")]
    pub num_vec_int_mul: usize,
    /// Startup latency of vector integer multiply operations.
    #[serde(default = "default_vec_int_mul_latency")]
    pub vec_int_mul_latency: u64,
    /// Number of vector integer divider units.
    #[serde(default = "default_num_vec_int_div")]
    pub num_vec_int_div: usize,
    /// Per-element latency of vector integer divide operations.
    #[serde(default = "default_vec_int_div_latency")]
    pub vec_int_div_latency: u64,
    /// Number of vector FP ALU units.
    #[serde(default = "default_num_vec_fp_alu")]
    pub num_vec_fp_alu: usize,
    /// Startup latency of vector FP ALU operations.
    #[serde(default = "default_vec_fp_alu_latency")]
    pub vec_fp_alu_latency: u64,
    /// Number of vector FP FMA units.
    #[serde(default = "default_num_vec_fp_fma")]
    pub num_vec_fp_fma: usize,
    /// Startup latency of vector FP FMA operations.
    #[serde(default = "default_vec_fp_fma_latency")]
    pub vec_fp_fma_latency: u64,
    /// Number of vector FP div/sqrt units.
    #[serde(default = "default_num_vec_fp_div_sqrt")]
    pub num_vec_fp_div_sqrt: usize,
    /// Per-element latency of vector FP div/sqrt operations.
    #[serde(default = "default_vec_fp_div_sqrt_latency")]
    pub vec_fp_div_sqrt_latency: u64,
    /// Number of vector memory units.
    #[serde(default = "default_num_vec_mem")]
    pub num_vec_mem: usize,
    /// Startup latency of vector memory operations.
    #[serde(default = "default_vec_mem_latency")]
    pub vec_mem_latency: u64,
    /// Number of vector permute units.
    #[serde(default = "default_num_vec_permute")]
    pub num_vec_permute: usize,
    /// Startup latency of vector permute operations.
    #[serde(default = "default_vec_permute_latency")]
    pub vec_permute_latency: u64,
}

impl Default for FuConfig {
    fn default() -> Self {
        Self {
            num_int_alu: 4,
            int_alu_latency: 1,
            num_int_mul: 1,
            int_mul_latency: 3,
            num_int_div: 1,
            int_div_latency: 35,
            num_fp_add: 2,
            fp_add_latency: 4,
            num_fp_mul: 2,
            fp_mul_latency: 5,
            num_fp_fma: 2,
            fp_fma_latency: 5,
            num_fp_div_sqrt: 1,
            fp_div_sqrt_latency: 21,
            num_branch: 2,
            branch_latency: 1,
            num_mem: 2,
            mem_latency: 1,
            num_vec_int_alu: default_num_vec_int_alu(),
            vec_int_alu_latency: default_vec_int_alu_latency(),
            num_vec_int_mul: default_num_vec_int_mul(),
            vec_int_mul_latency: default_vec_int_mul_latency(),
            num_vec_int_div: default_num_vec_int_div(),
            vec_int_div_latency: default_vec_int_div_latency(),
            num_vec_fp_alu: default_num_vec_fp_alu(),
            vec_fp_alu_latency: default_vec_fp_alu_latency(),
            num_vec_fp_fma: default_num_vec_fp_fma(),
            vec_fp_fma_latency: default_vec_fp_fma_latency(),
            num_vec_fp_div_sqrt: default_num_vec_fp_div_sqrt(),
            vec_fp_div_sqrt_latency: default_vec_fp_div_sqrt_latency(),
            num_vec_mem: default_num_vec_mem(),
            vec_mem_latency: default_vec_mem_latency(),
            num_vec_permute: default_num_vec_permute(),
            vec_permute_latency: default_vec_permute_latency(),
        }
    }
}

/// Backend type selection.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Deserialize)]
#[serde(rename_all = "PascalCase")]
pub enum BackendKind {
    /// In-order pipeline (default).
    #[default]
    InOrder,
    /// Out-of-order pipeline (future).
    OutOfOrder,
}

/// Forwarding policy. Selects how `forward_load` reacts to in-flight vec stores.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, serde::Deserialize, serde::Serialize)]
#[serde(rename_all = "snake_case")]
pub enum VecStoreForwarding {
    /// Per-line byte-mask forwarding (BOOM/Apple/Intel/AMD/ARM pattern). Default.
    #[default]
    ByteMask,
    /// Saturn pattern: never forward; stall on overlap; miss otherwise.
    Stall,
    /// Most conservative: stall on any older in-flight vec store.
    Off,
}

const fn default_num_vec_int_alu() -> usize {
    1
}

const fn default_vec_int_alu_latency() -> u64 {
    1
}

const fn default_num_vec_int_mul() -> usize {
    1
}

const fn default_vec_int_mul_latency() -> u64 {
    3
}

const fn default_num_vec_int_div() -> usize {
    1
}

const fn default_vec_int_div_latency() -> u64 {
    20
}

const fn default_num_vec_fp_alu() -> usize {
    1
}

const fn default_vec_fp_alu_latency() -> u64 {
    4
}

const fn default_num_vec_fp_fma() -> usize {
    1
}

const fn default_vec_fp_fma_latency() -> u64 {
    5
}

const fn default_num_vec_fp_div_sqrt() -> usize {
    1
}

const fn default_vec_fp_div_sqrt_latency() -> u64 {
    20
}

const fn default_num_vec_mem() -> usize {
    1
}

const fn default_vec_mem_latency() -> u64 {
    1
}

const fn default_num_vec_permute() -> usize {
    1
}

const fn default_vec_permute_latency() -> u64 {
    1
}