proofman-common 1.3.0-alpha

Shared proof/setup contexts, traces, and STARK metadata types for the PIL2 proofman framework
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
use std::collections::{HashMap, HashSet};
use std::ffi::c_void;

use proofman_fields::PrimeField64;
use proofman_starks_lib_c::{expressions_bin_new_c, expressions_bin_free_c};

use crate::format_bytes;
use crate::load_const_pols;
use crate::{GlobalInfo, GlobalInfoAir};
use crate::ProofmanError;
use crate::ProofmanResult;
use crate::Setup;
use crate::exec_header;
use crate::ProofType;

/// The slot an air's packed const pols occupy in the GPU const buffer. Airs with identical
/// fixed columns -- same verkey, i.e. same const-tree root -- share one slot; `owner` is the
/// first of the group and the only one that uploads.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct FixedGroup {
    pub owner: (usize, usize),
}

/// Columns the recursion's HOST trace buffer needs for one air.
///
/// On GPU it only stages the exec map's width for `widenCompactWitnessGPU` to widen device-side. On
/// CPU it is the air's full width: since the solve and the scatter are fused, `generate_witness`
/// takes a pooled buffer before the prover buffer is in scope, so the fill can no longer go straight
/// into the prover's own cm1 slot. Must agree with `recursion_trace_stride`, which decides the fill
/// from the same exec header -- sizing narrow while the fill goes wide overruns the buffer.
///
/// One function because three call sites decide this, and the one that did it inline sized the whole
/// pool at the air's full width while the other two were compact.
pub fn recursion_staging_cols<F: PrimeField64>(setup: &Setup<F>, gpu: bool) -> u64 {
    let cm1 = setup.stark_info.map_sections_n["cm1"];
    if !gpu {
        return cm1;
    }
    setup.exec_data.as_deref().map(|e| exec_header(e).map_cols).filter(|&m| m > 0 && m < cm1).unwrap_or(cm1)
}

pub struct SetupsVadcop<F: PrimeField64> {
    pub sctx_compressor: Option<SetupCtx<F>>,
    pub sctx_recursive1: Option<SetupCtx<F>>,
    pub sctx_recursive2: Option<SetupCtx<F>>,
    pub setup_vadcop_final: Option<Setup<F>>,
    pub setup_vadcop_final_compressed: Option<Setup<F>>,
    pub max_compact_trace_size: usize,
    pub max_const_size: usize,
    pub max_const_tree_size: usize,
    pub max_prover_buffer_size: usize,
    pub max_prover_recursive_buffer_size: usize,
    pub max_prover_recursive2_buffer_size: usize,
    pub max_pinned_proof_size: usize,
    pub max_n_bits_ext: usize,
    pub total_const_pols_size: usize,
    pub recurser_const_slot_size: usize,
}

unsafe impl<F: PrimeField64> Send for SetupsVadcop<F> {}
unsafe impl<F: PrimeField64> Sync for SetupsVadcop<F> {}

impl<F: PrimeField64> SetupsVadcop<F> {
    pub fn new(
        global_info: &GlobalInfo,
        verify_constraints: bool,
        aggregation: bool,
        gpu: bool,
    ) -> ProofmanResult<Self> {
        if aggregation {
            let sctx_compressor = SetupCtx::new(global_info, &ProofType::Compressor, verify_constraints, gpu)?;
            let sctx_recursive1 = SetupCtx::new(global_info, &ProofType::Recursive1, verify_constraints, gpu)?;
            let sctx_recursive2 = SetupCtx::new(global_info, &ProofType::Recursive2, verify_constraints, gpu)?;
            let setup_vadcop_final = Setup::new(
                &global_info.get_setup_path("vadcop_final"),
                0,
                0,
                &GlobalInfoAir::new("VadcopFinal".to_string()),
                &ProofType::VadcopFinal,
                verify_constraints,
                gpu,
                None,
            )?;

            // Only if the key says it carries the stage. `Setup::new` reads the starkinfo from disk
            // and panics if it is absent, so a key built without the compressed final -- which is
            // the default for blake3, where it measured a 2% smaller proof for a whole extra
            // recursion layer -- would fail here, before proving even starts.
            let setup_vadcop_final_compressed = if global_info.has_compressed_final {
                Some(Setup::new(
                    &global_info.get_setup_path("vadcop_final_compressed"),
                    0,
                    0,
                    &GlobalInfoAir::new("VadcopFinalCompressed".to_string()),
                    &ProofType::VadcopFinalCompressed,
                    verify_constraints,
                    gpu,
                    None,
                )?)
            } else {
                None
            };

            let recurser_const_slot_size = if gpu {
                let n_constants = setup_vadcop_final.stark_info.n_constants as usize;
                let n_rows = 1usize << setup_vadcop_final.stark_info.stark_struct.n_bits;
                1 + n_constants + n_rows * n_constants
            } else {
                0
            };

            let total_const_pols_size = sctx_compressor.total_const_pols_size
                + sctx_recursive1.total_const_pols_size
                + sctx_recursive2.total_const_pols_size
                + setup_vadcop_final.const_pols_size_packed
                + setup_vadcop_final_compressed.as_ref().map_or(0, |s| s.const_pols_size_packed)
                + recurser_const_slot_size;

            let max_const_size = sctx_compressor
                .max_const_size
                .max(sctx_recursive1.max_const_size)
                .max(sctx_recursive2.max_const_size)
                .max(setup_vadcop_final.const_pols_size);
            let max_const_tree_size = sctx_compressor
                .max_const_tree_size
                .max(sctx_recursive1.max_const_tree_size)
                .max(sctx_recursive2.max_const_tree_size)
                .max(setup_vadcop_final.const_tree_size)
                .max(setup_vadcop_final_compressed.as_ref().map_or(0, |s| s.const_tree_size));
            let max_prover_buffer_size = sctx_compressor
                .max_prover_buffer_size
                .max(sctx_recursive1.max_prover_buffer_size)
                .max(sctx_recursive2.max_prover_buffer_size)
                .max(setup_vadcop_final.prover_buffer_size as usize)
                .max(setup_vadcop_final_compressed.as_ref().map_or(0, |s| s.prover_buffer_size as usize));

            // Recursive-capable buffers = prover buffer (mapTotalN) + witness tail: the room past the
            // proof layout where the recursive proof's COMPACT host trace lands (staging cols x N:
            // each repository's `max_compact_trace_size`, the finals' own staging width) before the device
            // widens it into cm1; gen_recursive_proof_gpu checks the room at launch. Sized exactly
            // to that trace -- 4e44d5549 sized it at the full trace, ~1 GB more than ever lands.
            let vadcop_final_tail = (recursion_staging_cols(&setup_vadcop_final, gpu)
                * (1 << setup_vadcop_final.stark_info.stark_struct.n_bits))
                as usize;
            // Its OWN staging width, not vadcop_final's: the two airs have had identical geometry so
            // far, which is why pairing them never showed; it would undersize the buffer the moment
            // they diverged.
            let vadcop_final_compressed_tail = setup_vadcop_final_compressed
                .as_ref()
                .map_or(0, |s| (recursion_staging_cols(s, gpu) * (1 << s.stark_info.stark_struct.n_bits)) as usize);
            let max_prover_recursive2_buffer_size = (sctx_recursive2.max_prover_buffer_size
                + sctx_recursive2.max_compact_trace_size)
                .max(sctx_recursive1.max_prover_buffer_size + sctx_recursive1.max_compact_trace_size);

            let max_prover_recursive_buffer_size = (sctx_recursive2.max_prover_buffer_size
                + sctx_recursive2.max_compact_trace_size)
                .max(sctx_recursive1.max_prover_buffer_size + sctx_recursive1.max_compact_trace_size)
                .max(sctx_compressor.max_prover_buffer_size + sctx_compressor.max_compact_trace_size)
                .max(setup_vadcop_final.prover_buffer_size as usize + vadcop_final_tail)
                .max(
                    setup_vadcop_final_compressed
                        .as_ref()
                        .map_or(0, |c| c.prover_buffer_size as usize + vadcop_final_compressed_tail),
                );

            // This floors every non-recursive GPU stream class, so which term dominates decides
            // whether a regular class fits at all.
            tracing::debug!(
                "Recursive buffer requirement: compressor {}, recursive1 {}, recursive2 {}, vadcop_final {}, vadcop_final_compressed {}",
                format_bytes((sctx_compressor.max_prover_buffer_size + sctx_compressor.max_compact_trace_size) as f64 * 8.0),
                format_bytes((sctx_recursive1.max_prover_buffer_size + sctx_recursive1.max_compact_trace_size) as f64 * 8.0),
                format_bytes((sctx_recursive2.max_prover_buffer_size + sctx_recursive2.max_compact_trace_size) as f64 * 8.0),
                format_bytes((setup_vadcop_final.prover_buffer_size as usize + vadcop_final_tail) as f64 * 8.0),
                format_bytes(
                    setup_vadcop_final_compressed
                        .as_ref()
                        .map_or(0, |c| c.prover_buffer_size as usize + vadcop_final_compressed_tail) as f64
                        * 8.0
                ),
            );

            let max_pinned_proof_size = sctx_compressor
                .max_pinned_proof_size
                .max(sctx_recursive1.max_pinned_proof_size)
                .max(sctx_recursive2.max_pinned_proof_size)
                .max(setup_vadcop_final.proof_size as usize)
                .max(setup_vadcop_final_compressed.as_ref().map_or(0, |c| c.proof_size as usize));

            let max_n_bits_ext = sctx_compressor
                .max_n_bits_ext
                .max(sctx_recursive1.max_n_bits_ext)
                .max(sctx_recursive2.max_n_bits_ext)
                .max(setup_vadcop_final.stark_info.stark_struct.n_bits_ext as usize);

            // Largest compact (staging-width) trace over every recursive proof kind, compressor
            // included: they share one host pool (see MemoryHandlerRecursive::trace), and it is also
            // the witness tail each recursive-capable stream class reserves past mapTotalN.
            let max_compact_trace_size = sctx_recursive1
                .max_compact_trace_size
                .max(sctx_recursive2.max_compact_trace_size)
                .max(sctx_compressor.max_compact_trace_size)
                // Their STAGING width, not `vadcop_final_trace_size` -- that one is the prover
                // buffer's and stays full. Sizing the shared pool from it put every buffer at the
                // air's full width, which is what made the pool 7x what it holds.
                .max(vadcop_final_tail)
                .max(vadcop_final_compressed_tail);

            Ok(SetupsVadcop {
                sctx_compressor: Some(sctx_compressor),
                sctx_recursive1: Some(sctx_recursive1),
                sctx_recursive2: Some(sctx_recursive2),
                setup_vadcop_final: Some(setup_vadcop_final),
                setup_vadcop_final_compressed,
                max_const_tree_size,
                max_const_size,
                max_prover_buffer_size,
                max_prover_recursive_buffer_size,
                max_prover_recursive2_buffer_size,
                max_pinned_proof_size,
                max_n_bits_ext,
                max_compact_trace_size,
                total_const_pols_size,
                recurser_const_slot_size,
            })
        } else {
            Ok(SetupsVadcop {
                sctx_compressor: None,
                sctx_recursive1: None,
                sctx_recursive2: None,
                setup_vadcop_final: None,
                setup_vadcop_final_compressed: None,
                total_const_pols_size: 0,
                recurser_const_slot_size: 0,
                max_const_tree_size: 0,
                max_const_size: 0,
                max_prover_buffer_size: 0,
                max_prover_recursive_buffer_size: 0,
                max_prover_recursive2_buffer_size: 0,
                max_pinned_proof_size: 0,
                max_n_bits_ext: 0,
                max_compact_trace_size: 0,
            })
        }
    }

    /// Size of every pooled signalValues buffer (in u64 elements): the largest `total_signal_no`
    /// across the recursive/compressor/final setups, so every circuit takes a pooled buffer and
    /// none allocates its own.
    ///
    /// Sizing from the second largest would spare `n_streams x (largest - second)` of host memory
    /// -- 328 MB per stream on the blake3 key, where the sole compressor (Keccakf) is 3.1x the next
    /// -- and it is not worth it: that made the compressor allocate and first-touch 485 MB per
    /// proof, measured at +37.6 ms on its witness (204.1 vs 166.5 ms, n=36 each). 0 if no setup
    /// exposes a `total_signal_no`, and the pool then stays off.
    pub fn signal_pool_cap(&self) -> usize {
        let mut sizes: Vec<usize> = Vec::new();
        for sctx in [self.sctx_compressor.as_ref(), self.sctx_recursive1.as_ref(), self.sctx_recursive2.as_ref()]
            .into_iter()
            .flatten()
        {
            sizes.extend(sctx.total_signal_nos());
        }
        for setup in
            [self.setup_vadcop_final.as_ref(), self.setup_vadcop_final_compressed.as_ref()].into_iter().flatten()
        {
            if let Some(n) = setup.total_signal_no {
                sizes.push(n as usize);
            }
        }
        sizes.into_iter().max().unwrap_or(0)
    }

    pub fn get_setup(&self, airgroup_id: usize, air_id: usize, setup_type: &ProofType) -> ProofmanResult<&Setup<F>> {
        match setup_type {
            ProofType::Compressor => self.sctx_compressor.as_ref().unwrap().get_setup(airgroup_id, air_id),
            ProofType::Recursive1 => self.sctx_recursive1.as_ref().unwrap().get_setup(airgroup_id, air_id),
            ProofType::Recursive2 => self.sctx_recursive2.as_ref().unwrap().get_setup(airgroup_id, air_id),
            ProofType::VadcopFinal => Ok(self.setup_vadcop_final.as_ref().unwrap()),
            // The only optional setup of the family: keys built without the compressed-final
            // stage (blake3's default) legitimately have none.
            ProofType::VadcopFinalCompressed => self.setup_vadcop_final_compressed.as_ref().ok_or_else(|| {
                ProofmanError::InvalidSetup("Proving key was built without the vadcop_final_compressed stage".into())
            }),
            _ => Err(ProofmanError::InvalidSetup("Invalid setup type".into())),
        }
    }
}

#[derive(Debug)]
pub struct SetupRepository<F: PrimeField64> {
    setups: HashMap<(usize, usize), Setup<F>>,
    max_const_tree_size: usize,
    max_const_size: usize,
    max_prover_buffer_size: usize,
    prover_buffer_sizes: Vec<((usize, usize), usize)>,
    max_prover_contributions_size: usize,
    max_pinned_proof_size: usize,
    max_compact_trace_size: usize,
    total_const_pols_size: usize,
    total_custom_commits_reserved_words: usize,
    fixed_groups: HashMap<(usize, usize), FixedGroup>,
    global_bin: Option<*mut c_void>,
    global_info_file: String,
    max_n_bits_ext: usize,
    max_const_pols_size_packed: usize,
    const_slot_cache_slots: usize,
}

unsafe impl<F: PrimeField64> Send for SetupRepository<F> {}
unsafe impl<F: PrimeField64> Sync for SetupRepository<F> {}

impl<F: PrimeField64> Drop for SetupRepository<F> {
    fn drop(&mut self) {
        if let Some(global_bin_ptr) = self.global_bin {
            expressions_bin_free_c(global_bin_ptr);
        }
    }
}

impl<F: PrimeField64> SetupRepository<F> {
    pub fn new(
        global_info: &GlobalInfo,
        setup_type: &ProofType,
        verify_constraints: bool,
        gpu: bool,
    ) -> ProofmanResult<Self> {
        let mut setups = HashMap::new();

        let global_bin = match setup_type == &ProofType::Basic {
            true => {
                let global_bin_path =
                    &global_info.get_proving_key_path().join("pilout.globalConstraints.bin").display().to_string();
                Some(expressions_bin_new_c(global_bin_path.as_str(), true, false))
            }
            false => None,
        };

        let global_info_path = &global_info.get_proving_key_path().join("pilout.globalInfo.json");
        let global_info_file = global_info_path.to_str().unwrap().to_string();

        let mut max_const_tree_size = 0;
        let mut max_const_size = 0;
        let mut max_n_bits_ext = 0;
        let mut max_prover_contributions_size = 0;
        let mut max_prover_buffer_size = 0;
        let mut prover_buffer_sizes: Vec<((usize, usize), usize)> = Vec::new();
        let mut max_pinned_proof_size = 0;
        let mut total_const_pols_size = 0;
        let mut total_custom_commits_reserved_words = 0;
        let mut max_compact_trace_size = 0;
        let mut max_const_pols_size_packed = 0;
        let mut n_const_slots = 0;

        // Airs in the order load_device_const_pols walks them, and the slot each verkey maps to.
        let mut sized_airs: Vec<(usize, usize)> = Vec::new();
        let mut groups: HashMap<Vec<u64>, (usize, usize)> = HashMap::new();

        // Initialize Hashmap for each airgroup_id, air_id

        for (airgroup_id, air_group) in global_info.airs.iter().enumerate() {
            for (air_id, _) in air_group.iter().enumerate() {
                let setup_path = global_info.get_air_setup_path(airgroup_id, air_id, setup_type);
                let setup = Setup::new(
                    &setup_path,
                    airgroup_id,
                    air_id,
                    &global_info.airs[airgroup_id][air_id],
                    setup_type,
                    verify_constraints,
                    gpu,
                    Some(&global_info.get_air_setup_path(airgroup_id, 0, &ProofType::Recursive2)),
                )?;
                if setup_type != &ProofType::Compressor || global_info.get_air_has_compressor(airgroup_id, air_id) {
                    let n = 1 << setup.stark_info.stark_struct.n_bits;
                    let n_bits_ext = setup.stark_info.stark_struct.n_bits_ext;
                    if max_const_tree_size < setup.const_tree_size {
                        max_const_tree_size = setup.const_tree_size;
                    }
                    if max_const_size < setup.const_pols_size {
                        max_const_size = setup.const_pols_size;
                    }
                    if max_prover_buffer_size < setup.prover_buffer_size {
                        max_prover_buffer_size = setup.prover_buffer_size;
                    }
                    prover_buffer_sizes.push(((airgroup_id, air_id), setup.prover_buffer_size as usize));
                    if max_prover_contributions_size < setup.contributions_size {
                        max_prover_contributions_size = setup.contributions_size;
                    }
                    if setup.gpu {
                        sized_airs.push((airgroup_id, air_id));
                        if !setup.verkey.is_empty() {
                            groups.entry(setup.get_vk()).or_insert((airgroup_id, air_id));
                        }
                    }
                    max_pinned_proof_size = max_pinned_proof_size.max(setup.pinned_proof_size);
                    max_n_bits_ext = max_n_bits_ext.max(n_bits_ext);
                    // Basic airs never land a compact witness (no exec map; their trace goes straight
                    // into cm1), so their repository reports 0 rather than a meaningless full width.
                    if setup_type != &ProofType::Basic {
                        max_compact_trace_size =
                            max_compact_trace_size.max((recursion_staging_cols(&setup, gpu) * n) as usize);
                    }
                }
                setups.insert((airgroup_id, air_id), setup);
                if setup_type == &ProofType::Recursive2 {
                    break;
                }
            }
        }

        // Second pass: the group owner is only settled once every member has been seen. Sizes
        // one slot per group, in the same order the loader assigns offsets -- must not drift.
        let mut fixed_groups: HashMap<(usize, usize), FixedGroup> = HashMap::new();
        let mut sized_slots: HashSet<(usize, usize)> = HashSet::new();
        let mut shared_airs = 0;
        let mut saved = 0;
        for air in sized_airs {
            let setup = &setups[&air];
            let group = match groups.get(&setup.get_vk()) {
                Some(&owner) => FixedGroup { owner },
                // Nothing to fingerprint with (no verkey): the air is its own group.
                None => FixedGroup { owner: air },
            };
            fixed_groups.insert(air, group);
            max_const_pols_size_packed = max_const_pols_size_packed.max(setup.const_pols_size_packed);
            if sized_slots.insert(group.owner) {
                n_const_slots += 1;
                total_const_pols_size += setup.const_pols_size_packed;
                // Custom commits ride the same buffer; the slot is filled later, when
                // register_custom_commits supplies the file path.
                total_const_pols_size += setup.custom_commits_reserved_words;
                total_custom_commits_reserved_words += setup.custom_commits_reserved_words;
            } else {
                shared_airs += 1;
                saved += setup.const_pols_size_packed;
                saved += setup.custom_commits_reserved_words;
            }
        }
        if shared_airs > 0 {
            let proof_type: &str = (*setup_type).into();
            tracing::info!(
                "Sharing GPU const pols: {shared_airs} {proof_type} airs reuse another air's fixed columns, \
                 saving {} MB",
                saved * std::mem::size_of::<u64>() / (1024 * 1024)
            );
        }

        prover_buffer_sizes.sort_by(|(ka, sa), (kb, sb)| sb.cmp(sa).then(ka.cmp(kb)));

        // Recursive1 const pols are not resident per setup: the loader carves a slot cache of
        // RECURSIVE1_CONST_SLOTS (or fewer when there are fewer setups) slots of the largest packed
        // set, filled at launch from host pinned copies (see DeviceCommitBuffers::constCache). The
        // custom commits of this repository would need resident slots; the recursion setups have
        // none, which is asserted here rather than assumed.
        let const_slot_cache_slots = if gpu && *setup_type == ProofType::Recursive1 && n_const_slots > 0 {
            assert!(
                total_custom_commits_reserved_words == 0,
                "recursive1 setups with custom commits cannot use the const slot cache"
            );
            RECURSIVE1_CONST_SLOTS.min(n_const_slots)
        } else {
            0
        };
        if const_slot_cache_slots > 0 {
            total_const_pols_size = const_slot_cache_slots * max_const_pols_size_packed;
        }

        Ok(Self {
            setups,
            fixed_groups,
            global_bin,
            global_info_file,
            max_const_tree_size,
            max_const_size,
            max_prover_contributions_size: max_prover_contributions_size as usize,
            max_prover_buffer_size: max_prover_buffer_size as usize,
            prover_buffer_sizes,
            max_pinned_proof_size: max_pinned_proof_size as usize,
            total_const_pols_size,
            total_custom_commits_reserved_words,
            max_compact_trace_size,
            max_n_bits_ext: max_n_bits_ext as usize,
            max_const_pols_size_packed,
            const_slot_cache_slots,
        })
    }
}

/// Air instance context for managing air instances (traces)
#[allow(dead_code)]
pub struct SetupCtx<F: PrimeField64> {
    setup_repository: SetupRepository<F>,
    pub max_const_tree_size: usize,
    pub max_const_size: usize,
    pub max_prover_contributions_size: usize,
    pub max_prover_buffer_size: usize,
    /// Per-air prover buffer size, largest first. See `SetupRepository::prover_buffer_sizes`.
    pub prover_buffer_sizes: Vec<((usize, usize), usize)>,
    pub max_pinned_proof_size: usize,
    pub max_compact_trace_size: usize,
    pub max_n_bits_ext: usize,
    pub total_const_pols_size: usize,
    /// Included in `total_const_pols_size`, and tracked separately because it stays resident even
    /// in no-const-buf mode: nothing stages custom commits per switch.
    pub total_custom_commits_reserved_words: usize,
    /// Largest packed const-pols set of the repository (elements): the slot size of the const cache.
    pub max_const_pols_size_packed: usize,
    /// Slots of the const slot cache this repository uses (recursive1 on GPU), 0 = resident slots.
    pub const_slot_cache_slots: usize,
    setup_type: ProofType,
}

/// Slots of the recursive1 const slot cache: a block uses ~20 distinct recursive1 setups, so 20
/// makes the second use of an air within a block a hit while freeing (44 - 20) x 100 MiB.
pub const RECURSIVE1_CONST_SLOTS: usize = 20;

impl<F: PrimeField64> SetupCtx<F> {
    pub fn new(
        global_info: &GlobalInfo,
        setup_type: &ProofType,
        verify_constraints: bool,
        gpu: bool,
    ) -> ProofmanResult<Self> {
        let setup_repository = SetupRepository::new(global_info, setup_type, verify_constraints, gpu)?;
        let max_const_tree_size = setup_repository.max_const_tree_size;
        let max_const_size = setup_repository.max_const_size;
        let max_prover_contributions_size = setup_repository.max_prover_contributions_size;
        let max_prover_buffer_size = setup_repository.max_prover_buffer_size;
        let prover_buffer_sizes = setup_repository.prover_buffer_sizes.clone();
        let max_pinned_proof_size = setup_repository.max_pinned_proof_size;
        let total_const_pols_size = setup_repository.total_const_pols_size;
        let total_custom_commits_reserved_words = setup_repository.total_custom_commits_reserved_words;
        let max_compact_trace_size = setup_repository.max_compact_trace_size;
        let max_n_bits_ext = setup_repository.max_n_bits_ext;
        let max_const_pols_size_packed = setup_repository.max_const_pols_size_packed;
        let const_slot_cache_slots = setup_repository.const_slot_cache_slots;
        Ok(SetupCtx {
            setup_repository,
            max_const_tree_size,
            max_const_size,
            max_prover_contributions_size,
            max_prover_buffer_size,
            prover_buffer_sizes,
            max_compact_trace_size,
            max_pinned_proof_size,
            max_n_bits_ext,
            total_const_pols_size,
            total_custom_commits_reserved_words,
            max_const_pols_size_packed,
            const_slot_cache_slots,
            setup_type: *setup_type,
        })
    }

    pub fn get_setup(&self, airgroup_id: usize, air_id: usize) -> ProofmanResult<&Setup<F>> {
        match self.setup_repository.setups.get(&(airgroup_id, air_id)) {
            Some(setup) => Ok(setup),
            None => Err(ProofmanError::InvalidSetup(format!(
                "Setup not found for airgroup_id: {}, air_id: {}",
                airgroup_id, air_id
            ))),
        }
    }

    /// `None` for airs never uploaded to the GPU (CPU setups, compressor-less airs).
    pub fn get_fixed_group(&self, airgroup_id: usize, air_id: usize) -> Option<FixedGroup> {
        self.setup_repository.fixed_groups.get(&(airgroup_id, air_id)).copied()
    }

    pub fn get_fixed(&self, airgroup_id: usize, air_id: usize) -> ProofmanResult<Vec<F>> {
        match self.setup_repository.setups.get(&(airgroup_id, air_id)) {
            Some(setup) => {
                let mut const_pols: Vec<F> = vec![F::ZERO; setup.const_pols_size];
                load_const_pols(setup, &mut const_pols);
                Ok(const_pols)
            }
            None => Err(ProofmanError::InvalidSetup(format!(
                "Setup not found for airgroup_id: {}, air_id: {}",
                airgroup_id, air_id
            ))),
        }
    }

    pub fn get_setups_list(&self) -> Vec<(usize, usize)> {
        self.setup_repository.setups.keys().cloned().collect()
    }

    /// `total_signal_no` (circom signalValues length, u64 elements) for every setup
    /// in this repository. Used to size the signalValues buffer pool.
    pub fn total_signal_nos(&self) -> Vec<usize> {
        self.setup_repository.setups.values().filter_map(|s| s.total_signal_no.map(|n| n as usize)).collect()
    }

    pub fn get_global_bin(&self) -> *mut c_void {
        self.setup_repository.global_bin.unwrap()
    }

    pub fn get_global_info_file(&self) -> String {
        self.setup_repository.global_info_file.clone()
    }
}

#[cfg(test)]
mod staging_tests {
    /// Every recursive air's staging width has to come from `recursion_staging_cols`, or one of them
    /// sizes the shared pool at its full width and the others' compaction buys nothing. The final
    /// airs were computed inline and did exactly that: 6 buffers at 1.07 GB instead of 0.075 GB.
    ///
    /// Grep rather than a call, because the defect is a call site that does the arithmetic ITSELF --
    /// it cannot be caught by testing the function everyone else already uses.
    #[test]
    fn no_call_site_sizes_the_recursive_pool_by_hand() {
        let src = include_str!("setup_ctx.rs");
        // Every `let` that sizes a compact trace must go through recursion_staging_cols and never
        // read cm1 directly: the SetupsVadcop max, the per-repository max, and the two finals'
        // tails the SetupsVadcop max is assembled from.
        for decl in ["let max_compact_trace_size", "let vadcop_final_tail", "let vadcop_final_compressed_tail"] {
            for body in src.split(decl).skip(1) {
                // Only real declarations (`let x = ...`), not this test's own string literals.
                if !body.trim_start().starts_with('=') {
                    continue;
                }
                let body = &body[..body.find(';').unwrap_or(body.len())];
                assert!(
                    !body.contains("map_sections_n"),
                    "{decl} reads cm1 directly; it must go through recursion_staging_cols:\n{body}"
                );
                let via_rule = body.contains("recursion_staging_cols");
                let via_tails = body.contains("vadcop_final_tail") || body.contains("max_compact_trace_size");
                assert!(via_rule || via_tails, "{decl} must use the shared rule:\n{body}");
            }
        }
    }
}