zygo-cli 0.1.5

zygo — a daemonless, rootless warm sandbox runtime
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
// SPDX-License-Identifier: Apache-2.0
//! `zygo bench` — measure the warm path against the real pool.
//!
//! `crates/zygo-core/examples/poc3_warm_path.rs` measured p50 1887 µs against a
//! 2000 µs budget: **6% of headroom**. The
//! supervisor's queue, timers and metrics all land on this path, so
//! a number that is only ever measured by hand will be spent without anyone
//! noticing. This runs the same measurement against `zygo_core::pool`, which is
//! the code that ships.

use std::time::{Duration, Instant};

use anyhow::Context;
use zygo_core::pool::{CpuAccounting, Pool, PoolConfig};
use zygo_core::spec::{Layer, ResolveOptions, Spec};

use crate::cli::{BenchCommand, Cli};
use crate::output::{self, Style};

/// The warm-path budgets: p50 under 2 ms, p99 under 10 ms.
const WARM_P50_BUDGET_US: f64 = 2_000.0;
const WARM_P99_BUDGET_US: f64 = 10_000.0;

/// The acceptance criterion for warm-exec: "a Go binary under 3 ms".
///
/// A different budget because it is a different thing: an agent request is a
/// `fork()` of a warm interpreter, a warm-exec request is a fresh process
/// entered into the sandbox and `execve`d — six `setns` calls, the full
/// hardening sequence and the program's own start-up, every time. Measured at
/// p50 2.2 ms for `sh -c cat`; with `python3` per request it is 54 ms, which is
/// exactly why interpreters get an agent instead.
const EXEC_P50_BUDGET_US: f64 = 3_000.0;

/// The embedded-runtime roadmap's exit criterion for a runtime pool: a
/// request's p99 under 5 ms, with a different script every time.
///
/// A p99 rather than a p50, because that is what the roadmap asks and because
/// it is the number an embedder's own tail depends on. It is checked here as
/// a p50 budget too — set to the same figure — so a pool whose median is
/// already outside it is not reported as passing on a quiet percentile.
const POOL_P99_BUDGET_US: f64 = 5_000.0;

pub fn run(cli: &Cli, command: &BenchCommand) -> anyhow::Result<u8> {
    let measured = match command {
        BenchCommand::Warm {
            n,
            no_cgroup,
            rate,
            cpu,
            pool,
            scripts,
            cmd,
        } => warm(
            cli,
            *n,
            !no_cgroup,
            *rate,
            *cpu,
            (!cmd.is_empty()).then_some(cmd.as_slice()),
            pool.then_some(*scripts),
        )?,
        BenchCommand::Cold { n, image, command } => cold(cli, *n, image, command.as_deref())?,
        BenchCommand::Load {
            seconds,
            concurrency,
            cpu,
        } => load(cli, *seconds, *concurrency, *cpu)?,
        BenchCommand::All { quick } => return all(cli, *quick),
    };
    if cli.json {
        output::json(&measured.1)?;
    }
    Ok(measured.0)
}

/// An exit code of its own for "these numbers are not a verdict".
///
/// Distinct from 1, which means a budget was missed. A caller that treats
/// every non-zero exit as a regression would otherwise file a bug against the
/// code when the answer was a hot laptop.
const NOT_A_VERDICT: u8 = 2;

fn warm(
    cli: &Cli,
    n: u32,
    per_request_cgroup: bool,
    rate: Option<f64>,
    cpu: Option<f64>,
    cmd: Option<&[String]>,
    pool_scripts: Option<u32>,
) -> anyhow::Result<(u8, serde_json::Value)> {
    let paths = super::paths(cli);
    let pool = Pool::new(PoolConfig {
        per_request_cgroup,
        ..PoolConfig::new(paths.clone())
    })?;

    // An empty handler, so what is measured is overhead and nothing else —
    // or, for warm-exec, the smallest program that completes the contract.
    let dir = tempfile::tempdir()?;
    let handler = dir.path().join("handler.py");
    std::fs::write(&handler, "def handler(event):\n    return None\n")?;

    // The pool's scripts: the same empty handler, with a distinct constant in
    // each source so nothing can be deduped or cached across them. Built up
    // front, because building one is not what is being measured.
    let scripts: Vec<zygo_core::protocol::Script> = (0..pool_scripts.unwrap_or(0))
        .map(|i| {
            zygo_core::protocol::Script::inline(format!(
                "CONSTANT = {i}\n\n\ndef handler(event):\n    return None\n"
            ))
        })
        .collect();

    let spec = Spec::default();
    let resolved = spec.resolve(
        None,
        &Layer {
            entry: (cmd.is_none() && pool_scripts.is_none()).then(|| handler.clone()),
            cmd: cmd.map(<[String]>::to_vec),
            image: (cmd.is_some() || pool_scripts.is_some())
                .then(|| "python:3.12-slim".to_string()),
            runtime: pool_scripts.map(|_| {
                zygo_core::spec::Runtime::Builtin(zygo_core::spec::BuiltinRuntime::Python)
            }),
            cpu: cpu.map(zygo_core::spec::Cpu),
            ..Default::default()
        },
        &ResolveOptions {
            one_shot: true,
            pool: pool_scripts.is_some(),
            ..Default::default()
        },
    )?;

    let style = Style::stdout();
    eprintln!(
        "{} {}  {}{}",
        style.dim("image"),
        resolved.image,
        style.dim(if per_request_cgroup {
            "per-request cgroup"
        } else {
            "no per-request cgroup"
        }),
        match cmd {
            Some(c) => format!("  {}", style.dim(&format!("warm-exec: {}", c.join(" ")))),
            None => String::new(),
        }
    );

    let warmup_started = Instant::now();
    let function = pool
        .serve(&resolved)
        .with_context(|| format!("could not warm `{}`", resolved.image))?;
    let warmup = warmup_started.elapsed();

    let status = function.status();
    eprintln!(
        "{} {} in {:.0} ms (imports {:.1} ms, rss {} MB)",
        style.dim("warm"),
        status.runtime,
        warmup.as_secs_f64() * 1000.0,
        status.imports_ms,
        status.rss_kb / 1024
    );

    // A short warm-up run first: the first few requests pay for page faults in
    // the interpreter that every later one inherits.
    //
    // With a pool, every script is called once before anything is measured.
    // That is the roadmap's "p99 after the first call to each", and it is the
    // honest shape either way: a first call pays for whatever a first call
    // pays for, and an embedder's tenth thousandth does not.
    let settle = (n / 20).clamp(50, 500);
    if scripts.is_empty() {
        for _ in 0..settle {
            function.call(serde_json::Value::Null)?;
        }
    } else {
        for script in &scripts {
            function.call_script_with_timeout(
                serde_json::Value::Null,
                Some(script.clone()),
                Duration::from_secs(30),
            )?;
        }
    }

    let mut samples = Vec::with_capacity(n as usize);
    let mut phases: Vec<zygo_core::pool::CallTiming> = Vec::with_capacity(n as usize);
    let mut handler_us: Vec<f64> = Vec::with_capacity(n as usize);

    // The tenant's CPU quota, before and after. A run with no think time asks
    // for slightly more than one core — the agent and the child it is tearing
    // down overlap — so a `cpu = 1.0` tenant meets its own quota and CFS stops
    // it until the next period. Without these counters that tail reads as a
    // defect in the runtime; with them it reads as the limit working.
    let cpu_before = function.cpu_accounting();

    let interval = request_interval(rate);
    let started = Instant::now();
    for i in 0..n {
        // Paced from the start of the run rather than from the last request, so
        // a slow request is absorbed instead of shifting every one after it.
        if let Some(interval) = interval {
            let due = interval.mul_f64(f64::from(i));
            if let Some(wait) = due.checked_sub(started.elapsed()) {
                std::thread::sleep(wait);
            }
        }
        let t0 = Instant::now();
        // A different script every time, cycling: the point of a pool is that
        // request *n* and request *n+1* are different tenants' code, and a
        // benchmark that sent one script ten thousand times would be
        // measuring a cache nobody has.
        let (outcome, timing) = match scripts.is_empty() {
            true => function.call_timed(serde_json::Value::Null, Duration::from_secs(30))?,
            false => function.call_script_timed(
                serde_json::Value::Null,
                Some(scripts[i as usize % scripts.len()].clone()),
                Duration::from_secs(30),
            )?,
        };
        samples.push(t0.elapsed().as_secs_f64() * 1e6);
        phases.push(timing);
        // What the child says it spent inside the handler. The difference
        // between this and the `run` phase is plumbing: the pipe, the child's
        // start-up and its exit.
        handler_us.push(outcome.metrics.wall_ms * 1000.0);

        anyhow::ensure!(
            outcome.succeeded(),
            "request {i} failed: {}",
            outcome.error.unwrap_or_else(|| "no error given".into())
        );
        if n >= 1000 && i > 0 && i % (n / 10) == 0 {
            eprint!("\r  {}%", i * 100 / n);
        }
    }
    let elapsed = started.elapsed();
    let quota = match (cpu_before, function.cpu_accounting()) {
        (Some(before), Some(after)) => Some(after.since(&before)),
        _ => None,
    };
    if n >= 1000 {
        eprintln!("\r      ");
    }

    // What the machine itself costs for a fork and a teardown, with no sandbox
    // and no Zygo in the way. A p99 that merely tracks this floor is a property
    // of the host, not of the code — and on a busy or nested-virtualised
    // machine the floor can be most of the budget.
    let floor = measure_fork_floor(500);

    let mut report = Report::of(&samples, elapsed, n, quota);
    if cmd.is_some() {
        report.p50_budget = EXEC_P50_BUDGET_US;
        report.label = "warm-exec request overhead";
    }
    if !scripts.is_empty() {
        report.p50_budget = POOL_P99_BUDGET_US;
        report.p99_budget = POOL_P99_BUDGET_US;
        report.label = "pooled script request overhead";
    }
    let mut json = report.to_json();
    if !scripts.is_empty() {
        json["scripts"] = scripts.len().into();
    }
    json["warm_ms"] = (warmup.as_secs_f64() * 1000.0).into();
    json["imports_ms"] = status.imports_ms.into();
    json["rss_kb"] = status.rss_kb.into();
    json["runtime"] = status.runtime.clone().into();
    if !cli.json {
        report.print(&style);
        print_phases(&phases);
        if cmd.is_none() {
            print_handler_share(&phases, &handler_us);
        }
        print_floor(&style, &floor, &report);
        // Last: it is a conclusion about the breakdown, so it reads after it
        // rather than in the middle of it.
        print_cgroup_note(&phases, &report, &style);
    }

    let _ = function.shutdown();
    Ok((u8::from(!report.within_budget()), json))
}

// ---------------------------------------------------------------------------
// `zygo bench all`
// ---------------------------------------------------------------------------

/// The numbers the README and `docs/book/25-performance.md` print.
///
/// They are here, as constants, so that `zygo bench all` compares what it
/// measured with what is *claimed* rather than only with a budget. A published
/// number nobody can reproduce is a number that drifts silently: the budget
/// stays met, the documentation stays wrong, and the first person to notice is
/// a reader who tried it.
///
/// Every one of these was measured on 25 September 2026 in the Lima VM on an
/// Apple M1 Max (2 vCPU, Ubuntu 24.04, Linux 6.8, an ordinary user), except
/// where `docs/book/25-performance.md` says otherwise. A different machine will
/// not reproduce them, which is the point of printing the machine.
mod published {
    /// The warm path, at 250 requests a second. The p99 is a Linux 6.x cgroup
    /// cost (the book's "why 1 in 100 is slow on newer kernels"); on 5.10 it
    /// was 2.60 ms the same day.
    pub const WARM_P50_MS: f64 = 1.44;
    pub const WARM_P99_MS: f64 = 10.53;
    pub const WARM_RATE: f64 = 250.0;
    /// Sustained throughput at a concurrency of four.
    pub const LOAD_PER_SECOND: f64 = 1108.0;
    /// A one-shot `zygo run`, image already in the store.
    pub const COLD_P50_MS: f64 = 12.3;
    /// Warm-exec: a fresh process entered into a held sandbox.
    pub const EXEC_P50_MS: f64 = 1.40;
    /// A runtime pool, a different script with every request, at the same
    /// rate. `ROADMAP.md`'s exit criterion for Phase 1 is the p99 under 5 ms.
    pub const POOL_P50_MS: f64 = 1.91;
    pub const POOL_P99_MS: f64 = 11.36;
    /// What `zygo serve` costs once, for a Python handler with no imports, as
    /// `bench warm` measures it (the supervisor already running).
    pub const SERVE_MS: f64 = 34.0;
}

/// How far a measurement may be from the published number before it is worth
/// pointing at.
///
/// Wide on purpose. This is not a regression gate — the budgets are — it is a
/// "the documentation is describing a different machine" detector, and a
/// factor of two is what distinguishes that from ordinary variation between
/// hosts.
const PUBLISHED_TOLERANCE: f64 = 2.0;

/// `zygo bench all` — every published number, on this host, in one command.
fn all(cli: &Cli, quick: bool) -> anyhow::Result<u8> {
    let style = Style::stdout();
    let host = Host::describe(&super::paths(cli));

    if !cli.json {
        host.print(&style);
        println!();
    }

    // Sampled around the whole run, not around each measurement: a machine
    // that starts throttling during the cold-start benchmark has invalidated
    // the warm-path numbers that came before it too, because the cause is the
    // machine and not the order.
    let before = Thermal::sample();

    let (warm_n, cold_n, load_seconds) = if quick {
        (500, 10, 3)
    } else {
        (10_000, 50, 10)
    };

    let mut results = serde_json::Map::new();
    let mut failed = 0u8;

    let mut step = |name: &str, outcome: anyhow::Result<(u8, serde_json::Value)>| {
        match outcome {
            Ok((code, json)) => {
                failed |= code;
                results.insert(name.to_string(), json);
            }
            Err(e) => {
                // One measurement that cannot run must not take the other
                // three with it: a host without `runsc` still has a warm path,
                // and a report of three numbers and one reason beats no report.
                eprintln!("  {} {name}: {e:#}", Style::stdout().red("could not run"));
                failed |= 1;
                results.insert(
                    name.to_string(),
                    serde_json::json!({ "error": format!("{e:#}") }),
                );
            }
        }
    };

    if !cli.json {
        println!("{}", style.bold("1/5  the warm path"));
    }
    step(
        "warm",
        warm(
            cli,
            warm_n,
            true,
            Some(published::WARM_RATE),
            None,
            None,
            None,
        ),
    );

    if !cli.json {
        println!();
        println!("{}", style.bold("2/5  warm-exec"));
    }
    let exec_cmd: Vec<String> = ["sh", "-c", "cat"].iter().map(|s| s.to_string()).collect();
    step(
        "warm_exec",
        warm(
            cli,
            warm_n,
            true,
            Some(published::WARM_RATE),
            None,
            Some(&exec_cmd),
            None,
        ),
    );

    // The embedded-runtime shape, beside the function it is measured against:
    // the same instrument, the same host, the same rate, and a different
    // script with every request. The pair is the only honest way to say what
    // a pool costs — an absolute number would be about the machine.
    if !cli.json {
        println!();
        println!(
            "{}",
            style.bold("3/5  a runtime pool, a different script each request")
        );
    }
    step(
        "pool",
        warm(
            cli,
            warm_n,
            true,
            Some(published::WARM_RATE),
            None,
            None,
            Some(if quick { 50 } else { 1_000 }),
        ),
    );

    if !cli.json {
        println!();
        println!("{}", style.bold("4/5  a cold start"));
    }
    step("cold", cold(cli, cold_n, "python:3.12-slim", None));

    // The quota is lifted for the throughput run, and only for it. With the
    // spec's default `cpu = 1.0` a tenant is quota-bound long before the
    // runtime is — this host sustains about 430 requests a second and then
    // CFS stops it — so the number would be a measurement of the limit. The
    // latency runs above keep the default quota on purpose, because there
    // the limit is part of what is being reported.
    let load_cores = host.cores.clamp(1, 4) as f64;
    if !cli.json {
        println!();
        println!(
            "{}  {}",
            style.bold("5/5  sustained throughput"),
            style.dim(&format!(
                "with the tenant's CPU quota raised to {load_cores:.0} cores,                  so this measures the runtime and not the quota"
            ))
        );
    }
    step("load", load(cli, load_seconds, 4, Some(load_cores)));

    let after = Thermal::sample();
    let disturbance = Thermal::compare(&before, &after, &host);

    let comparison = compare_with_published(&results);

    if cli.json {
        output::json(&serde_json::json!({
            "host": host.to_json(),
            "results": results,
            "published": comparison.iter().map(Claim::to_json).collect::<Vec<_>>(),
            "disturbance": disturbance,
            "verdict": if !disturbance.is_empty() {
                "not a verdict: the host was throttled or busy"
            } else if failed != 0 {
                "a budget was missed"
            } else {
                "within budget"
            },
        }))?;
    } else {
        println!();
        println!("{}", style.bold("against the published numbers"));
        println!("  {:<34} {:>12} {:>12}", "", "published", "here");
        for claim in &comparison {
            let mark = if claim.close() {
                style.green("ok")
            } else {
                style.yellow("differs")
            };
            println!(
                "  {:<34} {:>12} {:>12}   {mark}",
                claim.what,
                format!("{:.2} {}", claim.published, claim.unit),
                match claim.measured {
                    Some(v) => format!("{v:.2} {}", claim.unit),
                    None => "—".to_string(),
                }
            );
        }
        println!();
        println!(
            "{}",
            style.dim(
                "  * the throughput run lifts the tenant's CPU quota; the latency runs keep it.\n\
                 \n  \
                 \"differs\" is not a failure. These were measured on the machines in\n  \
                 docs/book/25-performance.md; a different host produces different numbers, which\n  \
                 is why the one above is printed. What a budget says is in each section."
            )
        );

        println!();
        if !disturbance.is_empty() {
            println!("{} these numbers are not a verdict:", style.red("✗"));
            for reason in &disturbance {
                println!("    {}", style.yellow(reason));
            }
            println!(
                "{}",
                style.dim(
                    "  A run on a host that was throttled or busy measures the host. \n  \
                     Repeat it on an idle machine before drawing a conclusion."
                )
            );
        } else if failed != 0 {
            println!(
                "{} a budget was missed; see the sections above",
                style.red("✗")
            );
        } else {
            println!(
                "{} every budget met, on an undisturbed host",
                style.green("✓")
            );
        }
    }

    if !disturbance.is_empty() {
        return Ok(NOT_A_VERDICT);
    }
    Ok(failed)
}

/// One published number against what this host produced.
struct Claim {
    what: &'static str,
    unit: &'static str,
    published: f64,
    measured: Option<f64>,
    /// Whether a *larger* measurement is the good direction (throughput) or
    /// the bad one (latency).
    higher_is_better: bool,
}

impl Claim {
    /// Whether the measurement is near enough to the claim to call it
    /// reproduced on this host.
    fn close(&self) -> bool {
        let Some(measured) = self.measured else {
            return false;
        };
        let (a, b) = if self.higher_is_better {
            (self.published, measured)
        } else {
            (measured, self.published)
        };
        // Better than published is always fine; worse is compared against the
        // tolerance.
        a <= b * PUBLISHED_TOLERANCE
    }

    fn to_json(&self) -> serde_json::Value {
        serde_json::json!({
            "what": self.what,
            "unit": self.unit,
            "published": self.published,
            "measured": self.measured,
            "close": self.close(),
        })
    }
}

fn compare_with_published(results: &serde_json::Map<String, serde_json::Value>) -> Vec<Claim> {
    let number =
        |section: &str, key: &str| -> Option<f64> { results.get(section)?.get(key)?.as_f64() };
    let micros = |section: &str, key: &str| number(section, key).map(|v| v / 1000.0);

    vec![
        Claim {
            what: "warm request, median",
            unit: "ms",
            published: published::WARM_P50_MS,
            measured: micros("warm", "p50_us"),
            higher_is_better: false,
        },
        Claim {
            what: "warm request, 99th percentile",
            unit: "ms",
            published: published::WARM_P99_MS,
            measured: micros("warm", "p99_us"),
            higher_is_better: false,
        },
        Claim {
            what: "warm-exec request, median",
            unit: "ms",
            published: published::EXEC_P50_MS,
            measured: micros("warm_exec", "p50_us"),
            higher_is_better: false,
        },
        Claim {
            what: "pooled script request, median",
            unit: "ms",
            published: published::POOL_P50_MS,
            measured: micros("pool", "p50_us"),
            higher_is_better: false,
        },
        Claim {
            what: "pooled script request, 99th percentile",
            unit: "ms",
            published: published::POOL_P99_MS,
            measured: micros("pool", "p99_us"),
            higher_is_better: false,
        },
        Claim {
            what: "cold `zygo run`, median",
            unit: "ms",
            published: published::COLD_P50_MS,
            measured: number("cold", "p50_ms"),
            higher_is_better: false,
        },
        Claim {
            what: "throughput at concurrency 4 *",
            unit: "req/s",
            published: published::LOAD_PER_SECOND,
            measured: number("load", "requests_per_second"),
            higher_is_better: true,
        },
        Claim {
            what: "`zygo serve`, once",
            unit: "ms",
            published: published::SERVE_MS,
            measured: number("warm", "warm_ms"),
            higher_is_better: false,
        },
    ]
}

/// What the numbers are numbers *for*.
///
/// Printed above every run, because the single most common way a benchmark
/// misleads is by being quoted without the machine it came from. Everything
/// here is read, not attempted: it describes, it does not decide.
struct Host {
    kernel: String,
    arch: &'static str,
    cpu_model: String,
    cores: usize,
    memory_kb: Option<u64>,
    governor: Option<String>,
    max_mhz: Option<f64>,
    virtualised: Option<String>,
    data_root: String,
    version: &'static str,
}

impl Host {
    fn describe(paths: &zygo_core::paths::Paths) -> Host {
        Host {
            kernel: read_first_line("/proc/sys/kernel/osrelease")
                .unwrap_or_else(|| std::env::consts::OS.to_string()),
            arch: std::env::consts::ARCH,
            cpu_model: cpu_model().unwrap_or_else(|| "unknown".into()),
            cores: std::thread::available_parallelism()
                .map(std::num::NonZeroUsize::get)
                .unwrap_or(0),
            memory_kb: meminfo("MemTotal"),
            governor: read_first_line("/sys/devices/system/cpu/cpu0/cpufreq/scaling_governor"),
            max_mhz: read_first_line("/sys/devices/system/cpu/cpu0/cpufreq/cpuinfo_max_freq")
                .and_then(|v| v.parse::<f64>().ok())
                .map(|khz| khz / 1000.0),
            virtualised: virtualisation(),
            data_root: paths.data().display().to_string(),
            version: env!("CARGO_PKG_VERSION"),
        }
    }

    fn print(&self, style: &Style) {
        println!("{}", style.bold("the machine these numbers are about"));
        let line = |k: &str, v: String| println!("  {:<12} {v}", style.dim(k));
        line("zygo", format!("{} ({})", self.version, self.arch));
        line("kernel", self.kernel.clone());
        line(
            "cpu",
            format!(
                "{} × {}{}",
                self.cores,
                self.cpu_model,
                match self.max_mhz {
                    Some(mhz) => format!(", up to {mhz:.0} MHz"),
                    None => String::new(),
                }
            ),
        );
        if let Some(kb) = self.memory_kb {
            line("memory", format!("{:.1} GiB", kb as f64 / 1024.0 / 1024.0));
        }
        if let Some(g) = &self.governor {
            line("governor", g.clone());
        }
        if let Some(v) = &self.virtualised {
            line("virtual", v.clone());
        }
        line("data root", self.data_root.clone());
    }

    fn to_json(&self) -> serde_json::Value {
        serde_json::json!({
            "zygo": self.version,
            "arch": self.arch,
            "kernel": self.kernel,
            "cpu_model": self.cpu_model,
            "cores": self.cores,
            "memory_kb": self.memory_kb,
            "governor": self.governor,
            "max_mhz": self.max_mhz,
            "virtualised": self.virtualised,
            "data_root": self.data_root,
        })
    }
}

/// What the machine was doing to itself while the numbers were taken.
///
/// Two sources, because they catch different things and neither is everywhere:
/// the x86 per-core throttle counters in `/sys`, and the Raspberry Pi's
/// firmware flag, which is the only place an undervolted Pi says so. The load
/// average is not throttling at all, but a benchmark that shared the machine
/// with a compile is no more of a verdict than one that overheated.
struct Thermal {
    core_throttles: u64,
    pi_flags: Option<u32>,
    loadavg1: Option<f64>,
}

impl Thermal {
    fn sample() -> Thermal {
        Thermal {
            core_throttles: core_throttle_count(),
            pi_flags: pi_throttled_flags(),
            loadavg1: read_first_line("/proc/loadavg")
                .and_then(|l| l.split_whitespace().next()?.parse().ok()),
        }
    }

    /// Reasons these numbers are not a verdict. Empty is the good answer.
    fn compare(before: &Thermal, after: &Thermal, host: &Host) -> Vec<String> {
        let mut reasons = Vec::new();

        if after.core_throttles > before.core_throttles {
            reasons.push(format!(
                "the CPU throttled {} time(s) during the run (/sys/devices/system/cpu/*/thermal_throttle)",
                after.core_throttles - before.core_throttles
            ));
        }

        // The Pi's firmware word: bit 0 under-voltage now, bit 1 frequency
        // capped now, bit 2 throttled now, bit 3 soft temperature limit; the
        // same four at bits 16-19 mean "since boot". Only the live bits are
        // held against the run — a Pi that browned out last Tuesday is not
        // this measurement's problem.
        for (sample, when) in [(before, "before"), (after, "after")] {
            if let Some(flags) = sample.pi_flags
                && flags & 0xF != 0
            {
                reasons.push(format!(
                    "the firmware reports under-voltage or capped frequency {when} the run \
                     (vcgencmd get_throttled = {flags:#x})"
                ));
            }
        }

        // Half the cores busy with something else is enough to move a
        // percentile. Read before the run: afterwards it includes the run.
        if let Some(load) = before.loadavg1
            && host.cores > 0
            && load > host.cores as f64 / 2.0
        {
            reasons.push(format!(
                "the machine was already busy when the run started (load {load:.2} \
                 on {} cores)",
                host.cores
            ));
        }

        reasons
    }
}

/// The sum of every per-core and per-package throttle counter the kernel
/// keeps. x86 only; aarch64 has no equivalent, so the Pi's firmware flag is
/// what covers the other architecture Zygo is measured on.
fn core_throttle_count() -> u64 {
    let Ok(entries) = std::fs::read_dir("/sys/devices/system/cpu") else {
        return 0;
    };
    let mut total = 0;
    for entry in entries.flatten() {
        for name in ["core_throttle_count", "package_throttle_count"] {
            let path = entry.path().join("thermal_throttle").join(name);
            if let Ok(text) = std::fs::read_to_string(&path)
                && let Ok(n) = text.trim().parse::<u64>()
            {
                total += n;
            }
        }
    }
    total
}

/// `vcgencmd get_throttled`, where the tool exists.
///
/// It prints `throttled=0x50005`. Absent everywhere but a Raspberry Pi, and
/// absent there too unless `libraspberrypi-bin` is installed — which is why
/// this is `Option` rather than a zero.
fn pi_throttled_flags() -> Option<u32> {
    let output = std::process::Command::new("vcgencmd")
        .arg("get_throttled")
        .output()
        .ok()?;
    let text = String::from_utf8_lossy(&output.stdout);
    let value = text.trim().strip_prefix("throttled=")?;
    let digits = value.strip_prefix("0x").unwrap_or(value);
    u32::from_str_radix(digits, 16).ok()
}

fn read_first_line(path: &str) -> Option<String> {
    let text = std::fs::read_to_string(path).ok()?;
    Some(text.lines().next()?.trim().to_string())
}

fn meminfo(key: &str) -> Option<u64> {
    let text = std::fs::read_to_string("/proc/meminfo").ok()?;
    text.lines()
        .find(|l| l.starts_with(key))?
        .split_whitespace()
        .nth(1)?
        .parse()
        .ok()
}

/// What kind of processor this is, in the words of whichever file has them.
///
/// `/proc/cpuinfo` answers on x86 with `model name` and on a Raspberry Pi with
/// `Model`, and answers with neither inside a LinuxKit aarch64 VM — where the
/// device tree is the only place with a name. Three sources, because the line
/// this feeds is the difference between "these numbers are about a Pi" and
/// "these numbers are about an M1", and `unknown` is the one answer that helps
/// nobody.
fn cpu_model() -> Option<String> {
    if let Ok(text) = std::fs::read_to_string("/proc/cpuinfo") {
        for key in ["model name", "Model", "Hardware", "cpu model", "Processor"] {
            if let Some(line) = text
                .lines()
                .find(|l| l.trim_start().starts_with(key) && l.contains(':'))
                && let Some((_, value)) = line.split_once(':')
                && !value.trim().is_empty()
            {
                return Some(value.trim().to_string());
            }
        }
    }
    // The device tree, where there is one. The strings are NUL-terminated.
    for path in [
        "/sys/firmware/devicetree/base/model",
        "/proc/device-tree/model",
    ] {
        if let Ok(bytes) = std::fs::read(path) {
            let text = String::from_utf8_lossy(&bytes)
                .trim_end_matches('\0')
                .trim()
                .to_string();
            if !text.is_empty() {
                return Some(text);
            }
        }
    }
    None
}

/// What this is running inside, where the kernel says so.
///
/// `systemd-detect-virt` is the accurate answer and is not always installed;
/// the DMI product name is the fallback and is enough to tell a laptop from
/// Docker Desktop's LinuxKit VM, which is the distinction that matters for
/// every number here.
fn virtualisation() -> Option<String> {
    if let Ok(output) = std::process::Command::new("systemd-detect-virt").output()
        && output.status.success()
    {
        let what = String::from_utf8_lossy(&output.stdout).trim().to_string();
        if !what.is_empty() && what != "none" {
            return Some(what);
        }
        return None;
    }
    read_first_line("/sys/class/dmi/id/product_name")
}

/// The cold budget: a `run` with the image already in the store.
const COLD_BUDGET_MS: f64 = 50.0;

/// The acceptance criterion for throughput.
const LOAD_TARGET_PER_SECOND: f64 = 600.0;

/// `zygo bench cold` — build a sandbox, run a program, tear it down.
///
/// This is the cold budget, and the default command is the one it is written
/// about: starting a Python interpreter and exiting. The image must already be
/// in the store, because the budget is for the cached case — pulling is a
/// network measurement and belongs nowhere near this number.
fn cold(
    cli: &Cli,
    n: u32,
    image: &str,
    command: Option<&[String]>,
) -> anyhow::Result<(u8, serde_json::Value)> {
    use zygo_core::image::{Reference, Store};
    use zygo_core::sandbox::SandboxConfig;

    let paths = super::paths(cli);
    paths.ensure()?;
    let store = Store::new(paths.clone());
    let reference: Reference = image.parse()?;
    let entry = store.get(&reference).with_context(|| {
        format!("`{image}` is not in the store\n  → pull it first: zygo pull {image}")
    })?;

    let argv: Vec<String> = match command {
        Some(cmd) => cmd.to_vec(),
        None => ["python3", "-c", "pass"]
            .iter()
            .map(|s| s.to_string())
            .collect(),
    };

    let spec = Spec::default();
    let resolved = spec.resolve(
        None,
        &Layer {
            image: Some(image.to_string()),
            cmd: Some(argv.clone()),
            ..Default::default()
        },
        &ResolveOptions {
            one_shot: true,
            ..Default::default()
        },
    )?;

    let overlay = zygo_core::doctor::cached(&paths)
        .checks
        .iter()
        .any(|c| c.name == "overlayfs (userns)" && c.status == zygo_core::doctor::Status::Ok);
    let mount_points = zygo_core::sandbox::mount::required_mount_points(&resolved.mounts);
    let view = store.rootfs_view(&entry.layers, overlay, &mount_points)?;
    let backend = zygo_core::backend::for_isolation(resolved.isolation, &paths)?;

    let style = Style::stdout();
    eprintln!(
        "{} {image} {}  {}",
        style.dim("cold"),
        argv.join(" "),
        style.dim(if overlay { "overlayfs" } else { "flattened" })
    );

    let mut samples = Vec::with_capacity(n as usize);
    for i in 0..n {
        // A fresh root per iteration: reusing one would measure a warm page
        // cache for the mount points rather than what a real `run` does.
        let newroot = paths
            .tmp()
            .join(format!("bench-cold-{}-{i}", std::process::id()));
        std::fs::create_dir_all(&newroot)?;
        let config = SandboxConfig::from_resolved(&resolved, &view, &newroot, argv.clone(), &[]);

        let t0 = Instant::now();
        let mut sandbox = backend
            .start(&config)
            .with_context(|| format!("iteration {i} could not start"))?;
        let code = sandbox.wait()?;
        samples.push(t0.elapsed().as_secs_f64() * 1000.0);

        anyhow::ensure!(code == 0, "iteration {i} exited {code}");
        let _ = std::fs::remove_dir_all(&newroot);
    }

    samples.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));
    let q = |p: f64| percentile(&samples, p);
    let p50 = q(50.0);

    let json = serde_json::json!({
        "image": image,
        "command": argv,
        "runs": n,
        "p50_ms": p50,
        "p90_ms": q(90.0),
        "p99_ms": q(99.0),
        "max_ms": samples.last().copied().unwrap_or(0.0),
        "budget_ms": COLD_BUDGET_MS,
        "pass": p50 < COLD_BUDGET_MS,
        "rootfs": if overlay { "overlayfs" } else { "flattened" },
    });
    if !cli.json {
        println!("cold start over {n} runs, image already in the store");
        println!(
            "  p50 {:>7.1} ms   p90 {:>7.1}   p99 {:>7.1}   max {:>7.1}",
            p50,
            q(90.0),
            q(99.0),
            samples.last().copied().unwrap_or(0.0)
        );
        println!();
        let ok = p50 < COLD_BUDGET_MS;
        println!(
            "  p50 < {COLD_BUDGET_MS:.0} ms   {}",
            if ok {
                style.green("PASS")
            } else {
                style.red("FAIL")
            }
        );
        if !overlay {
            // Worth saying: a flattened rootfs is a different measurement, and
            // on this host it is the only one available.
            println!(
                "{}",
                style.dim(
                    "  note: this host has no unprivileged overlayfs, so the rootfs is\n  \
                     flattened — the same bind mount every run, which is the cheap case"
                )
            );
        }
    }
    let within_budget = p50 < COLD_BUDGET_MS;
    Ok((u8::from(!within_budget), json))
}

/// `zygo bench load` — sustained throughput through one warm function.
///
/// The acceptance criterion is ≥ 600 requests/s at a
/// concurrency of 4. Measuring it needs concurrent callers, and the interesting
/// number is not just the total: [`CallTiming::lock`](zygo_core::pool::CallTiming::lock) says how much of each
/// request was spent waiting for the connection, which is the difference
/// between "the machine is busy" and "the callers are queueing behind each
/// other".
fn load(
    cli: &Cli,
    seconds: u32,
    concurrency: u32,
    cpu: Option<f64>,
) -> anyhow::Result<(u8, serde_json::Value)> {
    use std::sync::Arc;
    use std::sync::atomic::{AtomicBool, Ordering};

    anyhow::ensure!(concurrency >= 1, "concurrency must be at least 1");

    let paths = super::paths(cli);
    let pool = Pool::new(PoolConfig::new(paths.clone()))?;

    let dir = tempfile::tempdir()?;
    let handler = dir.path().join("handler.py");
    std::fs::write(&handler, "def handler(event):\n    return None\n")?;

    let spec = Spec::default();
    let resolved = spec.resolve(
        None,
        &Layer {
            entry: Some(handler.clone()),
            cpu: cpu.map(zygo_core::spec::Cpu),
            concurrency: Some(concurrency),
            ..Default::default()
        },
        &ResolveOptions {
            one_shot: true,
            ..Default::default()
        },
    )?;

    let style = Style::stdout();
    let function = Arc::new(
        pool.serve(&resolved)
            .with_context(|| format!("could not warm `{}`", resolved.image))?,
    );
    eprintln!(
        "{} {} clients for {seconds}s",
        style.dim("load"),
        concurrency
    );

    for _ in 0..50 {
        function.call(serde_json::Value::Null)?;
    }

    let stop = Arc::new(AtomicBool::new(false));
    let cpu_before = function.cpu_accounting();
    let started = Instant::now();

    let workers: Vec<_> = (0..concurrency)
        .map(|_| {
            let (function, stop) = (Arc::clone(&function), Arc::clone(&stop));
            std::thread::spawn(move || {
                let mut samples = Vec::new();
                let mut lock_us = Vec::new();
                let mut failures = 0u64;
                let mut served = 0u64;
                while !stop.load(Ordering::Relaxed) {
                    served += 1;
                    let t0 = Instant::now();
                    match function.call_timed(serde_json::Value::Null, Duration::from_secs(30)) {
                        Ok((outcome, timing)) => {
                            samples.push(t0.elapsed().as_secs_f64() * 1e6);
                            lock_us.push(timing.lock.as_secs_f64() * 1e6);
                            if !outcome.succeeded() {
                                failures += 1;
                            }
                        }
                        Err(_) => failures += 1,
                    }
                }
                (samples, lock_us, failures, served)
            })
        })
        .collect();

    std::thread::sleep(Duration::from_secs(u64::from(seconds)));
    stop.store(true, Ordering::Relaxed);

    let mut samples = Vec::new();
    let mut lock_us = Vec::new();
    let mut failures = 0u64;
    // Per worker, because a total hides starvation completely: one client
    // taking every slot and three taking none looks exactly like four clients
    // sharing fairly.
    let mut per_worker = Vec::new();
    for w in workers {
        let (s, l, f, served) = w.join().map_err(|_| anyhow::anyhow!("a worker panicked"))?;
        samples.extend(s);
        lock_us.extend(l);
        failures += f;
        per_worker.push(served);
    }
    per_worker.sort_unstable();
    let elapsed = started.elapsed();
    let quota = match (cpu_before, function.cpu_accounting()) {
        (Some(before), Some(after)) => Some(after.since(&before)),
        _ => None,
    };

    let count = samples.len() as u32;
    let report = Report::of(&samples, elapsed, count, quota);
    lock_us.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));

    let met = report.per_second >= LOAD_TARGET_PER_SECOND;
    let mut json = report.to_json();
    json["concurrency"] = concurrency.into();
    json["failures"] = failures.into();
    json["target_per_second"] = LOAD_TARGET_PER_SECOND.into();
    json["meets_target"] = met.into();
    json["lock_p50_us"] = percentile(&lock_us, 50.0).into();
    json["lock_max_us"] = lock_us.last().copied().unwrap_or(0.0).into();
    json["requests_per_client"] = per_worker.clone().into();
    if !cli.json {
        println!(
            "sustained load: {concurrency} clients, {:.1}s, empty handler",
            elapsed.as_secs_f64()
        );
        println!(
            "  {:.0} requests/s   {count} requests   {failures} failed",
            report.per_second
        );
        println!(
            "  p50 {:>7.0} µs   p90 {:>7.0}   p99 {:>7.0}   max {:>8.0}",
            report.p50, report.p90, report.p99, report.max
        );
        println!();
        println!(
            "  >= {LOAD_TARGET_PER_SECOND:.0} requests/s   {}",
            if met {
                style.green("PASS")
            } else {
                style.red("FAIL")
            }
        );
        print_contention(&style, &lock_us, &report, concurrency);
        print_fairness(&style, &per_worker);
        report.print_quota(&style);
    }

    let _ = function.shutdown();
    Ok((u8::from(!met), json))
}

/// Show how the work was split between clients.
///
/// A throughput total says nothing about fairness, and fairness is where an
/// unfair lock shows up: one client taking nearly every slot while the others
/// wait reads as healthy throughput right up until you look at a percentile of
/// *their* latency.
fn print_fairness(style: &Style, per_worker: &[u64]) {
    if per_worker.len() < 2 {
        return;
    }
    let (min, max) = (per_worker[0], per_worker[per_worker.len() - 1]);
    let total: u64 = per_worker.iter().sum();
    let fair = total / per_worker.len() as u64;
    println!();
    println!("  requests per client: min {min}, max {max}, even would be {fair}");
    if min * 4 < max {
        println!(
            "{}",
            style.yellow(
                "  The work is not being shared. A warm function serialises on one
                   connection, and the lock guarding it is not fair, so a client that
                   releases and immediately re-acquires can starve the others for seconds
                   at a time. Until the agent can hold several forks at once, concurrency
                   above 1 buys nothing and costs fairness."
            )
        );
    }
}

/// Say how much of each request was spent waiting for the connection.
///
/// This is the number that says *why* a concurrency target is or is not met. A
/// warm function serialises on its own connection, because the wire protocol is
/// one request in flight per agent; adding callers therefore adds queueing
/// rather than throughput until the agent itself can hold several forks at once.
fn print_contention(style: &Style, lock_us: &[f64], report: &Report, concurrency: u32) {
    if lock_us.is_empty() {
        return;
    }
    let lock_p50 = percentile(lock_us, 50.0);
    let share = lock_p50 / report.p50.max(1.0) * 100.0;
    println!();
    println!(
        "  waiting for the connection: p50 {lock_p50:>6.0} µs   p99 {:>7.0}            max {:>9.0}   ({share:.0}% of p50)",
        percentile(lock_us, 99.0),
        lock_us.last().copied().unwrap_or(0.0)
    );

    // The tail, not the median, is where this hurts. Measured at concurrency 4:
    // the requests-per-client split was a reasonable 684 to 1166, while one
    // client still waited 2.17 s for the connection — so a percentile of the
    // pooled samples reports zero contention right up to the maximum.
    let lock_max = lock_us.last().copied().unwrap_or(0.0);
    if concurrency > 1 && lock_max > report.p50 * 20.0 {
        println!(
            "{}",
            style.yellow(
                "  One client waited far longer for the connection than a request takes.\n  \
                 A warm function serves one request at a time — the agent handles one\n  \
                 `EXEC` to completion before reading the next — and the lock guarding\n  \
                 that connection is not fair, so waiting is unbounded and badly skewed.\n  \
                 Until the agent can hold several forks at once, concurrency above 1\n  \
                 buys no throughput and costs a long latency tail."
            )
        );
    } else if concurrency > 1 && share > 30.0 {
        println!(
            "{}",
            style.yellow(
                "  Most of each request is spent queueing behind another one: the agent\n  \
                 handles one `EXEC` to completion before reading the next, so\n  \
                 `concurrency` bounds what the supervisor admits, not what the agent\n  \
                 can overlap."
            )
        );
    }
}

/// The gap between request starts for an offered load in requests/s.
///
/// `None` means unpaced, which drives the tenant to its own CPU quota — see
/// [`Report::saturated`].
fn request_interval(rate: Option<f64>) -> Option<Duration> {
    let rate = rate?;
    if !rate.is_finite() || rate <= 0.0 {
        return None;
    }
    Some(Duration::from_secs_f64(1.0 / rate))
}

struct Report {
    p50: f64,
    p90: f64,
    p99: f64,
    p999: f64,
    max: f64,
    mean: f64,
    per_second: f64,
    count: u32,
    elapsed: Duration,
    /// What the tenant's CPU quota did during the run, when it could be read.
    quota: Option<CpuAccounting>,
    /// The p50 budget this run is judged against: the agent's or warm-exec's.
    p50_budget: f64,
    /// The tail's budget. A pool's is `ROADMAP.md`'s 5 ms rather than the warm
    /// path's 10 ms, because that is the number Phase 1's exit is written in.
    p99_budget: f64,
    label: &'static str,
}

impl Report {
    fn of(samples: &[f64], elapsed: Duration, count: u32, quota: Option<CpuAccounting>) -> Report {
        let mut sorted = samples.to_vec();
        sorted.sort_by(|a, b| a.partial_cmp(b).expect("no NaN in a duration"));

        Report {
            p50: percentile(&sorted, 50.0),
            p90: percentile(&sorted, 90.0),
            p99: percentile(&sorted, 99.0),
            p999: percentile(&sorted, 99.9),
            max: sorted.last().copied().unwrap_or(0.0),
            mean: sorted.iter().sum::<f64>() / sorted.len().max(1) as f64,
            per_second: count as f64 / elapsed.as_secs_f64(),
            count,
            elapsed,
            quota,
            p50_budget: WARM_P50_BUDGET_US,
            p99_budget: WARM_P99_BUDGET_US,
            label: "warm request overhead",
        }
    }

    /// Whether the tenant spent the run stopped at its own CPU quota.
    ///
    /// When it did, the tail is the quota's enforcement period, not Zygo's
    /// cost, and the p99 measured here is not a number about this code.
    fn saturated(&self) -> bool {
        self.quota.is_some_and(|q| q.saturated())
    }

    /// Whether the p99 budget can be judged at all.
    fn p99_is_meaningful(&self) -> bool {
        !self.saturated()
    }

    /// `false` only for a budget that was both judged and missed.
    ///
    /// A saturated run is not a failure of this code, and reporting it as one
    /// would send anyone reading CI after the wrong thing — the same mistake
    /// the phase-0 report records twice under test methodology. It is not a
    /// pass either: the p99 simply was not measured, which the output says.
    fn within_budget(&self) -> bool {
        self.p50 < self.p50_budget && (!self.p99_is_meaningful() || self.p99 < self.p99_budget)
    }

    /// How much of the budget is left. The number worth watching: the first
    /// measured 6%, and the supervisor's own work still has to fit in it.
    fn headroom_percent(&self) -> f64 {
        (self.p50_budget - self.p50) / self.p50_budget * 100.0
    }

    fn print(&self, style: &Style) {
        println!("{} over {} requests, empty handler", self.label, self.count);
        println!(
            "  p50 {:>7.0} µs   p90 {:>7.0}   p99 {:>7.0}   p99.9 {:>8.0}   max {:>8.0}   mean {:>7.0}",
            self.p50, self.p90, self.p99, self.p999, self.max, self.mean
        );
        println!("  {:.0} requests/s, sequential", self.per_second);
        println!();

        let verdict = |ok: bool, text: &str| {
            if ok {
                style.green(text)
            } else {
                style.red(text)
            }
        };
        println!(
            "  p50 < {:.0} µs   {}",
            self.p50_budget,
            verdict(
                self.p50 < self.p50_budget,
                if self.p50 < self.p50_budget {
                    "PASS"
                } else {
                    "FAIL"
                }
            )
        );
        if self.p99_is_meaningful() {
            println!(
                "  p99 < {:.0} µs  {}",
                self.p99_budget,
                verdict(
                    self.p99 < self.p99_budget,
                    if self.p99 < self.p99_budget {
                        "PASS"
                    } else {
                        "FAIL"
                    }
                )
            );
        } else {
            println!(
                "  p99 < {:.0} µs  {}",
                self.p99_budget,
                style.yellow("NOT MEASURED — the tenant was at its CPU quota")
            );
        }

        let headroom = self.headroom_percent();
        let note = format!("  headroom at p50: {headroom:.0}%");
        // The supervisor's queue, timers and metrics have yet to be added to
        // this path; a thin margin now is a regression waiting to happen.
        println!(
            "{}",
            if headroom < 15.0 {
                style.yellow(&note)
            } else {
                note
            }
        );
        self.print_quota(style);
    }

    /// What the tenant's CPU quota did, and what it means for the numbers above.
    fn print_quota(&self, style: &Style) {
        let Some(q) = self.quota else {
            return;
        };
        println!();
        let demand = q.demand_cores(self.elapsed);
        let per_request = if self.count > 0 {
            q.usage_us as f64 / f64::from(self.count) / 1000.0
        } else {
            0.0
        };
        match q.quota_cores {
            Some(cores) => println!(
                "  cpu   {:.2} cores used of {:.2} quota ({:.0}%), {:.2} ms per request",
                demand,
                cores,
                demand / cores * 100.0,
                per_request
            ),
            None => println!(
                "  cpu   {demand:.2} cores used, no quota, {per_request:.2} ms per request"
            ),
        }
        if q.periods > 0 {
            println!(
                "        throttled in {} of {} periods, {:.0} ms stopped in total",
                q.throttled_periods,
                q.periods,
                q.throttled_us as f64 / 1000.0
            );
        }

        if !self.saturated() {
            return;
        }
        // The one thing a reader must not conclude is that Zygo is slow.
        let bound = q.period.as_secs_f64() * 1e6 / 2.0;
        println!();
        println!(
            "{}",
            style.yellow(
                "  This run saturated the tenant's own CPU quota, so the tail above is\n  \
                 the quota being enforced, not Zygo's cost. A throttled request waits out\n  \
                 the rest of the enforcement period:"
            )
        );
        println!(
            "        period {:.0} ms → an expected {:.0} µs of added latency per throttled request",
            q.period.as_secs_f64() * 1000.0,
            bound
        );
        let capacity = q.quota_cores.unwrap_or(1.0) * 1000.0 / per_request.max(0.01);
        println!(
            "        this tenant sustains about {capacity:.0} requests/s; \
             measure latency below that with --rate, or raise --cpu"
        );
    }

    fn to_json(&self) -> serde_json::Value {
        serde_json::json!({
            "requests": self.count,
            "p50_us": self.p50,
            "p90_us": self.p90,
            "p99_us": self.p99,
            "p999_us": self.p999,
            "max_us": self.max,
            "mean_us": self.mean,
            "requests_per_second": self.per_second,
            "budget": { "p50_us": self.p50_budget, "p99_us": self.p99_budget },
            "headroom_p50_percent": self.headroom_percent(),
            "pass": self.within_budget(),
            "p99_measured": self.p99_is_meaningful(),
            "cpu": self.quota.map(|q| serde_json::json!({
                "quota_cores": q.quota_cores,
                "period_us": q.period.as_micros() as u64,
                "used_cores": q.demand_cores(self.elapsed),
                "ms_per_request": q.usage_us as f64 / f64::from(self.count.max(1)) / 1000.0,
                "periods": q.periods,
                "throttled_periods": q.throttled_periods,
                "throttled_us": q.throttled_us,
                "saturated": q.saturated(),
            })),
        })
    }
}

/// Split the `run` phase into the handler and everything around it.
///
/// The child reports its own wall time, so the remainder is plumbing: the
/// child's start-up after `GO`, the result pipe, and its exit. A tail that
/// lives there is Zygo's; a tail in the handler is the handler's.
fn print_handler_share(phases: &[zygo_core::pool::CallTiming], handler_us: &[f64]) {
    if handler_us.len() != phases.len() {
        return;
    }
    let mut handler = handler_us.to_vec();
    let mut plumbing: Vec<f64> = phases
        .iter()
        .zip(handler_us)
        .map(|(t, h)| (t.run.as_secs_f64() * 1e6 - h).max(0.0))
        .collect();
    handler.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));
    plumbing.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));

    println!(
        "{:<24} p50 {:>7.0} µs   p99 {:>7.0}   max {:>8.0}",
        "    of which handler",
        percentile(&handler, 50.0),
        percentile(&handler, 99.0),
        handler.last().copied().unwrap_or(0.0)
    );
    println!(
        "{:<24} p50 {:>7.0} µs   p99 {:>7.0}   max {:>8.0}",
        "    of which plumbing",
        percentile(&plumbing, 50.0),
        percentile(&plumbing, 99.0),
        plumbing.last().copied().unwrap_or(0.0)
    );
}

/// Cost of a bare `fork()` + `waitpid()` on this machine, in microseconds.
///
/// Deliberately the smallest possible child: it exits immediately. Anything the
/// warm path does is on top of this.
#[cfg(unix)]
fn measure_fork_floor(iterations: usize) -> Vec<f64> {
    let mut samples = Vec::with_capacity(iterations);
    for _ in 0..iterations {
        let t0 = Instant::now();
        // SAFETY: the child does nothing but `_exit`, which is
        // async-signal-safe; the parent only waits for it.
        unsafe {
            match libc::fork() {
                0 => libc::_exit(0),
                -1 => return samples,
                pid => {
                    let mut status = 0;
                    libc::waitpid(pid, &mut status, 0);
                }
            }
        }
        samples.push(t0.elapsed().as_secs_f64() * 1e6);
    }
    samples
}

#[cfg(not(unix))]
fn measure_fork_floor(_iterations: usize) -> Vec<f64> {
    Vec::new()
}

/// Put the measurement next to what the machine can do at all.
fn print_floor(style: &Style, floor: &[f64], report: &Report) {
    if floor.is_empty() {
        return;
    }
    let mut sorted = floor.to_vec();
    sorted.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));

    let floor_p50 = percentile(&sorted, 50.0);
    let floor_p99 = percentile(&sorted, 99.0);
    println!();
    println!(
        "  {} bare fork+wait on this host: p50 {:.0} µs, p99 {:.0} µs",
        style.dim("floor"),
        floor_p50,
        floor_p99
    );

    // If the machine's own p99 is already most of the budget, the run says more
    // about the machine than about Zygo, and the number should say so.
    if floor_p99 > WARM_P99_BUDGET_US * 0.2 {
        println!(
            "{}",
            style.yellow(&format!(
                "  the host's own fork p99 is {:.0}% of the p99 budget — this measurement is \n                   dominated by the machine, not by the warm path. Re-run on an idle host.",
                floor_p99 / WARM_P99_BUDGET_US * 100.0
            ))
        );
    }
    let _ = report;
}

/// Where the time went, phase by phase.
///
/// A total percentile says a tail exists; only the breakdown says which phase
/// owns it. Three plausible explanations for one were measured and refuted
/// before this existed.
/// The share of the p99 spent creating and joining the request's own cgroup.
///
/// `admit` and `release` are the per-request cgroup and nothing else, so when
/// they own the tail the number above is about the per-request cgroup rather
/// than about the warm path. Returned as a fraction of the total p99.
fn cgroup_share_of_p99(phases: &[zygo_core::pool::CallTiming]) -> Option<f64> {
    if phases.is_empty() {
        return None;
    }
    let p99_of = |get: fn(&zygo_core::pool::CallTiming) -> Duration| {
        let mut v: Vec<f64> = phases.iter().map(|t| get(t).as_secs_f64() * 1e6).collect();
        v.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));
        percentile(&v, 99.0)
    };
    let total = p99_of(|t| t.lock + t.fork + t.admit + t.run + t.release);
    if total <= 0.0 {
        return None;
    }
    Some((p99_of(|t| t.admit) + p99_of(|t| t.release)) / total)
}

/// Say when the tail belongs to the per-request cgroup rather than to Zygo.
///
/// The same duty as the CPU-quota note: a reader must not conclude from a
/// failed budget that the runtime is slow, when what they measured is a
/// decision that is still open. `bench warm` on a default configuration is
/// the command the README points newcomers at, and while the per-request
/// cgroup's cost is still an open question it
/// prints `p99 FAIL` — so it has to say what the number is about, and how to
/// take the measurement without it.
fn print_cgroup_note(phases: &[zygo_core::pool::CallTiming], report: &Report, style: &Style) {
    let Some(share) = cgroup_share_of_p99(phases) else {
        return;
    };
    // Only when it owns the tail *and* the tail is what failed. A run inside
    // budget needs no excuse, and one that failed on p50 has another cause.
    if share < 0.5 || !report.p99_is_meaningful() || report.p99 < WARM_P99_BUDGET_US {
        return;
    }
    println!();
    println!(
        "{}",
        style.yellow(&format!(
            "  {:.0}% of the p99 above is `admit` and `release` — creating this request's\n  \
             own cgroup and removing it, not running the handler. That is the cost of\n  \
             per-request containment, not a property of the warm path.",
            share * 100.0
        ))
    );
    println!(
        "        measure without it: zygo bench warm --no-cgroup{}",
        match report.count {
            0 => String::new(),
            n => format!(" --n {n}"),
        }
    );
    // The usual cause on a newer kernel, and one a reader can remove: the
    // move into the request's cgroup waiting for the kernel (see
    // `zygo doctor`'s "cgroup moves" line).
    let mountinfo = std::fs::read_to_string("/proc/self/mountinfo").unwrap_or_default();
    if zygo_core::doctor::cgroup2_favors_moves(&mountinfo) == Some(false) {
        println!(
            "{}",
            style.yellow(
                "  cgroup2 here has no favordynmods, so on Linux 6.0+ moving a request into\n  \
                 its cgroup can wait several ms for the kernel. `zygo doctor` says how to\n  \
                 change that, and what it costs."
            )
        );
    }
}

fn print_phases(phases: &[zygo_core::pool::CallTiming]) {
    let micros = |d: Duration| d.as_secs_f64() * 1e6;
    /// One row of the breakdown: a label and the phase it reads.
    type Column = (&'static str, fn(&zygo_core::pool::CallTiming) -> Duration);

    let columns: [Column; 5] = [
        ("  lock    (contention)", |t| t.lock),
        ("  fork    (EXEC→FORKED)", |t| t.fork),
        ("  admit   (cgroup)", |t| t.admit),
        ("  run     (GO→DONE)", |t| t.run),
        ("  release (cgroup rm)", |t| t.release),
    ];

    println!();
    for (label, get) in columns {
        let mut values: Vec<f64> = phases.iter().map(|t| micros(get(t))).collect();
        values.sort_by(|a, b| a.partial_cmp(b).expect("no NaN"));
        println!(
            "{label:<24} p50 {:>7.0} µs   p99 {:>7.0}   max {:>8.0}",
            percentile(&values, 50.0),
            percentile(&values, 99.0),
            values.last().copied().unwrap_or(0.0)
        );
    }
}

fn percentile(sorted: &[f64], p: f64) -> f64 {
    if sorted.is_empty() {
        return 0.0;
    }
    let rank = (p / 100.0) * (sorted.len() - 1) as f64;
    let lo = rank.floor() as usize;
    let hi = rank.ceil() as usize;
    if lo == hi {
        sorted[lo]
    } else {
        sorted[lo] + (sorted[hi] - sorted[lo]) * (rank - lo as f64)
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    /// A run on a host with no readable cgroup: the latency is all there is.
    fn report(samples: &[f64]) -> Report {
        Report::of(samples, Duration::from_secs(1), samples.len() as u32, None)
    }

    /// A run whose tenant cgroup was read, throttled in `throttled` of
    /// `periods` enforcement periods.
    fn report_with_quota(samples: &[f64], periods: u64, throttled: u64) -> Report {
        Report::of(
            samples,
            Duration::from_secs(1),
            samples.len() as u32,
            Some(CpuAccounting {
                usage_us: 1_000_000,
                periods,
                throttled_periods: throttled,
                throttled_us: throttled * 50_000,
                quota_cores: Some(1.0),
                period: Duration::from_millis(100),
            }),
        )
    }

    /// `--rate` is a frequency; the loop needs a gap. Getting this the wrong
    /// way round does not fail, it just paces at 300 *seconds* per request —
    /// which is exactly what it did the first time.
    #[test]
    fn a_rate_becomes_the_gap_between_requests() {
        let gap = request_interval(Some(300.0)).expect("paced");
        assert!(
            (gap.as_secs_f64() - 1.0 / 300.0).abs() < 1e-9,
            "300 requests/s is 3.3 ms apart, not 300 s; got {gap:?}"
        );
        assert_eq!(request_interval(Some(1.0)), Some(Duration::from_secs(1)));
        assert_eq!(
            request_interval(Some(1000.0)),
            Some(Duration::from_millis(1))
        );
    }

    #[test]
    fn a_rate_that_cannot_be_paced_runs_unpaced() {
        assert_eq!(request_interval(None), None, "no --rate at all");
        assert_eq!(request_interval(Some(0.0)), None, "0/s would never finish");
        assert_eq!(request_interval(Some(-5.0)), None);
        assert_eq!(request_interval(Some(f64::NAN)), None);
        assert_eq!(
            request_interval(Some(f64::INFINITY)),
            None,
            "infinitely fast is just unpaced"
        );
    }

    #[test]
    fn percentiles_interpolate() {
        let sorted: Vec<f64> = (1..=100).map(|n| n as f64).collect();
        assert_eq!(percentile(&sorted, 0.0), 1.0);
        assert_eq!(percentile(&sorted, 100.0), 100.0);
        assert!((percentile(&sorted, 50.0) - 50.5).abs() < 0.01);
        assert_eq!(percentile(&[], 50.0), 0.0);
    }

    #[test]
    fn the_budget_is_the_design_documents() {
        // p50 < 2 ms, p99 < 10 ms.
        assert_eq!(WARM_P50_BUDGET_US, 2_000.0);
        assert_eq!(WARM_P99_BUDGET_US, 10_000.0);
    }

    #[test]
    fn a_run_inside_the_budget_passes_and_one_outside_does_not() {
        let fast: Vec<f64> = (0..1000).map(|i| 1000.0 + (i % 100) as f64).collect();
        assert!(report(&fast).within_budget());

        let slow: Vec<f64> = (0..1000).map(|_| 2500.0).collect();
        assert!(!report(&slow).within_budget());

        // p50 inside the budget but a long tail outside it still fails.
        let mut tailed: Vec<f64> = (0..1000).map(|_| 1000.0).collect();
        tailed.extend((0..50).map(|_| 50_000.0));
        assert!(!report(&tailed).within_budget(), "p99 was ignored");
    }

    /// The first measurement was 6%. If a change eats into that, the number should say
    /// so rather than the run merely still passing.
    #[test]
    fn headroom_is_reported_as_a_percentage_of_the_budget() {
        let r = report(&[1_880.0; 100]);
        assert!(
            (r.headroom_percent() - 6.0).abs() < 0.5,
            "{}",
            r.headroom_percent()
        );

        let r = report(&[1_000.0; 100]);
        assert!((r.headroom_percent() - 50.0).abs() < 0.5);
    }

    #[test]
    fn the_json_report_carries_the_budget_it_was_judged_against() {
        let v = report(&[1_500.0; 100]).to_json();
        assert_eq!(v["budget"]["p50_us"], 2_000.0);
        assert_eq!(v["pass"], true);
        assert!(v["headroom_p50_percent"].as_f64().unwrap() > 0.0);
    }

    #[test]
    fn exit_status_follows_the_verdict() {
        assert_eq!(u8::from(!report(&[1_000.0; 100]).within_budget()), 0);
        assert_eq!(u8::from(!report(&[9_000.0; 100]).within_budget()), 1);
    }

    // --- saturation ------------------------------------------------------
    //
    // The tail this guards was chased for a whole session as if it were a
    // defect in the runtime. It was the tenant's own CPU quota: a run with no
    // think time asks for slightly more than one core, so CFS stops it until
    // the next period. These tests hold the tool to reporting which of the two
    // it measured.

    /// The measured shape: p50 well inside the budget, a tail of half the CFS
    /// period, and the cgroup saying it was throttled in nearly every period.
    fn a_throttled_run() -> Report {
        let mut samples: Vec<f64> = (0..980).map(|_| 1_400.0).collect();
        samples.extend((0..20).map(|_| 48_000.0));
        report_with_quota(&samples, 28, 27)
    }

    #[test]
    fn a_tail_made_by_the_cpu_quota_is_not_reported_as_a_failure() {
        let r = a_throttled_run();
        assert!(r.saturated());
        assert!(
            r.p99 > WARM_P99_BUDGET_US,
            "the fixture should be over budget at p99"
        );
        assert!(
            !r.p99_is_meaningful(),
            "a saturated run cannot judge the p99 budget"
        );
        assert!(
            r.within_budget(),
            "the quota doing its job is not a failure of this code"
        );
        assert_eq!(u8::from(!r.within_budget()), 0, "and CI should not go red");
    }

    #[test]
    fn a_saturated_run_still_fails_on_p50() {
        // Saturation excuses the tail, not the median: a p50 over budget is a
        // real regression whatever the quota was doing.
        let r = report_with_quota(&[2_500.0; 1000], 28, 27);
        assert!(r.saturated());
        assert!(!r.within_budget());
    }

    #[test]
    fn an_unsaturated_run_is_judged_on_both_budgets() {
        let mut samples: Vec<f64> = (0..980).map(|_| 1_400.0).collect();
        samples.extend((0..20).map(|_| 48_000.0));
        let r = report_with_quota(&samples, 200, 1);
        assert!(!r.saturated(), "1 period in 200 is noise, not saturation");
        assert!(r.p99_is_meaningful());
        assert!(!r.within_budget(), "a real tail must still fail");
    }

    #[test]
    fn a_host_without_a_readable_cgroup_judges_both_budgets() {
        // macOS, or cgroup v1, or a run outside the hierarchy: no counters
        // means no excuse.
        let mut samples: Vec<f64> = (0..980).map(|_| 1_400.0).collect();
        samples.extend((0..20).map(|_| 48_000.0));
        let r = report(&samples);
        assert!(!r.saturated());
        assert!(!r.within_budget());
    }

    #[test]
    fn the_json_report_says_whether_the_p99_was_measured() {
        let v = a_throttled_run().to_json();
        assert_eq!(v["p99_measured"], false);
        assert_eq!(v["cpu"]["saturated"], true);
        assert_eq!(v["cpu"]["throttled_periods"], 27);
        assert_eq!(v["cpu"]["quota_cores"], 1.0);
        assert_eq!(v["cpu"]["period_us"], 100_000);
        // A consumer that only reads `pass` must not be told the p99 passed.
        assert!(v["p99_us"].as_f64().unwrap() > WARM_P99_BUDGET_US);

        let v = report(&[1_500.0; 100]).to_json();
        assert_eq!(v["p99_measured"], true);
        assert!(v["cpu"].is_null(), "no cgroup, no cpu section");
    }
}