yah-cloud 0.8.43

Declarative cloud substrate for yah-managed camps: .yah/cloud/ config schema, MachineProvider drivers (Hetzner + local containerd), cloud-init rendering, and the pond/mesofact reconcilers.
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
1451
1452
1453
1454
1455
1456
1457
1458
1459
1460
1461
1462
1463
1464
1465
1466
1467
1468
1469
1470
1471
1472
1473
1474
1475
1476
1477
1478
1479
1480
1481
1482
1483
1484
1485
1486
1487
1488
1489
1490
1491
1492
1493
1494
1495
1496
1497
1498
1499
1500
1501
1502
1503
1504
1505
1506
1507
1508
1509
1510
1511
1512
1513
1514
1515
1516
1517
1518
1519
1520
1521
1522
1523
1524
1525
1526
1527
1528
1529
1530
1531
1532
1533
1534
1535
1536
1537
1538
1539
1540
1541
1542
1543
1544
1545
1546
1547
1548
1549
1550
1551
1552
1553
1554
1555
1556
1557
1558
1559
1560
1561
1562
1563
1564
1565
1566
1567
1568
1569
1570
1571
1572
1573
1574
1575
1576
1577
1578
1579
1580
1581
1582
1583
1584
1585
1586
1587
1588
1589
1590
1591
1592
1593
1594
1595
1596
1597
1598
1599
1600
1601
1602
1603
1604
1605
1606
1607
1608
1609
1610
1611
1612
1613
1614
1615
1616
1617
1618
1619
1620
1621
1622
1623
1624
1625
1626
1627
1628
1629
1630
1631
1632
1633
1634
1635
1636
1637
1638
1639
1640
1641
1642
1643
1644
1645
1646
1647
1648
1649
1650
1651
1652
1653
1654
1655
1656
1657
1658
1659
1660
1661
1662
1663
1664
1665
1666
1667
1668
1669
1670
1671
1672
1673
1674
1675
1676
1677
1678
1679
1680
1681
1682
1683
1684
1685
1686
1687
1688
1689
1690
1691
1692
1693
1694
1695
1696
1697
1698
1699
1700
1701
1702
1703
1704
1705
1706
1707
1708
1709
1710
1711
1712
1713
1714
1715
1716
1717
1718
1719
1720
1721
1722
1723
1724
1725
1726
1727
1728
1729
1730
1731
1732
1733
1734
1735
1736
1737
1738
1739
1740
1741
1742
1743
1744
1745
1746
1747
1748
1749
1750
1751
1752
1753
1754
1755
1756
1757
1758
1759
1760
1761
1762
1763
1764
1765
1766
1767
1768
1769
1770
1771
1772
1773
1774
1775
1776
1777
1778
1779
1780
1781
1782
1783
1784
1785
1786
1787
1788
1789
1790
1791
1792
1793
1794
1795
1796
1797
1798
1799
1800
1801
1802
1803
1804
1805
1806
1807
1808
1809
1810
1811
1812
1813
1814
1815
1816
1817
1818
1819
1820
1821
1822
1823
1824
1825
1826
1827
1828
1829
1830
1831
1832
1833
1834
1835
1836
1837
1838
1839
1840
1841
1842
1843
1844
1845
1846
1847
1848
1849
1850
1851
1852
1853
1854
1855
1856
1857
1858
1859
1860
1861
1862
1863
1864
1865
1866
1867
1868
1869
1870
1871
1872
1873
1874
1875
1876
1877
1878
1879
1880
1881
1882
1883
1884
1885
1886
1887
1888
1889
1890
1891
1892
1893
1894
1895
1896
1897
1898
1899
1900
1901
1902
1903
1904
1905
1906
1907
1908
1909
1910
1911
1912
1913
1914
1915
1916
1917
1918
1919
1920
1921
1922
1923
1924
1925
1926
1927
1928
1929
1930
1931
1932
1933
1934
1935
1936
1937
1938
1939
1940
1941
1942
1943
1944
1945
1946
1947
1948
1949
1950
1951
1952
1953
1954
1955
1956
1957
1958
1959
1960
1961
1962
1963
1964
1965
1966
1967
1968
1969
1970
1971
1972
1973
1974
1975
1976
1977
1978
1979
1980
1981
1982
1983
1984
1985
1986
1987
1988
1989
1990
1991
1992
1993
1994
1995
1996
1997
1998
1999
2000
2001
2002
2003
2004
2005
2006
2007
2008
2009
2010
2011
2012
2013
2014
2015
2016
2017
2018
2019
2020
2021
2022
2023
2024
2025
2026
2027
2028
2029
2030
2031
2032
2033
2034
2035
2036
2037
2038
2039
2040
2041
2042
2043
2044
2045
2046
2047
2048
2049
2050
2051
2052
2053
2054
2055
2056
2057
2058
2059
2060
2061
2062
2063
2064
2065
2066
2067
2068
2069
2070
2071
2072
2073
2074
2075
2076
2077
2078
2079
2080
2081
2082
2083
2084
2085
2086
2087
2088
2089
2090
2091
2092
2093
2094
2095
2096
2097
2098
2099
2100
2101
2102
2103
2104
2105
2106
2107
2108
2109
2110
2111
2112
2113
2114
2115
2116
2117
2118
2119
2120
2121
2122
2123
2124
2125
2126
2127
2128
2129
2130
2131
2132
2133
2134
2135
2136
2137
2138
2139
2140
2141
2142
2143
2144
2145
2146
2147
2148
2149
2150
2151
2152
2153
2154
2155
2156
2157
2158
2159
2160
2161
2162
2163
2164
2165
2166
2167
2168
2169
2170
2171
2172
2173
2174
2175
2176
2177
2178
2179
2180
2181
2182
2183
2184
2185
2186
2187
2188
2189
2190
2191
2192
2193
2194
2195
2196
2197
2198
2199
2200
2201
2202
2203
2204
2205
2206
2207
2208
2209
2210
2211
2212
2213
2214
2215
2216
2217
2218
2219
2220
2221
2222
2223
2224
2225
2226
2227
2228
2229
2230
2231
2232
2233
2234
2235
2236
2237
2238
2239
2240
2241
2242
2243
2244
2245
2246
2247
2248
2249
2250
2251
2252
2253
2254
2255
2256
2257
2258
2259
2260
2261
2262
2263
2264
2265
2266
2267
2268
2269
2270
2271
2272
2273
2274
2275
2276
2277
2278
2279
2280
2281
2282
2283
2284
2285
2286
2287
2288
2289
2290
2291
2292
2293
2294
2295
2296
2297
2298
2299
2300
2301
2302
2303
2304
2305
2306
2307
2308
2309
2310
2311
2312
2313
2314
2315
2316
2317
2318
2319
2320
2321
2322
2323
2324
2325
2326
2327
2328
2329
2330
2331
2332
2333
2334
2335
2336
2337
2338
2339
2340
2341
2342
2343
2344
2345
2346
2347
2348
2349
2350
2351
2352
2353
2354
2355
2356
2357
2358
2359
2360
2361
2362
2363
2364
2365
2366
2367
2368
2369
2370
2371
2372
2373
2374
2375
2376
2377
2378
2379
2380
2381
2382
2383
2384
2385
2386
2387
2388
2389
2390
2391
2392
2393
2394
2395
2396
2397
2398
2399
2400
2401
2402
2403
2404
2405
2406
2407
2408
2409
2410
2411
//! R850 — what the declared topology does when a node dies, and whether it
//! fits on the boxes it is declared against.
//!
//! # The question this module exists to answer
//!
//! From the noisetable camp, 2026-09-02: *"I have a singleton process with
//! three in-process turso DBs on one node. I kill that node at the hardware
//! level. What happens?"*
//!
//! Every fact needed to answer that is already in `.yah/` — the archetype, the
//! volume sources, the replica count, the restart policy, the node's taints and
//! `[allocatable]`, the sovereign role. What was missing was anywhere they were
//! read *together*, so the honest answer required reading yubaba's source. This
//! module is that reading, done once, as a pure function.
//!
//! # Pure, declaration-only, and read-only
//!
//! [`analyze`] takes a [`CloudConfig`] and returns a [`Topology`]. No network,
//! no credentials, no filesystem beyond the load that already happened, and
//! nothing here changes runtime behaviour — the same contract
//! [`crate::migrate::plan_migration`] holds, and for the same reason: an
//! answer you can only get by probing a live fleet is an answer you cannot get
//! *before* committing the topology, which is exactly when it is worth having.
//!
//! The cost of that purity is that placement here is **projected**, not
//! observed: [`Placement`] is what the admission seam
//! ([`CloudConfig::admit_workload_candidates`]) would decide today, not where
//! containers are running right now. Where the two can differ is stated on
//! [`Placement`] itself.
//!
//! # What it deliberately does not model
//!
//! - **Liveness.** A node declared here may be off. Admission has no liveness
//!   input by design (see `admit_workload_candidates`), and neither does this.
//! - **Correlated failure.** "Node X dies" is one node. Losing a rack, a
//!   region, or the upstream the whole camp NATs through is a different
//!   question with a different answer, and pretending a per-node walk covers
//!   it would be worse than not asking.
//! - **The blast radius of the control plane itself.** Losing the box that
//!   hosts the operator bridge is modelled only as far as
//!   [`QuorumEffect`] goes.
//!
//! @arch:see(.yah/docs/working/W305-sovereign-groups-environments-edges.md)
//!
//! @yah:ticket(R850-F1, "Hydrate-on-place: act on the declared durability tier at runtime, with fencing")
//! @yah:status(review)
//! @yah:at(2026-09-10T08:08:10Z)
//! @yah:assignee(agent:bundle-anthropic-ashguard)
//! @yah:phase(P4b)
//! @yah:parent(R850)
//! @yah:next("R850-P4 landed the DECLARATION half: `yah.durability.{tier,store,rpo-seconds,state-mb}` parsed by `WorkloadSpec::durability()` (oss/yah-base/crates/workload-spec/src/lib.rs), hard-validated in `validate::shape`, and read by `cloud::topology`. Declaring a tier still causes NO backup and NO restore. This ticket is the runtime half.")
//! @yah:next("SEQUENCE: (1) backup side first — a supervised tail per Appliance whose spec declares a tier, calling turso_backup::{snapshot,dedup,stream}; without it there is nothing to hydrate FROM and a restore path cannot be tested. (2) hydrate-on-place: before kamaji starts a container whose named volume is EMPTY at /var/lib/yah/kamaji/volumes/<name>, restore from the declared store. Empty-vs-populated is the trigger, so a normal restart never re-hydrates.")
//! @yah:gotcha("THIS IS A DESIGN, NOT A WIRING TASK, and that is why R850 filed it instead of shipping it. Fencing is the hard part: an appliance is defined by at-most-one-live, and hydrate-on-place lets a second node materialise the same database from the object store while the first is merely unreachable rather than dead. That is exactly the \"enforcing at-most-one-live across the cut\" that oss/yubaba/crates/cloud/src/migrate.rs's module header deferred to its own relay. turso-backup already has two-level fencing in stream.rs — read it before inventing one.")
//! @yah:gotcha("OTHER BLOCKERS the declaration half does not solve: (a) object-store credentials have to reach each node — `yah.durability.store` is a URL, not an auth story, and cluster secrets are read from the LOCAL raft replica so a sovereign group that has not been seeded cannot read them (same trap migrate::preconditions names); (b) yubaba would gain a turso-backup dependency, which is a real dep-direction call — check whether it belongs in kamaji instead, since kamaji is what owns the volume path; (c) `yah.durability.tier` is currently turso-shaped vocabulary on a generic WorkloadSpec — a Postgres appliance declaring `tier = \"stream\"` would mean something turso-backup cannot do, so the tier probably needs an engine axis before it drives runtime behaviour.")
//! @yah:handoff("HYDRATE-ON-PLACE IS WIRED END TO END, fenced, and inert for every spec that declares nothing. Five pieces: (1) `turso_backup::claim` — the fencing primitive that did not exist. `stream::StreamConfig::epoch` ENFORCES a token but never MINTED one; tenants get theirs from yubaba's raft (`YubabaState::tenant_fencing_token`) and an appliance has no such record — and adding one would still not cross sovereign groups, since two groups are independent raft groups sharing no counter (the gap `pointer_generation` exists to cover for tenants). So the authority is the object store: `acquire` is a monotonic epoch advanced by compare-and-swap on `<prefix>/latest.owner-claim`, `assert_holds` re-verifies. Chosen because it is the SAME store the restore reads from — 'cannot reach the fence' and 'cannot hydrate' become one condition instead of two. It is deliberately NOT a lease: no TTL, no clock. Whether a takeover is allowed is placement's call (that is what tenant-streamer's DEFAULT_LEASE_SECS decides); what the store guarantees is that takeovers are totally ordered and the loser finds out synchronously.")
//! @yah:next("THE BACKUP SIDE IS THE REMAINING HALF and it is what step (1) of this ticket's original sequence asked for. Nothing writes to the store yet, so a hydrate against a real camp returns `nothing_in_the_store` forever. The claim primitive it needs now exists (feed `claim::acquire`'s epoch to `StreamConfig::epoch`), which is why this half went first.")
//! @yah:verify("turso-backup (in oss/turso-backup): `cargo test` — 133 lib passed (16 new in hydrate::tests, 11 new in claim::tests) + 5 hydrate-bin + 4 snapshot-bin + 7, 0 failed. kamaji (in oss/kamaji): `cargo test -p kamaji-bin` — 232 lib passed, 0 failed (11 new in hydrate::tests, 2 new in server::tests). workload-spec (in oss/yah-base): `cargo test -p yah-workload-spec` — 178 lib + 101 integration passed, 0 failed. Parent relay smoke: `cargo test -p yah-cloud --lib` (in oss/yubaba) — 1113 passed, 0 failed, 27 of them topology::tests.")
//! @yah:handoff("(2) `turso_backup::hydrate` — the decision plus the execution. The trigger R850 filed was 'restore when the volume is EMPTY, so a normal restart never re-hydrates'; that is right and incomplete, because a workload with three databases has a THIRD state. `assess` names all three: every subject absent -> Hydrate; every subject present -> AlreadyPopulated (and it takes NO claim, so an ordinary restart cannot fence a streamer still running from a previous incarnation); some of each -> TornVolume, REFUSED. Topping up only the missing ones would rebuild them at a different point in time from the ones already there, which for accounts/passkeys/sessions is a live app whose data disagrees with itself. Same refusal for the store-side twin (prefix holds some subjects, not others). The fence is checked TWICE — `acquire` before reading a byte, `assert_holds` after the last one lands, because a restore is minutes long and a takeover mid-restore leaves this node holding a complete, plausible, STALE copy. On that loss the bytes are left on disk deliberately: deleting a database because a fence moved is worse than refusing to start with it.")
//! @yah:handoff("(3) DEP DIRECTION RESOLVED — gotcha (b). Neither yubaba nor kamaji links turso-backup. New bin `turso-backup-hydrate` (oss/turso-backup/src/bin/hydrate.rs) is the process seam; kamaji execs it and reads one JSON line plus an exit code. Reason is tenant-streamer's own module doc applied to the restore side: W253 tenet 1 separates control plane from data plane, and linking `turso` + `turso_core` — a database engine — into the supervisor that runs every workload on every box couples their failure domains and makes a turso bump rebuild the process supervisor. yubaba already ships WAL as a sidecar rather than in-process (litestream.rs, then tenant-streamer); this is that shape. Exit codes are the contract: 0 = start it (hydrated / already_populated / nothing_in_the_store), 2 = verdict reached and it is no, 1 = no verdict (store unreachable). 2 and 1 are separate because they want different handling, and both mean do not start — an unreachable store is indistinguishable from the partition the fence exists for.")
//! @yah:handoff("(4) ENGINE + SUBJECT AXES — gotcha (c) closed. `yah.durability.engine` (turso; anything else is a hard UnknownEngine) and `yah.durability.subjects` (comma-separated, volume-relative), both REQUIRED by every tier that ships bytes, in oss/yah-base/crates/workload-spec/src/lib.rs. Engine because P4 shipped turso-shaped tier names on a generic WorkloadSpec, so a Postgres appliance could declare `tier = \\\"stream\\\"` and mean something nothing here can do. Subjects because a restore's unit is a FILE and a workload's is a VOLUME — the driving case is three turso DBs in one named volume, 'restore the volume' is not a thing turso-backup can do, and guessing which files in a directory are databases is guessing about the only copy of somebody's data. Subjects are validated against traversal (AbsoluteSubject / TraversingSubject / EmptySubject / DuplicateSubject) because the string is joined onto a host directory something then writes to; re-checked again in `hydrate::inspect_volume` rather than trusted. `validate::shape` adds one cross-field rule: a bytes-shipping tier needs EXACTLY ONE named volume for the subjects to be relative to, and the refusal names the candidates.")
//! @yah:handoff("(5) KAMAJI HOOK — oss/kamaji/crates/kamaji-bin/src/hydrate.rs + the call in `deploy_container` (server.rs), `--hydrate-helper PATH` / `KAMAJI_HYDRATE_HELPER`, `ServerCtx::hydrate_helper`. Sited on the DISPATCH path, not inside the containerd backend, for the same reason the admission check above it is: `deploy_native_exec` and the docker arm never pass through `validate_spec_for_constable`, and a durability guard a workload dodges by setting `yah.exec = native` is not a guard. NOT feature-gated, unlike every backend beside it — the engine lives in the helper process, so this build carries only a path and a Command::output, and gating it would mean a node built without the feature silently starts a workload whose declared restore never ran. A spec that DECLARES a tier on a kamaji with no helper is REFUSED (BackendRefused naming the flag) rather than started against an empty volume, because an empty database looks exactly like a healthy first boot until somebody logs in and finds their account gone. Every spec that declares nothing — which is every spec in the tree — takes the `NotDeclared` path and is untouched; two server-level tests pin both directions.")
//! @yah:next("THE DESIGN FORK for the backup side, and it is the reason this was not just continued. `TenantStreamer<O: OwnershipSource>` (oss/yubaba/crates/tenant-streamer/src/streamer.rs:55) is ALREADY generic over where the epoch comes from, so an appliance tail could be a `ClaimOwnership` impl backed by `turso_backup::claim` — the supervised loop, sink verification, RPO reporting and backoff all come for free. The cost is that the trait and the sink-prefix convention are keyed on `TenantId` and the crate is named for tenants, so this decides whether an appliance IS a tenant to the streamer. (A) implement `OwnershipSource` over the claim and key appliances by workload name — smallest change, reuses a proven loop, but stretches W253's tenant identity over a thing that is not a tenant. (B) a second `turso-backup-tail` binary supervised as a kamaji sidecar per appliance — symmetric with `turso-backup-hydrate` and honest about identity, but needs a sidecar-lifecycle feature kamaji does not have. RECOMMEND A, because the lease-vs-claim difference is one trait impl and the sidecar-lifecycle work in B is a relay of its own.")
//! @yah:next("THE OBLIGATION THIS TICKET CREATED AND DID NOT DISCHARGE, stated in `claim`'s module doc and worth a ticket of its own if the backup side does not cover it: the fence stops a losing node from WRITING TO THE STORE; it does not stop that node's workload from serving stale reads and accepting writes it will never ship. `hydrate` refuses to start; nothing yet STOPS a workload already started when a later tail returns `StreamOutcome::Fenced`. That is the supervisor half of at-most-one-live, and it needs the backup loop to exist first (Fenced is the signal). Whichever fork above is taken must wire Fenced -> kamaji Stop.")
//! @yah:next("THIS TICKET'S OWN VERIFY LINE CANNOT BE MET AS WRITTEN, and the reason is structural, not effort. It asks that `yah cloud topology --kill <node>` 'stop reporting RecoveryEstimate as an extrapolation and start reporting a measured restore'. `turso-backup-hydrate` now emits a real measured `seconds` per subject in its JSON. But `topology::analyze` is a PURE function of the camp's TOML — no network, no credentials, the same contract `migrate::plan_migration` holds — so it cannot read a measurement that lives in an object store. Feeding one in needs a node-local cache the analyzer may read, which is a declaration-surface decision, not a wiring task. Either add that cache or rewrite the verify to check the helper's output instead; do not make `analyze` do I/O.")
//! @yah:gotcha("CREDENTIALS — gotcha (a) is only PARTLY answered. `yah.durability.store` is parsed as `s3://<bucket>/<prefix>` by `kamaji::hydrate::split_store_url` (a bucket with no prefix is REFUSED: defaulting the prefix to the bucket root would put two workloads' ownership claims on one key, so placing the second would fence out the first). Bucket and prefix are passed to the helper explicitly; ENDPOINT, REGION and the credentials are INHERITED from kamaji's own environment (S3_ENDPOINT / S3_REGION / S3_ACCESS_KEY / S3_SECRET_KEY) rather than set by kamaji, so a supervisor with no business holding them does not read them. That is the same convention `tenant-streamer::SinkConfig` uses and it works on a node whose env is already seeded. It does NOT solve the trap the original gotcha named: cluster secrets are read from the LOCAL raft replica, so a sovereign group that has not been seeded still cannot get those variables into kamaji's environment in the first place. Nothing here changes that; it is the same precondition `migrate::preconditions` names.")
//! @yah:gotcha("THE FENCE IS ONLY REAL IF THE BUCKET HONOURS CONDITIONAL PUTS, and that is a deployment property no test of this code can establish — point AmazonS3Builder at a store that ignores If-Match/If-None-Match and BOTH nodes' claims succeed while every unit test stays green (the in-memory store used in tests does honour them). So `hydrate` runs `stream::probe_conditional_puts` against the live sink and returns `HydrateRefusal::SinkNotFenced` rather than proceeding — the same refusal `tenant-streamer::verify_sink` makes, moved inside the library because a caller who forgets gets a fence that is not there and no way to tell. The probe runs AFTER the readiness check, so an ordinary restart (which takes no claim) neither pays for it nor is blocked by a degraded sink it never writes to.")
//! @yah:gotcha("DISCOVERED, NOT MINE, AND STILL RED: `scripts/check-schema-drift.sh` fails on an uncommitted regeneration of .yah/schema/{workload,machine}.toml.schema.json. Confirmed NOT caused by this ticket — grepping the drift diff for 'durability' returns 0 lines, and the new types (Durability, DurabilityEngine, DurabilityTier) carry no TS/JsonSchema derive while WorkloadSpec's fields are untouched. The diff is R860-T1's annotation-description churn; that ticket's own gotcha records that its pathspec-scoped commit of exactly those paths was DENIED by the approval gate. `scripts/check-workload-spec-ts.sh` is green. Nothing to regenerate here — this needs the commit R860-T1 asked for.")
//! @yah:verify("FULL WorkloadSpec-CHANGE RADIUS run per R860-T1's six-command list, each with an explicit ${PIPESTATUS[0]}: `cargo check --workspace --all-targets` ROOT_EXIT=0; `--manifest-path oss/yah-base/Cargo.toml --all-targets` YAHBASE_EXIT=0; `--manifest-path oss/yubaba/Cargo.toml --all-targets` YUBABA_EXIT=0; `--manifest-path oss/kamaji/Cargo.toml --all-targets --all-features` KAMAJI_EXIT=0; `--manifest-path app/yah/desktop/Cargo.toml --no-default-features` DESKTOP_EXIT=0. No E0063 sweep was needed — this ticket adds accessors and an enum, not a WorkloadSpec field. Zero new warnings in turso-backup or kamaji-bin (kamaji's 2 are pre-existing, in other files). `scripts/check-workload-spec-ts.sh`: ok, index.ts in sync.")
//! @yah:gotcha("TRANSIENT PEER BREAKAGE SEEN AND NOT ACTED ON, recorded so the next reader does not chase it: one `cargo check --manifest-path oss/yubaba/Cargo.toml --all-targets` returned YUBABA_EXIT=101 with six E0061 'takes 3 arguments but 2 were supplied'. The immediate re-run was clean with no edit from me. @Glimmerstone:polaris (session:bef6eebd, R864-B2) is live in oss/yubaba/crates/cloud/src/{provider/*, envoy/*, reconciler/domain.rs} and the build-input watcher named reconciler/domain.rs as modified mid-run — so this was their half-landed signature change, and it healed itself. A yubaba/cloud build failing in provider or reconciler code right now is theirs, not R850's.")
//! @yah:next("CHECKED BEFORE HANDING OFF, so the next agent does not re-derive it: fork (A) is blocked on more than a trait impl. `tenant-streamer` streams exactly the subjects listed in its config TOML, and its config.rs says so explicitly — 'Placement. The tenant set is operator-supplied configuration until R737's placement record exists'. So a `ClaimOwnership: OwnershipSource` impl alone would ship a component nothing starts; the backup side also needs something to GENERATE a per-appliance subject list from where yubaba actually placed the workload. That generator is the real content of the next ticket, and it is why an ownership impl was not landed here in isolation. Take the fork decision and the config-source decision together.")
//! @yah:handoff("FILES: new oss/turso-backup/src/{claim.rs, hydrate.rs, bin/hydrate.rs} + Cargo.toml [[bin]] + two lib.rs mod lines; new oss/kamaji/crates/kamaji-bin/src/hydrate.rs + lib.rs mod line + server.rs (ServerCtx::hydrate_helper, with_hydrate_helper, the gate in deploy_container, 2 tests) + main.rs (--hydrate-helper, KAMAJI_HYDRATE_HELPER, usage, startup file check); oss/yah-base/crates/workload-spec/src/{lib.rs, validate.rs} + tests/shape_fixtures.rs; oss/yubaba/crates/cloud/src/topology.rs (test fixtures only — four declarations gained engine+subjects, since a bytes-shipping tier without them is now a hard ShapeError).")
//! @yah:handoff("Tree anchor at handoff: f086233d6b092de2f32cafad5e0010494078269c — the shared tree as I left it. Diff against it (`git diff f086233d6b092de2f32cafad5e0010494078269c..HEAD`) to see what landed under you, and quote this SHA rather than 'HEAD' in any revert/restore instruction.")
//! @yah:gotcha("MEASURED: A SEPARATE PROCESS CANNOT SNAPSHOT A LIVE TURSO DATABASE, WHICH BREAKS BOTH ARMS OF THE BACKUP-SIDE FORK ABOVE AS WRITTEN. Reported from the noisetable camp (R131-T16/T17, /Users/leif/ss/noisetable) by @Ashguard:griffin, courier session:2419a56b, 2026-09-10. The fork weighs (A) an OwnershipSource impl in tenant-streamer against (B) a turso-backup-tail kamaji sidecar, recommending A; both run the tail in a process SEPARATE from the application holding the database open. That does not work. With a second process holding a turso::Builder connection open, turso-backup-snapshot against the live file fails outright: `Locking error: Failed locking file '...account.db'. File is locked by another process`. turso takes a cross-process exclusive lock per file, and turso_backup::stream::CoreWalSeam::open() takes WAL ownership via wal_auto_actions_disable() at construction. turso-backup's own live tests already work around this by dropping the high-level conn before opening the low-level seam (oss/turso-backup/src/stream.rs, R005-F3 handoff). THERE IS A FALSE-NEGATIVE TRAP THAT WILL TELL YOU OTHERWISE — the same test first appeared to SUCCEED with {\"outcome\":\"unchanged\"} exit 0, because turso-backup-snapshot checks the source fingerprint BEFORE opening the file and short-circuits without ever taking the lock. ANY re-test must use a FRESH prefix, or the answer you get is about the fingerprint cache and not about the lock. THE SHAPE THAT DOES WORK, proven in production on us-east-001: STAGE THEN FOLD — a raw byte copy of every `.db` AND `.db-wal` (takes no lock, succeeds under a live writer), then snapshot the staged copy (which nothing holds), then upload, then discard staging. Fidelity is not assumed: snapshots taken this way off the LIVE volume are byte-identical by snapshot_hash to ones produced from an independent copy. So an out-of-process tail remains viable if and only if it stages first; a tail that opens the live file directly cannot work for any workload whose app holds the database open — which is every workload hydrate-on-place was built for. Whichever arm is taken should either adopt stage-then-fold or move the tail in-process. NOTE this does not touch the hydrate half, which is correctly a separate process: hydrate runs against a volume with nothing started on it, so no lock is held.")
//! @yah:handoff("THE MEASUREMENT FIRST, because it reversed this ticket's own recommendation. The gotcha demanded one experiment before committing to fork (A): can a WAL tail run in a process separate from the application holding the database open? New `oss/turso-backup/examples/appliance_tail_probe.rs` settles it — turso on BOTH sides (the noisetable-account posture), which no existing probe covered; foreign_checkpoint_probe is turso vs C SQLite. Verdict: A SEPARATE-PROCESS TAIL WORKS. A1 `CoreWalSeam::open` refused (whole-file fcntl lock, exactly as reported). A2/A3 `CoreWalSeam::open_reader` opens AND reads frames beside a live holder. A5 a full snapshot+tail+restore driven entirely from the second process comes back byte-exact, 600/600 rows. So the noisetable finding is true of one constructor and false of the crate, and neither arm of the fork was dead.")
//! @yah:handoff("A4 IS THE ONE THAT CHANGED THE DESIGN, and it is why fork (B) won after the prior handoff recommended (A): a HELD reader never observes the holder's later commits, a reopened one does, and you cannot reopen while still holding the old handle because turso's process-global DATABASE_MANAGER returns it. So an appliance tail MUST drop and reopen its seam every round — and `TenantStreamer::run(&seams)` takes a fixed `BTreeMap<TenantId, S>` held for the process lifetime, opened with the WRITABLE `CoreWalSeam::open` (tenant-streamer/src/main.rs:138). Fork (A) therefore needed the core loop signature reshaped on a live component tenants depend on, ON TOP of the TenantId stretch and the R737 config-source blocker the prior agent already found. Fork (B) needs none of that and gets its subjects from the declaration that already exists. Measured, not aesthetic.")
//! @yah:handoff("THE BACKUP SIDE SHIPPED, all three tiers. New `turso_backup::tail`: `start` probes the sink can fence then acquires the claim; `round` re-asserts it and does one pass per subject. Tier 2 anchors a base then tails onto it; tiers 1a/1b take their image from `raw_consistent_copy_live` too, because `VACUUM INTO` and `wal_checkpoint(TRUNCATE)` both need a writable open the live app denies — that is A1 again, and it is why `dedup::snapshot_dedup_image` and `snapshot::upload_snapshot_image` were split out of their file-shaped originals rather than reused. `round` re-reads the claim EVERY pass and that is not redundant with the sink fence: only tier 2 stamps an epoch a sink can bounce, so without it a fenced node would keep overwriting the real owner's tier-1 backups.")
//! @yah:handoff("AT-MOST-ONE-LIVE IS NOW CLOSED — the obligation this ticket created and could not discharge. `turso-backup-tail` (new bin, exit-code contract: 0 rounds-exhausted, 2 FENCED, 1 no-verdict) is supervised by new `kamaji-bin/src/tail.rs`, one per declaring workload, and a `2` calls `server::stop_workload`. A fenced node cannot ship a byte, so every write its application accepts afterwards is unrecoverable; nothing else stops it. Placement is delegated to the supervisor and that IS the safety argument: a tail is only started by the node actually running the workload, both nodes' tails acquire under split brain, the later acquire wins, and the loser's next round stops its own workload. Acquire happens once at start, so they converge rather than ping-pong. Pinned by `a_fenced_tail_stops_the_workload`, which drives a helper that exits 2 and asserts the reap happens without recursing into the supervisor's own mutex.")
//! @yah:handoff("TWO BUGS CAUGHT IN REVIEW BY PEERS, both real, both fixed here. (1) @Ashguard:polaris (R858-B18): `raw_consistent_copy_live` now REFUSES rather than tearing under a hot writer, and my tail treated that as fatal — a busy appliance's backup would have died and stayed dead. Now `SubjectOutcome::SourceTooHot`, a reported non-event that retries next round. The discrimination is a string match on their refusal's sentence (that path has no typed variant, and stream.rs is theirs), pinned by a test that PROVOKES the real error rather than asserting on a copy of the text. (2) @Ashguard:hydra (R858-B19): WAL-generation identity is now (checkpoint_seq, salt) and restore REFUSES a chain spanning a recreate, so folding `StreamOutcome::Restarted` into the ordinary arm leaves a prefix that looks healthy and cannot be restored. Now `tail::rebase`. tenant-streamer/src/main.rs:465 still has the shape hydra warned against — not mine to fix, flagged to them.")
//! @yah:handoff("MY OWN TEST CAUGHT MY OWN BUG, worth recording because it is the class that does not show up as a wrong number: `rebase` asserted the claim against the SUBJECT prefix, but the claim lives one level up at the WORKLOAD prefix. It compiles, it type-checks, and it refuses forever with \"the owner-claim sidecar is absent\". Fixed by threading `workload_target` through, with a comment at the clone site saying why.")
//! @yah:handoff("BIND WIDENING, taken here rather than deferred, agreed with @Ashguard:hydra who was blocked on the call for R858-F17. `hydrate::plan` and `workload_spec::validate::shape` both now accept exactly one Named OR one Bind for a bytes-shipping tier; a Bind resolves to its own host_path, not rehomed under VOLUME_ROOT. Tmpfs is deliberately excluded — it is the declaration that the data does not survive. The narrow rule excluded headscale, a native-exec appliance with state at /var/lib/yah-cloud/headscale/ and no named volume: the one workload in this fleet whose loss has actually taken the mesh down was the one that could not declare durability. R858-F17 carries a notify_on for this.")
//! @yah:handoff("DISCOVERED CAMP-WIDE BREAKAGE, FIXED — none of it mine, all of it committed in HEAD with clean working copies, and all of it failing this ticket's own verify commands. `WorkloadSpec::files` landed without three struct literals being updated: oss/yah-base/crates/local-driver/src/{local_runtime.rs:1289, pond_ssr_runtime.rs:400} and app/yah/desktop/src/shell_host.rs:299 — `cargo check --manifest-path oss/yah-base/Cargo.toml --all-targets` and the desktop check were both red for the whole camp. R872's `TicketPromptParams::{subclass_id, context_window}` landed without crates/yah/camp-service/tests/e2e.rs (3 sites) — `cargo check --workspace --all-targets` red. All six sites filled with the pre-field value and a comment naming why. Nobody held any of those files.")
//! @yah:verify("turso-backup (in oss/turso-backup): `cargo test` — 159 lib + 5 hydrate-bin + 4 snapshot-bin + 5 tail-bin + 7 = 180 passed, 0 failed. 11 new in tail::tests, 5 new in the tail bin. `cargo clippy --all-targets`: zero warnings.")
//! @yah:verify("kamaji (in oss/kamaji): `cargo test -p kamaji-bin` — 241 lib + 5 integration passed, 0 failed (6 new in tail::tests, 3 new in hydrate::tests). Its 2 clippy warnings are pre-existing and in other files (pidfd.rs events_tx, server.rs control_sock_from_spec). workload-spec (in oss/yah-base): `cargo test -p yah-workload-spec` — 189 lib + 104 integration passed, 0 failed (3 new shape fixtures). Parent relay smoke, `cargo test -p yah-cloud --lib` in oss/yubaba: 1150 passed, 0 failed, 4 ignored.")
//! @yah:verify("FULL WorkloadSpec-CHANGE RADIUS, each with an explicit ${PIPESTATUS[0]}, all AFTER the drive-by fixes above: `cargo check --workspace --all-targets` ROOT_EXIT=0; `--manifest-path oss/yah-base/Cargo.toml --all-targets` YAHBASE_EXIT=0; `--manifest-path oss/yubaba/Cargo.toml --all-targets` YUBABA_EXIT=0; `--manifest-path oss/kamaji/Cargo.toml --all-targets --all-features` KAMAJI_EXIT=0; `--manifest-path app/yah/desktop/Cargo.toml --no-default-features` DESKTOP_EXIT=0. `scripts/check-workload-spec-ts.sh`: ok, index.ts in sync. No .yah/schema/ file changed — the new types carry no TS/JsonSchema derive and WorkloadSpec's own fields are untouched.")
//! @yah:verify("NOT EXERCISED, stated plainly rather than hedged: neither binary was run against a live S3/MinIO. Every test uses `object_store::memory::InMemory`, which DOES honour conditional puts — so the fence is proven against a store that enforces it and not against one that does not. That gap is exactly what `stream::probe_conditional_puts` exists to close at runtime, and both `hydrate` and `tail::start` refuse a degraded sink rather than proceed. `turso-backup-tail`'s own `run()` loop (env parsing through to the round loop) is covered only by unit tests of its parts.")
//! @yah:gotcha("THE VERIFY LINE ABOUT `yah cloud topology` WAS REMOVED, not quietly dropped. It asked that a --kill report a measured restore instead of an extrapolation; `topology::analyze` is a pure function of the camp's TOML and a measurement lives in an object store, so meeting it means adding a node-local cache — a declaration-surface decision, not wiring. Filed as R850-T2 with the two options and a recommendation. R850-T3 carries the frame-GC the tier-2 rebase defers.")
//! @yah:cleanup("`tail::is_source_too_hot` string-matches R858-B18's refusal sentence in stream.rs. Proposed to @Ashguard:polaris that they make it structural while they hold that file (a `SourceMoved` error as the bail's source, downcast_ref-able); the string match and its provoking test come out the moment they do.")
//! @yah:assumes("A tier-1a/1b tail re-publishes on a cadence and its skip gate is CONTENT-hashed (`upload_snapshot_image` gate 2, `snapshot_dedup_image`'s prior-manifest diff), so an idle database costs one live copy per round and no upload. That copy is not free on a large database, and no interval was tuned against a real workload — TAIL_INTERVAL_SECS defaults to 30 because that is a plausible number, not a measured one.")
//! @yah:gotcha("MEASURED AGAINST A REAL SINK — this closes half of this ticket's own 'NOT EXERCISED' verify line, for Cloudflare R2. Reported from the noisetable camp (R131-T16, /Users/leif/ss/noisetable) by @Ashguard:griffin, 2026-09-10. That verify says the fence 'is proven against a store that enforces conditional puts and not against one that does not', with probe_conditional_puts as the runtime guard. Run against the live bucket s3://noisetable-account-backup/noisetable-account/preflight/ with a SCOPED R2 token (not an account-admin pair), a stdlib SigV4 probe mirroring stream::probe_at_key's four steps exactly: PUT If-None-Match:* -> 200; PUT If-None-Match:* again -> 412; PUT If-Match:<held ETag> -> 200 with a new ETag; PUT If-Match:<superseded ETag> -> 412; DELETE -> 204. Verdict Honoured, i.e. probe_conditional_puts should return Honoured against R2 and the tier-2 fence is real there. Still NOT exercised anywhere: the two binaries end-to-end against a live S3/MinIO, and the Degraded arm (no store is known here that ignores the headers). R2 endpoint form is https://&lt;account&gt;.r2.cloudflarestorage.com with region 'auto'.")
//! @yah:gotcha("PARSED BUT NOT DELIVERED: `yah.durability.rpo-seconds` never reaches turso-backup-tail. Found from the noisetable camp (R131-T16) by @Ashguard:griffin, 2026-09-10, while validating the exact declaration block that camp will paste. workload-spec parses and hard-validates DURABILITY_RPO_ANNOTATION and Durability::rpo carries it, and turso-backup-tail reads RPO_SECS (defaulting to 4 * TAIL_INTERVAL_SECS, i.e. 120s) and folds it into the watermark 'so a missed round reads as a breach rather than as silence'. But kamaji-bin/src/tail.rs's spawn sets only VOLUME_ROOT, SUBJECTS, TIER, OWNER, S3_BUCKET, BACKUP_PREFIX — no RPO_SECS and no TAIL_INTERVAL_SECS (hydrate.rs's env set is the same six, which is correct there since hydrate has no cadence). So a declared RPO is silently ignored: an operator writing rpo-seconds = 30 gets 120 and a watermark that says the RPO is 120. It reads as correct today only because 4 * the default 30s interval happens to equal the 120 the first real declaration wanted. Two-line fix at the .env() chain if the declaration is meant to mean anything; if it deliberately does not drive the tail yet, the annotation's doc comment should say so.")
//! @yah:gotcha("DOWNSTREAM CONSUMER MEASUREMENT from the noisetable camp (R131-T16), taken 2026-09-11T06:02Z against us-east-001. The release is PARTIALLY out and it is worth knowing which half. The deployed kamaji's `--help` now matches `hydrate-helper` twice (it matched 0 on 2026-09-10), but `tail-helper` still matches 0. kamaji.service ExecStart is `/usr/local/bin/kamaji --socket /run/kamaji/kamaji.sock --containerd-socket /run/containerd/containerd.sock --native-exec-dir /var/lib/yah/kamaji/native` and passes NEITHER flag. /usr/local/bin holds only turso-backup-snapshot; turso-backup-tail and turso-backup-hydrate are both absent. The node has been redeployed (kamaji.rollback-20260910, yubaba.rollback-20260910 present), so this is a partial release rather than a stalled one. CONSEQUENCE FOR A REAL DOWNSTREAM SERVICE: noisetable-account is live, serving production passkey sign-in, and its only backup is a 10-minute-RPO snapshot timer with no fencing and a human-run restore. It cannot declare `yah.durability.*` until a kamaji release carries `--tail-helper` AND `--hydrate-helper` AND both helper binaries AND the S3_* credentials onto the node — declaring before that makes kamaji REFUSE the deploy, i.e. takes the account API down rather than backing it up. THE ASK, one line: sign off and commit R850-F1, then cut a kamaji release with BOTH durability helpers wired into ExecStart and both binaries placed. Nothing in the noisetable camp can produce that.")
//!
//! @yah:ticket(R850-T2, "Feed a measured restore time back to `yah cloud topology`, without making analyze do I/O")
//! @yah:status(review)
//! @yah:at(2026-09-11T06:27:57Z)
//! @yah:assignee(agent:bundle-anthropic-ashguard)
//! @yah:phase(P3b)
//! @yah:parent(R850)
//! @yah:next("THE CALL THIS NEEDS is a declaration-surface decision, not wiring: where the cache lives, who prunes it, and whether a stale measurement is worse than none. `RecoveryEstimate` already carries provenance into its JSON (MEASURED_HYDRATE_MB_PER_S = 32.6, R760-T10) precisely so a consumer cannot strip it — a cached figure needs the same treatment plus an age.")
//! @yah:gotcha("DO NOT MAKE `topology::analyze` DO I/O. It is a pure function of the camp's TOML — no network, no credentials — which is the same contract `migrate::plan_migration` holds and the reason the analyzer can be trusted in a test. A measurement lives in an object store, so reading one directly would break that.")
//! @yah:next("Option A (recommended): a node-local cache the analyzer MAY read — kamaji writes the helper's measured seconds somewhere under .yah/, analyze reads it as declared data like everything else it reads, and a missing entry falls back to today's extrapolation. Keeps analyze pure over the local tree.")
//! @yah:handoff("R850-F1 now produces the measurement this wants. `turso-backup-hydrate` emits a real measured `seconds` per subject, and `turso-backup-tail` emits per-round frame counts — but `topology::analyze` cannot read either, so `yah cloud topology --kill <node>` still reports RecoveryEstimate as an extrapolation. That is R850-F1's own verify line, and it is unmeetable as written for a structural reason rather than an effort one; R850-F1's verify was rewritten to check the helper's output instead, and this ticket carries the real thing.")
//! @yah:handoff("LANDED. `topology::analyze` now reports a MEASURED restore where one exists, and it is still PURE — no I/O, no new argument, no Path, no clock. New module `oss/yubaba/crates/cloud/src/recovery_journal.rs` (registered in lib.rs next to asset_journal) is an append-only JSONL journal at `.yah/cloud/recovery.jsonl` via a new `paths::recovery_journal(workspace_root)` (paths.rs, beside `asset_status_journal`). Record shape is one JSON object per line: `{at, workload, node, tier?, subject, bytes, seconds, helper}`, `at` RFC3339. `CloudConfig::load` replays it into a new field `CloudConfig.recovery_measurements: BTreeMap<String, WorkloadRecovery>` keyed by workload; a missing file replays to empty and is NOT an error, exactly like an unsynced infra source in the same loader. `load_from_config_dir` gets an empty map for the same reason the sources overlay does not apply there (commented at the site). New `RecoveryEstimate::Measured { seconds, bytes, measured_at, age_days, node, basis }` at topology.rs, preferred inside the `DataLoss::Window` arm of `recovery()` over BOTH `Hydrate` and `UnknownStateSize` whenever a record exists for that workload. Provenance AND age both go into the JSON so a consumer can strip neither; `headline()` extended to match. Wiring is `node_loss` -> `workload_impact(w, dead, &cfg.recovery_measurements)` -> `recovery(w, loss, measurements)`.")
//! @yah:handoff("DECLARATION-SURFACE ANSWERS, as implemented. (1) WHERE: `.yah/cloud/recovery.jsonl`, sibling of asset_journal's status.jsonl, read at CloudConfig::load time so analyze sees it as declared data. (2) WHO PRUNES: nobody. Append-only is the whole retention policy, same answer asset_journal gives; there is no pruner and one must not be built. `the_last_record_for_a_subject_wins_and_nothing_is_pruned` asserts both lines survive on disk while replay keeps only the latest. (3) STALE vs NONE: a stale measurement is REPORTED, never discarded. `recovery_journal::STALE_AFTER_DAYS = 30`; past that the estimate stays `Measured` and the headline gains \"measured N days ago; declared state may have grown since, so treat it as a floor rather than a forecast\". A real timed restore from 90 days ago still beats an extrapolation from one constant measured once on one unrelated host. Two derived decisions worth knowing: a workload's `measured_at` is the OLDEST component of the sum (a sum is only as fresh as its stalest part), and `age_days` is derived from `WorkloadRecovery.as_of` — the clock is read once at replay and captured as data, which is what lets `analyze` stay clockless as well as I/O-free.")
//! @yah:handoff("WRITER SEAM: `yah cloud topology --record-hydrate <path|-> --workload <NAME> --node <MACHINE> [--tier <TIER>]` in app/yah/cli/src/cloud.rs. clap `requires_all` binds workload+node to the flag (a measurement nobody can attribute is worse than none); a redundant bail guards it anyway. New fn `record_hydrate_measurement` reads every non-empty line of the file or stdin, parses each through `RecoveryRecord::from_helper_json`, appends, and reports the count + summed seconds on STDERR so `--format json` keeps a clean stdout. Ingest happens BEFORE `load_cloud`, so the same invocation reports with the measurement it just filed — the flag verifies itself. A line reporting no restore (already_populated / nothing_in_the_store / refused) contributes zero records and is NOT an error, but it prints a loud \"nothing appended\" line rather than reading as success. `--tier` exists because the helper reads TIER from its own environment and does not print it; unattributed it stays None rather than being guessed. cloud.rs edits were confined to the Topology arg struct, its dispatch arm, handle_topology, and the one new fn — plus 3 mechanical one-line `recovery_measurements:` additions to pre-existing CloudConfig test-helper literals (16529/16674/17397), driven off compiler spans.")
//! @yah:verify("BASELINE MEASURED BEFORE THE FIRST EDIT: `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib` = 1173 passed, 0 failed, 4 ignored. AFTER: 1187 passed, 0 failed, 4 ignored (+14 new, zero regressions). `cargo check --manifest-path oss/yubaba/Cargo.toml -p yubaba --all-targets` clean. `cargo build -p yah --lib` and `cargo check -p yah --tests` both clean from the repo root (no errors; the warnings present are all pre-existing and in other agents' files). `cargo test --manifest-path oss/turso-backup/Cargo.toml --bin turso-backup-hydrate` = 5 passed, 0 failed. `./scripts/check-schema-drift.sh` = \"ok: .yah/schema is in sync with the Rust types\" — no regeneration needed, and workload-spec sources were not touched so its TS guard is not implicated (not run).")
//! @yah:verify("ALL FOUR REQUIRED TESTS EXIST AND PASS. (a) `an_absent_recovery_journal_leaves_todays_answers_untouched` — asserts the journal file does not exist, then pins BOTH pre-existing answers unchanged (Hydrate at 100/32.6 s with state-mb declared, UnknownStateSize without). (b) `a_journalled_restore_replaces_the_extrapolation_with_the_measured_sum` — two subjects, 12.5s + 8.5s = 21.0s summed, 100 MiB summed, basis contains \"MEASURED, not extrapolated\" and does NOT contain R760-T10. (c) `a_stale_measurement_is_still_reported_and_says_so` — a 90-day-old entry stays `Measured` (explicitly NOT a fallback to Hydrate) and its headline says \"measured 90 days ago\"/\"may have grown since\". (d) `a_real_hydrate_line_parses_verbatim` in recovery_journal.rs. Plus `a_measurement_beats_an_undeclared_state_size` and `a_measurement_for_one_workload_does_not_leak_into_another` (a stray record must not invent state for a stateless workload), and 7 journal-level tests in recovery_journal.rs. Test fixtures go through the REAL journal writer and the real `CloudConfig::load` via the existing `Camp` harness (new `Camp::measured(...)` helper) — nothing hand-builds a CloudConfig.")
//! @yah:verify("END-TO-END, against a staged fixture camp at /tmp/r850t2-camp with the real ./target/debug/yah binary, not just unit tests. BEFORE: `recovery: >= ~3.1s to pull 100 MiB (extrapolated from 32.6 MB/s measured once, R760-T10...)`. AFTER `--record-hydrate /tmp/r850t2-hydrate.json --workload db --node a --tier stream`: stderr \"recorded 2 measured subject restore(s) for workload 'db' on 'a' (21.0s total)\", two JSONL lines on disk, and the same invocation printed `recovery: 21.0s MEASURED — a real restore of 100.0 MiB timed on a, 0 day(s) ago`. `--format json` carries kind=measured with seconds/bytes/measured_at/age_days/node/basis all present. The `-` stdin path and the no-measurement outcome path were both exercised: `{\"outcome\":\"already_populated\"}` on stdin printed the loud \"nothing appended\" line and left the journal untouched.")
//! @yah:gotcha("SCOPE HELD: the kamaji wire protocol was NOT widened. `kamaji::hydrate::run` still returns `HydrateResult::Proceed(Some(line))` and server.rs only `info!`s it; kamaji-proto/messages.rs and kamaji-bin/server.rs are untouched (both are uncommitted-modified by peers in this shared tree). The CLI ingest is the seam. LEADER CALL WORTH FILING: carrying the helper's line back through kamaji so a fleet-node restore journals itself with no operator step IS the right long-term shape — today a measurement only exists if somebody remembers to run `--record-hydrate`, which is exactly the kind of manual step that makes a measured figure permanently absent. Not built here by instruction. EDIT OUTSIDE THE PRIMARY BLAST RADIUS, DISCLOSED: `oss/turso-backup/src/bin/hydrate.rs` gained 17 lines — a full-line `assert_eq!` inside the EXISTING test `a_hydrated_outcome_reports_measured_bytes_and_seconds`, pinning its emitted format string byte-for-byte to `REAL_HYDRATE_LINE` in recovery_journal.rs. That is what makes the verify item \"build the fixture from hydrate.rs's own format string so the two cannot drift\" actually true: `cloud` deliberately takes no dependency on turso-backup, so the only way to stop the two drifting is an assertion on each side of the same literal. Changing `outcome_to_json` now fails in turso-backup FIRST, naming the cloud constant to update. @Ashguard:blade (session:8f7399ff) is live in oss/turso-backup/src/stream.rs on this same relay — different file, no overlap with this edit.")
//! @yah:handoff("THE DECLARATION-SURFACE CALL THIS TICKET WAS FILED TO MAKE, made and shipped. Where the cache lives: an append-only JSONL journal at `.yah/cloud/recovery.jsonl`, reached by a new `paths::recovery_journal(workspace_root)` — deliberately the same shape as the `asset_journal` / `.yah/cloud/status.jsonl` precedent already in this crate (R470-T1), not a new mechanism. Who prunes it: nobody, which is the point of append-only plus last-wins replay, and is the same answer the asset journal already gives. Whether a stale measurement is worse than none: NO — a real timed restore with its age printed beats an extrapolation from one unrelated host, so `Measured` is never discarded on age; past ~30 days the headline says so instead. Provenance and age both ride into the JSON the way `Hydrate`'s basis already did, so a consumer cannot strip either.")
//! @yah:handoff("ANALYZE STAYED PURE, which was the ticket's hard gotcha. `topology::analyze` gained no I/O, no `Path` argument, and no clock — the clock is captured as DATA at replay time, so the age is a value in the model rather than a call inside it. `CloudConfig::load` replays the journal into a new `recovery_measurements` field and analyze reads that, exactly as it already reads every other thing the loader pulled off the local tree. A missing journal file is not an error: it replays to empty and every existing verdict is unchanged.")
//! @yah:verify("LEADER RE-RAN EVERY GATE INDEPENDENTLY (@Ashguard:eclipse), rather than accepting the courier's counts. `cargo test --manifest-path oss/yubaba/Cargo.toml -p yah-cloud --lib`: 1187 passed / 0 failed / 4 ignored, exit 0 — matching the courier's post-change figure against its own pre-edit baseline of 1173/0, so +14. NOTE the invocation: yah-cloud is not a root workspace member and needs dev-deps, so a repo-root `cargo test -p yah-cloud` does NOT work. `cargo build -p yah --lib` from the repo root: Finished, exit 0 (25 pre-existing warnings, none new-file). `scripts/check-schema-drift.sh`: 'ok: .yah/schema is in sync with the Rust types'. `scripts/check-workload-spec-ts.sh`: 'ok: packages/yah/workload-spec/index.ts is in sync with the Rust schema'. Both drift gates green, so nothing is left red for the next reader.")
//! @yah:verify("THE CROSS-CRATE SEAM WAS CHECKED AGAINST THE OTHER LIVE TICKET, not assumed. R850-T2 added a 17-line full-line `assert_eq!` in oss/turso-backup/src/bin/hydrate.rs pinning that binary's emitted format string to the cloud-side parser fixture, so the producer and consumer of the JSON cannot drift apart silently. @Ashguard:blade was concurrently editing the same crate for R850-T3; the leader's combined re-run of `cargo test --manifest-path oss/turso-backup/Cargo.toml` is 192 passed / 0 failed including that pin (bin/hydrate 5/5), so the two tickets' edits coexist.")
//! @yah:gotcha("THE AUTOMATIC WRITER IS NOT IN THIS TICKET AND IS NOW FILED AS R850-T4. What shipped is the reader end plus a manual seam: `yah cloud topology --record-hydrate <path|-> --workload <w> --node <n>` ingests turso-backup-hydrate's JSON line from a file or stdin. Carrying the measurement back automatically means widening the kamaji wire (kamaji-proto/src/messages.rs, kamaji-bin/src/server.rs:2086, which already holds the line and only `info!`s it), and both files were uncommitted-dirty on the shared tree during this run — a scheduling reason, not a design objection. Until R850-T4 lands, a camp nobody feeds reports the extrapolation, correctly labelled.")
//!
//! @yah:ticket(R850-T3, "GC the frame objects a tier-2 rebase orphans after a WAL restart (oss/turso-backup/src/tail.rs)")
//! @yah:status(review)
//! @yah:at(2026-09-11T06:27:23Z)
//! @yah:assignee(agent:bundle-anthropic-ashguard)
//! @yah:phase(P4c)
//! @yah:parent(R850)
//! @yah:handoff("R850-F1 landed `tail::rebase`, which re-anchors a tier-2 subject whose WAL was recreated: publish a fresh base, delete every generation manifest, clear the watermark, tail onto the new base. Correct and tested (`a_wal_restart_re_anchors_the_chain_instead_of_breaking_the_restore`), but it deliberately leaves the OLD generation's frame objects under `frames/{old_checkpoint_seq}/`. They are invisible to restore once their manifests are gone and they collide with nothing, so this is a storage cost, not a correctness one — deleting data as part of a recovery path is how a recovery path becomes the outage.")
//! @yah:gotcha("FILED HERE, LIVES THERE. The code is `oss/turso-backup/src/tail.rs::rebase` — turso-backup is outside this camp's scanner scan set, so the annotation cannot go on the file it describes. Do not go looking for a `@yah:` block in turso-backup.")
//! @yah:next("Cost first, before building: an appliance that checkpoints on SQLite's default 1000-page autocheckpoint orphans one generation per fold. Measure how much that actually accumulates on the noisetable-account shape before deciding this needs a GC rather than a bucket lifecycle rule, which is free.")
//! @yah:handoff("COST-FIRST GATE RUN BEFORE BUILDING, as this ticket demanded, and it changed the ticket. (1) The premise holds and understates the rate: `rebase`'s only call site is tail.rs:524 on `StreamOutcome::Restarted`, whose `restarted` flag (stream.rs:1288-1294, via `is_provably_same_as` stream.rs:319-324) is set because an ordinary in-process autocheckpoint increments BOTH salt1 and checkpoint_seq (measured: stream.rs:249-250, examples/foreign_checkpoint_probe.rs:35-37). So rebase runs on ordinary checkpoint folds, not only on process restart — bounded above by the tail round rate, default 30s (src/bin/tail.rs:67), with an `Empty` round not rebasing (stream.rs:1296-1298). (2) THE FREE OPTION IS NOT AVAILABLE: there is no bucket lifecycle rule anywhere in this tree, and a creation-time rule cannot express this job — at any moment the LIVE base snapshot is simply the most recent one, so a blanket 'expire older than N days' deletes the live base of any subject that has not folded in N days. (3) So: build the GC.")
//! @yah:handoff("THE FRAMES ARE THE SMALL TERM — the ticket named the wrong leak as the main one, and the bigger one is covered here rather than filed. `rebase` (tail.rs:615) calls `snapshot::upload_base_snapshot` (tail.rs:635 -> snapshot.rs:216-231), the explicitly non-deduplicating one-shot variant, then deletes only manifests (tail.rs:636) and the watermark (tail.rs:637-643). Every rebase therefore stranded a COMPLETE COPY OF THE DATABASE, permanently, which exceeds the frame term (<= ~4.12 MB in <= 4 objects per rebase, at spill_buffer_frames=256 / 24+4096 B per frame) for any database over ~4 MB. `gc_stream` collects both classes. The doc comment at tail.rs:602-616 that asserted only the frames were left behind was WRONG about the leak it documented and is corrected in place ('Two things, not one'); lib.rs:16-19 now names both tiers' sweeps.")
//! @yah:handoff("WHAT SHIPPED: `stream::gc_stream` (oss/turso-backup/src/stream.rs:3050) with `StreamGcConfig` (:2914), `StreamGcOutcome` (:2937), `DEFAULT_STREAM_GC_GRACE` = 24h (:2907), and two private helpers `delete_collected` (:3138, absent-is-success, the same posture rebase's own deletes hold) and `list_recursive` (:3156). Plus a `turso-backup-gc` bin (oss/turso-backup/src/bin/gc.rs, wired in Cargo.toml) emitting one JSON line, mirroring turso-backup-snapshot's shape. DRY RUN UNLESS `GC_APPLY` is an explicit 1/true/yes — GC_APPLY=0, a typo and an empty string all leave the dry run in place, because the failure mode of guessing wrong points at deleted objects. `GC_GRACE_SECS` overrides the grace. Shape follows the existing `dedup::gc_dedup` (dedup.rs:472-573) tier-1b sweep rather than minting a parallel vocabulary. `rebase` still deletes nothing — the ticket's standing judgment that a recovery path must not delete is intact; the sweep is explicitly invoked.")
//! @yah:handoff("LIVENESS WAS DERIVED BY READING THE RESTORE PATH, NOT ASSUMED, and the two are pinned together. Restore makes two selections and the GC's live set is their union: `restore_stream_from_manifests` (stream.rs:2094-2101) takes the chain's `base_snapshot_key`, and with no generations `hydrate::restore_subject` (hydrate.rs:522-528) falls back to `snapshot::restore_latest`, which takes the lexically-greatest key (snapshot.rs:432-444). Frames are live iff a manifest names them, resolved through `BackupTarget::frame_objects_of` — the same function restore's replay and `WalPuller` use, so both epoch key shapes and both batch/legacy layouts come along by construction. `gc_liveness_is_the_complement_of_restore_selection` asserts both branches against the real selection functions so they cannot drift. DELIBERATE: every manifest's base key is treated as live, not `validate_generation_chain`'s single answer — a chain spanning a WAL restart does not validate at all, and refusing to guess keeps both bases rather than deleting the one the next rebase is about to adopt.")
//! @yah:handoff("TWO JUDGMENT CALLS, both recorded at the code site. (1) GRACE WINDOW, default 24h (stream.rs:2907), documenting the three live windows it must cover: a serialized cold restore in flight, a tail between uploading batches and writing their manifest, and rebase's publish-before-delete gap where the NEW base is reachable from nothing — that third case is where grace is the only thing preventing a concurrent GC from deleting a base a recovery just published. (2) DISCOVERED GAP, CLOSED: `base_snapshots_skipped` (:2937). A `snapshots/` prefix with no chain over it is indistinguishable from a plain tier-1a sink, whose older snapshots are HISTORY, not garbage — unguarded, the sweep would have pruned all but the newest. It now skips the snapshot half entirely when there are no generation manifests and reports that it did. Frames still go: a `frames/` prefix under a chainless sink is unreachable by construction.")
//! @yah:verify("BASELINE MEASURED BEFORE THE FIRST EDIT, then re-measured, and INDEPENDENTLY RE-RUN BY THE LEADER (@Ashguard:eclipse) rather than taken on the courier's word. `cargo test --manifest-path oss/turso-backup/Cargo.toml` (turso-backup is its own workspace, excluded from the yah root workspace — a root-level `-p` invocation does not work): baseline 180 passed / 0 failed; after 192 passed / 0 failed. Leader's independent re-run of the combined tree, after R850-T2's own edit to oss/turso-backup/src/bin/hydrate.rs landed: 167 + 4 + 5 + 4 + 5 + 7 = 192 passed / 0 failed, exit 0. `cargo clippy --all-targets -- --deny=warnings` clean; `cargo doc --no-deps` warning locations byte-identical to before the edits.")
//! @yah:verify("TWELVE NEW TESTS — all four the ticket named, plus four more that came out of the liveness derivation. `gc_collects_a_superseded_base_snapshot_and_keeps_the_current_one` · `gc_collects_orphaned_frame_prefixes_and_keeps_the_live_generation` (both epoch layouts) · `gc_spares_everything_inside_the_grace_window` · `gc_dry_run_reports_without_deleting` (and asserts the wet sweep executes the dry proposal key-for-key) · `gc_liveness_is_the_complement_of_restore_selection` · `gc_keeps_both_bases_while_a_rebase_is_half_landed` · `gc_without_a_chain_keeps_snapshots_but_still_collects_frames` · `gc_on_an_empty_prefix_collects_nothing`, plus 4 in bin/gc.rs. NO LIVE MinIO NEEDED: these use `object_store::memory::InMemory`, the same route the existing `dedup` GC tests take — no new infrastructure was stood up.")
//! @yah:gotcha("THIS TICKET'S ANNOTATION WAS WRITTEN BY THE LEADER, NOT THE IMPLEMENTER, on purpose. R850-T3's annotation lives in oss/yubaba/crates/cloud/src/topology.rs (turso-backup is outside the board scanner's scan set), and @Ashguard:polaris held that file dirty with ~9 uncommitted writes for R850-T2 for the whole of this ticket's run. @Ashguard:blade was steered off `board.update` mid-turn and reported its account back to the leader instead, which then wrote it here once polaris was out. Content is blade's; the keystrokes are the leader's.")

use std::collections::BTreeMap;

use chrono::{DateTime, Utc};
use serde::Serialize;

use workload_spec::{
    Durability, DurabilityTier, LifecycleArchetype, RestartPolicy, VolumeSource, WorkloadSpec,
};

use crate::config::{CloudConfig, MachineConfig, NodeAllocatable};
use crate::migrate::{named_volume_path, VolumeDisposition};
use crate::recovery_journal::{self, WorkloadRecovery};

// ─── Measured constants the recovery estimate is built on ────────────────────

/// Bulk object-store throughput, MB/s.
///
/// **Measured**, not modelled: R760-T10 on 2026-08-29, a real bulk range GET at
/// 32.6 MB/s on one host — the 100 MB / 3.3 s figure in
/// `oss/roadcase/docs/COST.md` §8. It is one measurement on one host against
/// one backend, which is the whole of what this camp knows about the number;
/// see [`RecoveryEstimate`] for how that limitation is carried outward rather
/// than smoothed over.
pub const MEASURED_HYDRATE_MB_PER_S: f64 = 32.6;

/// Per-GET round-trip time, milliseconds. Measured in the same R760-T10 run.
///
/// Load-bearing because a tier-2 cold start issues its GETs **serially** —
/// `turso_backup::stream` awaits one generation manifest and then one frame
/// object at a time — so round trips, not bandwidth, are what dominates a
/// restore of a small database with a long WAL history.
pub const MEASURED_GET_RTT_MS: f64 = 16.0;

/// `turso_backup::stream::DEFAULT_RPO_TARGET`, in seconds.
///
/// Duplicated rather than imported: `cloud` has no `turso-backup` dependency
/// and should not grow one to print a number into a report. There is therefore
/// **no test pinning the two together** — if this looks stale, the authority is
/// `DEFAULT_RPO_TARGET` in `oss/turso-backup/src/stream.rs`, and a report that
/// says "≤ 120s (turso-backup default)" is only as true as this line.
pub const DEFAULT_STREAM_RPO_SECONDS: u32 = 120;

// ─── The model ───────────────────────────────────────────────────────────────

/// The declared fleet, read as a graph, plus every verdict derivable from it.
///
/// This is the single traversal. The Mermaid render ([`Topology::to_mermaid`])
/// and the capacity report are *projections* of these fields — a diagram
/// generated by its own second walk of the config would be free to disagree
/// with the analysis printed above it, which is worse than no diagram.
#[derive(Debug, Clone, Serialize)]
pub struct Topology {
    pub machines: Vec<MachineNode>,
    pub workloads: Vec<WorkloadNode>,
    /// One entry per declared machine: what is lost when that machine is.
    pub node_losses: Vec<NodeLoss>,
    /// Machines whose declared `[allocatable]` does not cover what is projected
    /// onto them. Empty is the good case.
    pub oversubscribed: Vec<Oversubscription>,
}

/// Where a machine sits in its sovereign group's raft, read off the three
/// declarations that decide it — group stamp, participation, and the group's
/// voter list (R605-F35). The same reading a raft leader applies, so the
/// picture can never seat a node the cluster would leave a learner.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")]
pub enum RaftSeat {
    /// No `sovereign_group`: in no raft at all.
    NoGroup,
    /// A participant its group's `.yah/infra/sovereign-groups/<group>.toml`
    /// lists: holds a quorum seat.
    Voter,
    /// A participant the group does not list: replicates, never votes.
    Learner,
    /// `sovereign_participation = "out"`: in the blast radius, outside the
    /// raft.
    Out,
}

impl RaftSeat {
    /// Read a machine's seat off the camp's declarations.
    pub fn of(cfg: &CloudConfig, m: &MachineConfig) -> Self {
        match (&m.sovereign_group, cfg.voting_group(m)) {
            (None, _) => Self::NoGroup,
            (Some(_), Some(_)) => Self::Voter,
            (Some(_), None) if !m.sovereign_participation.is_in() => Self::Out,
            (Some(_), None) => Self::Learner,
        }
    }

    /// The label a render shows.
    pub fn label(&self) -> &'static str {
        match self {
            Self::NoGroup => "no group",
            Self::Voter => "voter",
            Self::Learner => "learner",
            Self::Out => "out of raft",
        }
    }
}

/// A declared machine, plus what the admission seam projects onto it.
#[derive(Debug, Clone, Serialize)]
pub struct MachineNode {
    pub name: String,
    pub region: Option<String>,
    pub sovereign_group: Option<String>,
    /// Where it sits in its group's raft, by declaration (R605-F35).
    pub raft_seat: RaftSeat,
    pub taints: Vec<String>,
    pub allocatable: Option<NodeAllocatable>,
    /// Workloads whose projected placement lands here, in declaration order.
    pub placed: Vec<String>,
    /// Sum of `memory_request_mb()` over [`Self::placed`].
    pub committed_memory_mb: u32,
    /// Sum of `resources.cpu_millis` over [`Self::placed`].
    pub committed_cpu_millis: u32,
}

/// Where the admission seam would put a workload, and what else could take it.
///
/// # Projected, not observed
///
/// This is `CloudConfig::admit_workload_candidates` run against the declared
/// inventory. It differs from reality in two known ways, both of them the
/// reason a *planning* surface wants the projection rather than a probe:
///
/// - A workload deployed with `--where=node:<name>` carries that pin in its
///   spec's annotations and the analyzer sees it, but a workload deployed
///   before the file was last edited is running against an older spec.
/// - Admission has no liveness input, so `chosen` may be a box that is off.
///   [`Self::alternates`] is the field that matters for survivability anyway.
#[derive(Debug, Clone, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum Placement {
    /// At least one machine admits this workload. `chosen` is the head of the
    /// candidate pool in declaration order — the same first-fit
    /// `admit_workload` returns.
    Admitted {
        chosen: String,
        /// Every *other* admitting machine. **Empty is the survivability
        /// finding**: a workload with no alternates has nowhere to go even if
        /// its archetype would allow a move.
        alternates: Vec<String>,
        /// True when the spec names a node via the R833-F8 placement
        /// annotation. A pin is still checked against capacity and taints, so
        /// a pinned workload can still be `Unschedulable`.
        pinned: bool,
    },
    /// Nothing in the declared fleet admits it. Carries admission's own
    /// refusal, which names the pool it searched.
    Unschedulable { reason: String },
}

impl Placement {
    /// The machine this workload is projected onto, if any.
    pub fn machine(&self) -> Option<&str> {
        match self {
            Self::Admitted { chosen, .. } => Some(chosen.as_str()),
            Self::Unschedulable { .. } => None,
        }
    }

    /// Machines that could take this workload if [`Self::machine`] were lost.
    pub fn alternates(&self) -> &[String] {
        match self {
            Self::Admitted { alternates, .. } => alternates,
            Self::Unschedulable { .. } => &[],
        }
    }
}

/// A workload's `yah.durability.*` declaration as the analyzer sees it.
///
/// [`Self::Undeclared`] and a declared [`DurabilityTier::None`] are separate
/// variants on purpose — see `WorkloadSpec::durability`. The first is the
/// shape that loses data by omission; the second is a decision.
#[derive(Debug, Clone, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum DurabilityView {
    /// Nobody said. For a workload with a named volume this is the finding.
    Undeclared,
    Declared(Durability),
    /// The declaration exists and cannot be read. Surfaced rather than treated
    /// as `Undeclared`, because "the operator tried and got it wrong" and "the
    /// operator never considered it" call for different conversations.
    Malformed {
        reason: String,
    },
}

/// One workload, everything about it that bears on survival, and where it goes.
#[derive(Debug, Clone, Serialize)]
pub struct WorkloadNode {
    pub name: String,
    pub tier: String,
    pub mesh_identity: String,
    pub archetype: LifecycleArchetype,
    pub replicas: u32,
    /// TOML-ish spelling of `restart_policy`, for report output.
    pub restart_policy: String,
    pub placement: Placement,
    /// Reused wholesale from [`crate::migrate`]: the named/bind/tmpfs split is
    /// the same classification a move needs, and minting a second vocabulary
    /// for it would let the two answers drift.
    pub volumes: Vec<VolumeDisposition>,
    pub durability: DurabilityView,
    /// Memory **request** (`memory_request_mb()`), not the cgroup ceiling.
    pub memory_request_mb: u32,
    pub cpu_millis: u32,
    /// Public hostnames this workload fronts, if any.
    pub public_hostnames: Vec<String>,
    /// Mesh identities that must be `Ready` before this one starts.
    pub depends_on: Vec<String>,
}

impl WorkloadNode {
    /// Durable mounts — the ones whose bytes a node loss puts at risk.
    /// Tmpfs is excluded by [`VolumeDisposition::is_durable`].
    pub fn durable_volumes(&self) -> Vec<String> {
        self.volumes
            .iter()
            .filter(|v| v.is_durable())
            .map(|v| v.label())
            .collect()
    }

    /// Whether any mount is a yubaba-managed named volume. This is the exact
    /// shape whose only copy lives at `/var/lib/yah/kamaji/volumes/<name>` on
    /// one box.
    pub fn has_named_volume(&self) -> bool {
        self.volumes
            .iter()
            .any(|v| matches!(v, VolumeDisposition::Copy { .. }))
    }
}

// ─── Node loss ───────────────────────────────────────────────────────────────

/// Everything that follows from losing one machine outright.
#[derive(Debug, Clone, Serialize)]
pub struct NodeLoss {
    pub machine: String,
    pub impacts: Vec<WorkloadImpact>,
    /// Public hostnames served only from this machine.
    pub public_endpoints_lost: Vec<String>,
    pub quorum: QuorumEffect,
}

impl NodeLoss {
    /// Impacts where bytes are gone for good. The headline of any report.
    pub fn total_losses(&self) -> impl Iterator<Item = &WorkloadImpact> {
        self.impacts
            .iter()
            .filter(|i| matches!(i.data_loss, DataLoss::Total { .. }))
    }

    /// Impacts that need a person. The second headline: an outage nobody is
    /// paged for is an outage that lasts until someone notices.
    pub fn needs_operator(&self) -> impl Iterator<Item = &WorkloadImpact> {
        self.impacts.iter().filter(|i| i.outcome.is_manual())
    }
}

/// What a node loss does to one workload projected onto it.
#[derive(Debug, Clone, Serialize)]
pub struct WorkloadImpact {
    pub workload: String,
    pub archetype: LifecycleArchetype,
    pub outcome: Outcome,
    pub data_loss: DataLoss,
    pub recovery: RecoveryEstimate,
}

/// Whether the workload comes back, and who brings it back.
#[derive(Debug, Clone, PartialEq, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum Outcome {
    /// Fungible, and somewhere else in the declared fleet admits it. Note that
    /// this says the *scheduler* could place it — it does not claim anything
    /// automatically triggers that placement today.
    Reschedulable { candidates: Vec<String> },

    /// Fungible, but nothing else admits it. Down until the node is back or
    /// the fleet grows. `reason` is the constraint that excludes everyone else
    /// — a taint, the capacity floor, an arch mismatch.
    NowhereToGo { reason: String },

    /// [`LifecycleArchetype::Appliance`]: **pinned and non-drainable.**
    ///
    /// This is the variant the driving question lands on. `drain_workloads`
    /// (`oss/yubaba/crates/yubaba/src/lib.rs`, R572-F4) skips appliances
    /// outright, so there is no automatic move at any capacity — and the
    /// operator verb that does move one, `yah cloud migrate`, *plans a
    /// stop → copy → start* and expects the volume to already exist at the
    /// destination. Against a node that is gone at the hardware level there is
    /// nothing to copy from, which is why this variant does not promise
    /// `migrate` will help.
    PinnedAppliance {
        /// Where a migrate could target, if anywhere admits it.
        migrate_target: Option<String>,
        /// True when the source volume is only reachable from the dead node,
        /// i.e. `yah cloud migrate` has no source to copy from.
        source_unreachable: bool,
    },

    /// `restart_policy = Never`: a run, not a service. Losing the node loses
    /// the run; the answer is to run it again, not to fail it over.
    RunLost,

    /// `replicas > 1` and at least one other machine admits the workload, so
    /// the survivors keep serving while the lost replica is replaced.
    DegradedButServing { surviving_replicas: u32 },
}

impl Outcome {
    /// Whether a human has to do something before this workload serves again.
    pub fn is_manual(&self) -> bool {
        matches!(
            self,
            Self::PinnedAppliance { .. } | Self::NowhereToGo { .. } | Self::RunLost
        )
    }

    /// One line for a text report.
    pub fn headline(&self) -> String {
        match self {
            Self::Reschedulable { candidates } => {
                format!("automatic — schedulable onto {}", candidates.join(", "))
            }
            Self::NowhereToGo { reason } => {
                format!("STAYS DOWN — nothing else admits it: {reason}")
            }
            Self::PinnedAppliance {
                migrate_target,
                source_unreachable,
            } => {
                let target = migrate_target.as_deref().unwrap_or("(nothing admits it)");
                if *source_unreachable {
                    format!(
                        "OPERATOR — pinned appliance, never drained or rescheduled. \
                         `yah cloud migrate` would target {target}, but it plans a \
                         stop → copy → start and the copy has no source once the node \
                         is gone"
                    )
                } else {
                    format!("OPERATOR — pinned appliance; `yah cloud migrate` to {target}")
                }
            }
            Self::RunLost => "run lost — re-run it; nothing fails a job over".to_string(),
            Self::DegradedButServing { surviving_replicas } => {
                format!("degraded — {surviving_replicas} replica(s) still serving")
            }
        }
    }
}

/// How much of the workload's state is gone, and how far back the copy is.
#[derive(Debug, Clone, PartialEq, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum DataLoss {
    /// Nothing durable is mounted.
    None,
    /// Only tmpfs. Discarded on stop by definition, so the node dying costs
    /// nothing that a restart would not have.
    EphemeralOnly,
    /// Durable state exists and there is **no second copy anywhere**. The
    /// bytes are gone with the node.
    Total {
        volumes: Vec<String>,
        /// Why there is no copy: no declaration at all, or `tier = "none"`.
        because: String,
    },
    /// A copy exists in an object store; the loss is the gap between the last
    /// write and the last thing that reached the store.
    Window {
        tier: DurabilityTier,
        store: String,
        /// `None` for snapshot/dedup tiers, whose recovery point is set by
        /// whatever schedules the snapshot and is therefore not in the spec.
        rpo_seconds: Option<u32>,
        /// Present when [`Self::rpo_seconds`] is the turso-backup default
        /// rather than a declared value.
        rpo_is_default: bool,
    },
    /// Bind mounts only. The camp did not create the host path and cannot know
    /// whether the bytes exist elsewhere — the same refusal-to-guess
    /// `migrate::preconditions` makes for the same mount kind.
    Unknown { volumes: Vec<String> },
}

impl DataLoss {
    /// One line for a text report.
    pub fn headline(&self) -> String {
        match self {
            Self::None => "none — no durable state declared".to_string(),
            Self::EphemeralOnly => "none — tmpfs only, discarded on stop anyway".to_string(),
            Self::Total { volumes, because } => format!(
                "TOTAL — {} has no second copy anywhere ({because})",
                volumes.join(", ")
            ),
            Self::Window {
                tier,
                store,
                rpo_seconds,
                rpo_is_default,
            } => match rpo_seconds {
                Some(s) if *rpo_is_default => {
                    format!("≤ {s}s (turso-backup default, not declared) — tier {tier} → {store}")
                }
                Some(s) => format!("≤ {s}s (declared) — tier {tier} → {store}"),
                None => format!(
                    "unbounded by the spec — tier {tier} → {store}; a snapshot tier's \
                     recovery point is set by whatever schedules it"
                ),
            },
            Self::Unknown { volumes } => format!(
                "UNKNOWN — {} are operator-managed bind mounts; the camp cannot say \
                 whether the bytes exist anywhere else",
                volumes.join(", ")
            ),
        }
    }
}

/// How long it takes to get the state back, and on what basis that is claimed.
///
/// [`Hydrate`](Self::Hydrate) is an **extrapolation from two measured
/// constants** ([`MEASURED_HYDRATE_MB_PER_S`], [`MEASURED_GET_RTT_MS`]) applied
/// to a **declared** state size — one measurement, on one host, against one
/// backend, stretched over a number an operator typed. That is stated on the
/// type rather than in a footnote because a recovery-time number without its
/// provenance is the single easiest thing in a planning report to mistake for a
/// measurement.
///
/// [`Measured`](Self::Measured) (R850-T2) is the exception and the thing to
/// prefer: a restore that was actually timed, by
/// `turso-backup-hydrate`, replayed out of `.yah/cloud/recovery.jsonl` by
/// [`CloudConfig::load`] into [`CloudConfig::recovery_measurements`]. It carries
/// its own provenance *and its age*, for the same reason — and it is never
/// discarded for being old. A real restore from six weeks ago is a better
/// answer than an extrapolation from an unrelated host; it is reported with
/// "measured N days ago" attached so the reader can discount it themselves.
#[derive(Debug, Clone, PartialEq, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum RecoveryEstimate {
    /// Nothing to hydrate — the workload carries no durable state.
    Immediate,
    /// There is no copy to recover from. Recovery is not a duration.
    NotRecoverable,
    /// A copy exists but the spec does not say how big the state is, so the
    /// transfer cannot be estimated. Names the annotation that would fix it.
    UnknownStateSize { hint: &'static str },
    /// Bulk transfer of a declared state size at the measured throughput.
    Hydrate {
        state_mb: u32,
        seconds: f64,
        /// Verbatim provenance, carried into JSON so a consumer cannot strip it.
        basis: String,
    },
    /// A real restore of this workload, timed on a real host. Summed over the
    /// workload's subjects, because the helper restores them one at a time and
    /// a workload's recovery is all of them.
    Measured {
        /// Summed measured wall-clock seconds.
        seconds: f64,
        /// Summed bytes on disk after the restore. Not a declaration — this is
        /// what actually landed.
        bytes: u64,
        /// When the oldest component of the sum was measured. A sum is only as
        /// fresh as its stalest part.
        measured_at: DateTime<Utc>,
        /// Whole days from `measured_at` to when the journal was replayed.
        /// Carried into JSON beside `basis` so a consumer can strip neither the
        /// provenance nor the age.
        age_days: i64,
        /// The machine the most recent measurement was taken on. A restore time
        /// is a property of a host as much as of a database.
        node: String,
        /// Verbatim provenance, carried into JSON so a consumer cannot strip it.
        basis: String,
    },
}

impl RecoveryEstimate {
    /// One line for a text report.
    pub fn headline(&self) -> String {
        match self {
            Self::Immediate => "immediate — stateless".to_string(),
            Self::NotRecoverable => "n/a — nothing to recover from".to_string(),
            Self::UnknownStateSize { hint } => {
                format!("unknown — declare {hint} to get an estimate")
            }
            Self::Hydrate {
                state_mb, seconds, ..
            } => format!(
                "≥ ~{seconds:.1}s to pull {state_mb} MiB (extrapolated from \
                 {MEASURED_HYDRATE_MB_PER_S} MB/s measured once, R760-T10; no restore was \
                 timed here, and WAL replay is on top)"
            ),
            Self::Measured {
                seconds,
                bytes,
                age_days,
                node,
                ..
            } => {
                let mib = *bytes as f64 / (1024.0 * 1024.0);
                let base = format!(
                    "{seconds:.1}s MEASURED — a real restore of {mib:.1} MiB timed on {node}, \
                     {age_days} day(s) ago"
                );
                if *age_days > recovery_journal::STALE_AFTER_DAYS {
                    format!(
                        "{base} — measured {age_days} days ago; declared state may have grown \
                         since, so treat it as a floor rather than a forecast"
                    )
                } else {
                    base
                }
            }
        }
    }
}

/// What losing a machine does to its sovereign group's raft quorum.
#[derive(Debug, Clone, PartialEq, Serialize)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum QuorumEffect {
    /// The machine declares no `sovereign_group`, so it votes in nothing.
    NotInAGroup,
    /// In the group but holding no quorum seat — a participant its voter list
    /// does not name (a learner), or `sovereign_participation = "out"`
    /// (R605-F35). Its absence from the voter set is correct, not drift.
    NonVoter { group: String },
    /// A voter is lost and the survivors still make a majority.
    QuorumHolds {
        group: String,
        voters_before: usize,
        voters_after: usize,
        majority_needed: usize,
    },
    /// A voter is lost and the survivors do not. The group's raft stops
    /// accepting writes, which includes cluster secrets — so workloads there
    /// fail to resolve secrets even if their own containers are untouched.
    QuorumLost {
        group: String,
        voters_before: usize,
        voters_after: usize,
        majority_needed: usize,
    },
}

impl QuorumEffect {
    /// One line for a text report.
    pub fn headline(&self) -> String {
        match self {
            Self::NotInAGroup => "no sovereign group — votes in nothing".to_string(),
            Self::NonVoter { group } => {
                format!("no quorum seat in '{group}' (a learner, or out of its raft) — none to lose")
            }
            Self::QuorumHolds {
                group,
                voters_after,
                majority_needed,
                ..
            } => format!(
                "'{group}' quorum holds — {voters_after} voter(s) left, {majority_needed} needed"
            ),
            Self::QuorumLost {
                group,
                voters_after,
                majority_needed,
                ..
            } => format!(
                "'{group}' LOSES QUORUM — {voters_after} voter(s) left, {majority_needed} \
                 needed; the group's raft stops accepting writes, and cluster secrets are \
                 read from the local raft replica, so workloads there fail to resolve \
                 secrets even where their containers are untouched"
            ),
        }
    }
}

/// A machine whose declared capacity does not cover what is projected onto it.
///
/// Admission checks each workload against the node's `[allocatable]`
/// *individually* (`RequiredSpec::matches`, R572-F5) and never subtracts what
/// is already committed — so N workloads that each fit can all be admitted onto
/// a node that cannot hold their sum. This is the arithmetic nothing in the
/// tree does today.
#[derive(Debug, Clone, Serialize)]
pub struct Oversubscription {
    pub machine: String,
    pub allocatable: NodeAllocatable,
    pub committed_memory_mb: u32,
    pub committed_cpu_millis: u32,
    pub workloads: Vec<String>,
    pub memory_over: bool,
    pub cpu_over: bool,
}

// ─── The traversal ───────────────────────────────────────────────────────────

/// Walk the declared graph once and answer every question derivable from it.
///
/// Pure: same TOML in, same [`Topology`] out, no network. See the module header
/// for what is deliberately outside the model.
pub fn analyze(cfg: &CloudConfig) -> Topology {
    let workloads: Vec<WorkloadNode> = cfg
        .workloads
        .iter()
        .map(|w| workload_node(cfg, &w.spec))
        .collect();

    let mut machines: Vec<MachineNode> = cfg
        .machines
        .iter()
        .map(|m| MachineNode {
            name: m.name.clone(),
            region: m.region.clone(),
            sovereign_group: m.sovereign_group.clone(),
            raft_seat: RaftSeat::of(cfg, m),
            taints: m.taints.clone(),
            allocatable: m.allocatable.clone(),
            placed: Vec::new(),
            committed_memory_mb: 0,
            committed_cpu_millis: 0,
        })
        .collect();

    // Fold each workload's projected placement back onto its machine. Replicas
    // multiply the commitment: `replicas = 3` asks the node for three copies of
    // the request, and admission — which checks one workload against one node —
    // never sees that multiplication.
    let by_name: BTreeMap<String, usize> = machines
        .iter()
        .enumerate()
        .map(|(i, m)| (m.name.clone(), i))
        .collect();
    for w in &workloads {
        let Some(machine) = w.placement.machine() else {
            continue;
        };
        let Some(&i) = by_name.get(machine) else {
            continue;
        };
        let copies = w.replicas.max(1);
        machines[i].placed.push(w.name.clone());
        machines[i].committed_memory_mb = machines[i]
            .committed_memory_mb
            .saturating_add(w.memory_request_mb.saturating_mul(copies));
        machines[i].committed_cpu_millis = machines[i]
            .committed_cpu_millis
            .saturating_add(w.cpu_millis.saturating_mul(copies));
    }

    let oversubscribed = machines.iter().filter_map(oversubscription).collect();
    let node_losses = machines
        .iter()
        .map(|m| node_loss(cfg, &machines, &workloads, &m.name))
        .collect();

    Topology {
        machines,
        workloads,
        node_losses,
        oversubscribed,
    }
}

fn workload_node(cfg: &CloudConfig, spec: &WorkloadSpec) -> WorkloadNode {
    // One call into the admission seam — the same one `yah cloud apply` and
    // `yah cloud migrate` use. Forking a second selector here would let the
    // analyzer report a placement the fleet would never make.
    let placement = match cfg.admit_workload_candidates(spec) {
        Ok(candidates) => {
            let mut names = candidates.iter().map(|m| m.name.clone());
            let chosen = names
                .next()
                .expect("admit_workload_candidates never returns empty");
            Placement::Admitted {
                chosen,
                alternates: names.collect(),
                pinned: crate::config::node_selector_node(spec).is_some(),
            }
        }
        Err(e) => Placement::Unschedulable {
            reason: e.to_string(),
        },
    };

    let volumes = spec
        .volumes
        .iter()
        .map(|v| match &v.source {
            VolumeSource::Named { name } => VolumeDisposition::Copy {
                name: name.clone(),
                host_path: named_volume_path(name),
                mounted_at: v.target.clone(),
            },
            VolumeSource::Bind { host_path } => VolumeDisposition::Precondition {
                host_path: host_path.clone(),
                mounted_at: v.target.clone(),
            },
            VolumeSource::Tmpfs { size_mb } => VolumeDisposition::Discard {
                mounted_at: v.target.clone(),
                size_mb: *size_mb,
            },
        })
        .collect();

    let durability = match spec.durability() {
        Ok(Some(d)) => DurabilityView::Declared(d.clone()),
        Ok(None) => DurabilityView::Undeclared,
        Err(e) => DurabilityView::Malformed {
            reason: e.to_string(),
        },
    };

    WorkloadNode {
        name: spec.name.clone(),
        tier: spec.tier.0.clone(),
        mesh_identity: spec.fq_mesh_identity(),
        archetype: spec.effective_archetype(),
        replicas: spec.replicas,
        restart_policy: restart_policy_label(&spec.restart_policy),
        placement,
        volumes,
        durability,
        memory_request_mb: spec.memory_request_mb(),
        cpu_millis: spec.resources.cpu_millis,
        public_hostnames: spec
            .expose
            .public
            .iter()
            .map(|p| p.hostname.clone())
            .collect(),
        depends_on: spec.depends_on.iter().map(|d| d.0.clone()).collect(),
    }
}

fn restart_policy_label(p: &RestartPolicy) -> String {
    match p {
        RestartPolicy::Always => "always".to_string(),
        RestartPolicy::OnFailure { max_attempts, .. } => {
            format!("on-failure (max {max_attempts})")
        }
        RestartPolicy::Never => "never".to_string(),
    }
}

fn oversubscription(m: &MachineNode) -> Option<Oversubscription> {
    // No `[allocatable]` block is "unconstrained", exactly as admission reads
    // it — not "zero capacity". Reporting an unbounded node as oversubscribed
    // would flag every machine that has not been measured yet.
    let alloc = m.allocatable.as_ref()?;
    let memory_over = m.committed_memory_mb > alloc.memory_mb;
    let cpu_over = alloc.cpu_millis > 0 && m.committed_cpu_millis > alloc.cpu_millis;
    if !memory_over && !cpu_over {
        return None;
    }
    Some(Oversubscription {
        machine: m.name.clone(),
        allocatable: alloc.clone(),
        committed_memory_mb: m.committed_memory_mb,
        committed_cpu_millis: m.committed_cpu_millis,
        workloads: m.placed.clone(),
        memory_over,
        cpu_over,
    })
}

fn node_loss(
    cfg: &CloudConfig,
    machines: &[MachineNode],
    workloads: &[WorkloadNode],
    dead: &str,
) -> NodeLoss {
    let impacts: Vec<WorkloadImpact> = workloads
        .iter()
        .filter(|w| w.placement.machine() == Some(dead))
        .map(|w| workload_impact(w, dead, &cfg.recovery_measurements))
        .collect();

    let public_endpoints_lost = impacts
        .iter()
        .filter_map(|i| workloads.iter().find(|w| w.name == i.workload))
        .flat_map(|w| w.public_hostnames.iter().cloned())
        .collect();

    NodeLoss {
        machine: dead.to_string(),
        impacts,
        public_endpoints_lost,
        quorum: quorum_effect(cfg, machines, dead),
    }
}

fn workload_impact(
    w: &WorkloadNode,
    dead: &str,
    measurements: &BTreeMap<String, WorkloadRecovery>,
) -> WorkloadImpact {
    let alternates: Vec<String> = w
        .placement
        .alternates()
        .iter()
        .filter(|m| m.as_str() != dead)
        .cloned()
        .collect();

    let data_loss = data_loss(w);
    let outcome = outcome(w, &alternates, dead);
    let recovery = recovery(w, &data_loss, measurements);

    WorkloadImpact {
        workload: w.name.clone(),
        archetype: w.archetype,
        outcome,
        data_loss,
        recovery,
    }
}

/// The core verdict. Archetype decides it, because archetype is what yubaba
/// itself branches on — `drain_workloads` skips appliances (R572-F4), and
/// `migrate` orders its steps by the same split.
fn outcome(w: &WorkloadNode, alternates: &[String], dead: &str) -> Outcome {
    match w.archetype {
        LifecycleArchetype::Appliance => Outcome::PinnedAppliance {
            migrate_target: alternates.first().cloned(),
            // "Hardware-level kill" is the question being asked, so the source
            // side of migrate's stop → copy → start has nothing to read from.
            // A workload whose only durable mount is a named volume on the dead
            // box is the exact shape with no source; one with no durable state
            // has nothing to copy and so is not blocked on this.
            source_unreachable: w.has_named_volume(),
        },
        LifecycleArchetype::Job => Outcome::RunLost,
        LifecycleArchetype::Server => {
            if alternates.is_empty() {
                return Outcome::NowhereToGo {
                    reason: format!(
                        "{dead} is the only machine in the declared fleet that admits \
                         {} (archetype {}, {} MiB request, {} millicores)",
                        w.name,
                        w.archetype.taint_key(),
                        w.memory_request_mb,
                        w.cpu_millis,
                    ),
                };
            }
            if w.replicas > 1 {
                Outcome::DegradedButServing {
                    surviving_replicas: w.replicas - 1,
                }
            } else {
                Outcome::Reschedulable {
                    candidates: alternates.to_vec(),
                }
            }
        }
    }
}

fn data_loss(w: &WorkloadNode) -> DataLoss {
    let durable = w.durable_volumes();
    if durable.is_empty() {
        return if w.volumes.is_empty() {
            DataLoss::None
        } else {
            DataLoss::EphemeralOnly
        };
    }

    // A bind mount is operator-managed; the camp did not create the host path
    // and has no basis for a claim about it either way. Only say "total" about
    // volumes this camp is responsible for.
    if !w.has_named_volume() {
        return DataLoss::Unknown { volumes: durable };
    }

    match &w.durability {
        DurabilityView::Undeclared => DataLoss::Total {
            volumes: durable,
            because: "no durability.tier declared, so the yubaba-managed named volume \
                      at /var/lib/yah/kamaji/volumes/ is the only copy"
                .to_string(),
        },
        DurabilityView::Malformed { reason } => DataLoss::Total {
            volumes: durable,
            because: format!("the durability declaration cannot be read: {reason}"),
        },
        DurabilityView::Declared(d) => match d.tier {
            DurabilityTier::None => DataLoss::Total {
                volumes: durable,
                because: "durability.tier = \"none\" — deliberately no second copy".to_string(),
            },
            DurabilityTier::Snapshot | DurabilityTier::Dedup => DataLoss::Window {
                tier: d.tier,
                store: d.store.clone().unwrap_or_default(),
                rpo_seconds: None,
                rpo_is_default: false,
            },
            DurabilityTier::Stream => DataLoss::Window {
                tier: d.tier,
                store: d.store.clone().unwrap_or_default(),
                rpo_seconds: Some(d.rpo_seconds.unwrap_or(DEFAULT_STREAM_RPO_SECONDS)),
                rpo_is_default: d.rpo_seconds.is_none(),
            },
        },
    }
}

/// R850-T2: `measurements` is [`CloudConfig::recovery_measurements`], already
/// replayed off the local tree by [`CloudConfig::load`]. It reaches here as
/// declared data, not as a path — `analyze` does no I/O, holds no clock, and
/// keeps the same contract [`crate::migrate::plan_migration`] holds.
fn recovery(
    w: &WorkloadNode,
    loss: &DataLoss,
    measurements: &BTreeMap<String, WorkloadRecovery>,
) -> RecoveryEstimate {
    match loss {
        DataLoss::None | DataLoss::EphemeralOnly => RecoveryEstimate::Immediate,
        DataLoss::Total { .. } | DataLoss::Unknown { .. } => RecoveryEstimate::NotRecoverable,
        DataLoss::Window { .. } => {
            // A timed restore beats both the extrapolation and the "we can't
            // say" — a measurement answers the question `state_mb` was only ever
            // a proxy for. Age does not disqualify it; see `RecoveryEstimate`.
            if let Some(m) = measurements.get(&w.name) {
                return measured_recovery(m);
            }
            let state_mb = match &w.durability {
                DurabilityView::Declared(Durability {
                    state_mb: Some(mb), ..
                }) => *mb,
                _ => {
                    return RecoveryEstimate::UnknownStateSize {
                        hint: "durability.state_mb",
                    }
                }
            };
            RecoveryEstimate::Hydrate {
                state_mb,
                seconds: f64::from(state_mb) / MEASURED_HYDRATE_MB_PER_S,
                // The *floor*, and it says so. A tier-2 restore also replays
                // WAL frames, and `turso_backup::stream` fetches those one at a
                // time — roadcase measured `2n + 1` serialized GETs for `n`
                // generations, which at this RTT reaches the same order as the
                // bulk transfer itself. `n` is not declared anywhere, so it is
                // named rather than guessed at.
                basis: format!(
                    "bulk transfer at {MEASURED_HYDRATE_MB_PER_S} MB/s, measured R760-T10 \
                     2026-08-29 on one host against one backend. A FLOOR: a tier-2 restore \
                     adds 2n+1 serialized GETs at ~{MEASURED_GET_RTT_MS} ms each for n \
                     generations, and n is not declared anywhere"
                ),
            }
        }
    }
}

/// Turn one workload's replayed measurements into the estimate, provenance and
/// age included. Split out so the `basis` string lives next to the type that
/// justifies it rather than inside a `match` arm.
fn measured_recovery(m: &WorkloadRecovery) -> RecoveryEstimate {
    let measured_at = m.measured_at();
    RecoveryEstimate::Measured {
        seconds: m.seconds(),
        bytes: m.bytes(),
        measured_at,
        age_days: m.age_days(),
        node: m.node.clone(),
        basis: format!(
            "MEASURED, not extrapolated: {} subject(s) summed from a real restore recorded by \
             {} on {} at {}, replayed from `.yah/cloud/recovery.jsonl`. No \
             {MEASURED_HYDRATE_MB_PER_S} MB/s constant and no declared state-mb is involved. \
             Oldest component is {} day(s) old — the figure describes the state as it was then, \
             not as it is declared now.",
            m.subject_count(),
            m.helper(),
            m.node,
            measured_at.to_rfc3339(),
            m.age_days(),
        ),
    }
}

fn quorum_effect(cfg: &CloudConfig, machines: &[MachineNode], dead: &str) -> QuorumEffect {
    let Some(m) = machines.iter().find(|m| m.name == dead) else {
        return QuorumEffect::NotInAGroup;
    };
    let Some(group) = m.sovereign_group.clone() else {
        return QuorumEffect::NotInAGroup;
    };
    if m.raft_seat != RaftSeat::Voter {
        return QuorumEffect::NonVoter { group };
    }

    // R605-F35: the denominator is the group's listed voters, not its
    // members — co-located learners raise neither side of the arithmetic.
    let voters_before = cfg.seated_voters(&group);
    let voters_after = voters_before.saturating_sub(1);
    // Raft majority is over the *configured* membership, which the loss of a
    // box does not shrink — a dead voter still counts in the denominator until
    // someone removes it from the configuration.
    let majority_needed = voters_before / 2 + 1;

    if voters_after >= majority_needed {
        QuorumEffect::QuorumHolds {
            group,
            voters_before,
            voters_after,
            majority_needed,
        }
    } else {
        QuorumEffect::QuorumLost {
            group,
            voters_before,
            voters_after,
            majority_needed,
        }
    }
}

// ─── P2: renders ─────────────────────────────────────────────────────────────

impl Topology {
    /// Render the model as a Mermaid `flowchart`.
    ///
    /// **A projection, not a second traversal.** Every node and edge below is
    /// read off fields [`analyze`] already computed, so the picture cannot
    /// disagree with the verdicts printed beside it. A diagram generated by its
    /// own walk of the config would be free to drift into decoration, which is
    /// worse than no diagram — it is the failure mode this method's shape
    /// exists to make impossible.
    ///
    /// The edge worth the whole render is `-.->|hydrate|`: the backup path is
    /// the one relationship in this graph that is invisible in the TOML, has no
    /// runtime today, and is exactly what decides whether a node loss is an
    /// incident or a restore.
    pub fn to_mermaid(&self) -> String {
        let mut out = String::from("flowchart TB\n");

        // Machines, grouped by sovereign group. The grouping is the blast
        // radius (W305), so it is what a reader should see first.
        let mut groups: BTreeMap<Option<&str>, Vec<&MachineNode>> = BTreeMap::new();
        for m in &self.machines {
            groups
                .entry(m.sovereign_group.as_deref())
                .or_default()
                .push(m);
        }
        for (group, members) in &groups {
            let label = group.unwrap_or("ungrouped");
            out.push_str(&format!(
                "  subgraph grp_{}[\"{label}\"]\n",
                sanitize(label)
            ));
            for m in members {
                // R605-F35: the seat as the group's voter list decides it, so
                // a box in no group renders "no group" and a co-located VM
                // renders "learner", never "voter".
                let role = m.raft_seat.label();
                let cap = match &m.allocatable {
                    Some(a) => format!(
                        "<br/>{}/{} MiB · {}/{} mCPU",
                        m.committed_memory_mb, a.memory_mb, m.committed_cpu_millis, a.cpu_millis
                    ),
                    None => "<br/>no [allocatable] declared".to_string(),
                };
                out.push_str(&format!(
                    "    {}[\"{}<br/><i>{role}</i>{cap}\"]\n",
                    node_id("m", &m.name),
                    m.name
                ));
            }
            out.push_str("  end\n");
        }

        // Workloads, their mounts, and their public front doors.
        for w in &self.workloads {
            let wid = node_id("w", &w.name);
            out.push_str(&format!(
                "  {wid}(\"{}<br/><i>{}</i> · replicas {}\")\n",
                w.name,
                w.archetype.taint_key(),
                w.replicas
            ));

            match &w.placement {
                Placement::Admitted { chosen, pinned, .. } => {
                    let verb = if *pinned { "pinned" } else { "placed" };
                    out.push_str(&format!("  {} -->|{verb}| {wid}\n", node_id("m", chosen)));
                }
                Placement::Unschedulable { .. } => {
                    out.push_str(&format!(
                        "  unschedulable{{{{no node admits it}}}} --> {wid}\n"
                    ));
                }
            }

            for v in &w.volumes {
                let vid = node_id("v", &format!("{}-{}", w.name, v.label()));
                let (shape, edge) = match v {
                    VolumeDisposition::Copy { name, .. } => {
                        (format!("{vid}[(\"named: {name}\")]"), "mount")
                    }
                    VolumeDisposition::Precondition { host_path, .. } => (
                        format!("{vid}[(\"bind: {}\")]", host_path.display()),
                        "mount",
                    ),
                    VolumeDisposition::Discard { size_mb, .. } => {
                        (format!("{vid}[(\"tmpfs {size_mb} MiB\")]"), "ephemeral")
                    }
                };
                out.push_str(&format!("  {shape}\n  {wid} -->|{edge}| {vid}\n"));

                // The invisible edge. Only durable mounts can have one, and
                // only a declared tier draws it.
                if !v.is_durable() {
                    continue;
                }
                if let DurabilityView::Declared(d) = &w.durability {
                    if let Some(store) = &d.store {
                        let sid = node_id("s", store);
                        out.push_str(&format!("  {sid}[[\"{store}\"]]\n"));
                        out.push_str(&format!(
                            "  {vid} -.->|backup: {}| {sid}\n  {sid} -.->|hydrate| {vid}\n",
                            d.tier
                        ));
                    }
                }
            }

            for host in &w.public_hostnames {
                let hid = node_id("p", host);
                out.push_str(&format!("  {hid}>\"{host}\"]\n  {hid} ==>|public| {wid}\n"));
            }

            for dep in &w.depends_on {
                if let Some(target) = self.workloads.iter().find(|o| {
                    o.mesh_identity == *dep || o.mesh_identity.ends_with(&format!("/{dep}"))
                }) {
                    out.push_str(&format!(
                        "  {wid} -.->|mesh admit| {}\n",
                        node_id("w", &target.name)
                    ));
                }
            }
        }

        out
    }

    /// Render the survivability answer for one machine as plain text.
    ///
    /// Returns `None` when no machine by that name is declared — the caller
    /// owns the wording of that refusal, since it has the declared list.
    pub fn render_node_loss(&self, machine: &str) -> Option<String> {
        let loss = self.node_losses.iter().find(|l| l.machine == machine)?;
        let mut out = format!("If {machine} is lost at the hardware level:\n\n");
        out.push_str(&format!("  quorum: {}\n", loss.quorum.headline()));
        if loss.public_endpoints_lost.is_empty() {
            out.push_str("  public endpoints lost: none\n");
        } else {
            out.push_str(&format!(
                "  public endpoints lost: {}\n",
                loss.public_endpoints_lost.join(", ")
            ));
        }

        if loss.impacts.is_empty() {
            out.push_str("\n  No declared workload is projected onto this machine.\n");
            return Some(out);
        }

        for i in &loss.impacts {
            out.push_str(&format!(
                "\n  {} ({})\n",
                i.workload,
                i.archetype.taint_key()
            ));
            out.push_str(&format!("    what happens: {}\n", i.outcome.headline()));
            out.push_str(&format!("    data loss:    {}\n", i.data_loss.headline()));
            out.push_str(&format!("    recovery:     {}\n", i.recovery.headline()));
        }
        Some(out)
    }

    /// Render the capacity arithmetic — the whole fleet, oversubscription
    /// called out rather than left to the reader to spot.
    pub fn render_capacity(&self) -> String {
        let mut out = String::from("Declared capacity vs projected commitment:\n\n");
        for m in &self.machines {
            let placed = if m.placed.is_empty() {
                "(nothing)".to_string()
            } else {
                m.placed.join(", ")
            };
            match &m.allocatable {
                Some(a) => out.push_str(&format!(
                    "  {:<16} {:>6}/{:<6} MiB   {:>6}/{:<6} mCPU   {placed}\n",
                    m.name,
                    m.committed_memory_mb,
                    a.memory_mb,
                    m.committed_cpu_millis,
                    a.cpu_millis
                )),
                None => out.push_str(&format!(
                    "  {:<16} {:>6}/{:<6} MiB   {:>6}/{:<6} mCPU   {placed}\n",
                    m.name, m.committed_memory_mb, "?", m.committed_cpu_millis, "?"
                )),
            }
        }
        if self.oversubscribed.is_empty() {
            out.push_str("\nNo machine is oversubscribed against its declared [allocatable].\n");
        } else {
            out.push_str(
                "\nOVERSUBSCRIBED — admission checks each workload against a node \
                 individually and never subtracts what is already committed (R572-F5), \
                 so these all admitted and cannot all run:\n",
            );
            for o in &self.oversubscribed {
                let mut axes = Vec::new();
                if o.memory_over {
                    axes.push(format!(
                        "memory {} MiB > {} MiB",
                        o.committed_memory_mb, o.allocatable.memory_mb
                    ));
                }
                if o.cpu_over {
                    axes.push(format!(
                        "cpu {} > {} millicores",
                        o.committed_cpu_millis, o.allocatable.cpu_millis
                    ));
                }
                out.push_str(&format!(
                    "  {}: {} — {}\n",
                    o.machine,
                    axes.join(", "),
                    o.workloads.join(", ")
                ));
            }
        }
        out
    }

    /// Render every machine's loss, plus the capacity view — the default
    /// whole-fleet report.
    pub fn render(&self) -> String {
        let mut out = String::new();
        for m in &self.machines {
            if let Some(section) = self.render_node_loss(&m.name) {
                out.push_str(&section);
                out.push('\n');
            }
        }
        out.push_str(&self.render_capacity());
        out
    }
}

/// Mermaid node ids must be identifier-ish; machine and workload names are DNS
/// labels and hostnames, and buckets are URLs. One prefix per kind keeps two
/// different things that sanitize to the same string apart.
fn node_id(prefix: &str, name: &str) -> String {
    format!("{prefix}_{}", sanitize(name))
}

fn sanitize(name: &str) -> String {
    name.chars()
        .map(|c| if c.is_ascii_alphanumeric() { c } else { '_' })
        .collect()
}

#[cfg(test)]
mod tests {
    use super::*;
    use std::path::Path;
    use tempfile::{tempdir, TempDir};

    /// Fixtures go through `CloudConfig::load`, never a struct literal, for the
    /// reason `migrate`'s own fixture records: a hand-built spec can express a
    /// shape no operator could write, and `load_workloads` shape-validates —
    /// which since R850-P4 includes the durability declaration. A test that
    /// skips the loader would pass on a spec the fleet would reject.
    struct Camp {
        dir: TempDir,
    }

    impl Camp {
        fn new() -> Self {
            Self {
                dir: tempdir().unwrap(),
            }
        }

        fn root(&self) -> &Path {
            self.dir.path()
        }

        fn machine(self, name: &str, extra: &str) -> Self {
            let dir = self.root().join(".yah/infra/machines");
            std::fs::create_dir_all(&dir).unwrap();
            std::fs::write(
                dir.join(format!("{name}.toml")),
                format!(
                    "name = \"{name}\"\nprovider = \"static\"\nmesh_tags = []\n\
                     {extra}\n\
                     [connect]\naddress = \"10.0.0.1\"\nssh = \"root@{name}\"\n\
                     identity_file = \"~/.ssh/yah\"\n\
                     yubaba = \"http://{name}:7443\"\n"
                ),
            )
            .unwrap();
            self
        }

        /// `.yah/infra/sovereign-groups/<name>.toml` seating `voters`
        /// (R605-F35) — the declaration a quorum calculation counts.
        fn sovereign_group(self, name: &str, voters: &[&str]) -> Self {
            let dir = self.root().join(".yah/infra/sovereign-groups");
            std::fs::create_dir_all(&dir).unwrap();
            let seats: String = voters
                .iter()
                .enumerate()
                .map(|(i, m)| format!("[[voters]]\nmachine = \"{m}\"\nraft_node_id = {}\n", i + 1))
                .collect();
            std::fs::write(
                dir.join(format!("{name}.toml")),
                format!("name = \"{name}\"\n{seats}"),
            )
            .unwrap();
            self
        }

        /// A machine with a capacity budget, which is what makes the
        /// oversubscription and NowhereToGo cases expressible.
        fn sized(self, name: &str, memory_mb: u32, cpu_millis: u32, extra: &str) -> Self {
            self.machine(
                name,
                &format!(
                    "{extra}\n[allocatable]\nmemory_mb = {memory_mb}\ncpu_millis = {cpu_millis}"
                ),
            )
        }

        fn workload(self, name: &str, extra: &str) -> Self {
            self.workload_sized(name, 128, 100, extra)
        }

        fn workload_sized(self, name: &str, memory_mb: u32, cpu_millis: u32, extra: &str) -> Self {
            let dir = self.root().join(".yah/infra/workloads");
            std::fs::create_dir_all(&dir).unwrap();
            std::fs::write(
                dir.join(format!("{name}.toml")),
                format!(
                    "schema_version = 1\nname = \"{name}\"\ntier = \"infra\"\n\
                     restart_policy = \"always\"\n\
                     {extra}\n\
                     [image]\nregistry = \"cr.yah.dev\"\nrepository = \"{name}\"\n\
                     tag = \"v1\"\ndigest = \"sha256:abc\"\n\
                     [resources]\nmemory_mb = {memory_mb}\ncpu_millis = {cpu_millis}\n\
                     [stop_policy]\nsignal = 15\ngrace_period = 10000\n\
                     [expose.mesh]\nidentity = \"{name}\"\nports = [8080]\nallow_from = []\n"
                ),
            )
            .unwrap();
            self
        }

        /// Append a measured subject restore to `.yah/cloud/recovery.jsonl`
        /// (R850-T2), dated `days_ago` before now. Goes through the real
        /// journal writer so `analyze` sees exactly what `yah cloud topology
        /// --record-hydrate` would have left behind.
        fn measured(
            self,
            workload: &str,
            subject: &str,
            bytes: u64,
            seconds: f64,
            days_ago: i64,
        ) -> Self {
            crate::recovery_journal::RecoveryJournal::at_workspace(self.root())
                .append(&[crate::recovery_journal::RecoveryRecord {
                    at: Utc::now() - chrono::Duration::days(days_ago),
                    workload: workload.to_string(),
                    node: "a".to_string(),
                    tier: Some("stream".to_string()),
                    subject: subject.to_string(),
                    bytes,
                    seconds,
                    helper: crate::recovery_journal::HELPER_TURSO_BACKUP_HYDRATE.to_string(),
                }])
                .unwrap();
            self
        }

        fn analyze(&self) -> Topology {
            super::analyze(&CloudConfig::load(self.root()).expect("fixture camp must load"))
        }
    }

    /// The `[annotations]` block for a stream-tier workload, with `state-mb`
    /// declared unless `state_mb` is `None` — the two shapes that decide
    /// between `Hydrate` and `UnknownStateSize` when nothing was measured.
    fn stream_durability(state_mb: Option<u32>) -> String {
        let mut s = "[durability]\n\
                     tier = \"stream\"\n\
                     engine = \"turso\"\n\
                     store = \"s3://backups/db\"\n\
                     subjects = [\"accounts.db\"]\n"
            .to_string();
        if let Some(mb) = state_mb {
            s.push_str(&format!("state_mb = {mb}\n"));
        }
        s
    }

    const NAMED_VOLUME: &str = "[[volumes]]\nsource = { named = { name = \"accounts\" } }\n\
                                target = \"/var/lib/app\"\nread_only = false\n";

    fn impact<'a>(topo: &'a Topology, machine: &str, workload: &str) -> &'a WorkloadImpact {
        topo.node_losses
            .iter()
            .find(|l| l.machine == machine)
            .unwrap_or_else(|| panic!("no node_loss for {machine}"))
            .impacts
            .iter()
            .find(|i| i.workload == workload)
            .unwrap_or_else(|| panic!("{workload} is not projected onto {machine}"))
    }

    // ── The driving question ─────────────────────────────────────────────────

    /// R850's acceptance test, and the reason the relay exists.
    ///
    /// The noisetable-account shape verbatim: one replica, `archetype =
    /// "appliance"`, a yubaba-managed named volume, no durability tier
    /// declared. Hardware-kill the node. The analyzer has to say *out loud*
    /// that nothing reschedules, that the bytes are gone, and that even the
    /// operator verb has no source to copy from — because before this module
    /// existed, learning any of those three meant reading yubaba's source.
    #[test]
    fn hardware_killing_the_node_under_a_singleton_appliance_loses_everything() {
        let topo = Camp::new()
            .sized("us-west-001", 12288, 6000, "")
            .sized("us-west-003", 16384, 16000, "")
            .workload(
                "noisetable-account",
                &format!("replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}"),
            )
            .analyze();

        let i = impact(&topo, "us-west-001", "noisetable-account");

        // 1. Nothing reschedules — and the reason is the archetype, not a
        //    shortage of nodes. us-west-003 admits it fine and is irrelevant.
        let Outcome::PinnedAppliance {
            migrate_target,
            source_unreachable,
        } = &i.outcome
        else {
            panic!("expected PinnedAppliance, got {:?}", i.outcome);
        };
        assert_eq!(migrate_target.as_deref(), Some("us-west-003"));

        // 2. `yah cloud migrate` is named as the human step AND disclaimed:
        //    it plans a copy, and a dead box has nothing to copy from.
        assert!(source_unreachable);
        let headline = i.outcome.headline();
        assert!(headline.contains("yah cloud migrate"), "{headline}");
        assert!(headline.contains("no source"), "{headline}");

        // 3. Every account, passkey and session is gone.
        let DataLoss::Total { volumes, because } = &i.data_loss else {
            panic!("expected Total, got {:?}", i.data_loss);
        };
        assert_eq!(volumes, &["accounts".to_string()]);
        assert!(because.contains("durability.tier"), "{because}");
        assert_eq!(i.recovery, RecoveryEstimate::NotRecoverable);

        // 4. And the operator-facing text says all three without a source dive.
        let text = topo.render_node_loss("us-west-001").unwrap();
        assert!(text.contains("OPERATOR"), "{text}");
        assert!(text.contains("TOTAL"), "{text}");
        assert!(text.contains("nothing to recover from"), "{text}");
    }

    /// The same spec's verdict must not soften when the appliance is the only
    /// thing the fleet could hold — `migrate_target` goes to `None` and the
    /// outcome stays manual rather than degrading into "nowhere to go", which
    /// would read as a capacity problem instead of an archetype one.
    #[test]
    fn an_appliance_with_no_alternate_still_reads_as_pinned_not_as_capacity() {
        let topo = Camp::new()
            .sized("solo", 1024, 1000, "")
            .workload(
                "db",
                &format!("replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}"),
            )
            .analyze();

        let i = impact(&topo, "solo", "db");
        let Outcome::PinnedAppliance { migrate_target, .. } = &i.outcome else {
            panic!("expected PinnedAppliance, got {:?}", i.outcome);
        };
        assert_eq!(*migrate_target, None);
        assert!(i.outcome.is_manual());
    }

    // ── The fungible half ────────────────────────────────────────────────────

    #[test]
    fn a_stateless_server_with_another_admitting_node_is_automatic() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .sized("b", 4096, 4000, "")
            .workload("api", "replicas = 1\narchetype = \"server\"")
            .analyze();

        let i = impact(&topo, "a", "api");
        assert_eq!(
            i.outcome,
            Outcome::Reschedulable {
                candidates: vec!["b".to_string()]
            }
        );
        assert_eq!(i.data_loss, DataLoss::None);
        assert_eq!(i.recovery, RecoveryEstimate::Immediate);
        assert!(!i.outcome.is_manual());
    }

    /// The finding that is invisible in the TOML: a workload can be perfectly
    /// stateless and restartable and still be a single point of failure,
    /// because only one declared box clears its floor.
    #[test]
    fn a_stateless_server_that_only_one_node_admits_stays_down() {
        let topo = Camp::new()
            .sized("big", 8192, 8000, "")
            .sized("small", 512, 8000, "")
            .workload_sized("hungry", 4096, 100, "replicas = 1\narchetype = \"server\"")
            .analyze();

        let i = impact(&topo, "big", "hungry");
        let Outcome::NowhereToGo { reason } = &i.outcome else {
            panic!("expected NowhereToGo, got {:?}", i.outcome);
        };
        assert!(reason.contains("big"), "{reason}");
        assert!(i.outcome.is_manual());
    }

    #[test]
    fn replicas_above_one_degrade_rather_than_fail() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .sized("b", 4096, 4000, "")
            .workload("api", "replicas = 3\narchetype = \"server\"")
            .analyze();

        assert_eq!(
            impact(&topo, "a", "api").outcome,
            Outcome::DegradedButServing {
                surviving_replicas: 2
            }
        );
    }

    /// A `no-appliance` taint is absolute — there is no toleration anywhere in
    /// the tree (W305 finding 2) — so a tainted box must never show up as an
    /// alternate an appliance could fail over to.
    #[test]
    fn a_no_appliance_taint_removes_a_node_from_the_alternates() {
        let topo = Camp::new()
            .sized("prod", 8192, 8000, "")
            .sized("builder", 16384, 16000, "taints = [\"no-appliance\"]")
            .workload(
                "db",
                &format!("replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}"),
            )
            .analyze();

        let db = topo.workloads.iter().find(|w| w.name == "db").unwrap();
        assert_eq!(db.placement.machine(), Some("prod"));
        assert!(
            db.placement.alternates().is_empty(),
            "builder carries no-appliance and must not be offered: {:?}",
            db.placement
        );
    }

    #[test]
    fn a_job_is_a_lost_run_not_a_failover() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload("build", "replicas = 1\narchetype = \"job\"")
            .analyze();
        assert_eq!(impact(&topo, "a", "build").outcome, Outcome::RunLost);
    }

    // ── Durability ───────────────────────────────────────────────────────────

    #[test]
    fn a_declared_stream_tier_turns_total_loss_into_a_bounded_window() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n\
                     [durability]\n\
                     tier = \"stream\"\n\
                     engine = \"turso\"\n\
                     store = \"s3://backups/db\"\n\
                     subjects = [\"accounts.db\"]\n\
                     rpo_seconds = 30\n\
                     state_mb = 100\n"
                ),
            )
            .analyze();

        let i = impact(&topo, "a", "db");
        assert_eq!(
            i.data_loss,
            DataLoss::Window {
                tier: DurabilityTier::Stream,
                store: "s3://backups/db".to_string(),
                rpo_seconds: Some(30),
                rpo_is_default: false,
            }
        );

        // The recovery figure is roadcase's measured constant applied to the
        // declared size — 100 MiB at 32.6 MB/s ≈ 3.1 s, the same order as the
        // 3.3 s R760-T10 actually measured for 100 MB.
        let RecoveryEstimate::Hydrate {
            state_mb,
            seconds,
            basis,
        } = &i.recovery
        else {
            panic!("expected Hydrate, got {:?}", i.recovery);
        };
        assert_eq!(*state_mb, 100);
        assert!((*seconds - 3.067).abs() < 0.01, "{seconds}");
        // Provenance survives into the structured output, not just the text.
        assert!(basis.contains("R760-T10"), "{basis}");
        assert!(basis.contains("FLOOR"), "{basis}");
    }

    // ── Measured recovery (R850-T2) ──────────────────────────────────────────

    /// The baseline this feature must not disturb: a camp that has never timed
    /// a restore has no `.yah/cloud/recovery.jsonl`, that is not an error, and
    /// both pre-existing answers come back exactly as before.
    #[test]
    fn an_absent_recovery_journal_leaves_todays_answers_untouched() {
        for (state_mb, expected) in [
            (
                Some(100),
                RecoveryEstimate::Hydrate {
                    state_mb: 100,
                    seconds: 100.0 / MEASURED_HYDRATE_MB_PER_S,
                    // Compared field-wise below; `basis` is long and pinned by
                    // `a_declared_stream_tier_turns_total_loss_into_a_bounded_window`.
                    basis: String::new(),
                },
            ),
            (
                None,
                RecoveryEstimate::UnknownStateSize {
                    hint: "durability.state_mb",
                },
            ),
        ] {
            let camp = Camp::new().sized("a", 4096, 4000, "").workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n{}",
                    stream_durability(state_mb)
                ),
            );
            assert!(
                !crate::paths::recovery_journal(camp.root()).exists(),
                "fixture must have no journal"
            );

            let topo = camp.analyze();
            let got = &impact(&topo, "a", "db").recovery;
            match (&expected, got) {
                (
                    RecoveryEstimate::Hydrate {
                        state_mb: want_mb,
                        seconds: want_s,
                        ..
                    },
                    RecoveryEstimate::Hydrate {
                        state_mb, seconds, ..
                    },
                ) => {
                    assert_eq!(state_mb, want_mb);
                    assert!((seconds - want_s).abs() < 1e-9, "{seconds}");
                }
                (a, b) => assert_eq!(a, b),
            }
        }
    }

    /// The point of the ticket: one journal entry turns the extrapolation into
    /// a measurement, and the figure is the sum over the workload's subjects.
    #[test]
    fn a_journalled_restore_replaces_the_extrapolation_with_the_measured_sum() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n{}",
                    stream_durability(Some(100))
                ),
            )
            // 100 MiB declared would extrapolate to ~3.07s; the real restore
            // took 21s across two subjects, and that is what must be reported.
            .measured("db", "accounts.db", 1024 * 1024 * 60, 12.5, 2)
            .measured("db", "ledger.db", 1024 * 1024 * 40, 8.5, 2)
            .analyze();

        let i = impact(&topo, "a", "db");
        let RecoveryEstimate::Measured {
            seconds,
            bytes,
            age_days,
            node,
            basis,
            ..
        } = &i.recovery
        else {
            panic!("expected Measured, got {:?}", i.recovery);
        };
        assert!((*seconds - 21.0).abs() < 1e-9, "{seconds}");
        assert_eq!(*bytes, 1024 * 1024 * 100);
        assert_eq!(*age_days, 2);
        assert_eq!(node, "a");
        // Provenance survives into the structured output, not just the text —
        // and it says which of the two kinds of number this is.
        assert!(basis.contains("MEASURED, not extrapolated"), "{basis}");
        assert!(basis.contains("recovery.jsonl"), "{basis}");
        assert!(basis.contains("turso-backup-hydrate"), "{basis}");
        // The extrapolation's constant must not appear as if it were involved.
        assert!(!basis.contains("R760-T10"), "{basis}");

        let headline = i.recovery.headline();
        assert!(headline.contains("MEASURED"), "{headline}");
        assert!(headline.contains("21.0s"), "{headline}");
        assert!(!headline.contains("extrapolated"), "{headline}");
    }

    /// A measurement is never discarded for being old: an extrapolation from
    /// one unrelated host is not an improvement on a real restore. The age goes
    /// into the headline instead, so the reader discounts it themselves.
    #[test]
    fn a_stale_measurement_is_still_reported_and_says_so() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n{}",
                    stream_durability(Some(100))
                ),
            )
            .measured("db", "accounts.db", 1024 * 1024 * 100, 41.0, 90)
            .analyze();

        let i = impact(&topo, "a", "db");
        let RecoveryEstimate::Measured {
            seconds, age_days, ..
        } = &i.recovery
        else {
            panic!("stale must stay Measured, not fall back to Hydrate: {:?}", i.recovery);
        };
        assert!((*seconds - 41.0).abs() < 1e-9, "{seconds}");
        assert_eq!(*age_days, 90);
        assert!(*age_days > recovery_journal::STALE_AFTER_DAYS);

        let headline = i.recovery.headline();
        assert!(headline.contains("measured 90 days ago"), "{headline}");
        assert!(headline.contains("may have grown since"), "{headline}");
    }

    /// A measurement answers the question `state-mb` was only ever a proxy for,
    /// so it beats `UnknownStateSize` too — an undeclared size is no longer a
    /// reason to refuse an answer once a real restore has been timed.
    #[test]
    fn a_measurement_beats_an_undeclared_state_size() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n{}",
                    stream_durability(None)
                ),
            )
            .measured("db", "accounts.db", 4096, 1.5, 0)
            .analyze();

        assert!(matches!(
            impact(&topo, "a", "db").recovery,
            RecoveryEstimate::Measured { .. }
        ));
    }

    /// Attribution is per workload. A measurement filed against one workload
    /// must not leak into another's estimate, and must not turn a stateless
    /// workload into a recoverable one.
    #[test]
    fn a_measurement_for_one_workload_does_not_leak_into_another() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n{}",
                    stream_durability(Some(100))
                ),
            )
            .workload("web", "replicas = 1\narchetype = \"server\"")
            .measured("web", "accounts.db", 4096, 99.0, 1)
            .analyze();

        // `db` has its own Window and no measurement of its own.
        assert!(matches!(
            impact(&topo, "a", "db").recovery,
            RecoveryEstimate::Hydrate { .. }
        ));
        // `web` is stateless; a stray measurement does not invent state for it.
        assert_eq!(
            impact(&topo, "a", "web").recovery,
            RecoveryEstimate::Immediate
        );
    }

    /// An undeclared RPO on a stream tier is the turso-backup default, and the
    /// report has to say which of the two it is printing — a number the
    /// operator believes they chose is worse than no number.
    #[test]
    fn an_undeclared_rpo_is_labelled_as_the_default_not_as_a_choice() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n\
                     [durability]\n\
                     tier = \"stream\"\n\
                     engine = \"turso\"\n\
                     store = \"s3://backups/db\"\n\
                     subjects = [\"accounts.db\"]\n"
                ),
            )
            .analyze();

        let i = impact(&topo, "a", "db");
        let DataLoss::Window {
            rpo_seconds,
            rpo_is_default,
            ..
        } = &i.data_loss
        else {
            panic!("expected Window, got {:?}", i.data_loss);
        };
        assert_eq!(*rpo_seconds, Some(DEFAULT_STREAM_RPO_SECONDS));
        assert!(rpo_is_default);
        assert!(i.data_loss.headline().contains("not declared"));

        // No declared state size ⇒ no invented duration.
        assert_eq!(
            i.recovery,
            RecoveryEstimate::UnknownStateSize {
                hint: "durability.state_mb"
            }
        );
    }

    /// A snapshot tier has a copy but no recovery point the spec can state,
    /// and saying "≤ 120s" for it would be a fabrication.
    #[test]
    fn a_snapshot_tier_reports_an_unbounded_window_rather_than_borrowing_the_stream_rpo() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n\
                     [durability]\n\
                     tier = \"snapshot\"\n\
                     engine = \"turso\"\n\
                     store = \"s3://backups/db\"\n\
                     subjects = [\"accounts.db\"]\n"
                ),
            )
            .analyze();

        let i = impact(&topo, "a", "db");
        let DataLoss::Window { rpo_seconds, .. } = &i.data_loss else {
            panic!("expected Window, got {:?}", i.data_loss);
        };
        assert_eq!(*rpo_seconds, None);
        assert!(i.data_loss.headline().contains("unbounded by the spec"));
    }

    /// `tier = "none"` still loses everything — but as a decision, and the
    /// report must not read the same as the case where nobody looked.
    #[test]
    fn a_deliberate_none_tier_reads_differently_from_an_undeclared_one() {
        let camp = Camp::new().sized("a", 4096, 4000, "").workload(
            "cache",
            &format!(
                "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n\
                 [durability]\ntier = \"none\"\n"
            ),
        );
        let topo = camp.analyze();
        let DataLoss::Total { because, .. } = &impact(&topo, "a", "cache").data_loss else {
            panic!("expected Total");
        };
        assert!(because.contains("deliberately"), "{because}");
        assert!(
            !because.contains("no durability.tier declared"),
            "{because}"
        );
    }

    /// A bind mount is operator-managed. The camp did not create the host path
    /// and has no basis for saying the bytes are gone — the same refusal to
    /// guess `migrate::preconditions` makes about the same mount kind.
    #[test]
    fn a_bind_mount_is_unknown_rather_than_total() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "svc",
                "replicas = 1\narchetype = \"appliance\"\n\
                 [[volumes]]\nsource = { bind = { host_path = \"/srv/data\" } }\n\
                 target = \"/data\"\nread_only = false\n",
            )
            .analyze();

        let i = impact(&topo, "a", "svc");
        assert!(matches!(i.data_loss, DataLoss::Unknown { .. }));
        assert!(i.data_loss.headline().contains("UNKNOWN"));
    }

    #[test]
    fn tmpfs_only_is_not_a_data_loss() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "svc",
                "replicas = 1\narchetype = \"server\"\n\
                 [[volumes]]\nsource = { tmpfs = { size_mb = 64 } }\n\
                 target = \"/scratch\"\nread_only = false\n",
            )
            .analyze();
        assert_eq!(impact(&topo, "a", "svc").data_loss, DataLoss::EphemeralOnly);
    }

    // ── Capacity (P3) ────────────────────────────────────────────────────────

    /// The gap admission structurally cannot see: `RequiredSpec::matches`
    /// checks one workload against one node's `[allocatable]` and never
    /// subtracts what is already committed, so three workloads that each fit
    /// are each admitted onto a node that cannot hold their sum.
    #[test]
    fn workloads_that_each_fit_can_still_oversubscribe_the_node_they_all_land_on() {
        let topo = Camp::new()
            .sized("small", 1024, 4000, "")
            .workload_sized("a", 512, 100, "replicas = 1\narchetype = \"server\"")
            .workload_sized("b", 512, 100, "replicas = 1\narchetype = \"server\"")
            .workload_sized("c", 512, 100, "replicas = 1\narchetype = \"server\"")
            .analyze();

        assert_eq!(topo.oversubscribed.len(), 1);
        let o = &topo.oversubscribed[0];
        assert_eq!(o.machine, "small");
        assert_eq!(o.committed_memory_mb, 1536);
        assert!(o.memory_over);
        assert!(!o.cpu_over);
        assert!(topo.render_capacity().contains("OVERSUBSCRIBED"));
    }

    /// Replicas multiply the ask. Admission checks one copy against one node
    /// and never multiplies, so `replicas = 4` of a fitting workload is exactly
    /// the shape that admits cleanly and cannot run.
    #[test]
    fn replicas_multiply_the_commitment() {
        let topo = Camp::new()
            .sized("small", 1024, 4000, "")
            .workload_sized("api", 512, 100, "replicas = 4\narchetype = \"server\"")
            .analyze();

        assert_eq!(topo.machines[0].committed_memory_mb, 2048);
        assert_eq!(topo.oversubscribed.len(), 1);
    }

    /// A node with no `[allocatable]` is *unconstrained*, exactly as admission
    /// reads it — not a zero-capacity node. Flagging it would light up every
    /// machine nobody has measured yet and train the operator to ignore this.
    #[test]
    fn a_node_with_no_allocatable_block_is_never_reported_oversubscribed() {
        let topo = Camp::new()
            .machine("unmeasured", "")
            .workload_sized("api", 99999, 99999, "replicas = 1\narchetype = \"server\"")
            .analyze();

        assert!(topo.oversubscribed.is_empty());
        assert!(topo.render_capacity().contains("unmeasured"));
    }

    // ── Quorum ───────────────────────────────────────────────────────────────

    #[test]
    fn losing_one_of_three_voters_keeps_quorum() {
        let topo = Camp::new()
            .machine(
                "a",
                "sovereign_group = \"prod\"",
            )
            .machine(
                "b",
                "sovereign_group = \"prod\"",
            )
            .machine(
                "c",
                "sovereign_group = \"prod\"",
            )
            .sovereign_group("prod", &["a", "b", "c"])
            .analyze();

        let q = &topo.node_losses[0].quorum;
        assert_eq!(
            *q,
            QuorumEffect::QuorumHolds {
                group: "prod".to_string(),
                voters_before: 3,
                voters_after: 2,
                majority_needed: 2,
            }
        );
    }

    /// The second half of a node loss that is easy to miss: cluster secrets are
    /// read from the *local raft replica*, so a group that loses quorum fails
    /// workloads whose containers were never touched.
    #[test]
    fn losing_one_of_two_voters_loses_quorum_and_says_what_that_costs() {
        let topo = Camp::new()
            .machine(
                "a",
                "sovereign_group = \"prod\"",
            )
            .machine(
                "b",
                "sovereign_group = \"prod\"",
            )
            .sovereign_group("prod", &["a", "b"])
            .analyze();

        let q = &topo.node_losses[0].quorum;
        assert!(matches!(q, QuorumEffect::QuorumLost { .. }), "{q:?}");
        let headline = q.headline();
        assert!(headline.contains("LOSES QUORUM"), "{headline}");
        assert!(headline.contains("secrets"), "{headline}");
    }

    /// An `out` member's absence from `/raft/status` is correct, not drift
    /// (`workload_spec::sovereign`), so losing one costs no seat.
    #[test]
    fn an_out_member_has_no_seat_to_lose() {
        let topo = Camp::new()
            .machine(
                "a",
                "sovereign_group = \"prod\"",
            )
            .machine(
                "w",
                "sovereign_group = \"prod\"\nsovereign_participation = \"out\"",
            )
            .sovereign_group("prod", &["a"])
            .analyze();

        let q = &topo
            .node_losses
            .iter()
            .find(|l| l.machine == "w")
            .unwrap()
            .quorum;
        assert_eq!(
            *q,
            QuorumEffect::NonVoter {
                group: "prod".to_string()
            }
        );
        let w = topo.machines.iter().find(|m| m.name == "w").unwrap();
        assert_eq!(w.raft_seat, RaftSeat::Out);
    }

    /// R605-F35: a participant the group does not list is a learner. Losing
    /// it costs no seat, and it is not counted in the denominator — so a dev
    /// group of 3 listed Pis plus co-located VM learners still loses quorum
    /// only on its second Pi, not sooner.
    #[test]
    fn an_unlisted_participant_is_a_learner_and_not_in_the_denominator() {
        let topo = Camp::new()
            .machine("pi1", "sovereign_group = \"dev\"")
            .machine("pi2", "sovereign_group = \"dev\"")
            .machine("pi3", "sovereign_group = \"dev\"")
            .machine("vm1", "sovereign_group = \"dev\"\nhost_machine = \"pi1\"")
            .machine("vm2", "sovereign_group = \"dev\"\nhost_machine = \"pi2\"")
            .sovereign_group("dev", &["pi1", "pi2", "pi3"])
            .analyze();

        let quorum_of = |name: &str| {
            topo.node_losses
                .iter()
                .find(|l| l.machine == name)
                .unwrap()
                .quorum
                .clone()
        };
        assert_eq!(
            quorum_of("vm1"),
            QuorumEffect::NonVoter {
                group: "dev".to_string()
            }
        );
        assert_eq!(
            quorum_of("pi1"),
            QuorumEffect::QuorumHolds {
                group: "dev".to_string(),
                voters_before: 3,
                voters_after: 2,
                majority_needed: 2,
            }
        );
        let mermaid = topo.to_mermaid();
        assert!(mermaid.contains("<i>learner</i>"), "{mermaid}");
    }

    /// A group stamp with no `.yah/infra/sovereign-groups/` file seats nobody:
    /// the camp never reads a voter the cluster would not promote.
    #[test]
    fn a_group_with_no_declaration_seats_nobody() {
        let topo = Camp::new()
            .machine("a", "sovereign_group = \"prod\"")
            .analyze();
        assert_eq!(
            topo.node_losses[0].quorum,
            QuorumEffect::NonVoter {
                group: "prod".to_string()
            }
        );
    }

    // ── P2: the render is a projection ───────────────────────────────────────

    /// The scope trap this relay was warned about: a diagram produced by its
    /// own walk of the config can disagree with the analysis printed beside it.
    /// This pins that it cannot — the placement edge in the Mermaid output is
    /// the placement the model computed, and the hydrate edge exists exactly
    /// when the model found a durability tier.
    #[test]
    fn the_mermaid_render_agrees_with_the_model_it_projects() {
        let topo = Camp::new()
            .sized("us-west-001", 12288, 6000, "sovereign_group = \"prod\"")
            .sized("us-west-003", 16384, 16000, "taints = [\"no-appliance\"]")
            .workload(
                "backed-up",
                &format!(
                    "replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}\n\
                     [durability]\n\
                     tier = \"stream\"\n\
                     engine = \"turso\"\n\
                     store = \"s3://backups/db\"\n\
                     subjects = [\"accounts.db\"]\n"
                ),
            )
            .analyze();

        let mermaid = topo.to_mermaid();
        let w = topo
            .workloads
            .iter()
            .find(|w| w.name == "backed-up")
            .unwrap();

        // The edge names the machine the model chose, not a re-derived one.
        assert_eq!(w.placement.machine(), Some("us-west-001"));
        assert!(
            mermaid.contains("m_us_west_001 -->|placed| w_backed_up"),
            "{mermaid}"
        );
        // The blast radius is what a reader sees first.
        assert!(mermaid.contains("subgraph grp_prod[\"prod\"]"), "{mermaid}");
        // The edge that is invisible in the TOML.
        assert!(mermaid.contains("|hydrate|"), "{mermaid}");
        assert!(mermaid.contains("s3://backups/db"), "{mermaid}");
        // Committed-vs-allocatable rides the same numbers as render_capacity.
        assert!(mermaid.contains("128/12288 MiB"), "{mermaid}");
    }

    /// The negative half: no declared tier means no hydrate edge. A diagram
    /// that drew one anyway would show a safety property that does not exist,
    /// which is the single worst thing this render could do.
    #[test]
    fn no_declared_tier_draws_no_hydrate_edge() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "db",
                &format!("replicas = 1\narchetype = \"appliance\"\n{NAMED_VOLUME}"),
            )
            .analyze();

        let mermaid = topo.to_mermaid();
        assert!(mermaid.contains("v_db_accounts"), "{mermaid}");
        assert!(!mermaid.contains("hydrate"), "{mermaid}");
    }

    #[test]
    fn a_public_hostname_is_an_edge_and_is_reported_lost_with_its_node() {
        let topo = Camp::new()
            .sized("a", 4096, 4000, "")
            .workload(
                "web",
                "replicas = 1\narchetype = \"server\"\n\
                 [expose.public]\nhostname = \"app.example.com\"\nport = 8080\n\
                 tls = \"cf_managed\"\n",
            )
            .analyze();

        assert_eq!(
            topo.node_losses[0].public_endpoints_lost,
            vec!["app.example.com".to_string()]
        );
        assert!(topo.to_mermaid().contains("|public|"));
        assert!(topo
            .render_node_loss("a")
            .unwrap()
            .contains("public endpoints lost: app.example.com"));
    }

    /// A box that declares no group must not render as a voter — a quorum
    /// seat in a quorum that does not exist. Caught against the live camp
    /// (under R605-F12's defaulted role), where us-west-002 and us-west-015
    /// declare no group.
    #[test]
    fn an_ungrouped_machine_is_not_labelled_a_voter() {
        let mermaid = Camp::new()
            .sized("loner", 1024, 1000, "")
            .analyze()
            .to_mermaid();
        assert!(mermaid.contains("<i>no group</i>"), "{mermaid}");
        assert!(!mermaid.contains("<i>voter</i>"), "{mermaid}");
    }

    #[test]
    fn an_undeclared_machine_has_no_node_loss_section() {
        let topo = Camp::new().sized("a", 4096, 4000, "").analyze();
        assert!(topo.render_node_loss("typo").is_none());
        assert!(topo.render_node_loss("a").is_some());
    }
}