yah-cloud 0.8.43

Declarative cloud substrate for yah-managed camps: .yah/cloud/ config schema, MachineProvider drivers (Hetzner + local containerd), cloud-init rendering, and the pond/mesofact reconcilers.
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
//! Reconciler abstraction — bring a workload up against a mirror's
//! provider slots.
//!
//! A reconciler is the kind-specific code that knows how to deploy one
//! workload kind (`mesofact-static`, `container`, future `almanac`, …) to a
//! mirror. Selection: a [`ServiceComponent`](crate::ServiceComponent)'s
//! `kind` field picks which reconciler runs; the reconciler then dispatches
//! on the mirror's provider slot (e.g. `mesofact-static` →
//! `providers.static` slot → `miniflare-native` inline or `cloudflare` ref).
//!
//! [`MesofactStaticReconciler`] serves all four tiers through one door — the
//! compiled Worker bundle — and varies only the object store beneath it:
//! the `yah-s3-fs` driver at dev ([`dev_door`]), MinIO at pond ([`pond`]),
//! R2 at cloud and ha. Which one a mirror binds is declared in
//! `[drivers.s3]`, never branched on (W265, R584-F4).
//!
//!
//! @yah:ticket(R419-F2, "Implement CloudflareWorkerReconciler (kind=cloudflare-worker)")
//! @yah:assignee(agent:claude)
//! @yah:at(2026-06-03T08:02:49Z)
//! @yah:status(review)
//! @yah:parent(R419)
//! @yah:handoff("Landed CloudflareWorkerReconciler in crates/yah/cloud/src/reconciler/cloudflare_worker.rs and re-exported it through reconciler/mod.rs + cloud/src/lib.rs. up() validates the registry slot (must be `use = \"cloudflare\"` with non-empty `zone` + `domain`), reads workload.toml's `[build]` + `[[bindings]]`, matches each binding against a sibling mirror slot by `binding` field, runs build, reads the bundled entrypoint (default dist/index.js), idempotently lists+creates R2 buckets, deploys via deploy_worker_script with WorkerBinding::R2Bucket entries (R419-F1 surface), then attaches the custom domain via the new upsert_worker_custom_domain method on CloudflareClient. Returns RunningWorkload::adopted with public_url=https://<domain>. All config validation runs BEFORE any CF API call (R330-B5 fail-fast discipline).")
//! @yah:handoff("Added CloudflareClient::upsert_worker_custom_domain (cloudflare.rs) — separate from upsert_worker_route because Worker Routes are zone-scoped pattern matches and Custom Domains are account-scoped hostname attachments. List-first idempotency: skips PUT when the (hostname, service, zone_id) tuple is already bound.")
//! @yah:verify("cargo check -p cloud --lib — clean")
//! @yah:depends_on(R419-F1)
//!
//! @yah:ticket(R419-F3, "Register cloudflare-worker reconciler in CLI + desktop dispatch")
//! @yah:assignee(agent:claude)
//! @yah:at(2026-06-03T08:02:58Z)
//! @yah:status(review)
//! @yah:parent(R419)
//! @yah:handoff("Added match arm `\"cloudflare-worker\" => CloudflareWorkerReconciler::new().up(ctx)` in both dispatchers: app/yah/cli/src/cloud.rs:reconcile_component and app/yah/desktop/src/mirror_run.rs's component.kind.as_str() match. Imported CloudflareWorkerReconciler at the top of each file. Updated the desktop file-level docstring to list the new kind. Pre-existing fallback arm still produces a clean error for unknown kinds.")
//! @yah:verify("cargo check -p cloud --lib — clean")
//! @yah:verify("cargo check -p yah --lib --bins — clean (warnings unchanged from baseline)")
//! @yah:verify("cargo check -p desktop --lib — clean (warnings unchanged from baseline)")
//! @yah:depends_on(R419-F2)
//!
//! @yah:ticket(R419-F4, "Regression tests: misconfig fail-fast for cloudflare-worker reconciler")
//! @yah:assignee(agent:claude)
//! @yah:at(2026-06-03T08:03:09Z)
//! @yah:status(review)
//! @yah:parent(R419)
//! @yah:handoff("Three fail-fast tests in reconciler::cloudflare_worker::tests — Fixture builds an in-tempdir yah-cr-shaped workspace (writes workload.toml + .yah/infra/providers/cloudflare.toml). up_bails_on_registry_missing_domain (case 3) drops domain off the registry slot. up_bails_on_binding_name_drift (case 2) puts binding=\"STORAGE\" in the cache slot while workload.toml binds CACHE. up_bails_on_cache_slot_missing_bucket (case 1) keeps binding=\"CACHE\" but omits bucket. Each test asserts the error message names the offending field, the slot role, the service, and the env — no CF HTTP call is made because validation runs before any client construction.")
//! @yah:verify("cargo test -p cloud --lib reconciler::cloudflare_worker — 3 passed")
//! @yah:verify("cargo test -p cloud --lib — 246 passed (1 pre-existing failure cloud_init::tests::embedded_template_matches_workspace_canonical is unrelated R092-F2 template drift, not caused by R419)")
//! @yah:depends_on(R419-F2)
//!
//! @yah:relay(R458, "Cloud reconciler for .yah/domains/*.toml — R2 custom-domain shape")
//! @yah:at(2026-06-05T08:40:57Z)
//! @yah:next("F1: implement ensure_r2_custom_domain (mirror of ensure_r2_bucket) + wire into yah cloud apply as a post-services pass. Scope: domains with cdn_bucket set and no [[routes]] (today: cdn-yah-dev.toml). Worker-routed shape (yah-dev, app-yah-dev) is a separate surface.")
//! @arch:see(.yah/domains/cdn-yah-dev.toml)
//!
//! @yah:ticket(R458-F1, "ensure_r2_custom_domain (CF API) + apply-time orchestration")
//! @yah:at(2026-06-05T08:41:07Z)
//! @yah:status(review)
//! @yah:assignee(agent:claude)
//! @yah:parent(R458)
//! @arch:see(.yah/domains/cdn-yah-dev.toml)
//! @yah:next("R458 can be archived once F1 is signed off, unless we want to keep it open for the routed-domain (Worker) reconciler shape — that's a much larger surface (DNS + Worker route management + bundle deploy) than this F1's bucket-binding.")
//! @yah:handoff("Live-verified end-to-end. cdn.yah.dev now resolves to CF anycast IPs (104.21.43.100, 172.67.178.4) and curl HTTP/2 200s against https://cdn.yah.dev/yah-desktop/whisper/distil-large-v3-q5_1.bin (content-length 584567555 = the q5_1 bytes R422-F11 published). Second apply is idempotent — the list-first path skips the POST when the binding is already present. Files touched: (1) crates/yah/cloud/src/provider/cloudflare.rs — new R2CustomDomain output type + CloudflareClient::list_r2_custom_domains and CloudflareClient::add_r2_custom_domain methods (GET / POST /accounts/{id}/r2/buckets/{bucket}/domains/custom). The POST body needs zone_id even though the endpoint is bucket-scoped (CF rejects with 'JSON not well formed' otherwise — caught live and added on the second iteration). Re-exported through provider/mod.rs + cloud/src/lib.rs. (2) crates/yah/cloud/src/reconciler/domain.rs — new module with ensure_r2_custom_domain(account_id, bucket, domain) mirroring static_asset::ensure_r2_bucket's list-first idempotency. Resolves the parent zone id via the existing CloudflareClient::zone_id_for_name and a parent_zone_name(domain) heuristic that takes the last two labels (correct for every yah-owned zone today; the doc comment names the longest-suffix-match upgrade path for future three-label-apex zones). 3 unit tests cover apex / subdomain / deeper-subdomain. (3) crates/yah/cloud/src/reconciler/mod.rs — declared `pub mod domain` + re-exported ensure_r2_custom_domain. (4) app/yah/cli/src/cloud.rs — new DomainOutcome enum + a post-services domain pass in handle_apply that walks cfg.domains, dispatches the R2-custom-domain shape (cdn_bucket set, no [[routes]]), and routes routed-shape domains (yah-dev, app-yah-dev) into a Skipped row labelled 'has [[routes]] — Worker-routed shape'. Gated on a cloudflare provider being declared (pond-only setups print 'skip domain pass: no cloudflare provider declared'). Originally also gated on empty --service filter; that gate dropped on review — domains are workspace-scoped and the operator wants them reconciled even when narrowing services. New print_domain_summary mirrors print_apply_summary's table + JSON output. Required CF token scopes (verified live): Workers R2 Storage: Edit + Zone: Read. The existing cloudflare-api-token slot carries both.\n\nLive verification command + transcript:\n\n  $ ./target/debug/yah cloud apply --env cloud --service yah-desktop\n  ==> yah-desktop/cloud: reconciling 2 component(s)\n      component desktop (kind=binary)\n      component whisper-models (kind=static-asset)\n  ==> domain cdn-yah-dev (cdn.yah.dev): ensuring R2 custom-domain binding on bucket yah-dev\n  apply summary (cloud):\n    yah-desktop  ok       2 component(s) reconciled\n  domain summary (cloud):\n    cdn-yah-dev  ok       R2 custom domain bound\n  $ dig +short cdn.yah.dev\n  104.21.43.100\n  172.67.178.4\n  $ curl -sI https://cdn.yah.dev/yah-desktop/whisper/distil-large-v3-q5_1.bin | head -4\n  HTTP/2 200\n  content-length: 584567555\n\nUnblocks R422-T13's client-side `cdn_fallback = \"https://cdn.yah.dev/yah-desktop/whisper/{blake3}\"` — the URL now actually resolves and serves the bytes.")
//! @yah:verify("cargo test -p cloud --lib reconciler::domain --locked  # 3 pass")
//! @yah:verify("cargo check --workspace --locked  # clean")
//! @yah:verify("./target/debug/yah cloud apply --env cloud --service yah-desktop  # domain summary shows cdn-yah-dev=ok, yah-dev/app-yah-dev=skipped (routed)")
//! @yah:verify("curl -sI https://cdn.yah.dev/yah-desktop/whisper/distil-large-v3-q5_1.bin  # HTTP/2 200, content-length 584567555")
//!
//! @yah:ticket(R870-B25, "The cloud-init template drift guard is vacuously green, and the canonical mirror.yml it should guard is 128 lines stale")
//! @yah:status(review)
//! @yah:at(2026-09-11T00:24:58Z)
//! @yah:assignee(agent:bundle-anthropic-ashguard)
//! @yah:parent(R870)
//! @yah:severity(high)
//! @yah:next("OPERATIONAL QUESTION THIS RAISES, worth answering separately from the code fix: which is the provisioning path actually used, the embedded template or the repo-root canonical one? If any node was provisioned from the canonical copy since R858-F17 landed, it has no turso-backup helpers and its durability-declaring workloads will refuse to deploy. `rendered_runcmd_entries_are_all_strings` is the gate that genuinely catches the colon-space footgun and IS green — this ticket is about the twin-file guard beside it, not that one.")
//! @yah:verify("The drift guard must FAIL on today's tree (proving it now compares something real), then pass once the canonical `.yah/infra/cloud-init/mirror.yml` is brought level with the embedded template. Asserting only the post-fix green is the weak form — it passes for the same vacuous reason it does today.")
//! @yah:gotcha("Found by independent verification of R870-F23 phase 2 (@Ashguard:dove, session:60d4f41f), confirming a claim from @Ashguard:blade's implementation pass. NOT caused by R870-F23 — pre-existing, and F23's own change is green and unaffected. Do not read this as a phase-2 regression.")
//! @yah:next("THE DEFECT, read not inferred. `embedded_template_matches_workspace_canonical` exists to prove the mirror.yml compiled into the binary matches the canonical copy on disk. It resolves the workspace root by walking `CARGO_MANIFEST_DIR.ancestors()`, which lands on `oss/yubaba` — an independent Cargo workspace whose `.yah/` holds only a `.gitignore`. The canonical path therefore does not exist, the test takes its `canonical_path.exists()` bootstrap branch, and asserts NOTHING. It has been green for that reason, not because the files agree.")
//! @yah:next("WHAT THE GUARD IS MISSING, measured: the repo-root `.yah/infra/cloud-init/mirror.yml` is 128 diff-lines behind `oss/yubaba/crates/cloud/templates/mirror.yml`. It is missing the ENTIRE R858-F17 turso-backup block, and the YAML-quoting fix R870-F23 landed in the embedded copy (the two runcmd entries whose bare `: ` made cloud-init parse them as a Mapping and skip them). THE FIX IS TWO PARTS AND THE ORDER MATTERS: first make the guard non-vacuous — resolve the canonical path against the REPO root rather than the enclosing cargo workspace, or fail loudly when it cannot be found, so the bootstrap branch can no longer swallow a real absence. Then reconcile the two files. Doing only the second leaves the guard still asleep for the next drift.")
//! @yah:handoff("GUARD MADE NON-VACUOUS FIRST, then the files reconciled, and the intermediate RED was observed. New CanonicalHome enum + locate_canonical_home() in oss/yubaba/crates/cloud/src/cloud_init.rs (non-test code, since it is a real property of the layout): the ancestor walk is keyed on `.yah/infra/` instead of `.yah/`. The monorepo root has it; oss/yubaba, whose own .yah/ holds nothing but a .gitignore, does not, and neither does the standalone export where the repo root genuinely IS oss/yubaba. embedded_template_matches_workspace_canonical now matches on that: Monorepo(root) means the canonical file is REQUIRED (missing/unreadable panics with a message naming the path and saying provisioning reads it); StandaloneExport is the only branch allowed to skip. The old `if canonical_path.exists()` bootstrap branch that swallowed a real absence is gone.")
//! @yah:handoff("THE RED, run after step 1 and before step 2 exactly as the ticket asked: `cargo test -p yah-cloud --lib cloud_init::tests` from oss/yubaba gave 29 passed / 1 FAILED, the single failure being embedded_template_matches_workspace_canonical asserting on /Users/leif/ss/yah/.yah/infra/cloud-init/mirror.yml with the full 128-line diff in its message. That is the proof it now compares something real; the post-fix green on its own would have been indistinguishable from today's vacuous green.")
//! @yah:handoff("THE RECONCILE WAS NOT A ONE-WAY COPY, and this is the part worth reading. The canonical copy carried a DELIBERATE CORRECTION the embedded template never received: commit 45e39f0a (2026-08-03) changed `systemctl enable --now yubaba.slice` to `systemctl start yubaba.slice`, plus the comment explaining why - app/yah/cli/resources/yubaba.slice has no [Install] section, and systemctl enable on such a unit fails. Blindly overwriting canonical with embedded (which the ticket text reads like) would have regressed that fix into the file provisioning actually ships. So the fix went BOTH ways: the slice correction was applied to oss/yubaba/crates/cloud/templates/mirror.yml first, then that file was copied over .yah/infra/cloud-init/mirror.yml. The two are now byte-identical at 237 lines; canonical gained the whole R858-F17 turso-backup block, the R870-F23 YAML double-quoting of the two colon-space runcmd entries, the litestream 0.3.13 install, and the headscale 0.23.0 + litestream-headscale.service pre-stage.")
//! @yah:handoff("WHICH TEMPLATE PROVISIONING ACTUALLY USES - THE CANONICAL ON-DISK ONE, read not guessed. cloud_init::load_template() returns the on-disk <workspace_root>/.yah/infra/cloud-init/mirror.yml whenever it exists and falls back to the include_str! embedded copy only when it does not. provision::build_request() (oss/yubaba/crates/cloud/src/provision.rs:88) is its sole non-test caller, and its sole caller in turn is handle_provision at app/yah/cli/src/cloud.rs:14524, whose workspace_root is the camp repo root - which has the file. So the embedded template is the FALLBACK, not the live path, and the stale copy is the one every `yah cloud machine provision` has been shipping. load_template's doc comment now says so.")
//! @yah:handoff("FLEET CONSEQUENCE, dated from git so the window is bounded. NO node was touched - code, templates and tests only, per scope. Canonical was last updated 2026-08-03 (45e39f0a). The headscale 0.23.0 pre-stage, the litestream install and litestream-headscale.service staging landed in the EMBEDDED copy only on 2026-09-05 (4bed91fe), so any node provisioned between 2026-09-05 and today silently received none of them - that is the real damage this drift did. The turso-backup block landed in the embedded copy only TODAY (a8f0d501, 2026-09-10), so no node has ever received the durability helpers from either file; that gap is fleet-wide and predates this ticket rather than being caused by the drift. Remediating existing nodes is deliberately NOT done here.")
//! @yah:handoff("DISCOVERED WORK, fixed in this pass: the SECOND stale twin is .yah/infra/cloud-init/stand-up-yubaba.sh, mirror.yml's documented SSH transcription for LAN nodes (W257 step 6). It installed yubaba/kamaji/units from the release tarball but never the turso-backup helpers or the KAMAJI_HYDRATE_HELPER/KAMAJI_TAIL_HELPER drop-in, so every LAN node stood up by it would refuse any durability-declaring workload. Added a present-checked, non-fatal block in the script's own idiom ($SUDO install from $D, tee for the drop-in, a WARNING to stderr when the tarball predates R858-F17), using Environment= in a drop-in rather than ExecStart= flags for the reason kamaji.service's own comment gives. Verified against scripts/publish-yubaba-release.sh:326-327, which hard-asserts both binaries at exactly $STAGE_NAME/turso-backup-{hydrate,tail} - the path the script looks in. bash -n clean.")
//! @yah:verify("cargo test -p yah-cloud --lib, run from oss/yubaba (NOT the repo root - yah-cloud is not a root workspace member and needs dev-dependencies): 1166 passed / 0 failed / 4 ignored, against the 1163/0/4 baseline. The +3 are exactly the three tests added here; nothing regressed.")
//! @yah:verify("INTERMEDIATE RED, the verification this ticket actually asked for: after step 1 and before step 2, cargo test -p yah-cloud --lib cloud_init::tests = 29 passed / 1 failed, the single failure being embedded_template_matches_workspace_canonical naming .yah/infra/cloud-init/mirror.yml. Green afterwards.")
//! @yah:verify("Three new tests pin the guard against going vacuous again. locate_canonical_home_finds_monorepo_root_not_the_inner_workspace and locate_canonical_home_reports_standalone_export build both tree shapes in tempdirs (the monorepo one reproduces the inner oss/yubaba/.yah/.gitignore that caused the original miss). drift_guard_cannot_go_vacuous_in_the_monorepo keys off a signal INDEPENDENT of the .yah/infra/ marker - `git subtree split --prefix=oss/yubaba` strips that prefix, so a CARGO_MANIFEST_DIR still ending in oss/yubaba/crates/cloud proves we are in the monorepo - and panics if the guard resolved StandaloneExport there.")
//! @yah:verify("diff -u between the two mirror.yml copies is empty; both are 237 lines.")
//! @yah:verify("bash -n .yah/infra/cloud-init/stand-up-yubaba.sh clean.")
//! @yah:verify("rendered_runcmd_entries_are_all_strings left untouched and still green, as instructed - a different test and a different gate.")
//! @yah:gotcha("NOT FIXED, and deliberately left as an operator call rather than decided silently: stand-up-yubaba.sh still lacks the litestream 0.3.13 install and the headscale 0.23.0 / litestream-headscale.service pre-stage that mirror.yml now carries. Those exist because R858-T4 made EVERY node a coordinator candidate, and whether the LAN/appliance class (us-west-01x, mostly no-voter) belongs in that candidate set is a fleet-topology decision, not a transcription gap. The durability helpers WERE added because their consequence is unconditional - kamaji refuses the workload on any node - while these two only matter if the node can ever own the ingress role.")
//! @yah:gotcha("Uncommitted: this camp's git policy is `defer`, so the four changed files (oss/yubaba/crates/cloud/src/cloud_init.rs, both mirror.yml copies, .yah/infra/cloud-init/stand-up-yubaba.sh) are in the working tree for the camp's git sweep. No commit SHA to cite.")
//! @yah:verify("INDEPENDENTLY VERIFIED BY A SECOND COURIER (session:67d56cd3) who did not implement it. GUARD IS GENUINELY HONEST, confirmed by reading the branches: `locate_canonical_home` (cloud_init.rs:208) walks ancestors for `.yah/infra/`, and the Monorepo branch (cloud_init.rs:1013-1029) `read_to_string`-panics on a missing or unreadable canonical and `assert_eq`s full trimmed content against `DEFAULT_TEMPLATE` — so BOTH a deleted canonical AND a revert to its pre-fix state now fail hard. No skip branch survives. THE STANDALONE-EXPORT DISCRIMINATOR HOLDS: `.yah/infra/` exists only at the monorepo root and nowhere under `oss/` (oss/yubaba/.yah holds one tracked .gitignore), so the real subtree-split export takes the StandaloneExport branch legitimately; the only route to that branch from inside the monorepo is deleting `.yah/infra/`, which `drift_guard_cannot_go_vacuous_in_the_monorepo` catches off an INDEPENDENT path-suffix signal. Two residual holes, both benign in DIRECTION: a standalone clone placed under some `.yah/infra/` ancestor false-FAILS rather than skipping, and renaming the oss/yubaba path would disarm only the meta-guard, not the drift guard.")
//! @yah:verify("THE RECONCILIATION WAS NOT A ONE-WAY COPY, and that was checked rather than taken on trust — a blind embedded-to-canonical copy would have silently destroyed someone else's fix. Both copies are byte-identical at 237 lines, sha256 6bc46731e030bae8a1970bcb06cf3132323eb454ab72496d6ef9b2c6177585af. At HEAD the CANONICAL carried the 45e39f0a fix (`systemctl start yubaba.slice`, since the slice has no [Install] section) while the EMBEDDED still had `enable --now`; the working diff applies that exact hunk embedded-ward (+5/-3) against the canonical's +109/-1, proving the flow went both directions. No `enable --now yubaba.slice` remains in either file. The canonical's single deleted line was only the `chmod 0644` superseded by the version adding litestream-headscale.service. R858-F17's turso-backup block and R870-F23's two double-quoted runcmd entries are present in both, necessarily so given byte-identity. THE INTERMEDIATE RED WAS STRUCTURALLY NECESSARY, not stage-managed: the Monorepo branch compares entire trimmed contents and the two HEAD copies differed by 112 insertions / 6 deletions, so the assert could not have passed. `cloud_init::tests` is 30 tests, making the reported 29-pass/1-fail arithmetically consistent. (The \"128-line diff\" figure is 118 changed lines by numstat — cosmetic, not a defect.) COUNTS: `cargo test -p yah-cloud --lib` from oss/yubaba = 1166 passed / 0 failed / 4 ignored, the +3 over baseline being exactly the three new tests.")
//! @yah:verify("THE FLEET FACT, CONFIRMED WITH ONE DATE CORRECTED — this is the operationally consequential part and the correction matters. PROVISIONING READS THE CANONICAL ON-DISK COPY, not the embedded one: provision.rs:88 is `cloud_init::load_template(workspace_root)`, and `load_template` (cloud_init.rs:225-232) PREFERS `.yah/infra/cloud-init/mirror.yml`, falling back to the embedded copy only when absent; sole non-test caller chain is cloud.rs:14523 with the camp repo root. So every node was provisioned from the file that was 128 lines stale. CONSEQUENCE 1, date corrected: the headscale/litestream pre-stage entered the EMBEDDED copy on 2026-09-04 in b20a1e09 — NOT 2026-09-05, which was 4bed91fe merely refining it — and the canonical never carried it, so every node provisioned since 2026-09-04 missed that pre-stage. CONSEQUENCE 2 HOLDS AS STATED: no node has ever had the turso-backup helpers. `turso-backup-hydrate` first entered the embedded template today in a8f0d501 and the canonical only in this change; grep finds no other install path (only publish-yubaba-release.sh, kamaji's consumers, both mirror.yml copies, stand-up-yubaba.sh), and `.yah/infra/machines/us-west-001.toml:42` independently records \"NO prod node has them today\". NOT VERIFIED, stated rather than smoothed: no node was touched, so the actual SET of nodes provisioned since 2026-09-04 is unconfirmed.")
//! @yah:handoff("THIRD TWIN CLOSED — .yah/infra/cloud-init/stand-up-yubaba.sh is now both LEVEL and GUARDED. GUARD SHAPE CHOSEN: the assertion test, not generate-from-one-source, and the reason is that the twin is not a pure transcription. The script deliberately diverges from mirror.yml in ways that are CORRECT and load-bearing (enable+restart instead of `enable --now`, which is the only reason it can call itself idempotent — see its own :162-176 comment measured on us-west-013/014; write-if-absent journald ceiling so a Pi's tighter 200M bound wins; cluster-KEK install; loopback bind for no-mesh nodes; status block reading /health rather than the on-disk --version). A generator would therefore have to model the divergences, which is the whole difficulty, and mirror.yml is itself a static `{{ }}`-substituted file rather than a Rust-rendered one — so single-sourcing would mean either making mirror.yml generated (large) or parsing YAML to emit bash (fragile). Not contained; see @yah:next for what it would actually take.")
//! @yah:handoff("THE GUARD DERIVES ITS PINS FROM THE TEMPLATE rather than restating them, which is what stops it becoming the next thing that rots. `stand_up_script_carries_the_templates_install_steps` (oss/yubaba/crates/cloud/src/cloud_init.rs, next to the mirror.yml drift guard) reads the script via the new `paths::stand_up_script`, reusing `locate_canonical_home` so StandaloneExport is the same single permitted skip. Three assertion families: (1) STAND_UP_TWIN_ANCHORS, an explicit list of artifacts a node must end up carrying, each asserted against DEFAULT_TEMPLATE AS WELL so a stale anchor goes red on the template side instead of over-constraining the script; (2) every 64-char lowercase-hex sha256 pin extracted from DEFAULT_TEMPLATE must appear in the script (the four litestream/headscale amd64+arm64 checksums; `{{YAH_YUBABA_SHA256}}` is not hex so it is not picked up, and the extractor asserts it found >=4 so a broken extractor cannot make the guard vacuous); (3) every `https://github.com/OWNER/REPO/releases/download/TAG/` prefix in the template must appear in the script — owner/repo/tag pinned, arch-templated asset filename left free. Net effect: bumping litestream or headscale in mirror.yml alone now turns this red.")
//! @yah:handoff("THREE REDS DEMONSTRATED, not one, because the anchor half fires first and would have masked the other two. (a) Test written BEFORE the script fix: `cargo test -p yah-cloud --lib cloud_init::tests::stand_up_script` = 0 passed / 1 FAILED, naming `no litestream-headscale.service`. (b) With the script level, one hex char mutated in the script's headscale arm64 HS_SHA: FAILED, naming the missing pin 99fa9b29...e9fe. (c) The litestream URL tag moved to v0.3.14 while the template stays v0.3.13: FAILED, naming the unfetched .../download/v0.3.13/. Both mutations were reverted by hand and re-verified. SCRIPT CONTENT ADDED, in the script's own idiom: present-checked non-fatal `$SUDO install -m0644 \"$D/litestream-headscale.service\"` alongside the other three units (verified it really ships in the tarball — publish-yubaba-release.sh:287 stages it from $RESOURCES); an arch-cased litestream 0.3.13 fetch+sha256+`tar -C /usr/local/bin`; an arch-cased headscale 0.23.0 fetch+sha256+`install -m0755` into /var/lib/yah-cloud/headscale/headscale. Every failure path is a stderr WARNING naming the operational consequence, never an exit — a node without these cannot hold the ingress role, which is not a failed stand-up.")
//! @yah:handoff("DELIBERATE CHOICE WORTH REVIEWING: the two downloads are UNCONDITIONAL, not present-checked-skip. A re-run of this script is the documented upgrade path (W257 §8), and `if [ -x /usr/local/bin/litestream ]; then skip` would recreate exactly the stale-binary trap the script's own `restart`-not-`--now` comment was written for after us-west-013/014. Cost is re-downloading ~10MB litestream + ~51MB headscale on every re-run; both are sha-pinned so the repeat is idempotent, just not free. A version-checked skip was rejected because it would depend on `headscale version` / `litestream version` output formats I did not verify. Rationale is in the script's comments. VERIFIED: `bash -n` clean; both mirror.yml copies UNTOUCHED and still byte-identical at sha256 6bc46731e030bae8a1970bcb06cf3132323eb454ab72496d6ef9b2c6177585af; `cargo test -p yah-cloud --lib` from oss/yubaba = 1167 passed / 0 failed / 4 ignored, exactly +1 over the 1166 baseline and that +1 is the new test. rustfmt --check clean on cloud_init.rs (paths.rs has ONE pre-existing unformatted hunk in `infra_source_cache_dir`, not mine, left alone). NO FLEET CONTACT of any kind — script, test and one paths.rs helper only. Uncommitted: camp git policy is `defer`, so no SHA to cite.")
//! @yah:next("WHAT GENERATE-FROM-ONE-SOURCE WOULD ACTUALLY TAKE, having rejected it as out of scope here. The contained 20% is the third-party PINS: litestream 0.3.13 + 2 sha256s and headscale 0.23.0 + 2 sha256s now live in five places (both mirror.yml copies, stand-up-yubaba.sh, and for headscale also `cloud::mesh::HEADSCALE_VERSION` and `yubaba::DEFAULT_HEADSCALE_VERSION`, which mirror.yml's own comment admits are kept in lockstep by convention with no dependency edge). Making those Rust constants the one source and rendering them into mirror.yml as `{{ LITESTREAM_BLOCK }}` / `{{ HEADSCALE_BLOCK }}` would collapse five to one, but it requires touching BOTH mirror.yml copies in lockstep and a matching emitter for the script. The remaining 80% — the install STEPS — is not contained: the script's correct divergences (idempotent enable+restart, write-if-absent ceilings, KEK, loopback bind) mean a generator must model divergence, and the script's real destination is the `yah cloud machine bootstrap` command W242 Phase 1 already plans, which would emit both from one Rust model. That is the right home for this, and it is a relay, not a hunk.")
//! @yah:gotcha("SUPERSEDES the earlier gotcha beginning \"NOT FIXED, and deliberately left as an operator call\" — that entry is now STALE and should be read as history, not state. The litestream 0.3.13 install and the headscale 0.23.0 / litestream-headscale.service pre-stage ARE now in stand-up-yubaba.sh, added on the relay leader's explicit instruction in this follow-on pass. The fleet-topology question that entry deferred has therefore been ANSWERED IN THE AFFIRMATIVE BY DEFAULT: every LAN/appliance node stood up by this script from here on is provisioned as a coordinator candidate (headscale binary staged at /var/lib/yah-cloud/headscale/headscale, replication unit laid down, neither started — leader.rs still decides who runs it). If that is NOT wanted for the us-west-01x class, the lever is an opt-out env guard in the script, not reverting it, because reverting now goes red against `stand_up_script_carries_the_templates_install_steps`. Cost of the affirmative answer is bounded and staging-only: ~61MB fetched per run and two staged-but-inert artifacts.")
//! @yah:verify("THIRD-TWIN GUARD: `cargo test -p yah-cloud --lib cloud_init::` from oss/yubaba = 31 passed / 0 failed (30 before, +1 = stand_up_script_carries_the_templates_install_steps). Full lib suite = 1167 passed / 0 failed / 4 ignored vs the 1166/0/4 baseline. bash -n .yah/infra/cloud-init/stand-up-yubaba.sh clean. Both mirror.yml copies still sha256 6bc46731e030bae8a1970bcb06cf3132323eb454ab72496d6ef9b2c6177585af — the guard on THEM was not disturbed and is still green.")
//! @yah:handoff("ACCEPTED BY THE RELAY LEADER (@Ashguard:hydra, session:39386823). Both halves landed in the required order — guard made honest FIRST, then the two mirror.yml copies reconciled — plus a third twin (stand-up-yubaba.sh) closed that the fix itself exposed. Implemented by @Ashguard:polaris (session:9d2e59a3), independently verified by session:67d56cd3, third-twin follow-on by session:039a0f82. The operationally consequential finding is in the verify entries: provisioning reads the CANONICAL on-disk template, and that was the copy which had drifted.")
//! @yah:verify("COLUMN MOVED VIA `yah board move R870-B25 review` AFTER BOTH `board.review` VERBS REFUSED — recorded because the next agent will hit it too and the error message actively misleads. The MCP verb and `yah board review` both fail with \"ticket 'R870-B25' not found — it may have been archived or may not exist in this camp\", which is false: the CLI's fresh scan saw it fine, anchored at oss/yubaba/crates/cloud/src/reconciler/mod.rs:67. ROOT CAUSE, grounded not guessed: `arch.review_ticket` is a daemon-only gated write with NO in-process fallback — already filed as R606-T3 and stated verbatim at app/yah/cli/src/camp.rs:975, \"Board reads/updates fall back in-process; review/move/etc do not — that asymmetry is the bug\" — and the daemon resolves transition targets from its IN-MEMORY store rather than from disk (crates/yah/camp-service/src/service.rs:4199-4216, which emits that exact string). So \"exists on disk\" and \"the daemon can transition it\" are independent facts. `yah board move` turns out to be on the falls-back side despite that note, which is why it works. Contributing: the daemon is version-skewed and degraded (0.8.36+74874f3e-dirty vs CLI 0.8.37+e896d28a-dirty) and refused read probes with EAGAIN, the signature R606-S2 pinned to the 500ms fast-path RPC floor under load. ALSO CONFIRMED, since it was the other candidate explanation: `oss/yubaba` is NOT a subcamp — only cheers, mesofact, turso-backup and xlb carry `.yah/camp.toml` — so no `--path` is needed and the filing location was never the problem.")

use std::collections::{BTreeMap, VecDeque};
use std::path::{Path, PathBuf};
use std::sync::Arc;

use anyhow::{Context, Result};
use async_trait::async_trait;
use serde::{Deserialize, Serialize};
use tokio::sync::{oneshot, Mutex as AsyncMutex};

use workload_spec::{NamespaceId, TenantId};

use crate::{GitSource, MirrorConfig, MirrorProviderSlot, ServiceComponent, ServiceConfig};

pub mod bundle_store;
pub(crate) mod cf_creds;
pub mod cloudflare_worker;
pub mod container;
pub mod derive_cache_prune;
pub mod dev_door;
pub mod domain;
pub mod headscale;
pub mod ingress;
pub mod ingress_verify;
// R918-F5 — the local-process reconciler is Unix-only *by design*, not by
// accident of which syscalls it happened to reach for. Its contract is
// "replace the predecessor": `kill(pid, 0)` liveness plus a SIGTERM-grace-
// SIGKILL ladder. A build with that ladder removed would still spawn, and
// would silently double-spawn instead of reaping — exactly the failure that
// looks inert in a diff and strands processes on a node. So the module is
// gated whole rather than half-ported. A non-unix `yah` is a client (it talks
// to a camp and submits builds); it supervises no workloads, so it needs no
// local-process reconciler. Reversing this means implementing
// OpenProcess/TerminateProcess here, at the point someone actually wants a
// Windows fleet node — see .yah/docs/working/W352-windows-and-macos-build-targets.md.
#[cfg(unix)]
pub mod device;
#[cfg(unix)]
pub mod local_process;
pub mod mesofact_bundle;
pub mod mesofact_static;
// `pub(crate)` rather than private: `sanitize_ident` is the crate's ONE mesh-ident
// normalizer, and R870-F23's `inner_door::component_workload_ident` has to fold
// its derived ident the same way `local_process` folds its own. A second copy
// would be a second normalizer that can drift.
pub(crate) mod native_support;
pub mod pg_driver;
pub mod s3_driver;
pub mod smtp_driver;
pub mod pond;
pub mod pond_door;
pub mod pond_publish;
pub mod publish_beacon;
pub mod r2_publish;
pub mod service_discovery;
pub mod static_asset;
pub mod static_asset_prune;
pub mod sync_status;

#[cfg(test)]
mod lowering_golden;

pub use bundle_store::{publish_bundle_to_r2, PublishReport as BundlePublishReport};
pub use cloudflare_worker::CloudflareWorkerReconciler;
pub use container::{ContainerOptions, ContainerReconciler};
pub use derive_cache_prune::{
    collect_live_derive_hashes, compute_derive_cache_candidates, execute_derive_cache_prune,
    DeriveCacheLiveHashes, DerivePruneCandidate,
};
pub use domain::{
    deploy_domain_passway, diff_apex_records, ensure_passway_apex, ensure_r2_custom_domain,
    list_live_apex_records, plan_domain_passway, plan_passway_apex, public_origins, ApexRecordDiff,
    DomainPasswayPlan, LiveApexRecord, PasswayApexOutcome, PasswayOrigin,
};
pub use headscale::{
    DeclaredHeadscale, DeclaredPolicy, DeclaredPreauthKey, HeadscaleReconciler,
    WORKLOAD_KIND as HEADSCALE_WORKLOAD_KIND,
};
pub use ingress::{
    collate_front_doors, declared as ingress_declared, ensure_tunnel_ingress, machine_mesh_addrs,
    plan_ingress, publish_tunnel_ingress, resolve_ingress_candidates, resolve_ingress_placements,
    Collation, IngressPlan, IngressRule, NodeFrontDoor, PlannedEdge, TunnelIngressOutcome,
};
pub use ingress_verify::{
    apply_public_path, resolve_upstreams_reporting, verify_collation, BeaconFetch, DialOutcome,
    EndpointCheck, PublicReadings, RuleResolution, RuleResolutions, RuleVerdict,
    UndeclaredDoorProbe, VerifyFinding, VerifyReport, undeclared_door_probes,
};
#[cfg(unix)]
pub use local_process::LocalProcessReconciler;
#[cfg(unix)]
pub use device::DeviceReconciler;
pub use mesofact_bundle::{
    resolve_bundle_machines, BundleSlot, MesofactBundleReconciler, RevalidateSlot,
    SLOT_ROLE as BUNDLE_SLOT_ROLE,
};
pub use dev_door::{sync_dev_door, up_dev_door, DEFAULT_DEV_DOOR_PORT};
pub use mesofact_static::MesofactStaticReconciler;
pub use pond::{PondOptions, PondState};
pub use pond_door::{
    door_env, door_state_dir, ensure_pond_cert_as, plan_pond_door, pond_hostname,
    resolve_passway_binary, spawn_pond_door, CertPair, PondDoorPlan, DEFAULT_DOOR_PORT, POND_TLD,
};
// R918-F5 — `is_root` reads an effective uid; `ensure_pond_cert` is the wrapper
// that folds it in. Both unix-only. `ensure_pond_cert_as` above is the portable
// half, with the privilege decision injected.
#[cfg(unix)]
pub use pond_door::{ensure_pond_cert, is_root};
pub use pond_publish::{derive_minio_key, publish_to_pond, PondPublishReport};
pub use r2_publish::{
    publish_to_r2, R2PublishReport, R2PurgeOpts, R2_ACCESS_KEY_ENV, R2_ACCESS_KEY_SLOT,
    R2_SECRET_KEY_ENV, R2_SECRET_KEY_SLOT,
};
pub use service_discovery::{
    DiscoveredRecord, RecordVisibility, ServiceRecordFanout, UnknownReason,
};
pub use static_asset::StaticAssetReconciler;
pub use static_asset_prune::{
    compute_live_set, compute_prune_candidates, execute_prune, load_service_and_mirror,
    PruneCandidate, PruneOutcome, PruneReport,
};
pub use sync_status::{
    compute_cell, compute_service, new_sync_id, summarize, CellStatus, DriftEntry, HealthState,
    MirrorObservation, Runtime, ServiceStatus, StatusSummary, SyncHistoryEntry, SyncOutcome,
    SyncState, WireContainerStatus,
};

// ─── Log buffer ─────────────────────────────────────────────────────────────

const LOG_CAP: usize = 500;

#[derive(Debug, Default)]
struct LogRing {
    lines: VecDeque<String>,
    /// Monotonically increasing total lines ever pushed (never decrements).
    total: usize,
}

/// Bounded ring buffer for child-process stdout/stderr (R263-F3).
/// Shared between the reader tasks and the Tauri `mirror_run_logs` command.
#[derive(Debug, Clone, Default)]
pub struct LogBuffer(Arc<AsyncMutex<LogRing>>);

impl LogBuffer {
    pub fn new() -> Self {
        Self::default()
    }

    /// Append a line; drops the oldest entry when over capacity.
    pub async fn push(&self, line: String) {
        let mut ring = self.0.lock().await;
        ring.total += 1;
        ring.lines.push_back(line);
        if ring.lines.len() > LOG_CAP {
            ring.lines.pop_front();
        }
    }

    /// Return lines not yet seen by the caller.
    ///
    /// `since` is the `total` cursor from the previous call (0 = nothing
    /// seen yet). Returns `(new_lines, new_cursor)`. Pass `new_cursor` back
    /// on the next call to receive only incremental output.
    pub async fn since(&self, since: usize) -> (Vec<String>, usize) {
        let ring = self.0.lock().await;
        let oldest = ring.total.saturating_sub(ring.lines.len());
        let skip = since.saturating_sub(oldest);
        let new_lines: Vec<String> = ring.lines.iter().skip(skip).cloned().collect();
        (new_lines, ring.total)
    }

    /// Current write cursor — the `total` [`Self::since`] would hand back
    /// right now if nothing more were pushed. Lets a producer mark a boundary
    /// (e.g. "everything before this point was the build phase") for a later
    /// reader to seek past without re-reading lines it doesn't want.
    pub async fn cursor(&self) -> usize {
        self.0.lock().await.total
    }
}

/// Shared cell for one phase-boundary cursor a reconciler can publish
/// mid-`up()`, so a poller sees a multi-phase bring-up's internal transition
/// (e.g. build → run) before the whole call returns — the same problem
/// [`LogBuffer`] solves for output, for a single position instead of a ring.
/// `None` until the reconciler reaches that phase; a caller registers one
/// before calling `up()` to observe it live, same pattern as
/// [`LogBuffer::clone`]-and-hand-in.
#[derive(Debug, Clone, Default)]
pub struct PhaseCursor(Arc<AsyncMutex<Option<usize>>>);

impl PhaseCursor {
    pub fn new() -> Self {
        Self::default()
    }

    pub async fn set(&self, cursor: usize) {
        *self.0.lock().await = Some(cursor);
    }

    pub async fn get(&self) -> Option<usize> {
        *self.0.lock().await
    }
}

/// The `(tenant, namespace)` a bring-up is scoped to (W206). Reconcilers that
/// touch a credentialed provider resolve it at this scope
/// ([`CfProvider::resolve_scoped`](super::reconciler::cf_creds)) so a namespace's
/// Cloudflare zone/account/keystore slots come from its own scope rather than the
/// workspace-global defaults. Defaults to the singleton `(default, default)`,
/// which collapses every scoped lookup back to the historical global slots — so
/// single-namespace deployments are unaffected.
#[derive(Debug, Clone)]
pub struct ProviderScope {
    pub tenant: TenantId,
    pub namespace: NamespaceId,
}

impl ProviderScope {
    /// The degenerate single-tenant / single-namespace scope. Scoped provider
    /// lookups made against it resolve to the pre-W206 global keystore slots.
    pub fn singleton() -> Self {
        Self {
            tenant: TenantId::singleton(),
            namespace: NamespaceId::singleton(),
        }
    }
}

impl Default for ProviderScope {
    fn default() -> Self {
        Self::singleton()
    }
}

/// Inputs a reconciler sees for one bring-up.
pub struct ReconcileCtx<'a> {
    /// Workspace root (parent of `.yah/`). Used to resolve relative paths
    /// on the component.
    pub workspace_root: &'a Path,
    /// Service that owns the component.
    pub service: &'a ServiceConfig,
    /// Component being brought up.
    pub component: &'a ServiceComponent,
    /// Mirror manifest the bring-up targets.
    pub mirror: &'a MirrorConfig,
    /// Environment name (file stem of `mirrors/<env>.toml`).
    pub env: &'a str,
    /// `(tenant, namespace)` this bring-up is scoped to (W206). Credentialed
    /// providers resolve at this scope; defaults to [`ProviderScope::singleton`].
    pub scope: ProviderScope,
}

impl<'a> ReconcileCtx<'a> {
    /// Absolute path to the component's workload directory (the parent of
    /// `workload.toml`).
    ///
    /// In-tree components resolve to `<workspace_root>/<path>`. For
    /// `git`-sourced components (R561-F1, "BYO git") this points into the
    /// local clone — `<source_cache>/<subdir>/<path>` — which is empty until
    /// [`materialize`](Self::materialize) runs (approach A: clone-at-reconcile,
    /// so config load + validation stay offline).
    pub fn workload_dir(&self) -> PathBuf {
        match &self.component.git {
            None => self.workspace_root.join(&self.component.path),
            Some(git) => {
                let mut dir = self.source_cache_dir();
                if let Some(subdir) = &git.subdir {
                    dir = dir.join(subdir);
                }
                dir.join(&self.component.path)
            }
        }
    }

    /// Root of the local clone for a `git`-sourced component:
    /// `<workspace_root>/.yah/infra/state/sources/<service>/<component_id>`.
    fn source_cache_dir(&self) -> PathBuf {
        self.workspace_root
            .join(".yah/infra/state/sources")
            .join(&self.service.name)
            .join(&self.component.id)
    }

    /// Ensure a `git`-sourced component's code is present locally before build
    /// (R561-F1, approach A). No-op for in-tree components. Idempotent: clones
    /// on the first call, fetches + re-checks-out the pinned ref thereafter.
    ///
    /// Reconcilers MUST call this at the top of [`up`](Reconciler::up) before
    /// reading [`workload_dir`](Self::workload_dir) for a remote component.
    pub async fn materialize(&self) -> Result<()> {
        let Some(git) = &self.component.git else {
            return Ok(());
        };
        materialize_git_source(git, &self.source_cache_dir())
            .await
            .with_context(|| {
                format!(
                    "materializing git source {}@{} for {}/{}",
                    git.repo, git.r#ref, self.service.name, self.component.id
                )
            })
    }

    /// Read `<workload_dir>/workload.toml` and extract just the `kind`
    /// discriminator.
    ///
    /// Why this and not the strongly-typed [`workload_spec::Workload`]
    /// parse: reconcilers only need the kind to
    /// dispatch; per-kind tooling (e.g. `mesofact-dev`'s
    /// `WatchOptions::from_workload`) does its own parsing for the
    /// build/out_dir fields it cares about.
    pub fn workload_kind(&self) -> Result<String> {
        workload_kind(&self.workload_dir())
    }

    /// Look up a provider slot by role (e.g. `"static"`, `"compute"`).
    ///
    /// Tries the component-qualified key first (`"<role>:<component id>"`)
    /// before falling back to the bare role. A mirror role is normally
    /// service-wide — one `providers.static` slot serves every static-kind
    /// component — but a service can declare more than one component with
    /// the same role (e.g. two `mesofact-static`/`mesofact-spa` components
    /// sharing one mirror), and those need distinct ports to ever both come
    /// up. The qualified key is how a mirror opts a specific component out
    /// of sharing the bare-role slot:
    ///
    /// ```toml
    /// [providers."static:site"]
    /// kind = "miniflare-native"
    /// port = 4331
    ///
    /// [providers."static:app"]
    /// kind = "miniflare-native"
    /// port = 4332
    /// ```
    ///
    /// A mirror with only the bare role (the common, single-component case)
    /// is unaffected — the qualified lookup misses and falls through.
    pub fn slot(&self, role: &str) -> Option<&'a MirrorProviderSlot> {
        let qualified = format!("{role}:{}", self.component.id);
        self.mirror
            .providers
            .get(qualified.as_str())
            .or_else(|| self.mirror.providers.get(role))
    }

    /// This environment's `[build.<component id>]` override, if the mirror
    /// declares one that changes anything (R905).
    ///
    /// Keyed by component id rather than by role: a build belongs to the
    /// project that declares it, and two components sharing a role still build
    /// with two different commands.
    pub fn build_override(&self) -> Option<&'a crate::config::MirrorBuildOverride> {
        self.mirror.build_override(&self.component.id)
    }
}

/// An explicit teardown hook for a workload this process did not spawn as a
/// child — see [`RunningWorkload::with_teardown`] (R714-B1).
///
/// Boxed rather than a generic parameter because [`RunningWorkload`] is stored
/// in heterogeneous collections (the desktop's mirror registry) and cannot
/// carry a type parameter.
/// `Sync` on the boxed closure is load-bearing, not belt-and-braces: without
/// it `RunningWorkload` stops being `Sync`, so `&RunningWorkload` stops being
/// `Send`, and every desktop `#[tauri::command]` that holds one across an
/// `.await` (`mirror_run_logs` iterates the registry's handles) fails to
/// compile with "future cannot be sent between threads safely". The captures a
/// teardown needs — a container name and a workspace root — are `Sync` anyway.
type TeardownFuture = std::pin::Pin<Box<dyn std::future::Future<Output = Result<()>> + Send>>;
type TeardownFn = Box<dyn FnOnce() -> TeardownFuture + Send + Sync + 'static>;

/// Newtype so [`RunningWorkload`] can keep its `#[derive(Debug)]` — a boxed
/// closure is not `Debug`.
struct Teardown(TeardownFn);

impl std::fmt::Debug for Teardown {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        f.write_str("Teardown(<fn>)")
    }
}

/// Handle to a workload that's been brought up. Owns the lifecycle: drop
/// or call [`RunningWorkload::shutdown`] to take it back down.
#[derive(Debug)]
pub struct RunningWorkload {
    /// Workload kind that was reconciled (e.g. `"mesofact-static"`).
    pub kind: String,
    /// Slot role this workload occupies on the mirror (e.g. `"static"`).
    pub slot: String,
    /// Local URL the workload exposes, when applicable. `None` for
    /// workloads that publish to a non-local artifact store (e.g. R2).
    pub dev_url: Option<String>,
    /// Notional URL of the deployed artifact in the production case
    /// (e.g. `https://yah.dev`). `None` until Cloudflare/R2 wiring lands.
    pub public_url: Option<String>,
    /// Secondary local UI console, if the workload exposes one (e.g. MinIO
    /// console on a pond tier). Surfaced as a separate chip in the Services
    /// matrix next to `dev_url`.
    pub console_url: Option<String>,
    /// Ring buffer for stdout/stderr from the workload's child process.
    /// `None` for workloads that don't capture stdio (e.g. container-backed).
    pub log_buffer: Option<LogBuffer>,

    /// R546-B12: human-readable lines the reconciler wants the operator to see
    /// on THIS run — what it actually did, not what is configured. A clean
    /// static-asset reconcile used to print nothing at all, so a successful
    /// publish and a successful no-op were indistinguishable at the apply
    /// surface, and the one view that would have disambiguated them (`yah cloud
    /// status`) was itself blind.
    ///
    /// Deliberately NOT on [`RunningWorkloadSummary`]: that type is serialized
    /// across the process boundary to the desktop UI, and this is per-run
    /// console output, not state the UI should cache.
    pub notes: Vec<String>,

    /// Sender that signals the supervisor task to tear down. Closing the
    /// channel (drop) is equivalent to sending — supervisor exits on
    /// channel close.
    shutdown: Option<oneshot::Sender<()>>,
    /// Joinable task that owns any child process and reaps it on signal.
    supervisor: Option<tokio::task::JoinHandle<Result<()>>>,
    /// R714-B1: teardown for a workload that runs OUTSIDE this process, so
    /// there is no child to reap and no supervisor to signal.
    ///
    /// Deliberately run from [`RunningWorkload::shutdown`] only, never from
    /// `Drop`. The two are different intents and conflating them breaks both
    /// directions: a container the desktop started is meant to outlive the
    /// desktop (the next launch re-adopts it with `adopt_only`), so quitting
    /// the app must not `docker rm -f` it; but the ■ button IS an explicit
    /// stop, and must.
    teardown: Option<Teardown>,

    /// R715-F4: where to ask this workload for a live status document, when
    /// it declared a process-control channel (W315). `None` for a workload
    /// with no channel — polling is then simply skipped, not an error.
    ///
    /// The desktop holds `RunningWorkload` in-process (it links this crate
    /// directly, unlike the ad-hoc `run.spawn` path which lives behind the
    /// camp daemon's socket), so a live poll is [`crate::proc_control::fetch_status`]
    /// called straight against this endpoint — no RPC hop, no handle registry.
    control: Option<crate::proc_control::ControlEndpoint>,

    /// [`LogBuffer`] cursor marking the end of the build phase, for a
    /// component that was compiled before being spawned (`local-process`
    /// with `cargo_package` set). Lines before this index in `log_buffer` are
    /// `cargo build` output; lines at or after are the spawned process's own
    /// stdout/stderr. `None` when nothing was built (no `cargo_package`, or a
    /// reconciler that predates this field).
    pub build_log_end: Option<usize>,
}

impl RunningWorkload {
    /// Create a handle for a workload that's already running externally
    /// (e.g. embedded in yah-camp). No subprocess is owned; shutdown is a
    /// no-op so the caller can call `shutdown()` uniformly.
    pub fn adopted(
        kind: impl Into<String>,
        slot: impl Into<String>,
        dev_url: Option<String>,
    ) -> Self {
        Self {
            kind: kind.into(),
            slot: slot.into(),
            dev_url,
            public_url: None,
            console_url: None,
            log_buffer: None,
            notes: Vec::new(),
            shutdown: None,
            supervisor: None,
            teardown: None,
            control: None,
            build_log_end: None,
        }
    }

    /// Attach an explicit teardown to a handle for an externally-running
    /// workload (R714-B1).
    ///
    /// `adopted()` alone gives a handle whose `shutdown()` is a documented
    /// no-op. That is right for workloads another process owns and will keep
    /// re-asserting (pond containers under camp's yubaba), and wrong for ones
    /// this process started and nobody else will ever stop — for those the ■
    /// button reported success while the container kept running.
    ///
    /// `teardown` runs on `shutdown()` and NOT on `Drop`; see [`Self::teardown`].
    pub fn with_teardown<F, Fut>(mut self, teardown: F) -> Self
    where
        F: FnOnce() -> Fut + Send + Sync + 'static,
        Fut: std::future::Future<Output = Result<()>> + Send + 'static,
    {
        self.teardown = Some(Teardown(Box::new(move || Box::pin(teardown()))));
        self
    }

    /// Whether this handle carries a real teardown — i.e. whether a failed
    /// [`Self::shutdown`] means the workload is **still running**.
    ///
    /// The stop path needs this to decide between an operator-facing error and
    /// a log line, and R875-B1 is why it is a property of the handle rather
    /// than a list of kinds at the call site. That list started as
    /// `kind == "container"` when R714-B1 gave containers a teardown, and was
    /// silently wrong the moment a second kind grew one: an adopted
    /// mesofact-dev whose teardown failed reported a successful stop with the
    /// server still serving. A handle knows whether it owns the workload; a
    /// string comparison at the call site only knows what it was last taught.
    pub fn owns_teardown(&self) -> bool {
        self.teardown.is_some()
    }

    /// Attach per-run operator-facing lines (R546-B12). See [`Self::notes`].
    pub fn with_notes(mut self, notes: Vec<String>) -> Self {
        self.notes = notes;
        self
    }

    /// Record where to poll this workload's process-control channel, when it
    /// declared one (R715-F4). A no-op (leaves `control: None`) when the
    /// argument is `None` — callers can pass the reconciler's resolved
    /// endpoint straight through without an `if let`.
    pub fn with_control(mut self, control: Option<crate::proc_control::ControlEndpoint>) -> Self {
        self.control = control;
        self
    }

    /// Ask this workload's process-control channel for a live status
    /// document (R715-F4). `None` when no channel was declared, or when the
    /// endpoint didn't answer — both are "nothing new to show", not errors;
    /// see [`crate::proc_control::fetch_status`] for why a poll failure isn't
    /// itself meaningful.
    pub async fn poll_control(&self) -> Option<crate::proc_control::ProcStatus> {
        let endpoint = self.control.as_ref()?;
        crate::proc_control::fetch_status(endpoint).await.ok()
    }

    /// Whether the supervisor task that owns this workload's child process is
    /// still running. `spawn_native_log_supervisor` (native_support.rs) exits
    /// as soon as `NativeRuntime::get_workload` reports a terminal state —
    /// which itself comes from a real `child.wait()` in kamaji's native
    /// backend, so this catches a crash (segfault, panic, `SIGKILL`) the same
    /// way it catches a clean exit, not just an unresponsive process.
    ///
    /// A workload with no supervisor (`RunningWorkload::adopted` — runs
    /// outside this process, e.g. a pond container camp re-asserts) has
    /// nothing to check here and reports alive unconditionally; its liveness
    /// is whatever tracks it, not this handle.
    pub fn is_alive(&self) -> bool {
        self.supervisor.as_ref().is_none_or(|h| !h.is_finished())
    }

    /// Record where the build phase ends in `log_buffer`, for a component
    /// that was compiled before being spawned. See [`Self::build_log_end`].
    pub fn with_build_log_end(mut self, cursor: Option<usize>) -> Self {
        self.build_log_end = cursor;
        self
    }

    /// Set the public URL for a published workload (e.g. `"https://yah.dev"`).
    pub fn with_public_url(mut self, url: impl Into<String>) -> Self {
        self.public_url = Some(url.into());
        self
    }

    /// Set the console URL for a workload that exposes a secondary local UI
    /// (e.g. MinIO console on a pond tier).
    pub fn with_console_url(mut self, url: impl Into<String>) -> Self {
        self.console_url = Some(url.into());
        self
    }

    /// Gracefully tear down: signal the supervisor, await its exit, then run
    /// any explicit teardown hook.
    ///
    /// The hook runs LAST and its error propagates. A stop that could not tear
    /// the workload down must surface as an error, never as a silent success —
    /// that silence is the whole of R714-B1.
    pub async fn shutdown(mut self) -> Result<()> {
        if let Some(tx) = self.shutdown.take() {
            let _ = tx.send(());
        }
        if let Some(handle) = self.supervisor.take() {
            handle
                .await
                .context("joining workload supervisor")?
                .context("workload supervisor")?;
        }
        if let Some(Teardown(hook)) = self.teardown.take() {
            hook().await.context("workload teardown")?;
        }
        Ok(())
    }
}

impl Drop for RunningWorkload {
    fn drop(&mut self) {
        // Best-effort signal. The supervisor task is detached and will
        // reap its child when it observes the closed channel.
        if let Some(tx) = self.shutdown.take() {
            let _ = tx.send(());
        }
    }
}

/// Bring one workload up. Each impl handles one [`ServiceComponent::kind`].
#[async_trait]
pub trait Reconciler: Send + Sync {
    /// Workload kind this reconciler handles (matches `ServiceComponent.kind`).
    fn kind(&self) -> &'static str;

    /// Bring the workload up. Returns a handle whose lifecycle is tied to
    /// the mirror being up.
    async fn up(&self, ctx: ReconcileCtx<'_>) -> Result<RunningWorkload>;
}

/// Read `<workload_dir>/workload.toml` and return just the `kind` field.
/// See [`ReconcileCtx::workload_kind`] for why we don't deserialize through
/// the strong types yet.
pub fn workload_kind(workload_dir: &Path) -> Result<String> {
    let path = workload_dir.join("workload.toml");
    let src =
        std::fs::read_to_string(&path).with_context(|| format!("reading {}", path.display()))?;
    let value: toml::Value =
        toml::from_str(&src).with_context(|| format!("parsing {}", path.display()))?;
    let kind = value
        .get("kind")
        .and_then(|v| v.as_str())
        .with_context(|| format!("{}: missing `kind` field", path.display()))?;
    Ok(kind.to_string())
}

/// Shallow-clone (or update) a [`GitSource`] into `dir` (R561-F1). Idempotent:
/// clones when `dir/.git` is absent, otherwise fetches the pinned ref and
/// force-checks-it-out. Uses the system `git` so it inherits the operator's
/// credential helpers / SSH agent — no in-process git library.
///
/// `pub` (R615-T3 / W274): `yah infra sync` reuses this verbatim for
/// `InfraSourceKind::Git` sources rather than a second shallow-clone-or-pull
/// implementation — same "one git-source shape, reused" discipline R615-F1
/// already applied to the type.
pub async fn materialize_git_source(git: &GitSource, dir: &Path) -> Result<()> {
    use tokio::process::Command;

    async fn run_git(args: &[&std::ffi::OsStr]) -> Result<()> {
        let out = Command::new("git")
            .args(args)
            .output()
            .await
            .context("spawning git")?;
        if !out.status.success() {
            anyhow::bail!(
                "git {} failed: {}",
                args.iter()
                    .map(|a| a.to_string_lossy())
                    .collect::<Vec<_>>()
                    .join(" "),
                String::from_utf8_lossy(&out.stderr).trim()
            );
        }
        Ok(())
    }

    use std::ffi::OsStr;
    let dir_os = dir.as_os_str();
    let r#ref = git.r#ref.as_str();

    if dir.join(".git").is_dir() {
        // Existing checkout — update to the pinned ref.
        run_git(&[
            OsStr::new("-C"),
            dir_os,
            OsStr::new("fetch"),
            OsStr::new("--depth"),
            OsStr::new("1"),
            OsStr::new("origin"),
            OsStr::new(r#ref),
        ])
        .await?;
        run_git(&[
            OsStr::new("-C"),
            dir_os,
            OsStr::new("checkout"),
            OsStr::new("--force"),
            OsStr::new("FETCH_HEAD"),
        ])
        .await?;
    } else {
        if let Some(parent) = dir.parent() {
            tokio::fs::create_dir_all(parent)
                .await
                .with_context(|| format!("creating {}", parent.display()))?;
        }
        // `--branch` accepts a branch or tag. Pinning to a bare commit SHA is a
        // follow-up (needs clone-then-fetch); the common case is a branch/tag.
        run_git(&[
            OsStr::new("clone"),
            OsStr::new("--depth"),
            OsStr::new("1"),
            OsStr::new("--branch"),
            OsStr::new(r#ref),
            OsStr::new(git.repo.as_str()),
            dir_os,
        ])
        .await?;
    }
    Ok(())
}

/// Build a `RunningWorkload` from the pieces a reconciler produces.
pub(crate) fn into_running(
    kind: impl Into<String>,
    slot: impl Into<String>,
    dev_url: Option<String>,
    public_url: Option<String>,
    log_buffer: Option<LogBuffer>,
    shutdown: oneshot::Sender<()>,
    supervisor: tokio::task::JoinHandle<Result<()>>,
) -> RunningWorkload {
    RunningWorkload {
        kind: kind.into(),
        slot: slot.into(),
        dev_url,
        public_url,
        console_url: None,
        log_buffer,
        notes: Vec::new(),
        shutdown: Some(shutdown),
        supervisor: Some(supervisor),
        teardown: None,
        control: None,
        build_log_end: None,
    }
}

/// Wait for a TCP port to start accepting connections. Returns `true` if
/// the port came up within `timeout`, `false` otherwise. Useful for
/// reconcilers that spawn a server and need to know when it's reachable
/// before reporting success.
///
/// R918-F5 — unix-only: its only caller is [`local_process`], gated above.
#[cfg(unix)]
pub(crate) async fn wait_for_port(
    addr: std::net::SocketAddr,
    timeout: std::time::Duration,
) -> bool {
    let deadline = tokio::time::Instant::now() + timeout;
    loop {
        if tokio::net::TcpStream::connect(addr).await.is_ok() {
            return true;
        }
        if tokio::time::Instant::now() >= deadline {
            return false;
        }
        tokio::time::sleep(std::time::Duration::from_millis(50)).await;
    }
}

// `wait_for_http_ready` lived here pre-R374-F3 to back the MinIO health
// probe in pond's bring-up path. That logic moved to
// `local_driver::pond_minio::wait_for_http_ready` so yubaba + cloud share
// it. The mesofact-static reconciler arm uses [`wait_for_port`] for
// dev-tier port readiness; nothing else needs an HTTP-level probe today.

/// Pluck a `u16` out of a [`MirrorProviderSlot`]'s inline `fields` map.
/// Returns `None` if the key is absent or out of range.
pub(crate) fn slot_field_u16(fields: &BTreeMap<String, toml::Value>, key: &str) -> Option<u16> {
    fields
        .get(key)
        .and_then(|v| v.as_integer())
        .and_then(|n| u16::try_from(n).ok())
}

/// Serializable summary of a running workload — what the desktop / CLI
/// hands to the UI. Subset of [`RunningWorkload`] that's safe to cross
/// process boundaries.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
pub struct RunningWorkloadSummary {
    pub kind: String,
    pub slot: String,
    pub dev_url: Option<String>,
    pub public_url: Option<String>,
    pub console_url: Option<String>,
    /// Operator-facing lines from [`RunningWorkload::notes`] (R546-B12).
    ///
    /// These used to stop here — the summary dropped them, so every note a
    /// reconciler attached was written into a struct nobody read. They matter
    /// most for a workload with no `dev_url` to click (R715-T2): the notes are
    /// then the only structured thing the Run tab has to show about it.
    #[serde(default, skip_serializing_if = "Vec::is_empty")]
    pub notes: Vec<String>,
    /// See [`RunningWorkload::build_log_end`]. Lets a log-tail consumer skip
    /// straight to run-phase output, or show everything from 0 when the
    /// operator wants the build log too (e.g. after a compile failure).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub build_log_end: Option<usize>,
}

impl From<&RunningWorkload> for RunningWorkloadSummary {
    fn from(r: &RunningWorkload) -> Self {
        Self {
            kind: r.kind.clone(),
            slot: r.slot.clone(),
            dev_url: r.dev_url.clone(),
            public_url: r.public_url.clone(),
            console_url: r.console_url.clone(),
            notes: r.notes.clone(),
            build_log_end: r.build_log_end,
        }
    }
}

#[cfg(test)]
mod source_seam_tests {
    //! R561-F1 — the BYO-git source seam: path resolution + materialization.
    use super::*;
    use std::collections::BTreeMap;

    fn component(git: Option<GitSource>) -> ServiceComponent {
        ServiceComponent {
            mount: None,
            id: "site".into(),
            kind: "mesofact-static".into(),
            path: "site".into(),
            role: "static".into(),
            publishes: None,
            wave: 0,
            git,
            deploy: Default::default(),
        }
    }

    fn service(comp: ServiceComponent) -> ServiceConfig {
        ServiceConfig {
            schema_version: 1,
            name: "scrabcake".into(),
            address: crate::config::ServiceAddress::front_door("scrabcake.example"),
            description: None,
            components: vec![comp],
        }
    }

    fn mirror() -> MirrorConfig {
        MirrorConfig {
            schema_version: 1,
            shape: crate::MirrorShape::Local,
            providers: BTreeMap::new(),
            ingress: Default::default(),
            ingress_machines: Vec::new(),
            drivers: Default::default(),
            asset_aliases: BTreeMap::new(),
            build: Default::default(),
        }
    }

    fn ctx<'a>(ws: &'a Path, svc: &'a ServiceConfig, mir: &'a MirrorConfig) -> ReconcileCtx<'a> {
        ReconcileCtx {
            workspace_root: ws,
            service: svc,
            component: &svc.components[0],
            mirror: mir,
            env: "dev",
            scope: ProviderScope::singleton(),
        }
    }

    #[test]
    fn workload_dir_in_tree_joins_workspace_root() {
        let svc = service(component(None));
        let mir = mirror();
        assert_eq!(
            ctx(Path::new("/ws"), &svc, &mir).workload_dir(),
            Path::new("/ws/site")
        );
    }

    #[test]
    fn workload_dir_git_resolves_into_source_cache_with_subdir() {
        let git = GitSource {
            repo: "https://example.com/r.git".into(),
            r#ref: "main".into(),
            subdir: Some("apps".into()),
        };
        let svc = service(component(Some(git)));
        let mir = mirror();
        assert_eq!(
            ctx(Path::new("/ws"), &svc, &mir).workload_dir(),
            Path::new("/ws/.yah/infra/state/sources/scrabcake/site/apps/site")
        );
    }

    #[tokio::test]
    async fn materialize_is_noop_for_in_tree_component() {
        let svc = service(component(None));
        let mir = mirror();
        // No git source → Ok, and nothing is written under the workspace.
        ctx(Path::new("/nonexistent-ws"), &svc, &mir)
            .materialize()
            .await
            .unwrap();
    }

    #[tokio::test]
    async fn materialize_clones_git_source_offline() {
        fn git(args: &[&str], cwd: &Path) {
            let out = std::process::Command::new("git")
                .args(args)
                .current_dir(cwd)
                .env("GIT_AUTHOR_NAME", "t")
                .env("GIT_AUTHOR_EMAIL", "t@t")
                .env("GIT_COMMITTER_NAME", "t")
                .env("GIT_COMMITTER_EMAIL", "t@t")
                .output()
                .unwrap();
            assert!(
                out.status.success(),
                "git {args:?}: {}",
                String::from_utf8_lossy(&out.stderr)
            );
        }

        let tmp = tempfile::tempdir().unwrap();
        let src = tmp.path().join("src-repo");
        std::fs::create_dir_all(&src).unwrap();
        git(&["init", "-b", "main"], &src);
        std::fs::write(src.join("hello.txt"), "hi").unwrap();
        git(&["add", "."], &src);
        git(&["commit", "-m", "init"], &src);

        let source = GitSource {
            repo: format!("file://{}", src.display()),
            r#ref: "main".into(),
            subdir: None,
        };
        let dest = tmp.path().join("cache");

        // First call clones.
        materialize_git_source(&source, &dest).await.unwrap();
        assert!(dest.join("hello.txt").is_file());

        // Second call takes the update path and stays green (idempotent).
        materialize_git_source(&source, &dest).await.unwrap();
        assert!(dest.join("hello.txt").is_file());
    }
}

#[cfg(test)]
mod teardown_tests {
    //! R714-B1 — the explicit-teardown contract on [`RunningWorkload`].
    //!
    //! These are about WHEN the hook runs, not what it does. The bug being
    //! fixed was a `shutdown()` that reported success having done nothing, and
    //! the trap in fixing it is a `Drop` that tears down a container which is
    //! supposed to survive the process.
    use super::*;
    use std::sync::atomic::{AtomicUsize, Ordering};

    fn counting() -> (RunningWorkload, Arc<AtomicUsize>) {
        let hits = Arc::new(AtomicUsize::new(0));
        let seen = hits.clone();
        let w = RunningWorkload::adopted("container", "compute", None)
            .with_teardown(move || {
                let seen = seen.clone();
                async move {
                    seen.fetch_add(1, Ordering::SeqCst);
                    Ok(())
                }
            });
        (w, hits)
    }

    /// Attaching a teardown must not cost `RunningWorkload` its auto traits.
    /// The desktop stores these handles in a shared registry and its Tauri
    /// commands iterate them across `.await` points, which needs `Sync` — a
    /// hook that is `Send` but not `Sync` takes it away here and surfaces two
    /// crates over as "future cannot be sent between threads safely", with a
    /// span pointing at `mirror_run_logs` rather than at this file.
    #[test]
    fn a_teardown_does_not_cost_the_handle_send_or_sync() {
        fn assert_send_sync<T: Send + Sync>() {}
        assert_send_sync::<RunningWorkload>();
    }

    #[tokio::test]
    async fn shutdown_runs_the_teardown() {
        let (w, hits) = counting();
        w.shutdown().await.unwrap();
        assert_eq!(hits.load(Ordering::SeqCst), 1);
    }

    #[tokio::test]
    async fn dropping_the_handle_does_NOT_run_the_teardown() {
        // A container this process started is meant to outlive it — the next
        // launch re-adopts it. Quitting the app must not `docker rm -f` it.
        let (w, hits) = counting();
        drop(w);
        // Yield so a stray spawned task would have had a chance to run.
        tokio::task::yield_now().await;
        assert_eq!(hits.load(Ordering::SeqCst), 0);
    }

    #[tokio::test]
    async fn a_failing_teardown_makes_shutdown_fail() {
        // The whole of R714-B1: a stop that could not tear the workload down
        // must not report success.
        let w = RunningWorkload::adopted("container", "compute", None)
            .with_teardown(|| async { anyhow::bail!("docker stop refused") });
        let err = w.shutdown().await.unwrap_err();
        assert!(format!("{err:#}").contains("docker stop refused"), "{err:#}");
    }

    #[tokio::test]
    async fn an_adopted_handle_without_a_teardown_still_shuts_down_cleanly() {
        // Pond containers are owned by camp's yubaba and stop through
        // `workload.stop`; their no-op shutdown is correct and must stay.
        let w = RunningWorkload::adopted("mesofact-static", "static", None);
        w.shutdown().await.unwrap();
    }

    /// R875-B1. The stop path decides "still running, tell the operator" vs
    /// "untidy supervisor, log it" from this, so it has to answer for the
    /// handle in front of it rather than for the kind string it carries — the
    /// two disagreed for every adopted mesofact-dev.
    #[test]
    fn owns_teardown_tracks_the_hook_not_the_kind() {
        let (owned, _) = counting();
        assert!(owned.owns_teardown());
        assert!(
            !RunningWorkload::adopted("container", "compute", None).owns_teardown(),
            "a bare adopted handle owns nothing, whatever its kind says"
        );
        assert!(
            RunningWorkload::adopted("mesofact-static", "static", None)
                .with_teardown(|| async { Ok(()) })
                .owns_teardown(),
            "a non-container kind with a teardown owns its workload"
        );
    }

    #[tokio::test]
    async fn the_teardown_runs_after_the_supervisor_is_joined() {
        // Ordering matters for a workload that has both: reap the child first,
        // then remove the container it was talking to.
        let order = Arc::new(AsyncMutex::new(Vec::<&'static str>::new()));
        let (tx, rx) = oneshot::channel::<()>();
        let sup_order = order.clone();
        let supervisor = tokio::spawn(async move {
            let _ = rx.await;
            sup_order.lock().await.push("supervisor");
            Ok(())
        });
        let hook_order = order.clone();
        let w = into_running("container", "compute", None, None, None, tx, supervisor)
            .with_teardown(move || {
                let hook_order = hook_order.clone();
                async move {
                    hook_order.lock().await.push("teardown");
                    Ok(())
                }
            });

        w.shutdown().await.unwrap();
        assert_eq!(*order.lock().await, vec!["supervisor", "teardown"]);
    }
}